Compare commits

..
Author SHA1 Message Date
Garry TanandClaude Fable 5 89579780e0 fix(embed): extend #1717 model labeling to the embed-backfill stale path
src/core/embed-stale.ts (used by the embed-backfill minion handler) built
its merged ChunkInput without a model field, so every chunk on a touched
page — re-embedded AND preserved — was relabeled to the engine default on
each backfill pass. Mirror the embed.ts semantics: stamp the resolved
gateway label on re-embedded chunks, carry the existing label on untouched
ones. Pinned by a PGLite test that fails without the fix.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-22 11:06:16 -07:00
cf2deedfc6 fix(embed): label content_chunks.model with the model that produced the vector (#1717)
The embed paths (embedPage, embedAll, embedAllStale) and the inline
import/sync embed paths built ChunkInput[] without a model field, so the
engines' upsertChunks defaulted content_chunks.model to the hardcoded
DEFAULT_EMBEDDING_MODEL instead of the gateway-configured model that
actually produced the vector.

- New core helper resolveEmbeddingModelLabel() in src/core/embedding.ts
  (returns the resolved gateway model, undefined when unconfigured).
- embed.ts: stamp the label on (re)embedded chunks in all three paths;
  chunks preserved from a prior embed keep their existing model so a
  mixed-model page isn't relabeled wholesale.
- import-file.ts: stamp the label on inline-embedded markdown chunks and
  re-embedded code chunks; reused (incremental) code-chunk embeddings
  carry their existing model label forward.

Takeover of PR #1803 (rebased onto master over the pace-mode changes;
helper moved into core so import-file.ts can share it).

Co-authored-by: harjothkhara <harjothkhara@users.noreply.github.com>
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-21 14:42:17 -07:00
19 changed files with 148 additions and 508 deletions
+1 -3
View File
@@ -1411,8 +1411,7 @@ This is the dispatcher. Skills are the implementation. **Read the skill file bef
| "get more out of gbrain", "is my brain set up right", "weekly brain checkup", "advise me on my brain", "gbrain advisor" | `skills/gbrain-advisor/SKILL.md` |
| Save or load reports | `skills/reports/SKILL.md` |
| "Create a skill", "improve this skill" | `skills/skill-creator/SKILL.md` |
| "save this learning to the vault", "capture this skill in Obsidian", "record this workflow in my notes", "put this setup change in the vault" | `skills/skill-vault-capture-policy/SKILL.md` |
| "Skillify this", "is this a skill?", "make this proper", "add tests and evals for this" | `skills/skillify/SKILL.md` |
| "Skillify this", "is this a skill?", "make this proper" | `skills/skillify/SKILL.md` |
| "Compress my resolver", "AGENTS.md too large", "RESOLVER.md too big", "functional area dispatcher", "shrink routing table" | `skills/functional-area-resolver/SKILL.md` |
| "Is gbrain healthy?", morning health check, skillpack-check | `skills/skillpack-check/SKILL.md` |
| "harvest this skill into gbrain", "publish this skill to gbrain", "lift this skill upstream", "share this skill with other gbrain clients", "promote my skill to gbrain" | `skills/skillpack-harvest/SKILL.md` |
@@ -1430,7 +1429,6 @@ This is the dispatcher. Skills are the implementation. **Read the skill file bef
| "Set up GBrain", first boot | `skills/setup/SKILL.md` |
| "Now what?", "fill my brain", "cold start", "bootstrap", "import my data", "what should I import first" | `skills/cold-start/SKILL.md` |
| "Migrate from Obsidian/Notion/Logseq" | `skills/migrate/SKILL.md` |
| "Connect Obsidian to gbrain", "import my vault to gbrain", "sync vault and gbrain", "embed gbrain after vault update", "is gbrain synced with my vault" | `skills/obsidian-gbrain-safe-index/SKILL.md` |
| Brain health check, maintenance run | `skills/maintain/SKILL.md` |
| "Extract links", "build link graph", "populate timeline" | `skills/maintain/SKILL.md` (extraction sections) |
| "Run dream", "process today's session", "synthesize my conversations", "consolidate yesterday's conversations", "what patterns did you see", "did the dream cycle run" | `skills/maintain/SKILL.md` (dream cycle section) |
+2 -3
View File
@@ -39,8 +39,8 @@
"skills/briefing",
"skills/citation-fixer",
"skills/concept-synthesis",
"skills/cron-scheduler",
"skills/cross-modal-review",
"skills/cron-scheduler",
"skills/daily-task-manager",
"skills/daily-task-prep",
"skills/data-research",
@@ -54,11 +54,10 @@
"skills/media-ingest",
"skills/meeting-ingestion",
"skills/minion-orchestrator",
"skills/obsidian-gbrain-safe-index",
"skills/perplexity-research",
"skills/query",
"skills/repo-architecture",
"skills/reports",
"skills/repo-architecture",
"skills/signal-detector",
"skills/skill-creator",
"skills/skillify",
+1 -3
View File
@@ -60,8 +60,7 @@ This is the dispatcher. Skills are the implementation. **Read the skill file bef
| "get more out of gbrain", "is my brain set up right", "weekly brain checkup", "advise me on my brain", "gbrain advisor" | `skills/gbrain-advisor/SKILL.md` |
| Save or load reports | `skills/reports/SKILL.md` |
| "Create a skill", "improve this skill" | `skills/skill-creator/SKILL.md` |
| "save this learning to the vault", "capture this skill in Obsidian", "record this workflow in my notes", "put this setup change in the vault" | `skills/skill-vault-capture-policy/SKILL.md` |
| "Skillify this", "is this a skill?", "make this proper", "add tests and evals for this" | `skills/skillify/SKILL.md` |
| "Skillify this", "is this a skill?", "make this proper" | `skills/skillify/SKILL.md` |
| "Compress my resolver", "AGENTS.md too large", "RESOLVER.md too big", "functional area dispatcher", "shrink routing table" | `skills/functional-area-resolver/SKILL.md` |
| "Is gbrain healthy?", morning health check, skillpack-check | `skills/skillpack-check/SKILL.md` |
| "harvest this skill into gbrain", "publish this skill to gbrain", "lift this skill upstream", "share this skill with other gbrain clients", "promote my skill to gbrain" | `skills/skillpack-harvest/SKILL.md` |
@@ -79,7 +78,6 @@ This is the dispatcher. Skills are the implementation. **Read the skill file bef
| "Set up GBrain", first boot | `skills/setup/SKILL.md` |
| "Now what?", "fill my brain", "cold start", "bootstrap", "import my data", "what should I import first" | `skills/cold-start/SKILL.md` |
| "Migrate from Obsidian/Notion/Logseq" | `skills/migrate/SKILL.md` |
| "Connect Obsidian to gbrain", "import my vault to gbrain", "sync vault and gbrain", "embed gbrain after vault update", "is gbrain synced with my vault" | `skills/obsidian-gbrain-safe-index/SKILL.md` |
| Brain health check, maintenance run | `skills/maintain/SKILL.md` |
| "Extract links", "build link graph", "populate timeline" | `skills/maintain/SKILL.md` (extraction sections) |
| "Run dream", "process today's session", "synthesize my conversations", "consolidate yesterday's conversations", "what patterns did you see", "did the dream cycle run" | `skills/maintain/SKILL.md` (dream cycle section) |
+1 -11
View File
@@ -2,7 +2,7 @@
"name": "gbrain",
"version": "0.32.3.0",
"conformance_version": "1.0.0",
"description": "Personal knowledge brain with hybrid RAG search GStack mod for agent platforms",
"description": "Personal knowledge brain with hybrid RAG search \u2014 GStack mod for agent platforms",
"skills": [
{
"name": "ingest",
@@ -34,11 +34,6 @@
"path": "migrate/SKILL.md",
"description": "Universal migration from Obsidian, Notion, Logseq, markdown, CSV, JSON, Roam"
},
{
"name": "obsidian-gbrain-safe-index",
"path": "obsidian-gbrain-safe-index/SKILL.md",
"description": "Connect and sync an Obsidian-style Markdown vault with gbrain using a cost-controlled import-first workflow, explicit embedding gates, and durable skill/workflow capture."
},
{
"name": "setup",
"path": "setup/SKILL.md",
@@ -268,11 +263,6 @@
"name": "skill-optimizer",
"path": "skill-optimizer/SKILL.md",
"description": "Self-evolving skill optimization via gbrain skillopt — SkillOpt-paper-grounded text-space optimizer with validation gating (median-of-3 + epsilon=0.05), bundled-skill safety, bootstrap review sentinel, per-skill DB lock, and atomic versioned writes."
},
{
"name": "skill-vault-capture-policy",
"path": "skill-vault-capture-policy/SKILL.md",
"description": "Capture durable operational learnings, new agent skills, and environment/setup changes into the Obsidian vault instead of transient chat memory."
}
],
"dependencies": {
-137
View File
@@ -1,137 +0,0 @@
---
name: obsidian-gbrain-safe-index
version: 1.0.0
description: |
Connect, maintain, and sync an Obsidian-style Markdown vault with gbrain while preserving a cost-controlled workflow: the vault remains the source of truth, gbrain is the searchable/embedded index, durable skills/workflows are captured into the vault, and paid embedding runs only after explicit approval.
triggers:
- "connect Obsidian to gbrain"
- "import my vault to gbrain"
- "sync vault and gbrain"
- "capture this skill in my vault"
- "embed gbrain after vault update"
- "is gbrain synced with my vault"
tools:
- terminal
- read_file
- search_files
- write_file
- patch
mutating: true
---
# Obsidian → gbrain Safe Index and Capture
## Contract
This skill guarantees:
- Treats the user's Obsidian-style Markdown vault as the source of truth before gbrain indexing.
- Keeps gbrain in conservative mode unless the user explicitly approves a more expensive mode.
- Imports vault changes with `--no-embed` first, then embeds only after explicit approval for the specific paid action.
- Captures durable new skills, workflows, and environment learnings into the vault instead of leaving them only in chat memory.
- Verifies every sync with concrete `gbrain stats`, search mode, and, when embeddings run, exact embedded chunk counts.
## Phases
1. **Resolve the vault path.**
- Prefer an existing environment variable such as `OBSIDIAN_VAULT_PATH` or `WIKI_PATH`.
- If no path is configured, search likely note directories and ask the user before writing.
- Verify the directory exists and contains markdown files or an `.obsidian` directory.
2. **Read vault operating rules before writing.**
- If the vault has `SCHEMA.md`, `index.md`, `log.md`, `AGENTS.md`, or similar operating files, read them before ingest/query/major edit.
- Respect immutable source folders such as `raw/` when the vault declares them.
- Use the vault's native link convention, usually Obsidian `[[wikilinks]]`, for durable relationships.
3. **MECE/capture decision.**
- If new knowledge belongs on an existing page, update that page.
- If it is a distinct recurring workflow or operational policy, create a small meta or concept page following the vault schema.
- Update the vault index/catalog for every new page when the vault maintains one.
- Append a log entry for meaningful vault updates when the vault maintains a log.
4. **Safe gbrain import path.**
- Pre-check source directory; do not import a nonexistent path.
- Run `gbrain config set search.mode conservative` before/after risky reinit steps.
- Run `gbrain import "$OBSIDIAN_VAULT_PATH" --no-embed`.
- Run `gbrain extract links --source fs --dir "$OBSIDIAN_VAULT_PATH"` when wikilinks changed materially.
5. **Paid embedding gate.**
- Do not run `gbrain embed --stale` unless the user explicitly asks or a prior instruction clearly approved this exact paid action.
- Before embedding, verify provider readiness with `gbrain providers test --model <provider:model>`.
- Confirm the configured embedding dimensions match the local schema.
- After embedding, verify `gbrain stats` and record exact `Pages`, `Chunks`, `Embedded`, and `Links` counts.
6. **Final verification and vault echo.**
- Run `gbrain stats` and `gbrain search modes`.
- If vault files changed, re-import with `--no-embed`; if embedding was approved, embed stale chunks afterward.
- Report what changed, what was free/local, what used API billing, and what remains pending.
## Output Format
Use a compact status table:
| Item | Status |
|---|---|
| Vault path | `/path` |
| Vault updated | yes/no + files |
| gbrain mode | conservative/balanced/tokenmax |
| Import | `--no-embed` completed / skipped / failed |
| Pages/chunks | exact counts from `gbrain stats` |
| Embeddings | exact count; note whether this run used API billing |
| Links | exact count |
| Background jobs | none / list exact jobs |
Then include:
- **Safe next step:** free/local action.
- **Paid next step:** embedding/LLM action, if any, with explicit approval requirement.
## Anti-Patterns
- Creating a duplicate skill/page when an existing Obsidian, gbrain, or vault-ingest skill already covers the workflow.
- Running `gbrain embed --stale`, `gbrain dream`, `gbrain autopilot --install`, `gbrain onboard --auto`, or `tokenmax` without explicit cost approval.
- Importing a nonexistent or wrong directory and treating a zero-page import as success.
- Forgetting to update the vault index/catalog and log after creating or materially updating vault pages.
- Recording API keys, tokens, or raw secrets in the vault or final response.
## Tools Used
- `read_file` — read vault schema/index/log and target notes.
- `search_files` — find existing vault pages and avoid duplicates.
- `write_file` / `patch` — create or update vault pages.
- `terminal` — run `gbrain`, `git`, and environment checks with secret values redacted.
## Safe Commands
```bash
export PATH="$HOME/.bun/bin:$PATH"
gbrain config set search.mode conservative
gbrain import "$OBSIDIAN_VAULT_PATH" --no-embed
gbrain extract links --source fs --dir "$OBSIDIAN_VAULT_PATH"
gbrain stats
gbrain search modes
```
## Paid / Approval-Gated Commands
```bash
gbrain providers test --model <provider:model>
gbrain embed --stale
gbrain dream
gbrain autopilot --install
gbrain onboard --auto --max-usd 5
gbrain config set search.mode tokenmax
```
## Verification Checklist
- [ ] Vault path exists and is the intended source.
- [ ] Vault operating files were read before edits when present.
- [ ] Existing pages/skills were searched to avoid duplicates.
- [ ] New/updated vault pages follow the vault schema and link convention.
- [ ] Vault index/catalog updated for new pages when present.
- [ ] Vault log appended for meaningful actions when present.
- [ ] `gbrain import ... --no-embed` completed.
- [ ] `gbrain stats` recorded pages/chunks/embeddings/links.
- [ ] `gbrain search modes` confirms conservative mode unless a different mode was explicitly approved.
- [ ] No paid/background commands ran without approval.
@@ -1,11 +0,0 @@
// Routing eval fixtures for skills/obsidian-gbrain-safe-index.
// Positive cases: intents embed a trigger phrase in natural surrounding context
// (never verbatim-identical to a trigger — the fixture linter rejects tautologies).
{"intent": "help me connect Obsidian to gbrain for my notes", "expected_skill": "obsidian-gbrain-safe-index"}
{"intent": "import my vault to gbrain but skip embeddings for now", "expected_skill": "obsidian-gbrain-safe-index"}
{"intent": "please sync vault and gbrain after I edit notes", "expected_skill": "obsidian-gbrain-safe-index"}
{"intent": "run embed gbrain after vault update tonight", "expected_skill": "obsidian-gbrain-safe-index"}
{"intent": "hey is gbrain synced with my vault right now", "expected_skill": "obsidian-gbrain-safe-index"}
// Negative cases: related but owned by other skills. Assert NO route to this skill.
{"intent": "migrate my notes from Notion to gbrain", "expected_skill": null}
{"intent": "what is on my calendar tomorrow", "expected_skill": null}
-100
View File
@@ -1,100 +0,0 @@
---
name: skill-vault-capture-policy
version: 1.0.0
description: |
Use when the user wants a durable operational learning, new agent skill, or
important environment/setup change to be captured into the Obsidian vault
instead of left only in transient chat memory. Covers the capture rule,
preferred page patterns, and index/log update obligations.
triggers:
- "save this learning to the vault"
- "capture this skill in Obsidian"
- "record this workflow in my notes"
- "put this setup change in the vault"
- "should we add this to the knowledge base"
tools:
- read_file
- search_files
- write_file
- patch
mutating: true
---
# Skill Vault Capture Policy
## Contract
This skill guarantees:
- Durable operational learnings, newly adopted skills, and important environment/setup changes are captured into the Obsidian vault rather than left only in chat memory.
- An existing page is updated when the knowledge clearly belongs there; a new page is created only when the topic is distinct and likely to recur.
- Every new vault page is added to `index.md`.
- Every meaningful create/update appends a dated entry to `log.md`.
- Small linked pages are preferred over one giant running note.
## Phases
1. **Classify the learning.**
- New gbrain operating rule, cost control, or embedding/provider change.
- New agent-fork / harness / coding-tool integration fact.
- New Obsidian vault workflow or structure decision.
- New recurring agent skill that changes how the agent should operate here.
2. **Avoid duplicates.**
- Search the vault for an existing page that already owns the topic.
- If found, update it with a new section or dated note rather than creating a near-duplicate.
3. **Create when distinct.**
- Place new pages under the vault schema: `_meta/` for operating notes, `concepts/` for workflows, `entities/` for tools/people.
- Use YAML frontmatter and at least two `[[wikilinks]]` unless it is a short seed page.
4. **Update navigation.**
- Add the page to `index.md` under the correct type heading.
- Append a `## [YYYY-MM-DD] create|update | subject` entry to `log.md`.
5. **Report the capture.**
- State which files changed and whether `index.md` / `log.md` were updated.
## Output Format
Use a short status block:
| Item | Status |
|---|---|
| Learning classified | type |
| Page created/updated | path |
| index.md updated | yes/no |
| log.md updated | yes/no |
## Anti-Patterns
- Leaving durable learnings only in chat memory.
- Creating a near-duplicate page instead of updating the existing one.
- Forgetting to update `index.md` and `log.md`.
- Writing one giant running note instead of small linked pages.
- Recording secrets, API keys, or raw credentials in the vault.
## Tools Used
- `read_file` — read `SCHEMA.md`, `index.md`, `log.md`, and target pages.
- `search_files` — find existing pages to avoid duplicates.
- `write_file` / `patch` — create or update vault pages and navigation.
## Safe Commands
```bash
# inspect vault navigation before writing
read SCHEMA.md index.md log.md
# create or update a page, then refresh catalog/log
# index.md: add [[page-slug]] under the matching type heading
# log.md: append ## [YYYY-MM-DD] create|update | subject
```
## Verification Checklist
- [ ] Vault schema/read files were checked before writing.
- [ ] Existing pages were searched to avoid duplicates.
- [ ] New/updated page has frontmatter and wikilinks.
- [ ] `index.md` updated for new pages.
- [ ] `log.md` appended for meaningful actions.
- [ ] No secrets or raw credentials were written.
@@ -1,10 +0,0 @@
// Routing eval fixtures for skills/skill-vault-capture-policy.
// Positive cases: intents embed a trigger phrase in natural surrounding context
// (never verbatim-identical to a trigger — the fixture linter rejects tautologies).
{"intent": "please save this learning to the vault so we keep it", "expected_skill": "skill-vault-capture-policy", "ambiguous_with": ["idea-ingest"]}
{"intent": "we should capture this skill in Obsidian for reuse", "expected_skill": "skill-vault-capture-policy", "ambiguous_with": ["capture"]}
{"intent": "can you record this workflow in my notes for next time", "expected_skill": "skill-vault-capture-policy"}
{"intent": "put this setup change in the vault before we forget", "expected_skill": "skill-vault-capture-policy"}
// Negative cases: related but owned by other skills or out of scope.
{"intent": "connect my Obsidian vault to gbrain", "expected_skill": null}
{"intent": "what is on my calendar tomorrow", "expected_skill": null}
+14 -1
View File
@@ -1,5 +1,5 @@
import type { BrainEngine } from '../core/engine.ts';
import { embedBatch, currentEmbeddingSignature } from '../core/embedding.ts';
import { embedBatch, currentEmbeddingSignature, resolveEmbeddingModelLabel } from '../core/embedding.ts';
import type { ChunkInput } from '../core/types.ts';
import { chunkText } from '../core/chunkers/recursive.ts';
import { createProgress, type ProgressReporter } from '../core/progress.ts';
@@ -581,11 +581,16 @@ async function embedPage(
for (let j = 0; j < toEmbed.length; j++) {
embeddingMap.set(toEmbed[j].chunk_index, embeddings[j]);
}
// #1717: label each (re)embedded chunk with the model that actually
// produced its vector. Preserved chunks (not re-embedded this pass) keep
// their existing model so a mixed-model page isn't relabeled wholesale.
const embedModelLabel = resolveEmbeddingModelLabel();
const updated: ChunkInput[] = chunks.map(c => ({
chunk_index: c.chunk_index,
chunk_text: c.chunk_text,
chunk_source: c.chunk_source,
embedding: embeddingMap.get(c.chunk_index),
model: embeddingMap.has(c.chunk_index) && embedModelLabel ? embedModelLabel : c.model,
token_count: c.token_count || Math.ceil(c.chunk_text.length / 4),
}));
@@ -717,12 +722,16 @@ async function embedAll(
for (let j = 0; j < toEmbed.length; j++) {
embeddingMap.set(toEmbed[j].chunk_index, embeddings[j]);
}
// #1717: stamp the resolved embedding model on (re)embedded chunks;
// preserve the existing model on chunks left untouched.
const embedModelLabel = resolveEmbeddingModelLabel();
// Preserve ALL chunks, only update embeddings for stale ones
const updated: ChunkInput[] = chunks.map(c => ({
chunk_index: c.chunk_index,
chunk_text: c.chunk_text,
chunk_source: c.chunk_source,
embedding: embeddingMap.get(c.chunk_index) ?? undefined,
model: embeddingMap.has(c.chunk_index) && embedModelLabel ? embedModelLabel : c.model,
token_count: c.token_count || Math.ceil(c.chunk_text.length / 4),
}));
await observed(pacer, () => engine.upsertChunks(page.slug, updated, pageOpts));
@@ -1012,11 +1021,15 @@ async function embedAllStale(
for (let j = 0; j < stale.length; j++) {
staleIdxToEmbedding.set(stale[j].chunk_index, embeddings[j]);
}
// #1717: label the re-embedded (stale) chunks with the resolved
// model; preserve the existing model on the non-stale chunks.
const embedModelLabel = resolveEmbeddingModelLabel();
const merged: ChunkInput[] = existing.map(c => ({
chunk_index: c.chunk_index,
chunk_text: c.chunk_text,
chunk_source: c.chunk_source,
embedding: staleIdxToEmbedding.get(c.chunk_index) ?? undefined,
model: staleIdxToEmbedding.has(c.chunk_index) && embedModelLabel ? embedModelLabel : c.model,
token_count: c.token_count || Math.ceil(c.chunk_text.length / 4),
}));
await observed(pacer, () => engine.upsertChunks(slug, merged, { sourceId: keySourceId }));
+7
View File
@@ -20,6 +20,7 @@
import type { BrainEngine } from './engine.ts';
import type { ChunkInput } from './types.ts';
import { embedBatchWithBackoff } from '../commands/embed.ts';
import { resolveEmbeddingModelLabel } from './embedding.ts';
import { type DbPacer, createNoopPacer, observed } from './db-pacer.ts';
import { AbortError } from './abort-check.ts';
@@ -200,11 +201,17 @@ export async function embedStaleForSource(
for (let j = 0; j < stale.length; j++) {
staleIdxToEmbedding.set(stale[j].chunk_index, embeddings[j]);
}
// #1717: label re-embedded chunks with the model that produced the
// vector; preserved chunks keep their existing model. Without this,
// upsertChunks falls back to DEFAULT_EMBEDDING_MODEL for every chunk
// (the same mislabel the embed.ts paths fixed).
const embedModelLabel = resolveEmbeddingModelLabel();
const merged: ChunkInput[] = existing.map((c) => ({
chunk_index: c.chunk_index,
chunk_text: c.chunk_text,
chunk_source: c.chunk_source,
embedding: staleIdxToEmbedding.get(c.chunk_index) ?? undefined,
model: staleIdxToEmbedding.has(c.chunk_index) && embedModelLabel ? embedModelLabel : c.model,
token_count: c.token_count || Math.ceil(c.chunk_text.length / 4),
// Carry through per-chunk metadata. upsertChunks writes these as
// EXCLUDED.<col> (not COALESCE), so omitting them here resets image
+15
View File
@@ -113,6 +113,21 @@ export async function embedBatch(
return results;
}
/**
* Resolve the embedding model label (`provider:model`) to stamp onto
* `content_chunks.model`, so each chunk records the model that actually
* produced its vector instead of the engine's hardcoded default (#1717).
* Returns undefined if the gateway is unconfigured; callers then fall back
* to the chunk's existing model rather than mislabeling it.
*/
export function resolveEmbeddingModelLabel(): string | undefined {
try {
return gatewayGetModel();
} catch {
return undefined;
}
}
/** Currently-configured embedding model (short form without provider prefix). */
export function getEmbeddingModelName(): string {
return gatewayGetModel().split(':').slice(1).join(':') || 'text-embedding-3-large';
+11 -1
View File
@@ -8,7 +8,7 @@ import { chunkText } from './chunkers/recursive.ts';
import { chunkCodeText, chunkCodeTextFull, detectCodeLanguage, CHUNKER_VERSION } from './chunkers/code.ts';
import { findChunkForOffset } from './chunkers/edge-extractor.ts';
import { extractCodeRefs, imageOfCandidates } from './link-extraction.ts';
import { embedBatch, embedMultimodal, currentEmbeddingSignature } from './embedding.ts';
import { embedBatch, embedMultimodal, currentEmbeddingSignature, resolveEmbeddingModelLabel } from './embedding.ts';
import { slugifyPath, slugifyCodePath, isCodeFilePath } from './sync.ts';
import type { ChunkInput, PageInput, PageType } from './types.ts';
import { computeEffectiveDate } from './effective-date.ts';
@@ -716,8 +716,12 @@ export async function importFromContent(
? chunks.map((c) => wrapChunkForEmbedding(c.chunk_text, prefix, c.chunk_source))
: chunks.map((c) => c.chunk_text);
const embeddings = await embedBatch(wrappedTexts);
// #1717: label each chunk with the model that actually produced its
// vector, not the engine's hardcoded default.
const embedModelLabel = resolveEmbeddingModelLabel();
for (let i = 0; i < chunks.length; i++) {
chunks[i].embedding = embeddings[i];
if (embedModelLabel) chunks[i].model = embedModelLabel;
// token_count tracks the wrapped string length so cost reporting
// reflects what we actually sent to the embedder.
chunks[i].token_count = Math.ceil(wrappedTexts[i].length / 4);
@@ -1141,7 +1145,10 @@ export async function importCodeFile(
const matched = existingByKey.get(key);
if (matched && matched.embedding) {
// Reuse the existing embedding verbatim. No API call, no cost.
// #1717: carry the existing model label along with the reused vector
// so the upsert doesn't relabel it with the engine default.
chunks[i]!.embedding = matched.embedding as Float32Array;
chunks[i]!.model = matched.model ?? undefined;
chunks[i]!.token_count = matched.token_count ?? undefined;
} else {
needsEmbedIndexes.push(i);
@@ -1153,9 +1160,12 @@ export async function importCodeFile(
try {
const textsToEmbed = needsEmbedIndexes.map((i) => chunks[i]!.chunk_text);
const embeddings = await embedBatch(textsToEmbed);
// #1717: stamp the model that produced these vectors.
const embedModelLabel = resolveEmbeddingModelLabel();
for (let j = 0; j < needsEmbedIndexes.length; j++) {
const i = needsEmbedIndexes[j]!;
chunks[i]!.embedding = embeddings[j]!;
if (embedModelLabel) chunks[i]!.model = embedModelLabel;
chunks[i]!.token_count = Math.ceil(chunks[i]!.chunk_text.length / 4);
}
} catch (e: unknown) {
@@ -1,61 +0,0 @@
/**
* E2E smoke for skills/obsidian-gbrain-safe-index.
*
* Verifies the from-trigger-to-side-effect path that skillify requires:
* a real user trigger phrase routes to the skill, the resolver/check
* pipeline treats it as reachable, and the skill file exposes the
* gbrain commands the workflow actually runs.
*
* This stays local-only (no paid embedding, no external API): it asserts
* the documented safe/import path is present and parseable, not that it
* mutates a live brain.
*/
import { describe, expect, it } from 'bun:test';
import { existsSync, readFileSync } from 'fs';
import { join } from 'path';
const SKILLS = join(import.meta.dir, '..', '..', 'skills');
const SKILL_MD = join(SKILLS, 'obsidian-gbrain-safe-index', 'SKILL.md');
const RESOLVER = join(SKILLS, 'RESOLVER.md');
const TRIGGER_PHRASES = [
'connect my Obsidian vault to gbrain',
'import my vault to gbrain',
'sync vault and gbrain',
'capture this skill in my vault',
'embed gbrain after vault update',
'is gbrain synced with my vault',
];
describe('obsidian-gbrain-safe-index E2E', () => {
it('resolver maps real trigger phrasings to the skill', () => {
const resolver = readFileSync(RESOLVER, 'utf-8');
expect(resolver).toContain('obsidian-gbrain-safe-index/SKILL.md');
// Each representative phrase shares a token substring with a resolver row.
const rows = resolver
.split('\n')
.filter((l) => l.includes('obsidian-gbrain-safe-index/SKILL.md'))
.join('\n');
for (const phrase of TRIGGER_PHRASES) {
const hit = phrase
.toLowerCase()
.split(/\s+/)
.some((tok) => tok.length > 3 && rows.toLowerCase().includes(tok));
expect(hit, `no resolver token for: ${phrase}`).toBe(true);
}
});
it('skill documents the gbrain import-first safe path', () => {
const body = readFileSync(SKILL_MD, 'utf-8');
expect(body).toContain('gbrain import');
expect(body).toContain('--no-embed');
expect(body).toContain('gbrain config set search.mode conservative');
// Paid gate must be explicit, not a silent default.
expect(body).toContain('gbrain embed --stale');
});
it('skill is reachable from the skill tree', () => {
expect(existsSync(SKILL_MD)).toBe(true);
});
});
@@ -1,52 +0,0 @@
/**
* E2E smoke for skills/skill-vault-capture-policy.
*
* Verifies the from-trigger-to-side-effect path: a real capture request
* routes to the skill, and the skill documents the vault navigation
* obligations (index.md / log.md) that make a capture durable.
*
* Local-only: it asserts documented behavior, not live vault writes.
*/
import { describe, expect, it } from 'bun:test';
import { existsSync, readFileSync } from 'fs';
import { join } from 'path';
const SKILLS = join(import.meta.dir, '..', '..', 'skills');
const SKILL_MD = join(SKILLS, 'skill-vault-capture-policy', 'SKILL.md');
const RESOLVER = join(SKILLS, 'RESOLVER.md');
const TRIGGER_PHRASES = [
'save this learning to the vault',
'capture this skill in Obsidian',
'record this workflow in my notes',
'put this setup change in the knowledge base',
];
describe('skill-vault-capture-policy E2E', () => {
it('resolver maps real capture phrasings to the skill', () => {
const resolver = readFileSync(RESOLVER, 'utf-8');
expect(resolver).toContain('skill-vault-capture-policy/SKILL.md');
const rows = resolver
.split('\n')
.filter((l) => l.includes('skill-vault-capture-policy/SKILL.md'))
.join('\n');
for (const phrase of TRIGGER_PHRASES) {
const hit = phrase
.toLowerCase()
.split(/\s+/)
.some((tok) => tok.length > 3 && rows.toLowerCase().includes(tok));
expect(hit, `no resolver token for: ${phrase}`).toBe(true);
}
});
it('skill documents index.md and log.md update obligations', () => {
const body = readFileSync(SKILL_MD, 'utf-8');
expect(body).toContain('index.md');
expect(body).toContain('log.md');
});
it('skill is reachable from the skill tree', () => {
expect(existsSync(SKILL_MD)).toBe(true);
});
});
+46
View File
@@ -15,6 +15,7 @@ import { describe, test, expect, beforeAll, afterAll, beforeEach } from 'bun:tes
import { PGLiteEngine } from '../src/core/pglite-engine.ts';
import { resetPgliteState } from './helpers/reset-pglite.ts';
import { embedStaleForSource } from '../src/core/embed-stale.ts';
import { configureGateway, resetGateway } from '../src/core/ai/gateway.ts';
import type { ChunkInput } from '../src/core/types.ts';
let engine: PGLiteEngine;
@@ -276,4 +277,49 @@ describe('embedStaleForSource', () => {
// The stale text row actually got its embedding.
expect(txtRow.embedded_at).not.toBeNull();
});
// #1717: the backfill path must label re-embedded chunks with the model
// that produced the vector, and preserve the existing label on chunks it
// did not touch (before the fix, both were reset to the engine default).
test('labels re-embedded chunks with the gateway model, preserves untouched labels (#1717)', async () => {
configureGateway({
embedding_model: 'openai:text-embedding-3-large',
env: { OPENAI_API_KEY: 'sk-test-embed-stale-1717' },
});
try {
await engine.putPage('notes/model-label', {
type: 'note',
title: 'model-label',
compiled_truth: '# model-label\n\nseeded',
});
await engine.upsertChunks('notes/model-label', [
{
chunk_index: 0,
chunk_text: 'already embedded elsewhere',
chunk_source: 'compiled_truth',
embedding: new Float32Array(1536).fill(0.01),
model: 'voyage:voyage-3',
token_count: 4,
},
{
chunk_index: 1,
chunk_text: 'stale chunk needing embed',
chunk_source: 'compiled_truth',
token_count: 5,
embedding: undefined, // stale
},
]);
const result = await embedStaleForSource(engine, 'default', { embedFn: fakeEmbedFn });
expect(result.embedded).toBe(1);
const after = await engine.getChunks('notes/model-label');
const preserved = after.find((c) => c.chunk_index === 0)!;
const reembedded = after.find((c) => c.chunk_index === 1)!;
expect(reembedded.model).toBe('openai:text-embedding-3-large');
expect(preserved.model).toBe('voyage:voyage-3');
} finally {
resetGateway();
}
});
});
+33
View File
@@ -37,6 +37,8 @@ mock.module('../src/core/embedding.ts', () => ({
// setPageEmbeddingSignature / invalidateStaleSignatureEmbeddings resolve to
// null via the Proxy default, so the signature value is inert here.
currentEmbeddingSignature: () => 'test:model:1536',
// #1717: embed paths stamp this label on (re)embedded chunks.
resolveEmbeddingModelLabel: () => 'openai:text-embedding-3-large',
}));
// Import AFTER mocking.
@@ -803,3 +805,34 @@ describe('embedAllStale --source threading (D7)', () => {
expect((firstCallOpts as { sourceId?: string }).sourceId).toBe('media-corpus');
});
});
// #1717: content_chunks.model must record the model that actually produced
// each vector, not the gateway/engine default.
describe('content_chunks.model labeling (#1717)', () => {
test('stamps the resolved embedding model on re-embedded chunks, preserves it on untouched chunks', async () => {
let upserted: any[] | undefined;
// Chunk 0 is stale (no embedded_at) → gets re-embedded this pass.
// Chunk 1 is already embedded with a DIFFERENT model → must be preserved,
// not relabeled to the current model.
const chunks = [
{ chunk_index: 0, chunk_text: 'a', chunk_source: 'compiled_truth', embedded_at: null, model: 'zeroentropyai:zembed-1', token_count: 1 },
{ chunk_index: 1, chunk_text: 'b', chunk_source: 'compiled_truth', embedded_at: '2026-01-01', embedding: new Float32Array(1536), model: 'voyage:voyage-3', token_count: 1 },
];
const engine = mockEngine({
getPage: async () => ({ slug: 'notes/x', compiled_truth: 'a', timeline: '', source_id: 'default' }),
getChunks: async () => chunks,
upsertChunks: async (_slug: string, c: any[]) => { upserted = c; },
setPageEmbeddingSignature: async () => null,
});
await runEmbedCore(engine, { slugs: ['notes/x'] });
expect(upserted).toBeDefined();
const byIdx = Object.fromEntries(upserted!.map(c => [c.chunk_index, c]));
// Re-embedded chunk carries the model that produced its vector (was
// mislabeled with the default before the fix).
expect(byIdx[0].model).toBe('openai:text-embedding-3-large');
// Untouched chunk keeps its original model — no wholesale relabel.
expect(byIdx[1].model).toBe('voyage:voyage-3');
});
});
@@ -73,4 +73,21 @@ describe('importFromContent embedding_signature stamping (F1)', () => {
await importFromContent(engine, 'concepts/unstamped', '# Unstamped\n\nbody content.', { noEmbed: true });
expect(await signatureOf('concepts/unstamped')).toBeNull();
});
// #1717: content_chunks.model must record the model that produced the
// vector (the configured gateway model), not the engine's hardcoded
// default. The gateway here is configured to openai:text-embedding-3-large,
// which differs from DEFAULT_EMBEDDING_MODEL — so this fails without the
// import-path model stamping.
test('inline embed labels content_chunks.model with the configured model (#1717)', async () => {
await importFromContent(engine, 'concepts/labeled', '# Labeled\n\nsome body content to chunk and embed.', {});
const rows = await engine.executeRaw<{ model: string }>(
`SELECT cc.model FROM content_chunks cc
JOIN pages p ON p.id = cc.page_id
WHERE p.slug = $1 AND p.source_id = 'default'`,
['concepts/labeled'],
);
expect(rows.length).toBeGreaterThan(0);
for (const r of rows) expect(r.model).toBe('openai:text-embedding-3-large');
});
});
-51
View File
@@ -1,51 +0,0 @@
import { describe, expect, it } from 'bun:test';
import { readFileSync, existsSync } from 'fs';
import { join } from 'path';
const SKILL_DIR = join(import.meta.dir, '..', 'skills', 'obsidian-gbrain-safe-index');
const SKILL_MD = join(SKILL_DIR, 'SKILL.md');
const RESOLVER = join(import.meta.dir, '..', 'skills', 'RESOLVER.md');
function parseFrontmatter(raw: string): Record<string, unknown> {
const m = raw.match(/^---\n([\s\S]*?)\n---/);
if (!m) throw new Error('no frontmatter');
const out: Record<string, unknown> = {};
for (const line of m[1].split('\n')) {
const mm = line.match(/^([a-zA-Z_]+):\s*(.*)$/);
if (mm) out[mm[1]] = mm[2].trim();
}
return out;
}
describe('obsidian-gbrain-safe-index skill', () => {
it('has a SKILL.md with required frontmatter', () => {
expect(existsSync(SKILL_MD)).toBe(true);
const fm = parseFrontmatter(readFileSync(SKILL_MD, 'utf-8'));
expect(fm['name']).toBe('obsidian-gbrain-safe-index');
expect(fm['description']).toBeTruthy();
});
it('has the required conformance sections', () => {
const body = readFileSync(SKILL_MD, 'utf-8');
for (const section of ['## Contract', '## Phases', '## Output Format', '## Anti-Patterns']) {
expect(body.includes(section), `missing ${section}`).toBe(true);
}
});
it('is registered in RESOLVER.md', () => {
expect(existsSync(RESOLVER)).toBe(true);
const resolver = readFileSync(RESOLVER, 'utf-8');
expect(resolver.includes('obsidian-gbrain-safe-index/SKILL.md')).toBe(true);
});
it('has routing-eval fixtures that exercise real trigger phrasings', () => {
const evalPath = join(SKILL_DIR, 'routing-eval.jsonl');
expect(existsSync(evalPath)).toBe(true);
const lines = readFileSync(evalPath, 'utf-8')
.split('\n')
.filter((l) => l.trim() && !l.trim().startsWith('//'))
.map((l) => JSON.parse(l));
const positives = lines.filter((l) => l.expected_skill === 'obsidian-gbrain-safe-index');
expect(positives.length).toBeGreaterThanOrEqual(5);
});
});
-64
View File
@@ -1,64 +0,0 @@
import { describe, expect, it } from 'bun:test';
import { readFileSync, existsSync } from 'fs';
import { join } from 'path';
import { DEFAULT_PRIVATE_PATTERNS } from '../src/core/skillpack/harvest-lint.ts';
const SKILL_DIR = join(import.meta.dir, '..', 'skills', 'skill-vault-capture-policy');
const SKILL_MD = join(SKILL_DIR, 'SKILL.md');
const RESOLVER = join(import.meta.dir, '..', 'skills', 'RESOLVER.md');
function parseFrontmatter(raw: string): Record<string, unknown> {
const m = raw.match(/^---\n([\s\S]*?)\n---/);
if (!m) throw new Error('no frontmatter');
const out: Record<string, unknown> = {};
for (const line of m[1].split('\n')) {
const mm = line.match(/^([a-zA-Z_]+):\s*(.*)$/);
if (mm) out[mm[1]] = mm[2].trim();
}
return out;
}
describe('skill-vault-capture-policy skill', () => {
it('has a SKILL.md with required frontmatter', () => {
expect(existsSync(SKILL_MD)).toBe(true);
const fm = parseFrontmatter(readFileSync(SKILL_MD, 'utf-8'));
expect(fm['name']).toBe('skill-vault-capture-policy');
expect(fm['description']).toBeTruthy();
});
it('has the required conformance sections', () => {
const body = readFileSync(SKILL_MD, 'utf-8');
for (const section of ['## Contract', '## Phases', '## Output Format', '## Anti-Patterns']) {
expect(body.includes(section), `missing ${section}`).toBe(true);
}
});
it('is registered in RESOLVER.md', () => {
expect(existsSync(RESOLVER)).toBe(true);
expect(readFileSync(RESOLVER, 'utf-8').includes('skill-vault-capture-policy/SKILL.md')).toBe(true);
});
it('contains no private user or agent-fork names (privacy rule)', () => {
for (const file of [SKILL_MD, join(SKILL_DIR, 'routing-eval.jsonl')]) {
const body = readFileSync(file, 'utf-8');
// DEFAULT_PRIVATE_PATTERNS[0] is the banned fork-name pattern; sourced
// from harvest-lint so this file never contains the literal itself
// (scripts/check-privacy.sh would reject it).
const forkName = new RegExp(DEFAULT_PRIVATE_PATTERNS[0], 'i');
for (const name of [/\bAdam\b/, /\bHermes\b/, /\bHerdr\b/, /\bArk\b/, forkName]) {
expect(name.test(body), `private name ${name} in ${file}`).toBe(false);
}
}
});
it('has routing-eval fixtures', () => {
const evalPath = join(SKILL_DIR, 'routing-eval.jsonl');
expect(existsSync(evalPath)).toBe(true);
const positives = readFileSync(evalPath, 'utf-8')
.split('\n')
.filter((l) => l.trim() && !l.trim().startsWith('//'))
.map((l) => JSON.parse(l))
.filter((l) => l.expected_skill === 'skill-vault-capture-policy');
expect(positives.length).toBeGreaterThanOrEqual(4);
});
});