mirror of
https://github.com/garrytan/gbrain.git
synced 2026-08-16 01:42:23 +00:00
Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
88287775e5 |
@@ -71,8 +71,8 @@ GBrain is designed to be installed and operated by an AI agent. The fastest path
|
||||
|
||||
If you don't already have an AI agent platform running, start with one of these. Both are designed to read GBrain's install protocol and execute it:
|
||||
|
||||
- **[OpenClaw](https://github.com/openclaw/openclaw)** — deploy [AlphaClaw on Render](https://render.com/deploy?repo=https://github.com/chrysb/alphaclaw) (one click, 8GB+ RAM)
|
||||
- **[Hermes](https://github.com/NousResearch/hermes-agent)** — deploy on [Railway](https://github.com/praveen-ks-2001/hermes-agent-template) (one click)
|
||||
- **[OpenClaw](https://github.com/openclawagents/openclaw)** — deploy [AlphaClaw on Render](https://render.com/deploy?repo=https://github.com/chrysb/alphaclaw) (one click, 8GB+ RAM)
|
||||
- **[Hermes](https://github.com/openclawagents/hermes)** — deploy on [Railway](https://github.com/praveen-ks-2001/hermes-agent-template) (one click)
|
||||
|
||||
Then paste this into your agent:
|
||||
|
||||
|
||||
@@ -131,9 +131,7 @@ into gbrain so other clients can scaffold it. Default behavior:
|
||||
`~/.gbrain/harvest-private-patterns.txt` plus built-in defaults
|
||||
(canonical private fork name, common email regex, Slack channel pattern). Any
|
||||
match → rollback (delete the harvested files) and exit non-zero.
|
||||
- `openclaw.plugin.json` updated with the new slug, sorted. Harvest must preserve
|
||||
the top-level OpenClaw-native plugin fields (`id`, `configSchema`, `contracts`)
|
||||
because OpenClaw validates those before it can install the package.
|
||||
- `openclaw.plugin.json` updated with the new slug, sorted.
|
||||
- `--no-lint` bypasses the linter (after a manual editorial scrub).
|
||||
|
||||
Use the `skillpack-harvest` skill (its companion editorial workflow)
|
||||
|
||||
@@ -233,14 +233,13 @@ keep it or `git checkout` to throw it away. Nothing is committed for you.
|
||||
|
||||
**For a skill that ships with gbrain** (anything under the gbrain repo's own
|
||||
`skills/`): SkillOpt refuses to overwrite it by default and writes the winner to
|
||||
`skills/<name>/skillopt/proposed.md` instead (while keeping `best.md` as the
|
||||
optimizer's current-best pointer), so an optimization pass can never silently
|
||||
mutate a skill other people depend on. Two ways to handle that:
|
||||
`skills/<name>/skillopt/best.md` instead, so an optimization pass can never
|
||||
silently mutate a skill other people depend on. Two ways to handle that:
|
||||
|
||||
```bash
|
||||
# See the proposed improvement without touching SKILL.md (works for ANY skill):
|
||||
gbrain skillopt meeting-prep --split 1:1:1 --no-mutate
|
||||
# → writes skills/meeting-prep/skillopt/proposed.md, updates best.md, and prints the proposal path.
|
||||
# → writes skills/meeting-prep/skillopt/best.md (the proposed rewrite), prints its path. Copy what you want.
|
||||
|
||||
# Actually rewrite a bundled skill (explicit opt-in + an independent held-out set):
|
||||
gbrain skillopt brain-ops --split 1:1:1 --allow-mutate-bundled \
|
||||
|
||||
+2
-2
@@ -1565,8 +1565,8 @@ GBrain is designed to be installed and operated by an AI agent. The fastest path
|
||||
|
||||
If you don't already have an AI agent platform running, start with one of these. Both are designed to read GBrain's install protocol and execute it:
|
||||
|
||||
- **[OpenClaw](https://github.com/openclaw/openclaw)** — deploy [AlphaClaw on Render](https://render.com/deploy?repo=https://github.com/chrysb/alphaclaw) (one click, 8GB+ RAM)
|
||||
- **[Hermes](https://github.com/NousResearch/hermes-agent)** — deploy on [Railway](https://github.com/praveen-ks-2001/hermes-agent-template) (one click)
|
||||
- **[OpenClaw](https://github.com/openclawagents/openclaw)** — deploy [AlphaClaw on Render](https://render.com/deploy?repo=https://github.com/chrysb/alphaclaw) (one click, 8GB+ RAM)
|
||||
- **[Hermes](https://github.com/openclawagents/hermes)** — deploy on [Railway](https://github.com/praveen-ks-2001/hermes-agent-template) (one click)
|
||||
|
||||
Then paste this into your agent:
|
||||
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
{
|
||||
"id": "gbrain-context-engine",
|
||||
"name": "gbrain",
|
||||
"version": "0.32.3.0",
|
||||
"description": "Personal knowledge brain with Postgres + pgvector hybrid search",
|
||||
|
||||
@@ -266,5 +266,4 @@ editorial pass.
|
||||
(e.g. `src/commands/<slug>.ts` if the host SKILL.md declares it
|
||||
in frontmatter)
|
||||
- gbrain's `openclaw.plugin.json` — adds the slug to `skills:`
|
||||
array, sorted alphabetically, without removing OpenClaw-native plugin fields
|
||||
like `id`, `configSchema`, or `contracts`
|
||||
array, sorted alphabetically
|
||||
|
||||
@@ -57,8 +57,6 @@ This mode guarantees:
|
||||
- `skills/manifest.json` lists every skill directory
|
||||
- `skills/RESOLVER.md` references every skill in the manifest
|
||||
- `openclaw.plugin.json` `skills[]` round-trips with both
|
||||
- `openclaw.plugin.json` keeps OpenClaw install-required native plugin fields
|
||||
(`id`, object `configSchema`, and `contracts.contextEngines` when applicable)
|
||||
- No MECE violations (duplicate triggers across skills)
|
||||
|
||||
### Phases
|
||||
@@ -74,7 +72,7 @@ This mode guarantees:
|
||||
### Automation
|
||||
|
||||
```bash
|
||||
bun test test/skills-conformance.test.ts test/resolver.test.ts test/openclaw-plugin-manifest.test.ts
|
||||
bun test test/skills-conformance.test.ts test/resolver.test.ts
|
||||
```
|
||||
|
||||
The CI-gated check is the package.json `test` script.
|
||||
|
||||
+1
-50
@@ -54,7 +54,7 @@ export function bigintToStringReplacer(_key: string, value: unknown): unknown {
|
||||
}
|
||||
|
||||
// CLI-only commands that bypass the operation layer
|
||||
export const CLI_ONLY = new Set(['init', 'reinit-pglite', 'upgrade', 'post-upgrade', 'check-update', 'integrations', 'publish', 'check-backlinks', 'lint', 'report', 'import', 'export', 'files', 'embed', 'serve', 'call', 'config', 'doctor', 'migrate', 'eval', 'bench', 'sync', 'extract', 'extract-conversation-facts', 'enrich', 'features', 'autopilot', 'graph-query', 'jobs', 'agent', 'apply-migrations', 'skillpack-check', 'skillpack', 'resolvers', 'integrity', 'repair-jsonb', 'orphans', 'maintain', 'sources', 'mounts', 'dream', 'check-resolvable', 'routing-eval', 'skillify', 'smoke-test', 'providers', 'storage', 'repos', 'code-def', 'code-refs', 'reindex', 'reindex-code', 'reindex-frontmatter', 'code-callers', 'code-callees', 'reconcile-links', 'frontmatter', 'auth', 'friction', 'claw-test', 'book-mirror', 'takes', 'think', 'salience', 'anomalies', 'calibration', 'transcripts', 'models', 'remote', 'recall', 'forget', 'edges-backfill', 'cache', 'ze-switch', 'founder', 'brainstorm', 'lsd', 'schema', 'capture', 'onboard', 'conversation-parser', 'status', 'connect', 'skillopt', 'quarantine', 'self-upgrade', 'advisor', 'watch', 'reindex-search-vector']);
|
||||
export const CLI_ONLY = new Set(['init', 'reinit-pglite', 'upgrade', 'post-upgrade', 'check-update', 'integrations', 'publish', 'check-backlinks', 'lint', 'report', 'import', 'export', 'files', 'embed', 'serve', 'call', 'config', 'doctor', 'migrate', 'eval', 'sync', 'extract', 'extract-conversation-facts', 'enrich', 'features', 'autopilot', 'graph-query', 'jobs', 'agent', 'apply-migrations', 'skillpack-check', 'skillpack', 'resolvers', 'integrity', 'repair-jsonb', 'orphans', 'sources', 'mounts', 'dream', 'check-resolvable', 'routing-eval', 'skillify', 'smoke-test', 'providers', 'storage', 'repos', 'code-def', 'code-refs', 'reindex', 'reindex-code', 'reindex-frontmatter', 'code-callers', 'code-callees', 'reconcile-links', 'frontmatter', 'auth', 'friction', 'claw-test', 'book-mirror', 'takes', 'think', 'salience', 'anomalies', 'calibration', 'transcripts', 'models', 'remote', 'recall', 'forget', 'edges-backfill', 'cache', 'ze-switch', 'founder', 'brainstorm', 'lsd', 'schema', 'capture', 'onboard', 'conversation-parser', 'status', 'connect', 'skillopt', 'quarantine', 'self-upgrade', 'advisor', 'watch', 'reindex-search-vector']);
|
||||
// CLI-only commands whose handlers print their own --help text. These are
|
||||
// excluded from the generic short-circuit so detailed per-command and
|
||||
// per-subcommand usage stays reachable.
|
||||
@@ -78,8 +78,6 @@ const CLI_ONLY_SELF_HELP = new Set([
|
||||
'capture',
|
||||
// v0.42 self-upgrade ships its own usage (flags + the agent-skill story).
|
||||
'self-upgrade',
|
||||
// maintain (#3015) prints its own usage block (modes + not-auto-applied list).
|
||||
'maintain',
|
||||
// v0.43 (#2095): watch ships WATCH_HELP (flags + the stdin-turn protocol).
|
||||
'watch',
|
||||
// v0.37 fix wave (Lane D.4 + CDX2-12): sync's --no-embed flag was
|
||||
@@ -106,9 +104,6 @@ const CLI_ONLY_SELF_HELP = new Set([
|
||||
// `gbrain connect --help` prints its own usage (flags + examples) from
|
||||
// runConnect; route around the generic one-line short-circuit.
|
||||
'connect',
|
||||
// #1474: bench-publish ships its own detailed HELP (flags, exit codes,
|
||||
// the export → publish → gate loop). Route around the generic stub.
|
||||
'bench',
|
||||
]);
|
||||
|
||||
// v114 (#1941): alias -> operation lookup, kept separate from `cliOps` so
|
||||
@@ -1432,29 +1427,6 @@ async function handleCliOnly(command: string, args: string[]) {
|
||||
return;
|
||||
}
|
||||
|
||||
// #1474: `gbrain bench publish` is pure file I/O (reads a captured
|
||||
// eval-candidates NDJSON from `gbrain eval export`, writes a baseline
|
||||
// NDJSON). No DB access; bypass connectEngine entirely so the documented
|
||||
// export → publish → gate loop works on machines without a brain.
|
||||
// The v0.41.1 wave shipped bench-publish.ts + docs/eval-bench.md but this
|
||||
// dispatcher case was never added, so the command hit 'Unknown command'.
|
||||
if (command === 'bench') {
|
||||
if (args[0] === 'publish') {
|
||||
const { runBenchPublish } = await import('./commands/bench-publish.ts');
|
||||
await runBenchPublish(args.slice(1));
|
||||
return;
|
||||
}
|
||||
if (args.length === 0 || args[0] === '--help' || args[0] === '-h') {
|
||||
const { runBenchPublish } = await import('./commands/bench-publish.ts');
|
||||
await runBenchPublish(['--help']);
|
||||
return;
|
||||
}
|
||||
console.error(`Unknown bench subcommand: ${args[0]}`);
|
||||
console.error('Usage: gbrain bench publish --from <captured.ndjson> --to <baseline.ndjson> [flags]');
|
||||
console.error(' See docs/eval-bench.md for the full loop: eval export → bench publish → eval gate');
|
||||
process.exit(2);
|
||||
}
|
||||
|
||||
// v0.42.x (#2390): `gbrain eval chronicle` is deterministic — brings its own
|
||||
// in-memory PGLite, no DB/gateway. CI fixture gate runs anywhere.
|
||||
if (command === 'eval' && args[0] === 'chronicle') {
|
||||
@@ -1785,11 +1757,6 @@ async function handleCliOnly(command: string, args: string[]) {
|
||||
await runOrphans(engine, args);
|
||||
break;
|
||||
}
|
||||
case 'maintain': {
|
||||
const { runMaintain } = await import('./commands/maintain.ts');
|
||||
await runMaintain(engine, args);
|
||||
break;
|
||||
}
|
||||
// v0.32.7 CJK wave — post-upgrade markdown re-chunk sweep.
|
||||
// v0.36 Phase 3 wave — `gbrain reindex --multimodal` re-embeds content_chunks
|
||||
// into the unified Voyage multimodal-3 column.
|
||||
@@ -2250,22 +2217,6 @@ async function connectEngine(opts?: { probeOnly?: boolean }): Promise<BrainEngin
|
||||
if (merged.embedding_image_ocr_model !== undefined) {
|
||||
process.env.GBRAIN_EMBEDDING_IMAGE_OCR_MODEL = merged.embedding_image_ocr_model;
|
||||
}
|
||||
// #1475: stash the merged eval.* flags the same way. The capture gate
|
||||
// (isEvalCaptureEnabled / isEvalScrubEnabled) runs against ctx.config,
|
||||
// which is built from the sync file-plane loadConfig() in both the CLI
|
||||
// op path and MCP dispatch — it never sees the DB plane directly. The
|
||||
// gates consult this stash when the file plane is silent, so
|
||||
// `gbrain config set eval.capture true` actually turns capture on.
|
||||
// A pre-set env value wins over the DB plane (env-above-config, the
|
||||
// incident escape hatch) — unlike GBRAIN_EMBEDDING_MULTIMODAL these
|
||||
// keys have no loadConfig() env mapping, so without this guard the
|
||||
// DB stash would silently clobber an operator's export.
|
||||
if (process.env.GBRAIN_EVAL_CAPTURE === undefined && merged.eval?.capture !== undefined) {
|
||||
process.env.GBRAIN_EVAL_CAPTURE = String(merged.eval.capture);
|
||||
}
|
||||
if (process.env.GBRAIN_EVAL_SCRUB_PII === undefined && merged.eval?.scrub_pii !== undefined) {
|
||||
process.env.GBRAIN_EVAL_SCRUB_PII = String(merged.eval.scrub_pii);
|
||||
}
|
||||
// Always re-configure with merged values when DB merge succeeded. The
|
||||
// trigger used to be field-name-gated (only when embedding_multimodal_model
|
||||
// was set); that coupled the gate to the field set and would silently
|
||||
|
||||
+4
-34
@@ -581,7 +581,7 @@ async function embedPage(
|
||||
for (let j = 0; j < toEmbed.length; j++) {
|
||||
embeddingMap.set(toEmbed[j].chunk_index, embeddings[j]);
|
||||
}
|
||||
const updated: ChunkInput[] = chunks.map(c => preserveCodeMetadata(c, {
|
||||
const updated: ChunkInput[] = chunks.map(c => ({
|
||||
chunk_index: c.chunk_index,
|
||||
chunk_text: c.chunk_text,
|
||||
chunk_source: c.chunk_source,
|
||||
@@ -605,31 +605,6 @@ async function embedPage(
|
||||
slog(`${slug}: embedded ${toEmbed.length} chunks`);
|
||||
}
|
||||
|
||||
/**
|
||||
* Carry code-chunk metadata (language, symbol_name, symbol_type, line range,
|
||||
* parent scope, doc comment, qualified name) from a loaded Chunk back into a
|
||||
* ChunkInput destined for upsertChunks.
|
||||
*
|
||||
* Issue #769: every re-embed used to strip these fields, and upsertChunks
|
||||
* overwrites (does not COALESCE) the metadata columns from EXCLUDED, so
|
||||
* each pass clobbered code-def's primary index to NULL. Pulling the
|
||||
* preservation into one helper keeps the three re-embed call sites
|
||||
* (embedPage, embedAll non-stale, embedAllStale) in lock-step.
|
||||
*/
|
||||
function preserveCodeMetadata(loaded: any, base: ChunkInput): ChunkInput {
|
||||
return {
|
||||
...base,
|
||||
language: loaded.language ?? undefined,
|
||||
symbol_name: loaded.symbol_name ?? undefined,
|
||||
symbol_type: loaded.symbol_type ?? undefined,
|
||||
start_line: loaded.start_line ?? undefined,
|
||||
end_line: loaded.end_line ?? undefined,
|
||||
parent_symbol_path: loaded.parent_symbol_path ?? undefined,
|
||||
doc_comment: loaded.doc_comment ?? undefined,
|
||||
symbol_name_qualified: loaded.symbol_name_qualified ?? undefined,
|
||||
};
|
||||
}
|
||||
|
||||
async function embedAll(
|
||||
engine: BrainEngine,
|
||||
staleOnly: boolean,
|
||||
@@ -742,10 +717,8 @@ async function embedAll(
|
||||
for (let j = 0; j < toEmbed.length; j++) {
|
||||
embeddingMap.set(toEmbed[j].chunk_index, embeddings[j]);
|
||||
}
|
||||
// Preserve ALL chunks, only update embeddings for stale ones.
|
||||
// preserveCodeMetadata threads code-chunk metadata (#769) so re-embed
|
||||
// doesn't clobber language/symbol_name/symbol_type to NULL.
|
||||
const updated: ChunkInput[] = chunks.map(c => preserveCodeMetadata(c, {
|
||||
// Preserve ALL chunks, only update embeddings for stale ones
|
||||
const updated: ChunkInput[] = chunks.map(c => ({
|
||||
chunk_index: c.chunk_index,
|
||||
chunk_text: c.chunk_text,
|
||||
chunk_source: c.chunk_source,
|
||||
@@ -1039,10 +1012,7 @@ async function embedAllStale(
|
||||
for (let j = 0; j < stale.length; j++) {
|
||||
staleIdxToEmbedding.set(stale[j].chunk_index, embeddings[j]);
|
||||
}
|
||||
// preserveCodeMetadata threads code-chunk metadata (#769) so the
|
||||
// autopilot --stale path doesn't clobber language/symbol_name/etc
|
||||
// to NULL on every cycle.
|
||||
const merged: ChunkInput[] = existing.map(c => preserveCodeMetadata(c, {
|
||||
const merged: ChunkInput[] = existing.map(c => ({
|
||||
chunk_index: c.chunk_index,
|
||||
chunk_text: c.chunk_text,
|
||||
chunk_source: c.chunk_source,
|
||||
|
||||
@@ -1651,7 +1651,7 @@ async function extractTimelineFromDB(
|
||||
* make re-extraction idempotent). EVERY processed page is stamped, including
|
||||
* zero-link pages — they WERE processed.
|
||||
*/
|
||||
export async function extractStaleFromDB(
|
||||
async function extractStaleFromDB(
|
||||
engine: BrainEngine,
|
||||
opts: {
|
||||
dryRun: boolean;
|
||||
|
||||
@@ -1,224 +0,0 @@
|
||||
/**
|
||||
* gbrain maintain — conservative self-healing maintenance.
|
||||
*
|
||||
* This command automates the safe parts of the operator runbook:
|
||||
* - stale link/timeline extraction
|
||||
* - stale per-source dream cycles when doctor reports cycle_freshness
|
||||
*
|
||||
* It deliberately does NOT mutate source files, apply schema-pack upgrades, or
|
||||
* invent semantic hub links. Those need review or a separate command with an
|
||||
* auditable proposal surface.
|
||||
*/
|
||||
|
||||
import { existsSync } from 'fs';
|
||||
import type { BrainEngine } from '../core/engine.ts';
|
||||
import type { BrainHealth } from '../core/types.ts';
|
||||
import { buildChecks, computeDoctorReport, type DoctorReport, type Check } from './doctor.ts';
|
||||
import { extractStaleFromDB } from './extract.ts';
|
||||
import { runCycle, type CycleReport } from '../core/cycle.ts';
|
||||
|
||||
type ActionStatus = 'ok' | 'would_apply' | 'applied' | 'blocked' | 'skipped';
|
||||
|
||||
export interface MaintenanceAction {
|
||||
name: string;
|
||||
status: ActionStatus;
|
||||
message: string;
|
||||
details?: Record<string, unknown>;
|
||||
}
|
||||
|
||||
export interface MaintainOptions {
|
||||
json: boolean;
|
||||
safe: boolean;
|
||||
dryRun: boolean;
|
||||
help: boolean;
|
||||
}
|
||||
|
||||
export interface MaintainReport {
|
||||
mode: 'dry-run' | 'safe';
|
||||
before: {
|
||||
health: BrainHealth;
|
||||
doctor: DoctorReport;
|
||||
};
|
||||
actions: MaintenanceAction[];
|
||||
after: {
|
||||
health: BrainHealth;
|
||||
doctor: DoctorReport;
|
||||
};
|
||||
}
|
||||
|
||||
export function parseMaintainArgs(args: string[]): MaintainOptions {
|
||||
const safe = args.includes('--safe');
|
||||
return {
|
||||
json: args.includes('--json'),
|
||||
safe,
|
||||
dryRun: args.includes('--dry-run') || !safe,
|
||||
help: args.includes('--help') || args.includes('-h'),
|
||||
};
|
||||
}
|
||||
|
||||
export function extractCycleFreshnessSourceIds(checks: Check[]): string[] {
|
||||
const ids = new Set<string>();
|
||||
for (const check of checks) {
|
||||
if (check.name !== 'cycle_freshness' || check.status === 'ok') continue;
|
||||
const re = /Source '([^']+)' last cycled/g;
|
||||
for (const match of check.message.matchAll(re)) {
|
||||
const id = match[1]?.trim();
|
||||
if (id) ids.add(id);
|
||||
}
|
||||
}
|
||||
return [...ids].sort();
|
||||
}
|
||||
|
||||
async function buildDoctorReport(engine: BrainEngine): Promise<DoctorReport> {
|
||||
const checks = await buildChecks(engine, ['--json', '--scope=brain']);
|
||||
return computeDoctorReport(checks);
|
||||
}
|
||||
|
||||
async function runStaleExtraction(
|
||||
engine: BrainEngine,
|
||||
beforeHealth: BrainHealth,
|
||||
dryRun: boolean,
|
||||
): Promise<MaintenanceAction> {
|
||||
if (beforeHealth.stale_pages <= 0) {
|
||||
return { name: 'extract_stale', status: 'ok', message: 'No stale pages.' };
|
||||
}
|
||||
|
||||
if (dryRun) {
|
||||
return {
|
||||
name: 'extract_stale',
|
||||
status: 'would_apply',
|
||||
message: `Would run DB-backed stale extraction for ${beforeHealth.stale_pages} page(s).`,
|
||||
details: { stale_pages: beforeHealth.stale_pages },
|
||||
};
|
||||
}
|
||||
|
||||
const result = await extractStaleFromDB(engine, {
|
||||
dryRun: false,
|
||||
jsonMode: false,
|
||||
includeFrontmatter: false,
|
||||
catchUp: false,
|
||||
});
|
||||
|
||||
return {
|
||||
name: 'extract_stale',
|
||||
status: 'applied',
|
||||
message: `Processed ${result.pagesProcessed} stale page(s); ${result.staleRemaining} remain.`,
|
||||
details: {
|
||||
links_created: result.linksCreated,
|
||||
timeline_created: result.timelineCreated,
|
||||
pages_processed: result.pagesProcessed,
|
||||
stale_remaining: result.staleRemaining,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
async function runCycleFreshnessMaintenance(
|
||||
engine: BrainEngine,
|
||||
beforeDoctor: DoctorReport,
|
||||
dryRun: boolean,
|
||||
): Promise<MaintenanceAction[]> {
|
||||
const sourceIds = extractCycleFreshnessSourceIds(beforeDoctor.checks);
|
||||
if (sourceIds.length === 0) {
|
||||
return [{ name: 'cycle_freshness', status: 'ok', message: 'All sources cycled recently.' }];
|
||||
}
|
||||
|
||||
if (dryRun) {
|
||||
return sourceIds.map((sourceId) => ({
|
||||
name: 'cycle_freshness',
|
||||
status: 'would_apply',
|
||||
message: `Would run source-scoped dream cycle for ${sourceId}.`,
|
||||
details: { source_id: sourceId },
|
||||
}));
|
||||
}
|
||||
|
||||
const sources = await engine.listAllSources();
|
||||
const actions: MaintenanceAction[] = [];
|
||||
|
||||
for (const sourceId of sourceIds) {
|
||||
const source = sources.find((s) => s.id === sourceId);
|
||||
const localPath = source?.local_path ?? null;
|
||||
const brainDir = localPath && existsSync(localPath) ? localPath : null;
|
||||
const report: CycleReport = await runCycle(engine, {
|
||||
brainDir,
|
||||
dryRun: false,
|
||||
pull: false,
|
||||
sourceId,
|
||||
});
|
||||
actions.push({
|
||||
name: 'cycle_freshness',
|
||||
status: report.status === 'failed' ? 'blocked' : 'applied',
|
||||
message: `Ran source-scoped dream cycle for ${sourceId}: ${report.status}.`,
|
||||
details: {
|
||||
source_id: sourceId,
|
||||
brain_dir: brainDir,
|
||||
cycle_status: report.status,
|
||||
phases: report.phases.map((p) => ({ phase: p.phase, status: p.status })),
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
return actions;
|
||||
}
|
||||
|
||||
export async function runMaintain(engine: BrainEngine, args: string[]): Promise<MaintainReport | void> {
|
||||
const opts = parseMaintainArgs(args);
|
||||
if (opts.help) {
|
||||
console.log(`Usage: gbrain maintain [--safe] [--dry-run] [--json]
|
||||
|
||||
Conservative self-healing maintenance.
|
||||
|
||||
Modes:
|
||||
--dry-run Preview safe actions without writes. Default when --safe is absent.
|
||||
--safe Apply safe actions: stale extraction and source cycle freshness.
|
||||
--json Emit a structured before/action/after report.
|
||||
|
||||
Not auto-applied:
|
||||
source-file frontmatter fixes, schema-pack upgrades, atom-pack changes,
|
||||
semantic hub-link guesses, and destructive cleanup.
|
||||
`);
|
||||
return;
|
||||
}
|
||||
|
||||
const beforeHealth = await engine.getHealth();
|
||||
const beforeDoctor = await buildDoctorReport(engine);
|
||||
const actions: MaintenanceAction[] = [];
|
||||
|
||||
actions.push(await runStaleExtraction(engine, beforeHealth, opts.dryRun));
|
||||
actions.push(...await runCycleFreshnessMaintenance(engine, beforeDoctor, opts.dryRun));
|
||||
|
||||
const afterHealth = await engine.getHealth();
|
||||
const afterDoctor = await buildDoctorReport(engine);
|
||||
const report: MaintainReport = {
|
||||
mode: opts.dryRun ? 'dry-run' : 'safe',
|
||||
before: { health: beforeHealth, doctor: beforeDoctor },
|
||||
actions,
|
||||
after: { health: afterHealth, doctor: afterDoctor },
|
||||
};
|
||||
|
||||
if (opts.json) {
|
||||
console.log(JSON.stringify(report, null, 2));
|
||||
} else {
|
||||
printMaintainReport(report);
|
||||
}
|
||||
return report;
|
||||
}
|
||||
|
||||
function printMaintainReport(report: MaintainReport): void {
|
||||
console.log(`GBrain maintain (${report.mode})`);
|
||||
console.log(
|
||||
`Before: brain_score=${Math.round(report.before.health.brain_score)}/100 ` +
|
||||
`stale=${report.before.health.stale_pages} islands=${report.before.health.orphan_pages} ` +
|
||||
`doctor=${report.before.doctor.status}`,
|
||||
);
|
||||
for (const action of report.actions) {
|
||||
console.log(` ${action.status}: ${action.name} — ${action.message}`);
|
||||
}
|
||||
console.log(
|
||||
`After: brain_score=${Math.round(report.after.health.brain_score)}/100 ` +
|
||||
`stale=${report.after.health.stale_pages} islands=${report.after.health.orphan_pages} ` +
|
||||
`doctor=${report.after.doctor.status}`,
|
||||
);
|
||||
if (report.mode === 'dry-run') {
|
||||
console.log('Run `gbrain maintain --safe` to apply safe actions.');
|
||||
}
|
||||
}
|
||||
+55
-10
@@ -15,11 +15,6 @@
|
||||
import type { BrainEngine } from '../core/engine.ts';
|
||||
import { createProgress, startHeartbeat } from '../core/progress.ts';
|
||||
import { getCliOptions, cliOptsToProgressOptions } from '../core/cli-options.ts';
|
||||
import {
|
||||
shouldExcludeFromOrphanReporting,
|
||||
loadOrphanPolicyOverrides,
|
||||
type OrphanPolicyOverrides,
|
||||
} from '../core/orphan-policy.ts';
|
||||
|
||||
// --- Types ---
|
||||
|
||||
@@ -37,14 +32,65 @@ export interface OrphanResult {
|
||||
excluded: number;
|
||||
}
|
||||
|
||||
// --- Filter constants ---
|
||||
|
||||
/** Slug suffixes that are always auto-generated root files */
|
||||
const AUTO_SUFFIX_PATTERNS = ['/_index', '/log'];
|
||||
|
||||
/** Page slugs that are pseudo-pages by convention */
|
||||
const PSEUDO_SLUGS = new Set(['_atlas', '_index', '_stats', '_orphans', '_scratch', 'claude']);
|
||||
|
||||
/** Slug segment that marks raw sources */
|
||||
const RAW_SEGMENT = '/raw/';
|
||||
|
||||
/** Slug prefixes where no inbound links is expected */
|
||||
const DENY_PREFIXES = [
|
||||
'output/',
|
||||
'dashboards/',
|
||||
'scripts/',
|
||||
'templates/',
|
||||
'openclaw/config/',
|
||||
];
|
||||
|
||||
/** First slug segments where no inbound links is expected */
|
||||
const FIRST_SEGMENT_EXCLUSIONS = new Set([
|
||||
'scratch',
|
||||
'thoughts',
|
||||
'catalog',
|
||||
'entities',
|
||||
'raw',
|
||||
'atoms',
|
||||
'skills',
|
||||
]);
|
||||
|
||||
// --- Filter logic ---
|
||||
|
||||
/**
|
||||
* Returns true if a slug should be excluded from orphan reporting by default.
|
||||
* These are pages where having no inbound links is expected / not a content problem.
|
||||
*/
|
||||
export function shouldExclude(slug: string, overrides?: OrphanPolicyOverrides): boolean {
|
||||
return shouldExcludeFromOrphanReporting(slug, overrides);
|
||||
export function shouldExclude(slug: string): boolean {
|
||||
// Pseudo-pages (exact match)
|
||||
if (PSEUDO_SLUGS.has(slug)) return true;
|
||||
|
||||
// Auto-generated suffix patterns
|
||||
for (const suffix of AUTO_SUFFIX_PATTERNS) {
|
||||
if (slug.endsWith(suffix)) return true;
|
||||
}
|
||||
|
||||
// Raw source slugs
|
||||
if (slug.includes(RAW_SEGMENT)) return true;
|
||||
|
||||
// Deny-prefix slugs
|
||||
for (const prefix of DENY_PREFIXES) {
|
||||
if (slug.startsWith(prefix)) return true;
|
||||
}
|
||||
|
||||
// First-segment exclusions
|
||||
const firstSegment = slug.split('/')[0];
|
||||
if (FIRST_SEGMENT_EXCLUSIONS.has(firstSegment)) return true;
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -110,7 +156,6 @@ export async function findOrphans(
|
||||
let allOrphans: { slug: string; title: string; domain: string | null }[];
|
||||
let total: number;
|
||||
let excludedAll: number;
|
||||
const overrides = includePseudo ? undefined : await loadOrphanPolicyOverrides(engine);
|
||||
try {
|
||||
allOrphans = await engine.findOrphanPages(
|
||||
sourceIds ? { sourceIds } : sourceId ? { sourceId } : undefined,
|
||||
@@ -139,7 +184,7 @@ export async function findOrphans(
|
||||
total = liveRows.length;
|
||||
excludedAll = includePseudo
|
||||
? 0
|
||||
: liveRows.reduce((n, r) => n + (shouldExclude(r.slug, overrides) ? 1 : 0), 0);
|
||||
: liveRows.reduce((n, r) => n + (shouldExclude(r.slug) ? 1 : 0), 0);
|
||||
} finally {
|
||||
stopHb();
|
||||
progress.finish();
|
||||
@@ -147,7 +192,7 @@ export async function findOrphans(
|
||||
|
||||
const filtered = includePseudo
|
||||
? allOrphans
|
||||
: allOrphans.filter(row => !shouldExclude(row.slug, overrides));
|
||||
: allOrphans.filter(row => !shouldExclude(row.slug));
|
||||
|
||||
const orphans: OrphanPage[] = filtered.map(row => ({
|
||||
slug: row.slug,
|
||||
|
||||
@@ -18,8 +18,9 @@
|
||||
* at runtime.
|
||||
*/
|
||||
|
||||
import { chunkText as recursiveChunk } from './recursive.ts';
|
||||
import { chunkText as recursiveChunk, capByEstimatedTokens, DEFAULT_MAX_EST_TOKENS } from './recursive.ts';
|
||||
import { buildQualifiedName } from './qualified-names.ts';
|
||||
import { estimateEmbeddingTokens } from '../cjk.ts';
|
||||
|
||||
// Embed the tree-sitter runtime + per-language grammars as files.
|
||||
// `with { type: 'file' }` returns a path (string) at runtime. Bun bundles
|
||||
@@ -111,7 +112,15 @@ import G_ZIG from '../../assets/wasm/grammars/tree-sitter-zig.wasm' with { type:
|
||||
// chunks get the new columns populated. Without this, the v28 backfill
|
||||
// gives every existing chunk a search_vector but subsequent Layer 5 AST
|
||||
// work would silently no-op.
|
||||
export const CHUNKER_VERSION = 4;
|
||||
//
|
||||
// v5: estimated-token hard cap on AST-path chunks (capCodeChunks). A node
|
||||
// splitLargeNode can't subdivide (giant single-statement function, huge
|
||||
// literal) previously shipped WHOLE regardless of size and could overflow
|
||||
// strict per-request embedding-token limits (local llama-server crashes
|
||||
// past ~2,050 tokens, measured). Mirrors the markdown
|
||||
// chunker's v4 cap; fallback-path chunks are already capped inside
|
||||
// recursiveChunk.
|
||||
export const CHUNKER_VERSION = 5;
|
||||
|
||||
// Lazy-loaded tree-sitter module (v0.22.x API: Parser is default export)
|
||||
let Parser: typeof import('web-tree-sitter') | null = null;
|
||||
@@ -708,7 +717,7 @@ export async function chunkCodeTextFull(
|
||||
if (chunks.length === 0) {
|
||||
return { chunks: fallbackChunks(source, filePath, language, opts), edges: rawEdges };
|
||||
}
|
||||
return { chunks: mergeSmallSiblings(chunks, chunkTarget), edges: rawEdges };
|
||||
return { chunks: capCodeChunks(mergeSmallSiblings(chunks, chunkTarget)), edges: rawEdges };
|
||||
} catch {
|
||||
return { chunks: fallbackChunks(source, filePath, language, opts), edges: [] };
|
||||
} finally {
|
||||
@@ -791,6 +800,33 @@ function mergeSmallSiblings(chunks: CodeChunk[], chunkTarget: number): CodeChunk
|
||||
return merged;
|
||||
}
|
||||
|
||||
/**
|
||||
* v5 final safety pass for AST-path chunks: split any chunk whose
|
||||
* ESTIMATED embedding tokens (conservative per-char-class heuristic,
|
||||
* cjk.ts) exceed DEFAULT_MAX_EST_TOKENS. Reaches chunks the AST logic
|
||||
* can't subdivide — splitLargeNode returns [] for nodes with < 2 body
|
||||
* children (giant single-statement functions, huge literals), which
|
||||
* previously shipped whole at any size.
|
||||
*
|
||||
* Split pieces inherit the source chunk's metadata verbatim; start/end
|
||||
* lines become approximate for pieces after the first. Acceptable —
|
||||
* these chunks exist for embedding + retrieval, and the alternative was
|
||||
* an embedding request the server rejects (or worse, crashes on).
|
||||
*/
|
||||
function capCodeChunks(chunks: CodeChunk[]): CodeChunk[] {
|
||||
if (chunks.every((c) => estimateEmbeddingTokens(c.text) <= DEFAULT_MAX_EST_TOKENS)) {
|
||||
return chunks;
|
||||
}
|
||||
const out: CodeChunk[] = [];
|
||||
for (const c of chunks) {
|
||||
const pieces = capByEstimatedTokens(c.text, DEFAULT_MAX_EST_TOKENS);
|
||||
for (const piece of pieces) {
|
||||
out.push({ ...c, text: piece, index: out.length, metadata: { ...c.metadata } });
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
function buildMergedChunk(group: CodeChunk[], index: number): CodeChunk {
|
||||
const first = group[0]!;
|
||||
const last = group[group.length - 1]!;
|
||||
|
||||
@@ -17,7 +17,13 @@
|
||||
* Lossless invariant: non-overlapping portions reassemble to original.
|
||||
*/
|
||||
|
||||
import { countCJKAwareWords, CJK_SENTENCE_DELIMITERS, CJK_CLAUSE_DELIMITERS } from '../cjk.ts';
|
||||
import {
|
||||
countCJKAwareWords,
|
||||
CJK_SENTENCE_DELIMITERS,
|
||||
CJK_CLAUSE_DELIMITERS,
|
||||
charEmbedTokenWeight,
|
||||
estimateEmbeddingTokens,
|
||||
} from '../cjk.ts';
|
||||
|
||||
/**
|
||||
* Markdown chunker version. Folded into the per-page chunker_version column
|
||||
@@ -33,8 +39,20 @@ import { countCJKAwareWords, CJK_SENTENCE_DELIMITERS, CJK_CLAUSE_DELIMITERS } fr
|
||||
* re-embed (not re-chunk) so existing pages pick up the wrapper on the
|
||||
* post-upgrade reembed sweep. See
|
||||
* `src/core/contextual-retrieval-service.ts`.
|
||||
*
|
||||
* v4: estimated-token hard cap + whitespace-word undercount fix. The word
|
||||
* pipeline counted a 150-char URL as ONE whitespace word, so URL/phone/
|
||||
* email-dense docs (CJK density < 0.30 → whitespace fallback) produced
|
||||
* 3-4K-char chunks that overflow strict per-request embedding-token
|
||||
* limits (measured: local llama-server crashes past ~2,050 tokens; URL
|
||||
* soup tokenizes at ~1.6 chars/token). Two changes:
|
||||
* 1. countWords() floors the count at ceil(nonWhitespaceChars/6) so a
|
||||
* URL counts roughly per-character, not as one word.
|
||||
* 2. capByEstimatedTokens() final pass guarantees every chunk fits
|
||||
* `maxTokens` (default 1500) under a conservative per-char-class
|
||||
* token estimate, regardless of how word counting misjudged it.
|
||||
*/
|
||||
export const MARKDOWN_CHUNKER_VERSION = 3;
|
||||
export const MARKDOWN_CHUNKER_VERSION = 4;
|
||||
|
||||
const DELIMITERS: string[][] = [
|
||||
['\n\n'], // L0: paragraphs
|
||||
@@ -48,8 +66,20 @@ export interface ChunkOptions {
|
||||
chunkSize?: number; // target words per chunk (default 300)
|
||||
chunkOverlap?: number; // overlap words (default 50)
|
||||
maxChars?: number; // hard cap on any chunk's char length (default 6000)
|
||||
/**
|
||||
* v4: hard cap on any chunk's ESTIMATED embedding tokens (default 1500).
|
||||
* Estimate = conservative per-char-class weights (see cjk.ts
|
||||
* estimateEmbeddingTokens) — deliberately high, so the real tokenizer
|
||||
* count stays below this value. Default leaves headroom for the
|
||||
* contextual-retrieval wrapper (≤ ~630 chars) under a ~2,050-token
|
||||
* per-request embedding server limit.
|
||||
*/
|
||||
maxTokens?: number;
|
||||
}
|
||||
|
||||
/** v4 default for ChunkOptions.maxTokens — see the field doc above. */
|
||||
export const DEFAULT_MAX_EST_TOKENS = 1500;
|
||||
|
||||
export interface TextChunk {
|
||||
text: string;
|
||||
index: number;
|
||||
@@ -73,6 +103,7 @@ export function chunkText(text: string, opts?: ChunkOptions): TextChunk[] {
|
||||
const chunkSize = opts?.chunkSize || 300;
|
||||
const chunkOverlap = opts?.chunkOverlap || 50;
|
||||
const maxChars = opts?.maxChars || 6000;
|
||||
const maxTokens = opts?.maxTokens || DEFAULT_MAX_EST_TOKENS;
|
||||
|
||||
if (!text || text.trim().length === 0) return [];
|
||||
|
||||
@@ -89,8 +120,9 @@ export function chunkText(text: string, opts?: ChunkOptions): TextChunk[] {
|
||||
|
||||
const wordCount = countWords(stripped);
|
||||
if (wordCount <= chunkSize) {
|
||||
// Single-chunk path: still apply the maxChars cap.
|
||||
const capped = capByChars(stripped.trim(), maxChars);
|
||||
// Single-chunk path: still apply the maxChars + maxTokens caps.
|
||||
const capped = capByChars(stripped.trim(), maxChars)
|
||||
.flatMap((t) => capByEstimatedTokens(t, maxTokens));
|
||||
return capped.map((t, i) => ({ text: t, index: i }));
|
||||
}
|
||||
|
||||
@@ -101,9 +133,14 @@ export function chunkText(text: string, opts?: ChunkOptions): TextChunk[] {
|
||||
// v0.32.7: hard char cap. Catches pathological CJK + whitespace-less text
|
||||
// that the word-level pipeline can't bound (a single Chinese paragraph can
|
||||
// exceed 8192 OpenAI embedding tokens at any word count).
|
||||
// v4: estimated-token cap on top — the char cap alone passes token-dense
|
||||
// content (URL soup at ~1.6 chars/token) that overflows strict embedding
|
||||
// server limits.
|
||||
const capped: string[] = [];
|
||||
for (const chunk of withOverlap) {
|
||||
capped.push(...capByChars(chunk.trim(), maxChars));
|
||||
for (const piece of capByChars(chunk.trim(), maxChars)) {
|
||||
capped.push(...capByEstimatedTokens(piece, maxTokens));
|
||||
}
|
||||
}
|
||||
return capped.map((t, i) => ({ text: t, index: i }));
|
||||
}
|
||||
@@ -132,6 +169,68 @@ function capByChars(text: string, maxChars: number): string[] {
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* How far back (in chars) the token cap looks for a friendly cut point
|
||||
* before falling back to a hard cut. 300 covers typical rollup/list line
|
||||
* lengths so forced splits land at line starts, not mid-URL.
|
||||
*/
|
||||
const TOKEN_CAP_CUT_LOOKBACK = 300;
|
||||
|
||||
/**
|
||||
* v4: hard-cap a chunk's ESTIMATED embedding tokens. Final safety pass —
|
||||
* runs after capByChars on every chunk, so no upstream miscounting
|
||||
* (whitespace-word fallback, overlap inflation, char-cap survivors) can
|
||||
* emit a chunk past `maxTokens`.
|
||||
*
|
||||
* Cut placement prefers, within the last TOKEN_CAP_CUT_LOOKBACK chars of
|
||||
* the window: a newline, then any whitespace, then a hard cut. This keeps
|
||||
* forced splits off mid-line/mid-URL positions for list-shaped content
|
||||
* and inside code fences. No overlap is added (pieces stay lossless
|
||||
* modulo the trims the char cap already applies).
|
||||
*
|
||||
* @internal exported for the code chunker (code.ts) and tests.
|
||||
*/
|
||||
export function capByEstimatedTokens(text: string, maxTokens: number): string[] {
|
||||
if (text.length === 0) return [];
|
||||
if (estimateEmbeddingTokens(text) <= maxTokens) return [text];
|
||||
|
||||
const out: string[] = [];
|
||||
let start = 0;
|
||||
while (start < text.length) {
|
||||
// Greedily extend the window until the next char would break the cap.
|
||||
// Always take at least one char so the loop makes forward progress.
|
||||
let est = 0;
|
||||
let end = start;
|
||||
while (end < text.length) {
|
||||
const w = charEmbedTokenWeight(text.charCodeAt(end));
|
||||
if (est + w > maxTokens && end > start) break;
|
||||
est += w;
|
||||
end++;
|
||||
}
|
||||
|
||||
if (end < text.length) {
|
||||
const windowStart = Math.max(start + 1, end - TOKEN_CAP_CUT_LOOKBACK);
|
||||
let cut = text.lastIndexOf('\n', end - 1);
|
||||
if (cut < windowStart) {
|
||||
cut = -1;
|
||||
for (let i = end - 1; i >= windowStart; i--) {
|
||||
const code = text.charCodeAt(i);
|
||||
if (code === 0x20 || (code >= 0x09 && code <= 0x0d)) {
|
||||
cut = i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (cut >= windowStart) end = cut + 1;
|
||||
}
|
||||
|
||||
const slice = text.slice(start, end).trim();
|
||||
if (slice.length > 0) out.push(slice);
|
||||
start = end;
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
function recursiveSplit(text: string, level: number, target: number): string[] {
|
||||
if (level >= DELIMITERS.length) {
|
||||
// Level 4: split on whitespace
|
||||
@@ -317,7 +416,19 @@ function extractTrailingContext(text: string, targetWords: number): string {
|
||||
* Delegated to src/core/cjk.ts so the slugify whitelist, expansion
|
||||
* detection, and PGLite keyword fallback all agree on what "CJK enough"
|
||||
* means.
|
||||
*
|
||||
* v4: floored at ceil(nonWhitespaceChars/6). The whitespace fallback
|
||||
* counts a 150-char URL as ONE word, so URL/phone/email-dense docs
|
||||
* (whose ASCII mass pushes CJK density below the 0.30 threshold) were
|
||||
* sized at a fraction of their real bulk and merged into 3-4K-char
|
||||
* chunks. The floor makes long whitespace-less runs count roughly
|
||||
* per-character while leaving normal Latin prose untouched (average
|
||||
* English word ≈ 5 chars < 6, so the whitespace count still wins).
|
||||
* Kept local to the chunker — search/expansion.ts keeps the original
|
||||
* countCJKAwareWords semantics for its query-length check.
|
||||
*/
|
||||
function countWords(text: string): number {
|
||||
return countCJKAwareWords(text);
|
||||
const cjkAware = countCJKAwareWords(text);
|
||||
const nonWhitespace = text.replace(/\s/g, '').length;
|
||||
return Math.max(cjkAware, Math.ceil(nonWhitespace / 6));
|
||||
}
|
||||
|
||||
@@ -65,3 +65,64 @@ export function countCJKAwareWords(s: string): number {
|
||||
export function escapeLikePattern(s: string): string {
|
||||
return s.replace(/\\/g, '\\\\').replace(/%/g, '\\%').replace(/_/g, '\\_');
|
||||
}
|
||||
|
||||
/**
|
||||
* Conservative per-char-class embedding-token weights (markdown chunker v4).
|
||||
*
|
||||
* Why this exists: the chunker's "word" counting drastically UNDER-counts
|
||||
* whitespace-less ASCII runs (a 150-char URL = 1 whitespace word), so
|
||||
* word-based size targets can emit chunks that overflow an embedding
|
||||
* server's per-request token limit. Measured on a local Qwen3-embedding
|
||||
* llama-server stack:
|
||||
* - URL/phone/email-dense text tokenizes at ~1.6 chars/token
|
||||
* - base64-ish / minified blobs approach ~1.3 chars/token (worst case)
|
||||
* - Korean prose tokenizes NO WORSE than 1 char/token in practice
|
||||
*
|
||||
* Weights are deliberately HIGH (tokens are overestimated) so any cap
|
||||
* based on this estimate is safe against real tokenizers:
|
||||
* - CJK char → 1.0 token (real CJK prose is cheaper)
|
||||
* - other non-space → 0.75 token (≈1.33 chars/token, covers base64)
|
||||
* - whitespace → 0.1 token (mostly folds into neighbor tokens)
|
||||
*/
|
||||
export const EMBED_TOKEN_WEIGHT_CJK = 1.0;
|
||||
export const EMBED_TOKEN_WEIGHT_OTHER = 0.75;
|
||||
export const EMBED_TOKEN_WEIGHT_WS = 0.1;
|
||||
|
||||
/** BMP CJK check by UTF-16 code unit — same ranges as CJK_SLUG_CHARS. */
|
||||
export function isCJKCodeUnit(code: number): boolean {
|
||||
return (
|
||||
(code >= 0x4e00 && code <= 0x9fff) || // Han
|
||||
(code >= 0x3040 && code <= 0x309f) || // Hiragana
|
||||
(code >= 0x30a0 && code <= 0x30ff) || // Katakana
|
||||
(code >= 0xac00 && code <= 0xd7af) // Hangul Syllables
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Per-code-unit token weight. Unrecognized whitespace (exotic Unicode
|
||||
* spaces) intentionally falls into OTHER — that only overestimates.
|
||||
*/
|
||||
export function charEmbedTokenWeight(code: number): number {
|
||||
if (isCJKCodeUnit(code)) return EMBED_TOKEN_WEIGHT_CJK;
|
||||
if (
|
||||
code === 0x20 || (code >= 0x09 && code <= 0x0d) ||
|
||||
code === 0xa0 || code === 0x3000
|
||||
) {
|
||||
return EMBED_TOKEN_WEIGHT_WS;
|
||||
}
|
||||
return EMBED_TOKEN_WEIGHT_OTHER;
|
||||
}
|
||||
|
||||
/**
|
||||
* Tokenizer-free embedding-token estimate (conservative overestimate).
|
||||
* See weight docs above. Astral chars count as 2 OTHER code units —
|
||||
* another overestimate, which is the safe direction.
|
||||
*/
|
||||
export function estimateEmbeddingTokens(s: string): number {
|
||||
if (s.length === 0) return 0;
|
||||
let est = 0;
|
||||
for (let i = 0; i < s.length; i++) {
|
||||
est += charEmbedTokenWeight(s.charCodeAt(i));
|
||||
}
|
||||
return Math.ceil(est);
|
||||
}
|
||||
|
||||
@@ -816,25 +816,6 @@ export async function loadConfigWithEngine(
|
||||
merged.dream = mergedDream;
|
||||
}
|
||||
|
||||
// #1475: eval.* DB-plane merge. `gbrain config set eval.capture true`
|
||||
// writes the DB plane (both keys are in KNOWN_CONFIG_KEYS, so `set`
|
||||
// accepts them silently), but the capture gate (isEvalCaptureEnabled)
|
||||
// reads the merged config. Without this merge the DB value was written
|
||||
// and never read — capture only fired via GBRAIN_CONTRIBUTOR_MODE=1.
|
||||
// Sparse per-key merge: file/env wins per key, DB fills the gaps.
|
||||
const dbEvalCapture = await dbBool('eval.capture');
|
||||
const dbEvalScrub = await dbBool('eval.scrub_pii');
|
||||
const mergedEval: NonNullable<GBrainConfig['eval']> = { ...(merged.eval ?? {}) };
|
||||
if (mergedEval.capture === undefined && dbEvalCapture !== undefined) {
|
||||
mergedEval.capture = dbEvalCapture;
|
||||
}
|
||||
if (mergedEval.scrub_pii === undefined && dbEvalScrub !== undefined) {
|
||||
mergedEval.scrub_pii = dbEvalScrub;
|
||||
}
|
||||
if (Object.keys(mergedEval).length > 0) {
|
||||
merged.eval = mergedEval;
|
||||
}
|
||||
|
||||
return merged;
|
||||
}
|
||||
|
||||
|
||||
@@ -54,7 +54,6 @@ import {
|
||||
import {
|
||||
generatePerChunkSynopsis,
|
||||
SYNOPSIS_PROMPT_VERSION,
|
||||
SYNOPSIS_DOC_MAX_CHARS,
|
||||
type GeneratePerChunkSynopsisResult,
|
||||
} from './page-summary.ts';
|
||||
import {
|
||||
@@ -104,17 +103,8 @@ function getEmbeddingModelTag(): string {
|
||||
export function computeCorpusGeneration(args: {
|
||||
crMode: CRMode;
|
||||
haikuModel: string;
|
||||
/**
|
||||
* Resolved `SYNOPSIS_DOC_MAX_CHARS` for per_chunk_synopsis runs. When
|
||||
* present, folded into the hash so changes to
|
||||
* `GBRAIN_SYNOPSIS_DOC_MAX_CHARS` invalidate the prior cache cleanly.
|
||||
* Omit for `crMode !== 'per_chunk_synopsis'` — title / none modes
|
||||
* don't consult the cap and the field stays out of the hash for
|
||||
* back-compat with pre-cap embeddings.
|
||||
*/
|
||||
synopsisDocMaxChars?: number;
|
||||
}): string {
|
||||
const h = createHash('sha256')
|
||||
return createHash('sha256')
|
||||
.update(args.crMode)
|
||||
.update('|')
|
||||
.update(String(SYNOPSIS_PROMPT_VERSION))
|
||||
@@ -123,11 +113,9 @@ export function computeCorpusGeneration(args: {
|
||||
.update('|')
|
||||
.update(String(TITLE_WRAPPER_VERSION))
|
||||
.update('|')
|
||||
.update(getEmbeddingModelTag());
|
||||
if (args.synopsisDocMaxChars !== undefined) {
|
||||
h.update('|doc_cap=').update(String(args.synopsisDocMaxChars));
|
||||
}
|
||||
return h.digest('hex').slice(0, 16);
|
||||
.update(getEmbeddingModelTag())
|
||||
.digest('hex')
|
||||
.slice(0, 16);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -265,11 +253,7 @@ export async function reembedPageWithContextualRetrieval(
|
||||
args.pageSlug,
|
||||
args.sourceId,
|
||||
resolution.mode,
|
||||
computeCorpusGeneration({
|
||||
crMode: resolution.mode,
|
||||
haikuModel: args.haikuModel ?? DEFAULT_HAIKU_MODEL,
|
||||
synopsisDocMaxChars: resolution.mode === 'per_chunk_synopsis' ? SYNOPSIS_DOC_MAX_CHARS : undefined,
|
||||
}),
|
||||
computeCorpusGeneration({ crMode: resolution.mode, haikuModel: args.haikuModel ?? DEFAULT_HAIKU_MODEL }),
|
||||
);
|
||||
return { kind: 'skipped', reason: 'no_chunks' };
|
||||
}
|
||||
@@ -298,7 +282,6 @@ export async function reembedPageWithContextualRetrieval(
|
||||
const corpus_generation = computeCorpusGeneration({
|
||||
crMode: attemptMode,
|
||||
haikuModel,
|
||||
synopsisDocMaxChars: attemptMode === 'per_chunk_synopsis' ? SYNOPSIS_DOC_MAX_CHARS : undefined,
|
||||
});
|
||||
|
||||
// ── PHASE 2: single DB transaction ───────────────────────────
|
||||
|
||||
@@ -251,14 +251,6 @@ registerBackgroundWorkDrainer({
|
||||
export function isEvalCaptureEnabled(config: GBrainConfig | null | undefined): boolean {
|
||||
if (config?.eval?.capture === true) return true;
|
||||
if (config?.eval?.capture === false) return false;
|
||||
// #1475: DB-plane stash. `gbrain config set eval.capture true` lands in the
|
||||
// config table; connectEngine stamps the merged value here because
|
||||
// ctx.config is the sync file-plane load and never sees the DB plane.
|
||||
// Explicit per-key setting (file above, DB here) beats the broad
|
||||
// CONTRIBUTOR_MODE flag, matching how file-plane `false` already wins.
|
||||
// Doubles as a direct operator env knob.
|
||||
if (process.env.GBRAIN_EVAL_CAPTURE === 'true') return true;
|
||||
if (process.env.GBRAIN_EVAL_CAPTURE === 'false') return false;
|
||||
return process.env.GBRAIN_CONTRIBUTOR_MODE === '1';
|
||||
}
|
||||
|
||||
@@ -271,8 +263,5 @@ export function isEvalCaptureEnabled(config: GBrainConfig | null | undefined): b
|
||||
* have explicit `capture: true`.
|
||||
*/
|
||||
export function isEvalScrubEnabled(config: GBrainConfig | null | undefined): boolean {
|
||||
if (config?.eval?.scrub_pii === false) return false;
|
||||
if (config?.eval?.scrub_pii === true) return true;
|
||||
// #1475: DB-plane stash — see isEvalCaptureEnabled. Default stays true.
|
||||
return process.env.GBRAIN_EVAL_SCRUB_PII !== 'false';
|
||||
return config?.eval?.scrub_pii !== false;
|
||||
}
|
||||
|
||||
@@ -733,11 +733,6 @@ export async function importFromContent(
|
||||
: computeCorpusGeneration({
|
||||
crMode: effectiveCRMode,
|
||||
haikuModel: 'anthropic:claude-haiku-4-5-20251001',
|
||||
// Inline import-file path never uses per_chunk_synopsis (refuses
|
||||
// upstream); pass undefined so the doc-cap field stays out of
|
||||
// the hash here. Per_chunk_synopsis runs through the Minion
|
||||
// backfill handler which threads SYNOPSIS_DOC_MAX_CHARS through
|
||||
// the service layer.
|
||||
});
|
||||
|
||||
// Transaction wraps all DB writes. Every per-page tx call carries the
|
||||
|
||||
@@ -489,22 +489,7 @@ export async function extractPageLinks(
|
||||
// text inside `[[...]]` before any `|`), NOT the display alias
|
||||
// (ref.name = match[2]). `[[struktura|the project]]` must resolve
|
||||
// `struktura`, not "the project". The display text is for context only.
|
||||
//
|
||||
// The literal may be path-qualified (`[[notes/struktura]]`). The FS
|
||||
// path (resolveSlugAll) strips the dirname before its basename lookup,
|
||||
// but this path passed the raw literal to an index keyed by final
|
||||
// segments only — so every slash-containing wikilink outside
|
||||
// DIR_PATTERN silently resolved to nothing. Query by the final
|
||||
// segment, then use the written path as a disambiguation filter
|
||||
// (the analogue of the FS ancestor walk honoring the written path):
|
||||
// a match must end with the literal, so `[[notes/struktura]]` can
|
||||
// resolve to `vault/notes/struktura` but never to `wiki/struktura`.
|
||||
const slashIdx = ref.slug.lastIndexOf('/');
|
||||
const basename = slashIdx === -1 ? ref.slug : ref.slug.slice(slashIdx + 1);
|
||||
let matches = await resolver.resolveBasenameMatches(basename);
|
||||
if (slashIdx !== -1) {
|
||||
matches = matches.filter(m => m === ref.slug || m.endsWith(`/${ref.slug}`));
|
||||
}
|
||||
const matches = await resolver.resolveBasenameMatches(ref.slug);
|
||||
if (matches.length === 0) continue;
|
||||
const idx = content.indexOf(ref.slug);
|
||||
const context = idx >= 0 ? excerpt(content, idx, 240) : ref.name;
|
||||
|
||||
@@ -1,116 +0,0 @@
|
||||
/**
|
||||
* Shared orphan-reporting exclusion policy.
|
||||
*
|
||||
* These are pages where "no inbound links" is expected and should not count
|
||||
* against health. Keep this in core so the CLI orphan report and engine health
|
||||
* dashboard cannot drift.
|
||||
*
|
||||
* Defaults are GBrain-wide conventions only. Brain-specific exclusions
|
||||
* (private folder names, one-off fixture slugs) belong in the brain's own
|
||||
* config, not here:
|
||||
*
|
||||
* gbrain config set orphans.exclude_prefixes "my-private-folder/,archive/"
|
||||
* gbrain config set orphans.exclude_slugs "some-one-off-page"
|
||||
*/
|
||||
|
||||
const AUTO_SUFFIX_PATTERNS = ['/_index', '/log'];
|
||||
|
||||
const PSEUDO_SLUGS = new Set(['_atlas', '_index', '_stats', '_orphans', '_scratch', 'claude']);
|
||||
|
||||
const RAW_SEGMENT = '/raw/';
|
||||
|
||||
const DENY_PREFIXES = [
|
||||
'output/',
|
||||
'dashboards/',
|
||||
'scripts/',
|
||||
'templates/',
|
||||
'_templates/',
|
||||
'openclaw/config/',
|
||||
'extracts/',
|
||||
];
|
||||
|
||||
const FIRST_SEGMENT_EXCLUSIONS = new Set([
|
||||
'scratch',
|
||||
'thoughts',
|
||||
'catalog',
|
||||
'entities',
|
||||
'raw',
|
||||
'atoms',
|
||||
'skills',
|
||||
'dreaming',
|
||||
'daily',
|
||||
]);
|
||||
|
||||
const ROOT_DATE_SLUG = /^\d{4}-\d{2}-\d{2}(?:-.+)?$/;
|
||||
|
||||
function isAgentWorkspaceConvention(slug: string): boolean {
|
||||
if (!slug.startsWith('agents/')) return false;
|
||||
if (slug.includes('/memory/dreaming/')) return true;
|
||||
return /^agents\/[^/]+\/(?:agents|identity|soul|tools|user|heartbeat|dreams|dormant)$/.test(slug);
|
||||
}
|
||||
|
||||
/** Per-brain additions to the convention defaults (from config). */
|
||||
export interface OrphanPolicyOverrides {
|
||||
excludePrefixes?: string[];
|
||||
excludeSlugs?: string[];
|
||||
}
|
||||
|
||||
/** Config keys for per-brain orphan exclusions (comma-separated values). */
|
||||
export const ORPHAN_EXCLUDE_PREFIXES_KEY = 'orphans.exclude_prefixes';
|
||||
export const ORPHAN_EXCLUDE_SLUGS_KEY = 'orphans.exclude_slugs';
|
||||
|
||||
function parseList(value: string | null): string[] {
|
||||
if (!value) return [];
|
||||
return value.split(',').map(s => s.trim()).filter(Boolean);
|
||||
}
|
||||
|
||||
/**
|
||||
* Load per-brain orphan exclusions from the brain config table. Callers with
|
||||
* an engine in hand (getHealth, `gbrain orphans`) pass the result as the
|
||||
* second argument to shouldExcludeFromOrphanReporting.
|
||||
*/
|
||||
export async function loadOrphanPolicyOverrides(
|
||||
engine: { getConfig(key: string): Promise<string | null> },
|
||||
): Promise<OrphanPolicyOverrides> {
|
||||
const [prefixes, slugs] = await Promise.all([
|
||||
engine.getConfig(ORPHAN_EXCLUDE_PREFIXES_KEY),
|
||||
engine.getConfig(ORPHAN_EXCLUDE_SLUGS_KEY),
|
||||
]);
|
||||
return { excludePrefixes: parseList(prefixes), excludeSlugs: parseList(slugs) };
|
||||
}
|
||||
|
||||
export function shouldExcludeFromOrphanReporting(
|
||||
slug: string,
|
||||
overrides?: OrphanPolicyOverrides,
|
||||
): boolean {
|
||||
if (PSEUDO_SLUGS.has(slug)) return true;
|
||||
|
||||
for (const suffix of AUTO_SUFFIX_PATTERNS) {
|
||||
if (slug.endsWith(suffix)) return true;
|
||||
}
|
||||
|
||||
if (slug.includes(RAW_SEGMENT)) return true;
|
||||
if (slug.includes('/daily/')) return true;
|
||||
|
||||
for (const prefix of DENY_PREFIXES) {
|
||||
if (slug.startsWith(prefix)) return true;
|
||||
}
|
||||
|
||||
const firstSegment = slug.split('/')[0];
|
||||
if (FIRST_SEGMENT_EXCLUSIONS.has(firstSegment)) return true;
|
||||
|
||||
if (ROOT_DATE_SLUG.test(slug)) return true;
|
||||
|
||||
if (slug.startsWith('_brain-')) return true;
|
||||
|
||||
if (isAgentWorkspaceConvention(slug)) return true;
|
||||
|
||||
if (overrides) {
|
||||
if (overrides.excludeSlugs?.includes(slug)) return true;
|
||||
for (const prefix of overrides.excludePrefixes ?? []) {
|
||||
if (slug.startsWith(prefix)) return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
@@ -44,33 +44,6 @@ const HAIKU_MAX_TOKENS = 200;
|
||||
/** Default model when caller doesn't override. Resolves through the gateway. */
|
||||
const DEFAULT_SYNOPSIS_MODEL = 'anthropic:claude-haiku-4-5-20251001';
|
||||
|
||||
/**
|
||||
* Hard cap on `documentText` length (chars) before send.
|
||||
*
|
||||
* 2026-05-25 fix wave: small local chat models (Gemma 4 E2B, Qwen3 4B) get
|
||||
* dramatically slower on long contexts even with 131K-token windows declared.
|
||||
* A 73K-char page synopsis on Gemma 4 E2B takes 60-120s, exceeding the
|
||||
* worker's default 30s `lockDuration` and tripping `lock-lost` errors.
|
||||
*
|
||||
* Truncate to a budget that fits a small model's effective throughput while
|
||||
* preserving enough document context for the synopsis to be useful. Truncates
|
||||
* the TAIL because the head (title, frontmatter, intro) carries the
|
||||
* document-level anchor the synopsis needs.
|
||||
*
|
||||
* Override per workload via `GBRAIN_SYNOPSIS_DOC_MAX_CHARS`. Default 32768
|
||||
* (~8K tokens at 4 chars/tok) keeps small-model synopsis under ~30s.
|
||||
* Anthropic Haiku is unaffected at this cap; bump higher when running
|
||||
* frontier models if you want richer document anchoring.
|
||||
*/
|
||||
export const SYNOPSIS_DOC_MAX_CHARS = (() => {
|
||||
const env = process.env.GBRAIN_SYNOPSIS_DOC_MAX_CHARS;
|
||||
if (env && /^\d+$/.test(env)) {
|
||||
const n = parseInt(env, 10);
|
||||
if (n >= 512 && n <= 1_048_576) return n;
|
||||
}
|
||||
return 32768;
|
||||
})();
|
||||
|
||||
/**
|
||||
* Synopsis prompt version. Folded into corpus_generation so prompt edits
|
||||
* invalidate prior embeddings via the v0.40.3.0 query_cache.page_generations
|
||||
@@ -215,19 +188,11 @@ function buildUserPrompt(
|
||||
documentText: string,
|
||||
chunkText: string,
|
||||
): string {
|
||||
// Tail-truncate `documentText` to `SYNOPSIS_DOC_MAX_CHARS` so small local
|
||||
// chat models don't stall on >100KB pages. Head preserved (title block,
|
||||
// frontmatter, intro paragraphs carry the document-level anchor).
|
||||
let trimmedDoc = documentText;
|
||||
if (documentText.length > SYNOPSIS_DOC_MAX_CHARS) {
|
||||
trimmedDoc = documentText.slice(0, SYNOPSIS_DOC_MAX_CHARS) +
|
||||
`\n\n[... ${documentText.length - SYNOPSIS_DOC_MAX_CHARS} chars truncated for synopsis budget ...]`;
|
||||
}
|
||||
return [
|
||||
`<page_title>${pageTitle}</page_title>`,
|
||||
'',
|
||||
'<full_document>',
|
||||
trimmedDoc,
|
||||
documentText,
|
||||
'</full_document>',
|
||||
'',
|
||||
'<chunk>',
|
||||
|
||||
+19
-30
@@ -57,8 +57,6 @@ import { finalizeLastSeen } from './chronicle/last-seen.ts';
|
||||
import { computeAnomaliesFromBuckets } from './cycle/anomaly.ts';
|
||||
import { resolveBoostMap, resolveHardExcludes } from './search/source-boost.ts';
|
||||
import { buildSourceFactorCase, buildHardExcludeClause, buildVisibilityClause, buildRecencyComponentSql, buildBestPerPagePoolCte, buildOrFallbackWebsearchQuery } from './search/sql-ranking.ts';
|
||||
import { shouldExcludeFromOrphanReporting, loadOrphanPolicyOverrides } from './orphan-policy.ts';
|
||||
import { LINK_EXTRACTOR_VERSION_TS } from './link-extraction.ts';
|
||||
import {
|
||||
normalizeEngineColumn,
|
||||
buildVectorCastFragment,
|
||||
@@ -2324,10 +2322,6 @@ export class PGLiteEngine implements BrainEngine {
|
||||
// v0.40.3.0 D24 NULL→non-NULL race fix mirrors postgres-engine.ts. Two writers
|
||||
// racing on the same chunk previously raced last-write-wins; the fix lets the
|
||||
// fresher `embedded_at` win in the text-unchanged branch.
|
||||
//
|
||||
// Code-chunk metadata columns follow the same chunk_text-gated CASE pattern as `embedding`
|
||||
// (#769). Re-chunk trusts EXCLUDED outright; pure re-embed COALESCEs so a caller carrying
|
||||
// only embedding-shaped fields doesn't clobber metadata to NULL.
|
||||
await this.db.query(
|
||||
`INSERT INTO content_chunks ${cols} VALUES ${rowParts.join(', ')}
|
||||
ON CONFLICT (page_id, chunk_index) DO UPDATE SET
|
||||
@@ -2351,14 +2345,14 @@ export class PGLiteEngine implements BrainEngine {
|
||||
THEN EXCLUDED.embedded_at
|
||||
ELSE content_chunks.embedded_at
|
||||
END,
|
||||
language = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.language ELSE COALESCE(EXCLUDED.language, content_chunks.language) END,
|
||||
symbol_name = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_name ELSE COALESCE(EXCLUDED.symbol_name, content_chunks.symbol_name) END,
|
||||
symbol_type = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_type ELSE COALESCE(EXCLUDED.symbol_type, content_chunks.symbol_type) END,
|
||||
start_line = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.start_line ELSE COALESCE(EXCLUDED.start_line, content_chunks.start_line) END,
|
||||
end_line = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.end_line ELSE COALESCE(EXCLUDED.end_line, content_chunks.end_line) END,
|
||||
parent_symbol_path = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.parent_symbol_path ELSE COALESCE(EXCLUDED.parent_symbol_path, content_chunks.parent_symbol_path) END,
|
||||
doc_comment = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.doc_comment ELSE COALESCE(EXCLUDED.doc_comment, content_chunks.doc_comment) END,
|
||||
symbol_name_qualified = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_name_qualified ELSE COALESCE(EXCLUDED.symbol_name_qualified, content_chunks.symbol_name_qualified) END,
|
||||
language = EXCLUDED.language,
|
||||
symbol_name = EXCLUDED.symbol_name,
|
||||
symbol_type = EXCLUDED.symbol_type,
|
||||
start_line = EXCLUDED.start_line,
|
||||
end_line = EXCLUDED.end_line,
|
||||
parent_symbol_path = EXCLUDED.parent_symbol_path,
|
||||
doc_comment = EXCLUDED.doc_comment,
|
||||
symbol_name_qualified = EXCLUDED.symbol_name_qualified,
|
||||
modality = EXCLUDED.modality,
|
||||
embedding_image = COALESCE(EXCLUDED.embedding_image, content_chunks.embedding_image)`,
|
||||
params
|
||||
@@ -5213,10 +5207,15 @@ export class PGLiteEngine implements BrainEngine {
|
||||
(SELECT count(*) FROM pages) as page_count,
|
||||
(SELECT count(*) FROM content_chunks WHERE embedded_at IS NOT NULL)::float /
|
||||
GREATEST((SELECT count(*) FROM content_chunks), 1)::float as embed_coverage,
|
||||
0 as stale_pages,
|
||||
-- Bug 11 — orphan = islanded (no inbound AND no outbound). The raw
|
||||
-- list is filtered in TS using the shared orphan-reporting policy.
|
||||
0 as orphan_pages,
|
||||
(SELECT count(*) FROM pages p
|
||||
WHERE p.updated_at < (SELECT MAX(te.created_at) FROM timeline_entries te WHERE te.page_id = p.id)
|
||||
) as stale_pages,
|
||||
-- Bug 11 — orphan = islanded (no inbound AND no outbound).
|
||||
-- See BrainHealth.orphan_pages docstring; docs updated to match this.
|
||||
(SELECT count(*) FROM pages p
|
||||
WHERE NOT EXISTS (SELECT 1 FROM links l WHERE l.to_page_id = p.id)
|
||||
AND NOT EXISTS (SELECT 1 FROM links l WHERE l.from_page_id = p.id)
|
||||
) as orphan_pages,
|
||||
(SELECT count(*) FROM links l
|
||||
WHERE NOT EXISTS (SELECT 1 FROM pages p WHERE p.id = l.to_page_id)
|
||||
) as dead_links,
|
||||
@@ -5241,20 +5240,10 @@ export class PGLiteEngine implements BrainEngine {
|
||||
LIMIT 5
|
||||
`);
|
||||
|
||||
const { rows: islandedRows } = await this.db.query(`
|
||||
SELECT p.slug
|
||||
FROM pages p
|
||||
WHERE NOT EXISTS (SELECT 1 FROM links l WHERE l.to_page_id = p.id)
|
||||
AND NOT EXISTS (SELECT 1 FROM links l WHERE l.from_page_id = p.id)
|
||||
`);
|
||||
|
||||
const r = h as Record<string, unknown>;
|
||||
const pageCount = Number(r.page_count);
|
||||
const embedCoverage = Number(r.embed_coverage);
|
||||
const stalePages = await this.countStalePagesForExtraction({ versionTs: LINK_EXTRACTOR_VERSION_TS });
|
||||
const orphanOverrides = await loadOrphanPolicyOverrides(this);
|
||||
const orphanPages = (islandedRows as { slug: string }[])
|
||||
.filter(row => !shouldExcludeFromOrphanReporting(row.slug, orphanOverrides)).length;
|
||||
const orphanPages = Number(r.orphan_pages);
|
||||
const deadLinks = Number(r.dead_links);
|
||||
const linkCount = Number(r.link_count);
|
||||
const pagesWithTimeline = Number(r.pages_with_timeline);
|
||||
@@ -5282,7 +5271,7 @@ export class PGLiteEngine implements BrainEngine {
|
||||
return {
|
||||
page_count: pageCount,
|
||||
embed_coverage: embedCoverage,
|
||||
stale_pages: stalePages,
|
||||
stale_pages: Number(r.stale_pages),
|
||||
orphan_pages: orphanPages,
|
||||
missing_embeddings: Number(r.missing_embeddings),
|
||||
brain_score: brainScore,
|
||||
|
||||
+22
-33
@@ -67,8 +67,6 @@ import { resolveBoostMap, resolveHardExcludes } from './search/source-boost.ts';
|
||||
import { buildSourceFactorCase, buildHardExcludeClause, buildVisibilityClause, buildRecencyComponentSql, buildBestPerPagePoolCte, buildOrFallbackWebsearchQuery } from './search/sql-ranking.ts';
|
||||
import { DEFAULT_EMBEDDING_MODEL, DEFAULT_EMBEDDING_DIMENSIONS } from './ai/defaults.ts';
|
||||
import { DELETE_BATCH_SIZE } from './engine-constants.ts';
|
||||
import { shouldExcludeFromOrphanReporting, loadOrphanPolicyOverrides } from './orphan-policy.ts';
|
||||
import { LINK_EXTRACTOR_VERSION_TS } from './link-extraction.ts';
|
||||
|
||||
function escapeSqlStringLiteral(value: string): string {
|
||||
return value.replace(/'/g, "''");
|
||||
@@ -2475,13 +2473,6 @@ export class PostgresEngine implements BrainEngine {
|
||||
// - new is fresher (embedded_at > existing.embedded_at) → take new
|
||||
// - otherwise → keep existing (slower writer with stale embedding loses)
|
||||
// Mirrored in pglite-engine.ts; pinned by test/e2e/concurrent-embed-race.test.ts.
|
||||
//
|
||||
// Code-chunk metadata columns (language / symbol_name / symbol_type / line range /
|
||||
// parent_symbol_path / doc_comment / symbol_name_qualified) follow the SAME chunk_text-gated
|
||||
// CASE pattern as `embedding` (#769). Re-chunk (chunk_text changed) trusts EXCLUDED outright;
|
||||
// pure re-embed (chunk_text unchanged) COALESCEs so a caller that only carries embedding
|
||||
// doesn't clobber metadata to NULL. Without this, every embed --stale pass nuked code-def's
|
||||
// primary index for thousands of chunks at once.
|
||||
await sql.unsafe(
|
||||
`INSERT INTO content_chunks ${cols} VALUES ${rows.join(', ')}
|
||||
ON CONFLICT (page_id, chunk_index) DO UPDATE SET
|
||||
@@ -2505,14 +2496,14 @@ export class PostgresEngine implements BrainEngine {
|
||||
THEN EXCLUDED.embedded_at
|
||||
ELSE content_chunks.embedded_at
|
||||
END,
|
||||
language = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.language ELSE COALESCE(EXCLUDED.language, content_chunks.language) END,
|
||||
symbol_name = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_name ELSE COALESCE(EXCLUDED.symbol_name, content_chunks.symbol_name) END,
|
||||
symbol_type = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_type ELSE COALESCE(EXCLUDED.symbol_type, content_chunks.symbol_type) END,
|
||||
start_line = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.start_line ELSE COALESCE(EXCLUDED.start_line, content_chunks.start_line) END,
|
||||
end_line = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.end_line ELSE COALESCE(EXCLUDED.end_line, content_chunks.end_line) END,
|
||||
parent_symbol_path = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.parent_symbol_path ELSE COALESCE(EXCLUDED.parent_symbol_path, content_chunks.parent_symbol_path) END,
|
||||
doc_comment = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.doc_comment ELSE COALESCE(EXCLUDED.doc_comment, content_chunks.doc_comment) END,
|
||||
symbol_name_qualified = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_name_qualified ELSE COALESCE(EXCLUDED.symbol_name_qualified, content_chunks.symbol_name_qualified) END,
|
||||
language = EXCLUDED.language,
|
||||
symbol_name = EXCLUDED.symbol_name,
|
||||
symbol_type = EXCLUDED.symbol_type,
|
||||
start_line = EXCLUDED.start_line,
|
||||
end_line = EXCLUDED.end_line,
|
||||
parent_symbol_path = EXCLUDED.parent_symbol_path,
|
||||
doc_comment = EXCLUDED.doc_comment,
|
||||
symbol_name_qualified = EXCLUDED.symbol_name_qualified,
|
||||
modality = EXCLUDED.modality,
|
||||
embedding_image = COALESCE(EXCLUDED.embedding_image, content_chunks.embedding_image)`,
|
||||
params as Parameters<typeof sql.unsafe>[1],
|
||||
@@ -5322,9 +5313,11 @@ export class PostgresEngine implements BrainEngine {
|
||||
async getHealth(): Promise<BrainHealth> {
|
||||
const sql = this.sql;
|
||||
// Bug 11 doc-drift fix — orphan_pages means "islanded" (no inbound AND
|
||||
// no outbound links). The raw islanded list is filtered through the same
|
||||
// policy as `gbrain orphans` so convention pages do not count against
|
||||
// dashboard health.
|
||||
// no outbound links), aligning both engines with the user-facing
|
||||
// definition. The type comment previously said "no inbound" but the
|
||||
// SQL required both — docs now match code so users can trust the
|
||||
// number. A hub page that links out to many but has no back-references
|
||||
// is working as intended, not an orphan.
|
||||
const [h] = await sql`
|
||||
WITH entity_pages AS (
|
||||
SELECT id, slug FROM pages WHERE type IN ('person', 'company')
|
||||
@@ -5333,8 +5326,13 @@ export class PostgresEngine implements BrainEngine {
|
||||
(SELECT count(*) FROM pages) as page_count,
|
||||
(SELECT count(*) FROM content_chunks WHERE embedded_at IS NOT NULL)::float /
|
||||
GREATEST((SELECT count(*) FROM content_chunks), 1)::float as embed_coverage,
|
||||
0 as stale_pages,
|
||||
0 as orphan_pages,
|
||||
(SELECT count(*) FROM pages p
|
||||
WHERE p.updated_at < (SELECT MAX(te.created_at) FROM timeline_entries te WHERE te.page_id = p.id)
|
||||
) as stale_pages,
|
||||
(SELECT count(*) FROM pages p
|
||||
WHERE NOT EXISTS (SELECT 1 FROM links l WHERE l.to_page_id = p.id)
|
||||
AND NOT EXISTS (SELECT 1 FROM links l WHERE l.from_page_id = p.id)
|
||||
) as orphan_pages,
|
||||
(SELECT count(*) FROM links l
|
||||
WHERE NOT EXISTS (SELECT 1 FROM pages p WHERE p.id = l.to_page_id)
|
||||
) as dead_links,
|
||||
@@ -5358,18 +5356,9 @@ export class PostgresEngine implements BrainEngine {
|
||||
LIMIT 5
|
||||
`;
|
||||
|
||||
const islandedRows = await sql<{ slug: string }[]>`
|
||||
SELECT p.slug
|
||||
FROM pages p
|
||||
WHERE NOT EXISTS (SELECT 1 FROM links l WHERE l.to_page_id = p.id)
|
||||
AND NOT EXISTS (SELECT 1 FROM links l WHERE l.from_page_id = p.id)
|
||||
`;
|
||||
|
||||
const pageCount = Number(h.page_count);
|
||||
const embedCoverage = Number(h.embed_coverage);
|
||||
const stalePages = await this.countStalePagesForExtraction({ versionTs: LINK_EXTRACTOR_VERSION_TS });
|
||||
const orphanOverrides = await loadOrphanPolicyOverrides(this);
|
||||
const orphanPages = islandedRows.filter(row => !shouldExcludeFromOrphanReporting(row.slug, orphanOverrides)).length;
|
||||
const orphanPages = Number(h.orphan_pages);
|
||||
const deadLinks = Number(h.dead_links);
|
||||
const linkCount = Number(h.link_count);
|
||||
const pagesWithTimeline = Number(h.pages_with_timeline);
|
||||
@@ -5397,7 +5386,7 @@ export class PostgresEngine implements BrainEngine {
|
||||
return {
|
||||
page_count: pageCount,
|
||||
embed_coverage: embedCoverage,
|
||||
stale_pages: stalePages,
|
||||
stale_pages: Number(h.stale_pages),
|
||||
orphan_pages: orphanPages,
|
||||
missing_embeddings: Number(h.missing_embeddings),
|
||||
brain_score: brainScore,
|
||||
|
||||
@@ -93,13 +93,7 @@ import { resolveLrSchedule } from './lr-schedule.ts';
|
||||
import { preflight, formatPreflightReport } from './preflight.ts';
|
||||
import { isRejected, loadRejectedBuffer, makeRejectedEntry, saveRejectedBuffer } from './rejected-buffer.ts';
|
||||
import { runReflect, runOneShotRewrite, describeJudges } from './reflect.ts';
|
||||
import {
|
||||
acceptCandidate,
|
||||
proposedPath as proposedFilePath,
|
||||
revertAllPending,
|
||||
skillPath,
|
||||
writeProposed,
|
||||
} from './version-store.ts';
|
||||
import { acceptCandidate, bestPath, revertAllPending, skillPath, writeProposed } from './version-store.ts';
|
||||
import { runValidationGate, scoreSkillOnTasks } from './validate-gate.ts';
|
||||
import { ROLLOUT_SUCCESS_THRESHOLD } from './types.ts';
|
||||
import type { SkillOptOpts, EditOp, RunReceipt, BenchmarkTask } from './types.ts';
|
||||
@@ -708,9 +702,9 @@ async function runOptimizationLoop(
|
||||
// to the catch's assignment values only (it can't prove the async callback ran).
|
||||
const finalOutcome = outcome as 'accepted' | 'no_improvement' | 'aborted' | 'errored';
|
||||
if (!mutateDecision.mutate && finalOutcome === 'accepted') {
|
||||
// writeProposed() emitted both the best pointer and the stable review
|
||||
// artifact in the accept branch. SKILL.md remains untouched.
|
||||
proposedPath = proposedFilePath(skillsDir, skillName);
|
||||
// best.md was written by writeProposed() in the accept branch (no-mutate
|
||||
// path); it doubles as proposed.md for human review. SKILL.md untouched.
|
||||
proposedPath = bestPath(skillsDir, skillName);
|
||||
} else if (mutateDecision.mutate) {
|
||||
mutatedSkillFile = finalOutcome === 'accepted';
|
||||
}
|
||||
|
||||
@@ -23,7 +23,6 @@
|
||||
*
|
||||
* history.json
|
||||
* best.md
|
||||
* proposed.md
|
||||
* versions/
|
||||
* v0001_e1_s1.md
|
||||
* v0002_e1_s2.md
|
||||
@@ -53,10 +52,6 @@ export function bestPath(skillsDir: string, skillName: string): string {
|
||||
return path.join(skilloptDir(skillsDir, skillName), 'best.md');
|
||||
}
|
||||
|
||||
export function proposedPath(skillsDir: string, skillName: string): string {
|
||||
return path.join(skilloptDir(skillsDir, skillName), 'proposed.md');
|
||||
}
|
||||
|
||||
export function skillPath(skillsDir: string, skillName: string): string {
|
||||
return path.join(skillsDir, skillName, 'SKILL.md');
|
||||
}
|
||||
@@ -176,18 +171,17 @@ export function acceptCandidate(input: AcceptInput): AcceptResult {
|
||||
}
|
||||
|
||||
/**
|
||||
* Write the candidate to both `best.md` and `proposed.md` WITHOUT touching
|
||||
* SKILL.md or the history ledger. `best.md` remains the optimizer's current
|
||||
* best pointer; `proposed.md` is the stable human-review artifact promised by
|
||||
* `--no-mutate`. Returns the proposal path. Each write is atomic (.tmp + rename).
|
||||
* Write the candidate to `best.md` (which doubles as `proposed.md`) WITHOUT
|
||||
* touching SKILL.md or the history ledger. Used by the `--no-mutate` /
|
||||
* bundled-without-allow paths: the optimizer found a better candidate but the
|
||||
* caller opted out of in-place mutation, so we surface it for human review.
|
||||
* Returns the path written. Atomic (.tmp + rename).
|
||||
*/
|
||||
export function writeProposed(skillsDir: string, skillName: string, candidateText: string): string {
|
||||
const best = bestPath(skillsDir, skillName);
|
||||
const proposed = proposedPath(skillsDir, skillName);
|
||||
fs.mkdirSync(path.dirname(best), { recursive: true });
|
||||
atomicWrite(best, candidateText);
|
||||
atomicWrite(proposed, candidateText);
|
||||
return proposed;
|
||||
const p = bestPath(skillsDir, skillName);
|
||||
fs.mkdirSync(path.dirname(p), { recursive: true });
|
||||
atomicWrite(p, candidateText);
|
||||
return p;
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -63,26 +63,25 @@ interface PluginCtx {
|
||||
[key: string]: unknown;
|
||||
}
|
||||
|
||||
export function register(api: PluginApi) {
|
||||
api.registerContextEngine(ENGINE_ID, (ctx: PluginCtx) => {
|
||||
const hostResolver =
|
||||
typeof ctx.resolveEntities === 'function'
|
||||
? ctx.resolveEntities
|
||||
: typeof ctx.brainQuery === 'function'
|
||||
? ctx.brainQuery
|
||||
: undefined;
|
||||
return createGBrainContextEngine({
|
||||
workspaceDir: ctx.workspaceDir,
|
||||
resolveEntities: hostResolver,
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
const entry: PluginEntry = {
|
||||
id: 'gbrain-context-engine',
|
||||
name: 'GBrain Context Engine',
|
||||
description: 'Deterministic temporal/spatial context injection on every turn',
|
||||
register,
|
||||
|
||||
register(api: PluginApi) {
|
||||
api.registerContextEngine(ENGINE_ID, (ctx: PluginCtx) => {
|
||||
const hostResolver =
|
||||
typeof ctx.resolveEntities === 'function'
|
||||
? ctx.resolveEntities
|
||||
: typeof ctx.brainQuery === 'function'
|
||||
? ctx.brainQuery
|
||||
: undefined;
|
||||
return createGBrainContextEngine({
|
||||
workspaceDir: ctx.workspaceDir,
|
||||
resolveEntities: hostResolver,
|
||||
});
|
||||
});
|
||||
},
|
||||
};
|
||||
|
||||
export default entry;
|
||||
|
||||
@@ -1,67 +0,0 @@
|
||||
// #1474: the v0.41.1 wave shipped bench-publish.ts + docs/eval-bench.md
|
||||
// advertising `gbrain bench publish`, but the cli.ts dispatcher case was never
|
||||
// added — the documented command hit 'Unknown command'. These tests spawn the
|
||||
// real CLI (no DB needed; bench publish is pure file I/O) and fail on any
|
||||
// regression of the dispatcher wiring.
|
||||
import { describe, expect, test } from 'bun:test';
|
||||
import { mkdtempSync, writeFileSync, existsSync, rmSync } from 'node:fs';
|
||||
import { tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { spawnSync } from 'node:child_process';
|
||||
|
||||
function runCli(args: string[]): { stdout: string; stderr: string; code: number } {
|
||||
const result = spawnSync(process.execPath, ['run', 'src/cli.ts', 'bench', ...args], {
|
||||
encoding: 'utf8',
|
||||
cwd: process.cwd(),
|
||||
env: { ...process.env },
|
||||
});
|
||||
return { stdout: result.stdout ?? '', stderr: result.stderr ?? '', code: result.status ?? -1 };
|
||||
}
|
||||
|
||||
describe('gbrain bench dispatcher (#1474)', () => {
|
||||
test('bench --help reaches bench-publish help without a DB (was: Unknown command)', () => {
|
||||
const { stdout, stderr, code } = runCli(['--help']);
|
||||
expect(stderr).not.toContain('Unknown command');
|
||||
expect(code).toBe(0);
|
||||
expect(stdout).toContain('gbrain bench publish');
|
||||
expect(stdout).toContain('--from');
|
||||
});
|
||||
|
||||
test('unknown bench subcommand exits 2 with usage', () => {
|
||||
const { stderr, code } = runCli(['bogus']);
|
||||
expect(code).toBe(2);
|
||||
expect(stderr).toContain('Unknown bench subcommand');
|
||||
expect(stderr).toContain('bench publish');
|
||||
});
|
||||
|
||||
test('bench publish roundtrip: captured NDJSON in, baseline file out', () => {
|
||||
const tmp = mkdtempSync(join(tmpdir(), 'bench-cli-'));
|
||||
try {
|
||||
const row = {
|
||||
tool_name: 'query',
|
||||
query: 'hello world',
|
||||
retrieved_slugs: ['slug-a'],
|
||||
retrieved_chunk_ids: [1],
|
||||
source_ids: ['default'],
|
||||
expand_enabled: false,
|
||||
detail: 'medium',
|
||||
detail_resolved: 'medium',
|
||||
vector_enabled: true,
|
||||
expansion_applied: false,
|
||||
latency_ms: 100,
|
||||
remote: false,
|
||||
job_id: null,
|
||||
subagent_id: null,
|
||||
};
|
||||
const from = join(tmp, 'captured.ndjson');
|
||||
const to = join(tmp, 'personal.baseline.ndjson');
|
||||
writeFileSync(from, `${JSON.stringify(row)}\n`);
|
||||
const { code, stderr } = runCli(['publish', '--from', from, '--to', to]);
|
||||
expect(stderr).not.toContain('Unknown command');
|
||||
expect(code).toBe(0);
|
||||
expect(existsSync(to)).toBe(true);
|
||||
} finally {
|
||||
rmSync(tmp, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
});
|
||||
@@ -15,19 +15,22 @@ import { describe, test, expect } from 'bun:test';
|
||||
import { CHUNKER_VERSION } from '../src/core/chunkers/code.ts';
|
||||
|
||||
describe('Layer 12 — CHUNKER_VERSION constant', () => {
|
||||
test('bumped to 4 for Cathedral II', () => {
|
||||
test('bumped to 5 for the estimated-token hard cap', () => {
|
||||
// v3: v0.19.0 Chonkie parity (tokenizer + small-sibling merge).
|
||||
// v4: v0.20.0 Cathedral II (qualified names + parent scope + doc_comment
|
||||
// + fence extraction + chunk-grain FTS). Folded into content_hash
|
||||
// so any bump forces clean re-chunks on next sync.
|
||||
expect(CHUNKER_VERSION).toBe(4);
|
||||
// v5: estimated-token hard cap on AST-path chunks (capCodeChunks) so
|
||||
// un-subdividable giant nodes can't overflow strict embedding
|
||||
// server token limits.
|
||||
expect(CHUNKER_VERSION).toBe(5);
|
||||
});
|
||||
|
||||
test('is stable across imports (not recomputed at call time)', async () => {
|
||||
const a = (await import('../src/core/chunkers/code.ts')).CHUNKER_VERSION;
|
||||
const b = (await import('../src/core/chunkers/code.ts')).CHUNKER_VERSION;
|
||||
expect(a).toBe(b);
|
||||
expect(a).toBe(4);
|
||||
expect(a).toBe(5);
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
@@ -10,8 +10,8 @@ import { describe, test, expect } from 'bun:test';
|
||||
import { chunkCodeText, detectCodeLanguage, CHUNKER_VERSION } from '../../src/core/chunkers/code.ts';
|
||||
|
||||
describe('CHUNKER_VERSION', () => {
|
||||
test('v0.20.0 Cathedral II Layer 12 bumped to 4', () => {
|
||||
expect(CHUNKER_VERSION).toBe(4);
|
||||
test('v5: estimated-token hard cap on AST-path chunks', () => {
|
||||
expect(CHUNKER_VERSION).toBe(5);
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
@@ -135,13 +135,14 @@ describe('Recursive Text Chunker', () => {
|
||||
});
|
||||
|
||||
describe('CJK chunking (v0.32.7)', () => {
|
||||
test('MARKDOWN_CHUNKER_VERSION is 3', async () => {
|
||||
test('MARKDOWN_CHUNKER_VERSION is 4', async () => {
|
||||
// v0.40.3.0: bumped 2→3 to signal the post-upgrade reembed sweep that
|
||||
// contextual retrieval wrapping is now applied at embed time. Chunk
|
||||
// boundaries themselves are unchanged; the bump forces re-embed for
|
||||
// pages where chunker_version < 3.
|
||||
// contextual retrieval wrapping is now applied at embed time.
|
||||
// v4: estimated-token hard cap + whitespace-word undercount floor
|
||||
// (URL-dense docs produced chunks past strict embedding server token
|
||||
// limits). Boundary change → forces re-chunk for chunker_version < 4.
|
||||
const mod = await import('../../src/core/chunkers/recursive.ts');
|
||||
expect(mod.MARKDOWN_CHUNKER_VERSION).toBe(3);
|
||||
expect(mod.MARKDOWN_CHUNKER_VERSION).toBe(4);
|
||||
});
|
||||
|
||||
test('long pure-Chinese paragraph splits into multiple chunks', () => {
|
||||
|
||||
@@ -0,0 +1,195 @@
|
||||
/**
|
||||
* Markdown chunker v4 / code chunker v5 — estimated-token hard cap
|
||||
* regression tests.
|
||||
*
|
||||
* Reproduces a field failure: a local llama-server embedding backend
|
||||
* (`-ub 2048`) crashes deterministically (trace/BPT trap → EOF at the
|
||||
* client) when a single chunk exceeds ~2,050 real tokens. Two content
|
||||
* shapes triggered it:
|
||||
*
|
||||
* 1. Korean docs carrying one long source URL per line.
|
||||
* The URLs' ASCII mass pushes CJK density below 0.30, flipping
|
||||
* countCJKAwareWords to whitespace counting, where a 150-char URL
|
||||
* counts as ONE word → chunks ballooned to 3-4K chars ≈ 2,000+
|
||||
* real tokens (URL soup tokenizes at ~1.6 chars/token).
|
||||
*
|
||||
* 2. Large JSON code blocks (~7K chars) that the word pipeline
|
||||
* undercounts the same way (few whitespace tokens).
|
||||
*
|
||||
* The fix: every emitted chunk must satisfy
|
||||
* estimateEmbeddingTokens(chunk) <= maxTokens (default 1500)
|
||||
* where the estimate deliberately OVERSTATES real tokenizer counts.
|
||||
*/
|
||||
|
||||
import { describe, test, expect } from 'bun:test';
|
||||
import { chunkText, capByEstimatedTokens, DEFAULT_MAX_EST_TOKENS } from '../../src/core/chunkers/recursive.ts';
|
||||
import { chunkCodeText } from '../../src/core/chunkers/code.ts';
|
||||
import { estimateEmbeddingTokens } from '../../src/core/cjk.ts';
|
||||
|
||||
/** Synthesize the failing shape: Korean rollup lines each ending in a long Notion URL. */
|
||||
function urlDenseKoreanRollup(lines: number): string {
|
||||
const out: string[] = ['# 링크가 줄마다 붙는 한국어 예시 문서', ''];
|
||||
for (let i = 0; i < lines; i++) {
|
||||
const hex32 = (i * 2654435761 >>> 0).toString(16).padStart(8, '0').repeat(4);
|
||||
out.push(
|
||||
`- **항목 ${i}**: 이 줄은 청커 동작 검증을 위한 의미 없는 한국어 예시 문장입니다 · 전화 000-0000-${String(1000 + i)} · ` +
|
||||
`이메일 user${i}@example.com · 링크: https://docs.example.com/pages/${hex32}?v=abcdef0123456789&ref=sample`,
|
||||
);
|
||||
}
|
||||
return out.join('\n');
|
||||
}
|
||||
|
||||
/** Synthesize a large pretty-printed JSON block with CJK values. */
|
||||
function bigJsonBlock(targetChars: number): string {
|
||||
const entries: string[] = [];
|
||||
let i = 0;
|
||||
let len = 0;
|
||||
while (len < targetChars) {
|
||||
const row =
|
||||
` "item_${i}": { "name": "예시-${i}", "url": "https://example.com/api/v2/items/${i}?token=abc${i}def", "qty": ${i % 100}, "memo": "한국어 값이 섞인 예시 데이터" }`;
|
||||
entries.push(row);
|
||||
len += row.length;
|
||||
i++;
|
||||
}
|
||||
return `{\n${entries.join(',\n')}\n}`;
|
||||
}
|
||||
|
||||
describe('v4 estimated-token cap — URL-dense Korean doc (field-failure shape)', () => {
|
||||
test('every chunk stays under the estimated-token cap', () => {
|
||||
const md = urlDenseKoreanRollup(60);
|
||||
const chunks = chunkText(md);
|
||||
expect(chunks.length).toBeGreaterThan(0);
|
||||
for (const c of chunks) {
|
||||
expect(estimateEmbeddingTokens(c.text)).toBeLessThanOrEqual(DEFAULT_MAX_EST_TOKENS);
|
||||
}
|
||||
});
|
||||
|
||||
test('no chunk reaches the measured 3K-char danger zone for URL soup', () => {
|
||||
const md = urlDenseKoreanRollup(60);
|
||||
const chunks = chunkText(md);
|
||||
// 1500 est tokens at the OTHER weight (0.75/char) bounds chunks to
|
||||
// ~2,000 chars for pure ASCII — well under the ~3,300 chars where
|
||||
// URL-dense content crosses ~2,050 real tokens (1.6 chars/token).
|
||||
for (const c of chunks) {
|
||||
expect(c.text.length).toBeLessThanOrEqual(2600);
|
||||
}
|
||||
});
|
||||
|
||||
test('content is preserved (no lines dropped by the cap)', () => {
|
||||
const md = urlDenseKoreanRollup(60);
|
||||
const chunks = chunkText(md);
|
||||
const joined = chunks.map((c) => c.text).join('\n');
|
||||
// Spot-check first / middle / last rollup lines survive chunking.
|
||||
for (const marker of ['항목 0', '항목 30', '항목 59']) {
|
||||
expect(joined).toContain(marker);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('v4 estimated-token cap — large JSON blocks', () => {
|
||||
test('7K-char pretty JSON through the prose path stays under the cap', () => {
|
||||
const md = `설정 파일 원문 보존:\n\n\`\`\`\n${bigJsonBlock(7000)}\n\`\`\`\n`;
|
||||
const chunks = chunkText(md);
|
||||
expect(chunks.length).toBeGreaterThan(1);
|
||||
for (const c of chunks) {
|
||||
expect(estimateEmbeddingTokens(c.text)).toBeLessThanOrEqual(DEFAULT_MAX_EST_TOKENS);
|
||||
}
|
||||
});
|
||||
|
||||
test('7K-char minified JSON (single whitespace-less token) stays under the cap', () => {
|
||||
const minified = bigJsonBlock(7000).replace(/\n\s*/g, '');
|
||||
const chunks = chunkText(minified);
|
||||
expect(chunks.length).toBeGreaterThan(1);
|
||||
for (const c of chunks) {
|
||||
expect(estimateEmbeddingTokens(c.text)).toBeLessThanOrEqual(DEFAULT_MAX_EST_TOKENS);
|
||||
}
|
||||
});
|
||||
|
||||
test('json fence via the code chunker stays under the cap (+header slack)', async () => {
|
||||
const chunks = await chunkCodeText(bigJsonBlock(7000), 'fence.json');
|
||||
expect(chunks.length).toBeGreaterThan(0);
|
||||
for (const c of chunks) {
|
||||
// buildChunk prepends a short "[JSON] fence.json:…" header AFTER the
|
||||
// body-level cap; allow ~60 est tokens of header slack. Real-token
|
||||
// safety margin (2,050 − overestimated 1,500) absorbs this easily.
|
||||
expect(estimateEmbeddingTokens(c.text)).toBeLessThanOrEqual(DEFAULT_MAX_EST_TOKENS + 60);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('v4 word-count floor — behavior preserved for normal content', () => {
|
||||
test('Latin prose chunking is unchanged by the floor (avg word < 6 chars)', () => {
|
||||
const prose = Array.from({ length: 120 }, (_, i) =>
|
||||
`This is sentence number ${i} and it talks about ordinary things in plain words.`,
|
||||
).join(' ');
|
||||
const chunks = chunkText(prose);
|
||||
// Historical behavior: ~1,560 whitespace words → multiple ~300-word chunks.
|
||||
expect(chunks.length).toBeGreaterThan(3);
|
||||
for (const c of chunks) {
|
||||
const words = c.text.split(/\s+/).length;
|
||||
expect(words).toBeLessThanOrEqual(300 * 1.5 + 50); // merge cap + overlap
|
||||
}
|
||||
});
|
||||
|
||||
test('Korean prose (CJK-dense, no URLs) never triggers the token cap', () => {
|
||||
const prose = Array.from({ length: 80 }, (_, i) =>
|
||||
`이 문장은 순수 한국어 산문의 청킹 동작을 확인하기 위한 ${i}번째 예시 문장입니다.`,
|
||||
).join(' ');
|
||||
const chunks = chunkText(prose);
|
||||
expect(chunks.length).toBeGreaterThan(1);
|
||||
for (const c of chunks) {
|
||||
// CJK-dense chunks are char-counted (≈450 max) — nowhere near 1500.
|
||||
expect(estimateEmbeddingTokens(c.text)).toBeLessThanOrEqual(700);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('capByEstimatedTokens unit behavior', () => {
|
||||
test('returns input unchanged when under the cap', () => {
|
||||
expect(capByEstimatedTokens('short text', 1500)).toEqual(['short text']);
|
||||
expect(capByEstimatedTokens('', 1500)).toEqual([]);
|
||||
});
|
||||
|
||||
test('prefers newline cut points within the lookback window', () => {
|
||||
const line = 'x'.repeat(100);
|
||||
const text = Array.from({ length: 40 }, () => line).join('\n');
|
||||
const pieces = capByEstimatedTokens(text, 1000);
|
||||
expect(pieces.length).toBeGreaterThan(1);
|
||||
for (const p of pieces) {
|
||||
// Every piece should be whole lines (multiples of the 100-char line).
|
||||
for (const l of p.split('\n')) {
|
||||
expect(l).toBe(line);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
test('makes forward progress on whitespace-less input (hard cut)', () => {
|
||||
const blob = 'a'.repeat(10_000);
|
||||
const pieces = capByEstimatedTokens(blob, 1000);
|
||||
expect(pieces.length).toBeGreaterThan(1);
|
||||
expect(pieces.join('')).toBe(blob);
|
||||
for (const p of pieces) {
|
||||
expect(estimateEmbeddingTokens(p)).toBeLessThanOrEqual(1000);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('estimateEmbeddingTokens — weight sanity', () => {
|
||||
test('overestimates URL-dense ASCII (0.75/char ≥ measured ~0.63/char)', () => {
|
||||
const url = 'https://docs.example.com/pages/a1b2c3d4e5f6a1b2c3d4e5f6a1b2c3d4?v=abc&ref=sample';
|
||||
const est = estimateEmbeddingTokens(url);
|
||||
expect(est).toBeGreaterThanOrEqual(Math.floor(url.length * 0.7));
|
||||
});
|
||||
|
||||
test('counts CJK at 1 token/char', () => {
|
||||
expect(estimateEmbeddingTokens('가나다라마')).toBe(5);
|
||||
});
|
||||
|
||||
test('whitespace is nearly free', () => {
|
||||
expect(estimateEmbeddingTokens(' \n\t ')).toBeLessThanOrEqual(1);
|
||||
});
|
||||
|
||||
test('empty string is 0', () => {
|
||||
expect(estimateEmbeddingTokens('')).toBe(0);
|
||||
});
|
||||
});
|
||||
@@ -172,59 +172,6 @@ describe('issue #972 — DB-source (gbrain extract links --source db)', () => {
|
||||
expect(strk!.link_type).toBe('wikilink_basename');
|
||||
});
|
||||
|
||||
test('flag ON → path-qualified wikilink outside DIR_PATTERN resolves via DB path', async () => {
|
||||
// `[[notes/struktura]]` — `notes` is not in DIR_PATTERN, so the ref
|
||||
// reaches the generic pass with its dirname intact. Regression: the DB
|
||||
// path queried the basename index with the raw literal (which is keyed
|
||||
// by final segments only), so path-qualified wikilinks outside
|
||||
// DIR_PATTERN silently produced zero edges while the FS path resolved
|
||||
// the identical content.
|
||||
await engine.putPage('notes/struktura', {
|
||||
type: 'concept' as any, title: 'Struktura Notes',
|
||||
compiled_truth: '', timeline: '',
|
||||
});
|
||||
await engine.putPage('concepts/knowledge-graph', {
|
||||
type: 'concept', title: 'Knowledge Graph',
|
||||
compiled_truth: 'Background in [[notes/struktura]].', timeline: '',
|
||||
});
|
||||
await engine.setConfig('link_resolution.global_basename', 'true');
|
||||
|
||||
await runExtract(engine, ['links', '--source', 'db']);
|
||||
|
||||
const outLinks = await engine.getLinks('concepts/knowledge-graph');
|
||||
const strk = outLinks.find(l => l.to_slug === 'notes/struktura');
|
||||
expect(strk).toBeDefined();
|
||||
expect(strk!.link_type).toBe('wikilink_basename');
|
||||
expect(strk!.link_source).toBe('wikilink-resolved');
|
||||
});
|
||||
|
||||
test('path-qualified wikilink never attaches to a basename-only sibling', async () => {
|
||||
// Both notes/struktura and wiki/struktura exist. The author wrote
|
||||
// `[[notes/struktura]]` — the written path must exclude wiki/struktura
|
||||
// (a bare `[[struktura]]` would legitimately match both).
|
||||
await engine.putPage('notes/struktura', {
|
||||
type: 'concept' as any, title: 'Struktura Notes',
|
||||
compiled_truth: '', timeline: '',
|
||||
});
|
||||
await engine.putPage('wiki/struktura', {
|
||||
type: 'concept' as any, title: 'Struktura Wiki',
|
||||
compiled_truth: '', timeline: '',
|
||||
});
|
||||
await engine.putPage('concepts/x', {
|
||||
type: 'concept', title: 'X',
|
||||
compiled_truth: 'See [[notes/struktura]].', timeline: '',
|
||||
});
|
||||
await engine.setConfig('link_resolution.global_basename', 'true');
|
||||
|
||||
await runExtract(engine, ['links', '--source', 'db']);
|
||||
|
||||
const outLinks = await engine.getLinks('concepts/x');
|
||||
const basenameLinks = outLinks
|
||||
.filter(l => l.link_type === 'wikilink_basename')
|
||||
.map(l => l.to_slug);
|
||||
expect(basenameLinks).toEqual(['notes/struktura']);
|
||||
});
|
||||
|
||||
test('flag OFF → no basename edges via DB path (back-compat)', async () => {
|
||||
await engine.putPage('projects/struktura', {
|
||||
type: 'project', title: 'Struktura',
|
||||
|
||||
@@ -39,7 +39,6 @@ import { runSkillOpt } from '../../src/core/skillopt/orchestrator.ts';
|
||||
import {
|
||||
bestPath,
|
||||
loadHistory,
|
||||
proposedPath,
|
||||
skillPath,
|
||||
} from '../../src/core/skillopt/version-store.ts';
|
||||
import { loadRejectedBuffer } from '../../src/core/skillopt/rejected-buffer.ts';
|
||||
@@ -742,7 +741,7 @@ describe('skillopt T3 — F11 held-out gate, ablation opts, no-DB-pollution', ()
|
||||
} finally { fixture.cleanup(); }
|
||||
});
|
||||
|
||||
test('--no-mutate writes proposed.md and best.md, leaves SKILL.md untouched', async () => {
|
||||
test('--no-mutate writes proposed.md (best.md), leaves SKILL.md untouched', async () => {
|
||||
const fixture = setupFixture(SKILL_PEOPLE_ONLY, CITATIONS_BENCHMARK);
|
||||
try {
|
||||
installStub({
|
||||
@@ -754,9 +753,10 @@ describe('skillopt T3 — F11 held-out gate, ablation opts, no-DB-pollution', ()
|
||||
const result = await runOnce(fixture, { noMutate: true });
|
||||
expect(result.outcome).toBe('accepted');
|
||||
expect(result.mutatedSkillFile).toBe(false);
|
||||
expect(result.proposedPath).toBe(proposedPath(fixture.skillsDir, SKILL));
|
||||
expect(result.proposedPath).toBeDefined();
|
||||
// proposed.md (best.md) exists and carries the improvement.
|
||||
expect(fs.existsSync(result.proposedPath!)).toBe(true);
|
||||
expect(fs.readFileSync(result.proposedPath!, 'utf8')).toContain('## Citations');
|
||||
expect(fs.readFileSync(bestPath(fixture.skillsDir, SKILL), 'utf8')).toContain('## Citations');
|
||||
// SKILL.md on disk is UNCHANGED (still People-only).
|
||||
const skill = fs.readFileSync(skillPath(fixture.skillsDir, SKILL), 'utf8');
|
||||
expect(skill).not.toContain('## Citations');
|
||||
|
||||
@@ -803,107 +803,3 @@ describe('embedAllStale --source threading (D7)', () => {
|
||||
expect((firstCallOpts as { sourceId?: string }).sourceId).toBe('media-corpus');
|
||||
});
|
||||
});
|
||||
|
||||
// ────────────────────────────────────────────────────────────────
|
||||
// Code metadata preservation across re-embed (regression for #769)
|
||||
// ────────────────────────────────────────────────────────────────
|
||||
//
|
||||
// gbrain v0.30.1 and earlier silently clobbered code-chunk metadata
|
||||
// (language, symbol_name, symbol_type, start_line, end_line,
|
||||
// parent_symbol_path, doc_comment, symbol_name_qualified) on every
|
||||
// re-embed pass. The chunker populated those columns at import time,
|
||||
// but embed.ts loaded chunks via getChunks then mapped them to a
|
||||
// stripped ChunkInput carrying only 5 fields. upsertChunks then
|
||||
// OVERWROTE (not COALESCEd) the metadata columns from EXCLUDED, so
|
||||
// re-embed wiped them to NULL. End result on a real brain: 4875 code
|
||||
// pages, 47866 chunks, all with NULL language/symbol_name/symbol_type;
|
||||
// code-def returned 0 hits across every indexed repo.
|
||||
//
|
||||
// All three runEmbed paths (--stale autopilot, --all, --slugs) must
|
||||
// thread metadata through the re-upsert. Tests below assert that the
|
||||
// engine.upsertChunks call carries the same metadata it loaded.
|
||||
|
||||
describe('runEmbed preserves code-chunk metadata across re-embed (regression for #769)', () => {
|
||||
const fullCodeChunk = {
|
||||
chunk_index: 0,
|
||||
chunk_text: '[Java] foo/Bar.java:10-20 method baz',
|
||||
chunk_source: 'compiled_truth' as const,
|
||||
embedded_at: null,
|
||||
token_count: 12,
|
||||
language: 'java',
|
||||
symbol_name: 'baz',
|
||||
symbol_type: 'function',
|
||||
start_line: 10,
|
||||
end_line: 20,
|
||||
parent_symbol_path: ['Bar'],
|
||||
doc_comment: 'does the thing',
|
||||
symbol_name_qualified: 'Bar.baz',
|
||||
};
|
||||
|
||||
function metadataOf(chunk: any) {
|
||||
return {
|
||||
language: chunk.language,
|
||||
symbol_name: chunk.symbol_name,
|
||||
symbol_type: chunk.symbol_type,
|
||||
start_line: chunk.start_line,
|
||||
end_line: chunk.end_line,
|
||||
parent_symbol_path: chunk.parent_symbol_path,
|
||||
doc_comment: chunk.doc_comment,
|
||||
symbol_name_qualified: chunk.symbol_name_qualified,
|
||||
};
|
||||
}
|
||||
|
||||
test('--stale (autopilot path) carries code metadata into upsertChunks', async () => {
|
||||
const stale = [{
|
||||
slug: 'code-page',
|
||||
chunk_index: 0,
|
||||
chunk_text: fullCodeChunk.chunk_text,
|
||||
chunk_source: 'compiled_truth',
|
||||
model: null,
|
||||
token_count: 12,
|
||||
}];
|
||||
let upsertChunkArgs: any[] | null = null;
|
||||
const engine = mockEngine({
|
||||
countStaleChunks: async () => 1,
|
||||
listStaleChunks: async () => stale,
|
||||
getChunks: async () => [fullCodeChunk],
|
||||
upsertChunks: async (_slug: string, chunks: any[]) => { upsertChunkArgs = chunks; },
|
||||
});
|
||||
|
||||
await runEmbed(engine, ['--stale']);
|
||||
|
||||
expect(upsertChunkArgs).not.toBeNull();
|
||||
expect(upsertChunkArgs!).toHaveLength(1);
|
||||
expect(metadataOf(upsertChunkArgs![0])).toEqual(metadataOf(fullCodeChunk));
|
||||
});
|
||||
|
||||
test('--all (full re-embed) carries code metadata into upsertChunks', async () => {
|
||||
let upsertChunkArgs: any[] | null = null;
|
||||
const engine = mockEngine({
|
||||
listPages: async () => [{ slug: 'code-page' }],
|
||||
getChunks: async () => [fullCodeChunk],
|
||||
upsertChunks: async (_slug: string, chunks: any[]) => { upsertChunkArgs = chunks; },
|
||||
});
|
||||
|
||||
await runEmbed(engine, ['--all']);
|
||||
|
||||
expect(upsertChunkArgs).not.toBeNull();
|
||||
expect(upsertChunkArgs!).toHaveLength(1);
|
||||
expect(metadataOf(upsertChunkArgs![0])).toEqual(metadataOf(fullCodeChunk));
|
||||
});
|
||||
|
||||
test('--slugs (per-page embed) carries code metadata into upsertChunks', async () => {
|
||||
let upsertChunkArgs: any[] | null = null;
|
||||
const engine = mockEngine({
|
||||
getPage: async () => ({ slug: 'code-page', compiled_truth: 'x', timeline: '' }),
|
||||
getChunks: async () => [fullCodeChunk],
|
||||
upsertChunks: async (_slug: string, chunks: any[]) => { upsertChunkArgs = chunks; },
|
||||
});
|
||||
|
||||
await runEmbed(engine, ['--slugs', 'code-page']);
|
||||
|
||||
expect(upsertChunkArgs).not.toBeNull();
|
||||
expect(upsertChunkArgs!).toHaveLength(1);
|
||||
expect(metadataOf(upsertChunkArgs![0])).toEqual(metadataOf(fullCodeChunk));
|
||||
});
|
||||
});
|
||||
|
||||
@@ -309,62 +309,3 @@ describe('isEvalCaptureEnabled / isEvalScrubEnabled (CONTRIBUTOR_MODE-gated)', (
|
||||
} finally { restore(); }
|
||||
});
|
||||
});
|
||||
|
||||
describe('DB-plane stash (#1475): GBRAIN_EVAL_CAPTURE / GBRAIN_EVAL_SCRUB_PII', () => {
|
||||
// connectEngine stamps `gbrain config set eval.capture` (DB plane) onto
|
||||
// these env vars because ctx.config is the sync file-plane load. Without
|
||||
// the stash check the DB value was written and never read.
|
||||
const origCapture = process.env.GBRAIN_EVAL_CAPTURE;
|
||||
const origScrub = process.env.GBRAIN_EVAL_SCRUB_PII;
|
||||
const origMode = process.env.GBRAIN_CONTRIBUTOR_MODE;
|
||||
const restore = () => {
|
||||
if (origCapture === undefined) delete process.env.GBRAIN_EVAL_CAPTURE;
|
||||
else process.env.GBRAIN_EVAL_CAPTURE = origCapture;
|
||||
if (origScrub === undefined) delete process.env.GBRAIN_EVAL_SCRUB_PII;
|
||||
else process.env.GBRAIN_EVAL_SCRUB_PII = origScrub;
|
||||
if (origMode === undefined) delete process.env.GBRAIN_CONTRIBUTOR_MODE;
|
||||
else process.env.GBRAIN_CONTRIBUTOR_MODE = origMode;
|
||||
};
|
||||
|
||||
test('stash=true turns capture on when file plane is silent (the #1475 repro)', () => {
|
||||
delete process.env.GBRAIN_CONTRIBUTOR_MODE;
|
||||
process.env.GBRAIN_EVAL_CAPTURE = 'true';
|
||||
try {
|
||||
expect(isEvalCaptureEnabled(null)).toBe(true);
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
const noEval: any = { engine: 'pglite' };
|
||||
expect(isEvalCaptureEnabled(noEval)).toBe(true);
|
||||
} finally { restore(); }
|
||||
});
|
||||
|
||||
test('stash=false wins over CONTRIBUTOR_MODE=1 (explicit per-key beats broad flag)', () => {
|
||||
process.env.GBRAIN_CONTRIBUTOR_MODE = '1';
|
||||
process.env.GBRAIN_EVAL_CAPTURE = 'false';
|
||||
try {
|
||||
expect(isEvalCaptureEnabled(null)).toBe(false);
|
||||
} finally { restore(); }
|
||||
});
|
||||
|
||||
test('file-plane explicit value still wins over the stash', () => {
|
||||
process.env.GBRAIN_EVAL_CAPTURE = 'true';
|
||||
try {
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
const disabled: any = { engine: 'pglite', eval: { capture: false } };
|
||||
expect(isEvalCaptureEnabled(disabled)).toBe(false);
|
||||
} finally { restore(); }
|
||||
});
|
||||
|
||||
test('scrub stash: false disables, file plane wins, default stays true', () => {
|
||||
process.env.GBRAIN_EVAL_SCRUB_PII = 'false';
|
||||
try {
|
||||
expect(isEvalScrubEnabled(null)).toBe(false);
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
const fileWins: any = { engine: 'pglite', eval: { scrub_pii: true } };
|
||||
expect(isEvalScrubEnabled(fileWins)).toBe(true);
|
||||
} finally { restore(); }
|
||||
delete process.env.GBRAIN_EVAL_SCRUB_PII;
|
||||
try {
|
||||
expect(isEvalScrubEnabled(null)).toBe(true);
|
||||
} finally { restore(); }
|
||||
});
|
||||
});
|
||||
|
||||
@@ -403,77 +403,6 @@ describe('extractPageLinks', () => {
|
||||
expect(candidates).toEqual([]);
|
||||
});
|
||||
|
||||
test('path-qualified wikilink outside DIR_PATTERN queries by final segment', async () => {
|
||||
// `[[notes/struktura]]` (dir not in DIR_PATTERN) falls to the generic
|
||||
// pass. The resolver's basename index is keyed by final path segments,
|
||||
// so the lookup must strip the dirname — mirroring the FS path
|
||||
// (resolveSlugAll). Regression: the raw literal was passed through,
|
||||
// which never matched, so these links silently dropped.
|
||||
const seen: string[] = [];
|
||||
const resolver: SlugResolver = {
|
||||
resolve: async () => null,
|
||||
resolveBasenameMatches: async (name) => {
|
||||
seen.push(name);
|
||||
return name === 'struktura' ? ['notes/struktura'] : [];
|
||||
},
|
||||
};
|
||||
const { candidates } = await extractPageLinks(
|
||||
'concepts/x', 'See [[notes/struktura]].',
|
||||
{}, 'concept', resolver, { globalBasename: true },
|
||||
);
|
||||
expect(seen).toContain('struktura');
|
||||
expect(seen).not.toContain('notes/struktura');
|
||||
expect(candidates.map(c => c.targetSlug)).toEqual(['notes/struktura']);
|
||||
expect(candidates[0].linkType).toBe('wikilink_basename');
|
||||
expect(candidates[0].linkSource).toBe('wikilink-resolved');
|
||||
});
|
||||
|
||||
test('path-qualified wikilink keeps only matches ending with the written path', async () => {
|
||||
// The written path disambiguates: `[[notes/struktura]]` must never
|
||||
// attach to `wiki/struktura` even though both share the basename.
|
||||
const resolver: SlugResolver = {
|
||||
resolve: async () => null,
|
||||
resolveBasenameMatches: async (name) =>
|
||||
name === 'struktura' ? ['notes/struktura', 'wiki/struktura'] : [],
|
||||
};
|
||||
const { candidates } = await extractPageLinks(
|
||||
'concepts/x', 'See [[notes/struktura]].',
|
||||
{}, 'concept', resolver, { globalBasename: true },
|
||||
);
|
||||
expect(candidates.map(c => c.targetSlug)).toEqual(['notes/struktura']);
|
||||
});
|
||||
|
||||
test('path-qualified wikilink matches a deeper real slug by path suffix', async () => {
|
||||
// The page lives at vault/notes/struktura; the author wrote the shorter
|
||||
// tail `[[notes/struktura]]`. Suffix matching connects them, while the
|
||||
// basename-only sibling `wiki/struktura` stays excluded.
|
||||
const resolver: SlugResolver = {
|
||||
resolve: async () => null,
|
||||
resolveBasenameMatches: async (name) =>
|
||||
name === 'struktura' ? ['vault/notes/struktura', 'wiki/struktura'] : [],
|
||||
};
|
||||
const { candidates } = await extractPageLinks(
|
||||
'concepts/x', 'See [[notes/struktura]].',
|
||||
{}, 'concept', resolver, { globalBasename: true },
|
||||
);
|
||||
expect(candidates.map(c => c.targetSlug)).toEqual(['vault/notes/struktura']);
|
||||
});
|
||||
|
||||
test('path-qualified self-link is dropped like the bare form', async () => {
|
||||
// `[[notes/struktura]]` written on notes/struktura itself must not
|
||||
// produce a self-loop (same guard as the bare `[[own-tail]]` case).
|
||||
const resolver: SlugResolver = {
|
||||
resolve: async () => null,
|
||||
resolveBasenameMatches: async (name) =>
|
||||
name === 'struktura' ? ['notes/struktura'] : [],
|
||||
};
|
||||
const { candidates } = await extractPageLinks(
|
||||
'notes/struktura', 'See [[notes/struktura]].',
|
||||
{}, 'concept', resolver, { globalBasename: true },
|
||||
);
|
||||
expect(candidates).toEqual([]);
|
||||
});
|
||||
|
||||
test('bare wikilink resolution does not interfere with DIR_PATTERN wikilinks', async () => {
|
||||
// 2b refs (people/alice) take the verb-inferred type;
|
||||
// 2c refs (struktura) take wikilink_basename. Same call.
|
||||
|
||||
@@ -302,29 +302,4 @@ describe('loadConfigWithEngine (Phase 4 / F3)', () => {
|
||||
expect(merged?.engine).toBe('pglite');
|
||||
});
|
||||
});
|
||||
|
||||
describe('eval.* DB-plane merge (#1475)', () => {
|
||||
test('gbrain config set eval.capture true reaches the merged config', async () => {
|
||||
// The #1475 repro: DB plane has eval.capture=true, file plane silent.
|
||||
// Pre-fix the merge skipped eval.* entirely and capture never fired.
|
||||
const base: GBrainConfig = { engine: 'pglite' };
|
||||
const engine = makeEngine({ 'eval.capture': 'true', 'eval.scrub_pii': 'false' });
|
||||
const merged = await loadConfigWithEngine(engine, base);
|
||||
expect(merged?.eval?.capture).toBe(true);
|
||||
expect(merged?.eval?.scrub_pii).toBe(false);
|
||||
});
|
||||
|
||||
test('file plane wins per key; DB fills only the gaps', async () => {
|
||||
const base: GBrainConfig = { engine: 'pglite', eval: { capture: false } };
|
||||
const engine = makeEngine({ 'eval.capture': 'true', 'eval.scrub_pii': 'false' });
|
||||
const merged = await loadConfigWithEngine(engine, base);
|
||||
expect(merged?.eval?.capture).toBe(false); // file wins
|
||||
expect(merged?.eval?.scrub_pii).toBe(false); // DB fills the gap
|
||||
});
|
||||
|
||||
test('no eval keys anywhere leaves cfg.eval undefined', async () => {
|
||||
const merged = await loadConfigWithEngine(makeEngine({}), { engine: 'pglite' });
|
||||
expect(merged?.eval).toBeUndefined();
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
@@ -1,62 +0,0 @@
|
||||
import { describe, expect, test } from 'bun:test';
|
||||
import {
|
||||
extractCycleFreshnessSourceIds,
|
||||
parseMaintainArgs,
|
||||
} from '../src/commands/maintain.ts';
|
||||
import type { Check } from '../src/commands/doctor.ts';
|
||||
|
||||
describe('maintain args', () => {
|
||||
test('defaults to dry-run unless --safe is explicit', () => {
|
||||
expect(parseMaintainArgs([])).toMatchObject({
|
||||
safe: false,
|
||||
dryRun: true,
|
||||
json: false,
|
||||
});
|
||||
});
|
||||
|
||||
test('--safe enables mutating safe mode', () => {
|
||||
expect(parseMaintainArgs(['--safe', '--json'])).toMatchObject({
|
||||
safe: true,
|
||||
dryRun: false,
|
||||
json: true,
|
||||
});
|
||||
});
|
||||
|
||||
test('--dry-run wins over --safe', () => {
|
||||
expect(parseMaintainArgs(['--safe', '--dry-run'])).toMatchObject({
|
||||
safe: true,
|
||||
dryRun: true,
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe('cycle freshness source extraction', () => {
|
||||
test('extracts stale source ids from doctor messages', () => {
|
||||
const checks: Check[] = [
|
||||
{
|
||||
name: 'cycle_freshness',
|
||||
status: 'fail',
|
||||
message: "Source 'brain-sync-remote-teffur' last cycled 40h ago. Run `gbrain dream --source <id>`.",
|
||||
},
|
||||
{
|
||||
name: 'cycle_freshness',
|
||||
status: 'fail',
|
||||
message: "Source 'wiki' last cycled 25h ago. Source 'wiki' last cycled 25h ago.",
|
||||
},
|
||||
];
|
||||
|
||||
expect(extractCycleFreshnessSourceIds(checks)).toEqual([
|
||||
'brain-sync-remote-teffur',
|
||||
'wiki',
|
||||
]);
|
||||
});
|
||||
|
||||
test('ignores ok and unrelated checks', () => {
|
||||
const checks: Check[] = [
|
||||
{ name: 'cycle_freshness', status: 'ok', message: "Source 'fresh' last cycled recently." },
|
||||
{ name: 'frontmatter_integrity', status: 'warn', message: "Source 'wiki' has frontmatter issues." },
|
||||
];
|
||||
|
||||
expect(extractCycleFreshnessSourceIds(checks)).toEqual([]);
|
||||
});
|
||||
});
|
||||
@@ -1,17 +0,0 @@
|
||||
import { describe, expect, it } from 'bun:test';
|
||||
import { readFileSync } from 'fs';
|
||||
import { join } from 'path';
|
||||
|
||||
describe('root OpenClaw plugin manifest', () => {
|
||||
it('declares the id required by OpenClaw plugin installs', () => {
|
||||
const manifest = JSON.parse(readFileSync(join(import.meta.dir, '..', 'openclaw.plugin.json'), 'utf8'));
|
||||
const entrySource = readFileSync(join(import.meta.dir, '..', 'src', 'openclaw-context-engine.ts'), 'utf8');
|
||||
const entryId = entrySource.match(/id:\s*'([^']+)'/)?.[1];
|
||||
|
||||
expect(manifest.id).toBe(entryId);
|
||||
expect(manifest.configSchema).toBeDefined();
|
||||
expect(typeof manifest.configSchema).toBe('object');
|
||||
expect(manifest.contracts?.contextEngines).toContain('gbrain-context');
|
||||
expect(entrySource).toContain('export function register');
|
||||
});
|
||||
});
|
||||
@@ -186,67 +186,11 @@ describe('shouldExclude — orphan filter regression (preserve curation)', () =>
|
||||
expect(shouldExclude('entities/anonymous')).toBe(true);
|
||||
expect(shouldExclude('atoms/fact-123')).toBe(true);
|
||||
expect(shouldExclude('skills/gbrain-operations')).toBe(true);
|
||||
expect(shouldExclude('dreaming/light/2026-07-20')).toBe(true);
|
||||
expect(shouldExclude('daily/2026-07-20')).toBe(true);
|
||||
expect(shouldExclude('agent-openclaw/daily/2026-07-20')).toBe(true);
|
||||
});
|
||||
|
||||
test('workspace convention slugs are excluded', () => {
|
||||
expect(shouldExclude('_brain-conventions')).toBe(true);
|
||||
expect(shouldExclude('_templates/decision')).toBe(true);
|
||||
expect(shouldExclude('extracts/2026-06-30/takes.proposed/round-single')).toBe(true);
|
||||
expect(shouldExclude('2026-07-20')).toBe(true);
|
||||
expect(shouldExclude('2026-07-20-qa-sweep')).toBe(true);
|
||||
expect(shouldExclude('agents/arya/identity')).toBe(true);
|
||||
expect(shouldExclude('agents/arya/memory/dreaming/deep/2026-07-20')).toBe(true);
|
||||
});
|
||||
|
||||
test('regular slugs are NOT excluded', () => {
|
||||
expect(shouldExclude('people/alice')).toBe(false);
|
||||
expect(shouldExclude('companies/acme')).toBe(false);
|
||||
expect(shouldExclude('writing/post-1')).toBe(false);
|
||||
expect(shouldExclude('agents/arya/qa-reports/launch-review')).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('getHealth orphan_pages uses shared exclusion policy', () => {
|
||||
test('excluded convention islands do not count against health', async () => {
|
||||
await engine.putPage('_templates/decision', {
|
||||
type: 'template', title: 'Decision', compiled_truth: 'template', timeline: '', frontmatter: {},
|
||||
});
|
||||
await engine.putPage('skills/arya/source-check', {
|
||||
type: 'concept', title: 'Skill', compiled_truth: 'skill', timeline: '', frontmatter: {},
|
||||
});
|
||||
await engine.putPage('agents/arya/identity', {
|
||||
type: 'note', title: 'Identity', compiled_truth: 'identity', timeline: '', frontmatter: {},
|
||||
});
|
||||
await engine.putPage('people/alice', {
|
||||
type: 'person', title: 'Alice', compiled_truth: 'real island', timeline: '', frontmatter: {},
|
||||
});
|
||||
|
||||
const health = await engine.getHealth();
|
||||
|
||||
expect(health.orphan_pages).toBe(1);
|
||||
});
|
||||
|
||||
test('per-brain config overrides (orphans.exclude_*) also apply to health', async () => {
|
||||
await engine.putPage('my-private-folder/secret-ref', {
|
||||
type: 'note', title: 'Ref', compiled_truth: 'ref', timeline: '', frontmatter: {},
|
||||
});
|
||||
await engine.putPage('one-off-fixture-page', {
|
||||
type: 'note', title: 'Fixture', compiled_truth: 'fixture', timeline: '', frontmatter: {},
|
||||
});
|
||||
await engine.putPage('people/alice', {
|
||||
type: 'person', title: 'Alice', compiled_truth: 'real island', timeline: '', frontmatter: {},
|
||||
});
|
||||
|
||||
expect((await engine.getHealth()).orphan_pages).toBe(3);
|
||||
|
||||
await engine.setConfig('orphans.exclude_prefixes', 'my-private-folder/');
|
||||
await engine.setConfig('orphans.exclude_slugs', 'one-off-fixture-page');
|
||||
expect((await engine.getHealth()).orphan_pages).toBe(1);
|
||||
|
||||
await engine.unsetConfig('orphans.exclude_prefixes');
|
||||
await engine.unsetConfig('orphans.exclude_slugs');
|
||||
});
|
||||
});
|
||||
|
||||
@@ -66,10 +66,6 @@ describe('shouldExclude', () => {
|
||||
expect(shouldExclude('templates/meeting-note')).toBe(true);
|
||||
});
|
||||
|
||||
test('excludes deny-prefix: _templates/', () => {
|
||||
expect(shouldExclude('_templates/meeting-note')).toBe(true);
|
||||
});
|
||||
|
||||
test('excludes deny-prefix: openclaw/config/', () => {
|
||||
expect(shouldExclude('openclaw/config/agent')).toBe(true);
|
||||
});
|
||||
@@ -90,44 +86,10 @@ describe('shouldExclude', () => {
|
||||
expect(shouldExclude('entities/product-hunt')).toBe(true);
|
||||
});
|
||||
|
||||
test('excludes first-segment: skills, dreaming, and daily', () => {
|
||||
expect(shouldExclude('skills/arya/source-check')).toBe(true);
|
||||
expect(shouldExclude('dreaming/light/2026-07-20')).toBe(true);
|
||||
expect(shouldExclude('daily/2026-07-20')).toBe(true);
|
||||
expect(shouldExclude('agent-openclaw/daily/2026-07-20')).toBe(true);
|
||||
});
|
||||
|
||||
test('excludes root date logs and agent workspace conventions', () => {
|
||||
expect(shouldExclude('_brain-conventions')).toBe(true);
|
||||
expect(shouldExclude('2026-07-20')).toBe(true);
|
||||
expect(shouldExclude('2026-07-20-qa-sweep')).toBe(true);
|
||||
expect(shouldExclude('agents/arya/identity')).toBe(true);
|
||||
expect(shouldExclude('agents/arya/memory/dreaming/deep/2026-07-20')).toBe(true);
|
||||
});
|
||||
|
||||
test('excludes generated extracts', () => {
|
||||
expect(shouldExclude('extracts/2026-06-30/takes.proposed/round-single')).toBe(true);
|
||||
});
|
||||
|
||||
test('brain-specific exclusions come from config overrides, not global defaults', () => {
|
||||
// No baked-in defaults for these:
|
||||
expect(shouldExclude('my-private-folder/some-secret-ref.md')).toBe(false);
|
||||
expect(shouldExclude('one-off-fixture-page')).toBe(false);
|
||||
// The per-brain config plane (orphans.exclude_prefixes / exclude_slugs):
|
||||
const overrides = {
|
||||
excludePrefixes: ['my-private-folder/'],
|
||||
excludeSlugs: ['one-off-fixture-page'],
|
||||
};
|
||||
expect(shouldExclude('my-private-folder/some-secret-ref.md', overrides)).toBe(true);
|
||||
expect(shouldExclude('one-off-fixture-page', overrides)).toBe(true);
|
||||
expect(shouldExclude('people/jane-doe', overrides)).toBe(false);
|
||||
});
|
||||
|
||||
test('does NOT exclude a normal content page', () => {
|
||||
expect(shouldExclude('companies/acme')).toBe(false);
|
||||
expect(shouldExclude('people/jane-doe')).toBe(false);
|
||||
expect(shouldExclude('projects/gbrain')).toBe(false);
|
||||
expect(shouldExclude('agents/arya/qa-reports/launch-review')).toBe(false);
|
||||
});
|
||||
|
||||
test('does NOT exclude a page ending with log-like text that is not /log', () => {
|
||||
|
||||
@@ -218,21 +218,15 @@ describe('progress reporter', () => {
|
||||
test('only one process-level signal handler installed across many reporters', () => {
|
||||
// Baseline: one handler already installed by prior tests in this file.
|
||||
const installedBefore = __signalHandlerInstalledForTest();
|
||||
// liveReporters is process-global: earlier test files in the same shard
|
||||
// can leave a live entry behind (e.g. a production path that skips
|
||||
// finish() on an error branch). Assert NET-zero leak from THIS test's
|
||||
// lifecycles, not an absolute zero we don't control — same tolerance
|
||||
// the handler assertion below already applies via `installedBefore`.
|
||||
const liveBefore = __liveReporterCountForTest();
|
||||
const { stream } = sink(false);
|
||||
for (let i = 0; i < 50; i++) {
|
||||
const p = createProgress({ mode: 'json', stream, minIntervalMs: 0, minItems: 1 });
|
||||
p.start(`phase_${i}`, 1);
|
||||
p.finish();
|
||||
}
|
||||
// After 50 reporter lifecycles, still exactly one handler and zero NEWLY leaked live entries.
|
||||
// After 50 reporter lifecycles, still exactly one handler and zero leaked live entries.
|
||||
expect(__signalHandlerInstalledForTest()).toBe(installedBefore || true);
|
||||
expect(__liveReporterCountForTest()).toBe(liveBefore);
|
||||
expect(__liveReporterCountForTest()).toBe(0);
|
||||
});
|
||||
|
||||
test('startHeartbeat() fires heartbeats and stop() clears', async () => {
|
||||
|
||||
@@ -12,11 +12,9 @@ import {
|
||||
bestPath,
|
||||
historyPath,
|
||||
loadHistory,
|
||||
proposedPath,
|
||||
revertAllPending,
|
||||
skillPath,
|
||||
versionsDir,
|
||||
writeProposed,
|
||||
} from '../../src/core/skillopt/version-store.ts';
|
||||
|
||||
let tmpDir: string;
|
||||
@@ -81,19 +79,6 @@ describe('acceptCandidate (D8 two-phase commit)', () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe('writeProposed', () => {
|
||||
test('writes distinct best and proposed artifacts without mutating SKILL.md (#2635)', () => {
|
||||
const candidate = '---\nname: test\n---\nproposed body\n';
|
||||
|
||||
const written = writeProposed(tmpDir, SKILL, candidate);
|
||||
|
||||
expect(written).toBe(proposedPath(tmpDir, SKILL));
|
||||
expect(fs.readFileSync(bestPath(tmpDir, SKILL), 'utf8')).toBe(candidate);
|
||||
expect(fs.readFileSync(proposedPath(tmpDir, SKILL), 'utf8')).toBe(candidate);
|
||||
expect(fs.readFileSync(skillPath(tmpDir, SKILL), 'utf8')).toContain('baseline body');
|
||||
});
|
||||
});
|
||||
|
||||
describe('revertAllPending (D8 crash recovery)', () => {
|
||||
test('no-op when no pending rows', () => {
|
||||
const reverted = revertAllPending(tmpDir, SKILL);
|
||||
|
||||
Reference in New Issue
Block a user