mirror of
https://github.com/garrytan/gbrain.git
synced 2026-08-16 09:52:22 +00:00
Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7b8676be3e | ||
|
|
b8376f7327 |
@@ -206,7 +206,11 @@ jobs:
|
||||
needs: cache-check
|
||||
if: needs.cache-check.outputs.hit != 'true'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
# 20 (was 15): shard 4 runs ~14.5 min on master (dream.test.ts ~29s/test
|
||||
# dominates it) and hits the 15-min ceiling on slower runners, cancelling
|
||||
# mid-run with 0 test failures. Rebalancing via
|
||||
# scripts/mine-shard-weights.ts is the real fix; this stops the bleeding.
|
||||
timeout-minutes: 20
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
|
||||
@@ -1743,15 +1743,7 @@ async function extractStaleFromDB(
|
||||
// `page.updated_at.toISOString()` — the JS Date is ms-truncated, so the
|
||||
// µs-precision DB updated_at stayed strictly greater and the page never
|
||||
// cleared on Postgres. Stamping the exact value makes them equal.
|
||||
//
|
||||
// Version-arm floor: a page last edited BEFORE LINK_EXTRACTOR_VERSION_TS
|
||||
// would otherwise be stamped below the version watermark and stay
|
||||
// permanently stale (`links_extracted_at < versionTs` re-fires every run).
|
||||
// Stamp max(updated_at, versionTs) — versionTs is always a past release
|
||||
// date, so a concurrent edit's now() still exceeds the stamp and D4 holds.
|
||||
// Tie at ms precision picks updated_at_iso (its µs ≥ versionTs's .000000).
|
||||
const stampTs = new Date(page.updated_at_iso) >= new Date(versionTs) ? page.updated_at_iso : versionTs;
|
||||
processedRefs.push({ slug: page.slug, source_id: page.source_id, extractedAt: stampTs });
|
||||
processedRefs.push({ slug: page.slug, source_id: page.source_id, extractedAt: page.updated_at_iso });
|
||||
}
|
||||
|
||||
// Flush NON-swallowing (CDX-4): a throw here propagates out of the sweep so
|
||||
|
||||
@@ -28,7 +28,7 @@ import { ensureWellFormed } from './text-safe.ts';
|
||||
* OR updated_at > links_extracted_at`. It is an ISO-8601 string (NOT a number) —
|
||||
* the column is TIMESTAMPTZ and the predicate binds it as `::timestamptz`.
|
||||
*/
|
||||
export const LINK_EXTRACTOR_VERSION_TS = '2026-07-21T00:00:00Z';
|
||||
export const LINK_EXTRACTOR_VERSION_TS = '2026-05-31T00:00:00Z';
|
||||
|
||||
// ─── Entity references ──────────────────────────────────────────
|
||||
|
||||
@@ -80,10 +80,10 @@ export type LinkResolutionType = 'qualified' | 'unqualified';
|
||||
* Directory prefix whitelist. These are the top-level slug dirs the extractor
|
||||
* recognizes as entity references. Upstream canonical + our extensions:
|
||||
* - Gbrain canonical: people, companies, meetings, concepts, deal, civic, project, source, media, yc, projects
|
||||
* - Our domain extensions: tech, finance, personal, openclaw, ops (domain-organized wikis)
|
||||
* - Our domain extensions: tech, finance, personal, openclaw (domain-organized wikis)
|
||||
* - Our entity prefix: entities (we kept some legacy entities/projects/ pages)
|
||||
*/
|
||||
const DIR_PATTERN = '(?:people|companies|meetings|concepts|deal|civic|project|projects|source|media|yc|tech|finance|personal|openclaw|entities|ops)';
|
||||
const DIR_PATTERN = '(?:people|companies|meetings|concepts|deal|civic|project|projects|source|media|yc|tech|finance|personal|openclaw|entities)';
|
||||
|
||||
/**
|
||||
* Match `[Name](path)` markdown links pointing to entity directories.
|
||||
@@ -865,16 +865,7 @@ export function queryBasenameIndex(idx: Map<string, string[]>, name: string): st
|
||||
if (!name || typeof name !== 'string') return [];
|
||||
const trimmed = name.trim();
|
||||
if (!trimmed) return [];
|
||||
let hit = idx.get(trimmed) ?? idx.get(trimmed.toLowerCase()) ?? idx.get(normalizeBasename(trimmed));
|
||||
// Issue #2576 bug 2: path-style refs (`runbooks/2026-05-01-x`) from dirs
|
||||
// outside DIR_PATTERN reach here, but normalizeBasename strips slashes
|
||||
// into a garbage key (`runbooks2026-05-01-x`) that can never hit the
|
||||
// tail-keyed index. Fall back to the path tail so qualified refs resolve
|
||||
// by basename like everything else.
|
||||
if (!hit && trimmed.includes('/')) {
|
||||
const tail = trimmed.slice(trimmed.lastIndexOf('/') + 1).trim();
|
||||
if (tail) hit = idx.get(tail) ?? idx.get(tail.toLowerCase()) ?? idx.get(normalizeBasename(tail));
|
||||
}
|
||||
const hit = idx.get(trimmed) ?? idx.get(trimmed.toLowerCase()) ?? idx.get(normalizeBasename(trimmed));
|
||||
return hit ? [...hit].sort(basenameSort) : [];
|
||||
}
|
||||
|
||||
|
||||
@@ -23,6 +23,15 @@
|
||||
* hold conventions and shared rule files, not skills. Files like
|
||||
* `_brain-filing-rules.md` live at the root and are not considered
|
||||
* skills by either loader.
|
||||
*
|
||||
* ClawHub-installed workspace skills (#1767): a skill dir carrying
|
||||
* `.clawhub/origin.json` is an externally-managed runtime integration
|
||||
* (e.g. an email or catalog skill), not a gbrain-routable skill. The
|
||||
* derive path SKIPS those so `gbrain doctor` resolver_health doesn't
|
||||
* hard-fail on them — UNLESS the skill's SKILL.md frontmatter declares
|
||||
* `triggers:`, which is the explicit opt-in to gbrain routing (and the
|
||||
* same surface that makes it reachable). An explicit manifest.json that
|
||||
* lists a ClawHub skill also keeps strict checking (verbatim path).
|
||||
*/
|
||||
|
||||
import { existsSync, readFileSync, readdirSync, statSync } from 'fs';
|
||||
@@ -60,9 +69,27 @@ function parseSkillName(skillMdPath: string): string | null {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Does the SKILL.md frontmatter declare a `triggers:` key? A ClawHub-
|
||||
* installed skill that ships gbrain `triggers:` has explicitly opted in
|
||||
* to gbrain routing and gets full resolver checks (#1767).
|
||||
*/
|
||||
function declaresTriggers(skillMdPath: string): boolean {
|
||||
try {
|
||||
const content = readFileSync(skillMdPath, 'utf-8');
|
||||
const fmMatch = content.match(/^---\n([\s\S]*?)\n---/);
|
||||
if (!fmMatch) return false;
|
||||
return /^triggers:/m.test(fmMatch[1]);
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Walk skillsDir, return every `<skillsDir>/<dir>/SKILL.md` as a
|
||||
* ManifestEntry. Dotfile and underscore-prefixed dirs are skipped.
|
||||
* ManifestEntry. Dotfile and underscore-prefixed dirs are skipped, as
|
||||
* are ClawHub-installed external skills that haven't opted in to gbrain
|
||||
* routing via `triggers:` frontmatter (#1767).
|
||||
*/
|
||||
function deriveManifest(skillsDir: string): ManifestEntry[] {
|
||||
const out: ManifestEntry[] = [];
|
||||
@@ -93,6 +120,12 @@ function deriveManifest(skillsDir: string): ManifestEntry[] {
|
||||
const skillMd = join(subdirAbs, 'SKILL.md');
|
||||
if (!existsSync(skillMd)) continue;
|
||||
|
||||
// ClawHub-installed external skill (#1767): skip unless it opts in
|
||||
// to gbrain routing by declaring `triggers:` in its frontmatter.
|
||||
if (existsSync(join(subdirAbs, '.clawhub', 'origin.json')) && !declaresTriggers(skillMd)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const frontmatterName = parseSkillName(skillMd);
|
||||
const name = frontmatterName && frontmatterName !== '' ? frontmatterName : entry;
|
||||
out.push({ name, path: `${entry}/SKILL.md` });
|
||||
|
||||
@@ -382,6 +382,35 @@ describe("DRY detection — checkResolvable", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("#1767 — ClawHub workspace skills are not resolver-required", () => {
|
||||
let dir: string;
|
||||
afterEachCleanup(() => dir && rmSync(dir, { recursive: true, force: true }));
|
||||
|
||||
test("ClawHub skill without gbrain metadata produces no unreachable/mece_gap", () => {
|
||||
dir = mkdtempSync(join(tmpdir(), "gbrain-clawhub-"));
|
||||
// Native gbrain skill: routable via frontmatter triggers. No manifest.json
|
||||
// (the OpenClaw derive path from the issue repro).
|
||||
mkdirSync(join(dir, "query"), { recursive: true });
|
||||
writeFileSync(
|
||||
join(dir, "query", "SKILL.md"),
|
||||
`---\nname: query\ndescription: test\ntriggers:\n - "what do we know"\n---\n\n# query\n`
|
||||
);
|
||||
// ClawHub-installed integration: no triggers, no resolver row.
|
||||
mkdirSync(join(dir, "agentmail", ".clawhub"), { recursive: true });
|
||||
writeFileSync(
|
||||
join(dir, "agentmail", ".clawhub", "origin.json"),
|
||||
JSON.stringify({ registry: "https://clawhub.ai", slug: "agentmail" })
|
||||
);
|
||||
writeFileSync(join(dir, "agentmail", "SKILL.md"), `---\nname: agentmail\ndescription: email integration\n---\n\n# agentmail\n`);
|
||||
|
||||
const report = checkResolvable(dir);
|
||||
const agentmailIssues = report.issues.filter(i => i.skill === "agentmail");
|
||||
expect(agentmailIssues).toEqual([]);
|
||||
expect(report.ok).toBe(true);
|
||||
expect(report.summary.total_skills).toBe(1);
|
||||
});
|
||||
});
|
||||
|
||||
describe("v0.22.4 regression — actual repo skills/ has 0 errors", () => {
|
||||
test("repo skills/ pass check-resolvable cleanly (zero errors AND zero warnings)", () => {
|
||||
// The v0.22.4 (Part A) contract was zero warnings AND zero errors.
|
||||
|
||||
@@ -209,24 +209,6 @@ describe('gbrain extract --stale', () => {
|
||||
expect(usRows[0]?.eq).toBe(true);
|
||||
});
|
||||
|
||||
test('version-arm floor: page edited BEFORE LINK_EXTRACTOR_VERSION_TS clears after --stale (issue #2576 bug 3)', async () => {
|
||||
// A page whose updated_at predates the version watermark used to be
|
||||
// stamped at its updated_at (< versionTs), so the version arm re-fired
|
||||
// every run — permanently stale. The sweep now floors the stamp at
|
||||
// versionTs. (The #1768 test above also covers this since the v0.42.x
|
||||
// VERSION_TS bump moved its date below the watermark, but this pins the
|
||||
// behavior explicitly so a date "repair" there can't drop coverage.)
|
||||
await engine.putPage('people/alice', personPage('Alice'));
|
||||
await engine.executeRaw(`UPDATE pages SET updated_at = '2000-01-01T00:00:00Z' WHERE slug = 'people/alice'`);
|
||||
expect(await engine.countStalePagesForExtraction({ versionTs: LINK_EXTRACTOR_VERSION_TS })).toBe(1);
|
||||
|
||||
await runExtract(engine, ['--stale']);
|
||||
// Pre-floor this stayed 1 forever (stamp < versionTs → version arm re-fires).
|
||||
expect(await engine.countStalePagesForExtraction({ versionTs: LINK_EXTRACTOR_VERSION_TS })).toBe(0);
|
||||
await runExtract(engine, ['--stale']);
|
||||
expect(await engine.countStalePagesForExtraction({ versionTs: LINK_EXTRACTOR_VERSION_TS })).toBe(0);
|
||||
});
|
||||
|
||||
test('CDX-4 (D2): a link-flush throw aborts the sweep and leaves pages UNSTAMPED', async () => {
|
||||
await engine.putPage('people/alice', personPage('Alice'));
|
||||
await engine.putPage('companies/acme', companyPage('Acme', '[Alice](people/alice) founded [Acme](companies/acme).'));
|
||||
|
||||
@@ -140,15 +140,6 @@ describe('extractEntityRefs', () => {
|
||||
expect(wikiRefs[0].needsResolution).toBe(true);
|
||||
});
|
||||
|
||||
test('recognizes ops/ qualified wikilinks (issue #2576 bug 2)', () => {
|
||||
// `ops` was missing from DIR_PATTERN, so [[ops/...]] fell through to
|
||||
// the generic 2c pass (needsResolution) instead of being a real ref.
|
||||
const refs = extractEntityRefs('Deployed via [[ops/services/pointer-agent]].');
|
||||
expect(refs.length).toBe(1);
|
||||
expect(refs[0].slug).toBe('ops/services/pointer-agent');
|
||||
expect(refs[0].needsResolution).toBeUndefined();
|
||||
});
|
||||
|
||||
test('skips qualified-syntax tokens (those belong to 2a)', () => {
|
||||
// [[wiki:topics/ai]] looks like 2a's qualified shape — even though
|
||||
// it wouldn't satisfy DIR_PATTERN, 2c must not claim it either
|
||||
@@ -1078,15 +1069,6 @@ describe('makeResolver — fallback chain', () => {
|
||||
]);
|
||||
});
|
||||
|
||||
test('resolveBasenameMatches: path-style ref falls back to the tail (issue #2576 bug 2)', async () => {
|
||||
// normalizeBasename strips slashes, so `runbooks/2026-05-01-pointer-agent`
|
||||
// used to normalize to a garbage key that never hit the tail-keyed index.
|
||||
const engine = makeFakeEngineWithSlugs(['ops/changes/2026-05-01-pointer-agent']);
|
||||
const r = makeResolver(engine);
|
||||
expect(await r.resolveBasenameMatches!('runbooks/2026-05-01-pointer-agent'))
|
||||
.toEqual(['ops/changes/2026-05-01-pointer-agent']);
|
||||
});
|
||||
|
||||
test('resolveBasenameMatches: case-insensitive fallback', async () => {
|
||||
const engine = makeFakeEngineWithSlugs(['companies/fast-weigh']);
|
||||
const r = makeResolver(engine);
|
||||
|
||||
@@ -166,6 +166,55 @@ describe('loadOrDeriveManifest', () => {
|
||||
expect(r.skills.map(s => s.name)).toEqual(['apple', 'mango', 'zebra']);
|
||||
});
|
||||
|
||||
// #1767 — ClawHub-installed workspace skills are external integrations,
|
||||
// not gbrain-routable skills. The derive path skips them unless they
|
||||
// opt in via `triggers:` frontmatter.
|
||||
it('skips ClawHub-origin skills without triggers frontmatter (#1767)', () => {
|
||||
const dir = scratch();
|
||||
writeSkill(dir, 'query', 'query');
|
||||
writeSkill(dir, 'agentmail', 'agentmail');
|
||||
mkdirSync(join(dir, 'agentmail', '.clawhub'), { recursive: true });
|
||||
writeFileSync(
|
||||
join(dir, 'agentmail', '.clawhub', 'origin.json'),
|
||||
JSON.stringify({ registry: 'https://clawhub.ai', slug: 'agentmail' })
|
||||
);
|
||||
const r = loadOrDeriveManifest(dir);
|
||||
expect(r.derived).toBe(true);
|
||||
expect(r.skills.map(s => s.name)).toEqual(['query']);
|
||||
});
|
||||
|
||||
it('includes ClawHub-origin skills that opt in via triggers frontmatter (#1767)', () => {
|
||||
const dir = scratch();
|
||||
writeSkill(dir, 'agentmail', 'agentmail');
|
||||
mkdirSync(join(dir, 'agentmail', '.clawhub'), { recursive: true });
|
||||
writeFileSync(
|
||||
join(dir, 'agentmail', '.clawhub', 'origin.json'),
|
||||
JSON.stringify({ registry: 'https://clawhub.ai', slug: 'agentmail' })
|
||||
);
|
||||
writeFileSync(
|
||||
join(dir, 'agentmail', 'SKILL.md'),
|
||||
`---\nname: agentmail\ndescription: test\ntriggers:\n - "send email"\n---\n\n# agentmail\n`
|
||||
);
|
||||
const r = loadOrDeriveManifest(dir);
|
||||
expect(r.derived).toBe(true);
|
||||
expect(r.skills.map(s => s.name)).toEqual(['agentmail']);
|
||||
});
|
||||
|
||||
it('keeps ClawHub-origin skills listed in an explicit manifest.json (#1767)', () => {
|
||||
// Explicit manifest.json is a deliberate declaration — strict checking stays.
|
||||
const dir = scratch();
|
||||
writeSkill(dir, 'agentmail', 'agentmail');
|
||||
mkdirSync(join(dir, 'agentmail', '.clawhub'), { recursive: true });
|
||||
writeFileSync(
|
||||
join(dir, 'agentmail', '.clawhub', 'origin.json'),
|
||||
JSON.stringify({ registry: 'https://clawhub.ai', slug: 'agentmail' })
|
||||
);
|
||||
writeManifest(dir, { skills: [{ name: 'agentmail', path: 'agentmail/SKILL.md' }] });
|
||||
const r = loadOrDeriveManifest(dir);
|
||||
expect(r.derived).toBe(false);
|
||||
expect(r.skills.map(s => s.name)).toEqual(['agentmail']);
|
||||
});
|
||||
|
||||
it('treats dirs without SKILL.md as not-a-skill', () => {
|
||||
const dir = scratch();
|
||||
writeSkill(dir, 'query', 'query');
|
||||
|
||||
Reference in New Issue
Block a user