mirror of
https://github.com/garrytan/gbrain.git
synced 2026-08-17 10:22:34 +00:00
Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9391fb9317 |
@@ -233,14 +233,13 @@ keep it or `git checkout` to throw it away. Nothing is committed for you.
|
||||
|
||||
**For a skill that ships with gbrain** (anything under the gbrain repo's own
|
||||
`skills/`): SkillOpt refuses to overwrite it by default and writes the winner to
|
||||
`skills/<name>/skillopt/proposed.md` instead (while keeping `best.md` as the
|
||||
optimizer's current-best pointer), so an optimization pass can never silently
|
||||
mutate a skill other people depend on. Two ways to handle that:
|
||||
`skills/<name>/skillopt/best.md` instead, so an optimization pass can never
|
||||
silently mutate a skill other people depend on. Two ways to handle that:
|
||||
|
||||
```bash
|
||||
# See the proposed improvement without touching SKILL.md (works for ANY skill):
|
||||
gbrain skillopt meeting-prep --split 1:1:1 --no-mutate
|
||||
# → writes skills/meeting-prep/skillopt/proposed.md, updates best.md, and prints the proposal path.
|
||||
# → writes skills/meeting-prep/skillopt/best.md (the proposed rewrite), prints its path. Copy what you want.
|
||||
|
||||
# Actually rewrite a bundled skill (explicit opt-in + an independent held-out set):
|
||||
gbrain skillopt brain-ops --split 1:1:1 --allow-mutate-bundled \
|
||||
|
||||
@@ -142,12 +142,16 @@ export async function runOnboard(engine: BrainEngine, args: string[]): Promise<v
|
||||
|
||||
// --auto path: runs through the T2 library orchestrator. Hooks emit CLI
|
||||
// progress to stderr; the final result lands as JSON on stdout (or human
|
||||
// summary).
|
||||
// summary). extraRemediations (gathered above from runAllOnboardChecks)
|
||||
// is threaded into the runner so the onboard-check remediations
|
||||
// (extract-ner, extract-timeline-from-meetings, etc.) reach the planner
|
||||
// — the same wiring the --check path uses above.
|
||||
const result = await runRemediation(
|
||||
engine,
|
||||
{
|
||||
targetScore,
|
||||
maxUsd,
|
||||
extraRemediations,
|
||||
// --auto --yes opts into the prompt_required tier too; library
|
||||
// doesn't distinguish auto_apply vs prompt_required, it just runs
|
||||
// every remediation in the plan. The plan-building side (T12 render)
|
||||
|
||||
@@ -4957,7 +4957,7 @@ const run_onboard: Operation = {
|
||||
// typo, the underlying queue.add would reject. Defense-in-depth.
|
||||
const result = await runRemediation(
|
||||
ctx.engine,
|
||||
{ targetScore, maxUsd },
|
||||
{ targetScore, maxUsd, extraRemediations: allowedExtras },
|
||||
{},
|
||||
);
|
||||
|
||||
|
||||
@@ -66,9 +66,10 @@ export async function runRemediation(
|
||||
} = await import('../remediation-checkpoint.ts');
|
||||
|
||||
const ctx = await loadRecommendationContext(engine);
|
||||
const extraRemediations = opts.extraRemediations ?? [];
|
||||
|
||||
// Pre-flight ceiling check via the shared plan computation.
|
||||
const initialPlan = await computeRemediationPlan(engine, { targetScore });
|
||||
const initialPlan = await computeRemediationPlan(engine, { targetScore, extraRemediations });
|
||||
if (initialPlan.target_unreachable) {
|
||||
hooks.onTargetUnreachable?.(targetScore, initialPlan.max_reachable_score);
|
||||
return {
|
||||
@@ -87,7 +88,7 @@ export async function runRemediation(
|
||||
}
|
||||
|
||||
const initialHealth = await engine.getHealth();
|
||||
let recs: RemediationStep[] = computeRecommendations(initialHealth, ctx)
|
||||
let recs: RemediationStep[] = computeRecommendations(initialHealth, ctx, extraRemediations)
|
||||
.filter((r) => r.status === 'remediable');
|
||||
if (recs.length === 0) {
|
||||
hooks.onNothingToDo?.(initialHealth.brain_score, targetScore);
|
||||
@@ -305,7 +306,13 @@ export async function runRemediation(
|
||||
// steps with bumped retry suffix (D1).
|
||||
if (recs.length === 0 || stepCount >= maxJobs) break;
|
||||
const freshHealth = await engine.getHealth();
|
||||
recs = computeRecommendations(freshHealth, ctx).filter((r) => r.status === 'remediable');
|
||||
// Extras carry a static status:'remediable' — a fresh health snapshot
|
||||
// never ages them out the way health-derived steps drop. Filter out
|
||||
// ids this run already processed (any terminal status), or the recheck
|
||||
// would resubmit completed extras every iteration, forever.
|
||||
const processedIds = new Set(submitted.map((s) => s.id));
|
||||
const pendingExtras = extraRemediations.filter((r) => !processedIds.has(r.id));
|
||||
recs = computeRecommendations(freshHealth, ctx, pendingExtras).filter((r) => r.status === 'remediable');
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
@@ -63,6 +63,16 @@ export interface RemediationOpts {
|
||||
resumePlanHash?: string;
|
||||
/** Whether to attempt resume at all (default false). */
|
||||
resume?: boolean;
|
||||
/**
|
||||
* Caller-supplied RemediationStep entries threaded into the planner.
|
||||
* Mirrors RemediationPlanOpts.extraRemediations so onboard's --apply
|
||||
* --auto path (and MCP run_onboard auto modes) forward the same
|
||||
* onboard-check remediations the --check path already passes through
|
||||
* computeRemediationPlan. Without this the runner saw only generic
|
||||
* brain_score remediations and reported "Nothing to do" whenever the
|
||||
* only applicable work was an extra (e.g. extract-ner).
|
||||
*/
|
||||
extraRemediations?: RemediationStep[];
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -93,13 +93,7 @@ import { resolveLrSchedule } from './lr-schedule.ts';
|
||||
import { preflight, formatPreflightReport } from './preflight.ts';
|
||||
import { isRejected, loadRejectedBuffer, makeRejectedEntry, saveRejectedBuffer } from './rejected-buffer.ts';
|
||||
import { runReflect, runOneShotRewrite, describeJudges } from './reflect.ts';
|
||||
import {
|
||||
acceptCandidate,
|
||||
proposedPath as proposedFilePath,
|
||||
revertAllPending,
|
||||
skillPath,
|
||||
writeProposed,
|
||||
} from './version-store.ts';
|
||||
import { acceptCandidate, bestPath, revertAllPending, skillPath, writeProposed } from './version-store.ts';
|
||||
import { runValidationGate, scoreSkillOnTasks } from './validate-gate.ts';
|
||||
import { ROLLOUT_SUCCESS_THRESHOLD } from './types.ts';
|
||||
import type { SkillOptOpts, EditOp, RunReceipt, BenchmarkTask } from './types.ts';
|
||||
@@ -708,9 +702,9 @@ async function runOptimizationLoop(
|
||||
// to the catch's assignment values only (it can't prove the async callback ran).
|
||||
const finalOutcome = outcome as 'accepted' | 'no_improvement' | 'aborted' | 'errored';
|
||||
if (!mutateDecision.mutate && finalOutcome === 'accepted') {
|
||||
// writeProposed() emitted both the best pointer and the stable review
|
||||
// artifact in the accept branch. SKILL.md remains untouched.
|
||||
proposedPath = proposedFilePath(skillsDir, skillName);
|
||||
// best.md was written by writeProposed() in the accept branch (no-mutate
|
||||
// path); it doubles as proposed.md for human review. SKILL.md untouched.
|
||||
proposedPath = bestPath(skillsDir, skillName);
|
||||
} else if (mutateDecision.mutate) {
|
||||
mutatedSkillFile = finalOutcome === 'accepted';
|
||||
}
|
||||
|
||||
@@ -23,7 +23,6 @@
|
||||
*
|
||||
* history.json
|
||||
* best.md
|
||||
* proposed.md
|
||||
* versions/
|
||||
* v0001_e1_s1.md
|
||||
* v0002_e1_s2.md
|
||||
@@ -53,10 +52,6 @@ export function bestPath(skillsDir: string, skillName: string): string {
|
||||
return path.join(skilloptDir(skillsDir, skillName), 'best.md');
|
||||
}
|
||||
|
||||
export function proposedPath(skillsDir: string, skillName: string): string {
|
||||
return path.join(skilloptDir(skillsDir, skillName), 'proposed.md');
|
||||
}
|
||||
|
||||
export function skillPath(skillsDir: string, skillName: string): string {
|
||||
return path.join(skillsDir, skillName, 'SKILL.md');
|
||||
}
|
||||
@@ -176,18 +171,17 @@ export function acceptCandidate(input: AcceptInput): AcceptResult {
|
||||
}
|
||||
|
||||
/**
|
||||
* Write the candidate to both `best.md` and `proposed.md` WITHOUT touching
|
||||
* SKILL.md or the history ledger. `best.md` remains the optimizer's current
|
||||
* best pointer; `proposed.md` is the stable human-review artifact promised by
|
||||
* `--no-mutate`. Returns the proposal path. Each write is atomic (.tmp + rename).
|
||||
* Write the candidate to `best.md` (which doubles as `proposed.md`) WITHOUT
|
||||
* touching SKILL.md or the history ledger. Used by the `--no-mutate` /
|
||||
* bundled-without-allow paths: the optimizer found a better candidate but the
|
||||
* caller opted out of in-place mutation, so we surface it for human review.
|
||||
* Returns the path written. Atomic (.tmp + rename).
|
||||
*/
|
||||
export function writeProposed(skillsDir: string, skillName: string, candidateText: string): string {
|
||||
const best = bestPath(skillsDir, skillName);
|
||||
const proposed = proposedPath(skillsDir, skillName);
|
||||
fs.mkdirSync(path.dirname(best), { recursive: true });
|
||||
atomicWrite(best, candidateText);
|
||||
atomicWrite(proposed, candidateText);
|
||||
return proposed;
|
||||
const p = bestPath(skillsDir, skillName);
|
||||
fs.mkdirSync(path.dirname(p), { recursive: true });
|
||||
atomicWrite(p, candidateText);
|
||||
return p;
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -39,7 +39,6 @@ import { runSkillOpt } from '../../src/core/skillopt/orchestrator.ts';
|
||||
import {
|
||||
bestPath,
|
||||
loadHistory,
|
||||
proposedPath,
|
||||
skillPath,
|
||||
} from '../../src/core/skillopt/version-store.ts';
|
||||
import { loadRejectedBuffer } from '../../src/core/skillopt/rejected-buffer.ts';
|
||||
@@ -742,7 +741,7 @@ describe('skillopt T3 — F11 held-out gate, ablation opts, no-DB-pollution', ()
|
||||
} finally { fixture.cleanup(); }
|
||||
});
|
||||
|
||||
test('--no-mutate writes proposed.md and best.md, leaves SKILL.md untouched', async () => {
|
||||
test('--no-mutate writes proposed.md (best.md), leaves SKILL.md untouched', async () => {
|
||||
const fixture = setupFixture(SKILL_PEOPLE_ONLY, CITATIONS_BENCHMARK);
|
||||
try {
|
||||
installStub({
|
||||
@@ -754,9 +753,10 @@ describe('skillopt T3 — F11 held-out gate, ablation opts, no-DB-pollution', ()
|
||||
const result = await runOnce(fixture, { noMutate: true });
|
||||
expect(result.outcome).toBe('accepted');
|
||||
expect(result.mutatedSkillFile).toBe(false);
|
||||
expect(result.proposedPath).toBe(proposedPath(fixture.skillsDir, SKILL));
|
||||
expect(result.proposedPath).toBeDefined();
|
||||
// proposed.md (best.md) exists and carries the improvement.
|
||||
expect(fs.existsSync(result.proposedPath!)).toBe(true);
|
||||
expect(fs.readFileSync(result.proposedPath!, 'utf8')).toContain('## Citations');
|
||||
expect(fs.readFileSync(bestPath(fixture.skillsDir, SKILL), 'utf8')).toContain('## Citations');
|
||||
// SKILL.md on disk is UNCHANGED (still People-only).
|
||||
const skill = fs.readFileSync(skillPath(fixture.skillsDir, SKILL), 'utf8');
|
||||
expect(skill).not.toContain('## Citations');
|
||||
|
||||
@@ -0,0 +1,108 @@
|
||||
// test/remediation-run-extras.serial.test.ts
|
||||
// Regression for PR #2161 takeover: `gbrain onboard --apply --auto` dropped
|
||||
// onboard-check extraRemediations. Two distinct halves of the bug:
|
||||
// 1. runRemediation built the pre-flight plan + initial recs WITHOUT the
|
||||
// extras, so an extras-only plan reported "Nothing to do".
|
||||
// 2. The D7 mid-run recheck rebuilt recs WITHOUT the extras after every
|
||||
// completed step, so with 2+ plannable steps all remaining extras were
|
||||
// dropped after step 1. The recheck must also filter out extras this
|
||||
// run already processed — extras carry static status:'remediable', so
|
||||
// unfiltered threading would resubmit completed extras forever.
|
||||
//
|
||||
// SERIAL: mock.module (queue + wait-for-completion stubs, R2) + GBRAIN_HOME
|
||||
// env mutation so checkpoint files land in a tmpdir, not ~/.gbrain.
|
||||
|
||||
import { describe, expect, test, beforeAll, afterAll, mock } from 'bun:test';
|
||||
import { mkdtempSync, rmSync } from 'node:fs';
|
||||
import { tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { PGLiteEngine } from '../src/core/pglite-engine.ts';
|
||||
import { makeRemediationStep } from '../src/core/remediation-step.ts';
|
||||
|
||||
// Stub the Minion queue: every submitted job is immediately 'completed'.
|
||||
// runRemediation only calls queue.add + waitForCompletion(queue, id).
|
||||
let nextJobId = 1;
|
||||
const submittedJobs: Array<{ name: string }> = [];
|
||||
mock.module('../src/core/minions/queue.ts', () => ({
|
||||
MinionQueue: class {
|
||||
async add(name: string) {
|
||||
submittedJobs.push({ name });
|
||||
return { id: nextJobId++, status: 'completed' };
|
||||
}
|
||||
},
|
||||
}));
|
||||
mock.module('../src/core/minions/wait-for-completion.ts', () => ({
|
||||
waitForCompletion: async () => ({ status: 'completed' }),
|
||||
}));
|
||||
|
||||
let engine: PGLiteEngine;
|
||||
let home: string;
|
||||
const prevHome = process.env.GBRAIN_HOME;
|
||||
|
||||
beforeAll(async () => {
|
||||
home = mkdtempSync(join(tmpdir(), 'gbrain-remextras-'));
|
||||
process.env.GBRAIN_HOME = home;
|
||||
engine = new PGLiteEngine();
|
||||
await engine.connect({});
|
||||
await engine.initSchema();
|
||||
}, 120_000);
|
||||
|
||||
afterAll(async () => {
|
||||
await engine.disconnect();
|
||||
if (prevHome === undefined) delete process.env.GBRAIN_HOME;
|
||||
else process.env.GBRAIN_HOME = prevHome;
|
||||
rmSync(home, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
function extra(id: string, job: string) {
|
||||
return makeRemediationStep({
|
||||
id,
|
||||
job,
|
||||
params: {},
|
||||
severity: 'medium',
|
||||
est_seconds: 5,
|
||||
est_usd_cost: 0,
|
||||
rationale: 'synthetic onboard-check extra',
|
||||
status: 'remediable',
|
||||
});
|
||||
}
|
||||
|
||||
describe('runRemediation extraRemediations threading', () => {
|
||||
test('extras-only plan runs BOTH extras and terminates (no Nothing-to-do, no resubmit loop)', async () => {
|
||||
// Empty PGLite brain → zero health-derived recommendations. Without the
|
||||
// fix, half 1 makes this run return submitted: [] via onNothingToDo.
|
||||
// With only half 1 (the original PR #2161 diff), the mid-run recheck
|
||||
// drops the second extra after step 1 — submitted has 1 entry, not 2.
|
||||
const { runRemediation } = await import('../src/core/remediation/run.ts');
|
||||
let nothingToDo = false;
|
||||
const result = await runRemediation(
|
||||
engine,
|
||||
{
|
||||
targetScore: 1,
|
||||
extraRemediations: [
|
||||
extra('onboard.extract_ner', 'extract-ner'),
|
||||
extra('onboard.extract_timeline', 'extract-timeline-from-meetings'),
|
||||
],
|
||||
// Safety bound: an unfiltered recheck would resubmit completed
|
||||
// extras forever; maxJobs turns that regression into a fast fail
|
||||
// (extra count > 1 below) instead of a hung test.
|
||||
maxJobs: 5,
|
||||
},
|
||||
{ onNothingToDo: () => { nothingToDo = true; } },
|
||||
);
|
||||
|
||||
expect(nothingToDo).toBe(false);
|
||||
const ids = result.submitted.map((s) => s.id);
|
||||
expect(ids).toContain('onboard.extract_ner');
|
||||
expect(ids).toContain('onboard.extract_timeline');
|
||||
// Each extra ran exactly once — the recheck must not re-plan extras the
|
||||
// run already processed.
|
||||
expect(ids.filter((i) => i === 'onboard.extract_ner').length).toBe(1);
|
||||
expect(ids.filter((i) => i === 'onboard.extract_timeline').length).toBe(1);
|
||||
expect(result.submitted.every((s) => s.status === 'completed')).toBe(true);
|
||||
expect(submittedJobs.map((j) => j.name).sort()).toEqual([
|
||||
'extract-ner',
|
||||
'extract-timeline-from-meetings',
|
||||
]);
|
||||
});
|
||||
});
|
||||
@@ -12,11 +12,9 @@ import {
|
||||
bestPath,
|
||||
historyPath,
|
||||
loadHistory,
|
||||
proposedPath,
|
||||
revertAllPending,
|
||||
skillPath,
|
||||
versionsDir,
|
||||
writeProposed,
|
||||
} from '../../src/core/skillopt/version-store.ts';
|
||||
|
||||
let tmpDir: string;
|
||||
@@ -81,19 +79,6 @@ describe('acceptCandidate (D8 two-phase commit)', () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe('writeProposed', () => {
|
||||
test('writes distinct best and proposed artifacts without mutating SKILL.md (#2635)', () => {
|
||||
const candidate = '---\nname: test\n---\nproposed body\n';
|
||||
|
||||
const written = writeProposed(tmpDir, SKILL, candidate);
|
||||
|
||||
expect(written).toBe(proposedPath(tmpDir, SKILL));
|
||||
expect(fs.readFileSync(bestPath(tmpDir, SKILL), 'utf8')).toBe(candidate);
|
||||
expect(fs.readFileSync(proposedPath(tmpDir, SKILL), 'utf8')).toBe(candidate);
|
||||
expect(fs.readFileSync(skillPath(tmpDir, SKILL), 'utf8')).toContain('baseline body');
|
||||
});
|
||||
});
|
||||
|
||||
describe('revertAllPending (D8 crash recovery)', () => {
|
||||
test('no-op when no pending rows', () => {
|
||||
const reverted = revertAllPending(tmpDir, SKILL);
|
||||
|
||||
Reference in New Issue
Block a user