mirror of
https://github.com/garrytan/gbrain.git
synced 2026-08-16 01:42:23 +00:00
Compare commits
3
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
92656a221b | ||
|
|
6ec762cbcd | ||
|
|
4a81c017a0 |
Vendored
+56
File diff suppressed because one or more lines are too long
Vendored
-56
File diff suppressed because one or more lines are too long
Vendored
+1
-1
@@ -7,7 +7,7 @@
|
||||
<link rel="preconnect" href="https://fonts.googleapis.com" />
|
||||
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin />
|
||||
<link href="https://fonts.googleapis.com/css2?family=Inter:wght@400;500;600&family=JetBrains+Mono:wght@400;500&display=swap" rel="stylesheet" />
|
||||
<script type="module" crossorigin src="/admin/assets/index-CoGEje3-.js"></script>
|
||||
<script type="module" crossorigin src="/admin/assets/index-BpDk4NI4.js"></script>
|
||||
<link rel="stylesheet" crossorigin href="/admin/assets/index-GxkWX7v3.css">
|
||||
</head>
|
||||
<body>
|
||||
|
||||
+6
-2
@@ -5,13 +5,14 @@ import { AgentsPage } from './pages/Agents';
|
||||
import { RequestLogPage } from './pages/RequestLog';
|
||||
import { CalibrationPage } from './pages/Calibration';
|
||||
import { JobsWatchPage } from './pages/JobsWatch';
|
||||
import { SourcesPage } from './pages/Sources';
|
||||
import { api } from './api';
|
||||
|
||||
type Page = 'login' | 'dashboard' | 'agents' | 'log' | 'calibration' | 'jobs';
|
||||
type Page = 'login' | 'dashboard' | 'agents' | 'sources' | 'log' | 'calibration' | 'jobs';
|
||||
|
||||
function getPage(): Page {
|
||||
const hash = window.location.hash.replace('#', '') || 'dashboard';
|
||||
if (['login', 'dashboard', 'agents', 'log', 'calibration', 'jobs'].includes(hash)) return hash as Page;
|
||||
if (['login', 'dashboard', 'agents', 'sources', 'log', 'calibration', 'jobs'].includes(hash)) return hash as Page;
|
||||
return 'dashboard';
|
||||
}
|
||||
|
||||
@@ -54,6 +55,8 @@ export function App() {
|
||||
onClick={() => navigate('dashboard')}>Dashboard</a>
|
||||
<a className={`nav-item ${page === 'agents' ? 'active' : ''}`}
|
||||
onClick={() => navigate('agents')}>Agents</a>
|
||||
<a className={`nav-item ${page === 'sources' ? 'active' : ''}`}
|
||||
onClick={() => navigate('sources')}>Sources</a>
|
||||
<a className={`nav-item ${page === 'log' ? 'active' : ''}`}
|
||||
onClick={() => navigate('log')}>Request Log</a>
|
||||
<a className={`nav-item ${page === 'calibration' ? 'active' : ''}`}
|
||||
@@ -83,6 +86,7 @@ export function App() {
|
||||
<main className="main">
|
||||
{page === 'dashboard' && <DashboardPage />}
|
||||
{page === 'agents' && <AgentsPage />}
|
||||
{page === 'sources' && <SourcesPage />}
|
||||
{page === 'log' && <RequestLogPage />}
|
||||
{page === 'calibration' && <CalibrationPage />}
|
||||
{page === 'jobs' && <JobsWatchPage />}
|
||||
|
||||
@@ -52,4 +52,22 @@ export const api = {
|
||||
apiFetchText(`/admin/api/calibration/charts/${encodeURIComponent(type)}${holder ? `?holder=${encodeURIComponent(holder)}` : ''}`),
|
||||
// v0.41 D2 — live minion-jobs dashboard snapshot.
|
||||
jobsWatch: () => apiFetch('/admin/api/jobs/watch'),
|
||||
// v0.41.29 Sources tab + federated-read management
|
||||
sources: () => apiFetch('/admin/api/sources'),
|
||||
agentsFederatedRead: () => apiFetch('/admin/api/agents/federated-read'),
|
||||
grantRead: (clientId: string, sourceId: string) =>
|
||||
apiFetch(`/admin/api/agents/${encodeURIComponent(clientId)}/grant-read`, {
|
||||
method: 'POST',
|
||||
body: JSON.stringify({ source_id: sourceId }),
|
||||
}),
|
||||
revokeRead: (clientId: string, sourceId: string) =>
|
||||
apiFetch(`/admin/api/agents/${encodeURIComponent(clientId)}/revoke-read`, {
|
||||
method: 'POST',
|
||||
body: JSON.stringify({ source_id: sourceId }),
|
||||
}),
|
||||
setFederatedRead: (clientId: string, sourceIds: string[]) =>
|
||||
apiFetch(`/admin/api/agents/${encodeURIComponent(clientId)}/set-federated-read`, {
|
||||
method: 'POST',
|
||||
body: JSON.stringify({ source_ids: sourceIds }),
|
||||
}),
|
||||
};
|
||||
|
||||
@@ -381,8 +381,16 @@ function CredentialsModal({ credentials, onClose }: {
|
||||
);
|
||||
}
|
||||
|
||||
interface FederationState {
|
||||
source_id: string | null;
|
||||
federated_read: string[];
|
||||
}
|
||||
|
||||
function AgentDrawer({ agent, onClose, onRevoked }: { agent: Agent; onClose: () => void; onRevoked: () => void }) {
|
||||
const [tab, setTab] = useState<'claude-code' | 'chatgpt' | 'claude-cowork' | 'perplexity' | 'cursor' | 'json'>('claude-code');
|
||||
const [federation, setFederation] = useState<FederationState | null>(null);
|
||||
const [allSources, setAllSources] = useState<string[]>([]);
|
||||
const [showFederation, setShowFederation] = useState(false);
|
||||
const copy = (text: string) => navigator.clipboard.writeText(text);
|
||||
const serverUrl = window.location.origin;
|
||||
|
||||
@@ -390,6 +398,30 @@ function AgentDrawer({ agent, onClose, onRevoked }: { agent: Agent; onClose: ()
|
||||
const isOAuth = agent.auth_type === 'oauth';
|
||||
const agentName = agent.name || agent.client_name || 'unknown';
|
||||
|
||||
// Lazy-load federation state when the drawer opens for an OAuth client.
|
||||
// The /admin/api/agents endpoint doesn't carry source_id / federated_read,
|
||||
// so we fetch /admin/api/agents/federated-read separately and pair by id.
|
||||
useEffect(() => {
|
||||
if (!isOAuth || !cid) return;
|
||||
let cancelled = false;
|
||||
Promise.all([
|
||||
api.agentsFederatedRead().catch(() => ({ clients: [] })),
|
||||
api.sources().catch(() => ({ sources: [] })),
|
||||
]).then(([feds, srcs]: any) => {
|
||||
if (cancelled) return;
|
||||
const me = (feds.clients || []).find((c: any) => c.client_id === cid);
|
||||
setFederation(me ? { source_id: me.source_id, federated_read: me.federated_read || [] } : null);
|
||||
setAllSources((srcs.sources || []).map((s: any) => s.source_id));
|
||||
});
|
||||
return () => { cancelled = true; };
|
||||
}, [cid, isOAuth]);
|
||||
|
||||
const reloadFederation = async () => {
|
||||
const feds: any = await api.agentsFederatedRead().catch(() => ({ clients: [] }));
|
||||
const me = (feds.clients || []).find((c: any) => c.client_id === cid);
|
||||
setFederation(me ? { source_id: me.source_id, federated_read: me.federated_read || [] } : null);
|
||||
};
|
||||
|
||||
// For API keys, we can't show the actual token (it was shown once at creation).
|
||||
// For OAuth, we show the client_id and tell them to use their secret.
|
||||
|
||||
@@ -553,6 +585,34 @@ function AgentDrawer({ agent, onClose, onRevoked }: { agent: Agent; onClose: ()
|
||||
<span>{agent.token_ttl ? (agent.token_ttl >= 31536000 ? 'No expiry' : agent.token_ttl >= 86400 ? `${Math.floor(agent.token_ttl / 86400)}d` : agent.token_ttl >= 3600 ? `${Math.floor(agent.token_ttl / 3600)}h` : `${agent.token_ttl}s`) : '1h (default)'}</span>
|
||||
</div>
|
||||
|
||||
{isOAuth && federation && (
|
||||
<>
|
||||
<div className="section-title" style={{ display: 'flex', alignItems: 'center', justifyContent: 'space-between' }}>
|
||||
<span>Federation</span>
|
||||
<button
|
||||
className="btn btn-secondary"
|
||||
style={{ padding: '4px 10px', fontSize: 12 }}
|
||||
onClick={() => setShowFederation(true)}
|
||||
>
|
||||
Manage reads
|
||||
</button>
|
||||
</div>
|
||||
<div style={{ display: 'grid', gridTemplateColumns: '120px 1fr', gap: '6px 12px', fontSize: 13 }}>
|
||||
<span style={{ color: 'var(--text-secondary)' }}>Write source</span>
|
||||
<span className="mono">{federation.source_id || '(none)'}</span>
|
||||
<span style={{ color: 'var(--text-secondary)' }}>Federated reads</span>
|
||||
<span style={{ fontSize: 12 }}>
|
||||
{federation.federated_read.length === 0
|
||||
? <span style={{ color: 'var(--text-muted)' }}>(empty — no federated reads)</span>
|
||||
: federation.federated_read.map((s) => (
|
||||
<span key={s} className="badge badge-read" style={{ marginRight: 4, marginBottom: 2 }}>{s}</span>
|
||||
))
|
||||
}
|
||||
</span>
|
||||
</div>
|
||||
</>
|
||||
)}
|
||||
|
||||
{/*
|
||||
Config Export visible for both auth_type=oauth AND auth_type=api_key.
|
||||
Claude Code + Cursor + JSON tabs render real snippets regardless
|
||||
@@ -628,6 +688,142 @@ function AgentDrawer({ agent, onClose, onRevoked }: { agent: Agent; onClose: ()
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{showFederation && federation && (
|
||||
<FederationModal
|
||||
clientId={cid}
|
||||
clientName={agentName}
|
||||
allSources={allSources}
|
||||
currentReads={federation.federated_read}
|
||||
writeSource={federation.source_id}
|
||||
onClose={() => setShowFederation(false)}
|
||||
onSaved={async () => {
|
||||
await reloadFederation();
|
||||
setShowFederation(false);
|
||||
}}
|
||||
/>
|
||||
)}
|
||||
</>
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* FederationModal — admin counterpart of `gbrain auth set-federated-read`.
|
||||
* Source checkbox list; "Save" submits the full new list via the
|
||||
* race-safe atomic SQL path in setFederatedReadCore. Per the CLI's
|
||||
* documented contract, this is wholesale-replace semantics — concurrent
|
||||
* grant/revoke from a CLI operator would be last-writer-wins against
|
||||
* a Save here.
|
||||
*/
|
||||
function FederationModal({
|
||||
clientId, clientName, allSources, currentReads, writeSource, onClose, onSaved,
|
||||
}: {
|
||||
clientId: string;
|
||||
clientName: string;
|
||||
allSources: string[];
|
||||
currentReads: string[];
|
||||
writeSource: string | null;
|
||||
onClose: () => void;
|
||||
onSaved: () => Promise<void> | void;
|
||||
}) {
|
||||
const [selected, setSelected] = useState<Set<string>>(new Set(currentReads));
|
||||
const [saving, setSaving] = useState(false);
|
||||
const [error, setError] = useState<string | null>(null);
|
||||
|
||||
const toggle = (id: string) => {
|
||||
const next = new Set(selected);
|
||||
if (next.has(id)) next.delete(id); else next.add(id);
|
||||
setSelected(next);
|
||||
};
|
||||
|
||||
const handleSave = async () => {
|
||||
setSaving(true);
|
||||
setError(null);
|
||||
try {
|
||||
await api.setFederatedRead(clientId, Array.from(selected));
|
||||
await onSaved();
|
||||
} catch (e: any) {
|
||||
setError(e.message || 'save failed');
|
||||
setSaving(false);
|
||||
}
|
||||
};
|
||||
|
||||
// Union: all known sources + any current reads not in the source list
|
||||
// (e.g. orphan entries from before the source was deleted). The latter
|
||||
// surface as "(missing source)" so operators can revoke them.
|
||||
const allKnown = new Set([...allSources, ...currentReads]);
|
||||
const ordered = Array.from(allKnown).sort();
|
||||
|
||||
return (
|
||||
<div className="modal-overlay" onClick={onClose}>
|
||||
<div className="modal" onClick={(e) => e.stopPropagation()} style={{ maxWidth: 520 }}>
|
||||
<div className="modal-header">
|
||||
<div style={{ fontSize: 16, fontWeight: 600 }}>Manage federated reads</div>
|
||||
<div style={{ fontSize: 13, color: 'var(--text-secondary)', marginTop: 4 }}>
|
||||
<strong>{clientName}</strong> — pick which sources this client can read in addition to its
|
||||
{writeSource ? <> write source <code className="mono">{writeSource}</code></> : <> write source</>}.
|
||||
</div>
|
||||
</div>
|
||||
<div className="modal-body" style={{ maxHeight: '50vh', overflowY: 'auto' }}>
|
||||
{ordered.length === 0 && (
|
||||
<div style={{ color: 'var(--text-muted)', fontSize: 13 }}>
|
||||
No sources registered. Use <code>gbrain sources add <id> --path <dir></code> from the CLI first.
|
||||
</div>
|
||||
)}
|
||||
{ordered.map((id) => {
|
||||
const isOrphan = !allSources.includes(id);
|
||||
const isWriteSource = id === writeSource;
|
||||
return (
|
||||
<label
|
||||
key={id}
|
||||
style={{
|
||||
display: 'flex',
|
||||
alignItems: 'center',
|
||||
gap: 10,
|
||||
padding: '8px 10px',
|
||||
borderBottom: '1px solid var(--border)',
|
||||
cursor: 'pointer',
|
||||
fontSize: 13,
|
||||
}}
|
||||
>
|
||||
<input
|
||||
type="checkbox"
|
||||
checked={selected.has(id)}
|
||||
onChange={() => toggle(id)}
|
||||
style={{ width: 16, height: 16, margin: 0, flexShrink: 0, cursor: 'pointer' }}
|
||||
/>
|
||||
<span className="mono" style={{ flex: 1, minWidth: 0, overflow: 'hidden', textOverflow: 'ellipsis', whiteSpace: 'nowrap' }}>{id}</span>
|
||||
{isWriteSource && <span className="badge badge-write" style={{ fontSize: 10, flexShrink: 0 }}>write source</span>}
|
||||
{isOrphan && <span className="badge badge-danger" style={{ fontSize: 10, flexShrink: 0 }}>missing source</span>}
|
||||
</label>
|
||||
);
|
||||
})}
|
||||
</div>
|
||||
{error && (
|
||||
<div style={{
|
||||
background: 'rgba(239,68,68,0.08)',
|
||||
border: '1px solid rgba(239,68,68,0.3)',
|
||||
color: '#ef4444',
|
||||
padding: '10px 12px',
|
||||
borderRadius: 6,
|
||||
margin: '12px 0',
|
||||
fontSize: 12,
|
||||
}}>
|
||||
{error}
|
||||
</div>
|
||||
)}
|
||||
<div className="modal-footer">
|
||||
<button type="button" className="btn btn-secondary" onClick={onClose} disabled={saving}>Cancel</button>
|
||||
<button
|
||||
type="button"
|
||||
className="btn btn-primary"
|
||||
onClick={handleSave}
|
||||
disabled={saving}
|
||||
>
|
||||
{saving ? 'Saving…' : `Save (${selected.size} source${selected.size === 1 ? '' : 's'})`}
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,195 @@
|
||||
import React, { useState, useEffect } from 'react';
|
||||
import { api } from '../api';
|
||||
|
||||
interface SourceRow {
|
||||
source_id: string;
|
||||
name: string;
|
||||
local_path: string | null;
|
||||
sync_enabled: boolean;
|
||||
last_sync_at: string | null;
|
||||
staleness_hours: number | null;
|
||||
staleness_class: 'fresh' | 'stale' | 'severe' | 'unknown';
|
||||
last_commit: string | null;
|
||||
pages: number;
|
||||
chunks_total: number;
|
||||
chunks_unembedded: number;
|
||||
embedding_coverage_pct: number;
|
||||
}
|
||||
|
||||
interface FederatedClient {
|
||||
client_id: string;
|
||||
client_name: string;
|
||||
source_id: string | null;
|
||||
federated_read: string[];
|
||||
}
|
||||
|
||||
function timeAgo(iso: string | null): string {
|
||||
if (!iso) return 'never';
|
||||
const s = Math.floor((Date.now() - new Date(iso).getTime()) / 1000);
|
||||
if (s < 0) return 'in the future?';
|
||||
if (s < 60) return 'just now';
|
||||
if (s < 3600) return `${Math.floor(s / 60)}m ago`;
|
||||
if (s < 86400) return `${Math.floor(s / 3600)}h ago`;
|
||||
return `${Math.floor(s / 86400)}d ago`;
|
||||
}
|
||||
|
||||
function stalenessColor(cls: string): string {
|
||||
switch (cls) {
|
||||
case 'fresh': return '#4ade80';
|
||||
case 'stale': return '#fbbf24';
|
||||
case 'severe': return '#ef4444';
|
||||
default: return 'var(--text-muted)';
|
||||
}
|
||||
}
|
||||
|
||||
function coverageColor(pct: number): string {
|
||||
if (pct >= 99) return '#4ade80';
|
||||
if (pct >= 90) return '#fbbf24';
|
||||
return '#ef4444';
|
||||
}
|
||||
|
||||
export function SourcesPage() {
|
||||
const [sources, setSources] = useState<SourceRow[]>([]);
|
||||
const [clients, setClients] = useState<FederatedClient[]>([]);
|
||||
const [loading, setLoading] = useState(true);
|
||||
const [error, setError] = useState<string | null>(null);
|
||||
|
||||
const load = async () => {
|
||||
setLoading(true);
|
||||
setError(null);
|
||||
try {
|
||||
const [srcReport, clientsResp] = await Promise.all([
|
||||
api.sources(),
|
||||
api.agentsFederatedRead(),
|
||||
]);
|
||||
setSources(srcReport.sources || []);
|
||||
setClients(clientsResp.clients || []);
|
||||
} catch (e: any) {
|
||||
setError(e.message || 'load failed');
|
||||
} finally {
|
||||
setLoading(false);
|
||||
}
|
||||
};
|
||||
|
||||
useEffect(() => { load(); }, []);
|
||||
|
||||
// Reverse-lookup: for each source, which clients can read it?
|
||||
const readersBySource = (sourceId: string): string[] =>
|
||||
clients.filter((c) => c.federated_read.includes(sourceId)).map((c) => c.client_name);
|
||||
|
||||
// Reverse-lookup: which clients WRITE to this source (source_id == sourceId)?
|
||||
const writersBySource = (sourceId: string): string[] =>
|
||||
clients.filter((c) => c.source_id === sourceId).map((c) => c.client_name);
|
||||
|
||||
return (
|
||||
<div style={{ padding: 24, maxWidth: 1200 }}>
|
||||
<div style={{ display: 'flex', alignItems: 'center', justifyContent: 'space-between', marginBottom: 24 }}>
|
||||
<h1 style={{ fontSize: 24, margin: 0 }}>Sources</h1>
|
||||
<button
|
||||
onClick={load}
|
||||
style={{
|
||||
background: 'transparent',
|
||||
border: '1px solid var(--border)',
|
||||
color: 'var(--text-secondary)',
|
||||
padding: '6px 12px',
|
||||
borderRadius: 6,
|
||||
fontSize: 12,
|
||||
cursor: 'pointer',
|
||||
}}
|
||||
>
|
||||
Refresh
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{loading && <div style={{ color: 'var(--text-muted)' }}>Loading…</div>}
|
||||
{error && (
|
||||
<div style={{
|
||||
background: 'rgba(239,68,68,0.08)',
|
||||
border: '1px solid rgba(239,68,68,0.3)',
|
||||
color: '#ef4444',
|
||||
padding: 12,
|
||||
borderRadius: 6,
|
||||
marginBottom: 16,
|
||||
fontSize: 13,
|
||||
}}>
|
||||
Failed to load sources: {error}
|
||||
</div>
|
||||
)}
|
||||
|
||||
{!loading && !error && sources.length === 0 && (
|
||||
<div style={{ color: 'var(--text-muted)', padding: 16 }}>
|
||||
No active sources with a local_path. Use{' '}
|
||||
<code style={{ background: 'var(--bg-elevated)', padding: '2px 6px', borderRadius: 4 }}>
|
||||
gbrain sources add <id> --path <dir>
|
||||
</code>{' '}
|
||||
to register one.
|
||||
</div>
|
||||
)}
|
||||
|
||||
{!loading && !error && sources.length > 0 && (
|
||||
<div style={{ overflowX: 'auto' }}>
|
||||
<table style={{ width: '100%', borderCollapse: 'collapse', fontSize: 13 }}>
|
||||
<thead>
|
||||
<tr style={{ borderBottom: '1px solid var(--border)', textAlign: 'left', color: 'var(--text-muted)' }}>
|
||||
<th style={{ padding: '10px 12px' }}>ID</th>
|
||||
<th style={{ padding: '10px 12px', textAlign: 'right' }}>Pages</th>
|
||||
<th style={{ padding: '10px 12px', textAlign: 'right' }}>Chunks</th>
|
||||
<th style={{ padding: '10px 12px', textAlign: 'right' }}>Embed%</th>
|
||||
<th style={{ padding: '10px 12px' }}>Last Sync</th>
|
||||
<th style={{ padding: '10px 12px' }}>Writers</th>
|
||||
<th style={{ padding: '10px 12px' }}>Readers (federated)</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{sources.map((s) => {
|
||||
const readers = readersBySource(s.source_id);
|
||||
const writers = writersBySource(s.source_id);
|
||||
return (
|
||||
<tr key={s.source_id} style={{ borderBottom: '1px solid var(--border)' }}>
|
||||
<td style={{ padding: '10px 12px', fontFamily: 'JetBrains Mono, monospace' }}>
|
||||
<div>{s.source_id}</div>
|
||||
{s.name !== s.source_id && (
|
||||
<div style={{ fontSize: 11, color: 'var(--text-muted)', fontFamily: 'inherit' }}>{s.name}</div>
|
||||
)}
|
||||
</td>
|
||||
<td style={{ padding: '10px 12px', textAlign: 'right', fontFamily: 'JetBrains Mono, monospace' }}>{s.pages.toLocaleString()}</td>
|
||||
<td style={{ padding: '10px 12px', textAlign: 'right', fontFamily: 'JetBrains Mono, monospace' }}>{s.chunks_total.toLocaleString()}</td>
|
||||
<td style={{ padding: '10px 12px', textAlign: 'right', color: coverageColor(s.embedding_coverage_pct), fontFamily: 'JetBrains Mono, monospace' }}>
|
||||
{s.embedding_coverage_pct.toFixed(0)}%
|
||||
</td>
|
||||
<td style={{ padding: '10px 12px', color: s.local_path == null ? 'var(--text-muted)' : stalenessColor(s.staleness_class) }}>
|
||||
{s.local_path == null ? 'push-only' : timeAgo(s.last_sync_at)}
|
||||
</td>
|
||||
<td style={{ padding: '10px 12px', fontSize: 12, color: 'var(--text-secondary)' }}>
|
||||
{writers.length === 0 ? <span style={{ color: 'var(--text-muted)' }}>none</span> : writers.join(', ')}
|
||||
</td>
|
||||
<td style={{ padding: '10px 12px', fontSize: 12, color: 'var(--text-secondary)' }}>
|
||||
{readers.length === 0 ? <span style={{ color: 'var(--text-muted)' }}>none</span> : readers.join(', ')}
|
||||
</td>
|
||||
</tr>
|
||||
);
|
||||
})}
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
)}
|
||||
|
||||
<div style={{
|
||||
marginTop: 24,
|
||||
padding: 12,
|
||||
background: 'var(--bg-elevated)',
|
||||
border: '1px solid var(--border)',
|
||||
borderRadius: 6,
|
||||
fontSize: 12,
|
||||
color: 'var(--text-muted)',
|
||||
lineHeight: 1.6,
|
||||
}}>
|
||||
<strong style={{ color: 'var(--text-secondary)' }}>Two scopes per OAuth client:</strong>{' '}
|
||||
<em>Writers</em> = clients with this source as their <code>source_id</code> (write authority).{' '}
|
||||
<em>Readers</em> = clients with this source in their <code>federated_read</code> list (read access via federation).
|
||||
Manage federation per-client from the <a href="#agents" style={{ color: '#60a5fa' }}>Agents</a> tab using the
|
||||
"Manage reads" action.
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -1,13 +1,13 @@
|
||||
// AUTO-GENERATED — do not edit by hand.
|
||||
// Run `bun run scripts/build-admin-embedded.ts` to regenerate.
|
||||
// Source: admin/dist/ at 2026-05-27.
|
||||
// Source: admin/dist/ at 2026-07-22.
|
||||
//
|
||||
// Bun resolves the file: imports to a path that works at runtime even
|
||||
// inside a compiled binary (`bun build --compile`). The manifest maps
|
||||
// the request path the express handler sees to (resolved-path, mime).
|
||||
|
||||
// @ts-ignore — type: 'file' is Bun ESM, not in lib.d.ts
|
||||
import A_0_assets_index_CoGEje3__js from '../admin/dist/assets/index-CoGEje3-.js' with { type: 'file' };
|
||||
import A_0_assets_index_BpDk4NI4_js from '../admin/dist/assets/index-BpDk4NI4.js' with { type: 'file' };
|
||||
// @ts-ignore — type: 'file' is Bun ESM, not in lib.d.ts
|
||||
import A_1_assets_index_GxkWX7v3_css from '../admin/dist/assets/index-GxkWX7v3.css' with { type: 'file' };
|
||||
// @ts-ignore — type: 'file' is Bun ESM, not in lib.d.ts
|
||||
@@ -19,7 +19,7 @@ export interface AdminAsset {
|
||||
}
|
||||
|
||||
export const ADMIN_ASSETS: Record<string, AdminAsset> = {
|
||||
"/admin/assets/index-CoGEje3-.js": { path: A_0_assets_index_CoGEje3__js as unknown as string, mime: "application/javascript; charset=utf-8" },
|
||||
"/admin/assets/index-BpDk4NI4.js": { path: A_0_assets_index_BpDk4NI4_js as unknown as string, mime: "application/javascript; charset=utf-8" },
|
||||
"/admin/assets/index-GxkWX7v3.css": { path: A_1_assets_index_GxkWX7v3_css as unknown as string, mime: "text/css; charset=utf-8" },
|
||||
"/admin/index.html": { path: A_2_index_html as unknown as string, mime: "text/html; charset=utf-8" },
|
||||
};
|
||||
|
||||
+586
-1
@@ -24,6 +24,8 @@ import { loadConfig, toEngineConfig } from '../core/config.ts';
|
||||
import { createEngine } from '../core/engine-factory.ts';
|
||||
import type { BrainEngine } from '../core/engine.ts';
|
||||
import { sqlQueryForEngine, executeRawJsonb, type SqlQuery } from '../core/sql-query.ts';
|
||||
import { pgArray } from '../core/oauth-provider.ts';
|
||||
import { assertValidSourceId } from '../core/source-id.ts';
|
||||
|
||||
function hashToken(token: string): string {
|
||||
return createHash('sha256').update(token).digest('hex');
|
||||
@@ -165,6 +167,100 @@ async function list() {
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* `gbrain auth list-clients [--json]` — read surface for OAuth 2.1 clients.
|
||||
*
|
||||
* The existing `gbrain auth list` shows LEGACY bearer tokens from
|
||||
* `access_tokens`; this is the parallel for v0.26+ OAuth clients. Separate
|
||||
* commands rather than merged output because the two models have different
|
||||
* field sets (legacy: lifecycle dates; OAuth: scopes + source_id +
|
||||
* federated_read).
|
||||
*
|
||||
* Human output is card-style (multi-line per client) instead of a fixed-
|
||||
* width table — federated_read can hold many ids per client and a wide
|
||||
* single-line layout truncates / wraps badly on terminals < 200 cols.
|
||||
* JSON output uses a `schema_version: 1` envelope; additive only.
|
||||
*/
|
||||
async function listClients(args: string[]) {
|
||||
const json = args.includes('--json');
|
||||
const includeDeleted = args.includes('--include-deleted');
|
||||
await withConfiguredSql(async (sql) => {
|
||||
// Codex finding #2 (medium): default-hide soft-deleted clients so admin
|
||||
// soft-deletes are honored by the CLI surface. Opt-in via flag.
|
||||
const rows = includeDeleted
|
||||
? await sql`
|
||||
SELECT client_id, client_name, scope, source_id, federated_read,
|
||||
grant_types, created_at, deleted_at
|
||||
FROM oauth_clients
|
||||
ORDER BY client_name
|
||||
`
|
||||
: await sql`
|
||||
SELECT client_id, client_name, scope, source_id, federated_read,
|
||||
grant_types, created_at, deleted_at
|
||||
FROM oauth_clients
|
||||
WHERE deleted_at IS NULL
|
||||
ORDER BY client_name
|
||||
`;
|
||||
if (json) {
|
||||
const clients = rows.map((r) => ({
|
||||
client_id: String(r.client_id),
|
||||
client_name: String(r.client_name),
|
||||
scope: r.scope == null ? null : String(r.scope),
|
||||
source_id: r.source_id == null ? null : String(r.source_id),
|
||||
federated_read: Array.isArray(r.federated_read)
|
||||
? (r.federated_read as string[]).map(String)
|
||||
: [],
|
||||
grant_types: Array.isArray(r.grant_types)
|
||||
? (r.grant_types as string[]).map(String)
|
||||
: [],
|
||||
created_at:
|
||||
r.created_at instanceof Date
|
||||
? r.created_at.toISOString()
|
||||
: r.created_at == null
|
||||
? null
|
||||
: String(r.created_at),
|
||||
deleted_at:
|
||||
r.deleted_at instanceof Date
|
||||
? r.deleted_at.toISOString()
|
||||
: r.deleted_at == null
|
||||
? null
|
||||
: String(r.deleted_at),
|
||||
}));
|
||||
process.stdout.write(JSON.stringify({ schema_version: 1, clients }, null, 2) + '\n');
|
||||
return;
|
||||
}
|
||||
if (rows.length === 0) {
|
||||
console.log(
|
||||
includeDeleted
|
||||
? 'No OAuth clients found (including deleted). Register one: gbrain auth register-client <name>'
|
||||
: 'No active OAuth clients found. Register one: gbrain auth register-client <name>'
|
||||
+ '\n(Use --include-deleted to also show soft-deleted clients.)',
|
||||
);
|
||||
return;
|
||||
}
|
||||
for (let i = 0; i < rows.length; i++) {
|
||||
const r = rows[i];
|
||||
const fed = Array.isArray(r.federated_read)
|
||||
? (r.federated_read as string[]).map(String)
|
||||
: [];
|
||||
const grants = Array.isArray(r.grant_types)
|
||||
? (r.grant_types as string[]).map(String)
|
||||
: [];
|
||||
const deletedAt = r.deleted_at;
|
||||
const status = deletedAt == null
|
||||
? ''
|
||||
: ` [SOFT-DELETED ${deletedAt instanceof Date ? deletedAt.toISOString() : String(deletedAt)}]`;
|
||||
console.log(`${sanitizeForTerminal(String(r.client_name))}${status}`);
|
||||
console.log(` client_id: ${sanitizeForTerminal(String(r.client_id))}`);
|
||||
console.log(` scope: ${r.scope == null ? '(none)' : sanitizeForTerminal(String(r.scope))}`);
|
||||
console.log(` grant types: ${grants.length ? sanitizeForTerminal(grants.join(', ')) : '(none)'}`);
|
||||
console.log(` write source: ${r.source_id == null ? '(none)' : sanitizeForTerminal(String(r.source_id))}`);
|
||||
console.log(` federated: ${fed.length ? sanitizeForTerminal(fed.join(', ')) : '(empty)'}`);
|
||||
if (i < rows.length - 1) console.log('');
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
async function revoke(name: string) {
|
||||
if (!name) { console.error('Usage: auth revoke <name>'); process.exit(1); }
|
||||
await withConfiguredSql(async (sql) => {
|
||||
@@ -301,6 +397,475 @@ async function test(url: string, token: string) {
|
||||
console.log(`\n🧠 Your brain is live! (${elapsed}s)`);
|
||||
}
|
||||
|
||||
/**
|
||||
* Strip ANSI escapes + C0/C1 control characters from a string before
|
||||
* printing it to the operator's terminal. Defense for the
|
||||
* codex-flagged terminal-control-injection class: a client_name or
|
||||
* source_id registered via DCR with `\x1b[2J` (clear-screen) or
|
||||
* `\x1b]0;TITLE\x07` (OSC title-change) would poison
|
||||
* `gbrain auth list-clients` output otherwise.
|
||||
*
|
||||
* Replaces unsafe bytes with their `\xNN` hex escape so the operator
|
||||
* sees that something weird is in the field, instead of silent
|
||||
* mutilation. Tab and newline are preserved as-is so legitimate
|
||||
* multi-line values render.
|
||||
*/
|
||||
export function sanitizeForTerminal(s: string): string {
|
||||
// ALL C0/C1 controls + DEL get escaped. Codex re-review caught that
|
||||
// preserving `\n` lets a DCR-registered client_name spoof additional
|
||||
// human-output lines in list-clients (a real attack — newline in the
|
||||
// name visually adds a fake row to the operator's terminal). Tab is
|
||||
// also escaped for the same reason — field-separator spoofing.
|
||||
// C0: 0x00-0x1F. DEL: 0x7F. C1: 0x80-0x9F.
|
||||
return s.replace(/[\x00-\x1f\x7f-\x9f]/g, (ch) =>
|
||||
`\\x${ch.charCodeAt(0).toString(16).padStart(2, '0')}`,
|
||||
);
|
||||
}
|
||||
|
||||
export interface ResolvedClient {
|
||||
client_id: string;
|
||||
client_name: string;
|
||||
source_id: string | null;
|
||||
federated_read: string[];
|
||||
deleted_at: Date | string | null;
|
||||
}
|
||||
|
||||
export type FederatedReadOutcome =
|
||||
| { kind: 'noop'; reason: 'already-granted' | 'not-present' | 'same-list'; client: ResolvedClient; current: string[] }
|
||||
| { kind: 'updated'; client: ResolvedClient; before: string[]; after: string[] };
|
||||
|
||||
/**
|
||||
* Resolve an OAuth client by client_id (exact) or client_name (unique).
|
||||
* Errors on no-match and on ambiguous client_name (>1 row). client_id
|
||||
* takes precedence — if a long hash is passed and matches, returns
|
||||
* immediately without ever querying by name.
|
||||
*
|
||||
* Legacy bearer tokens in `access_tokens` are NOT searched. Federated read
|
||||
* scope is an OAuth-client concept (oauth_clients.federated_read column);
|
||||
* legacy bearers have no source scope.
|
||||
*/
|
||||
/**
|
||||
* Resolve an OAuth client. Codex finding #2 (medium): default-hide
|
||||
* soft-deleted clients so admin-soft-deleted rows aren't mutated by the
|
||||
* CLI. The `includeDeleted` opt is reserved for future read-side surfaces;
|
||||
* grant/revoke/set ALWAYS filter active rows only.
|
||||
*/
|
||||
export async function resolveClient(
|
||||
sql: SqlQuery,
|
||||
nameOrId: string,
|
||||
opts: { includeDeleted?: boolean } = {},
|
||||
): Promise<ResolvedClient> {
|
||||
const allowDeleted = opts.includeDeleted === true;
|
||||
const byId = allowDeleted
|
||||
? await sql`
|
||||
SELECT client_id, client_name, source_id, federated_read, deleted_at
|
||||
FROM oauth_clients WHERE client_id = ${nameOrId} LIMIT 1
|
||||
`
|
||||
: await sql`
|
||||
SELECT client_id, client_name, source_id, federated_read, deleted_at
|
||||
FROM oauth_clients WHERE client_id = ${nameOrId} AND deleted_at IS NULL LIMIT 1
|
||||
`;
|
||||
if (byId.length === 1) return normalizeClientRow(byId[0]);
|
||||
const byName = allowDeleted
|
||||
? await sql`
|
||||
SELECT client_id, client_name, source_id, federated_read, deleted_at
|
||||
FROM oauth_clients WHERE client_name = ${nameOrId}
|
||||
`
|
||||
: await sql`
|
||||
SELECT client_id, client_name, source_id, federated_read, deleted_at
|
||||
FROM oauth_clients WHERE client_name = ${nameOrId} AND deleted_at IS NULL
|
||||
`;
|
||||
if (byName.length === 0) {
|
||||
throw new Error(
|
||||
`No active OAuth client found with name or id "${nameOrId}". ` +
|
||||
`Run \`gbrain auth register-client <name>\` to create one, ` +
|
||||
`or \`gbrain auth list-clients\` to see what exists. ` +
|
||||
`(Soft-deleted clients are hidden by default.)`,
|
||||
);
|
||||
}
|
||||
if (byName.length > 1) {
|
||||
const ids = byName.map((r) => ` ${String(r.client_id)}`).join('\n');
|
||||
throw new Error(
|
||||
`Multiple active OAuth clients named "${nameOrId}". Pass the full client_id instead:\n${ids}`,
|
||||
);
|
||||
}
|
||||
return normalizeClientRow(byName[0]);
|
||||
}
|
||||
|
||||
function normalizeClientRow(row: Record<string, unknown>): ResolvedClient {
|
||||
const fed = row.federated_read;
|
||||
return {
|
||||
client_id: String(row.client_id),
|
||||
client_name: String(row.client_name),
|
||||
source_id: row.source_id == null ? null : String(row.source_id),
|
||||
federated_read: Array.isArray(fed) ? (fed as string[]).map(String) : [],
|
||||
deleted_at: row.deleted_at == null
|
||||
? null
|
||||
: (row.deleted_at as Date | string),
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Validate the source_id shape AND DB existence. Codex finding #3 (medium):
|
||||
* a manually-INSERTed source row with weird chars (e.g. comma, quote)
|
||||
* would otherwise land in oauth_clients.federated_read as a never-deletable
|
||||
* malformed entry. Fail at the boundary before the existence query so
|
||||
* malformed input gets the validator's hint, not a "does not exist" hint
|
||||
* pointing at a non-creatable id.
|
||||
*/
|
||||
export async function assertSourceExists(sql: SqlQuery, sourceId: string): Promise<void> {
|
||||
assertValidSourceId(sourceId);
|
||||
const rows = await sql`SELECT id FROM sources WHERE id = ${sourceId} LIMIT 1`;
|
||||
if (rows.length === 0) {
|
||||
throw new Error(
|
||||
`Source "${sourceId}" does not exist. Run \`gbrain sources list\` to see registered sources, ` +
|
||||
`or \`gbrain sources add ${sourceId}\` to create it.`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Atomic append: array_append + NOT-ANY guard so the row-lock fully
|
||||
* serializes concurrent grant/revoke against the same client. Codex
|
||||
* finding #1 (HIGH): the previous read-modify-write shape allowed a
|
||||
* concurrent revoke to be silently UNDONE by a racing grant.
|
||||
*
|
||||
* Returns the post-write federated_read array, or null when no rows
|
||||
* matched (already-granted, soft-deleted, or missing client). Callers
|
||||
* disambiguate via prior resolveClient + includes() check.
|
||||
*
|
||||
* `WHERE deleted_at IS NULL` is part of the atomic guard so a client
|
||||
* soft-deleted between resolveClient and the UPDATE can't be mutated.
|
||||
*/
|
||||
async function appendFederatedReadAtomic(
|
||||
sql: SqlQuery,
|
||||
clientId: string,
|
||||
sourceId: string,
|
||||
): Promise<string[] | null> {
|
||||
const rows = await sql`
|
||||
UPDATE oauth_clients
|
||||
SET federated_read = array_append(federated_read, ${sourceId})
|
||||
WHERE client_id = ${clientId}
|
||||
AND deleted_at IS NULL
|
||||
AND NOT (${sourceId} = ANY(federated_read))
|
||||
RETURNING federated_read
|
||||
`;
|
||||
if (rows.length === 0) return null;
|
||||
const fed = rows[0].federated_read;
|
||||
return Array.isArray(fed) ? (fed as string[]).map(String) : [];
|
||||
}
|
||||
|
||||
/**
|
||||
* Atomic remove: array_remove + ANY guard. Same race-correctness story
|
||||
* as appendFederatedReadAtomic. Returns post-write array or null.
|
||||
*/
|
||||
async function removeFederatedReadAtomic(
|
||||
sql: SqlQuery,
|
||||
clientId: string,
|
||||
sourceId: string,
|
||||
): Promise<string[] | null> {
|
||||
const rows = await sql`
|
||||
UPDATE oauth_clients
|
||||
SET federated_read = array_remove(federated_read, ${sourceId})
|
||||
WHERE client_id = ${clientId}
|
||||
AND deleted_at IS NULL
|
||||
AND ${sourceId} = ANY(federated_read)
|
||||
RETURNING federated_read
|
||||
`;
|
||||
if (rows.length === 0) return null;
|
||||
const fed = rows[0].federated_read;
|
||||
return Array.isArray(fed) ? (fed as string[]).map(String) : [];
|
||||
}
|
||||
|
||||
/**
|
||||
* Wholesale array overwrite for `set-federated-read`. Honors the
|
||||
* deleted_at filter. Last-writer-wins semantics under concurrent
|
||||
* `set` calls is acceptable — the user is asserting "this exact list"
|
||||
* intent; concurrent set+set just means whichever ran second wins.
|
||||
* Concurrent set+grant or set+revoke is also last-writer-wins, which
|
||||
* is the documented contract for `set`.
|
||||
*/
|
||||
async function replaceFederatedReadAtomic(
|
||||
sql: SqlQuery,
|
||||
clientId: string,
|
||||
next: string[],
|
||||
): Promise<string[] | null> {
|
||||
// TEXT[] binding via pgArray() string-literal escaping (see helper
|
||||
// for the security note). Our narrow SqlQuery surface
|
||||
// (src/core/sql-query.ts) doesn't bind JS arrays directly.
|
||||
const literal = pgArray(next);
|
||||
const rows = await sql`
|
||||
UPDATE oauth_clients
|
||||
SET federated_read = ${literal}
|
||||
WHERE client_id = ${clientId}
|
||||
AND deleted_at IS NULL
|
||||
RETURNING federated_read
|
||||
`;
|
||||
if (rows.length === 0) return null;
|
||||
const fed = rows[0].federated_read;
|
||||
return Array.isArray(fed) ? (fed as string[]).map(String) : [];
|
||||
}
|
||||
|
||||
/**
|
||||
* Pure helper: dedupe a comma-separated source-id list while preserving
|
||||
* insertion order. Empty input → empty array. Exported so the CLI parser
|
||||
* and tests share one normalizer.
|
||||
*/
|
||||
export function parseSourceCsv(csv: string): string[] {
|
||||
const requested = csv.split(',').map((s) => s.trim()).filter(Boolean);
|
||||
const seen = new Set<string>();
|
||||
const out: string[] = [];
|
||||
for (const s of requested) {
|
||||
if (!seen.has(s)) {
|
||||
seen.add(s);
|
||||
out.push(s);
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
export interface FederatedReadOpts {
|
||||
/** When true, compute the outcome but skip the persisting UPDATE. */
|
||||
dryRun?: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
* Core: append a source to the client's federated_read.
|
||||
*
|
||||
* Atomicity contract (Codex finding #1, HIGH):
|
||||
* The actual write goes through `appendFederatedReadAtomic` which
|
||||
* serializes at the row-lock so concurrent grant/revoke against the
|
||||
* same client cannot lose updates. The race vector that previously
|
||||
* silently restored revoked access is closed: under two operators
|
||||
* racing `revoke-read sensitive` + `grant-read harmless`, postgres
|
||||
* serializes the two UPDATEs and BOTH ops apply (sensitive removed,
|
||||
* harmless added), instead of one clobbering the other.
|
||||
*
|
||||
* The reported `before` is the snapshot at resolveClient time, which
|
||||
* may be stale relative to a concurrent racer. The `after` reflects
|
||||
* the post-UPDATE state from RETURNING (always fresh).
|
||||
*/
|
||||
export async function grantReadCore(
|
||||
sql: SqlQuery,
|
||||
nameOrId: string,
|
||||
sourceId: string,
|
||||
opts: FederatedReadOpts = {},
|
||||
): Promise<FederatedReadOutcome> {
|
||||
const client = await resolveClient(sql, nameOrId);
|
||||
await assertSourceExists(sql, sourceId);
|
||||
if (client.federated_read.includes(sourceId)) {
|
||||
return { kind: 'noop', reason: 'already-granted', client, current: client.federated_read };
|
||||
}
|
||||
if (opts.dryRun) {
|
||||
// Compute the would-be result without touching the row. Last-known
|
||||
// snapshot is best-effort under concurrent writes.
|
||||
const projected = [...client.federated_read, sourceId];
|
||||
return { kind: 'updated', client, before: client.federated_read, after: projected };
|
||||
}
|
||||
const after = await appendFederatedReadAtomic(sql, client.client_id, sourceId);
|
||||
if (after === null) {
|
||||
// Two equivalent failure modes: (a) racing grant-read already added
|
||||
// the source and the NOT-ANY guard suppressed our UPDATE, or
|
||||
// (b) the client was soft-deleted between resolveClient and UPDATE.
|
||||
// (a) is the more common path. Re-resolve to confirm + report.
|
||||
const reresolved = await resolveClient(sql, client.client_id, { includeDeleted: true });
|
||||
if (reresolved.deleted_at != null) {
|
||||
throw new Error(`Client "${client.client_name}" was soft-deleted before write could land.`);
|
||||
}
|
||||
return { kind: 'noop', reason: 'already-granted', client: reresolved, current: reresolved.federated_read };
|
||||
}
|
||||
return { kind: 'updated', client, before: client.federated_read, after };
|
||||
}
|
||||
|
||||
/**
|
||||
* Core: remove a source from the client's federated_read. Atomic via
|
||||
* array_remove + ANY-guard. Same race-correctness rationale as
|
||||
* grantReadCore — concurrent ops serialize at the row lock.
|
||||
*/
|
||||
export async function revokeReadCore(
|
||||
sql: SqlQuery,
|
||||
nameOrId: string,
|
||||
sourceId: string,
|
||||
opts: FederatedReadOpts = {},
|
||||
): Promise<FederatedReadOutcome> {
|
||||
const client = await resolveClient(sql, nameOrId);
|
||||
if (!client.federated_read.includes(sourceId)) {
|
||||
return { kind: 'noop', reason: 'not-present', client, current: client.federated_read };
|
||||
}
|
||||
if (opts.dryRun) {
|
||||
const projected = client.federated_read.filter((s) => s !== sourceId);
|
||||
return { kind: 'updated', client, before: client.federated_read, after: projected };
|
||||
}
|
||||
const after = await removeFederatedReadAtomic(sql, client.client_id, sourceId);
|
||||
if (after === null) {
|
||||
// Same disambiguation as grant: either a concurrent revoke already
|
||||
// removed the source (most common) or the client was soft-deleted.
|
||||
const reresolved = await resolveClient(sql, client.client_id, { includeDeleted: true });
|
||||
if (reresolved.deleted_at != null) {
|
||||
throw new Error(`Client "${client.client_name}" was soft-deleted before write could land.`);
|
||||
}
|
||||
return { kind: 'noop', reason: 'not-present', client: reresolved, current: reresolved.federated_read };
|
||||
}
|
||||
return { kind: 'updated', client, before: client.federated_read, after };
|
||||
}
|
||||
|
||||
/**
|
||||
* Core: replace the whole federated_read list. Idempotent on same list.
|
||||
*
|
||||
* Race semantics: wholesale-overwrite + deleted_at guard. Concurrent
|
||||
* set+set is last-writer-wins (documented contract for `set` — the
|
||||
* operator is asserting the exact list). Concurrent set+grant or
|
||||
* set+revoke is also last-writer-wins. If a strict-merge semantics is
|
||||
* needed, use grant-read / revoke-read individually.
|
||||
*/
|
||||
export async function setFederatedReadCore(
|
||||
sql: SqlQuery,
|
||||
nameOrId: string,
|
||||
sourceCsv: string,
|
||||
opts: FederatedReadOpts = {},
|
||||
): Promise<FederatedReadOutcome> {
|
||||
const next = parseSourceCsv(sourceCsv);
|
||||
const client = await resolveClient(sql, nameOrId);
|
||||
for (const s of next) {
|
||||
await assertSourceExists(sql, s);
|
||||
}
|
||||
const prev = client.federated_read;
|
||||
const same = prev.length === next.length && prev.every((v, i) => v === next[i]);
|
||||
if (same) {
|
||||
return { kind: 'noop', reason: 'same-list', client, current: prev };
|
||||
}
|
||||
if (opts.dryRun) {
|
||||
return { kind: 'updated', client, before: prev, after: next };
|
||||
}
|
||||
const after = await replaceFederatedReadAtomic(sql, client.client_id, next);
|
||||
if (after === null) {
|
||||
throw new Error(`Client "${client.client_name}" was soft-deleted before write could land.`);
|
||||
}
|
||||
return { kind: 'updated', client, before: prev, after };
|
||||
}
|
||||
|
||||
function printOutcome(
|
||||
verb: 'grant' | 'revoke' | 'set',
|
||||
sourceArg: string,
|
||||
outcome: FederatedReadOutcome,
|
||||
dryRun: boolean,
|
||||
): void {
|
||||
// Terminal-injection defense (Codex finding #5, low): a client_name
|
||||
// registered via DCR with ANSI escapes or control chars would
|
||||
// otherwise poison this output. Sanitize ALL strings that round-trip
|
||||
// from the DB before printing.
|
||||
const s = sanitizeForTerminal;
|
||||
const prefix = dryRun ? '[dry-run] ' : '';
|
||||
if (outcome.kind === 'noop') {
|
||||
const name = s(outcome.client.client_name);
|
||||
if (outcome.reason === 'already-granted') {
|
||||
console.log(`${prefix}No change: "${name}" already reads "${s(sourceArg)}".`);
|
||||
} else if (outcome.reason === 'not-present') {
|
||||
console.log(`${prefix}No change: "${name}" did not read "${s(sourceArg)}".`);
|
||||
} else {
|
||||
console.log(`${prefix}No change: "${name}" federated_read already matches.`);
|
||||
}
|
||||
console.log(` federated_read: ${outcome.current.map(s).join(', ') || '(empty)'}`);
|
||||
return;
|
||||
}
|
||||
const { client, before, after } = outcome;
|
||||
const name = s(client.client_name);
|
||||
const wouldOrDid = dryRun ? 'Would' : 'Did';
|
||||
if (verb === 'grant') {
|
||||
console.log(`${prefix}${wouldOrDid} grant: "${name}" can now read "${s(sourceArg)}".`);
|
||||
console.log(` federated_read: ${after.map(s).join(', ')}`);
|
||||
} else if (verb === 'revoke') {
|
||||
console.log(`${prefix}${wouldOrDid} revoke: "${name}" no longer reads "${s(sourceArg)}".`);
|
||||
console.log(` federated_read: ${after.map(s).join(', ') || '(empty — client has no federated reads)'}`);
|
||||
} else {
|
||||
console.log(`${prefix}${wouldOrDid} update "${name}" federated_read:`);
|
||||
console.log(` before: ${before.map(s).join(', ') || '(empty)'}`);
|
||||
console.log(` after: ${after.map(s).join(', ') || '(empty)'}`);
|
||||
}
|
||||
if (after.length === 0) {
|
||||
console.log(
|
||||
'Warning: client now reads no sources via federation. Queries through this ' +
|
||||
'client will only see content scoped explicitly via its write source.',
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Strip `--dry-run` from a positional-arg list. Returns the filtered list
|
||||
* plus the flag value. Kept positional-tolerant — the existing
|
||||
* `auth grant-read alice source` shape MUST keep working, AND
|
||||
* `auth grant-read alice source --dry-run` AND `auth grant-read --dry-run alice source`.
|
||||
*/
|
||||
export function extractDryRun(args: string[]): { dryRun: boolean; rest: string[] } {
|
||||
let dryRun = false;
|
||||
const rest: string[] = [];
|
||||
for (const a of args) {
|
||||
if (a === '--dry-run') {
|
||||
dryRun = true;
|
||||
continue;
|
||||
}
|
||||
rest.push(a);
|
||||
}
|
||||
return { dryRun, rest };
|
||||
}
|
||||
|
||||
async function grantRead(args: string[]): Promise<void> {
|
||||
const { dryRun, rest } = extractDryRun(args);
|
||||
const [nameOrId, sourceId] = rest;
|
||||
if (!nameOrId || !sourceId) {
|
||||
console.error('Usage: gbrain auth grant-read <client-name-or-id> <source-id> [--dry-run]');
|
||||
process.exit(1);
|
||||
}
|
||||
try {
|
||||
await withConfiguredSql(async (sql) => {
|
||||
const outcome = await grantReadCore(sql, nameOrId, sourceId, { dryRun });
|
||||
printOutcome('grant', sourceId, outcome, dryRun);
|
||||
});
|
||||
} catch (e: any) {
|
||||
console.error('Error:', e.message);
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
async function revokeRead(args: string[]): Promise<void> {
|
||||
const { dryRun, rest } = extractDryRun(args);
|
||||
const [nameOrId, sourceId] = rest;
|
||||
if (!nameOrId || !sourceId) {
|
||||
console.error('Usage: gbrain auth revoke-read <client-name-or-id> <source-id> [--dry-run]');
|
||||
process.exit(1);
|
||||
}
|
||||
try {
|
||||
await withConfiguredSql(async (sql) => {
|
||||
const outcome = await revokeReadCore(sql, nameOrId, sourceId, { dryRun });
|
||||
printOutcome('revoke', sourceId, outcome, dryRun);
|
||||
});
|
||||
} catch (e: any) {
|
||||
console.error('Error:', e.message);
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
async function setFederatedRead(args: string[]): Promise<void> {
|
||||
const { dryRun, rest } = extractDryRun(args);
|
||||
const [nameOrId, sourceCsv] = rest;
|
||||
if (!nameOrId || sourceCsv === undefined) {
|
||||
console.error(
|
||||
'Usage: gbrain auth set-federated-read <client-name-or-id> <source-id1,source-id2,...> [--dry-run]',
|
||||
);
|
||||
console.error('Pass an empty string ("") to clear all federated reads.');
|
||||
process.exit(1);
|
||||
}
|
||||
try {
|
||||
await withConfiguredSql(async (sql) => {
|
||||
const outcome = await setFederatedReadCore(sql, nameOrId, sourceCsv, { dryRun });
|
||||
printOutcome('set', sourceCsv, outcome, dryRun);
|
||||
});
|
||||
} catch (e: any) {
|
||||
console.error('Error:', e.message);
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
async function revokeClient(clientId: string) {
|
||||
if (!clientId) {
|
||||
console.error('Usage: auth revoke-client <client_id>');
|
||||
@@ -319,7 +884,7 @@ async function revokeClient(clientId: string) {
|
||||
console.error(`No client found with id "${clientId}"`);
|
||||
process.exit(1);
|
||||
}
|
||||
console.log(`OAuth client revoked: "${rows[0].client_name}" (${clientId})`);
|
||||
console.log(`OAuth client revoked: "${sanitizeForTerminal(String(rows[0].client_name))}" (${clientId})`);
|
||||
console.log('Tokens and authorization codes purged via cascade.');
|
||||
});
|
||||
} catch (e: any) {
|
||||
@@ -440,6 +1005,15 @@ export function parseRegisterClientArgs(args: string[]): RegisterClientArgs {
|
||||
if (!grantTypesSet && out.redirectUris.length > 0) {
|
||||
out.grantTypes = ['authorization_code', 'refresh_token'];
|
||||
}
|
||||
// Codex re-review (medium): validate source_id shape at the CLI boundary
|
||||
// so register-client can't seed malformed entries into source_id /
|
||||
// federated_read that subsequent grant/revoke/set commands can't manage.
|
||||
assertValidSourceId(out.sourceId);
|
||||
if (out.federatedRead) {
|
||||
for (const s of out.federatedRead) {
|
||||
assertValidSourceId(s);
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
@@ -557,6 +1131,10 @@ export async function runAuth(args: string[]): Promise<void> {
|
||||
}
|
||||
case 'register-client': await registerClient(rest[0], rest.slice(1)); return;
|
||||
case 'revoke-client': await revokeClient(rest[0]); return;
|
||||
case 'list-clients': await listClients(rest); return;
|
||||
case 'grant-read': await grantRead(rest); return;
|
||||
case 'revoke-read': await revokeRead(rest); return;
|
||||
case 'set-federated-read': await setFederatedRead(rest); return;
|
||||
case 'test': {
|
||||
const tokenIdx = rest.indexOf('--token');
|
||||
const url = rest.find(a => !a.startsWith('--') && a !== rest[tokenIdx + 1]);
|
||||
@@ -594,6 +1172,13 @@ Usage:
|
||||
--bound-max-concurrent <n> Bound submit_agent concurrency (default: 1)
|
||||
--budget-usd-per-day <usd> Bound submit_agent daily spend cap
|
||||
gbrain auth revoke-client <client_id> Hard-delete an OAuth 2.1 client (cascades to tokens + codes)
|
||||
gbrain auth list-clients [--json] List OAuth 2.1 clients with scope + write source + federated_read.
|
||||
gbrain auth grant-read <name|client_id> <source-id> [--dry-run]
|
||||
Add a source to the client's federated_read list (idempotent).
|
||||
gbrain auth revoke-read <name|client_id> <source-id> [--dry-run]
|
||||
Remove a source from the client's federated_read list (idempotent).
|
||||
gbrain auth set-federated-read <name|client_id> "<id1,id2,...>" [--dry-run]
|
||||
Replace the client's whole federated_read list. Pass "" to clear.
|
||||
gbrain auth test <url> --token <token> Smoke-test a remote MCP server
|
||||
`);
|
||||
}
|
||||
|
||||
@@ -365,6 +365,42 @@ export interface AgentClientSpend {
|
||||
inflight_count: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* `/admin/api/sources` source list — the input rows for buildSyncStatusReport.
|
||||
*
|
||||
* Queries the JSONB config column directly (listSources doesn't carry it,
|
||||
* but buildSyncStatusReport needs syncEnabled / strategy fields).
|
||||
*
|
||||
* Deliberately does NOT filter on local_path: in a push-only deployment
|
||||
* (content arrives via MCP put_page / capture / ingest, not `gbrain sync`
|
||||
* of a server checkout) every source has a null local_path — filtering on
|
||||
* it would empty both the Sources tab AND the federation source-picker.
|
||||
* buildSyncStatusReport does no disk I/O, so null-local_path sources
|
||||
* report fine (pages/chunks from SQL, staleness 'unknown' / never-synced).
|
||||
*/
|
||||
export async function queryAdminSources(engine: BrainEngine): Promise<
|
||||
Array<{ id: string; name: string; local_path: string | null; config: Record<string, unknown> }>
|
||||
> {
|
||||
const rows = await engine.executeRaw<{
|
||||
id: string;
|
||||
name: string;
|
||||
local_path: string | null;
|
||||
config: Record<string, unknown> | string | null;
|
||||
}>(
|
||||
`SELECT id, name, local_path, config FROM sources
|
||||
WHERE archived IS NOT TRUE
|
||||
ORDER BY id`,
|
||||
);
|
||||
return rows.map((r) => ({
|
||||
id: r.id,
|
||||
name: r.name,
|
||||
local_path: r.local_path,
|
||||
config: typeof r.config === 'string'
|
||||
? (JSON.parse(r.config) as Record<string, unknown>)
|
||||
: (r.config ?? {}),
|
||||
}));
|
||||
}
|
||||
|
||||
export async function queryAgentClientSpend(engine: BrainEngine): Promise<AgentClientSpend[]> {
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
const rows = await sql`
|
||||
@@ -1511,6 +1547,111 @@ export async function runServeHttp(engine: BrainEngine, options: ServeHttpOption
|
||||
}
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Sources tab — read-only view of registered sources with sync + embed
|
||||
// coverage stats. Drives the admin SPA's `Sources` page.
|
||||
//
|
||||
// Returns the same shape `gbrain sources status --json` prints, so the
|
||||
// SPA stays in lockstep with the CLI surface.
|
||||
// ---------------------------------------------------------------------------
|
||||
app.get('/admin/api/sources', requireAdmin, async (_req: Request, res: Response) => {
|
||||
try {
|
||||
const { buildSyncStatusReport } = await import('./sync.ts');
|
||||
const report = await buildSyncStatusReport(engine, await queryAdminSources(engine));
|
||||
res.json(report);
|
||||
} catch (e) {
|
||||
const msg = e instanceof Error ? e.message : String(e);
|
||||
res.status(503).json({ error: 'service_unavailable', detail: msg });
|
||||
}
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Federated-read management (admin-side counterparts of the CLI commands
|
||||
// `gbrain auth grant-read / revoke-read / set-federated-read`). All three
|
||||
// route through the same *Core helpers as the CLI so race-safety,
|
||||
// soft-delete filter, and source-id shape validation apply uniformly.
|
||||
//
|
||||
// The admin SPA's `Agents` page renders "Manage reads" actions per
|
||||
// client backed by these endpoints.
|
||||
// ---------------------------------------------------------------------------
|
||||
app.get('/admin/api/agents/federated-read', requireAdmin, async (_req: Request, res: Response) => {
|
||||
try {
|
||||
const rows = await sql`
|
||||
SELECT client_id, client_name, source_id, federated_read
|
||||
FROM oauth_clients
|
||||
WHERE deleted_at IS NULL
|
||||
ORDER BY client_name
|
||||
`;
|
||||
const clients = rows.map((r) => ({
|
||||
client_id: String(r.client_id),
|
||||
client_name: String(r.client_name),
|
||||
source_id: r.source_id == null ? null : String(r.source_id),
|
||||
federated_read: Array.isArray(r.federated_read)
|
||||
? (r.federated_read as string[]).map(String)
|
||||
: [],
|
||||
}));
|
||||
res.json({ clients });
|
||||
} catch (e) {
|
||||
const msg = e instanceof Error ? e.message : String(e);
|
||||
res.status(503).json({ error: 'service_unavailable', detail: msg });
|
||||
}
|
||||
});
|
||||
|
||||
app.post('/admin/api/agents/:clientId/grant-read', requireAdmin, express.json(), async (req: Request, res: Response) => {
|
||||
const clientId = String(req.params.clientId ?? '');
|
||||
const sourceId = String(req.body?.source_id ?? '').trim();
|
||||
if (!clientId || !sourceId) {
|
||||
res.status(400).json({ error: 'invalid_request', detail: 'clientId path param + source_id body required' });
|
||||
return;
|
||||
}
|
||||
try {
|
||||
const { grantReadCore } = await import('./auth.ts');
|
||||
const outcome = await grantReadCore(sql, clientId, sourceId);
|
||||
res.json({ outcome });
|
||||
} catch (e) {
|
||||
const msg = e instanceof Error ? e.message : String(e);
|
||||
res.status(400).json({ error: 'mutation_failed', detail: msg });
|
||||
}
|
||||
});
|
||||
|
||||
app.post('/admin/api/agents/:clientId/revoke-read', requireAdmin, express.json(), async (req: Request, res: Response) => {
|
||||
const clientId = String(req.params.clientId ?? '');
|
||||
const sourceId = String(req.body?.source_id ?? '').trim();
|
||||
if (!clientId || !sourceId) {
|
||||
res.status(400).json({ error: 'invalid_request', detail: 'clientId path param + source_id body required' });
|
||||
return;
|
||||
}
|
||||
try {
|
||||
const { revokeReadCore } = await import('./auth.ts');
|
||||
const outcome = await revokeReadCore(sql, clientId, sourceId);
|
||||
res.json({ outcome });
|
||||
} catch (e) {
|
||||
const msg = e instanceof Error ? e.message : String(e);
|
||||
res.status(400).json({ error: 'mutation_failed', detail: msg });
|
||||
}
|
||||
});
|
||||
|
||||
app.post('/admin/api/agents/:clientId/set-federated-read', requireAdmin, express.json(), async (req: Request, res: Response) => {
|
||||
const clientId = String(req.params.clientId ?? '');
|
||||
const rawIds = req.body?.source_ids;
|
||||
if (!clientId || !Array.isArray(rawIds)) {
|
||||
res.status(400).json({ error: 'invalid_request', detail: 'clientId path param + source_ids[] body required' });
|
||||
return;
|
||||
}
|
||||
// Encode the array as CSV so the same setFederatedReadCore signature
|
||||
// (string CSV input) the CLI uses applies here. Empty array → empty
|
||||
// string → clears the list.
|
||||
const csv = rawIds.map((s) => String(s).trim()).filter(Boolean).join(',');
|
||||
try {
|
||||
const { setFederatedReadCore } = await import('./auth.ts');
|
||||
const outcome = await setFederatedReadCore(sql, clientId, csv);
|
||||
res.json({ outcome });
|
||||
} catch (e) {
|
||||
const msg = e instanceof Error ? e.message : String(e);
|
||||
res.status(400).json({ error: 'mutation_failed', detail: msg });
|
||||
}
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// SSE live activity feed
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
@@ -18,9 +18,8 @@
|
||||
* at runtime.
|
||||
*/
|
||||
|
||||
import { chunkText as recursiveChunk, capByEstimatedTokens, DEFAULT_MAX_EST_TOKENS } from './recursive.ts';
|
||||
import { chunkText as recursiveChunk } from './recursive.ts';
|
||||
import { buildQualifiedName } from './qualified-names.ts';
|
||||
import { estimateEmbeddingTokens } from '../cjk.ts';
|
||||
|
||||
// Embed the tree-sitter runtime + per-language grammars as files.
|
||||
// `with { type: 'file' }` returns a path (string) at runtime. Bun bundles
|
||||
@@ -112,15 +111,7 @@ import G_ZIG from '../../assets/wasm/grammars/tree-sitter-zig.wasm' with { type:
|
||||
// chunks get the new columns populated. Without this, the v28 backfill
|
||||
// gives every existing chunk a search_vector but subsequent Layer 5 AST
|
||||
// work would silently no-op.
|
||||
//
|
||||
// v5: estimated-token hard cap on AST-path chunks (capCodeChunks). A node
|
||||
// splitLargeNode can't subdivide (giant single-statement function, huge
|
||||
// literal) previously shipped WHOLE regardless of size and could overflow
|
||||
// strict per-request embedding-token limits (local llama-server crashes
|
||||
// past ~2,050 tokens, measured). Mirrors the markdown
|
||||
// chunker's v4 cap; fallback-path chunks are already capped inside
|
||||
// recursiveChunk.
|
||||
export const CHUNKER_VERSION = 5;
|
||||
export const CHUNKER_VERSION = 4;
|
||||
|
||||
// Lazy-loaded tree-sitter module (v0.22.x API: Parser is default export)
|
||||
let Parser: typeof import('web-tree-sitter') | null = null;
|
||||
@@ -717,7 +708,7 @@ export async function chunkCodeTextFull(
|
||||
if (chunks.length === 0) {
|
||||
return { chunks: fallbackChunks(source, filePath, language, opts), edges: rawEdges };
|
||||
}
|
||||
return { chunks: capCodeChunks(mergeSmallSiblings(chunks, chunkTarget)), edges: rawEdges };
|
||||
return { chunks: mergeSmallSiblings(chunks, chunkTarget), edges: rawEdges };
|
||||
} catch {
|
||||
return { chunks: fallbackChunks(source, filePath, language, opts), edges: [] };
|
||||
} finally {
|
||||
@@ -800,33 +791,6 @@ function mergeSmallSiblings(chunks: CodeChunk[], chunkTarget: number): CodeChunk
|
||||
return merged;
|
||||
}
|
||||
|
||||
/**
|
||||
* v5 final safety pass for AST-path chunks: split any chunk whose
|
||||
* ESTIMATED embedding tokens (conservative per-char-class heuristic,
|
||||
* cjk.ts) exceed DEFAULT_MAX_EST_TOKENS. Reaches chunks the AST logic
|
||||
* can't subdivide — splitLargeNode returns [] for nodes with < 2 body
|
||||
* children (giant single-statement functions, huge literals), which
|
||||
* previously shipped whole at any size.
|
||||
*
|
||||
* Split pieces inherit the source chunk's metadata verbatim; start/end
|
||||
* lines become approximate for pieces after the first. Acceptable —
|
||||
* these chunks exist for embedding + retrieval, and the alternative was
|
||||
* an embedding request the server rejects (or worse, crashes on).
|
||||
*/
|
||||
function capCodeChunks(chunks: CodeChunk[]): CodeChunk[] {
|
||||
if (chunks.every((c) => estimateEmbeddingTokens(c.text) <= DEFAULT_MAX_EST_TOKENS)) {
|
||||
return chunks;
|
||||
}
|
||||
const out: CodeChunk[] = [];
|
||||
for (const c of chunks) {
|
||||
const pieces = capByEstimatedTokens(c.text, DEFAULT_MAX_EST_TOKENS);
|
||||
for (const piece of pieces) {
|
||||
out.push({ ...c, text: piece, index: out.length, metadata: { ...c.metadata } });
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
function buildMergedChunk(group: CodeChunk[], index: number): CodeChunk {
|
||||
const first = group[0]!;
|
||||
const last = group[group.length - 1]!;
|
||||
|
||||
@@ -17,13 +17,7 @@
|
||||
* Lossless invariant: non-overlapping portions reassemble to original.
|
||||
*/
|
||||
|
||||
import {
|
||||
countCJKAwareWords,
|
||||
CJK_SENTENCE_DELIMITERS,
|
||||
CJK_CLAUSE_DELIMITERS,
|
||||
charEmbedTokenWeight,
|
||||
estimateEmbeddingTokens,
|
||||
} from '../cjk.ts';
|
||||
import { countCJKAwareWords, CJK_SENTENCE_DELIMITERS, CJK_CLAUSE_DELIMITERS } from '../cjk.ts';
|
||||
|
||||
/**
|
||||
* Markdown chunker version. Folded into the per-page chunker_version column
|
||||
@@ -39,20 +33,8 @@ import {
|
||||
* re-embed (not re-chunk) so existing pages pick up the wrapper on the
|
||||
* post-upgrade reembed sweep. See
|
||||
* `src/core/contextual-retrieval-service.ts`.
|
||||
*
|
||||
* v4: estimated-token hard cap + whitespace-word undercount fix. The word
|
||||
* pipeline counted a 150-char URL as ONE whitespace word, so URL/phone/
|
||||
* email-dense docs (CJK density < 0.30 → whitespace fallback) produced
|
||||
* 3-4K-char chunks that overflow strict per-request embedding-token
|
||||
* limits (measured: local llama-server crashes past ~2,050 tokens; URL
|
||||
* soup tokenizes at ~1.6 chars/token). Two changes:
|
||||
* 1. countWords() floors the count at ceil(nonWhitespaceChars/6) so a
|
||||
* URL counts roughly per-character, not as one word.
|
||||
* 2. capByEstimatedTokens() final pass guarantees every chunk fits
|
||||
* `maxTokens` (default 1500) under a conservative per-char-class
|
||||
* token estimate, regardless of how word counting misjudged it.
|
||||
*/
|
||||
export const MARKDOWN_CHUNKER_VERSION = 4;
|
||||
export const MARKDOWN_CHUNKER_VERSION = 3;
|
||||
|
||||
const DELIMITERS: string[][] = [
|
||||
['\n\n'], // L0: paragraphs
|
||||
@@ -66,20 +48,8 @@ export interface ChunkOptions {
|
||||
chunkSize?: number; // target words per chunk (default 300)
|
||||
chunkOverlap?: number; // overlap words (default 50)
|
||||
maxChars?: number; // hard cap on any chunk's char length (default 6000)
|
||||
/**
|
||||
* v4: hard cap on any chunk's ESTIMATED embedding tokens (default 1500).
|
||||
* Estimate = conservative per-char-class weights (see cjk.ts
|
||||
* estimateEmbeddingTokens) — deliberately high, so the real tokenizer
|
||||
* count stays below this value. Default leaves headroom for the
|
||||
* contextual-retrieval wrapper (≤ ~630 chars) under a ~2,050-token
|
||||
* per-request embedding server limit.
|
||||
*/
|
||||
maxTokens?: number;
|
||||
}
|
||||
|
||||
/** v4 default for ChunkOptions.maxTokens — see the field doc above. */
|
||||
export const DEFAULT_MAX_EST_TOKENS = 1500;
|
||||
|
||||
export interface TextChunk {
|
||||
text: string;
|
||||
index: number;
|
||||
@@ -103,7 +73,6 @@ export function chunkText(text: string, opts?: ChunkOptions): TextChunk[] {
|
||||
const chunkSize = opts?.chunkSize || 300;
|
||||
const chunkOverlap = opts?.chunkOverlap || 50;
|
||||
const maxChars = opts?.maxChars || 6000;
|
||||
const maxTokens = opts?.maxTokens || DEFAULT_MAX_EST_TOKENS;
|
||||
|
||||
if (!text || text.trim().length === 0) return [];
|
||||
|
||||
@@ -120,9 +89,8 @@ export function chunkText(text: string, opts?: ChunkOptions): TextChunk[] {
|
||||
|
||||
const wordCount = countWords(stripped);
|
||||
if (wordCount <= chunkSize) {
|
||||
// Single-chunk path: still apply the maxChars + maxTokens caps.
|
||||
const capped = capByChars(stripped.trim(), maxChars)
|
||||
.flatMap((t) => capByEstimatedTokens(t, maxTokens));
|
||||
// Single-chunk path: still apply the maxChars cap.
|
||||
const capped = capByChars(stripped.trim(), maxChars);
|
||||
return capped.map((t, i) => ({ text: t, index: i }));
|
||||
}
|
||||
|
||||
@@ -133,14 +101,9 @@ export function chunkText(text: string, opts?: ChunkOptions): TextChunk[] {
|
||||
// v0.32.7: hard char cap. Catches pathological CJK + whitespace-less text
|
||||
// that the word-level pipeline can't bound (a single Chinese paragraph can
|
||||
// exceed 8192 OpenAI embedding tokens at any word count).
|
||||
// v4: estimated-token cap on top — the char cap alone passes token-dense
|
||||
// content (URL soup at ~1.6 chars/token) that overflows strict embedding
|
||||
// server limits.
|
||||
const capped: string[] = [];
|
||||
for (const chunk of withOverlap) {
|
||||
for (const piece of capByChars(chunk.trim(), maxChars)) {
|
||||
capped.push(...capByEstimatedTokens(piece, maxTokens));
|
||||
}
|
||||
capped.push(...capByChars(chunk.trim(), maxChars));
|
||||
}
|
||||
return capped.map((t, i) => ({ text: t, index: i }));
|
||||
}
|
||||
@@ -169,68 +132,6 @@ function capByChars(text: string, maxChars: number): string[] {
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* How far back (in chars) the token cap looks for a friendly cut point
|
||||
* before falling back to a hard cut. 300 covers typical rollup/list line
|
||||
* lengths so forced splits land at line starts, not mid-URL.
|
||||
*/
|
||||
const TOKEN_CAP_CUT_LOOKBACK = 300;
|
||||
|
||||
/**
|
||||
* v4: hard-cap a chunk's ESTIMATED embedding tokens. Final safety pass —
|
||||
* runs after capByChars on every chunk, so no upstream miscounting
|
||||
* (whitespace-word fallback, overlap inflation, char-cap survivors) can
|
||||
* emit a chunk past `maxTokens`.
|
||||
*
|
||||
* Cut placement prefers, within the last TOKEN_CAP_CUT_LOOKBACK chars of
|
||||
* the window: a newline, then any whitespace, then a hard cut. This keeps
|
||||
* forced splits off mid-line/mid-URL positions for list-shaped content
|
||||
* and inside code fences. No overlap is added (pieces stay lossless
|
||||
* modulo the trims the char cap already applies).
|
||||
*
|
||||
* @internal exported for the code chunker (code.ts) and tests.
|
||||
*/
|
||||
export function capByEstimatedTokens(text: string, maxTokens: number): string[] {
|
||||
if (text.length === 0) return [];
|
||||
if (estimateEmbeddingTokens(text) <= maxTokens) return [text];
|
||||
|
||||
const out: string[] = [];
|
||||
let start = 0;
|
||||
while (start < text.length) {
|
||||
// Greedily extend the window until the next char would break the cap.
|
||||
// Always take at least one char so the loop makes forward progress.
|
||||
let est = 0;
|
||||
let end = start;
|
||||
while (end < text.length) {
|
||||
const w = charEmbedTokenWeight(text.charCodeAt(end));
|
||||
if (est + w > maxTokens && end > start) break;
|
||||
est += w;
|
||||
end++;
|
||||
}
|
||||
|
||||
if (end < text.length) {
|
||||
const windowStart = Math.max(start + 1, end - TOKEN_CAP_CUT_LOOKBACK);
|
||||
let cut = text.lastIndexOf('\n', end - 1);
|
||||
if (cut < windowStart) {
|
||||
cut = -1;
|
||||
for (let i = end - 1; i >= windowStart; i--) {
|
||||
const code = text.charCodeAt(i);
|
||||
if (code === 0x20 || (code >= 0x09 && code <= 0x0d)) {
|
||||
cut = i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (cut >= windowStart) end = cut + 1;
|
||||
}
|
||||
|
||||
const slice = text.slice(start, end).trim();
|
||||
if (slice.length > 0) out.push(slice);
|
||||
start = end;
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
function recursiveSplit(text: string, level: number, target: number): string[] {
|
||||
if (level >= DELIMITERS.length) {
|
||||
// Level 4: split on whitespace
|
||||
@@ -416,19 +317,7 @@ function extractTrailingContext(text: string, targetWords: number): string {
|
||||
* Delegated to src/core/cjk.ts so the slugify whitelist, expansion
|
||||
* detection, and PGLite keyword fallback all agree on what "CJK enough"
|
||||
* means.
|
||||
*
|
||||
* v4: floored at ceil(nonWhitespaceChars/6). The whitespace fallback
|
||||
* counts a 150-char URL as ONE word, so URL/phone/email-dense docs
|
||||
* (whose ASCII mass pushes CJK density below the 0.30 threshold) were
|
||||
* sized at a fraction of their real bulk and merged into 3-4K-char
|
||||
* chunks. The floor makes long whitespace-less runs count roughly
|
||||
* per-character while leaving normal Latin prose untouched (average
|
||||
* English word ≈ 5 chars < 6, so the whitespace count still wins).
|
||||
* Kept local to the chunker — search/expansion.ts keeps the original
|
||||
* countCJKAwareWords semantics for its query-length check.
|
||||
*/
|
||||
function countWords(text: string): number {
|
||||
const cjkAware = countCJKAwareWords(text);
|
||||
const nonWhitespace = text.replace(/\s/g, '').length;
|
||||
return Math.max(cjkAware, Math.ceil(nonWhitespace / 6));
|
||||
return countCJKAwareWords(text);
|
||||
}
|
||||
|
||||
@@ -65,64 +65,3 @@ export function countCJKAwareWords(s: string): number {
|
||||
export function escapeLikePattern(s: string): string {
|
||||
return s.replace(/\\/g, '\\\\').replace(/%/g, '\\%').replace(/_/g, '\\_');
|
||||
}
|
||||
|
||||
/**
|
||||
* Conservative per-char-class embedding-token weights (markdown chunker v4).
|
||||
*
|
||||
* Why this exists: the chunker's "word" counting drastically UNDER-counts
|
||||
* whitespace-less ASCII runs (a 150-char URL = 1 whitespace word), so
|
||||
* word-based size targets can emit chunks that overflow an embedding
|
||||
* server's per-request token limit. Measured on a local Qwen3-embedding
|
||||
* llama-server stack:
|
||||
* - URL/phone/email-dense text tokenizes at ~1.6 chars/token
|
||||
* - base64-ish / minified blobs approach ~1.3 chars/token (worst case)
|
||||
* - Korean prose tokenizes NO WORSE than 1 char/token in practice
|
||||
*
|
||||
* Weights are deliberately HIGH (tokens are overestimated) so any cap
|
||||
* based on this estimate is safe against real tokenizers:
|
||||
* - CJK char → 1.0 token (real CJK prose is cheaper)
|
||||
* - other non-space → 0.75 token (≈1.33 chars/token, covers base64)
|
||||
* - whitespace → 0.1 token (mostly folds into neighbor tokens)
|
||||
*/
|
||||
export const EMBED_TOKEN_WEIGHT_CJK = 1.0;
|
||||
export const EMBED_TOKEN_WEIGHT_OTHER = 0.75;
|
||||
export const EMBED_TOKEN_WEIGHT_WS = 0.1;
|
||||
|
||||
/** BMP CJK check by UTF-16 code unit — same ranges as CJK_SLUG_CHARS. */
|
||||
export function isCJKCodeUnit(code: number): boolean {
|
||||
return (
|
||||
(code >= 0x4e00 && code <= 0x9fff) || // Han
|
||||
(code >= 0x3040 && code <= 0x309f) || // Hiragana
|
||||
(code >= 0x30a0 && code <= 0x30ff) || // Katakana
|
||||
(code >= 0xac00 && code <= 0xd7af) // Hangul Syllables
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Per-code-unit token weight. Unrecognized whitespace (exotic Unicode
|
||||
* spaces) intentionally falls into OTHER — that only overestimates.
|
||||
*/
|
||||
export function charEmbedTokenWeight(code: number): number {
|
||||
if (isCJKCodeUnit(code)) return EMBED_TOKEN_WEIGHT_CJK;
|
||||
if (
|
||||
code === 0x20 || (code >= 0x09 && code <= 0x0d) ||
|
||||
code === 0xa0 || code === 0x3000
|
||||
) {
|
||||
return EMBED_TOKEN_WEIGHT_WS;
|
||||
}
|
||||
return EMBED_TOKEN_WEIGHT_OTHER;
|
||||
}
|
||||
|
||||
/**
|
||||
* Tokenizer-free embedding-token estimate (conservative overestimate).
|
||||
* See weight docs above. Astral chars count as 2 OTHER code units —
|
||||
* another overestimate, which is the safe direction.
|
||||
*/
|
||||
export function estimateEmbeddingTokens(s: string): number {
|
||||
if (s.length === 0) return 0;
|
||||
let est = 0;
|
||||
for (let i = 0; i < s.length; i++) {
|
||||
est += charEmbedTokenWeight(s.charCodeAt(i));
|
||||
}
|
||||
return Math.ceil(est);
|
||||
}
|
||||
|
||||
@@ -55,7 +55,7 @@ export interface AgentClientBindings {
|
||||
* `redirect_uri` containing `,`) would be parsed by Postgres as MULTIPLE
|
||||
* array elements, smuggling values past validation. See CSO finding #5.
|
||||
*/
|
||||
function pgArray(arr: string[]): string {
|
||||
export function pgArray(arr: string[]): string {
|
||||
if (!arr || arr.length === 0) return '{}';
|
||||
const escaped = arr.map(s => `"${s.replace(/\\/g, '\\\\').replace(/"/g, '\\"')}"`);
|
||||
return `{${escaped.join(',')}}`;
|
||||
|
||||
@@ -0,0 +1,82 @@
|
||||
import { describe, it, expect, beforeAll, afterAll, beforeEach } from 'bun:test';
|
||||
import { PGLiteEngine } from '../src/core/pglite-engine.ts';
|
||||
import { resetPgliteState } from './helpers/reset-pglite.ts';
|
||||
import { queryAdminSources } from '../src/commands/serve-http.ts';
|
||||
import { buildSyncStatusReport } from '../src/commands/sync.ts';
|
||||
|
||||
/**
|
||||
* v0.41.29 Sources tab — `/admin/api/sources` endpoint SQL.
|
||||
*
|
||||
* The endpoint is a thin Express handler over `queryAdminSources` +
|
||||
* `buildSyncStatusReport`; the source-selection SQL is the load-bearing
|
||||
* surface (same pattern as test/admin-agents-spend.test.ts).
|
||||
*
|
||||
* Pinned behaviors:
|
||||
* - Excludes archived sources
|
||||
* - INCLUDES sources with null local_path (push-only brains: filtering
|
||||
* on local_path emptied the Sources tab + federation source-picker)
|
||||
* - JSONB config surfaces as an object, defaulting to {}
|
||||
* - Deterministic ORDER BY id
|
||||
* - buildSyncStatusReport accepts the rows (no disk I/O on null paths)
|
||||
*/
|
||||
|
||||
let engine: PGLiteEngine;
|
||||
|
||||
beforeAll(async () => {
|
||||
engine = new PGLiteEngine();
|
||||
await engine.connect({});
|
||||
await engine.initSchema();
|
||||
});
|
||||
|
||||
afterAll(async () => {
|
||||
await engine.disconnect();
|
||||
});
|
||||
|
||||
beforeEach(async () => {
|
||||
await resetPgliteState(engine);
|
||||
});
|
||||
|
||||
describe('queryAdminSources (/admin/api/sources SQL)', () => {
|
||||
it('includes push-only sources with null local_path', async () => {
|
||||
await engine.executeRaw(
|
||||
`INSERT INTO sources (id, name, local_path, config)
|
||||
VALUES ('push-only', 'push-only', NULL, '{}'::jsonb)`,
|
||||
);
|
||||
const sources = await queryAdminSources(engine);
|
||||
const ids = sources.map((s) => s.id);
|
||||
expect(ids).toContain('push-only');
|
||||
expect(sources.find((s) => s.id === 'push-only')!.local_path).toBe(null);
|
||||
});
|
||||
|
||||
it('excludes archived sources', async () => {
|
||||
await engine.executeRaw(
|
||||
`INSERT INTO sources (id, name, archived) VALUES ('gone', 'gone', true)`,
|
||||
);
|
||||
const sources = await queryAdminSources(engine);
|
||||
expect(sources.map((s) => s.id)).not.toContain('gone');
|
||||
});
|
||||
|
||||
it('surfaces JSONB config as an object and orders by id', async () => {
|
||||
await engine.executeRaw(
|
||||
`INSERT INTO sources (id, name, config)
|
||||
VALUES ('bbb', 'bbb', '{"syncEnabled": true}'::jsonb),
|
||||
('aaa', 'aaa', '{}'::jsonb)`,
|
||||
);
|
||||
const sources = await queryAdminSources(engine);
|
||||
const ids = sources.map((s) => s.id);
|
||||
expect(ids.indexOf('aaa')).toBeLessThan(ids.indexOf('bbb'));
|
||||
expect(sources.find((s) => s.id === 'bbb')!.config).toEqual({ syncEnabled: true });
|
||||
expect(sources.find((s) => s.id === 'aaa')!.config).toEqual({});
|
||||
});
|
||||
|
||||
it('buildSyncStatusReport accepts the rows (null local_path does not throw)', async () => {
|
||||
await engine.executeRaw(
|
||||
`INSERT INTO sources (id, name, local_path, config)
|
||||
VALUES ('push-only', 'push-only', NULL, '{}'::jsonb)`,
|
||||
);
|
||||
const report = await buildSyncStatusReport(engine, await queryAdminSources(engine));
|
||||
expect(report.schema_version).toBe(1);
|
||||
const row = report.sources.find((s) => s.source_id === 'push-only');
|
||||
expect(row).toBeDefined();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,568 @@
|
||||
/**
|
||||
* Tests for `gbrain auth grant-read|revoke-read|set-federated-read`.
|
||||
*
|
||||
* Pure helper: parseSourceCsv (no DB).
|
||||
* DB-coupled: resolveClient, assertSourceExists, *Core fns — exercised
|
||||
* against a real PGLite via the canonical block.
|
||||
*/
|
||||
import { describe, expect, test, beforeAll, afterAll, beforeEach } from 'bun:test';
|
||||
import { PGLiteEngine } from '../src/core/pglite-engine.ts';
|
||||
import { resetPgliteState } from './helpers/reset-pglite.ts';
|
||||
import { sqlQueryForEngine } from '../src/core/sql-query.ts';
|
||||
import { pgArray } from '../src/core/oauth-provider.ts';
|
||||
import {
|
||||
parseSourceCsv,
|
||||
resolveClient,
|
||||
assertSourceExists,
|
||||
grantReadCore,
|
||||
revokeReadCore,
|
||||
setFederatedReadCore,
|
||||
extractDryRun,
|
||||
sanitizeForTerminal,
|
||||
} from '../src/commands/auth.ts';
|
||||
|
||||
let engine: PGLiteEngine;
|
||||
|
||||
beforeAll(async () => {
|
||||
engine = new PGLiteEngine();
|
||||
await engine.connect({});
|
||||
await engine.initSchema();
|
||||
});
|
||||
|
||||
afterAll(async () => {
|
||||
await engine.disconnect();
|
||||
});
|
||||
|
||||
beforeEach(async () => {
|
||||
await resetPgliteState(engine);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// pure helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
describe('parseSourceCsv', () => {
|
||||
test('splits and trims', () => {
|
||||
expect(parseSourceCsv('a,b,c')).toEqual(['a', 'b', 'c']);
|
||||
expect(parseSourceCsv(' a , b ')).toEqual(['a', 'b']);
|
||||
});
|
||||
|
||||
test('drops empty segments', () => {
|
||||
expect(parseSourceCsv('a,,b,')).toEqual(['a', 'b']);
|
||||
expect(parseSourceCsv(',,')).toEqual([]);
|
||||
expect(parseSourceCsv('')).toEqual([]);
|
||||
});
|
||||
|
||||
test('dedupes while preserving first-seen order', () => {
|
||||
expect(parseSourceCsv('a,b,a,c,b')).toEqual(['a', 'b', 'c']);
|
||||
});
|
||||
});
|
||||
|
||||
describe('extractDryRun', () => {
|
||||
test('absent flag → false', () => {
|
||||
expect(extractDryRun(['alice', 'proj-x'])).toEqual({
|
||||
dryRun: false,
|
||||
rest: ['alice', 'proj-x'],
|
||||
});
|
||||
});
|
||||
|
||||
test('flag at end', () => {
|
||||
expect(extractDryRun(['alice', 'proj-x', '--dry-run'])).toEqual({
|
||||
dryRun: true,
|
||||
rest: ['alice', 'proj-x'],
|
||||
});
|
||||
});
|
||||
|
||||
test('flag at start', () => {
|
||||
expect(extractDryRun(['--dry-run', 'alice', 'proj-x'])).toEqual({
|
||||
dryRun: true,
|
||||
rest: ['alice', 'proj-x'],
|
||||
});
|
||||
});
|
||||
|
||||
test('flag in middle', () => {
|
||||
expect(extractDryRun(['alice', '--dry-run', 'proj-x'])).toEqual({
|
||||
dryRun: true,
|
||||
rest: ['alice', 'proj-x'],
|
||||
});
|
||||
});
|
||||
|
||||
test('no args', () => {
|
||||
expect(extractDryRun([])).toEqual({ dryRun: false, rest: [] });
|
||||
});
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// DB-coupled
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
async function seedSource(id: string): Promise<void> {
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
await sql`INSERT INTO sources (id, name) VALUES (${id}, ${id}) ON CONFLICT (id) DO NOTHING`;
|
||||
}
|
||||
|
||||
async function seedClient(name: string, federated: string[] = []): Promise<string> {
|
||||
// Ensure write source FK is satisfied — every seeded client points at 'default'.
|
||||
await seedSource('default');
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
const clientId = `gbrain_cl_test_${name}_${Date.now()}_${Math.random().toString(16).slice(2, 8)}`;
|
||||
const fedLit = pgArray(federated);
|
||||
await sql`
|
||||
INSERT INTO oauth_clients (client_id, client_name, client_secret_hash,
|
||||
redirect_uris, grant_types, scope,
|
||||
client_id_issued_at, source_id, federated_read)
|
||||
VALUES (${clientId}, ${name}, ${'dummy-hash'},
|
||||
${pgArray([])}, ${pgArray(['client_credentials'])}, ${'read'},
|
||||
${Date.now()}, ${'default'}, ${fedLit})
|
||||
`;
|
||||
return clientId;
|
||||
}
|
||||
|
||||
async function readFederated(clientId: string): Promise<string[]> {
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
const rows = await sql`SELECT federated_read FROM oauth_clients WHERE client_id = ${clientId}`;
|
||||
const fed = rows[0]?.federated_read;
|
||||
return Array.isArray(fed) ? (fed as string[]).map(String) : [];
|
||||
}
|
||||
|
||||
describe('resolveClient', () => {
|
||||
test('matches by client_id', async () => {
|
||||
await seedSource('default');
|
||||
const id = await seedClient('alice', ['default']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
const c = await resolveClient(sql, id);
|
||||
expect(c.client_name).toBe('alice');
|
||||
expect(c.federated_read).toEqual(['default']);
|
||||
});
|
||||
|
||||
test('matches by client_name', async () => {
|
||||
await seedSource('default');
|
||||
await seedClient('alice', ['default']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
const c = await resolveClient(sql, 'alice');
|
||||
expect(c.client_name).toBe('alice');
|
||||
});
|
||||
|
||||
test('errors loudly on no-match', async () => {
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
await expect(resolveClient(sql, 'nobody')).rejects.toThrow(/No active OAuth client found/);
|
||||
});
|
||||
|
||||
test('errors loudly on ambiguous client_name', async () => {
|
||||
await seedSource('default');
|
||||
await seedClient('bob', ['default']);
|
||||
await seedClient('bob', ['default']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
await expect(resolveClient(sql, 'bob')).rejects.toThrow(/Multiple active OAuth clients named/);
|
||||
});
|
||||
|
||||
test('null source_id is preserved as null (legacy row tolerance)', async () => {
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
const clientId = `gbrain_cl_test_null_${Date.now()}`;
|
||||
await sql`
|
||||
INSERT INTO oauth_clients (client_id, client_name, client_secret_hash,
|
||||
redirect_uris, grant_types, scope,
|
||||
client_id_issued_at, source_id, federated_read)
|
||||
VALUES (${clientId}, ${'legacy'}, ${'dummy'},
|
||||
${pgArray([])}, ${pgArray(['client_credentials'])}, ${'read'},
|
||||
${Date.now()}, ${null}, ${pgArray([])})
|
||||
`;
|
||||
const c = await resolveClient(sql, clientId);
|
||||
expect(c.source_id).toBeNull();
|
||||
expect(c.federated_read).toEqual([]);
|
||||
});
|
||||
});
|
||||
|
||||
describe('assertSourceExists', () => {
|
||||
test('passes when present', async () => {
|
||||
await seedSource('proj-x');
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
await expect(assertSourceExists(sql, 'proj-x')).resolves.toBeUndefined();
|
||||
});
|
||||
|
||||
test('throws with paste-ready hint when missing', async () => {
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
await expect(assertSourceExists(sql, 'ghost')).rejects.toThrow(
|
||||
/Source "ghost" does not exist.*gbrain sources add ghost/s,
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
describe('grantReadCore', () => {
|
||||
test('appends when not present and persists', async () => {
|
||||
await seedSource('default');
|
||||
await seedSource('proj-x');
|
||||
const id = await seedClient('alice', ['default']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
const outcome = await grantReadCore(sql, 'alice', 'proj-x');
|
||||
expect(outcome.kind).toBe('updated');
|
||||
if (outcome.kind === 'updated') {
|
||||
expect(outcome.before).toEqual(['default']);
|
||||
expect(outcome.after).toEqual(['default', 'proj-x']);
|
||||
}
|
||||
expect(await readFederated(id)).toEqual(['default', 'proj-x']);
|
||||
});
|
||||
|
||||
test('is idempotent — second call is a noop, list unchanged', async () => {
|
||||
await seedSource('default');
|
||||
await seedSource('proj-x');
|
||||
const id = await seedClient('alice', ['default']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
await grantReadCore(sql, 'alice', 'proj-x');
|
||||
const outcome = await grantReadCore(sql, 'alice', 'proj-x');
|
||||
expect(outcome.kind).toBe('noop');
|
||||
if (outcome.kind === 'noop') {
|
||||
expect(outcome.reason).toBe('already-granted');
|
||||
}
|
||||
expect(await readFederated(id)).toEqual(['default', 'proj-x']);
|
||||
});
|
||||
|
||||
test('refuses unknown source (fails BEFORE mutating)', async () => {
|
||||
await seedSource('default');
|
||||
const id = await seedClient('alice', ['default']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
await expect(grantReadCore(sql, 'alice', 'ghost')).rejects.toThrow(/does not exist/);
|
||||
expect(await readFederated(id)).toEqual(['default']);
|
||||
});
|
||||
|
||||
test('refuses unknown client', async () => {
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
await expect(grantReadCore(sql, 'nobody', 'whatever')).rejects.toThrow(/No active OAuth client found/);
|
||||
});
|
||||
|
||||
test('accepts client_id resolution too', async () => {
|
||||
await seedSource('default');
|
||||
await seedSource('proj-x');
|
||||
const id = await seedClient('alice', ['default']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
await grantReadCore(sql, id, 'proj-x');
|
||||
expect(await readFederated(id)).toEqual(['default', 'proj-x']);
|
||||
});
|
||||
|
||||
test('rejects malformed source_id BEFORE existence check (Codex finding #3)', async () => {
|
||||
await seedSource('default');
|
||||
const id = await seedClient('alice', ['default']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
// Even with a row in `sources` having a weird id, the validator at the
|
||||
// boundary refuses. Closes the "manual SQL plants a row, CLI lets it
|
||||
// become unmanageable in federated_read" vector.
|
||||
await sql`INSERT INTO sources (id, name) VALUES (${'has,"weird"-bits'}, ${'weird'})`;
|
||||
await expect(grantReadCore(sql, 'alice', 'has,"weird"-bits')).rejects.toThrow(/Invalid source_id/);
|
||||
// DB unchanged.
|
||||
expect(await readFederated(id)).toEqual(['default']);
|
||||
});
|
||||
});
|
||||
|
||||
describe('revokeReadCore', () => {
|
||||
test('removes when present', async () => {
|
||||
await seedSource('default');
|
||||
await seedSource('proj-x');
|
||||
const id = await seedClient('alice', ['default', 'proj-x']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
const outcome = await revokeReadCore(sql, 'alice', 'proj-x');
|
||||
expect(outcome.kind).toBe('updated');
|
||||
expect(await readFederated(id)).toEqual(['default']);
|
||||
});
|
||||
|
||||
test('is idempotent — second call is a noop, list unchanged', async () => {
|
||||
await seedSource('default');
|
||||
const id = await seedClient('alice', ['default']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
const outcome = await revokeReadCore(sql, 'alice', 'ghost-source');
|
||||
expect(outcome.kind).toBe('noop');
|
||||
if (outcome.kind === 'noop') {
|
||||
expect(outcome.reason).toBe('not-present');
|
||||
}
|
||||
expect(await readFederated(id)).toEqual(['default']);
|
||||
});
|
||||
|
||||
test('allows clearing the list down to empty (no implicit guard)', async () => {
|
||||
await seedSource('default');
|
||||
const id = await seedClient('alice', ['default']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
await revokeReadCore(sql, 'alice', 'default');
|
||||
expect(await readFederated(id)).toEqual([]);
|
||||
});
|
||||
|
||||
test('does NOT validate the source exists — operator may revoke stale references', async () => {
|
||||
await seedSource('default');
|
||||
// federated_read carries 'proj-x' but the source row was deleted.
|
||||
const id = await seedClient('alice', ['default', 'proj-x']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
const outcome = await revokeReadCore(sql, 'alice', 'proj-x');
|
||||
expect(outcome.kind).toBe('updated');
|
||||
expect(await readFederated(id)).toEqual(['default']);
|
||||
});
|
||||
});
|
||||
|
||||
describe('setFederatedReadCore', () => {
|
||||
test('replaces list wholesale', async () => {
|
||||
await seedSource('a');
|
||||
await seedSource('b');
|
||||
await seedSource('c');
|
||||
const id = await seedClient('alice', ['a']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
const outcome = await setFederatedReadCore(sql, 'alice', 'b,c');
|
||||
expect(outcome.kind).toBe('updated');
|
||||
expect(await readFederated(id)).toEqual(['b', 'c']);
|
||||
});
|
||||
|
||||
test('dedupes CSV input', async () => {
|
||||
await seedSource('a');
|
||||
await seedSource('b');
|
||||
const id = await seedClient('alice', []);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
await setFederatedReadCore(sql, 'alice', 'a,b,a,b,a');
|
||||
expect(await readFederated(id)).toEqual(['a', 'b']);
|
||||
});
|
||||
|
||||
test('empty string clears the list', async () => {
|
||||
await seedSource('a');
|
||||
const id = await seedClient('alice', ['a']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
await setFederatedReadCore(sql, 'alice', '');
|
||||
expect(await readFederated(id)).toEqual([]);
|
||||
});
|
||||
|
||||
test('noop when result equals current list', async () => {
|
||||
await seedSource('a');
|
||||
await seedSource('b');
|
||||
const id = await seedClient('alice', ['a', 'b']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
const outcome = await setFederatedReadCore(sql, 'alice', 'a,b');
|
||||
expect(outcome.kind).toBe('noop');
|
||||
if (outcome.kind === 'noop') {
|
||||
expect(outcome.reason).toBe('same-list');
|
||||
}
|
||||
expect(await readFederated(id)).toEqual(['a', 'b']);
|
||||
});
|
||||
|
||||
test('refuses unknown source (fails BEFORE mutating)', async () => {
|
||||
await seedSource('a');
|
||||
const id = await seedClient('alice', ['a']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
await expect(setFederatedReadCore(sql, 'alice', 'a,ghost')).rejects.toThrow(/does not exist/);
|
||||
// Original list preserved.
|
||||
expect(await readFederated(id)).toEqual(['a']);
|
||||
});
|
||||
|
||||
test('order in CSV is the order persisted', async () => {
|
||||
await seedSource('a');
|
||||
await seedSource('b');
|
||||
await seedSource('c');
|
||||
const id = await seedClient('alice', ['a']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
await setFederatedReadCore(sql, 'alice', 'c,a,b');
|
||||
expect(await readFederated(id)).toEqual(['c', 'a', 'b']);
|
||||
});
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// --dry-run semantics
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Codex fixes: soft-delete filter, atomic-SQL race-safety, sanitizer
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
describe('sanitizeForTerminal', () => {
|
||||
test('preserves printable ASCII unchanged', () => {
|
||||
expect(sanitizeForTerminal('alice')).toBe('alice');
|
||||
expect(sanitizeForTerminal('a b-c_d.e/f@g')).toBe('a b-c_d.e/f@g');
|
||||
});
|
||||
|
||||
test('escapes ANSI escape sequences', () => {
|
||||
expect(sanitizeForTerminal('\x1b[2J')).toBe('\\x1b[2J');
|
||||
expect(sanitizeForTerminal('\x1b]0;TITLE\x07')).toBe('\\x1b]0;TITLE\\x07');
|
||||
});
|
||||
|
||||
test('escapes ALL C0 controls including tab and newline', () => {
|
||||
// Codex re-review: preserving \n lets a DCR-registered name spoof
|
||||
// additional rows in list-clients output. Tab spoofs field separators.
|
||||
// Both are now escaped.
|
||||
expect(sanitizeForTerminal('\x00\x07\x08')).toBe('\\x00\\x07\\x08');
|
||||
expect(sanitizeForTerminal('line1\nline2')).toBe('line1\\x0aline2');
|
||||
expect(sanitizeForTerminal('col1\tcol2')).toBe('col1\\x09col2');
|
||||
});
|
||||
|
||||
test('escapes DEL and C1 controls', () => {
|
||||
expect(sanitizeForTerminal('\x7f')).toBe('\\x7f');
|
||||
expect(sanitizeForTerminal('\x9b[31m')).toBe('\\x9b[31m');
|
||||
});
|
||||
|
||||
test('passes through unicode', () => {
|
||||
expect(sanitizeForTerminal('café')).toBe('café');
|
||||
expect(sanitizeForTerminal('日本語')).toBe('日本語');
|
||||
});
|
||||
});
|
||||
|
||||
describe('soft-delete filter (Codex finding #2)', () => {
|
||||
async function softDeleteClient(clientId: string): Promise<void> {
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
await sql`UPDATE oauth_clients SET deleted_at = now() WHERE client_id = ${clientId}`;
|
||||
}
|
||||
|
||||
test('resolveClient hides soft-deleted clients by default', async () => {
|
||||
await seedSource('default');
|
||||
const id = await seedClient('alice', ['default']);
|
||||
await softDeleteClient(id);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
await expect(resolveClient(sql, 'alice')).rejects.toThrow(/No active OAuth client found/);
|
||||
await expect(resolveClient(sql, id)).rejects.toThrow(/No active OAuth client found/);
|
||||
});
|
||||
|
||||
test('resolveClient with includeDeleted finds soft-deleted clients', async () => {
|
||||
await seedSource('default');
|
||||
const id = await seedClient('alice', ['default']);
|
||||
await softDeleteClient(id);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
const c = await resolveClient(sql, id, { includeDeleted: true });
|
||||
expect(c.client_name).toBe('alice');
|
||||
expect(c.deleted_at).not.toBeNull();
|
||||
});
|
||||
|
||||
test('grantReadCore refuses to mutate soft-deleted clients', async () => {
|
||||
await seedSource('default');
|
||||
await seedSource('proj-x');
|
||||
const id = await seedClient('alice', ['default']);
|
||||
await softDeleteClient(id);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
await expect(grantReadCore(sql, 'alice', 'proj-x')).rejects.toThrow(/No active OAuth client found/);
|
||||
expect(await readFederated(id)).toEqual(['default']);
|
||||
});
|
||||
|
||||
test('revokeReadCore refuses to mutate soft-deleted clients', async () => {
|
||||
await seedSource('default');
|
||||
const id = await seedClient('alice', ['default']);
|
||||
await softDeleteClient(id);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
await expect(revokeReadCore(sql, 'alice', 'default')).rejects.toThrow(/No active OAuth client found/);
|
||||
expect(await readFederated(id)).toEqual(['default']);
|
||||
});
|
||||
|
||||
test('two clients with same name but only one active resolves to the active one', async () => {
|
||||
await seedSource('default');
|
||||
// Seed two clients with the same name; soft-delete the older one.
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
const oldId = await seedClient('alice', ['default']);
|
||||
await softDeleteClient(oldId);
|
||||
const newId = await seedClient('alice', ['default']); // same name, new row
|
||||
const c = await resolveClient(sql, 'alice');
|
||||
expect(c.client_id).toBe(newId); // active row wins; ambiguity error suppressed
|
||||
});
|
||||
});
|
||||
|
||||
describe('atomic SQL race-safety (Codex finding #1, HIGH)', () => {
|
||||
test('grant+revoke serialize at row-lock — sensitive stays revoked', async () => {
|
||||
await seedSource('default');
|
||||
await seedSource('sensitive');
|
||||
await seedSource('harmless');
|
||||
const id = await seedClient('alice', ['default', 'sensitive']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
|
||||
// Simulate concurrent revoke(sensitive) + grant(harmless). Real concurrency
|
||||
// would race at the JS event loop boundary; here we await sequentially but
|
||||
// each call goes through the ATOMIC SQL path. The contract: regardless of
|
||||
// ordering, the final state has sensitive REMOVED and harmless ADDED.
|
||||
await revokeReadCore(sql, 'alice', 'sensitive');
|
||||
await grantReadCore(sql, 'alice', 'harmless');
|
||||
const final1 = await readFederated(id);
|
||||
expect(final1.sort()).toEqual(['default', 'harmless']);
|
||||
|
||||
// Reverse order, same final state. The pre-fix read-modify-write shape
|
||||
// would have produced ['default', 'sensitive', 'harmless'] here (the
|
||||
// resurrection bug Codex caught).
|
||||
const id2 = await seedClient('bob', ['default', 'sensitive']);
|
||||
await grantReadCore(sql, 'bob', 'harmless');
|
||||
await revokeReadCore(sql, 'bob', 'sensitive');
|
||||
const final2 = await readFederated(id2);
|
||||
expect(final2.sort()).toEqual(['default', 'harmless']);
|
||||
});
|
||||
|
||||
test('grant uses RETURNING to surface the post-write state', async () => {
|
||||
await seedSource('default');
|
||||
await seedSource('proj-x');
|
||||
await seedClient('alice', ['default']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
const outcome = await grantReadCore(sql, 'alice', 'proj-x');
|
||||
expect(outcome.kind).toBe('updated');
|
||||
if (outcome.kind === 'updated') {
|
||||
// The `after` came from RETURNING, not from computing prev+sourceId
|
||||
// in JS — proves the atomic path returned authoritative state.
|
||||
expect(outcome.after).toEqual(['default', 'proj-x']);
|
||||
}
|
||||
});
|
||||
|
||||
test('grant noop path still survives without writing', async () => {
|
||||
await seedSource('default');
|
||||
const id = await seedClient('alice', ['default']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
const outcome = await grantReadCore(sql, 'alice', 'default');
|
||||
expect(outcome.kind).toBe('noop');
|
||||
if (outcome.kind === 'noop') expect(outcome.reason).toBe('already-granted');
|
||||
expect(await readFederated(id)).toEqual(['default']);
|
||||
});
|
||||
});
|
||||
|
||||
describe('dryRun mode', () => {
|
||||
test('grantReadCore returns "updated" outcome but skips the write', async () => {
|
||||
await seedSource('default');
|
||||
await seedSource('proj-x');
|
||||
const id = await seedClient('alice', ['default']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
const outcome = await grantReadCore(sql, 'alice', 'proj-x', { dryRun: true });
|
||||
expect(outcome.kind).toBe('updated');
|
||||
if (outcome.kind === 'updated') {
|
||||
expect(outcome.before).toEqual(['default']);
|
||||
expect(outcome.after).toEqual(['default', 'proj-x']);
|
||||
}
|
||||
// Crucially: the DB row is UNCHANGED.
|
||||
expect(await readFederated(id)).toEqual(['default']);
|
||||
});
|
||||
|
||||
test('revokeReadCore returns "updated" outcome but skips the write', async () => {
|
||||
await seedSource('default');
|
||||
await seedSource('proj-x');
|
||||
const id = await seedClient('alice', ['default', 'proj-x']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
const outcome = await revokeReadCore(sql, 'alice', 'proj-x', { dryRun: true });
|
||||
expect(outcome.kind).toBe('updated');
|
||||
expect(await readFederated(id)).toEqual(['default', 'proj-x']);
|
||||
});
|
||||
|
||||
test('setFederatedReadCore returns "updated" outcome but skips the write', async () => {
|
||||
await seedSource('a');
|
||||
await seedSource('b');
|
||||
await seedSource('c');
|
||||
const id = await seedClient('alice', ['a']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
const outcome = await setFederatedReadCore(sql, 'alice', 'b,c', { dryRun: true });
|
||||
expect(outcome.kind).toBe('updated');
|
||||
expect(await readFederated(id)).toEqual(['a']);
|
||||
});
|
||||
|
||||
test('noop outcomes are surfaced identically with or without dryRun', async () => {
|
||||
await seedSource('default');
|
||||
await seedSource('proj-x');
|
||||
await seedClient('alice', ['default', 'proj-x']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
const live = await grantReadCore(sql, 'alice', 'proj-x', { dryRun: false });
|
||||
const dry = await grantReadCore(sql, 'alice', 'proj-x', { dryRun: true });
|
||||
expect(live.kind).toBe('noop');
|
||||
expect(dry.kind).toBe('noop');
|
||||
});
|
||||
|
||||
test('errors still fire in dryRun (operator sees the problem before commit)', async () => {
|
||||
await seedSource('default');
|
||||
const id = await seedClient('alice', ['default']);
|
||||
const sql = sqlQueryForEngine(engine);
|
||||
await expect(
|
||||
grantReadCore(sql, 'alice', 'ghost', { dryRun: true }),
|
||||
).rejects.toThrow(/does not exist/);
|
||||
await expect(
|
||||
grantReadCore(sql, 'nobody', 'default', { dryRun: true }),
|
||||
).rejects.toThrow(/No active OAuth client found/);
|
||||
// DB unchanged.
|
||||
expect(await readFederated(id)).toEqual(['default']);
|
||||
});
|
||||
});
|
||||
@@ -174,6 +174,29 @@ describe('parseRegisterClientArgs', () => {
|
||||
});
|
||||
|
||||
describe('error cases', () => {
|
||||
test('--source with malformed id throws (validates source_id shape — codex re-review)', () => {
|
||||
// Defense for the "register-client seeds an unmanageable
|
||||
// federated_read entry" vector. assertValidSourceId fires before the
|
||||
// function returns so DB never sees a row with bad source scope.
|
||||
expect(() => parseRegisterClientArgs(['--source', 'has,weird,bits'])).toThrow(/Invalid source_id/);
|
||||
expect(() => parseRegisterClientArgs(['--source', 'UPPER'])).toThrow(/Invalid source_id/);
|
||||
expect(() => parseRegisterClientArgs(['--source', ''])).toThrow(/Invalid source_id|requires a value/);
|
||||
});
|
||||
|
||||
test('--federated-read with any malformed id throws', () => {
|
||||
// Single-item bad.
|
||||
expect(() => parseRegisterClientArgs(['--federated-read', 'bad,source!'])).toThrow(/Invalid source_id/);
|
||||
// Mixed valid + invalid — fails on the first bad one.
|
||||
expect(() => parseRegisterClientArgs(['--federated-read', 'good,bad source'])).toThrow(/Invalid source_id/);
|
||||
});
|
||||
|
||||
test('--source default + --federated-read default,team passes (regression — common case)', () => {
|
||||
// Sanity: the canonical real-world invocation still parses cleanly.
|
||||
const out = parseRegisterClientArgs(['--source', 'default', '--federated-read', 'default,team']);
|
||||
expect(out.sourceId).toBe('default');
|
||||
expect(out.federatedRead).toEqual(['default', 'team']);
|
||||
});
|
||||
|
||||
test('--redirect-uri without value → throws', () => {
|
||||
expect(() => parseRegisterClientArgs(['--redirect-uri'])).toThrow(/requires a value/);
|
||||
});
|
||||
|
||||
@@ -15,22 +15,19 @@ import { describe, test, expect } from 'bun:test';
|
||||
import { CHUNKER_VERSION } from '../src/core/chunkers/code.ts';
|
||||
|
||||
describe('Layer 12 — CHUNKER_VERSION constant', () => {
|
||||
test('bumped to 5 for the estimated-token hard cap', () => {
|
||||
test('bumped to 4 for Cathedral II', () => {
|
||||
// v3: v0.19.0 Chonkie parity (tokenizer + small-sibling merge).
|
||||
// v4: v0.20.0 Cathedral II (qualified names + parent scope + doc_comment
|
||||
// + fence extraction + chunk-grain FTS). Folded into content_hash
|
||||
// so any bump forces clean re-chunks on next sync.
|
||||
// v5: estimated-token hard cap on AST-path chunks (capCodeChunks) so
|
||||
// un-subdividable giant nodes can't overflow strict embedding
|
||||
// server token limits.
|
||||
expect(CHUNKER_VERSION).toBe(5);
|
||||
expect(CHUNKER_VERSION).toBe(4);
|
||||
});
|
||||
|
||||
test('is stable across imports (not recomputed at call time)', async () => {
|
||||
const a = (await import('../src/core/chunkers/code.ts')).CHUNKER_VERSION;
|
||||
const b = (await import('../src/core/chunkers/code.ts')).CHUNKER_VERSION;
|
||||
expect(a).toBe(b);
|
||||
expect(a).toBe(5);
|
||||
expect(a).toBe(4);
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
@@ -10,8 +10,8 @@ import { describe, test, expect } from 'bun:test';
|
||||
import { chunkCodeText, detectCodeLanguage, CHUNKER_VERSION } from '../../src/core/chunkers/code.ts';
|
||||
|
||||
describe('CHUNKER_VERSION', () => {
|
||||
test('v5: estimated-token hard cap on AST-path chunks', () => {
|
||||
expect(CHUNKER_VERSION).toBe(5);
|
||||
test('v0.20.0 Cathedral II Layer 12 bumped to 4', () => {
|
||||
expect(CHUNKER_VERSION).toBe(4);
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
@@ -135,14 +135,13 @@ describe('Recursive Text Chunker', () => {
|
||||
});
|
||||
|
||||
describe('CJK chunking (v0.32.7)', () => {
|
||||
test('MARKDOWN_CHUNKER_VERSION is 4', async () => {
|
||||
test('MARKDOWN_CHUNKER_VERSION is 3', async () => {
|
||||
// v0.40.3.0: bumped 2→3 to signal the post-upgrade reembed sweep that
|
||||
// contextual retrieval wrapping is now applied at embed time.
|
||||
// v4: estimated-token hard cap + whitespace-word undercount floor
|
||||
// (URL-dense docs produced chunks past strict embedding server token
|
||||
// limits). Boundary change → forces re-chunk for chunker_version < 4.
|
||||
// contextual retrieval wrapping is now applied at embed time. Chunk
|
||||
// boundaries themselves are unchanged; the bump forces re-embed for
|
||||
// pages where chunker_version < 3.
|
||||
const mod = await import('../../src/core/chunkers/recursive.ts');
|
||||
expect(mod.MARKDOWN_CHUNKER_VERSION).toBe(4);
|
||||
expect(mod.MARKDOWN_CHUNKER_VERSION).toBe(3);
|
||||
});
|
||||
|
||||
test('long pure-Chinese paragraph splits into multiple chunks', () => {
|
||||
|
||||
@@ -1,195 +0,0 @@
|
||||
/**
|
||||
* Markdown chunker v4 / code chunker v5 — estimated-token hard cap
|
||||
* regression tests.
|
||||
*
|
||||
* Reproduces a field failure: a local llama-server embedding backend
|
||||
* (`-ub 2048`) crashes deterministically (trace/BPT trap → EOF at the
|
||||
* client) when a single chunk exceeds ~2,050 real tokens. Two content
|
||||
* shapes triggered it:
|
||||
*
|
||||
* 1. Korean docs carrying one long source URL per line.
|
||||
* The URLs' ASCII mass pushes CJK density below 0.30, flipping
|
||||
* countCJKAwareWords to whitespace counting, where a 150-char URL
|
||||
* counts as ONE word → chunks ballooned to 3-4K chars ≈ 2,000+
|
||||
* real tokens (URL soup tokenizes at ~1.6 chars/token).
|
||||
*
|
||||
* 2. Large JSON code blocks (~7K chars) that the word pipeline
|
||||
* undercounts the same way (few whitespace tokens).
|
||||
*
|
||||
* The fix: every emitted chunk must satisfy
|
||||
* estimateEmbeddingTokens(chunk) <= maxTokens (default 1500)
|
||||
* where the estimate deliberately OVERSTATES real tokenizer counts.
|
||||
*/
|
||||
|
||||
import { describe, test, expect } from 'bun:test';
|
||||
import { chunkText, capByEstimatedTokens, DEFAULT_MAX_EST_TOKENS } from '../../src/core/chunkers/recursive.ts';
|
||||
import { chunkCodeText } from '../../src/core/chunkers/code.ts';
|
||||
import { estimateEmbeddingTokens } from '../../src/core/cjk.ts';
|
||||
|
||||
/** Synthesize the failing shape: Korean rollup lines each ending in a long Notion URL. */
|
||||
function urlDenseKoreanRollup(lines: number): string {
|
||||
const out: string[] = ['# 링크가 줄마다 붙는 한국어 예시 문서', ''];
|
||||
for (let i = 0; i < lines; i++) {
|
||||
const hex32 = (i * 2654435761 >>> 0).toString(16).padStart(8, '0').repeat(4);
|
||||
out.push(
|
||||
`- **항목 ${i}**: 이 줄은 청커 동작 검증을 위한 의미 없는 한국어 예시 문장입니다 · 전화 000-0000-${String(1000 + i)} · ` +
|
||||
`이메일 user${i}@example.com · 링크: https://docs.example.com/pages/${hex32}?v=abcdef0123456789&ref=sample`,
|
||||
);
|
||||
}
|
||||
return out.join('\n');
|
||||
}
|
||||
|
||||
/** Synthesize a large pretty-printed JSON block with CJK values. */
|
||||
function bigJsonBlock(targetChars: number): string {
|
||||
const entries: string[] = [];
|
||||
let i = 0;
|
||||
let len = 0;
|
||||
while (len < targetChars) {
|
||||
const row =
|
||||
` "item_${i}": { "name": "예시-${i}", "url": "https://example.com/api/v2/items/${i}?token=abc${i}def", "qty": ${i % 100}, "memo": "한국어 값이 섞인 예시 데이터" }`;
|
||||
entries.push(row);
|
||||
len += row.length;
|
||||
i++;
|
||||
}
|
||||
return `{\n${entries.join(',\n')}\n}`;
|
||||
}
|
||||
|
||||
describe('v4 estimated-token cap — URL-dense Korean doc (field-failure shape)', () => {
|
||||
test('every chunk stays under the estimated-token cap', () => {
|
||||
const md = urlDenseKoreanRollup(60);
|
||||
const chunks = chunkText(md);
|
||||
expect(chunks.length).toBeGreaterThan(0);
|
||||
for (const c of chunks) {
|
||||
expect(estimateEmbeddingTokens(c.text)).toBeLessThanOrEqual(DEFAULT_MAX_EST_TOKENS);
|
||||
}
|
||||
});
|
||||
|
||||
test('no chunk reaches the measured 3K-char danger zone for URL soup', () => {
|
||||
const md = urlDenseKoreanRollup(60);
|
||||
const chunks = chunkText(md);
|
||||
// 1500 est tokens at the OTHER weight (0.75/char) bounds chunks to
|
||||
// ~2,000 chars for pure ASCII — well under the ~3,300 chars where
|
||||
// URL-dense content crosses ~2,050 real tokens (1.6 chars/token).
|
||||
for (const c of chunks) {
|
||||
expect(c.text.length).toBeLessThanOrEqual(2600);
|
||||
}
|
||||
});
|
||||
|
||||
test('content is preserved (no lines dropped by the cap)', () => {
|
||||
const md = urlDenseKoreanRollup(60);
|
||||
const chunks = chunkText(md);
|
||||
const joined = chunks.map((c) => c.text).join('\n');
|
||||
// Spot-check first / middle / last rollup lines survive chunking.
|
||||
for (const marker of ['항목 0', '항목 30', '항목 59']) {
|
||||
expect(joined).toContain(marker);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('v4 estimated-token cap — large JSON blocks', () => {
|
||||
test('7K-char pretty JSON through the prose path stays under the cap', () => {
|
||||
const md = `설정 파일 원문 보존:\n\n\`\`\`\n${bigJsonBlock(7000)}\n\`\`\`\n`;
|
||||
const chunks = chunkText(md);
|
||||
expect(chunks.length).toBeGreaterThan(1);
|
||||
for (const c of chunks) {
|
||||
expect(estimateEmbeddingTokens(c.text)).toBeLessThanOrEqual(DEFAULT_MAX_EST_TOKENS);
|
||||
}
|
||||
});
|
||||
|
||||
test('7K-char minified JSON (single whitespace-less token) stays under the cap', () => {
|
||||
const minified = bigJsonBlock(7000).replace(/\n\s*/g, '');
|
||||
const chunks = chunkText(minified);
|
||||
expect(chunks.length).toBeGreaterThan(1);
|
||||
for (const c of chunks) {
|
||||
expect(estimateEmbeddingTokens(c.text)).toBeLessThanOrEqual(DEFAULT_MAX_EST_TOKENS);
|
||||
}
|
||||
});
|
||||
|
||||
test('json fence via the code chunker stays under the cap (+header slack)', async () => {
|
||||
const chunks = await chunkCodeText(bigJsonBlock(7000), 'fence.json');
|
||||
expect(chunks.length).toBeGreaterThan(0);
|
||||
for (const c of chunks) {
|
||||
// buildChunk prepends a short "[JSON] fence.json:…" header AFTER the
|
||||
// body-level cap; allow ~60 est tokens of header slack. Real-token
|
||||
// safety margin (2,050 − overestimated 1,500) absorbs this easily.
|
||||
expect(estimateEmbeddingTokens(c.text)).toBeLessThanOrEqual(DEFAULT_MAX_EST_TOKENS + 60);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('v4 word-count floor — behavior preserved for normal content', () => {
|
||||
test('Latin prose chunking is unchanged by the floor (avg word < 6 chars)', () => {
|
||||
const prose = Array.from({ length: 120 }, (_, i) =>
|
||||
`This is sentence number ${i} and it talks about ordinary things in plain words.`,
|
||||
).join(' ');
|
||||
const chunks = chunkText(prose);
|
||||
// Historical behavior: ~1,560 whitespace words → multiple ~300-word chunks.
|
||||
expect(chunks.length).toBeGreaterThan(3);
|
||||
for (const c of chunks) {
|
||||
const words = c.text.split(/\s+/).length;
|
||||
expect(words).toBeLessThanOrEqual(300 * 1.5 + 50); // merge cap + overlap
|
||||
}
|
||||
});
|
||||
|
||||
test('Korean prose (CJK-dense, no URLs) never triggers the token cap', () => {
|
||||
const prose = Array.from({ length: 80 }, (_, i) =>
|
||||
`이 문장은 순수 한국어 산문의 청킹 동작을 확인하기 위한 ${i}번째 예시 문장입니다.`,
|
||||
).join(' ');
|
||||
const chunks = chunkText(prose);
|
||||
expect(chunks.length).toBeGreaterThan(1);
|
||||
for (const c of chunks) {
|
||||
// CJK-dense chunks are char-counted (≈450 max) — nowhere near 1500.
|
||||
expect(estimateEmbeddingTokens(c.text)).toBeLessThanOrEqual(700);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('capByEstimatedTokens unit behavior', () => {
|
||||
test('returns input unchanged when under the cap', () => {
|
||||
expect(capByEstimatedTokens('short text', 1500)).toEqual(['short text']);
|
||||
expect(capByEstimatedTokens('', 1500)).toEqual([]);
|
||||
});
|
||||
|
||||
test('prefers newline cut points within the lookback window', () => {
|
||||
const line = 'x'.repeat(100);
|
||||
const text = Array.from({ length: 40 }, () => line).join('\n');
|
||||
const pieces = capByEstimatedTokens(text, 1000);
|
||||
expect(pieces.length).toBeGreaterThan(1);
|
||||
for (const p of pieces) {
|
||||
// Every piece should be whole lines (multiples of the 100-char line).
|
||||
for (const l of p.split('\n')) {
|
||||
expect(l).toBe(line);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
test('makes forward progress on whitespace-less input (hard cut)', () => {
|
||||
const blob = 'a'.repeat(10_000);
|
||||
const pieces = capByEstimatedTokens(blob, 1000);
|
||||
expect(pieces.length).toBeGreaterThan(1);
|
||||
expect(pieces.join('')).toBe(blob);
|
||||
for (const p of pieces) {
|
||||
expect(estimateEmbeddingTokens(p)).toBeLessThanOrEqual(1000);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('estimateEmbeddingTokens — weight sanity', () => {
|
||||
test('overestimates URL-dense ASCII (0.75/char ≥ measured ~0.63/char)', () => {
|
||||
const url = 'https://docs.example.com/pages/a1b2c3d4e5f6a1b2c3d4e5f6a1b2c3d4?v=abc&ref=sample';
|
||||
const est = estimateEmbeddingTokens(url);
|
||||
expect(est).toBeGreaterThanOrEqual(Math.floor(url.length * 0.7));
|
||||
});
|
||||
|
||||
test('counts CJK at 1 token/char', () => {
|
||||
expect(estimateEmbeddingTokens('가나다라마')).toBe(5);
|
||||
});
|
||||
|
||||
test('whitespace is nearly free', () => {
|
||||
expect(estimateEmbeddingTokens(' \n\t ')).toBeLessThanOrEqual(1);
|
||||
});
|
||||
|
||||
test('empty string is 0', () => {
|
||||
expect(estimateEmbeddingTokens('')).toBe(0);
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user