feat: Enhance model context window handling and improve inference logic for custom models (#1214)

This commit is contained in:
Haze
2026-08-03 11:38:39 +08:00
committed by GitHub
parent d0dfe84463
commit 8335ca329c
5 changed files with 305 additions and 42 deletions
+169 -29
View File
@@ -1,50 +1,190 @@
export type ModelInputModality = 'text' | 'image';
type ContextWindowRule = {
/** Human-readable family label; kept so the table reads as documentation. */
label: string;
pattern: RegExp;
contextWindow: number;
};
/**
* Conservative context-window defaults for well-known model families, applied
* to custom-provider model rows that would otherwise carry no `contextWindow`.
* Context-window defaults for well-known model families, applied to model rows
* that would otherwise carry no `contextWindow`.
*
* Why this matters: when a model row has neither `contextTokens` nor
* `contextWindow`, OpenClaw's embedded runner skips preemptive compaction and
* context-window guarding entirely, so long sessions only fail at the provider
* with "Context overflow: prompt too large" instead of being compacted early.
*
* Accuracy matters in both directions. Under-reporting is not the safe choice:
* it makes the runner start preflight compaction long before it is needed, and
* a compaction that times out aborts the whole turn. Over-reporting pushes the
* failure to the provider as a hard overflow. Prefer the vendor's published
* figure for the family rather than a defensive guess.
*
* Ordering contract: rules are evaluated top-down and the first match wins, so
* a specific variant MUST appear above its family fallback. Note that `\b`
* treats `.` and `-` as boundaries, so /\bgpt-5\b/ also matches `gpt-5.6-sol`;
* the generation-specific rules above it are what keep that correct.
*/
const CUSTOM_MODEL_CONTEXT_WINDOW_RULES: Array<{ pattern: RegExp; contextWindow: number }> = [
{ pattern: /\bgpt-5/, contextWindow: 272_000 },
{ pattern: /\b(?:gpt-4\.1|gpt-4o|o[134])\b/, contextWindow: 128_000 },
{ pattern: /\bclaude\b|\bclaude-/, contextWindow: 200_000 },
{ pattern: /\bgemini\b/, contextWindow: 1_048_576 },
{ pattern: /\bkimi\b|moonshot/, contextWindow: 256_000 },
{ pattern: /minimax/, contextWindow: 204_800 },
{ pattern: /\bglm-5(?:\.|\b)/, contextWindow: 1_000_000 },
{ pattern: /\bglm-4/, contextWindow: 200_000 },
const CONTEXT_WINDOW_RULES: ContextWindowRule[] = [
// ── OpenAI ──────────────────────────────────────────────────────────────
{ label: 'GPT-5.6 Luna (low-latency tier)', pattern: /\bgpt-5\.6-luna\b/, contextWindow: 272_000 },
{ label: 'GPT-5.6 Sol / Terra', pattern: /\bgpt-5\.6\b/, contextWindow: 1_050_000 },
{ label: 'GPT-5.5', pattern: /\bgpt-5\.5\b/, contextWindow: 1_000_000 },
{ label: 'GPT-5 lightweight variants', pattern: /\bgpt-5[\w.]*-(?:mini|nano|turbo)\b/, contextWindow: 272_000 },
{ label: 'GPT-5 flagship', pattern: /\bgpt-5\b/, contextWindow: 400_000 },
{ label: 'GPT-4.x and o-series', pattern: /\b(?:gpt-4\.1|gpt-4o|o[134])\b/, contextWindow: 128_000 },
// ── Anthropic ───────────────────────────────────────────────────────────
{ label: 'Claude Fable 5 / Opus 5 / Sonnet 5', pattern: /\bclaude-(?:fable|opus|sonnet)-5\b/, contextWindow: 1_000_000 },
{ label: 'Claude Opus 4.8+', pattern: /\bclaude-opus-4[.-][89]\b/, contextWindow: 1_000_000 },
{ label: 'Claude Sonnet 4.6+', pattern: /\bclaude-sonnet-4[.-][6-9]\b/, contextWindow: 1_000_000 },
{ label: 'Claude Haiku and legacy Claude', pattern: /\bclaude\b|\bclaude-/, contextWindow: 200_000 },
// ── Google ──────────────────────────────────────────────────────────────
{ label: 'Gemini 1.0 (pre-million era)', pattern: /\bgemini-1\.0\b/, contextWindow: 32_768 },
{ label: 'Gemini 1.5 and newer', pattern: /\bgemini\b/, contextWindow: 1_048_576 },
// ── DeepSeek ────────────────────────────────────────────────────────────
// `deepseek-chat` / `deepseek-reasoner` are compatibility aliases that route
// to V4-Flash, so they inherit the V4 window rather than the V3 one.
{ label: 'DeepSeek V3 / R1', pattern: /\bdeepseek-(?:v3|r1)\b/, contextWindow: 128_000 },
{ label: 'DeepSeek V4 and aliases', pattern: /\bdeepseek\b/, contextWindow: 1_000_000 },
// ── Moonshot / Kimi ─────────────────────────────────────────────────────
// Only K3 reached a million tokens; K2.x tops out at 262,144.
{ label: 'Kimi K3', pattern: /\bkimi-k3\b/, contextWindow: 1_000_000 },
{ label: 'Kimi K2.x and other Moonshot', pattern: /\bkimi\b|moonshot/, contextWindow: 262_144 },
// ── Alibaba Qwen ────────────────────────────────────────────────────────
{ label: 'Qwen-Long (bulk document tier)', pattern: /\bqwen-long\b/, contextWindow: 10_000_000 },
{ label: 'Qwen 3.6+ hosted API', pattern: /\bqwen-?3\.[6-9]\b/, contextWindow: 1_000_000 },
{ label: 'Qwen 3.5 / Qwen3-Next', pattern: /\bqwen-?3\.5\b|\bqwen3-next\b/, contextWindow: 262_144 },
{ label: 'Qwen open-weight base', pattern: /\bqwen/, contextWindow: 131_072 },
// ── Z.AI GLM ────────────────────────────────────────────────────────────
{ label: 'GLM-5.2+', pattern: /\bglm-5\.[2-9]\b/, contextWindow: 1_000_000 },
{ label: 'GLM-5.0 / 5.1', pattern: /\bglm-5(?:\.[01])?\b/, contextWindow: 200_000 },
{ label: 'GLM-4.x', pattern: /\bglm-4/, contextWindow: 200_000 },
// ── MiniMax ─────────────────────────────────────────────────────────────
{ label: 'MiniMax M3+', pattern: /\bminimax-m[3-9]\b/, contextWindow: 524_288 },
{ label: 'MiniMax M2.x and earlier', pattern: /minimax/, contextWindow: 204_800 },
];
/** Safe floor for unknown custom models: high enough to avoid compaction spam. */
export const DEFAULT_CUSTOM_MODEL_CONTEXT_WINDOW = 131_072;
/**
* Fallback for hosted models we do not recognise. Set at the low end of the
* current frontier rather than at the old 128K floor: nearly every model a
* user can point a hosted provider at at this point clears 200K, and guessing
* too low triggers needless compaction on long sessions.
*/
export const DEFAULT_CUSTOM_MODEL_CONTEXT_WINDOW = 200_000;
export function inferCustomModelContextWindow(modelId: string): number {
const normalized = modelId.trim().toLowerCase();
for (const rule of CUSTOM_MODEL_CONTEXT_WINDOW_RULES) {
if (rule.pattern.test(normalized)) return rule.contextWindow;
}
return DEFAULT_CUSTOM_MODEL_CONTEXT_WINDOW;
/**
* Ceiling for locally hosted runtimes (Ollama and friends). A local `qwen3`
* tag is a quantised small model, not the hosted flagship of the same name, so
* family rules must not hand it a frontier-sized window. Kept at 128K because
* ClawX seeds `compaction.reserveTokensFloor = 50000` — dropping the ceiling
* near or below that floor leaves the runner no usable budget.
*/
export const LOCAL_MODEL_CONTEXT_WINDOW = 131_072;
/**
* Ceiling for ChatGPT subscription transports (`openai-chatgpt-responses`).
*
* OAuth against a ChatGPT plan does not get the API-tier window: the backend
* enforces a far smaller per-session budget than `gpt-5.6-sol`'s published
* 1.05M. OpenClaw's own Codex catalog hard-codes 272,000 for every model on
* this transport, so we mirror that figure rather than inventing our own.
*
* This matters because ClawX writes OAuth rows into `models.providers.openai`
* while OpenClaw's cap lives on its separate `codex` provider — nothing else
* would clamp the value we write.
*/
export const CHATGPT_OAUTH_CONTEXT_WINDOW = 272_000;
/** Runtime provider keys are suffixed per instance, e.g. `ollama-a1b2c3`. */
const LOCAL_PROVIDER_KEY_PATTERN = /^ollama(?:-|$)/;
/** Current and legacy spellings of the ChatGPT subscription transport. */
const SUBSCRIPTION_API_PROTOCOLS = new Set([
'openai-chatgpt-responses',
'openai-codex-responses',
]);
export type ModelCapabilityContext = {
/** OpenClaw runtime provider key, used to detect locally hosted models. */
providerKey?: string;
/** `models.providers.*.api` value, used to detect subscription transports. */
apiProtocol?: string;
};
/**
* Model ids reach us in several shapes: bare (`gpt-5.6-sol`), vendor-prefixed
* from aggregators (`openai/gpt-5.6-sol`, `deepseek-ai/DeepSeek-V3`), and
* Ollama-tagged (`qwen3:latest`). Patterns are written against the bare family
* name, so expose both forms and let callers test each.
*/
function normalizeModelId(modelId: string): { bare: string; full: string } {
const full = modelId.trim().toLowerCase();
const withoutVendor = full.slice(full.lastIndexOf('/') + 1);
const [bare] = withoutVendor.split(':');
return { bare: bare || full, full };
}
function matchesModelId(pattern: RegExp, modelId: string): boolean {
const { bare, full } = normalizeModelId(modelId);
return pattern.test(bare) || pattern.test(full);
}
function isLocalProviderKey(providerKey: string | undefined): boolean {
return providerKey != null && LOCAL_PROVIDER_KEY_PATTERN.test(providerKey.trim().toLowerCase());
}
function isSubscriptionApiProtocol(apiProtocol: string | undefined): boolean {
return apiProtocol != null && SUBSCRIPTION_API_PROTOCOLS.has(apiProtocol.trim().toLowerCase());
}
/**
* Family rules describe what the vendor's API tier offers. The transport a
* given account actually uses can be far more restrictive, so clamp rather
* than trusting the published figure.
*/
function resolveContextWindowCeiling(context: ModelCapabilityContext): number {
const ceilings: number[] = [];
if (isLocalProviderKey(context.providerKey)) ceilings.push(LOCAL_MODEL_CONTEXT_WINDOW);
if (isSubscriptionApiProtocol(context.apiProtocol)) ceilings.push(CHATGPT_OAUTH_CONTEXT_WINDOW);
return ceilings.length > 0 ? Math.min(...ceilings) : Number.POSITIVE_INFINITY;
}
export function inferCustomModelContextWindow(
modelId: string,
context: ModelCapabilityContext = {},
): number {
const ceiling = resolveContextWindowCeiling(context);
for (const rule of CONTEXT_WINDOW_RULES) {
if (matchesModelId(rule.pattern, modelId)) return Math.min(rule.contextWindow, ceiling);
}
return Math.min(DEFAULT_CUSTOM_MODEL_CONTEXT_WINDOW, ceiling);
}
const VISION_MODEL_PATTERNS: RegExp[] = [
/\b(?:gpt-4o|gpt-4\.1|gpt-[5-9]|o[134])\b/,
/\bclaude-(?:3|4|fable|sonnet|opus|haiku)\b/,
/\bgemini\b/,
/\b(?:qwen[\w.-]*-?vl|qwen-vl)\b/,
/\b(?:vision|llava|pixtral|internvl|mllama|minicpm-v|glm-4v)\b/,
/(?:^|[-_/])vl(?:[-_/]|$)/,
];
/**
* Mirrors OpenClaw 2026.5.20 custom-provider onboarding inference.
* Unknown models use the same conservative text-only fallback as non-interactive onboarding.
*/
export function inferCustomModelInputModalities(modelId: string): ModelInputModality[] {
const normalized = modelId.trim().toLowerCase();
const supportsImageInput = (
/\b(?:gpt-4o|gpt-4\.1|gpt-[5-9]|o[134])\b/.test(normalized)
|| /\bclaude-(?:3|4|sonnet|opus|haiku)\b/.test(normalized)
|| /\bgemini\b/.test(normalized)
|| /\b(?:qwen[\w.-]*-?vl|qwen-vl)\b/.test(normalized)
|| /\b(?:vision|llava|pixtral|internvl|mllama|minicpm-v|glm-4v)\b/.test(normalized)
|| /(?:^|[-_/])vl(?:[-_/]|$)/.test(normalized)
);
const supportsImageInput = VISION_MODEL_PATTERNS.some((pattern) => matchesModelId(pattern, modelId));
return supportsImageInput ? ['text', 'image'] : ['text'];
}
+2 -2
View File
@@ -138,7 +138,7 @@ export const PROVIDER_DEFINITIONS: ProviderDefinition[] = [
reasoning: false,
input: ['text'],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 256000,
contextWindow: 262144,
maxTokens: 8192,
},
],
@@ -171,7 +171,7 @@ export const PROVIDER_DEFINITIONS: ProviderDefinition[] = [
reasoning: false,
input: ['text'],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 256000,
contextWindow: 262144,
maxTokens: 8192,
},
],
+12 -3
View File
@@ -926,7 +926,10 @@ function backfillCustomProviderModelContextWindows(config: Record<string, unknow
for (const row of rows) {
if (!isPlainRecord(row) || typeof row.id !== 'string' || !row.id) continue;
if (typeof row.contextWindow === 'number' || typeof row.contextTokens === 'number') continue;
row.contextWindow = inferCustomModelContextWindow(row.id);
row.contextWindow = inferCustomModelContextWindow(row.id, {
providerKey,
apiProtocol: typeof entry.api === 'string' ? entry.api : undefined,
});
backfilled.push(`${providerKey}/${row.id}`);
}
}
@@ -1919,7 +1922,10 @@ function upsertOpenClawProviderEntry(
input: inferCustomModelInputModalities(id),
// Without an explicit contextWindow OpenClaw cannot budget compaction
// for custom providers and long sessions die with context overflow.
contextWindow: inferCustomModelContextWindow(id),
contextWindow: inferCustomModelContextWindow(id, {
providerKey: provider,
apiProtocol: options.api,
}),
}
: {}),
}));
@@ -2819,7 +2825,10 @@ async function updateModelsJsonProviderEntriesForAgents(
&& typeof base.contextWindow !== 'number'
&& typeof base.contextTokens !== 'number'
) {
base.contextWindow = inferCustomModelContextWindow(m.id);
base.contextWindow = inferCustomModelContextWindow(m.id, {
providerKey: providerType,
apiProtocol: entry.api,
});
}
return {
...base,
+2 -2
View File
@@ -985,7 +985,7 @@ describe('syncProviderConfigToOpenClaw', () => {
expect(models).toEqual([
expect.objectContaining({
id: 'gpt-5.5',
contextWindow: 272000,
contextWindow: 1000000,
}),
]);
});
@@ -2497,7 +2497,7 @@ describe('batchSyncConfigFields', () => {
const custom = (providers['custom-enterpri'] as Record<string, unknown>).models as Array<Record<string, unknown>>;
const moonshot = (providers.moonshot as Record<string, unknown>).models as Array<Record<string, unknown>>;
expect(custom[0]).toEqual(expect.objectContaining({ id: 'gpt-5.5', contextWindow: 272000 }));
expect(custom[0]).toEqual(expect.objectContaining({ id: 'gpt-5.5', contextWindow: 1000000 }));
// Rows with explicit contextTokens are user-owned — leave untouched.
expect(custom[1].contextWindow).toBeUndefined();
expect(custom[1].contextTokens).toBe(32000);
+120 -6
View File
@@ -1,7 +1,9 @@
import { describe, expect, it } from 'vitest';
import {
CHATGPT_OAUTH_CONTEXT_WINDOW,
DEFAULT_CUSTOM_MODEL_CONTEXT_WINDOW,
LOCAL_MODEL_CONTEXT_WINDOW,
inferCustomModelContextWindow,
inferCustomModelInputModalities,
} from '@electron/shared/providers/model-capabilities';
@@ -9,10 +11,13 @@ import {
describe('inferCustomModelInputModalities', () => {
it.each([
'gpt-4o',
'gpt-5.6-sol',
'claude-opus-4-6',
'claude-fable-5',
'gemini-3-flash',
'qwen2.5-vl',
'glm-4v',
'openai/gpt-5.6-sol',
])('marks known vision model %s as image-capable', (modelId) => {
expect(inferCustomModelInputModalities(modelId)).toEqual(['text', 'image']);
});
@@ -29,21 +34,130 @@ describe('inferCustomModelInputModalities', () => {
describe('inferCustomModelContextWindow', () => {
it.each([
['gpt-5.5', 272_000],
// OpenAI: generation-specific rules must win over the bare gpt-5 family.
['gpt-5.6-sol', 1_050_000],
['gpt-5.6-terra', 1_050_000],
['gpt-5.6-luna', 272_000],
['gpt-5.5', 1_000_000],
['GPT-5.4-Mini', 272_000],
['gpt-5', 400_000],
['gpt-4o', 128_000],
// Anthropic
['claude-fable-5', 1_000_000],
['claude-opus-4-8', 1_000_000],
['claude-sonnet-4-6', 1_000_000],
['claude-opus-4-6', 200_000],
['gemini-3-flash', 1_048_576],
['kimi-k2.6', 256_000],
['MiniMax-M3', 204_800],
['claude-haiku-4-5', 200_000],
// Google
['gemini-3.1-pro-preview', 1_048_576],
['gemini-1.0-pro', 32_768],
// DeepSeek: V4 and its aliases are 1M, V3 is not.
['deepseek-v4-flash', 1_000_000],
['deepseek-v4-pro', 1_000_000],
['deepseek-chat', 1_000_000],
['deepseek-v3', 128_000],
// Moonshot: only K3 reached a million tokens.
['kimi-k3', 1_000_000],
['kimi-k2.6', 262_144],
// Qwen
['qwen-long', 10_000_000],
['qwen3.6-plus', 1_000_000],
['qwen3.5-397b', 262_144],
['qwen3-next-80b', 262_144],
// Z.AI GLM — mirrors the explicit rows in the provider registry.
['glm-5.2', 1_000_000],
['glm-5.1', 1_000_000],
['glm-5.1', 200_000],
['glm-4.7', 200_000],
// MiniMax
['MiniMax-M3', 524_288],
['MiniMax-M2.7', 204_800],
])('maps known family %s to %d tokens', (modelId, expected) => {
expect(inferCustomModelContextWindow(modelId)).toBe(expected);
});
it('falls back to the conservative default for unknown models', () => {
it.each([
['openai/gpt-5.6-sol', 1_050_000],
['deepseek-ai/DeepSeek-V3', 128_000],
['moonshotai/kimi-k3', 1_000_000],
])('resolves the family behind vendor-prefixed id %s', (modelId, expected) => {
expect(inferCustomModelContextWindow(modelId)).toBe(expected);
});
it('falls back to the frontier-era default for unknown models', () => {
expect(inferCustomModelContextWindow('unknown-private-model')).toBe(DEFAULT_CUSTOM_MODEL_CONTEXT_WINDOW);
expect(DEFAULT_CUSTOM_MODEL_CONTEXT_WINDOW).toBe(200_000);
});
describe('locally hosted providers', () => {
it('caps family inference so a local tag cannot inherit a frontier window', () => {
expect(inferCustomModelContextWindow('deepseek-v4-flash', { providerKey: 'ollama-a1b2c3' }))
.toBe(LOCAL_MODEL_CONTEXT_WINDOW);
});
it('strips the Ollama tag before matching the family', () => {
expect(inferCustomModelContextWindow('qwen3:latest', { providerKey: 'ollama-a1b2c3' }))
.toBe(131_072);
});
it('keeps a window smaller than the local ceiling', () => {
expect(inferCustomModelContextWindow('gpt-4o', { providerKey: 'ollama-a1b2c3' }))
.toBe(128_000);
});
it('does not cap hosted providers', () => {
expect(inferCustomModelContextWindow('deepseek-v4-flash', { providerKey: 'deepseek' }))
.toBe(1_000_000);
});
});
describe('ChatGPT subscription transport', () => {
it.each([
'openai-chatgpt-responses',
'openai-codex-responses',
])('caps API-tier windows on %s', (apiProtocol) => {
expect(inferCustomModelContextWindow('gpt-5.6-sol', { providerKey: 'openai', apiProtocol }))
.toBe(CHATGPT_OAUTH_CONTEXT_WINDOW);
expect(inferCustomModelContextWindow('gpt-5.5', { providerKey: 'openai', apiProtocol }))
.toBe(CHATGPT_OAUTH_CONTEXT_WINDOW);
});
it('leaves the API-key transport at the published window', () => {
expect(inferCustomModelContextWindow('gpt-5.6-sol', {
providerKey: 'openai',
apiProtocol: 'openai-responses',
})).toBe(1_050_000);
});
it('keeps a window already below the subscription ceiling', () => {
expect(inferCustomModelContextWindow('gpt-4o', {
providerKey: 'openai',
apiProtocol: 'openai-chatgpt-responses',
})).toBe(128_000);
});
it('caps the unknown-model default too', () => {
expect(inferCustomModelContextWindow('some-internal-preview', {
providerKey: 'openai',
apiProtocol: 'openai-chatgpt-responses',
})).toBe(DEFAULT_CUSTOM_MODEL_CONTEXT_WINDOW);
expect(DEFAULT_CUSTOM_MODEL_CONTEXT_WINDOW).toBeLessThan(CHATGPT_OAUTH_CONTEXT_WINDOW);
});
});
describe('MiniMax OAuth', () => {
// Device OAuth hits the same platform API as a key, so no transport cap.
it('keeps the platform window for oauth-backed MiniMax', () => {
expect(inferCustomModelContextWindow('MiniMax-M3', {
providerKey: 'minimax',
apiProtocol: 'anthropic-messages',
})).toBe(524_288);
});
});
});