Compare commits

...
Author SHA1 Message Date
4b9948250b feat(engine): add DeepSeek as a first-class cloud provider + fix #335 over-permissive cloud fallback (#545)
* feat(engine): add DeepSeek as a first-class cloud provider

Adds DEEPSEEK_API_KEY support to the cloud engine, wiring DeepSeek's
OpenAI-compatible API (api.deepseek.com/v1) alongside the existing
MiniMax, OpenRouter, Anthropic, and Google providers.

- Add _DEEPSEEK_MODELS list (deepseek-v4-flash, deepseek-v4-pro)
- Add _is_deepseek_model() routing predicate
- Init self._deepseek_client from DEEPSEEK_API_KEY in _init_clients()
- Add _generate_deepseek() and _stream_deepseek() methods
- Wire DeepSeek into generate(), stream(), _stream_full_openai(),
  list_models(), and health()
- Add approximate pricing entries for both models

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>

* fix(engine): strict cloud model routing + deepseek can_serve branch

Builds on the DeepSeek provider (PR #504) with two routing-correctness
fixes to CloudEngine._client_for_model:

1. Add the missing DeepSeek branch so can_serve('deepseek-*') agrees with
   list_models()/health() when only DEEPSEEK_API_KEY is set (mirrors the
   minimax branch). Without it the engine advertised deepseek models via
   list_models() but refused to serve them (the #532 can_serve contract).

2. Fix #335: _client_for_model previously fell through to the OpenAI client
   for ANY unrecognized model name, so an OpenAI key (even a dummy
   sk-dummy... one) made can_serve('qwen3.5:0.8b') return True. With the
   local engine transiently down (classic post-Windows-restart Ollama not
   yet up), model-aware get_engine then mis-selected the cloud engine for a
   local model and died with "OpenAI client not available". Add a positive
   _is_openai_model predicate (gpt-/chatgpt-/o1/o3/o4 + _OPENAI_MODELS) and
   return None for unrecognized names, so can_serve declines them. generate()
   and stream() keep their OpenAI fall-through, preserving loud failure for an
   explicitly-requested unknown cloud model.

Tests: DeepSeek detection/pricing/health/list_models/generate-routing/
can_serve and a #335 regression (can_serve rejects local names with an
OpenAI key; unknown model not served even with all clients set; end-to-end
get_engine does not misroute a local model with a dummy OpenAI key).

Fixes #335

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>

---------

Co-authored-by: Jen Huls <me@jenhuls.com>
Co-authored-by: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-14 18:21:56 -07:00
3 changed files with 366 additions and 4 deletions
+166 -4
View File
@@ -1,4 +1,7 @@
"""Cloud inference engine — OpenAI, Anthropic, Google, and MiniMax API backends."""
"""Cloud inference engine.
OpenAI, Anthropic, Google, MiniMax, and DeepSeek API backends.
"""
from __future__ import annotations
@@ -48,6 +51,8 @@ PRICING: Dict[str, tuple[float, float]] = {
"MiniMax-M2.7-highspeed": (0.60, 2.40),
"MiniMax-M2.5": (0.30, 1.20),
"MiniMax-M2.5-highspeed": (0.60, 2.40),
"deepseek-v4-flash": (0.27, 1.10),
"deepseek-v4-pro": (0.55, 2.19),
}
# Well-known model IDs per provider
@@ -83,6 +88,10 @@ _MINIMAX_MODELS = [
"MiniMax-M2.5",
"MiniMax-M2.5-highspeed",
]
_DEEPSEEK_MODELS = [
"deepseek-v4-flash",
"deepseek-v4-pro",
]
# OpenRouter models — prefixed with "openrouter/" so they can be identified
_OPENROUTER_POPULAR = [
@@ -111,6 +120,10 @@ def _is_minimax_model(model: str) -> bool:
return model.lower().startswith("minimax")
def _is_deepseek_model(model: str) -> bool:
return model.lower().startswith("deepseek")
def _is_openrouter_model(model: str) -> bool:
return model.startswith("openrouter/")
@@ -127,6 +140,35 @@ def _is_google_model(model: str) -> bool:
return "gemini" in model.lower() and not _is_openrouter_model(model)
# Positive prefix predicate for genuine OpenAI models. Kept in sync with
# ``server/cloud_router.py:_OPENAI_PREFIXES`` so local-vs-cloud classification
# agrees across the codebase. Used by ``_client_for_model``/``can_serve`` so the
# cloud engine never claims it can serve an unrecognized (e.g. local Ollama)
# model name just because an OpenAI key happens to be present (see #335).
_OPENAI_PREFIXES = ("gpt-", "chatgpt-", "o1", "o3", "o4")
def _is_openai_model(model: str) -> bool:
"""True only for genuine OpenAI models (gpt-*, chatgpt-*, o1/o3/o4 series).
Defined positively so that an unrecognized model name (a local Ollama model
like ``qwen3.5:0.8b``, or a typo) is NOT treated as an OpenAI model. This is
the routing surface ``can_serve`` relies on; ``generate``/``stream`` keep
their OpenAI fall-through so an explicitly-requested unknown cloud model
still errors loudly at call time.
Caveat: a user may repoint the OpenAI client at an OpenAI-compatible server
(vLLM/LM Studio) via ``OPENAI_BASE_URL`` and legitimately serve non-gpt
names. That path is undocumented/untested in this engine; if it is added,
this predicate (or ``_client_for_model``) should treat a configured custom
base_url as "serves anything".
"""
m = model.lower()
if m in (name.lower() for name in _OPENAI_MODELS):
return True
return m.startswith(_OPENAI_PREFIXES)
def _is_openai_reasoning_model(model: str) -> bool:
"""Check if model is an OpenAI reasoning model that restricts temperature."""
m = model.lower()
@@ -269,7 +311,7 @@ def _convert_tools_to_google(
@EngineRegistry.register("cloud")
class CloudEngine(InferenceEngine):
"""Cloud inference via OpenAI, Anthropic, Google, and MiniMax SDKs."""
"""Cloud inference via OpenAI, Anthropic, Google, MiniMax, and DeepSeek SDKs."""
engine_id = "cloud"
is_cloud = True
@@ -280,6 +322,7 @@ class CloudEngine(InferenceEngine):
self._google_client: Any = None
self._openrouter_client: Any = None
self._minimax_client: Any = None
self._deepseek_client: Any = None
self._codex_client: Any = None
# Gemini thought_signatures: tool_call_id -> signature bytes
self._thought_sigs: Dict[str, bytes] = {}
@@ -332,6 +375,17 @@ class CloudEngine(InferenceEngine):
)
except ImportError:
pass
deepseek_key = os.environ.get("DEEPSEEK_API_KEY")
if deepseek_key:
try:
import openai
self._deepseek_client = openai.OpenAI(
base_url="https://api.deepseek.com/v1",
api_key=deepseek_key,
)
except ImportError:
pass
# Codex — uses the OpenAI Responses API.
# Supports both standard API keys (api.openai.com) and ChatGPT
# OAuth tokens (chatgpt.com) via OPENAI_CODEX_BASE_URL override.
@@ -985,6 +1039,56 @@ class CloudEngine(InferenceEngine):
]
return result
def _generate_deepseek(
self,
messages: Sequence[Message],
*,
model: str,
temperature: float,
max_tokens: int,
**kwargs: Any,
) -> Dict[str, Any]:
if self._deepseek_client is None:
raise EngineConnectionError(
"DeepSeek client not available — set DEEPSEEK_API_KEY"
)
kwargs.pop("response_format", None)
create_kwargs: Dict[str, Any] = {
"model": model,
"messages": messages_to_dicts(messages),
"max_tokens": max_tokens,
"temperature": temperature,
}
t0 = time.monotonic()
resp = self._deepseek_client.chat.completions.create(**create_kwargs)
elapsed = time.monotonic() - t0
choice = resp.choices[0]
usage = resp.usage
prompt_tokens = usage.prompt_tokens if usage else 0
completion_tokens = usage.completion_tokens if usage else 0
result: Dict[str, Any] = {
"content": choice.message.content or "",
"usage": {
"prompt_tokens": prompt_tokens,
"completion_tokens": completion_tokens,
"total_tokens": (usage.total_tokens if usage else 0),
},
"model": resp.model,
"finish_reason": choice.finish_reason or "stop",
"cost_usd": estimate_cost(model, prompt_tokens, completion_tokens),
"ttft": elapsed,
}
if hasattr(choice.message, "tool_calls") and choice.message.tool_calls:
result["tool_calls"] = [
{
"id": tc.id,
"name": tc.function.name,
"arguments": tc.function.arguments,
}
for tc in choice.message.tool_calls
]
return result
def generate(
self,
messages: Sequence[Message],
@@ -1006,6 +1110,8 @@ class CloudEngine(InferenceEngine):
return self._generate_openrouter(messages, **kw)
if _is_minimax_model(model):
return self._generate_minimax(messages, **kw)
if _is_deepseek_model(model):
return self._generate_deepseek(messages, **kw)
if _is_anthropic_model(model):
return self._generate_anthropic(messages, **kw)
if _is_google_model(model):
@@ -1036,6 +1142,9 @@ class CloudEngine(InferenceEngine):
elif _is_minimax_model(model):
async for token in self._stream_minimax(messages, **kw):
yield token
elif _is_deepseek_model(model):
async for token in self._stream_deepseek(messages, **kw):
yield token
elif _is_anthropic_model(model):
async for token in self._stream_anthropic(messages, **kw):
yield token
@@ -1254,6 +1363,30 @@ class CloudEngine(InferenceEngine):
if delta and delta.content:
yield delta.content
async def _stream_deepseek(
self,
messages: Sequence[Message],
*,
model: str,
temperature: float,
max_tokens: int,
**kwargs: Any,
) -> AsyncIterator[str]:
if self._deepseek_client is None:
raise EngineConnectionError("DeepSeek client not available")
create_kwargs: Dict[str, Any] = {
"model": model,
"messages": messages_to_dicts(messages),
"max_tokens": max_tokens,
"temperature": temperature,
"stream": True,
}
resp = self._deepseek_client.chat.completions.create(**create_kwargs)
for chunk in resp:
delta = chunk.choices[0].delta if chunk.choices else None
if delta and delta.content:
yield delta.content
# -- stream_full: rich streaming with tool_calls support ----------------
async def _stream_full_openai(
@@ -1307,6 +1440,18 @@ class CloudEngine(InferenceEngine):
"stream": True,
**kwargs,
}
elif _is_deepseek_model(model):
client = self._deepseek_client
if client is None:
raise EngineConnectionError("DeepSeek client not available")
create_kwargs = {
"model": model,
"messages": messages_to_dicts(messages),
"max_tokens": max_tokens,
"temperature": temperature,
"stream": True,
**kwargs,
}
else:
client = self._openai_client
if client is None:
@@ -1473,24 +1618,40 @@ class CloudEngine(InferenceEngine):
models.extend(_OPENROUTER_POPULAR)
if self._minimax_client is not None:
models.extend(_MINIMAX_MODELS)
if self._deepseek_client is not None:
models.extend(_DEEPSEEK_MODELS)
if self._codex_client is not None:
models.extend(_CODEX_MODELS)
return models
def _client_for_model(self, model: str) -> Any:
"""Return the provider client ``generate``/``stream`` will dispatch to
for *model* (mirrors the routing in those methods)."""
for *model*, or ``None`` for a model this engine cannot route.
Mirrors the routing in ``generate``/``stream``, but is intentionally
*stricter* on the OpenAI fall-through: only genuine OpenAI models map to
the OpenAI client. Unrecognized names (e.g. a local Ollama model like
``qwen3.5:0.8b``) return ``None`` so ``can_serve`` declines them and the
cloud engine is not mis-selected as a fallback when the local engine is
transiently down and any (even dummy) ``OPENAI_API_KEY`` is set (#335).
``generate``/``stream`` keep their OpenAI fall-through, so an
explicitly-requested unknown cloud model still fails loudly at call time.
"""
if _is_codex_model(model):
return self._codex_client
if _is_openrouter_model(model):
return self._openrouter_client
if _is_minimax_model(model):
return self._minimax_client
if _is_deepseek_model(model):
return self._deepseek_client
if _is_anthropic_model(model):
return self._anthropic_client
if _is_google_model(model):
return self._google_client
return self._openai_client
if _is_openai_model(model):
return self._openai_client
return None
def can_serve(self, model: str) -> bool:
"""Return ``True`` only if the provider client for *model* exists.
@@ -1512,6 +1673,7 @@ class CloudEngine(InferenceEngine):
or self._google_client is not None
or self._openrouter_client is not None
or self._minimax_client is not None
or self._deepseek_client is not None
or self._codex_client is not None
)
+152
View File
@@ -9,9 +9,13 @@ import pytest
from openjarvis.core.registry import EngineRegistry
from openjarvis.core.types import Message, Role
from openjarvis.engine._base import EngineConnectionError
from openjarvis.engine.cloud import (
CloudEngine,
_is_codex_model,
_is_deepseek_model,
_is_openai_model,
_is_openrouter_model,
estimate_cost,
)
@@ -457,6 +461,7 @@ class TestCloudEngineCanServe:
"_google_client",
"_openrouter_client",
"_minimax_client",
"_deepseek_client",
"_codex_client",
):
setattr(eng, name, clients.get(name))
@@ -469,7 +474,154 @@ class TestCloudEngineCanServe:
assert eng.can_serve("gemini-2.5-pro") is False
assert eng.can_serve("openrouter/openai/gpt-4o") is False
def test_openai_key_does_not_claim_local_models(self) -> None:
"""#335: with only the OpenAI client set (e.g. a present-but-dummy
OPENAI_API_KEY), the cloud engine must NOT claim it can serve a local
Ollama model name — otherwise it gets mis-selected as a fallback when
the local engine is transiently down and dies with "OpenAI client not
available". Only genuine OpenAI models route to the OpenAI client.
"""
eng = self._engine(_openai_client=object())
# Local Ollama / unrecognized names are NOT served by the cloud engine.
assert eng.can_serve("qwen3.5:0.8b") is False
assert eng.can_serve("llama3.2") is False
assert eng.can_serve("mistral") is False
assert eng.can_serve("phi3:mini") is False
assert eng.can_serve("some-unknown-model") is False
# Genuine OpenAI families still served.
assert eng.can_serve("gpt-4o") is True
assert eng.can_serve("gpt-5.4") is True
assert eng.can_serve("o3-mini") is True
def test_unknown_model_not_served_even_with_all_clients(self) -> None:
"""#335: an unrecognized model is declined regardless of how many
provider clients are configured — it never falls through to OpenAI."""
eng = self._engine(
_openai_client=object(),
_anthropic_client=object(),
_google_client=object(),
_minimax_client=object(),
_deepseek_client=object(),
)
assert eng.can_serve("qwen3.5:0.8b") is False
assert eng.can_serve("totally-made-up") is False
def test_anthropic_only_serves_anthropic_models(self) -> None:
eng = self._engine(_anthropic_client=object())
assert eng.can_serve("claude-sonnet-4") is True
assert eng.can_serve("gpt-4o") is False
def test_deepseek_only_serves_deepseek_models(self) -> None:
"""The DeepSeek client serves deepseek-* models (and only those)."""
eng = self._engine(_deepseek_client=object())
assert eng.can_serve("deepseek-v4-flash") is True
assert eng.can_serve("deepseek-v4-pro") is True
assert eng.can_serve("DeepSeek-V4-Pro") is True # case-insensitive
assert eng.can_serve("gpt-4o") is False
# OpenRouter-prefixed deepseek is NOT the direct DeepSeek provider.
assert eng.can_serve("openrouter/deepseek/deepseek-r1") is False
class TestCloudEngineDeepSeek:
"""PR #504: DeepSeek as a first-class cloud provider (OpenAI-compatible)."""
def test_is_deepseek_model_predicate(self) -> None:
assert _is_deepseek_model("deepseek-v4-flash") is True
assert _is_deepseek_model("deepseek-v4-pro") is True
assert _is_deepseek_model("DeepSeek-V4-Pro") is True # case-insensitive
assert _is_deepseek_model("gpt-4o") is False
# No predicate collision: openrouter/deepseek/* belongs to OpenRouter.
assert _is_deepseek_model("openrouter/deepseek/deepseek-r1") is False
assert _is_openrouter_model("openrouter/deepseek/deepseek-r1") is True
# And a deepseek name is not mistaken for an OpenAI model.
assert _is_openai_model("deepseek-v4-pro") is False
def test_pricing_entries_present(self) -> None:
assert estimate_cost("deepseek-v4-flash", 1_000_000, 1_000_000) == (
pytest.approx(1.37) # 0.27 + 1.10
)
assert estimate_cost("deepseek-v4-pro", 1_000_000, 1_000_000) == (
pytest.approx(2.74) # 0.55 + 2.19
)
def test_init_wires_deepseek_client(self, monkeypatch: pytest.MonkeyPatch) -> None:
"""DEEPSEEK_API_KEY builds an openai client pointed at api.deepseek.com."""
for var in ("OPENAI_API_KEY", "ANTHROPIC_API_KEY"):
monkeypatch.delenv(var, raising=False)
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-deepseek-test")
fake_openai = mock.MagicMock()
with mock.patch.dict("sys.modules", {"openai": fake_openai}):
EngineRegistry.register_value("cloud", CloudEngine)
engine = CloudEngine()
fake_openai.OpenAI.assert_any_call(
base_url="https://api.deepseek.com/v1",
api_key="sk-deepseek-test",
)
assert engine._deepseek_client is not None
def test_health_and_list_models_gated_on_deepseek_key(
self, monkeypatch: pytest.MonkeyPatch
) -> None:
for var in ("OPENAI_API_KEY", "ANTHROPIC_API_KEY"):
monkeypatch.delenv(var, raising=False)
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-deepseek-test")
fake_openai = mock.MagicMock()
with mock.patch.dict("sys.modules", {"openai": fake_openai}):
EngineRegistry.register_value("cloud", CloudEngine)
engine = CloudEngine()
assert engine.health() is True
models = engine.list_models()
assert "deepseek-v4-flash" in models
assert "deepseek-v4-pro" in models
# can_serve must agree with list_models (regression for the missing
# _client_for_model deepseek branch flagged by the #504 verifier).
assert engine.can_serve("deepseek-v4-pro") is True
assert engine.can_serve("deepseek-v4-flash") is True
def test_generate_routes_to_deepseek_client(
self, monkeypatch: pytest.MonkeyPatch
) -> None:
for var in ("OPENAI_API_KEY", "ANTHROPIC_API_KEY", "DEEPSEEK_API_KEY"):
monkeypatch.delenv(var, raising=False)
fake_usage = SimpleNamespace(
prompt_tokens=7, completion_tokens=3, total_tokens=10
)
fake_choice = SimpleNamespace(
message=SimpleNamespace(content="ds-hello"),
finish_reason="stop",
)
fake_resp = SimpleNamespace(
choices=[fake_choice], usage=fake_usage, model="deepseek-v4-pro"
)
fake_client = mock.MagicMock()
fake_client.chat.completions.create.return_value = fake_resp
EngineRegistry.register_value("cloud", CloudEngine)
engine = CloudEngine()
engine._deepseek_client = fake_client
result = engine.generate(
[Message(role=Role.USER, content="Hi")], model="deepseek-v4-pro"
)
assert result["content"] == "ds-hello"
assert result["usage"]["prompt_tokens"] == 7
# Routed to the DeepSeek client, not OpenAI.
fake_client.chat.completions.create.assert_called_once()
def test_generate_without_client_raises(
self, monkeypatch: pytest.MonkeyPatch
) -> None:
for var in ("OPENAI_API_KEY", "ANTHROPIC_API_KEY", "DEEPSEEK_API_KEY"):
monkeypatch.delenv(var, raising=False)
EngineRegistry.register_value("cloud", CloudEngine)
engine = CloudEngine()
assert engine._deepseek_client is None
with pytest.raises(EngineConnectionError):
engine.generate(
[Message(role=Role.USER, content="Hi")], model="deepseek-v4-pro"
)
+48
View File
@@ -204,6 +204,54 @@ class TestGetEngine:
assert result is not None
assert result[0] == "local"
def test_dummy_openai_key_does_not_misroute_local_model(
self, monkeypatch: object
) -> None:
"""#335: a present-but-dummy OPENAI_API_KEY + a down local engine must
NOT cause a local Ollama model to be routed to the cloud engine.
Before the fix, CloudEngine.can_serve('qwen3.5:0.8b') returned True
whenever any OpenAI client existed (even a junk key), so get_engine
picked 'cloud' and the request later died with "OpenAI client not
available". With the strict _client_for_model fall-through it returns
None for unrecognized names, so get_engine declines cloud and (with the
local engine down) returns None — surfacing a "start your local engine"
failure instead.
"""
from openjarvis.engine.cloud import CloudEngine
_reg("ollama", "ollama")
EngineRegistry.register_value("cloud", CloudEngine)
cfg = JarvisConfig()
cfg.engine.default = "ollama"
def _make(k, c): # noqa: ANN001
if k == "ollama":
# Local engine is down (post-restart Ollama not yet up).
return _FakeEngine(healthy=False, models=["qwen3.5:0.8b"])
# Real CloudEngine with only a (dummy) OpenAI client wired.
eng = CloudEngine.__new__(CloudEngine)
for name in (
"_openai_client",
"_anthropic_client",
"_google_client",
"_openrouter_client",
"_minimax_client",
"_deepseek_client",
"_codex_client",
):
setattr(eng, name, object() if name == "_openai_client" else None)
return eng
with mock.patch(
"openjarvis.engine._discovery._make_engine",
side_effect=_make,
):
result = get_engine(cfg, model="qwen3.5:0.8b")
# Cloud must NOT be selected for a local model name.
assert result is None
def test_model_none_preserves_model_agnostic_selection(self) -> None:
"""model=None keeps the legacy behaviour: first healthy engine wins."""
_reg("primary", "primary")