mirror of
https://github.com/open-jarvis/OpenJarvis.git
synced 2026-08-15 09:21:56 +00:00
Compare commits
8
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8ef1ab1928 | ||
|
|
28e75cb513 | ||
|
|
4b9948250b | ||
|
|
79e23719d4 | ||
|
|
7ba334b5f0 | ||
|
|
48a2627c9a | ||
|
|
8625f4f95f | ||
|
|
cf08f164c0 |
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"schemaVersion": 1,
|
||||
"label": "Git Clones",
|
||||
"message": "110,366",
|
||||
"message": "117,047",
|
||||
"color": "green",
|
||||
"namedLogo": "git"
|
||||
}
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"total_clones": 110366,
|
||||
"last_updated": "2026-06-11T07:44:18Z",
|
||||
"total_clones": 117047,
|
||||
"last_updated": "2026-06-14T07:38:47Z",
|
||||
"daily": {
|
||||
"2026-03-27": 2189,
|
||||
"2026-03-28": 1874,
|
||||
@@ -77,6 +77,9 @@
|
||||
"2026-06-07": 1174,
|
||||
"2026-06-08": 2369,
|
||||
"2026-06-09": 1361,
|
||||
"2026-06-10": 1310
|
||||
"2026-06-10": 1310,
|
||||
"2026-06-11": 2564,
|
||||
"2026-06-12": 1313,
|
||||
"2026-06-13": 2804
|
||||
}
|
||||
}
|
||||
|
||||
@@ -11,22 +11,33 @@ concurrency:
|
||||
group: claude-issues-${{ github.event.issue.number || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
# Least-privilege: only what the issue-fixer job actually needs.
|
||||
# id-token (OIDC) is intentionally omitted — claude-code-action@v1 is passed
|
||||
# github_token directly, so OIDC is unused here.
|
||||
permissions:
|
||||
contents: write
|
||||
pull-requests: write
|
||||
issues: write
|
||||
id-token: write
|
||||
|
||||
jobs:
|
||||
fix:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 60
|
||||
timeout-minutes: 15
|
||||
# Security gate: this job reaches secrets.ANTHROPIC_API_KEY and holds a
|
||||
# write-scoped GITHUB_TOKEN. `issues` / `issue_comment` are public,
|
||||
# attacker-controllable events that run in the base-repo context with full
|
||||
# secret access, so the human-triggered paths are restricted to actors with
|
||||
# write-level association (OWNER / MEMBER / COLLABORATOR). This blocks
|
||||
# external / first-time contributors from draining the API budget or
|
||||
# creating branches/PRs, while leaving maintainer use unaffected.
|
||||
if: |
|
||||
github.event_name == 'workflow_dispatch' ||
|
||||
(github.event_name == 'issues' &&
|
||||
contains(fromJSON('["OWNER", "MEMBER", "COLLABORATOR"]'), github.event.issue.author_association) &&
|
||||
(contains(github.event.issue.labels.*.name, 'bug') ||
|
||||
contains(github.event.issue.labels.*.name, 'autofix'))) ||
|
||||
(github.event_name == 'issue_comment' &&
|
||||
contains(fromJSON('["OWNER", "MEMBER", "COLLABORATOR"]'), github.event.comment.author_association) &&
|
||||
!github.event.issue.pull_request &&
|
||||
contains(github.event.comment.body, '@claude') &&
|
||||
github.actor != 'claude[bot]')
|
||||
|
||||
@@ -11,23 +11,33 @@ concurrency:
|
||||
group: claude-review-${{ github.event.pull_request.number || github.event.issue.number || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
# Least-privilege: PR review only needs to post comments on the PR.
|
||||
# id-token (OIDC) is omitted — claude-code-action@v1 is passed github_token
|
||||
# directly, so OIDC is unused here.
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: write
|
||||
issues: write
|
||||
id-token: write
|
||||
|
||||
jobs:
|
||||
review:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
# Security gate: this job reaches secrets.ANTHROPIC_API_KEY. Both
|
||||
# issue_comment and pull_request_review_comment are public,
|
||||
# attacker-controllable events that run in the base-repo context with full
|
||||
# secret access, so the @claude paths are restricted to actors with
|
||||
# write-level association (OWNER / MEMBER / COLLABORATOR). External /
|
||||
# first-time contributors cannot trigger the key; maintainers are unaffected.
|
||||
if: |
|
||||
github.event_name == 'workflow_dispatch' ||
|
||||
(github.event_name == 'issue_comment' &&
|
||||
contains(fromJSON('["OWNER", "MEMBER", "COLLABORATOR"]'), github.event.comment.author_association) &&
|
||||
github.event.issue.pull_request &&
|
||||
contains(github.event.comment.body, '@claude') &&
|
||||
github.actor != 'claude[bot]') ||
|
||||
(github.event_name == 'pull_request_review_comment' &&
|
||||
contains(fromJSON('["OWNER", "MEMBER", "COLLABORATOR"]'), github.event.comment.author_association) &&
|
||||
contains(github.event.comment.body, '@claude') &&
|
||||
github.actor != 'claude[bot]')
|
||||
steps:
|
||||
|
||||
@@ -8,6 +8,19 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
**Vision input for `jarvis ask`** — attach images to a query with
|
||||
`-i`/`--image` (repeatable) or capture the current screen with
|
||||
`-S`/`--screen`, for vision-capable models such as `gemma3:4b`. Images flow
|
||||
through `Message.images` into Ollama's `/api/chat` `images` field; text-only
|
||||
requests are unaffected. A privacy guard warns before any image is sent to a
|
||||
non-local engine, and the security guardrail now preserves images when it
|
||||
sanitizes a flagged prompt. Screen capture uses the built-in Windows .NET
|
||||
stack with `mss`/`Pillow` fallbacks on other platforms. Adds the
|
||||
`JARVIS_NUM_CTX` environment variable to tune the Ollama context window
|
||||
(default `16384`).
|
||||
|
||||
## [1.0.2] - 2026-05-24
|
||||
|
||||
A patch release that fixes a packaging bug which broke the v1.0.1
|
||||
|
||||
@@ -66,6 +66,8 @@ jarvis ask "What is the capital of France?"
|
||||
| `--no-context` | flag | off | Disable memory context injection |
|
||||
| `-a`, `--agent AGENT` | string | none | Agent to use (`simple`, `orchestrator`) |
|
||||
| `--tools TOOLS` | string | none | Comma-separated tool names to enable |
|
||||
| `-i`, `--image PATH` | path | none | Image file for a vision model (e.g. `gemma3:4b`); repeatable |
|
||||
| `-S`, `--screen` | flag | off | Capture the current screen and send it to the vision model |
|
||||
|
||||
### Direct Mode vs Agent Mode
|
||||
|
||||
@@ -105,6 +107,39 @@ jarvis ask --no-context "Tell me about Python"
|
||||
jarvis ask --max-tokens 2048 "Write a detailed essay about AI"
|
||||
```
|
||||
|
||||
### Vision Input
|
||||
|
||||
Vision-capable models (such as `gemma3:4b`) can read images alongside your
|
||||
text prompt. Attach one or more image files with `-i`/`--image`, or capture
|
||||
the current screen with `-S`/`--screen`:
|
||||
|
||||
```bash
|
||||
# Ask about a local image
|
||||
jarvis ask -i screenshot.png "What is shown in this image?"
|
||||
|
||||
# Send multiple images (the flag is repeatable)
|
||||
jarvis ask -i chart-a.png -i chart-b.png "Compare these two charts"
|
||||
|
||||
# Capture the current screen and ask about it
|
||||
jarvis ask --screen "Summarize what's on my screen"
|
||||
```
|
||||
|
||||
Vision runs in **direct mode** only. If you also pass `--agent`, the image is
|
||||
ignored and a note is printed — re-run with `--agent ""` to force direct mode.
|
||||
|
||||
The Ollama context window can be tuned for large images or long prompts with
|
||||
the `JARVIS_NUM_CTX` environment variable (default `16384`):
|
||||
|
||||
```bash
|
||||
JARVIS_NUM_CTX=8192 jarvis ask --screen "What's on my screen?"
|
||||
```
|
||||
|
||||
!!! note "Keep vision on-device"
|
||||
Images are sensitive. OpenJarvis prints a privacy warning before sending
|
||||
an image to a non-local engine, so a screenshot never leaves your machine
|
||||
unnoticed. Use a local engine (e.g. `ollama` with `gemma3:4b`) to keep
|
||||
vision fully local.
|
||||
|
||||
### JSON Output Format
|
||||
|
||||
When using `--json` in **direct mode**, the output includes:
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import type { ResearchEvent, SSEEvent } from '../types';
|
||||
import { getBase } from './api';
|
||||
import { getBase, authHeaders } from './api';
|
||||
|
||||
export interface ChatRequest {
|
||||
model: string;
|
||||
@@ -16,7 +16,7 @@ export async function* streamChat(
|
||||
const base = getBase();
|
||||
const response = await fetch(`${base}/v1/chat/completions`, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
headers: authHeaders({ 'Content-Type': 'application/json' }),
|
||||
body: JSON.stringify(request),
|
||||
signal,
|
||||
});
|
||||
@@ -67,7 +67,7 @@ export async function* streamResearch(
|
||||
const base = getBase().replace(/\/v1\/?$/, '');
|
||||
const response = await fetch(`${base}/api/research`, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
headers: authHeaders({ 'Content-Type': 'application/json' }),
|
||||
body: JSON.stringify({ query }),
|
||||
signal,
|
||||
});
|
||||
@@ -106,3 +106,4 @@ export async function* streamResearch(
|
||||
reader.releaseLock();
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,79 @@
|
||||
"""Screen capture for vision input (``jarvis ask --screen``).
|
||||
|
||||
Captures the primary monitor to a temporary PNG so it can be handed to a
|
||||
vision-capable model. On Windows this uses the built-in .NET
|
||||
``System.Drawing`` stack (no third-party dependency). Other platforms fall
|
||||
back to ``mss`` or ``Pillow`` if installed.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
|
||||
# PowerShell: capture the PRIMARY monitor (more legible for a vision model
|
||||
# than a downscaled multi-monitor grab). {path} is filled in with forward
|
||||
# slashes, which .NET accepts on Windows and which avoids backslash escaping.
|
||||
_PS_CAPTURE = """
|
||||
Add-Type -AssemblyName System.Windows.Forms, System.Drawing
|
||||
$b = [System.Windows.Forms.Screen]::PrimaryScreen.Bounds
|
||||
$bmp = New-Object System.Drawing.Bitmap($b.Width, $b.Height)
|
||||
$g = [System.Drawing.Graphics]::FromImage($bmp)
|
||||
$g.CopyFromScreen($b.X, $b.Y, 0, 0, $bmp.Size)
|
||||
$bmp.Save("{path}", [System.Drawing.Imaging.ImageFormat]::Png)
|
||||
$g.Dispose(); $bmp.Dispose()
|
||||
"""
|
||||
|
||||
|
||||
def capture_screen_to_temp() -> str:
|
||||
"""Capture the screen to a temp PNG and return its absolute path.
|
||||
|
||||
Raises ``RuntimeError`` with actionable guidance if capture fails or the
|
||||
platform has no available backend.
|
||||
"""
|
||||
fd, path = tempfile.mkstemp(prefix="jarvis_screen_", suffix=".png")
|
||||
os.close(fd)
|
||||
|
||||
if sys.platform.startswith("win"):
|
||||
script = _PS_CAPTURE.replace("{path}", path.replace("\\", "/"))
|
||||
proc = subprocess.run(
|
||||
["powershell", "-NoProfile", "-NonInteractive", "-Command", script],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=30,
|
||||
)
|
||||
if (
|
||||
proc.returncode != 0
|
||||
or not os.path.exists(path)
|
||||
or not os.path.getsize(path)
|
||||
):
|
||||
raise RuntimeError(
|
||||
"screen capture failed: "
|
||||
+ (proc.stderr.strip() or "empty image written")
|
||||
)
|
||||
return path
|
||||
|
||||
# Non-Windows: optional backends.
|
||||
try:
|
||||
import mss # type: ignore
|
||||
|
||||
with mss.mss() as sct:
|
||||
sct.shot(mon=-1, output=path)
|
||||
return path
|
||||
except ImportError:
|
||||
pass
|
||||
try:
|
||||
from PIL import ImageGrab # type: ignore
|
||||
|
||||
ImageGrab.grab().save(path)
|
||||
return path
|
||||
except Exception as exc: # noqa: BLE001
|
||||
raise RuntimeError(
|
||||
"screen capture on this platform needs 'mss' or 'Pillow' "
|
||||
"(try: pip install mss)"
|
||||
) from exc
|
||||
|
||||
|
||||
__all__ = ["capture_screen_to_temp"]
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import json as json_mod
|
||||
import logging
|
||||
import sys
|
||||
@@ -619,6 +620,21 @@ def _print_profile(
|
||||
"(default: ~/.openjarvis/knowledge.db)."
|
||||
),
|
||||
)
|
||||
@click.option(
|
||||
"-i",
|
||||
"--image",
|
||||
"image_paths",
|
||||
multiple=True,
|
||||
type=click.Path(exists=True, dir_okay=False),
|
||||
help="Image file for a vision model (e.g. gemma3). Repeatable.",
|
||||
)
|
||||
@click.option(
|
||||
"-S",
|
||||
"--screen",
|
||||
"capture_screen",
|
||||
is_flag=True,
|
||||
help="Capture the current screen and send it to the vision model.",
|
||||
)
|
||||
@click.option(
|
||||
"--persona",
|
||||
"persona_name",
|
||||
@@ -645,6 +661,8 @@ def ask(
|
||||
research_mode: bool,
|
||||
knowledge_db: str | None,
|
||||
persona_name: str | None,
|
||||
image_paths: tuple[str, ...] = (),
|
||||
capture_screen: bool = False,
|
||||
) -> None:
|
||||
"""Ask Jarvis a question."""
|
||||
quiet = (ctx.obj or {}).get("quiet", False) or output_json
|
||||
@@ -652,6 +670,27 @@ def ask(
|
||||
console = Console(stderr=True)
|
||||
query_text = " ".join(query)
|
||||
|
||||
# Vision: collect base64 images from --image files and/or --screen.
|
||||
image_b64: list[str] = []
|
||||
for _img_path in image_paths:
|
||||
try:
|
||||
with open(_img_path, "rb") as _fh:
|
||||
image_b64.append(base64.b64encode(_fh.read()).decode("ascii"))
|
||||
except OSError as exc:
|
||||
console.print(f"[red]Could not read image {_img_path}: {exc}[/red]")
|
||||
sys.exit(1)
|
||||
if capture_screen:
|
||||
try:
|
||||
from openjarvis.cli._screen import capture_screen_to_temp
|
||||
|
||||
_shot = capture_screen_to_temp()
|
||||
with open(_shot, "rb") as _fh:
|
||||
image_b64.append(base64.b64encode(_fh.read()).decode("ascii"))
|
||||
logger.debug("Captured screen to %s", _shot)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
console.print(f"[red]Screen capture failed:[/red] {exc}")
|
||||
sys.exit(1)
|
||||
|
||||
wall_start = time.monotonic() if enable_profile else None
|
||||
|
||||
# Load config
|
||||
@@ -671,11 +710,26 @@ def ask(
|
||||
# Without this fallback, `[agent].default_system_prompt` and the
|
||||
# SOUL.md / MEMORY.md / USER.md persona system are silently bypassed for
|
||||
# the most common command (`jarvis ask "..."`).
|
||||
agent_explicitly_set = agent_name is not None
|
||||
if agent_name is None:
|
||||
configured_default = (config.agent.default_agent or "").strip()
|
||||
if configured_default:
|
||||
agent_name = configured_default
|
||||
|
||||
# Vision flows only through direct-to-engine mode. If an image/screenshot
|
||||
# was supplied without an explicit --agent, route to direct mode so the
|
||||
# picture reaches the model; if an agent was explicitly requested, say
|
||||
# plainly that the image is being skipped rather than dropping it silently.
|
||||
if image_b64:
|
||||
if not agent_explicitly_set:
|
||||
agent_name = ""
|
||||
else:
|
||||
console.print(
|
||||
"[yellow]Note:[/yellow] --image/--screen only works in direct "
|
||||
"mode; the image is ignored with --agent set. Re-run with "
|
||||
'`--agent ""` to use vision.'
|
||||
)
|
||||
|
||||
# Track whether the user explicitly set --max-tokens
|
||||
user_set_max_tokens = max_tokens is not None
|
||||
|
||||
@@ -871,6 +925,27 @@ def ask(
|
||||
return
|
||||
|
||||
# Direct-to-engine mode (no agent)
|
||||
# Privacy guard: a screenshot/image is sensitive, and OpenJarvis is
|
||||
# local-first. If the active engine isn't local, warn before the image
|
||||
# leaves the machine rather than silently uploading it to a third party.
|
||||
_LOCAL_ENGINES = {
|
||||
"ollama",
|
||||
"llamacpp",
|
||||
"vllm",
|
||||
"sglang",
|
||||
"exo",
|
||||
"nexa",
|
||||
"uzu",
|
||||
"apple_fm",
|
||||
"gemma_cpp",
|
||||
}
|
||||
if image_b64 and engine_name not in _LOCAL_ENGINES:
|
||||
console.print(
|
||||
f"[yellow]Privacy warning:[/yellow] sending {len(image_b64)} "
|
||||
f"image(s) to a non-local engine ('{engine_name}'). The image will "
|
||||
"leave this machine. Use a local engine (e.g. ollama) to keep "
|
||||
"vision on-device."
|
||||
)
|
||||
messages = [Message(role=Role.USER, content=query_text)]
|
||||
|
||||
# Memory-augmented context injection
|
||||
@@ -897,6 +972,15 @@ def ask(
|
||||
except Exception as exc:
|
||||
logger.debug("Failed to inject memory context: %s", exc)
|
||||
|
||||
# Vision: attach images to the final user message *after* any context
|
||||
# injection (which may rebuild the list). messages_to_dicts() forwards
|
||||
# the "images" field to Ollama's /api/chat.
|
||||
if image_b64:
|
||||
for _m in reversed(messages):
|
||||
if _m.role == Role.USER:
|
||||
_m.images = image_b64
|
||||
break
|
||||
|
||||
# Generate (InstrumentedEngine handles telemetry + energy recording)
|
||||
try:
|
||||
with console.status("[bold green]Generating...[/bold green]"):
|
||||
|
||||
@@ -946,8 +946,10 @@ class AgentConfig:
|
||||
system_prompt_path: str = "" # path to system prompt file (.txt, .md)
|
||||
context_from_memory: bool = True # inject relevant memory context into prompts
|
||||
default_system_prompt: str = (
|
||||
"You are a helpful AI assistant running locally on the user's own "
|
||||
"hardware through OpenJarvis. You are not a cloud service. Respond "
|
||||
"You are OpenJarvis, a helpful AI assistant running locally on the "
|
||||
"user's own hardware. You are not a cloud service, and you are not "
|
||||
"Claude, ChatGPT, Gemini, or any other branded assistant. If asked "
|
||||
"who or what you are, identify yourself as OpenJarvis. Respond "
|
||||
"helpfully, concisely, and accurately."
|
||||
)
|
||||
|
||||
|
||||
@@ -68,6 +68,10 @@ class Message:
|
||||
tool_calls: Optional[List[ToolCall]] = None
|
||||
tool_call_id: Optional[str] = None
|
||||
metadata: Dict[str, Any] = field(default_factory=dict)
|
||||
# Base64-encoded image data for vision-capable models (e.g. gemma3,
|
||||
# qwen2.5-vl). Forwarded to Ollama's /api/chat "images" field; None or
|
||||
# empty for text-only messages (the common case).
|
||||
images: Optional[List[str]] = None
|
||||
|
||||
|
||||
@dataclass(slots=True)
|
||||
|
||||
@@ -34,6 +34,10 @@ def messages_to_dicts(messages: Sequence[Message]) -> List[Dict[str, Any]]:
|
||||
]
|
||||
if m.tool_call_id:
|
||||
d["tool_call_id"] = m.tool_call_id
|
||||
# Vision: forward base64 images to the engine. Ollama's /api/chat
|
||||
# accepts an "images" array on a message; text messages skip this.
|
||||
if getattr(m, "images", None):
|
||||
d["images"] = list(m.images)
|
||||
out.append(d)
|
||||
return out
|
||||
|
||||
|
||||
@@ -1,4 +1,7 @@
|
||||
"""Cloud inference engine — OpenAI, Anthropic, Google, and MiniMax API backends."""
|
||||
"""Cloud inference engine.
|
||||
|
||||
OpenAI, Anthropic, Google, MiniMax, and DeepSeek API backends.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -48,6 +51,8 @@ PRICING: Dict[str, tuple[float, float]] = {
|
||||
"MiniMax-M2.7-highspeed": (0.60, 2.40),
|
||||
"MiniMax-M2.5": (0.30, 1.20),
|
||||
"MiniMax-M2.5-highspeed": (0.60, 2.40),
|
||||
"deepseek-v4-flash": (0.27, 1.10),
|
||||
"deepseek-v4-pro": (0.55, 2.19),
|
||||
}
|
||||
|
||||
# Well-known model IDs per provider
|
||||
@@ -83,6 +88,10 @@ _MINIMAX_MODELS = [
|
||||
"MiniMax-M2.5",
|
||||
"MiniMax-M2.5-highspeed",
|
||||
]
|
||||
_DEEPSEEK_MODELS = [
|
||||
"deepseek-v4-flash",
|
||||
"deepseek-v4-pro",
|
||||
]
|
||||
|
||||
# OpenRouter models — prefixed with "openrouter/" so they can be identified
|
||||
_OPENROUTER_POPULAR = [
|
||||
@@ -111,6 +120,10 @@ def _is_minimax_model(model: str) -> bool:
|
||||
return model.lower().startswith("minimax")
|
||||
|
||||
|
||||
def _is_deepseek_model(model: str) -> bool:
|
||||
return model.lower().startswith("deepseek")
|
||||
|
||||
|
||||
def _is_openrouter_model(model: str) -> bool:
|
||||
return model.startswith("openrouter/")
|
||||
|
||||
@@ -127,6 +140,35 @@ def _is_google_model(model: str) -> bool:
|
||||
return "gemini" in model.lower() and not _is_openrouter_model(model)
|
||||
|
||||
|
||||
# Positive prefix predicate for genuine OpenAI models. Kept in sync with
|
||||
# ``server/cloud_router.py:_OPENAI_PREFIXES`` so local-vs-cloud classification
|
||||
# agrees across the codebase. Used by ``_client_for_model``/``can_serve`` so the
|
||||
# cloud engine never claims it can serve an unrecognized (e.g. local Ollama)
|
||||
# model name just because an OpenAI key happens to be present (see #335).
|
||||
_OPENAI_PREFIXES = ("gpt-", "chatgpt-", "o1", "o3", "o4")
|
||||
|
||||
|
||||
def _is_openai_model(model: str) -> bool:
|
||||
"""True only for genuine OpenAI models (gpt-*, chatgpt-*, o1/o3/o4 series).
|
||||
|
||||
Defined positively so that an unrecognized model name (a local Ollama model
|
||||
like ``qwen3.5:0.8b``, or a typo) is NOT treated as an OpenAI model. This is
|
||||
the routing surface ``can_serve`` relies on; ``generate``/``stream`` keep
|
||||
their OpenAI fall-through so an explicitly-requested unknown cloud model
|
||||
still errors loudly at call time.
|
||||
|
||||
Caveat: a user may repoint the OpenAI client at an OpenAI-compatible server
|
||||
(vLLM/LM Studio) via ``OPENAI_BASE_URL`` and legitimately serve non-gpt
|
||||
names. That path is undocumented/untested in this engine; if it is added,
|
||||
this predicate (or ``_client_for_model``) should treat a configured custom
|
||||
base_url as "serves anything".
|
||||
"""
|
||||
m = model.lower()
|
||||
if m in (name.lower() for name in _OPENAI_MODELS):
|
||||
return True
|
||||
return m.startswith(_OPENAI_PREFIXES)
|
||||
|
||||
|
||||
def _is_openai_reasoning_model(model: str) -> bool:
|
||||
"""Check if model is an OpenAI reasoning model that restricts temperature."""
|
||||
m = model.lower()
|
||||
@@ -269,7 +311,7 @@ def _convert_tools_to_google(
|
||||
|
||||
@EngineRegistry.register("cloud")
|
||||
class CloudEngine(InferenceEngine):
|
||||
"""Cloud inference via OpenAI, Anthropic, Google, and MiniMax SDKs."""
|
||||
"""Cloud inference via OpenAI, Anthropic, Google, MiniMax, and DeepSeek SDKs."""
|
||||
|
||||
engine_id = "cloud"
|
||||
is_cloud = True
|
||||
@@ -280,6 +322,7 @@ class CloudEngine(InferenceEngine):
|
||||
self._google_client: Any = None
|
||||
self._openrouter_client: Any = None
|
||||
self._minimax_client: Any = None
|
||||
self._deepseek_client: Any = None
|
||||
self._codex_client: Any = None
|
||||
# Gemini thought_signatures: tool_call_id -> signature bytes
|
||||
self._thought_sigs: Dict[str, bytes] = {}
|
||||
@@ -332,6 +375,17 @@ class CloudEngine(InferenceEngine):
|
||||
)
|
||||
except ImportError:
|
||||
pass
|
||||
deepseek_key = os.environ.get("DEEPSEEK_API_KEY")
|
||||
if deepseek_key:
|
||||
try:
|
||||
import openai
|
||||
|
||||
self._deepseek_client = openai.OpenAI(
|
||||
base_url="https://api.deepseek.com/v1",
|
||||
api_key=deepseek_key,
|
||||
)
|
||||
except ImportError:
|
||||
pass
|
||||
# Codex — uses the OpenAI Responses API.
|
||||
# Supports both standard API keys (api.openai.com) and ChatGPT
|
||||
# OAuth tokens (chatgpt.com) via OPENAI_CODEX_BASE_URL override.
|
||||
@@ -985,6 +1039,56 @@ class CloudEngine(InferenceEngine):
|
||||
]
|
||||
return result
|
||||
|
||||
def _generate_deepseek(
|
||||
self,
|
||||
messages: Sequence[Message],
|
||||
*,
|
||||
model: str,
|
||||
temperature: float,
|
||||
max_tokens: int,
|
||||
**kwargs: Any,
|
||||
) -> Dict[str, Any]:
|
||||
if self._deepseek_client is None:
|
||||
raise EngineConnectionError(
|
||||
"DeepSeek client not available — set DEEPSEEK_API_KEY"
|
||||
)
|
||||
kwargs.pop("response_format", None)
|
||||
create_kwargs: Dict[str, Any] = {
|
||||
"model": model,
|
||||
"messages": messages_to_dicts(messages),
|
||||
"max_tokens": max_tokens,
|
||||
"temperature": temperature,
|
||||
}
|
||||
t0 = time.monotonic()
|
||||
resp = self._deepseek_client.chat.completions.create(**create_kwargs)
|
||||
elapsed = time.monotonic() - t0
|
||||
choice = resp.choices[0]
|
||||
usage = resp.usage
|
||||
prompt_tokens = usage.prompt_tokens if usage else 0
|
||||
completion_tokens = usage.completion_tokens if usage else 0
|
||||
result: Dict[str, Any] = {
|
||||
"content": choice.message.content or "",
|
||||
"usage": {
|
||||
"prompt_tokens": prompt_tokens,
|
||||
"completion_tokens": completion_tokens,
|
||||
"total_tokens": (usage.total_tokens if usage else 0),
|
||||
},
|
||||
"model": resp.model,
|
||||
"finish_reason": choice.finish_reason or "stop",
|
||||
"cost_usd": estimate_cost(model, prompt_tokens, completion_tokens),
|
||||
"ttft": elapsed,
|
||||
}
|
||||
if hasattr(choice.message, "tool_calls") and choice.message.tool_calls:
|
||||
result["tool_calls"] = [
|
||||
{
|
||||
"id": tc.id,
|
||||
"name": tc.function.name,
|
||||
"arguments": tc.function.arguments,
|
||||
}
|
||||
for tc in choice.message.tool_calls
|
||||
]
|
||||
return result
|
||||
|
||||
def generate(
|
||||
self,
|
||||
messages: Sequence[Message],
|
||||
@@ -1006,6 +1110,8 @@ class CloudEngine(InferenceEngine):
|
||||
return self._generate_openrouter(messages, **kw)
|
||||
if _is_minimax_model(model):
|
||||
return self._generate_minimax(messages, **kw)
|
||||
if _is_deepseek_model(model):
|
||||
return self._generate_deepseek(messages, **kw)
|
||||
if _is_anthropic_model(model):
|
||||
return self._generate_anthropic(messages, **kw)
|
||||
if _is_google_model(model):
|
||||
@@ -1036,6 +1142,9 @@ class CloudEngine(InferenceEngine):
|
||||
elif _is_minimax_model(model):
|
||||
async for token in self._stream_minimax(messages, **kw):
|
||||
yield token
|
||||
elif _is_deepseek_model(model):
|
||||
async for token in self._stream_deepseek(messages, **kw):
|
||||
yield token
|
||||
elif _is_anthropic_model(model):
|
||||
async for token in self._stream_anthropic(messages, **kw):
|
||||
yield token
|
||||
@@ -1254,6 +1363,30 @@ class CloudEngine(InferenceEngine):
|
||||
if delta and delta.content:
|
||||
yield delta.content
|
||||
|
||||
async def _stream_deepseek(
|
||||
self,
|
||||
messages: Sequence[Message],
|
||||
*,
|
||||
model: str,
|
||||
temperature: float,
|
||||
max_tokens: int,
|
||||
**kwargs: Any,
|
||||
) -> AsyncIterator[str]:
|
||||
if self._deepseek_client is None:
|
||||
raise EngineConnectionError("DeepSeek client not available")
|
||||
create_kwargs: Dict[str, Any] = {
|
||||
"model": model,
|
||||
"messages": messages_to_dicts(messages),
|
||||
"max_tokens": max_tokens,
|
||||
"temperature": temperature,
|
||||
"stream": True,
|
||||
}
|
||||
resp = self._deepseek_client.chat.completions.create(**create_kwargs)
|
||||
for chunk in resp:
|
||||
delta = chunk.choices[0].delta if chunk.choices else None
|
||||
if delta and delta.content:
|
||||
yield delta.content
|
||||
|
||||
# -- stream_full: rich streaming with tool_calls support ----------------
|
||||
|
||||
async def _stream_full_openai(
|
||||
@@ -1307,6 +1440,18 @@ class CloudEngine(InferenceEngine):
|
||||
"stream": True,
|
||||
**kwargs,
|
||||
}
|
||||
elif _is_deepseek_model(model):
|
||||
client = self._deepseek_client
|
||||
if client is None:
|
||||
raise EngineConnectionError("DeepSeek client not available")
|
||||
create_kwargs = {
|
||||
"model": model,
|
||||
"messages": messages_to_dicts(messages),
|
||||
"max_tokens": max_tokens,
|
||||
"temperature": temperature,
|
||||
"stream": True,
|
||||
**kwargs,
|
||||
}
|
||||
else:
|
||||
client = self._openai_client
|
||||
if client is None:
|
||||
@@ -1473,24 +1618,40 @@ class CloudEngine(InferenceEngine):
|
||||
models.extend(_OPENROUTER_POPULAR)
|
||||
if self._minimax_client is not None:
|
||||
models.extend(_MINIMAX_MODELS)
|
||||
if self._deepseek_client is not None:
|
||||
models.extend(_DEEPSEEK_MODELS)
|
||||
if self._codex_client is not None:
|
||||
models.extend(_CODEX_MODELS)
|
||||
return models
|
||||
|
||||
def _client_for_model(self, model: str) -> Any:
|
||||
"""Return the provider client ``generate``/``stream`` will dispatch to
|
||||
for *model* (mirrors the routing in those methods)."""
|
||||
for *model*, or ``None`` for a model this engine cannot route.
|
||||
|
||||
Mirrors the routing in ``generate``/``stream``, but is intentionally
|
||||
*stricter* on the OpenAI fall-through: only genuine OpenAI models map to
|
||||
the OpenAI client. Unrecognized names (e.g. a local Ollama model like
|
||||
``qwen3.5:0.8b``) return ``None`` so ``can_serve`` declines them and the
|
||||
cloud engine is not mis-selected as a fallback when the local engine is
|
||||
transiently down and any (even dummy) ``OPENAI_API_KEY`` is set (#335).
|
||||
``generate``/``stream`` keep their OpenAI fall-through, so an
|
||||
explicitly-requested unknown cloud model still fails loudly at call time.
|
||||
"""
|
||||
if _is_codex_model(model):
|
||||
return self._codex_client
|
||||
if _is_openrouter_model(model):
|
||||
return self._openrouter_client
|
||||
if _is_minimax_model(model):
|
||||
return self._minimax_client
|
||||
if _is_deepseek_model(model):
|
||||
return self._deepseek_client
|
||||
if _is_anthropic_model(model):
|
||||
return self._anthropic_client
|
||||
if _is_google_model(model):
|
||||
return self._google_client
|
||||
return self._openai_client
|
||||
if _is_openai_model(model):
|
||||
return self._openai_client
|
||||
return None
|
||||
|
||||
def can_serve(self, model: str) -> bool:
|
||||
"""Return ``True`` only if the provider client for *model* exists.
|
||||
@@ -1512,6 +1673,7 @@ class CloudEngine(InferenceEngine):
|
||||
or self._google_client is not None
|
||||
or self._openrouter_client is not None
|
||||
or self._minimax_client is not None
|
||||
or self._deepseek_client is not None
|
||||
or self._codex_client is not None
|
||||
)
|
||||
|
||||
|
||||
@@ -23,6 +23,19 @@ from openjarvis.engine._stubs import StreamChunk
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _default_num_ctx() -> int:
|
||||
"""Default context window (tokens). Override with ``JARVIS_NUM_CTX``.
|
||||
|
||||
Raised above Ollama's 4k default so an image (which costs many tokens)
|
||||
plus a real conversation fit. 16k is comfortable for small models on a
|
||||
typical consumer GPU.
|
||||
"""
|
||||
try:
|
||||
return int(os.environ.get("JARVIS_NUM_CTX", "16384"))
|
||||
except ValueError:
|
||||
return 16384
|
||||
|
||||
|
||||
@EngineRegistry.register("ollama")
|
||||
class OllamaEngine(InferenceEngine):
|
||||
"""Ollama backend via its native HTTP API."""
|
||||
@@ -73,7 +86,7 @@ class OllamaEngine(InferenceEngine):
|
||||
"options": {
|
||||
"temperature": temperature,
|
||||
"num_predict": max_tokens,
|
||||
"num_ctx": kwargs.get("num_ctx", 8192),
|
||||
"num_ctx": kwargs.get("num_ctx", _default_num_ctx()),
|
||||
},
|
||||
}
|
||||
# Disable extended thinking by default (Qwen3.5 etc.).
|
||||
@@ -189,7 +202,7 @@ class OllamaEngine(InferenceEngine):
|
||||
"options": {
|
||||
"temperature": temperature,
|
||||
"num_predict": max_tokens,
|
||||
"num_ctx": kwargs.get("num_ctx", 8192),
|
||||
"num_ctx": kwargs.get("num_ctx", _default_num_ctx()),
|
||||
},
|
||||
}
|
||||
# Mirror generate()'s default: disable extended thinking unless the
|
||||
@@ -268,7 +281,7 @@ class OllamaEngine(InferenceEngine):
|
||||
"options": {
|
||||
"temperature": temperature,
|
||||
"num_predict": max_tokens,
|
||||
"num_ctx": kwargs.get("num_ctx", 8192),
|
||||
"num_ctx": kwargs.get("num_ctx", _default_num_ctx()),
|
||||
},
|
||||
}
|
||||
if "think" not in kwargs:
|
||||
|
||||
@@ -192,6 +192,7 @@ class GuardrailsEngine(InferenceEngine):
|
||||
tool_calls=msg.tool_calls,
|
||||
tool_call_id=msg.tool_call_id,
|
||||
metadata=msg.metadata,
|
||||
images=msg.images,
|
||||
)
|
||||
messages = processed
|
||||
|
||||
|
||||
@@ -43,6 +43,51 @@ def _to_messages(chat_messages) -> list[Message]:
|
||||
return messages
|
||||
|
||||
|
||||
def _ensure_identity_prompt(messages: list[Message], app_config) -> list[Message]:
|
||||
"""Prepend OpenJarvis's identity system prompt when the client omits one.
|
||||
|
||||
The desktop UI's chat backend posts only user/assistant turns to
|
||||
``/v1/chat/completions`` (see ``frontend/.../Chat/InputArea.tsx``), so
|
||||
nothing grounds the model's identity. Without a system prompt the model
|
||||
answers from its training identity (e.g. "I'm Claude", "I am Qwen"),
|
||||
which is what #540 reported. The CLI paths inject this via
|
||||
``SystemPromptBuilder`` / ``BaseAgent``; the engine-direct server paths
|
||||
did not. This mirrors the agent fallback in ``agents/_stubs.py``.
|
||||
|
||||
If any message already carries a system role, the caller has supplied
|
||||
their own grounding and we leave the list untouched (no double-prompting).
|
||||
|
||||
Resolution of the identity text: ``app_config.agent.default_system_prompt``
|
||||
when a config is wired onto ``app.state``; otherwise fall back to
|
||||
``load_config()``. Config resolution is wrapped so a broken/missing
|
||||
config degrades to "no injection" rather than crashing the endpoint, but
|
||||
the failure is logged (per REVIEW.md — never silently swallow).
|
||||
"""
|
||||
if any(m.role == Role.SYSTEM for m in messages):
|
||||
return messages
|
||||
|
||||
prompt = ""
|
||||
try:
|
||||
if app_config is not None:
|
||||
prompt = app_config.agent.default_system_prompt or ""
|
||||
else:
|
||||
from openjarvis.core.config import load_config
|
||||
|
||||
prompt = load_config().agent.default_system_prompt or ""
|
||||
except Exception:
|
||||
logging.getLogger("openjarvis.server").debug(
|
||||
"Identity system prompt resolution failed; "
|
||||
"serving request without identity grounding",
|
||||
exc_info=True,
|
||||
)
|
||||
return messages
|
||||
|
||||
if not prompt:
|
||||
return messages
|
||||
|
||||
return [Message(role=Role.SYSTEM, content=prompt), *messages]
|
||||
|
||||
|
||||
@router.post("/v1/chat/completions")
|
||||
async def chat_completions(request_body: ChatCompletionRequest, request: Request):
|
||||
"""Handle chat completion requests (streaming and non-streaming)."""
|
||||
@@ -149,7 +194,7 @@ async def chat_completions(request_body: ChatCompletionRequest, request: Request
|
||||
# from the engine for true real-time output.
|
||||
if request_body.tools:
|
||||
return await _handle_stream_tools(
|
||||
engine, model, request_body, complexity_info
|
||||
engine, model, request_body, complexity_info, app_config=config
|
||||
)
|
||||
return await _handle_stream(
|
||||
engine,
|
||||
@@ -157,6 +202,7 @@ async def chat_completions(request_body: ChatCompletionRequest, request: Request
|
||||
request_body,
|
||||
complexity_info,
|
||||
trace_store=getattr(request.app.state, "trace_store", None),
|
||||
app_config=config,
|
||||
)
|
||||
|
||||
# Non-streaming: use agent if available, otherwise direct engine call.
|
||||
@@ -192,6 +238,7 @@ async def chat_completions(request_body: ChatCompletionRequest, request: Request
|
||||
request_body,
|
||||
bus=bus,
|
||||
complexity_info=complexity_info,
|
||||
app_config=config,
|
||||
)
|
||||
|
||||
|
||||
@@ -201,9 +248,11 @@ def _handle_direct(
|
||||
req: ChatCompletionRequest,
|
||||
bus=None,
|
||||
complexity_info=None,
|
||||
app_config=None,
|
||||
) -> ChatCompletionResponse:
|
||||
"""Direct engine call without agent."""
|
||||
messages = _to_messages(req.messages)
|
||||
messages = _ensure_identity_prompt(messages, app_config)
|
||||
kwargs: dict[str, Any] = {}
|
||||
if req.tools:
|
||||
kwargs["tools"] = req.tools
|
||||
@@ -380,6 +429,8 @@ async def _handle_stream_tools(
|
||||
model: str,
|
||||
req: ChatCompletionRequest,
|
||||
complexity_info=None,
|
||||
*,
|
||||
app_config=None,
|
||||
):
|
||||
"""Stream a raw OpenAI-compat function-calling response via SSE.
|
||||
|
||||
@@ -397,6 +448,7 @@ async def _handle_stream_tools(
|
||||
from openjarvis.server.cloud_router import is_cloud_model
|
||||
|
||||
messages = _to_messages(req.messages)
|
||||
messages = _ensure_identity_prompt(messages, app_config)
|
||||
chunk_id = f"chatcmpl-{uuid.uuid4().hex[:12]}"
|
||||
use_cloud = is_cloud_model(model)
|
||||
|
||||
@@ -491,6 +543,7 @@ async def _handle_stream(
|
||||
complexity_info=None,
|
||||
*,
|
||||
trace_store=None,
|
||||
app_config=None,
|
||||
):
|
||||
"""Stream response using SSE format.
|
||||
|
||||
@@ -509,6 +562,7 @@ async def _handle_stream(
|
||||
)
|
||||
|
||||
messages = _to_messages(req.messages)
|
||||
messages = _ensure_identity_prompt(messages, app_config)
|
||||
chunk_id = f"chatcmpl-{uuid.uuid4().hex[:12]}"
|
||||
|
||||
# Last user message — recorded as the trace query.
|
||||
|
||||
@@ -0,0 +1,128 @@
|
||||
"""CLI-level regression tests for ``jarvis ask`` vision input.
|
||||
|
||||
The unit tests in ``tests/test_vision.py`` cover the ``Message.images`` ->
|
||||
``messages_to_dicts`` serialization contract in isolation. These tests lock
|
||||
the *end-to-end CLI wiring*: that ``--image`` reads a file, base64-encodes it,
|
||||
attaches it to the final user ``Message``, and that the bytes actually reach
|
||||
``engine.generate()`` -- and that the local-first privacy guard fires only for
|
||||
non-local engines.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import importlib
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from click.testing import CliRunner
|
||||
|
||||
from openjarvis.cli import cli
|
||||
from openjarvis.core.config import JarvisConfig
|
||||
from openjarvis.core.types import Role
|
||||
|
||||
# Import the module (not the Click command attribute) so we can monkeypatch
|
||||
# the names it looks up at call time.
|
||||
_ask_mod = importlib.import_module("openjarvis.cli.ask")
|
||||
|
||||
# A minimal but valid 1x1 PNG so ``click.Path(exists=True)`` is satisfied and
|
||||
# the bytes are deterministic.
|
||||
_PNG_BYTES = base64.b64decode(
|
||||
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNk"
|
||||
"+M9QDwADhgGAWjR9awAAAABJRU5ErkJggg=="
|
||||
)
|
||||
|
||||
|
||||
class _RecordingEngine:
|
||||
"""A fake engine that records the messages handed to ``generate()``."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self.engine_id = "mock"
|
||||
self.received: list[Any] = []
|
||||
|
||||
def health(self) -> bool:
|
||||
return True
|
||||
|
||||
def list_models(self) -> list[str]:
|
||||
return ["test-model"]
|
||||
|
||||
def generate(self, messages, *, model=None, **kwargs):
|
||||
# Capture the exact Message objects the CLI built so the test can
|
||||
# assert the image bytes reached the engine boundary.
|
||||
self.received = list(messages)
|
||||
return {
|
||||
"content": "a 1x1 pixel",
|
||||
"usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2},
|
||||
"model": "test-model",
|
||||
"finish_reason": "stop",
|
||||
}
|
||||
|
||||
|
||||
def _patch_ask(monkeypatch, tmp_path: Path, *, engine_name: str) -> _RecordingEngine:
|
||||
"""Wire ``jarvis ask`` to a recording engine reported under ``engine_name``."""
|
||||
cfg = JarvisConfig()
|
||||
cfg.telemetry.db_path = str(tmp_path / "telemetry.db")
|
||||
# Keep memory context out of the picture so the user message we inspect is
|
||||
# the one the CLI built directly from the query + image.
|
||||
cfg.agent.context_from_memory = False
|
||||
monkeypatch.setattr(_ask_mod, "load_config", lambda: cfg)
|
||||
|
||||
engine = _RecordingEngine()
|
||||
monkeypatch.setattr(_ask_mod, "get_engine", lambda *a, **kw: (engine_name, engine))
|
||||
monkeypatch.setattr(_ask_mod, "discover_engines", lambda c: [(engine_name, engine)])
|
||||
monkeypatch.setattr(
|
||||
_ask_mod, "discover_models", lambda e: {engine_name: ["test-model"]}
|
||||
)
|
||||
return engine
|
||||
|
||||
|
||||
def _write_png(tmp_path: Path) -> tuple[Path, str]:
|
||||
img = tmp_path / "pixel.png"
|
||||
img.write_bytes(_PNG_BYTES)
|
||||
return img, base64.b64encode(_PNG_BYTES).decode("ascii")
|
||||
|
||||
|
||||
def test_image_reaches_engine_payload(monkeypatch, tmp_path: Path) -> None:
|
||||
engine = _patch_ask(monkeypatch, tmp_path, engine_name="ollama")
|
||||
img, expected_b64 = _write_png(tmp_path)
|
||||
|
||||
result = CliRunner().invoke(
|
||||
cli,
|
||||
["ask", "-i", str(img), "--no-context", "--agent", "", "describe this"],
|
||||
)
|
||||
|
||||
assert result.exit_code == 0, result.output
|
||||
# The CLI must have routed to direct mode and called the engine.
|
||||
assert engine.received, "engine.generate() was never called"
|
||||
user_msgs = [m for m in engine.received if m.role == Role.USER]
|
||||
assert user_msgs, "no USER message reached the engine"
|
||||
assert user_msgs[-1].images == [expected_b64]
|
||||
|
||||
|
||||
def test_privacy_warning_for_non_local_engine(monkeypatch, tmp_path: Path) -> None:
|
||||
engine = _patch_ask(monkeypatch, tmp_path, engine_name="openai")
|
||||
img, expected_b64 = _write_png(tmp_path)
|
||||
|
||||
result = CliRunner().invoke(
|
||||
cli,
|
||||
["ask", "-i", str(img), "--no-context", "--agent", "", "describe this"],
|
||||
)
|
||||
|
||||
assert result.exit_code == 0, result.output
|
||||
assert "Privacy warning" in result.output
|
||||
# The warning is informational; the image must still be delivered.
|
||||
user_msgs = [m for m in engine.received if m.role == Role.USER]
|
||||
assert user_msgs and user_msgs[-1].images == [expected_b64]
|
||||
|
||||
|
||||
def test_no_privacy_warning_for_local_engine(monkeypatch, tmp_path: Path) -> None:
|
||||
_patch_ask(monkeypatch, tmp_path, engine_name="ollama")
|
||||
img, _ = _write_png(tmp_path)
|
||||
|
||||
result = CliRunner().invoke(
|
||||
cli,
|
||||
["ask", "-i", str(img), "--no-context", "--agent", "", "describe this"],
|
||||
)
|
||||
|
||||
assert result.exit_code == 0, result.output
|
||||
assert "Privacy warning" not in result.output
|
||||
@@ -237,6 +237,14 @@ class TestAgentConfigNew:
|
||||
or isinstance(getattr(ac.__class__, "temperature", None), property) is False
|
||||
)
|
||||
|
||||
def test_default_system_prompt_anchors_identity(self) -> None:
|
||||
"""#540: the hardened wording must name OpenJarvis and explicitly
|
||||
deny the model's training identity so distilled models stop
|
||||
claiming to be Claude/ChatGPT/etc."""
|
||||
prompt = AgentConfig().default_system_prompt
|
||||
assert "OpenJarvis" in prompt
|
||||
assert "not Claude" in prompt
|
||||
|
||||
|
||||
class TestNestedEngineConfig:
|
||||
def test_nested_access(self) -> None:
|
||||
@@ -561,6 +569,7 @@ class TestWhatsAppBaileysChannelConfig:
|
||||
|
||||
def test_mining_config_absent_means_none(tmp_path):
|
||||
from openjarvis.core.config import load_config
|
||||
|
||||
cfg_path = tmp_path / "config.toml"
|
||||
cfg_path.write_text("") # empty config
|
||||
cfg = load_config(cfg_path)
|
||||
|
||||
@@ -9,9 +9,13 @@ import pytest
|
||||
|
||||
from openjarvis.core.registry import EngineRegistry
|
||||
from openjarvis.core.types import Message, Role
|
||||
from openjarvis.engine._base import EngineConnectionError
|
||||
from openjarvis.engine.cloud import (
|
||||
CloudEngine,
|
||||
_is_codex_model,
|
||||
_is_deepseek_model,
|
||||
_is_openai_model,
|
||||
_is_openrouter_model,
|
||||
estimate_cost,
|
||||
)
|
||||
|
||||
@@ -457,6 +461,7 @@ class TestCloudEngineCanServe:
|
||||
"_google_client",
|
||||
"_openrouter_client",
|
||||
"_minimax_client",
|
||||
"_deepseek_client",
|
||||
"_codex_client",
|
||||
):
|
||||
setattr(eng, name, clients.get(name))
|
||||
@@ -469,7 +474,154 @@ class TestCloudEngineCanServe:
|
||||
assert eng.can_serve("gemini-2.5-pro") is False
|
||||
assert eng.can_serve("openrouter/openai/gpt-4o") is False
|
||||
|
||||
def test_openai_key_does_not_claim_local_models(self) -> None:
|
||||
"""#335: with only the OpenAI client set (e.g. a present-but-dummy
|
||||
OPENAI_API_KEY), the cloud engine must NOT claim it can serve a local
|
||||
Ollama model name — otherwise it gets mis-selected as a fallback when
|
||||
the local engine is transiently down and dies with "OpenAI client not
|
||||
available". Only genuine OpenAI models route to the OpenAI client.
|
||||
"""
|
||||
eng = self._engine(_openai_client=object())
|
||||
# Local Ollama / unrecognized names are NOT served by the cloud engine.
|
||||
assert eng.can_serve("qwen3.5:0.8b") is False
|
||||
assert eng.can_serve("llama3.2") is False
|
||||
assert eng.can_serve("mistral") is False
|
||||
assert eng.can_serve("phi3:mini") is False
|
||||
assert eng.can_serve("some-unknown-model") is False
|
||||
# Genuine OpenAI families still served.
|
||||
assert eng.can_serve("gpt-4o") is True
|
||||
assert eng.can_serve("gpt-5.4") is True
|
||||
assert eng.can_serve("o3-mini") is True
|
||||
|
||||
def test_unknown_model_not_served_even_with_all_clients(self) -> None:
|
||||
"""#335: an unrecognized model is declined regardless of how many
|
||||
provider clients are configured — it never falls through to OpenAI."""
|
||||
eng = self._engine(
|
||||
_openai_client=object(),
|
||||
_anthropic_client=object(),
|
||||
_google_client=object(),
|
||||
_minimax_client=object(),
|
||||
_deepseek_client=object(),
|
||||
)
|
||||
assert eng.can_serve("qwen3.5:0.8b") is False
|
||||
assert eng.can_serve("totally-made-up") is False
|
||||
|
||||
def test_anthropic_only_serves_anthropic_models(self) -> None:
|
||||
eng = self._engine(_anthropic_client=object())
|
||||
assert eng.can_serve("claude-sonnet-4") is True
|
||||
assert eng.can_serve("gpt-4o") is False
|
||||
|
||||
def test_deepseek_only_serves_deepseek_models(self) -> None:
|
||||
"""The DeepSeek client serves deepseek-* models (and only those)."""
|
||||
eng = self._engine(_deepseek_client=object())
|
||||
assert eng.can_serve("deepseek-v4-flash") is True
|
||||
assert eng.can_serve("deepseek-v4-pro") is True
|
||||
assert eng.can_serve("DeepSeek-V4-Pro") is True # case-insensitive
|
||||
assert eng.can_serve("gpt-4o") is False
|
||||
# OpenRouter-prefixed deepseek is NOT the direct DeepSeek provider.
|
||||
assert eng.can_serve("openrouter/deepseek/deepseek-r1") is False
|
||||
|
||||
|
||||
class TestCloudEngineDeepSeek:
|
||||
"""PR #504: DeepSeek as a first-class cloud provider (OpenAI-compatible)."""
|
||||
|
||||
def test_is_deepseek_model_predicate(self) -> None:
|
||||
assert _is_deepseek_model("deepseek-v4-flash") is True
|
||||
assert _is_deepseek_model("deepseek-v4-pro") is True
|
||||
assert _is_deepseek_model("DeepSeek-V4-Pro") is True # case-insensitive
|
||||
assert _is_deepseek_model("gpt-4o") is False
|
||||
# No predicate collision: openrouter/deepseek/* belongs to OpenRouter.
|
||||
assert _is_deepseek_model("openrouter/deepseek/deepseek-r1") is False
|
||||
assert _is_openrouter_model("openrouter/deepseek/deepseek-r1") is True
|
||||
# And a deepseek name is not mistaken for an OpenAI model.
|
||||
assert _is_openai_model("deepseek-v4-pro") is False
|
||||
|
||||
def test_pricing_entries_present(self) -> None:
|
||||
assert estimate_cost("deepseek-v4-flash", 1_000_000, 1_000_000) == (
|
||||
pytest.approx(1.37) # 0.27 + 1.10
|
||||
)
|
||||
assert estimate_cost("deepseek-v4-pro", 1_000_000, 1_000_000) == (
|
||||
pytest.approx(2.74) # 0.55 + 2.19
|
||||
)
|
||||
|
||||
def test_init_wires_deepseek_client(self, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
"""DEEPSEEK_API_KEY builds an openai client pointed at api.deepseek.com."""
|
||||
for var in ("OPENAI_API_KEY", "ANTHROPIC_API_KEY"):
|
||||
monkeypatch.delenv(var, raising=False)
|
||||
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-deepseek-test")
|
||||
|
||||
fake_openai = mock.MagicMock()
|
||||
with mock.patch.dict("sys.modules", {"openai": fake_openai}):
|
||||
EngineRegistry.register_value("cloud", CloudEngine)
|
||||
engine = CloudEngine()
|
||||
|
||||
fake_openai.OpenAI.assert_any_call(
|
||||
base_url="https://api.deepseek.com/v1",
|
||||
api_key="sk-deepseek-test",
|
||||
)
|
||||
assert engine._deepseek_client is not None
|
||||
|
||||
def test_health_and_list_models_gated_on_deepseek_key(
|
||||
self, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
for var in ("OPENAI_API_KEY", "ANTHROPIC_API_KEY"):
|
||||
monkeypatch.delenv(var, raising=False)
|
||||
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-deepseek-test")
|
||||
|
||||
fake_openai = mock.MagicMock()
|
||||
with mock.patch.dict("sys.modules", {"openai": fake_openai}):
|
||||
EngineRegistry.register_value("cloud", CloudEngine)
|
||||
engine = CloudEngine()
|
||||
|
||||
assert engine.health() is True
|
||||
models = engine.list_models()
|
||||
assert "deepseek-v4-flash" in models
|
||||
assert "deepseek-v4-pro" in models
|
||||
# can_serve must agree with list_models (regression for the missing
|
||||
# _client_for_model deepseek branch flagged by the #504 verifier).
|
||||
assert engine.can_serve("deepseek-v4-pro") is True
|
||||
assert engine.can_serve("deepseek-v4-flash") is True
|
||||
|
||||
def test_generate_routes_to_deepseek_client(
|
||||
self, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
for var in ("OPENAI_API_KEY", "ANTHROPIC_API_KEY", "DEEPSEEK_API_KEY"):
|
||||
monkeypatch.delenv(var, raising=False)
|
||||
|
||||
fake_usage = SimpleNamespace(
|
||||
prompt_tokens=7, completion_tokens=3, total_tokens=10
|
||||
)
|
||||
fake_choice = SimpleNamespace(
|
||||
message=SimpleNamespace(content="ds-hello"),
|
||||
finish_reason="stop",
|
||||
)
|
||||
fake_resp = SimpleNamespace(
|
||||
choices=[fake_choice], usage=fake_usage, model="deepseek-v4-pro"
|
||||
)
|
||||
fake_client = mock.MagicMock()
|
||||
fake_client.chat.completions.create.return_value = fake_resp
|
||||
|
||||
EngineRegistry.register_value("cloud", CloudEngine)
|
||||
engine = CloudEngine()
|
||||
engine._deepseek_client = fake_client
|
||||
|
||||
result = engine.generate(
|
||||
[Message(role=Role.USER, content="Hi")], model="deepseek-v4-pro"
|
||||
)
|
||||
assert result["content"] == "ds-hello"
|
||||
assert result["usage"]["prompt_tokens"] == 7
|
||||
# Routed to the DeepSeek client, not OpenAI.
|
||||
fake_client.chat.completions.create.assert_called_once()
|
||||
|
||||
def test_generate_without_client_raises(
|
||||
self, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
for var in ("OPENAI_API_KEY", "ANTHROPIC_API_KEY", "DEEPSEEK_API_KEY"):
|
||||
monkeypatch.delenv(var, raising=False)
|
||||
EngineRegistry.register_value("cloud", CloudEngine)
|
||||
engine = CloudEngine()
|
||||
assert engine._deepseek_client is None
|
||||
with pytest.raises(EngineConnectionError):
|
||||
engine.generate(
|
||||
[Message(role=Role.USER, content="Hi")], model="deepseek-v4-pro"
|
||||
)
|
||||
|
||||
@@ -204,6 +204,54 @@ class TestGetEngine:
|
||||
assert result is not None
|
||||
assert result[0] == "local"
|
||||
|
||||
def test_dummy_openai_key_does_not_misroute_local_model(
|
||||
self, monkeypatch: object
|
||||
) -> None:
|
||||
"""#335: a present-but-dummy OPENAI_API_KEY + a down local engine must
|
||||
NOT cause a local Ollama model to be routed to the cloud engine.
|
||||
|
||||
Before the fix, CloudEngine.can_serve('qwen3.5:0.8b') returned True
|
||||
whenever any OpenAI client existed (even a junk key), so get_engine
|
||||
picked 'cloud' and the request later died with "OpenAI client not
|
||||
available". With the strict _client_for_model fall-through it returns
|
||||
None for unrecognized names, so get_engine declines cloud and (with the
|
||||
local engine down) returns None — surfacing a "start your local engine"
|
||||
failure instead.
|
||||
"""
|
||||
from openjarvis.engine.cloud import CloudEngine
|
||||
|
||||
_reg("ollama", "ollama")
|
||||
EngineRegistry.register_value("cloud", CloudEngine)
|
||||
|
||||
cfg = JarvisConfig()
|
||||
cfg.engine.default = "ollama"
|
||||
|
||||
def _make(k, c): # noqa: ANN001
|
||||
if k == "ollama":
|
||||
# Local engine is down (post-restart Ollama not yet up).
|
||||
return _FakeEngine(healthy=False, models=["qwen3.5:0.8b"])
|
||||
# Real CloudEngine with only a (dummy) OpenAI client wired.
|
||||
eng = CloudEngine.__new__(CloudEngine)
|
||||
for name in (
|
||||
"_openai_client",
|
||||
"_anthropic_client",
|
||||
"_google_client",
|
||||
"_openrouter_client",
|
||||
"_minimax_client",
|
||||
"_deepseek_client",
|
||||
"_codex_client",
|
||||
):
|
||||
setattr(eng, name, object() if name == "_openai_client" else None)
|
||||
return eng
|
||||
|
||||
with mock.patch(
|
||||
"openjarvis.engine._discovery._make_engine",
|
||||
side_effect=_make,
|
||||
):
|
||||
result = get_engine(cfg, model="qwen3.5:0.8b")
|
||||
# Cloud must NOT be selected for a local model name.
|
||||
assert result is None
|
||||
|
||||
def test_model_none_preserves_model_agnostic_selection(self) -> None:
|
||||
"""model=None keeps the legacy behaviour: first healthy engine wins."""
|
||||
_reg("primary", "primary")
|
||||
|
||||
@@ -487,6 +487,167 @@ class TestChatCompletions:
|
||||
assert data["choices"][0]["finish_reason"] == "stop"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Identity system-prompt injection (#540)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _make_capturing_engine(captured: list):
|
||||
"""Like ``_make_engine`` but records the messages each path receives.
|
||||
|
||||
``engine.generate`` is a MagicMock so ``call_args`` works on the
|
||||
direct/non-stream path. ``engine.stream`` / ``engine.stream_full`` are
|
||||
plain async-generator FUNCTIONS, so they capture their ``messages``
|
||||
argument into the shared *captured* list from inside the generator body
|
||||
(``call_args`` does not apply to plain functions).
|
||||
"""
|
||||
engine = MagicMock()
|
||||
engine.engine_id = "mock"
|
||||
engine.health.return_value = True
|
||||
engine.list_models.return_value = ["test-model"]
|
||||
engine.generate.return_value = {
|
||||
"content": "ok",
|
||||
"usage": {"prompt_tokens": 5, "completion_tokens": 3, "total_tokens": 8},
|
||||
"model": "test-model",
|
||||
"finish_reason": "stop",
|
||||
}
|
||||
|
||||
async def mock_stream(messages, *, model, temperature=0.7, max_tokens=1024, **kw):
|
||||
captured.append(messages)
|
||||
for token in ["Hello", " ", "world"]:
|
||||
yield token
|
||||
|
||||
async def mock_stream_full(
|
||||
messages, *, model, temperature=0.7, max_tokens=1024, **kw
|
||||
):
|
||||
from openjarvis.engine._stubs import StreamChunk
|
||||
|
||||
captured.append(messages)
|
||||
yield StreamChunk(content="ok", finish_reason="stop")
|
||||
|
||||
engine.stream = mock_stream
|
||||
engine.stream_full = mock_stream_full
|
||||
return engine
|
||||
|
||||
|
||||
class TestIdentityPromptInjection:
|
||||
"""Regression for #540.
|
||||
|
||||
The desktop UI posts only user/assistant turns to the
|
||||
OpenAI-compatible ``/v1/chat/completions`` endpoint, so the engine never
|
||||
saw OpenJarvis's identity system prompt and the model answered from its
|
||||
training identity ("I'm Claude", "I am Qwen", ...). The engine-direct
|
||||
server handlers must now inject ``agent.default_system_prompt`` whenever
|
||||
the client omits a system message — and must NOT inject a second one when
|
||||
the client already supplies their own.
|
||||
"""
|
||||
|
||||
def test_stream_injects_identity_when_absent(self):
|
||||
captured: list = []
|
||||
engine = _make_capturing_engine(captured)
|
||||
client = TestClient(create_app(engine, "test-model"))
|
||||
|
||||
resp = client.post(
|
||||
"/v1/chat/completions",
|
||||
json={
|
||||
"model": "test-model",
|
||||
"messages": [{"role": "user", "content": "who are you?"}],
|
||||
"stream": True,
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
# Drain the stream so the generator body runs and records messages.
|
||||
_ = resp.text
|
||||
assert captured, "engine.stream was never called"
|
||||
msgs = captured[-1]
|
||||
assert msgs[0].role.value == "system"
|
||||
assert "OpenJarvis" in msgs[0].content
|
||||
|
||||
def test_stream_no_double_injection_when_client_supplies_system(self):
|
||||
captured: list = []
|
||||
engine = _make_capturing_engine(captured)
|
||||
client = TestClient(create_app(engine, "test-model"))
|
||||
|
||||
resp = client.post(
|
||||
"/v1/chat/completions",
|
||||
json={
|
||||
"model": "test-model",
|
||||
"messages": [
|
||||
{"role": "system", "content": "Be terse."},
|
||||
{"role": "user", "content": "who are you?"},
|
||||
],
|
||||
"stream": True,
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
_ = resp.text
|
||||
msgs = captured[-1]
|
||||
system_msgs = [m for m in msgs if m.role.value == "system"]
|
||||
assert len(system_msgs) == 1
|
||||
assert system_msgs[0].content == "Be terse."
|
||||
|
||||
def test_direct_injects_identity_when_absent(self):
|
||||
captured: list = []
|
||||
engine = _make_capturing_engine(captured)
|
||||
# No agent -> non-stream request goes through _handle_direct.
|
||||
client = TestClient(create_app(engine, "test-model"))
|
||||
|
||||
resp = client.post(
|
||||
"/v1/chat/completions",
|
||||
json={
|
||||
"model": "test-model",
|
||||
"messages": [{"role": "user", "content": "who are you?"}],
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
assert engine.generate.called
|
||||
msgs = engine.generate.call_args.args[0]
|
||||
assert msgs[0].role.value == "system"
|
||||
assert "OpenJarvis" in msgs[0].content
|
||||
|
||||
def test_direct_no_double_injection_when_client_supplies_system(self):
|
||||
captured: list = []
|
||||
engine = _make_capturing_engine(captured)
|
||||
client = TestClient(create_app(engine, "test-model"))
|
||||
|
||||
resp = client.post(
|
||||
"/v1/chat/completions",
|
||||
json={
|
||||
"model": "test-model",
|
||||
"messages": [
|
||||
{"role": "system", "content": "Be terse."},
|
||||
{"role": "user", "content": "who are you?"},
|
||||
],
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
msgs = engine.generate.call_args.args[0]
|
||||
system_msgs = [m for m in msgs if m.role.value == "system"]
|
||||
assert len(system_msgs) == 1
|
||||
assert system_msgs[0].content == "Be terse."
|
||||
|
||||
def test_stream_tools_injects_identity_when_absent(self):
|
||||
captured: list = []
|
||||
engine = _make_capturing_engine(captured)
|
||||
client = TestClient(create_app(engine, "test-model"))
|
||||
|
||||
resp = client.post(
|
||||
"/v1/chat/completions",
|
||||
json={
|
||||
"model": "test-model",
|
||||
"messages": [{"role": "user", "content": "who are you?"}],
|
||||
"tools": [{"type": "function", "function": {"name": "calc"}}],
|
||||
"stream": True,
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
_ = resp.text
|
||||
assert captured, "engine.stream_full was never called"
|
||||
msgs = captured[-1]
|
||||
assert msgs[0].role.value == "system"
|
||||
assert "OpenJarvis" in msgs[0].content
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Models endpoint tests
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -0,0 +1,94 @@
|
||||
"""Tests for vision input support: ``Message.images`` -> Ollama payload.
|
||||
|
||||
These cover the data-flow contract that makes vision work end to end:
|
||||
a ``Message`` can carry base64 images, the engine serializer forwards them
|
||||
to Ollama's ``/api/chat`` ``images`` field, and text-only messages are
|
||||
completely unaffected. The security guardrail must preserve images when it
|
||||
rewrites a flagged message.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from types import SimpleNamespace
|
||||
|
||||
import openjarvis.engine.ollama as ollama_mod
|
||||
from openjarvis.core.types import Message, Role
|
||||
from openjarvis.engine._base import messages_to_dicts
|
||||
|
||||
|
||||
def test_message_defaults_to_no_images() -> None:
|
||||
assert Message(role=Role.USER, content="hi").images is None
|
||||
|
||||
|
||||
def test_messages_to_dicts_omits_images_for_text() -> None:
|
||||
dicts = messages_to_dicts([Message(role=Role.USER, content="hi")])
|
||||
assert "images" not in dicts[0]
|
||||
|
||||
|
||||
def test_messages_to_dicts_forwards_images() -> None:
|
||||
b64 = "aGVsbG8=" # "hello"
|
||||
dicts = messages_to_dicts(
|
||||
[Message(role=Role.USER, content="what is this?", images=[b64])]
|
||||
)
|
||||
assert dicts[0]["role"] == "user"
|
||||
assert dicts[0]["content"] == "what is this?"
|
||||
assert dicts[0]["images"] == [b64]
|
||||
|
||||
|
||||
def test_messages_to_dicts_empty_images_treated_as_text() -> None:
|
||||
dicts = messages_to_dicts([Message(role=Role.USER, content="hi", images=[])])
|
||||
assert "images" not in dicts[0]
|
||||
|
||||
|
||||
def test_default_num_ctx_default_and_override(monkeypatch) -> None:
|
||||
monkeypatch.delenv("JARVIS_NUM_CTX", raising=False)
|
||||
assert ollama_mod._default_num_ctx() == 16384
|
||||
|
||||
monkeypatch.setenv("JARVIS_NUM_CTX", "8000")
|
||||
assert ollama_mod._default_num_ctx() == 8000
|
||||
|
||||
# A non-integer override must fall back to the safe default, not crash.
|
||||
monkeypatch.setenv("JARVIS_NUM_CTX", "not-an-int")
|
||||
assert ollama_mod._default_num_ctx() == 16384
|
||||
|
||||
|
||||
def test_guardrails_preserves_images_when_sanitizing() -> None:
|
||||
"""A flagged message gets rewritten; its image must survive the rewrite."""
|
||||
from openjarvis.security.guardrails import GuardrailsEngine
|
||||
|
||||
class _RecordingEngine:
|
||||
"""Captures the messages the guardrail forwards to the real engine."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self.received: list[Message] = []
|
||||
|
||||
def generate(self, messages, *, model, **kwargs):
|
||||
self.received = list(messages)
|
||||
return {"content": "ok"}
|
||||
|
||||
class _AlwaysFlag:
|
||||
"""A scanner that flags everything, forcing the sanitize rewrite path."""
|
||||
|
||||
def scan(self, text: str):
|
||||
finding = SimpleNamespace(
|
||||
pattern_name="test",
|
||||
threat_level=SimpleNamespace(value="low"),
|
||||
description="always flags",
|
||||
)
|
||||
return SimpleNamespace(findings=[finding])
|
||||
|
||||
def redact(self, text: str) -> str:
|
||||
return text
|
||||
|
||||
engine = _RecordingEngine()
|
||||
guarded = GuardrailsEngine(
|
||||
engine,
|
||||
scanners=[_AlwaysFlag()],
|
||||
scan_input=True,
|
||||
scan_output=False,
|
||||
)
|
||||
msg = Message(role=Role.USER, content="suspicious", images=["aGVsbG8="])
|
||||
|
||||
guarded.generate([msg], model="x")
|
||||
|
||||
assert engine.received[0].images == ["aGVsbG8="]
|
||||
Reference in New Issue
Block a user