mirror of
https://github.com/open-jarvis/OpenJarvis.git
synced 2026-08-14 08:52:06 +00:00
Compare commits
303
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d9725fbb6a | ||
|
|
23f04264f9 | ||
|
|
8b59eb87e0 | ||
|
|
2e68e227b7 | ||
|
|
fc98614437 | ||
|
|
b1c5aba6fd | ||
|
|
6240c59ca3 | ||
|
|
9f3c7fd086 | ||
|
|
657c8dd26b | ||
|
|
4ef296e9d0 | ||
|
|
d5d06ca0e5 | ||
|
|
213ee4ff7e | ||
|
|
215ab76e5f | ||
|
|
0812ee0701 | ||
|
|
0b140110d8 | ||
|
|
2623f9e0f4 | ||
|
|
d454c41500 | ||
|
|
e3fb816d12 | ||
|
|
3486f27357 | ||
|
|
928776a71c | ||
|
|
2c2a4b6ae4 | ||
|
|
26bc7efb09 | ||
|
|
f3954e087a | ||
|
|
d865b4bed4 | ||
|
|
c686517cc7 | ||
|
|
904133cb25 | ||
|
|
299dee1f40 | ||
|
|
44ff286005 | ||
|
|
be51eb8684 | ||
|
|
420908401c | ||
|
|
b70be55681 | ||
|
|
a0187e40e6 | ||
|
|
d32f20f9b3 | ||
|
|
0b552cbcb5 | ||
|
|
19fd3c8d2b | ||
|
|
1fa80d8ecd | ||
|
|
b3f90691bf | ||
|
|
4ebf0839e7 | ||
|
|
b1e93d4ed0 | ||
|
|
eb2b612c7c | ||
|
|
560ec860df | ||
|
|
e7c46c1985 | ||
|
|
00d1e39b6d | ||
|
|
843375d6ef | ||
|
|
8d33cb58fa | ||
|
|
e4c4bcbae3 | ||
|
|
993c24c8b9 | ||
|
|
5bc8d3a2f6 | ||
|
|
433d10db5e | ||
|
|
9b7b3681f6 | ||
|
|
6dbe5461bb | ||
|
|
a65592fecb | ||
|
|
0513fbdb84 | ||
|
|
d4eb6308b1 | ||
|
|
2853a0001d | ||
|
|
3c99481975 | ||
|
|
4bf39af9bd | ||
|
|
0a3e812751 | ||
|
|
eb46febad5 | ||
|
|
81482b45d4 | ||
|
|
3e2f4bcdb4 | ||
|
|
a35b21195f | ||
|
|
f9d1bc8c27 | ||
|
|
dfa908c358 | ||
|
|
8ef1ab1928 | ||
|
|
28e75cb513 | ||
|
|
4b9948250b | ||
|
|
79e23719d4 | ||
|
|
7ba334b5f0 | ||
|
|
48a2627c9a | ||
|
|
8625f4f95f | ||
|
|
cf08f164c0 | ||
|
|
b21463aab6 | ||
|
|
0cac61d3bb | ||
|
|
50993dfa4d | ||
|
|
527f84f960 | ||
|
|
8eaeb3a754 | ||
|
|
90b7d0cb9b | ||
|
|
d7053c35d5 | ||
|
|
03c5ec3e40 | ||
|
|
4218258486 | ||
|
|
176ed3029b | ||
|
|
590addae4c | ||
|
|
bc9fa6b3c8 | ||
|
|
18d9897317 | ||
|
|
a61d3c09e1 | ||
|
|
c76c80d6b4 | ||
|
|
d908372eb5 | ||
|
|
1b2b0a06e1 | ||
|
|
9e4504cce4 | ||
|
|
d8d4985eb5 | ||
|
|
3359629456 | ||
|
|
726445433b | ||
|
|
b30bf43557 | ||
|
|
a99ee7c473 | ||
|
|
4523715bff | ||
|
|
f7c948fe12 | ||
|
|
30a014faf4 | ||
|
|
bb90480430 | ||
|
|
3ab7764e06 | ||
|
|
68b27654a8 | ||
|
|
1827c1578b | ||
|
|
28ef745384 | ||
|
|
50780a6a1d | ||
|
|
f72abadf68 | ||
|
|
945dabbce7 | ||
|
|
7506bcc0e4 | ||
|
|
739bff417c | ||
|
|
de3e86c544 | ||
|
|
58e822ff91 | ||
|
|
b3cfa398b4 | ||
|
|
53fb42104e | ||
|
|
f922811cba | ||
|
|
097795567b | ||
|
|
98b79eca72 | ||
|
|
58f7f717ae | ||
|
|
4b4fd587b2 | ||
|
|
690b95e050 | ||
|
|
d1ad3316b6 | ||
|
|
2b14315f3c | ||
|
|
09b19193fe | ||
|
|
b5926400e2 | ||
|
|
3398e0c10b | ||
|
|
bd1ab115ec | ||
|
|
5270149ea5 | ||
|
|
a13d8a909d | ||
|
|
0fcee8236e | ||
|
|
a08282c3e9 | ||
|
|
d13006263e | ||
|
|
6fc644e209 | ||
|
|
21f1becb4d | ||
|
|
d9af1eca45 | ||
|
|
a901fcc12a | ||
|
|
f833193454 | ||
|
|
0591a80448 | ||
|
|
d8a8cd8df7 | ||
|
|
c48784e304 | ||
|
|
e29ed9986e | ||
|
|
a7b9e09305 | ||
|
|
efa64beda8 | ||
|
|
a79cd6f5e9 | ||
|
|
b6a8280d68 | ||
|
|
caa3adbbce | ||
|
|
ad7c86495f | ||
|
|
f1b0df6b6b | ||
|
|
c8970cc387 | ||
|
|
48dcb40be4 | ||
|
|
94c515bba5 | ||
|
|
c9290aa77f | ||
|
|
cf29ef710e | ||
|
|
3aa24b2709 | ||
|
|
c6f172cbdc | ||
|
|
00b85b71dd | ||
|
|
6a011b4abb | ||
|
|
20f90f3ca0 | ||
|
|
4fa9163ea9 | ||
|
|
23ed17fe12 | ||
|
|
26285e56fa | ||
|
|
c6448f7cf2 | ||
|
|
6a27094658 | ||
|
|
c14c7c24c0 | ||
|
|
75d26d23ea | ||
|
|
4ece9a933e | ||
|
|
00065cd960 | ||
|
|
0702f2793c | ||
|
|
f6156f480b | ||
|
|
eb08987dae | ||
|
|
5acc86d7cc | ||
|
|
f827a8165d | ||
|
|
0d3cc6db51 | ||
|
|
b0f133a42d | ||
|
|
9cd760ad1e | ||
|
|
87e978ef20 | ||
|
|
47ca09f1a3 | ||
|
|
7e8db280e4 | ||
|
|
e3f2b008d2 | ||
|
|
cdae426d48 | ||
|
|
ae9727599b | ||
|
|
a86d022947 | ||
|
|
8e6bc343d8 | ||
|
|
56c9a59f8d | ||
|
|
bed3524b62 | ||
|
|
eb5d1e467f | ||
|
|
b43f4d2291 | ||
|
|
658effb964 | ||
|
|
0c34817ecd | ||
|
|
c06633e26f | ||
|
|
deb0dd509c | ||
|
|
a76319841a | ||
|
|
cf8d3e2cbe | ||
|
|
bced6cc274 | ||
|
|
612d3e1f88 | ||
|
|
58f7b753bf | ||
|
|
158fc83feb | ||
|
|
52e830e1a4 | ||
|
|
0483f5377d | ||
|
|
2d6d6ed7bd | ||
|
|
71c5b2bce1 | ||
|
|
440915c822 | ||
|
|
c84f1fddee | ||
|
|
fecbc51767 | ||
|
|
07b6cba6df | ||
|
|
7b6fe86c91 | ||
|
|
a6f18dff43 | ||
|
|
5a74fd82b8 | ||
|
|
9d0bb25516 | ||
|
|
e963321114 | ||
|
|
363bdbfa41 | ||
|
|
75cb70526e | ||
|
|
e087773b36 | ||
|
|
9c9c883f24 | ||
|
|
ef267abcc1 | ||
|
|
5f685a2e3b | ||
|
|
1d37528637 | ||
|
|
86c42e4e4a | ||
|
|
a65cd12ea2 | ||
|
|
36482b1a27 | ||
|
|
333cb9a370 | ||
|
|
4c37aa83df | ||
|
|
c8f1b3c54a | ||
|
|
855ed51780 | ||
|
|
4652b6e8a8 | ||
|
|
30f8d57398 | ||
|
|
9a42e62172 | ||
|
|
90f009e906 | ||
|
|
0217a901dd | ||
|
|
c7f451fbc8 | ||
|
|
30ac635e83 | ||
|
|
31efd9e672 | ||
|
|
8b5a304222 | ||
|
|
df566d4186 | ||
|
|
4177b5c50e | ||
|
|
f21eec6c86 | ||
|
|
4a7817509b | ||
|
|
2b0e6e7130 | ||
|
|
289fcb00e2 | ||
|
|
4668ea0b7f | ||
|
|
a58d6b6c48 | ||
|
|
af21bc18ea | ||
|
|
fea8d3e872 | ||
|
|
5b847cd081 | ||
|
|
a5029e297f | ||
|
|
9af0dfe336 | ||
|
|
b863cbb07b | ||
|
|
7081be7bd3 | ||
|
|
df4332b0e8 | ||
|
|
e97088f199 | ||
|
|
50b887ba25 | ||
|
|
9540e85dfb | ||
|
|
a7b8597856 | ||
|
|
eacf34e500 | ||
|
|
10b7ef3d6c | ||
|
|
44815e8544 | ||
|
|
b8c3b417ec | ||
|
|
da993eedf8 | ||
|
|
eb097062b0 | ||
|
|
d7888caf7a | ||
|
|
5044cf2bc8 | ||
|
|
797ee6b761 | ||
|
|
fec270eecf | ||
|
|
d314f3ea8c | ||
|
|
57151327b3 | ||
|
|
37f4942b07 | ||
|
|
b8245136db | ||
|
|
ab46c89660 | ||
|
|
f705a46b39 | ||
|
|
f0ab045cad | ||
|
|
9e7366fc05 | ||
|
|
4935053ac7 | ||
|
|
8fb9d5ece3 | ||
|
|
daaa5577f0 | ||
|
|
d74fb68b68 | ||
|
|
a689be0c69 | ||
|
|
f41cf420be | ||
|
|
e79bc1e196 | ||
|
|
a3ba63d148 | ||
|
|
ac2deb9d9d | ||
|
|
5a97e1197c | ||
|
|
d7624510bb | ||
|
|
aaaf00a951 | ||
|
|
bd1e87373f | ||
|
|
5e0123ffd7 | ||
|
|
363cb637cd | ||
|
|
2d3e5ad8b6 | ||
|
|
259f97b81c | ||
|
|
cdb04cdecb | ||
|
|
f16bf734cf | ||
|
|
3f2f46e406 | ||
|
|
deabda0c4b | ||
|
|
43b037a796 | ||
|
|
1403c7b843 | ||
|
|
f4d544d09e | ||
|
|
65a340bb7a | ||
|
|
a3a23125b4 | ||
|
|
7ba095ebf1 | ||
|
|
a26f7b2e48 | ||
|
|
83bcbb803c | ||
|
|
484d0f090b | ||
|
|
5a5df909c1 | ||
|
|
02565fb787 | ||
|
|
b4f208deb6 | ||
|
|
adf06593b7 | ||
|
|
43f068830e |
@@ -0,0 +1,16 @@
|
||||
# CODEOWNERS — gates which approvals satisfy the "Require review from
|
||||
# Code Owners" branch ruleset on `main`.
|
||||
#
|
||||
# Anyone listed here may approve PRs against the patterns they own.
|
||||
# Combined with the matching branch ruleset toggle, only their approvals
|
||||
# count toward the merge requirement. Non-owners can still leave reviews
|
||||
# and comments; their approvals simply do not unblock merge.
|
||||
#
|
||||
# See: https://docs.github.com/repositories/managing-your-repositories-settings-and-features/customizing-your-repository/about-code-owners
|
||||
#
|
||||
# To add more owners, append GitHub handles (`@username`) or team slugs
|
||||
# (`@open-jarvis/<team>`) to the line below. To gate specific paths
|
||||
# differently, add a more-specific pattern beneath it (later, more
|
||||
# specific rules win).
|
||||
|
||||
* @jonsaadfalcon @ANarayan @robbym-dev
|
||||
@@ -0,0 +1,98 @@
|
||||
name: Pearl Model Validation
|
||||
description: Track conversion and validation of a Pearl-compatible mining model
|
||||
labels: ["type:feature", "area:mining"]
|
||||
body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: |
|
||||
Use this template when promoting a raw Hugging Face model to a
|
||||
Pearl-compatible `pearl-ai/*-pearl` mining model. A model should remain
|
||||
`planned` in OpenJarvis until this checklist is complete.
|
||||
- type: input
|
||||
id: raw_model
|
||||
attributes:
|
||||
label: Raw model
|
||||
placeholder: e.g., Qwen/Qwen3.5-9B
|
||||
validations:
|
||||
required: true
|
||||
- type: input
|
||||
id: pearl_model
|
||||
attributes:
|
||||
label: Pearl model artifact
|
||||
placeholder: e.g., pearl-ai/Qwen3.5-9B-pearl
|
||||
validations:
|
||||
required: true
|
||||
- type: dropdown
|
||||
id: target_provider
|
||||
attributes:
|
||||
label: Target provider
|
||||
options:
|
||||
- vllm-pearl
|
||||
- cpu-pearl
|
||||
- apple-mps-pearl
|
||||
validations:
|
||||
required: true
|
||||
- type: textarea
|
||||
id: quantization_recipe
|
||||
attributes:
|
||||
label: Quantization recipe
|
||||
description: Link or paste the recipe used to create the Pearl model artifact.
|
||||
placeholder: |
|
||||
- compressed-tensors config:
|
||||
- 7-bit mining layers:
|
||||
- 8-bit non-mining layers:
|
||||
- calibration data:
|
||||
- SmoothQuant settings:
|
||||
validations:
|
||||
required: true
|
||||
- type: textarea
|
||||
id: hardware
|
||||
attributes:
|
||||
label: Validation hardware
|
||||
placeholder: |
|
||||
- GPU:
|
||||
- VRAM:
|
||||
- driver/CUDA:
|
||||
- Docker image/tag:
|
||||
- Pearl commit/ref:
|
||||
validations:
|
||||
required: true
|
||||
- type: checkboxes
|
||||
id: acceptance
|
||||
attributes:
|
||||
label: Acceptance checks
|
||||
options:
|
||||
- label: Model loads in Pearl's vLLM miner container
|
||||
required: true
|
||||
- label: vLLM registers Pearl's quantization plugin
|
||||
required: true
|
||||
- label: Mining layers use int7 NoisyGEMM
|
||||
required: true
|
||||
- label: Non-mining layers use int8 vanilla Pearl GEMM
|
||||
required: true
|
||||
- label: `jarvis mine init --model <pearl-model-id>` succeeds
|
||||
required: true
|
||||
- label: `jarvis mine start` succeeds
|
||||
required: true
|
||||
- label: `jarvis ask` succeeds through the mining engine
|
||||
required: true
|
||||
- label: `jarvis mine status` reports gateway/mining metrics
|
||||
required: true
|
||||
- label: `jarvis mine validate-model --allow-planned` passes
|
||||
required: true
|
||||
- label: Gateway/miner logs show no submission errors
|
||||
required: true
|
||||
- type: textarea
|
||||
id: artifacts
|
||||
attributes:
|
||||
label: Artifacts
|
||||
description: Attach logs, metrics, model config, and command output.
|
||||
placeholder: |
|
||||
- /v1/models output:
|
||||
- `jarvis mine status` output:
|
||||
- `jarvis mine validate-model --output` JSON:
|
||||
- gateway metrics excerpt:
|
||||
- miner logs:
|
||||
- PR/commit that flips status to validated:
|
||||
validations:
|
||||
required: true
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"schemaVersion": 1,
|
||||
"label": "Git Clones",
|
||||
"message": "27,547",
|
||||
"message": "159,322",
|
||||
"color": "green",
|
||||
"namedLogo": "git"
|
||||
}
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"total_clones": 27547,
|
||||
"last_updated": "2026-04-21T07:47:01Z",
|
||||
"total_clones": 159322,
|
||||
"last_updated": "2026-07-16T08:08:14Z",
|
||||
"daily": {
|
||||
"2026-03-27": 2189,
|
||||
"2026-03-28": 1874,
|
||||
@@ -26,6 +26,92 @@
|
||||
"2026-04-17": 2341,
|
||||
"2026-04-18": 1487,
|
||||
"2026-04-19": 1339,
|
||||
"2026-04-20": 1428
|
||||
"2026-04-20": 1428,
|
||||
"2026-04-21": 1216,
|
||||
"2026-04-22": 1768,
|
||||
"2026-04-23": 1050,
|
||||
"2026-04-24": 1245,
|
||||
"2026-04-25": 1116,
|
||||
"2026-04-26": 1211,
|
||||
"2026-04-27": 1606,
|
||||
"2026-04-28": 1090,
|
||||
"2026-04-29": 1332,
|
||||
"2026-04-30": 943,
|
||||
"2026-05-01": 1252,
|
||||
"2026-05-02": 1326,
|
||||
"2026-05-03": 1832,
|
||||
"2026-05-04": 1830,
|
||||
"2026-05-05": 3854,
|
||||
"2026-05-06": 1521,
|
||||
"2026-05-07": 1216,
|
||||
"2026-05-08": 661,
|
||||
"2026-05-09": 796,
|
||||
"2026-05-10": 814,
|
||||
"2026-05-11": 1008,
|
||||
"2026-05-12": 1390,
|
||||
"2026-05-13": 1397,
|
||||
"2026-05-14": 846,
|
||||
"2026-05-15": 1671,
|
||||
"2026-05-16": 2264,
|
||||
"2026-05-17": 654,
|
||||
"2026-05-18": 1425,
|
||||
"2026-05-19": 850,
|
||||
"2026-05-20": 954,
|
||||
"2026-05-21": 1605,
|
||||
"2026-05-22": 612,
|
||||
"2026-05-23": 2437,
|
||||
"2026-05-24": 4900,
|
||||
"2026-05-25": 1319,
|
||||
"2026-05-26": 1199,
|
||||
"2026-05-27": 898,
|
||||
"2026-05-28": 1276,
|
||||
"2026-05-29": 2950,
|
||||
"2026-05-30": 4338,
|
||||
"2026-05-31": 1887,
|
||||
"2026-06-01": 2072,
|
||||
"2026-06-02": 1847,
|
||||
"2026-06-03": 2164,
|
||||
"2026-06-04": 2632,
|
||||
"2026-06-05": 2127,
|
||||
"2026-06-06": 2204,
|
||||
"2026-06-07": 1174,
|
||||
"2026-06-08": 2369,
|
||||
"2026-06-09": 1361,
|
||||
"2026-06-10": 1310,
|
||||
"2026-06-11": 2564,
|
||||
"2026-06-12": 1313,
|
||||
"2026-06-13": 2804,
|
||||
"2026-06-14": 1543,
|
||||
"2026-06-15": 1379,
|
||||
"2026-06-16": 1317,
|
||||
"2026-06-17": 1170,
|
||||
"2026-06-18": 1408,
|
||||
"2026-06-19": 1350,
|
||||
"2026-06-20": 1437,
|
||||
"2026-06-21": 1426,
|
||||
"2026-06-22": 1350,
|
||||
"2026-06-23": 1468,
|
||||
"2026-06-24": 1635,
|
||||
"2026-06-25": 1640,
|
||||
"2026-06-26": 1338,
|
||||
"2026-06-27": 1338,
|
||||
"2026-06-28": 1028,
|
||||
"2026-06-29": 765,
|
||||
"2026-06-30": 951,
|
||||
"2026-07-01": 1134,
|
||||
"2026-07-02": 593,
|
||||
"2026-07-03": 537,
|
||||
"2026-07-04": 411,
|
||||
"2026-07-05": 485,
|
||||
"2026-07-06": 555,
|
||||
"2026-07-07": 905,
|
||||
"2026-07-08": 1171,
|
||||
"2026-07-09": 1857,
|
||||
"2026-07-10": 1181,
|
||||
"2026-07-11": 2185,
|
||||
"2026-07-12": 1917,
|
||||
"2026-07-13": 2102,
|
||||
"2026-07-14": 2337,
|
||||
"2026-07-15": 2362
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,92 @@
|
||||
name: Auto-tag on main push
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
actions: write
|
||||
|
||||
jobs:
|
||||
tag:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Compute dev version
|
||||
id: version
|
||||
run: |
|
||||
set -euo pipefail
|
||||
# Base version is the next patch above the latest plain release tag
|
||||
# (vX.Y.Z) reachable from HEAD. pyproject.toml no longer carries a
|
||||
# static version (#526 switched it to hatch-vcs), so the release tag
|
||||
# is the source of truth. `.devN`/`.rcN`/`desktop-*` tags are excluded
|
||||
# so they can't be mistaken for the release base.
|
||||
# Any future manual `X.Y.Z` release will outrank every `X.Y.Z.devN`
|
||||
# autotag — PEP 440 sorts dev releases strictly below the final.
|
||||
LATEST_RELEASE=$(git tag --list 'v[0-9]*' --merged HEAD \
|
||||
| grep -E '^v[0-9]+\.[0-9]+\.[0-9]+$' | sort -V | tail -1)
|
||||
if [[ -z "$LATEST_RELEASE" ]]; then
|
||||
echo "::error::No release tag (vX.Y.Z) reachable from HEAD"
|
||||
exit 1
|
||||
fi
|
||||
BASE="${LATEST_RELEASE#v}"
|
||||
MAJOR=$(echo "$BASE" | cut -d. -f1)
|
||||
MINOR=$(echo "$BASE" | cut -d. -f2)
|
||||
PATCH=$(echo "$BASE" | cut -d. -f3 | sed -E 's/[^0-9].*$//')
|
||||
NEXT_PATCH=$((PATCH + 1))
|
||||
BUILD=$(git rev-list --count HEAD)
|
||||
VERSION="${MAJOR}.${MINOR}.${NEXT_PATCH}.dev${BUILD}"
|
||||
TAG="v${VERSION}"
|
||||
echo "version=${VERSION}" >> "$GITHUB_OUTPUT"
|
||||
echo "tag=${TAG}" >> "$GITHUB_OUTPUT"
|
||||
echo "Computed ${TAG} (base=${BASE})"
|
||||
|
||||
- name: Create and push tag
|
||||
id: tag
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
TAG: ${{ steps.version.outputs.tag }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if git rev-parse "$TAG" >/dev/null 2>&1; then
|
||||
EXISTING_SHA=$(git rev-parse "$TAG")
|
||||
HEAD_SHA=$(git rev-parse HEAD)
|
||||
if [[ "$EXISTING_SHA" != "$HEAD_SHA" ]]; then
|
||||
echo "::error::Tag $TAG already exists at $EXISTING_SHA but HEAD is $HEAD_SHA"
|
||||
exit 1
|
||||
fi
|
||||
echo "Tag $TAG already exists at HEAD, skipping creation"
|
||||
echo "created=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
git config user.name "github-actions[bot]"
|
||||
git config user.email "github-actions[bot]@users.noreply.github.com"
|
||||
git tag "$TAG"
|
||||
git push origin "$TAG"
|
||||
echo "Created and pushed $TAG"
|
||||
echo "created=true" >> "$GITHUB_OUTPUT"
|
||||
|
||||
# Tag pushes made with the default GITHUB_TOKEN do NOT trigger other
|
||||
# workflows (recursion prevention). workflow_dispatch is the documented
|
||||
# exception, so we explicitly dispatch the downstream CD workflows here.
|
||||
# See: https://docs.github.com/en/actions/security-guides/automatic-token-authentication
|
||||
- name: Dispatch downstream workflows
|
||||
if: steps.tag.outputs.created == 'true'
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
TAG: ${{ steps.version.outputs.tag }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
echo "Dispatching pypi-publish.yml @ ${TAG}"
|
||||
gh workflow run pypi-publish.yml \
|
||||
--ref "${TAG}" \
|
||||
-f tag="${TAG}"
|
||||
echo "Dispatching desktop.yml @ ${TAG}"
|
||||
gh workflow run desktop.yml \
|
||||
--ref "${TAG}" \
|
||||
-f tag="${TAG}"
|
||||
@@ -0,0 +1,27 @@
|
||||
name: Bash tests
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
paths:
|
||||
- 'scripts/install/**'
|
||||
- 'tests/install/bash/**'
|
||||
- '.github/workflows/bash-tests.yml'
|
||||
push:
|
||||
branches: [main]
|
||||
paths:
|
||||
- 'scripts/install/**'
|
||||
- 'tests/install/bash/**'
|
||||
|
||||
jobs:
|
||||
bats:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install bats-core
|
||||
run: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y bats
|
||||
|
||||
- name: Run bats tests
|
||||
run: bats tests/install/bash/
|
||||
@@ -23,13 +23,18 @@ jobs:
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v8.0.0
|
||||
with:
|
||||
enable-cache: true
|
||||
|
||||
- name: Install dependencies
|
||||
run: uv sync --extra dev
|
||||
run: uv sync --extra dev --extra framework-comparison --extra server
|
||||
|
||||
- name: Ruff check
|
||||
run: uv run ruff check src/ tests/
|
||||
|
||||
- name: Ruff format check
|
||||
run: uv run ruff format --check src/ tests/
|
||||
|
||||
test:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
@@ -55,16 +60,23 @@ jobs:
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v8.0.0
|
||||
with:
|
||||
enable-cache: true
|
||||
|
||||
- name: Install dependencies
|
||||
run: uv sync --extra dev
|
||||
run: uv sync --extra dev --extra framework-comparison --extra server
|
||||
|
||||
- name: Build Rust extension
|
||||
run: uv run maturin develop --manifest-path rust/crates/openjarvis-python/Cargo.toml
|
||||
|
||||
- name: Run tests
|
||||
# COVERAGE_CORE=sysmon uses CPython 3.12's sys.monitoring backend,
|
||||
# which is dramatically cheaper than the default C trace function.
|
||||
# -n auto fans the suite out across all runner cores via pytest-xdist.
|
||||
env:
|
||||
COVERAGE_CORE: sysmon
|
||||
run: |
|
||||
uv run pytest tests/ -v --tb=short -m "not live and not cloud" \
|
||||
uv run pytest tests/ -n auto -q --tb=short -m "not live and not cloud and not hub" \
|
||||
--cov=openjarvis \
|
||||
--cov-report=term-missing \
|
||||
--cov-report=xml \
|
||||
@@ -72,12 +84,73 @@ jobs:
|
||||
|
||||
- name: Upload coverage report
|
||||
if: always()
|
||||
continue-on-error: true
|
||||
timeout-minutes: 5
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: coverage-xml
|
||||
path: coverage.xml
|
||||
if-no-files-found: warn
|
||||
|
||||
# Windows job — empirically exercises the platform-specific code paths that
|
||||
# the Ubuntu `test` job can never reach: GlobalMemoryStatusEx RAM detection
|
||||
# (#373) and the cp9xx -> UTF-8 stdout reconfigure (#293). Also the only CI
|
||||
# job that builds + imports the mandatory `openjarvis_rust` PyO3 extension
|
||||
# on Windows. Public repo -> Windows runner minutes are free.
|
||||
#
|
||||
# All `run:` steps use static commands only (no `github.event.*`
|
||||
# interpolation), so there is no workflow-injection surface here.
|
||||
test-windows:
|
||||
runs-on: windows-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
# 3.12 = common; 3.13 = the supported ceiling — installing there guards
|
||||
# against a numpy/native wheel gap at the top of the range (#350), which
|
||||
# is exactly how the source-build failure slips in on Windows.
|
||||
python-version: ["3.12", "3.13"]
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v6
|
||||
with:
|
||||
python-version: ${{ matrix.python-version }}
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v8.0.0
|
||||
with:
|
||||
enable-cache: true
|
||||
|
||||
- name: Install dependencies
|
||||
run: uv sync --extra dev --extra server
|
||||
|
||||
# Pure-Python check — runs before the Rust build so a flaky toolchain
|
||||
# install can never mask the actual RAM-detection verification.
|
||||
- name: Verify Windows RAM detection (#373)
|
||||
shell: bash
|
||||
run: |
|
||||
uv run python -c "from openjarvis.core.config import _total_ram_gb; ram = _total_ram_gb(); print(f'GlobalMemoryStatusEx RAM = {ram} GB'); assert ram > 0, f'Windows RAM detection returned {ram}, expected > 0'"
|
||||
|
||||
- name: Run Windows-specific tests (hardware + CLI)
|
||||
shell: bash
|
||||
run: |
|
||||
uv run pytest tests/hardware/test_hardware_profiles.py tests/cli/test_cli.py -v -m "not live and not cloud"
|
||||
|
||||
- name: Install Rust toolchain
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
|
||||
- name: Build + import the PyO3 extension on Windows
|
||||
shell: bash
|
||||
run: |
|
||||
uv run maturin develop --manifest-path rust/crates/openjarvis-python/Cargo.toml
|
||||
uv run python -c "import openjarvis_rust; print('openjarvis_rust imports on Windows OK')"
|
||||
|
||||
- name: Smoke-test CLI
|
||||
shell: bash
|
||||
run: |
|
||||
uv run jarvis --version
|
||||
|
||||
rust:
|
||||
runs-on: ubuntu-latest
|
||||
defaults:
|
||||
|
||||
@@ -11,22 +11,33 @@ concurrency:
|
||||
group: claude-issues-${{ github.event.issue.number || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
# Least-privilege: only what the issue-fixer job actually needs.
|
||||
# id-token (OIDC) is intentionally omitted — claude-code-action@v1 is passed
|
||||
# github_token directly, so OIDC is unused here.
|
||||
permissions:
|
||||
contents: write
|
||||
pull-requests: write
|
||||
issues: write
|
||||
id-token: write
|
||||
|
||||
jobs:
|
||||
fix:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 60
|
||||
timeout-minutes: 15
|
||||
# Security gate: this job reaches secrets.ANTHROPIC_API_KEY and holds a
|
||||
# write-scoped GITHUB_TOKEN. `issues` / `issue_comment` are public,
|
||||
# attacker-controllable events that run in the base-repo context with full
|
||||
# secret access, so the human-triggered paths are restricted to actors with
|
||||
# write-level association (OWNER / MEMBER / COLLABORATOR). This blocks
|
||||
# external / first-time contributors from draining the API budget or
|
||||
# creating branches/PRs, while leaving maintainer use unaffected.
|
||||
if: |
|
||||
github.event_name == 'workflow_dispatch' ||
|
||||
(github.event_name == 'issues' &&
|
||||
contains(fromJSON('["OWNER", "MEMBER", "COLLABORATOR"]'), github.event.issue.author_association) &&
|
||||
(contains(github.event.issue.labels.*.name, 'bug') ||
|
||||
contains(github.event.issue.labels.*.name, 'autofix'))) ||
|
||||
(github.event_name == 'issue_comment' &&
|
||||
contains(fromJSON('["OWNER", "MEMBER", "COLLABORATOR"]'), github.event.comment.author_association) &&
|
||||
!github.event.issue.pull_request &&
|
||||
contains(github.event.comment.body, '@claude') &&
|
||||
github.actor != 'claude[bot]')
|
||||
|
||||
@@ -1,8 +1,6 @@
|
||||
name: Claude PR Review
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
types: [opened, synchronize]
|
||||
issue_comment:
|
||||
types: [created]
|
||||
pull_request_review_comment:
|
||||
@@ -13,24 +11,33 @@ concurrency:
|
||||
group: claude-review-${{ github.event.pull_request.number || github.event.issue.number || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
# Least-privilege: PR review only needs to post comments on the PR.
|
||||
# id-token (OIDC) is omitted — claude-code-action@v1 is passed github_token
|
||||
# directly, so OIDC is unused here.
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: write
|
||||
issues: write
|
||||
id-token: write
|
||||
|
||||
jobs:
|
||||
review:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
# Security gate: this job reaches secrets.ANTHROPIC_API_KEY. Both
|
||||
# issue_comment and pull_request_review_comment are public,
|
||||
# attacker-controllable events that run in the base-repo context with full
|
||||
# secret access, so the @claude paths are restricted to actors with
|
||||
# write-level association (OWNER / MEMBER / COLLABORATOR). External /
|
||||
# first-time contributors cannot trigger the key; maintainers are unaffected.
|
||||
if: |
|
||||
github.event_name == 'pull_request' ||
|
||||
github.event_name == 'workflow_dispatch' ||
|
||||
(github.event_name == 'issue_comment' &&
|
||||
contains(fromJSON('["OWNER", "MEMBER", "COLLABORATOR"]'), github.event.comment.author_association) &&
|
||||
github.event.issue.pull_request &&
|
||||
contains(github.event.comment.body, '@claude') &&
|
||||
github.actor != 'claude[bot]') ||
|
||||
(github.event_name == 'pull_request_review_comment' &&
|
||||
contains(fromJSON('["OWNER", "MEMBER", "COLLABORATOR"]'), github.event.comment.author_association) &&
|
||||
contains(github.event.comment.body, '@claude') &&
|
||||
github.actor != 'claude[bot]')
|
||||
steps:
|
||||
|
||||
+114
-16
@@ -2,11 +2,8 @@ name: Desktop Build & Release
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths:
|
||||
- 'frontend/**'
|
||||
- '.github/workflows/desktop.yml'
|
||||
tags:
|
||||
- 'v*'
|
||||
- 'desktop-v*'
|
||||
pull_request:
|
||||
branches: [main]
|
||||
@@ -14,9 +11,14 @@ on:
|
||||
- 'frontend/**'
|
||||
- '.github/workflows/desktop.yml'
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
tag:
|
||||
description: 'Tag to build (e.g. v1.0.2.dev500). If set, autotag dispatches use this. github.ref still controls the checkout.'
|
||||
required: false
|
||||
type: string
|
||||
|
||||
concurrency:
|
||||
group: desktop-${{ github.ref }}
|
||||
group: desktop-${{ inputs.tag || github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
permissions:
|
||||
@@ -38,7 +40,8 @@ jobs:
|
||||
libappindicator3-dev \
|
||||
librsvg2-dev \
|
||||
patchelf \
|
||||
libxdo-dev
|
||||
libxdo-dev \
|
||||
libdbus-1-dev
|
||||
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v6
|
||||
@@ -64,23 +67,27 @@ jobs:
|
||||
- name: Create frontend dist stub
|
||||
run: mkdir -p frontend/dist && echo '<html><body></body></html>' > frontend/dist/index.html
|
||||
|
||||
- name: Cargo check
|
||||
# `cargo test` builds the crate (same coverage as the old `cargo check`)
|
||||
# and runs the unit tests, including the #331 uv-sync error-formatting
|
||||
# helpers. Static command, no untrusted input — no injection surface.
|
||||
- name: Cargo test
|
||||
working-directory: frontend/src-tauri
|
||||
run: cargo check
|
||||
run: cargo test
|
||||
|
||||
# Remove stale artifacts from the desktop-latest pre-release so that
|
||||
# only the current build's files are available for download.
|
||||
# Remove stale artifacts from the desktop-edge rolling pre-release so that
|
||||
# only the current build's files are available for download. (The stable
|
||||
# `desktop-latest` channel the installed app polls is never cleaned here.)
|
||||
clean-release:
|
||||
needs: [validate]
|
||||
if: github.event_name == 'push' || github.event_name == 'workflow_dispatch'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Delete old assets from desktop-latest
|
||||
- name: Delete old assets from desktop-edge
|
||||
if: "!startsWith(github.ref, 'refs/tags/')"
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
run: |
|
||||
TAG="desktop-latest"
|
||||
TAG="desktop-edge"
|
||||
# List all asset IDs on the release and delete them
|
||||
ASSET_IDS=$(gh api "repos/${{ github.repository }}/releases/tags/${TAG}" \
|
||||
--jq '.assets[].id' 2>/dev/null || true)
|
||||
@@ -108,6 +115,11 @@ jobs:
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
# Full history + tags so the workflow_dispatch fallback in
|
||||
# "Determine release info" can derive the dev version from the
|
||||
# latest release tag (#526).
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Install system dependencies (Linux)
|
||||
if: matrix.platform == 'ubuntu-22.04'
|
||||
@@ -119,7 +131,8 @@ jobs:
|
||||
libappindicator3-dev \
|
||||
librsvg2-dev \
|
||||
patchelf \
|
||||
libxdo-dev
|
||||
libxdo-dev \
|
||||
libdbus-1-dev
|
||||
|
||||
- name: Install Rust stable
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
@@ -156,14 +169,55 @@ jobs:
|
||||
shell: bash
|
||||
run: |
|
||||
if [[ "${{ github.ref }}" == refs/tags/desktop-v* ]]; then
|
||||
# Explicit stable desktop release tag
|
||||
VERSION="${{ github.ref_name }}"
|
||||
VERSION="${VERSION#desktop-v}"
|
||||
echo "tag=${{ github.ref_name }}" >> "$GITHUB_OUTPUT"
|
||||
echo "name=Desktop ${{ github.ref_name }}" >> "$GITHUB_OUTPUT"
|
||||
echo "prerelease=false" >> "$GITHUB_OUTPUT"
|
||||
elif [[ "${{ github.ref }}" == refs/tags/v* ]]; then
|
||||
# Auto-tagged rolling build from autotag.yml — use the same
|
||||
# version as the CLI/PyPI release so all surfaces stay in sync.
|
||||
# Rolling/dev builds go to the `desktop-edge` channel, NOT the
|
||||
# `desktop-latest` channel the installed app polls — so users on
|
||||
# stable are never auto-updated onto an unvetted dev build.
|
||||
VERSION="${{ github.ref_name }}"
|
||||
VERSION="${VERSION#v}"
|
||||
echo "tag=desktop-edge" >> "$GITHUB_OUTPUT"
|
||||
echo "name=Desktop (Edge Build)" >> "$GITHUB_OUTPUT"
|
||||
echo "prerelease=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "tag=desktop-latest" >> "$GITHUB_OUTPUT"
|
||||
echo "name=Desktop (Latest Build)" >> "$GITHUB_OUTPUT"
|
||||
# workflow_dispatch fallback (manual UI dispatch without --ref).
|
||||
# Derive a PEP 440 dev version aligned with autotag.yml so we
|
||||
# don't burn the X.Y.Z release-version namespace.
|
||||
# pyproject.toml no longer carries a static version (#526), so the
|
||||
# base comes from the latest plain release tag (vX.Y.Z), matching
|
||||
# autotag.yml. .dev/.rc/desktop-* tags are excluded.
|
||||
LATEST_RELEASE=$(git tag --list 'v[0-9]*' --merged HEAD \
|
||||
| grep -E '^v[0-9]+\.[0-9]+\.[0-9]+$' | sort -V | tail -1)
|
||||
if [[ -z "$LATEST_RELEASE" ]]; then
|
||||
echo "::error::No release tag (vX.Y.Z) reachable from HEAD"
|
||||
exit 1
|
||||
fi
|
||||
BASE="${LATEST_RELEASE#v}"
|
||||
MAJOR=$(echo "$BASE" | cut -d. -f1)
|
||||
MINOR=$(echo "$BASE" | cut -d. -f2)
|
||||
PATCH=$(echo "$BASE" | cut -d. -f3 | sed -E 's/[^0-9].*$//')
|
||||
NEXT_PATCH=$((PATCH + 1))
|
||||
BUILD=$(git rev-list --count HEAD)
|
||||
VERSION="${MAJOR}.${MINOR}.${NEXT_PATCH}.dev${BUILD}"
|
||||
# Manual dispatches are also dev builds -> the edge channel.
|
||||
echo "tag=desktop-edge" >> "$GITHUB_OUTPUT"
|
||||
echo "name=Desktop (Edge Build)" >> "$GITHUB_OUTPUT"
|
||||
echo "prerelease=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
echo "version=${VERSION}" >> "$GITHUB_OUTPUT"
|
||||
# Tauri's build script requires strict SemVer (MAJOR.MINOR.PATCH[-pre][+build]).
|
||||
# PEP 440 dev releases (`1.0.2.dev661`) are NOT valid SemVer, so we
|
||||
# translate `.devN` to the SemVer-equivalent `-dev.N` prerelease form.
|
||||
# PyPI keeps the PEP 440 form; only the Tauri bundle uses SemVer.
|
||||
TAURI_VERSION="${VERSION/.dev/-dev.}"
|
||||
echo "tauri_version=${TAURI_VERSION}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Configure Apple signing
|
||||
if: runner.os == 'macOS'
|
||||
@@ -199,7 +253,11 @@ jobs:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
TAURI_SIGNING_PRIVATE_KEY: ${{ secrets.TAURI_SIGNING_PRIVATE_KEY }}
|
||||
TAURI_SIGNING_PRIVATE_KEY_PASSWORD: ${{ secrets.TAURI_SIGNING_PRIVATE_KEY_PASSWORD }}
|
||||
TAURI_CONFIG: '{"bundle":{"externalBin":["binaries/ollama"]}}'
|
||||
TAURI_CONFIG: '{"version":"${{ steps.release-info.outputs.tauri_version }}","bundle":{"externalBin":["binaries/ollama"]}}'
|
||||
# tauri-action runs beforeBuildCommand (npm run build:tauri -> vite
|
||||
# build), which requires this at build time (#587). Strict for
|
||||
# releases: a missing/empty secret fails the build by design.
|
||||
VITE_SUPABASE_ANON_KEY: ${{ secrets.VITE_SUPABASE_ANON_KEY }}
|
||||
with:
|
||||
projectPath: frontend
|
||||
tauriScript: npx tauri
|
||||
@@ -210,3 +268,43 @@ jobs:
|
||||
prerelease: ${{ steps.release-info.outputs.prerelease }}
|
||||
includeUpdaterJson: true
|
||||
args: ${{ matrix.args }}
|
||||
|
||||
# When a stable `desktop-v*` release is published, repoint the
|
||||
# `desktop-latest` auto-update channel (the endpoint the installed app
|
||||
# polls) at it. The stable release's own `latest.json` already references
|
||||
# this release's signed assets, so we copy it verbatim — installed apps are
|
||||
# only ever offered vetted stable builds, never `desktop-edge` dev builds.
|
||||
refresh-stable-channel:
|
||||
needs: [build-and-release]
|
||||
if: startsWith(github.ref, 'refs/tags/desktop-v')
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Mirror stable latest.json into desktop-latest
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
STABLE_TAG: ${{ github.ref_name }}
|
||||
REPO: ${{ github.repository }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
# The stable release's updater manifest may take a moment to become
|
||||
# downloadable after tauri-action publishes it; retry briefly.
|
||||
URL="https://github.com/${REPO}/releases/download/${STABLE_TAG}/latest.json"
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if curl -fsSL -o latest.json "$URL"; then
|
||||
echo "Fetched ${STABLE_TAG}/latest.json on attempt ${attempt}"
|
||||
break
|
||||
fi
|
||||
echo "latest.json not ready yet (attempt ${attempt}); sleeping 15s"
|
||||
sleep 15
|
||||
done
|
||||
test -s latest.json || { echo "::error::Could not fetch ${URL}"; exit 1; }
|
||||
# Ensure the channel release exists (prerelease so it never usurps
|
||||
# the stable "Latest" badge), then replace its manifest in place.
|
||||
if ! gh release view desktop-latest --repo "$REPO" >/dev/null 2>&1; then
|
||||
gh release create desktop-latest --repo "$REPO" \
|
||||
--prerelease \
|
||||
--title "Desktop Auto-Update Channel" \
|
||||
--notes "Auto-update channel pointer for the desktop app. Mirrors the latest stable \`desktop-v*\` release; the in-app updater polls this \`latest.json\`. Download the app from the latest stable release, not here."
|
||||
fi
|
||||
gh release upload desktop-latest latest.json --repo "$REPO" --clobber
|
||||
echo "desktop-latest now mirrors ${STABLE_TAG}"
|
||||
|
||||
@@ -41,6 +41,26 @@ jobs:
|
||||
- name: Install dependencies
|
||||
run: uv sync --extra docs
|
||||
|
||||
# Inject the public Supabase anon key so the savings leaderboard works on
|
||||
# the published docs site. Missing/empty (e.g. fork PRs) leaves the
|
||||
# leaderboard gracefully disabled. The key is read from env (not inlined)
|
||||
# and JSON-encoded into a JS string literal to avoid any injection.
|
||||
- name: Inject leaderboard Supabase anon key
|
||||
env:
|
||||
OPENJARVIS_LEADERBOARD_ANON: ${{ secrets.VITE_SUPABASE_ANON_KEY }}
|
||||
run: |
|
||||
python3 - <<'PY'
|
||||
import json, os, pathlib
|
||||
|
||||
key = os.environ.get("OPENJARVIS_LEADERBOARD_ANON", "")
|
||||
pathlib.Path("docs/javascripts/leaderboard-config.js").write_text(
|
||||
"// Generated at docs-build time from the VITE_SUPABASE_ANON_KEY secret.\n"
|
||||
"window.OPENJARVIS_SUPABASE_ANON_KEY = " + json.dumps(key) + ";\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
print("leaderboard anon key:", "set" if key else "empty (leaderboard disabled)")
|
||||
PY
|
||||
|
||||
- name: Build documentation
|
||||
run: uv run mkdocs build
|
||||
|
||||
|
||||
@@ -35,3 +35,8 @@ jobs:
|
||||
- run: npm ci
|
||||
- run: npx tsc --noEmit
|
||||
- run: npm run build
|
||||
env:
|
||||
# Optional: when the secret is unset the build still succeeds and the
|
||||
# leaderboard is disabled (see src/lib/supabase.ts). No placeholder,
|
||||
# so a keyless CI build doesn't bake in a bogus anon key.
|
||||
VITE_SUPABASE_ANON_KEY: ${{ secrets.VITE_SUPABASE_ANON_KEY }}
|
||||
|
||||
@@ -0,0 +1,118 @@
|
||||
name: Installer integration
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
paths:
|
||||
- 'scripts/install/**'
|
||||
- 'src/openjarvis/cli/**'
|
||||
- 'tests/install/**'
|
||||
- '.github/workflows/installer-integration.yml'
|
||||
schedule:
|
||||
- cron: '0 6 * * *'
|
||||
|
||||
jobs:
|
||||
container-matrix:
|
||||
name: ${{ matrix.image }}
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
image:
|
||||
- ubuntu:22.04
|
||||
- ubuntu:24.04
|
||||
- fedora:40
|
||||
|
||||
container: ${{ matrix.image }}
|
||||
|
||||
steps:
|
||||
- name: Install prereqs (Ubuntu/Debian)
|
||||
if: contains(matrix.image, 'ubuntu') || contains(matrix.image, 'debian')
|
||||
run: |
|
||||
apt-get update
|
||||
apt-get install -y curl git python3 sudo
|
||||
|
||||
- name: Install prereqs (Fedora)
|
||||
if: contains(matrix.image, 'fedora')
|
||||
run: |
|
||||
dnf install -y curl git python3 sudo
|
||||
|
||||
- name: Create non-root user
|
||||
run: useradd -m -s /bin/bash testuser
|
||||
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
path: openjarvis-src
|
||||
|
||||
- name: Install Ollama mock
|
||||
run: install -m 755 "$GITHUB_WORKSPACE/openjarvis-src/tests/install/bash/stubs/ollama-mock" /usr/local/bin/ollama
|
||||
|
||||
- name: Run installer
|
||||
run: |
|
||||
chown -R testuser:testuser /home/testuser "$GITHUB_WORKSPACE/openjarvis-src"
|
||||
su testuser -c '
|
||||
export OPENJARVIS_REPO_URL=file://'"$GITHUB_WORKSPACE"'/openjarvis-src
|
||||
cd '"$GITHUB_WORKSPACE"'/openjarvis-src
|
||||
bash scripts/install/install.sh --no-bg-orchestrator
|
||||
'
|
||||
|
||||
- name: Verify install state
|
||||
run: |
|
||||
su testuser -c '
|
||||
test -d ~/.openjarvis/src
|
||||
test -d ~/.openjarvis/.venv
|
||||
test -f ~/.openjarvis/config.toml
|
||||
test -f ~/.openjarvis/.state/install-state.json
|
||||
test -L ~/.local/bin/jarvis
|
||||
'
|
||||
|
||||
- name: Verify jarvis --version
|
||||
run: su testuser -c '~/.local/bin/jarvis --version'
|
||||
|
||||
- name: Verify jarvis doctor exits 0
|
||||
run: su testuser -c '~/.local/bin/jarvis doctor'
|
||||
|
||||
- name: Verify uninstall is clean
|
||||
run: |
|
||||
su testuser -c '
|
||||
~/.local/bin/jarvis-uninstall
|
||||
test ! -d ~/.openjarvis
|
||||
test ! -L ~/.local/bin/jarvis
|
||||
'
|
||||
|
||||
macos:
|
||||
name: ${{ matrix.os }}
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
os: [macos-14, macos-15]
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install Ollama mock
|
||||
run: sudo install -m 755 tests/install/bash/stubs/ollama-mock /usr/local/bin/ollama
|
||||
|
||||
- name: Run installer
|
||||
run: |
|
||||
export OPENJARVIS_REPO_URL=file://$(pwd)
|
||||
bash scripts/install/install.sh --no-bg-orchestrator
|
||||
|
||||
- name: Verify install state
|
||||
run: |
|
||||
test -d ~/.openjarvis/src
|
||||
test -d ~/.openjarvis/.venv
|
||||
test -f ~/.openjarvis/config.toml
|
||||
test -L ~/.local/bin/jarvis
|
||||
|
||||
- name: Verify jarvis --version
|
||||
run: ~/.local/bin/jarvis --version
|
||||
|
||||
- name: Verify jarvis doctor exits 0
|
||||
run: ~/.local/bin/jarvis doctor
|
||||
|
||||
- name: Verify uninstall is clean
|
||||
run: |
|
||||
~/.local/bin/jarvis-uninstall
|
||||
test ! -d ~/.openjarvis
|
||||
test ! -L ~/.local/bin/jarvis
|
||||
@@ -7,6 +7,16 @@ on:
|
||||
tags:
|
||||
- "v*"
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
tag:
|
||||
description: 'Tag to publish (e.g. v1.0.2.dev500). Overrides github.ref.'
|
||||
required: false
|
||||
type: string
|
||||
dry_run:
|
||||
description: 'Dry run: build + validate, then publish to TestPyPI instead of PyPI (no production upload).'
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -17,13 +27,88 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
environment: pypi
|
||||
steps:
|
||||
- name: Resolve target ref
|
||||
id: ref
|
||||
env:
|
||||
INPUT_TAG: ${{ inputs.tag }}
|
||||
DEFAULT_REF: ${{ github.ref_name }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [[ -n "$INPUT_TAG" ]]; then
|
||||
echo "ref=${INPUT_TAG}" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "ref=${DEFAULT_REF}" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
ref: ${{ steps.ref.outputs.ref }}
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v8.0.0
|
||||
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v6
|
||||
with:
|
||||
node-version: '22'
|
||||
cache: 'npm'
|
||||
cache-dependency-path: frontend/package-lock.json
|
||||
|
||||
- name: Build frontend and bundle into package
|
||||
env:
|
||||
VITE_SUPABASE_ANON_KEY: ${{ secrets.VITE_SUPABASE_ANON_KEY }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
cd frontend
|
||||
npm ci
|
||||
# Vite is configured (frontend/vite.config.ts) with
|
||||
# `outDir: '../src/openjarvis/server/static'` and
|
||||
# `emptyOutDir: true`, so the build writes directly into the
|
||||
# Python package's static dir and clears stale assets itself.
|
||||
# No rm/cp is needed — and the previous `dist/`-assuming logic
|
||||
# was broken because `frontend/dist/` is never produced.
|
||||
npm run build
|
||||
STATIC=../src/openjarvis/server/static
|
||||
test -s "$STATIC/index.html" || {
|
||||
echo "::error::${STATIC}/index.html missing or empty after build"
|
||||
exit 1
|
||||
}
|
||||
|
||||
- name: Resolve build version from tag
|
||||
env:
|
||||
REF: ${{ steps.ref.outputs.ref }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
# Strip leading "v" (e.g. v1.0.3.dev825 -> 1.0.3.dev825).
|
||||
VERSION="${REF#v}"
|
||||
if ! [[ "$VERSION" =~ ^[0-9]+\.[0-9]+\.[0-9]+ ]]; then
|
||||
echo "::error::ref '$REF' is not a version tag (expected vX.Y.Z[.devN]); pass -f tag=vX.Y.Z"
|
||||
exit 1
|
||||
fi
|
||||
# pyproject.toml is now dynamic = ["version"] via hatch-vcs (#526), so
|
||||
# there is no static line to sed. setuptools_scm cannot bump custom
|
||||
# `.devN` tags, so we pin the exact build version explicitly — the
|
||||
# published version always equals the pushed tag.
|
||||
echo "SETUPTOOLS_SCM_PRETEND_VERSION=${VERSION}" >> "$GITHUB_ENV"
|
||||
echo "Building version ${VERSION}"
|
||||
|
||||
- name: Build package
|
||||
run: uv build
|
||||
|
||||
- name: Publish to TestPyPI (dry run)
|
||||
if: ${{ inputs.dry_run }}
|
||||
env:
|
||||
UV_PUBLISH_TOKEN: ${{ secrets.TEST_PYPI_API_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [[ -z "${UV_PUBLISH_TOKEN:-}" ]]; then
|
||||
echo "::warning::TEST_PYPI_API_TOKEN is not set — skipping the TestPyPI upload."
|
||||
echo "Build + twine check passed, which validated version derivation and packaging end to end."
|
||||
echo "To exercise a real upload, add a TEST_PYPI_API_TOKEN secret (or a TestPyPI trusted publisher)."
|
||||
exit 0
|
||||
fi
|
||||
uv publish --publish-url https://test.pypi.org/legacy/
|
||||
|
||||
- name: Publish to PyPI
|
||||
if: ${{ !inputs.dry_run }}
|
||||
run: uv publish
|
||||
|
||||
+20
-1
@@ -39,6 +39,9 @@ Thumbs.db
|
||||
# Secrets
|
||||
.env
|
||||
.env.*
|
||||
# ...but keep checked-in example/templates (never contain real secrets)
|
||||
!.env.example
|
||||
!**/.env.example
|
||||
|
||||
# Project
|
||||
*.sqlite
|
||||
@@ -48,7 +51,11 @@ Thumbs.db
|
||||
*.log
|
||||
results/
|
||||
logs/
|
||||
traces/
|
||||
# Anchored to repo root — DO NOT use the unanchored form `traces/`.
|
||||
# hatchling honors .gitignore when building the wheel; an unanchored
|
||||
# `traces/` pattern matches src/openjarvis/traces/ and silently drops
|
||||
# the runtime module from the wheel (issue #372).
|
||||
/traces/
|
||||
coding_task_*
|
||||
get-pip.py
|
||||
# Junk from mocked-path tests that write to their mock's __repr__ as a path
|
||||
@@ -60,6 +67,10 @@ site/
|
||||
# Frontend
|
||||
frontend/node_modules/
|
||||
frontend/dist/
|
||||
|
||||
# Desktop
|
||||
desktop/node_modules/
|
||||
desktop/dist/
|
||||
src/openjarvis/server/static/
|
||||
|
||||
# Desktop (Tauri)
|
||||
@@ -95,6 +106,9 @@ src/openjarvis/channels/whatsapp_baileys_bridge/node_modules/
|
||||
# SQLite in-memory artifacts
|
||||
:memory:
|
||||
|
||||
# Dogfood reports (regenerated locally; not for VCS)
|
||||
dogfood_report*.md
|
||||
|
||||
# Second Repos
|
||||
Inline/
|
||||
scratch/
|
||||
@@ -111,3 +125,8 @@ learning.db
|
||||
**/learning/benchmarks/
|
||||
**/teacher_traces/
|
||||
*.session.json
|
||||
|
||||
# Local dev artifacts (hybrid worker logs + cli debug dumps)
|
||||
minion_logs/
|
||||
*.oj-debug.json
|
||||
oj-debug.*.json
|
||||
|
||||
+287
-9
@@ -10,7 +10,253 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
|
||||
|
||||
### Added
|
||||
|
||||
#### Skills System (Plans 1, 2A, 2B)
|
||||
**Vision input for `jarvis ask`** — attach images to a query with
|
||||
`-i`/`--image` (repeatable) or capture the current screen with
|
||||
`-S`/`--screen`, for vision-capable models such as `gemma3:4b`. Images flow
|
||||
through `Message.images` into Ollama's `/api/chat` `images` field; text-only
|
||||
requests are unaffected. A privacy guard warns before any image is sent to a
|
||||
non-local engine, and the security guardrail now preserves images when it
|
||||
sanitizes a flagged prompt. Screen capture uses the built-in Windows .NET
|
||||
stack with `mss`/`Pillow` fallbacks on other platforms. Adds the
|
||||
`JARVIS_NUM_CTX` environment variable to tune the Ollama context window
|
||||
(default `16384`).
|
||||
|
||||
## [1.0.2] - 2026-05-24
|
||||
|
||||
A patch release that fixes a packaging bug which broke the v1.0.1
|
||||
wheel on PyPI, silences a noisy startup warning, restores a working
|
||||
install path while `openjarvis.ai` is down, improves desktop
|
||||
first-boot diagnostics on Windows, and ships the RAM-detection fix
|
||||
for Windows that missed the v1.0.1 cutoff.
|
||||
|
||||
### Fixed
|
||||
|
||||
**`openjarvis/traces/` missing from the v1.0.1 PyPI wheel** (#372).
|
||||
The `.gitignore` carried an unanchored `traces/` pattern, which
|
||||
hatchling honored at wheel-build time and matched the runtime module
|
||||
`src/openjarvis/traces/` — silently dropping the whole package. Every
|
||||
fresh `pip install openjarvis==1.0.1` then failed at import with
|
||||
`ModuleNotFoundError: No module named 'openjarvis.traces'` on the
|
||||
first `jarvis ask`, learning, or server call. Anchored the pattern to
|
||||
`/traces/`. Verified: a clean `uv build` now produces a wheel
|
||||
containing all four `traces/` files.
|
||||
|
||||
**`pynvml` deprecation `FutureWarning` on every command** (#389).
|
||||
Switched the dependency from the legacy `pynvml` package to NVIDIA's
|
||||
official `nvidia-ml-py` (same `pynvml` module name, no warning shim),
|
||||
and added defensive `warnings.filterwarnings` at every `import pynvml`
|
||||
site to suppress the warning even when `pynvml` is pulled in
|
||||
transitively.
|
||||
|
||||
**Windows RAM detection returning `0.0 GB`** (#373). The Windows
|
||||
branch of `_total_ram_gb()` (via `GlobalMemoryStatusEx`) landed after
|
||||
the v1.0.1 cutoff, so v1.0.1 users still saw `0.0 GB` from `jarvis
|
||||
init`. Now shipping in the wheel. A new `windows-latest` CI job runs
|
||||
the real `GlobalMemoryStatusEx` path on every PR as a regression
|
||||
guard.
|
||||
|
||||
**Desktop first-boot hung on "did not become healthy in time"**
|
||||
(#331). The Tauri boot path ran `uv sync` with stderr discarded and
|
||||
the exit code ignored, so a failed dependency install surfaced only
|
||||
as a generic 600-second health-check timeout. Now captures stderr,
|
||||
checks the exit status, and surfaces the actual `uv sync` error
|
||||
(with the diagnostic tail) before the long wait. The error-formatting
|
||||
logic is covered by unit tests.
|
||||
|
||||
### Changed
|
||||
|
||||
**Install URL moved to GitHub Pages** (#337, #352). The documented
|
||||
`openjarvis.ai/install.sh` URL was failing with `sslv3 alert
|
||||
handshake failure` (the domain is community-operated and had a broken
|
||||
TLS config). The canonical installer is now served from the
|
||||
project-controlled GitHub Pages site at
|
||||
`https://open-jarvis.github.io/OpenJarvis/install.sh`, generated from
|
||||
the same `scripts/install/install.sh` at docs-build time. The README
|
||||
also documents the WSL2 path for Windows and the `uv` prerequisite
|
||||
for the desktop binary, and the installer bails early with a clear
|
||||
message when run under Git Bash / MSYS2 / Cygwin.
|
||||
|
||||
## [1.0.1] - 2026-05-17
|
||||
|
||||
A patch release that closes the auto-update gap so the analytics
|
||||
module added in #351 actually reaches users on the desktop, adds
|
||||
runtime opt-out for that analytics, fixes the misleading upgrade
|
||||
hint the CLI was printing, and lands the ACE optimizer alongside
|
||||
DSPy and GEPA.
|
||||
|
||||
### Added
|
||||
|
||||
**ACE agent optimizer** (`learning/agents/ace_optimizer.py`). Adds
|
||||
[ACE](https://github.com/ace-agent/ace) as a third agent-learning
|
||||
policy alongside DSPy and GEPA. Where DSPy bootstraps few-shot
|
||||
examples and GEPA evolves prompt populations, ACE evolves a textual
|
||||
*playbook* of strategies the agent reads at inference time, updated
|
||||
by a Generator / Reflector / Curator triad. Pick via
|
||||
`[learning.agent] policy = "ace"`. Setup is manual (ACE isn't on
|
||||
PyPI and isn't a properly-packaged Python project as of v1.0.1) —
|
||||
see `docs/learning/ace.md` for the install path and trace-adapter
|
||||
behavior.
|
||||
|
||||
**`jarvis self-update`** subcommand. Detects how OpenJarvis was
|
||||
installed (pip, uv tool, editable git checkout) by inspecting
|
||||
`openjarvis.__file__`, then runs the right upgrade command. Supports
|
||||
`--check` (print the command without running) and `-y` (skip the
|
||||
confirmation prompt). The post-command "new version available" hint
|
||||
now points users at this command instead of guessing at the right
|
||||
flow.
|
||||
|
||||
**Desktop auto-update endpoint wired to the rolling
|
||||
`desktop-latest` GitHub release.** The Tauri updater plugin was
|
||||
configured on the build side (`createUpdaterArtifacts: true`,
|
||||
`includeUpdaterJson: true`, signing key in `TAURI_SIGNING_PRIVATE_KEY`)
|
||||
but inert on the runtime side (`active: false`, `endpoints: []`). The
|
||||
installed desktop app would never check. Both are now fixed; the app
|
||||
polls `releases/download/desktop-latest/latest.json` every 30 minutes
|
||||
and signature-verifies downloads against the minisign pubkey baked
|
||||
into the app. Full flow, key-rotation runbook, and dev escape hatch
|
||||
(`OPENJARVIS_NO_UPDATER=1`) documented in `docs/desktop-auto-update.md`.
|
||||
|
||||
**Analytics env-var opt-out** (`DO_NOT_TRACK`, `OPENJARVIS_NO_ANALYTICS`).
|
||||
Tanvir's analytics module (#351) only respected the
|
||||
`[analytics] enabled` config-file setting. Both env vars are now
|
||||
honored in `is_analytics_enabled()` and in the install.sh beacon
|
||||
script. Any truthy value (`1`, `true`, `yes`, `on`) disables for
|
||||
that process; env opt-out takes precedence over the config file.
|
||||
Documented under a new "Opting out" section in `docs/telemetry.md`.
|
||||
|
||||
### Changed
|
||||
|
||||
**Version-check trigger widened.** The "new version available" hint
|
||||
in `_version_check.py` used to fire only on `{ask, chat, serve}` and
|
||||
hardcoded the wrong upgrade command (`git pull && uv sync` — only
|
||||
correct for editable installs). Now fires on every interactive
|
||||
command (`doctor`, `init`, `quickstart`, `model`, `agents`, `skill`,
|
||||
`memory`, `bench`, `telemetry`, `config`, `eval`, `optimize`, plus
|
||||
the original three) and uses install-detection to print the right
|
||||
upgrade command. Honors `JARVIS_NO_UPDATE_CHECK=1` and `CI=true` to
|
||||
stay silent in automation.
|
||||
|
||||
**Desktop app version bumped 0.1.0 → 1.0.1** across
|
||||
`tauri.conf.json`, `frontend/package.json`, and
|
||||
`frontend/src-tauri/Cargo.toml` so the Python and desktop release
|
||||
streams are aligned and the auto-updater has a real version to
|
||||
compare against.
|
||||
|
||||
### Migration from 1.0.0
|
||||
|
||||
- **Importing `is_analytics_enabled`?** Same signature; behavior now
|
||||
short-circuits on env opt-out before checking the config. Callers
|
||||
that want the raw "is the config flag set" semantic should read
|
||||
`cfg.enabled` directly.
|
||||
- **Editable-git users running `jarvis self-update`** get the
|
||||
detected `git pull && uv sync` command pointed at their actual
|
||||
checkout, not `~/OpenJarvis`. If you'd come to rely on the
|
||||
hardcoded path, update your muscle memory.
|
||||
|
||||
## [1.0.0] - 2026-05-15
|
||||
|
||||
The five-primitive architecture (Intelligence, Engine, Agents,
|
||||
Tools & Memory, Learning) is now stable, with efficiency and
|
||||
on-device learning as first-class capabilities alongside accuracy.
|
||||
Companion blog post:
|
||||
[From Minions to OpenJarvis: A Retrospective on Two Years in Local AI](https://hazyresearch.stanford.edu/blog/2026-05-19-minions-to-openjarvis-retrospective).
|
||||
|
||||
### Highlights
|
||||
|
||||
**Five composable primitives.** Intelligence, Engine, Agents, Tools & Memory,
|
||||
and Learning each sit behind a single typed interface — any slot is
|
||||
substitutable without touching the rest. The composition layer is
|
||||
`JarvisSystem` in `src/openjarvis/system.py`, driven by a TOML config.
|
||||
|
||||
**Built-in agents across three execution modes.** Eight agents spanning a
|
||||
single-turn chat baseline, a deep-research agent with inline citations,
|
||||
a CodeAct-style coder, and a continuous monitor with memory compression
|
||||
for long-horizon workflows. Execution modes cover on-demand, scheduled,
|
||||
and continuous.
|
||||
|
||||
**Starter presets.** Eight preset configs installable via
|
||||
`jarvis init --preset <name>` bundle an agent with a hardware-appropriate
|
||||
engine, connectors, and tools. Variants cover Apple Silicon, Linux GPU
|
||||
servers, and CPU-only laptops, plus a quickstart for LLM-guided spec search.
|
||||
|
||||
**Inference engines.** Four first-class local engines (Ollama, vLLM, SGLang,
|
||||
llama.cpp) and five cloud providers (OpenAI, Anthropic, Google Gemini,
|
||||
OpenRouter, MiniMax) sit behind a single `Engine` interface. Discovery
|
||||
in `engine/_discovery.py` picks a sensible default per host.
|
||||
|
||||
### Added — hybrid local-cloud capabilities
|
||||
|
||||
**Per-query routing via a query-complexity analyzer**
|
||||
(`src/openjarvis/learning/routing/complexity.py`). Produces a 0.0–1.0
|
||||
complexity score with code/math/reasoning signals and a suggested token
|
||||
budget, populating `RoutingContext` so easy queries stay local and only
|
||||
queries that need frontier capability escalate.
|
||||
|
||||
**LLM-guided spec search** (`src/openjarvis/learning/spec_search/`).
|
||||
`SpecSearchOrchestrator` wires diagnose → plan → execute → gate into a
|
||||
single learning session: a frontier model reads traces, proposes
|
||||
coordinated edits across all five primitives, and a held-out benchmark
|
||||
gate (`gate/benchmark_gate.py`, `gate/regression.py`, `gate/cold_start.py`)
|
||||
accepts only non-regressing edits. Ships with the `spec-search-quickstart`
|
||||
preset and a runnable tutorial at `examples/openjarvis/spec_search_quickstart.py`.
|
||||
|
||||
**Six hybrid coordination paradigms** in `src/openjarvis/agents/hybrid/`.
|
||||
Each paradigm pairs a local student with a frontier cloud teacher under
|
||||
a different orchestration shape, as `LocalCloudAgent` subclasses:
|
||||
|
||||
- `minions` — reactive single-local + single-cloud loop
|
||||
- `conductor` — static DAG planner
|
||||
- `advisors` — executor ↔ advisor loop
|
||||
- `archon` — generate → rank → fuse
|
||||
- `skillorchestra` — per-query router across local skills
|
||||
- `toolorchestra` — RL'd local model with a tool pool
|
||||
|
||||
A runner CLI (`python -m openjarvis.agents.hybrid.runner --cell <name>`)
|
||||
and a 35-cell experiment registry (one TOML per method × benchmark ×
|
||||
model triple) let researchers run, score, and compare these on equal
|
||||
footing. Includes a Modal-backed SWE-bench-Verified harness scorer
|
||||
(`evals/scorers/swebench_harness.py`).
|
||||
|
||||
### Added — efficiency as a first-class constraint
|
||||
|
||||
**Hardware-agnostic energy telemetry at 50ms resolution** across NVIDIA
|
||||
(`telemetry/energy_nvidia.py`), AMD (`telemetry/energy_amd.py`), Apple
|
||||
Silicon (`telemetry/energy_apple.py`), and Intel RAPL
|
||||
(`telemetry/energy_rapl.py`). Energy, dollar cost, FLOPs, and latency
|
||||
are treated as evaluation targets alongside accuracy.
|
||||
|
||||
**Instrumentation for FLOPs, batch, steady-state, ITL, phase energy, and
|
||||
vLLM-specific metrics.** Joined per-query by the aggregator
|
||||
(`telemetry/aggregator.py`) so traces carry accuracy + efficiency together.
|
||||
|
||||
### Added — local learning loop
|
||||
|
||||
**Closed-loop optimization across the stack** — model weights via SFT
|
||||
(`learning/intelligence/sft_trainer.py`) and GRPO
|
||||
(`learning/intelligence/grpo_trainer.py` plus an orchestrator-specific
|
||||
variant under `learning/intelligence/orchestrator/`), prompts via DSPy
|
||||
(`learning/agents/dspy_optimizer.py`), agent logic via GEPA
|
||||
(`learning/agents/gepa_optimizer.py`), and engine + stack configuration
|
||||
via LLM-guided spec search. `LearningOrchestrator` coordinates triggers
|
||||
and applies optimizer overlays at discovery time so improvements compound
|
||||
across primitives.
|
||||
|
||||
### Added — cross-framework evaluation
|
||||
|
||||
**External agentic-framework evaluation via subprocess.** The
|
||||
`evals/backends/external/` subpackage wraps Hermes Agent and OpenClaw as
|
||||
one-shot subprocess backends behind the existing `InferenceBackend` ABC.
|
||||
The `evals/comparison/` toolkit provides path + commit-pin enforcement
|
||||
(`third_party.py`), config templating (`make_configs.py`), and LaTeX
|
||||
table generation (`table_gen.py`).
|
||||
|
||||
Ships with a new optional extra `framework-comparison` (depends on
|
||||
`polars`), a `live_external` pytest marker for integration tests
|
||||
requiring real foreign-framework installations, and a `ToolOrchestra`
|
||||
evaluation dataset (`evals/datasets/toolorchestra.py`) alongside the
|
||||
existing 30+ benchmark suite.
|
||||
|
||||
### Added — Skills System (Plans 1, 2A, 2B)
|
||||
|
||||
- **Skills core** — every skill is a tool. Skills appear in a system prompt catalog, agents invoke them on demand, content (pipeline results, markdown instructions, or both) gets injected into context.
|
||||
- `SkillManifest` + `SkillStep` types with tags, depends, invocation flags, markdown content
|
||||
@@ -66,13 +312,45 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
|
||||
- `docs/getting-started/configuration.md` — expanded with skills config sections
|
||||
- `CLAUDE.md` — updated architecture section
|
||||
|
||||
### Examples & Tutorials
|
||||
|
||||
- `examples/openjarvis/spec_search_quickstart.py` — runnable end-to-end
|
||||
LLM-guided spec search session.
|
||||
- `docs/user-guide/llm-guided-spec-search.md` — paper-aligned user guide.
|
||||
- `docs/architecture/learning.md` — Learning primitive deep-dive covering
|
||||
routing, spec search, optimizers, and the orchestrator.
|
||||
- `docs/tutorials/` — code-companion, deep-research, messaging-hub,
|
||||
scheduled-ops, and skills-workflow walkthroughs.
|
||||
- `src/openjarvis/agents/hybrid/registry/*.toml` — 35-cell registry of
|
||||
paradigm × benchmark × model experiments.
|
||||
|
||||
### Migration from 0.x
|
||||
|
||||
- **`learning/distillation/` is now `learning/spec_search/`.** The
|
||||
subsystem was renamed to match the LLM-guided spec search semantics
|
||||
documented in the companion paper. Update any imports
|
||||
(`from openjarvis.learning.distillation.*` →
|
||||
`from openjarvis.learning.spec_search.*`). The `jarvis distillation`
|
||||
CLI command is removed; use `spec_search`-prefixed config keys instead.
|
||||
- **`_third_party.toml` no longer ships default paths.** Set
|
||||
`HERMES_AGENT_PATH` and `OPENCLAW_PATH` env vars to point at your
|
||||
local checkouts before running the framework-comparison harness;
|
||||
missing or empty paths now raise `ThirdPartyNotFoundError` with an
|
||||
actionable hint.
|
||||
- **Engine `generate_full` return shape extended.**
|
||||
`JarvisAgentBackend.generate_full` and `JarvisDirectBackend.generate_full`
|
||||
now return the spec §6.2 extended fields (`energy_joules`,
|
||||
`peak_power_w`, `tool_calls`, `turn_count`, `framework`,
|
||||
`framework_commit`, `error`). Existing callers that didn't read these
|
||||
fields are unaffected; new callers can rely on cross-framework parity.
|
||||
|
||||
### Fixed
|
||||
|
||||
- **Trace metadata flow** — `ToolResult.metadata` now propagates through `TOOL_CALL_END` event to `TraceStep.metadata` (was silently dropped at the event-bus boundary)
|
||||
- **TaintSet JSON serialization** — `ToolExecutor._json_safe_metadata()` filters non-JSON-serializable values (like `TaintSet`) from event payloads before they reach `TraceStore`
|
||||
- **Non-dict YAML frontmatter** — source resolvers handle `yaml.safe_load()` returning a string instead of a dict (discovered on real OpenClaw imports)
|
||||
- **OpenClaw category/name queries** — `jarvis skill install openclaw:owner/slug` now correctly splits into category + name match
|
||||
- **SkillDiscovery trace compatibility** — `_extract_tool_sequence` reads from `step.input["tool"]` (the actual `TraceStep` format), not the nonexistent `step.tool_name` attribute
|
||||
- **LearningOrchestrator skill trigger** — `_maybe_optimize_skills` runs BEFORE the SFT-data short-circuit (skills are tagged via trace metadata, not mined as SFT pairs)
|
||||
- **PinchBenchScorer constructor** — `SkillBenchmarkRunner` constructs `PinchBenchScorer(judge_backend, model)` instead of no-args
|
||||
- **EvalRunner results access** — reads per-task data from `eval_runner.results` property, not nonexistent `summary.results`
|
||||
- **Trace metadata flow** — `ToolResult.metadata` now propagates through `TOOL_CALL_END` event to `TraceStep.metadata` (was silently dropped at the event-bus boundary).
|
||||
- **TaintSet JSON serialization** — `ToolExecutor._json_safe_metadata()` filters non-JSON-serializable values (like `TaintSet`) from event payloads before they reach `TraceStore`.
|
||||
- **Non-dict YAML frontmatter** — source resolvers handle `yaml.safe_load()` returning a string instead of a dict (discovered on real OpenClaw imports).
|
||||
- **OpenClaw category/name queries** — `jarvis skill install openclaw:owner/slug` now correctly splits into category + name match.
|
||||
- **SkillDiscovery trace compatibility** — `_extract_tool_sequence` reads from `step.input["tool"]` (the actual `TraceStep` format), not the nonexistent `step.tool_name` attribute.
|
||||
- **LearningOrchestrator skill trigger** — `_maybe_optimize_skills` runs BEFORE the SFT-data short-circuit (skills are tagged via trace metadata, not mined as SFT pairs).
|
||||
- **PinchBenchScorer constructor** — `SkillBenchmarkRunner` constructs `PinchBenchScorer(judge_backend, model)` instead of no-args.
|
||||
- **EvalRunner results access** — reads per-task data from `eval_runner.results` property, not nonexistent `summary.results`.
|
||||
|
||||
-1
Submodule Inline deleted from 03673aaa42
@@ -0,0 +1,19 @@
|
||||
.PHONY: setup build test lint format
|
||||
|
||||
# Mirrors .github/workflows/ci.yml so `make test` matches CI locally.
|
||||
|
||||
setup:
|
||||
uv sync --extra dev --extra framework-comparison --extra server
|
||||
|
||||
build:
|
||||
uv run maturin develop --manifest-path rust/crates/openjarvis-python/Cargo.toml
|
||||
|
||||
test: build
|
||||
uv run pytest tests/ -n auto -q --tb=short -m "not live and not cloud and not hub"
|
||||
|
||||
lint:
|
||||
uv run ruff check src/ tests/
|
||||
uv run ruff format --check src/ tests/
|
||||
|
||||
format:
|
||||
uv run ruff format src/ tests/
|
||||
@@ -4,19 +4,28 @@
|
||||
<p><i>Personal AI, On Personal Devices.</i></p>
|
||||
|
||||
<p>
|
||||
<a href="https://scalingintelligence.stanford.edu/blogs/openjarvis/"><img src="https://img.shields.io/badge/project-OpenJarvis-blue" alt="Project"></a>
|
||||
<a href="https://openjarvis.stanford.edu/"><img src="https://img.shields.io/badge/project-OpenJarvis-blue" alt="Project"></a>
|
||||
<a href="https://open-jarvis.github.io/OpenJarvis/"><img src="https://img.shields.io/badge/docs-mkdocs-blue" alt="Docs"></a>
|
||||
<img src="https://img.shields.io/badge/python-%3E%3D3.10-blue" alt="Python">
|
||||
<img src="https://img.shields.io/badge/license-Apache%202.0-green" alt="License">
|
||||
<a href="https://discord.gg/YZZRxCAhmm"><img src="https://img.shields.io/badge/discord-join-7289da?logo=discord&logoColor=white" alt="Discord"></a>
|
||||
<a href="https://discord.gg/CMVBmDQ5Fj"><img src="https://img.shields.io/badge/discord-join-7289da?logo=discord&logoColor=white" alt="Discord"></a>
|
||||
<a href="https://x.com/OpenJarvisAI"><img src="https://img.shields.io/badge/X-@OpenJarvisAI-black?logo=x&logoColor=white" alt="X / Twitter"></a>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
---
|
||||
|
||||
<div align="center">
|
||||
<img alt="OpenJarvis demo reel" src="assets/openjarvis_demo_reel.webp" width="75%">
|
||||
</div>
|
||||
|
||||
---
|
||||
|
||||
> **[Documentation](https://open-jarvis.github.io/OpenJarvis/)**
|
||||
>
|
||||
> **[Project Site](https://scalingintelligence.stanford.edu/blogs/openjarvis/)**
|
||||
> **[Project Site](https://openjarvis.stanford.edu/)**
|
||||
>
|
||||
> **[Paper](https://arxiv.org/abs/2605.17172)**
|
||||
>
|
||||
> **[Leaderboard](https://open-jarvis.github.io/OpenJarvis/leaderboard/)**
|
||||
>
|
||||
@@ -26,87 +35,49 @@
|
||||
|
||||
Personal AI agents are exploding in popularity, but nearly all of them still route intelligence through cloud APIs. Your "personal" AI continues to depend on someone else's server. At the same time, our [Intelligence Per Watt](https://www.intelligence-per-watt.ai/) research showed that local language models already handle 88.7% of single-turn chat and reasoning queries, with intelligence efficiency improving 5.3× from 2023 to 2025. The models and hardware are increasingly ready. What has been missing is the software stack to make local-first personal AI practical.
|
||||
|
||||
OpenJarvis is that stack. It is an opinionated framework for local-first personal AI, built around three core ideas: shared primitives for building on-device agents; evaluations that treat energy, FLOPs, latency, and dollar cost as first-class constraints alongside accuracy; and a learning loop that improves models using local trace data. The goal is simple: make it possible to build personal AI agents that run locally by default, calling the cloud only when truly necessary. OpenJarvis aims to be both a research platform and a production foundation for local AI, in the spirit of PyTorch.
|
||||
OpenJarvis is that stack. It is a framework for local-first personal AI, built around three core ideas: shared primitives for building on-device agents; evaluations that treat energy, FLOPs, latency, and dollar cost as first-class constraints alongside accuracy; and a learning loop that improves models using local trace data. The goal is simple: make it possible to build personal AI agents that run locally by default, calling the cloud only when truly necessary. OpenJarvis aims to be both a research platform and a production foundation for local AI, in the spirit of PyTorch.
|
||||
|
||||
## Installation
|
||||
|
||||
### Prerequisites
|
||||
Pick your platform and run one command. Each installer handles [uv](https://docs.astral.sh/uv/), the Python venv, Ollama, and a starter model — about 3 minutes on broadband.
|
||||
|
||||
| Tool | Install |
|
||||
|------|---------|
|
||||
| **Python 3.10+** | [python.org](https://www.python.org/downloads/) |
|
||||
| **uv** (Python package manager) | `curl -LsSf https://astral.sh/uv/install.sh \| sh` — or `brew install uv` on macOS |
|
||||
| **Rust** | `curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs \| sh` |
|
||||
| **Git** | [git-scm.com](https://git-scm.com/) — or `brew install git` on macOS |
|
||||
| Platform | One-liner |
|
||||
|---|---|
|
||||
| **macOS · Linux · WSL2** | `curl -fsSL https://open-jarvis.github.io/OpenJarvis/install.sh \| bash` |
|
||||
| **Native Windows** | `irm https://open-jarvis.github.io/OpenJarvis/install.ps1 \| iex` |
|
||||
| **Desktop GUI** | Download `.exe` / `.dmg` / `.deb` / `.rpm` / `.AppImage` from the [latest release](https://github.com/open-jarvis/OpenJarvis/releases) |
|
||||
|
||||
> **macOS users:** see the full [macOS Installation Guide](https://open-jarvis.github.io/OpenJarvis/getting-started/macos/) for a step-by-step walkthrough including Homebrew setup.
|
||||
Then `jarvis` to start. The Rust extension and larger models continue downloading in the background; `jarvis doctor` shows status.
|
||||
|
||||
### Setup
|
||||
|
||||
```bash
|
||||
git clone https://github.com/open-jarvis/OpenJarvis.git
|
||||
cd OpenJarvis
|
||||
uv sync # core framework
|
||||
uv sync --extra server # + FastAPI server
|
||||
|
||||
# Build the Rust extension
|
||||
uv run maturin develop -m rust/crates/openjarvis-python/Cargo.toml
|
||||
```
|
||||
|
||||
> **Python 3.14+:** set `PYO3_USE_ABI3_FORWARD_COMPATIBILITY=1` before the `maturin` command.
|
||||
|
||||
You also need a local inference backend: [Ollama](https://ollama.com), [vLLM](https://github.com/vllm-project/vllm), [SGLang](https://github.com/sgl-project/sglang), or [llama.cpp](https://github.com/ggerganov/llama.cpp). Alternatively, use the `cloud` engine with [OpenAI](https://openai.com), [Anthropic](https://anthropic.com), [Google Gemini](https://ai.google.dev), [OpenRouter](https://openrouter.ai), or [MiniMax](https://www.minimax.io) by setting the corresponding API key environment variable.
|
||||
Platform-specific notes (WSL2 setup, native-Windows scheduled-task service, desktop prerequisites, manual / contributor install): see the [installation docs](https://open-jarvis.github.io/OpenJarvis/getting-started/install/).
|
||||
|
||||
## Quick Start
|
||||
|
||||
```bash
|
||||
# 1. Install and detect hardware
|
||||
git clone https://github.com/open-jarvis/OpenJarvis.git
|
||||
cd OpenJarvis
|
||||
uv sync
|
||||
uv run jarvis init
|
||||
|
||||
# 2. Start Ollama and pull a model
|
||||
curl -fsSL https://ollama.com/install.sh | sh
|
||||
ollama serve &
|
||||
ollama pull qwen3:8b
|
||||
|
||||
# 3. Ask a question
|
||||
uv run jarvis ask "What is the capital of France?"
|
||||
jarvis # start chatting (default: chat-simple)
|
||||
jarvis init --preset <name> # switch to a starter config
|
||||
```
|
||||
|
||||
`jarvis init` auto-detects your hardware and recommends the best engine. Run `uv run jarvis doctor` at any time to diagnose issues.
|
||||
> Prefix `jarvis ...` with `uv run`, or `source .venv/bin/activate` first.
|
||||
|
||||
## Starter Configs
|
||||
| Preset | What it does |
|
||||
|---|---|
|
||||
| `morning-digest-mac` / `morning-digest-linux` / `morning-digest-minimal` | Spoken daily briefing from email, calendar, health, news |
|
||||
| `deep-research` | Multi-hop research across indexed docs with citations |
|
||||
| `code-assistant` | Agent with code execution, file I/O, and shell access |
|
||||
| `scheduled-monitor` | Stateful agent on a schedule with memory |
|
||||
| `chat-simple` | Lightweight conversation, no tools |
|
||||
|
||||
Install any preset with one command:
|
||||
Example:
|
||||
|
||||
```bash
|
||||
jarvis init --preset morning-digest-mac # or any preset below
|
||||
```
|
||||
|
||||
| Preset | Use Case | What it does |
|
||||
|--------|----------|-------------|
|
||||
| `morning-digest-mac` | Daily Briefing (Mac) | Spoken briefing from email, calendar, health, news with Jarvis voice |
|
||||
| `morning-digest-linux` | Daily Briefing (Linux) | Same, with vLLM support for GPU servers |
|
||||
| `morning-digest-minimal` | Daily Briefing (minimal) | Just Gmail + Calendar, runs on any machine |
|
||||
| `deep-research` | Research Assistant | Multi-hop research across indexed docs with citations |
|
||||
| `code-assistant` | Code Companion | Agent with code execution, file I/O, and shell access |
|
||||
| `scheduled-monitor` | Persistent Monitor | Stateful agent that runs on a schedule with memory |
|
||||
| `chat-simple` | Simple Chat | Lightweight conversation, no tools needed |
|
||||
|
||||
```bash
|
||||
# Example: Morning Digest on Mac
|
||||
jarvis init --preset morning-digest-mac
|
||||
jarvis connect gdrive # one OAuth flow covers Gmail, Calendar, Tasks
|
||||
jarvis digest --fresh # generate and play your first briefing
|
||||
|
||||
# Example: Deep Research
|
||||
jarvis init --preset deep-research
|
||||
jarvis memory index ./docs/ # index your documents
|
||||
jarvis ask "Summarize all emails about Project X"
|
||||
jarvis connect gdrive # one OAuth covers Gmail / Calendar / Tasks
|
||||
jarvis digest --fresh # generate and play your first briefing
|
||||
```
|
||||
|
||||
Per-preset deep dives: [morning digest](https://open-jarvis.github.io/OpenJarvis/user-guide/morning-digest/) · [deep research](https://open-jarvis.github.io/OpenJarvis/user-guide/deep-research/) · [code assistant](https://open-jarvis.github.io/OpenJarvis/user-guide/code-assistant/) · [scheduled monitor](https://open-jarvis.github.io/OpenJarvis/user-guide/scheduled-monitor/) · [chat simple](https://open-jarvis.github.io/OpenJarvis/user-guide/chat-simple/) · or the full [quickstart guide](https://open-jarvis.github.io/OpenJarvis/getting-started/quickstart/).
|
||||
|
||||
### Skills
|
||||
|
||||
Skills teach agents how to better use tools and improve their reasoning. Every skill is a tool — agents discover them from a catalog and invoke them on demand.
|
||||
@@ -132,6 +103,8 @@ See the [Skills User Guide](https://open-jarvis.github.io/OpenJarvis/user-guide/
|
||||
|
||||
### Built-in Agents
|
||||
|
||||
OpenJarvis ships with eight built-in agents across three execution modes (on-demand, scheduled, continuous):
|
||||
|
||||
| Agent | Type | What it does |
|
||||
|-------|------|-------------|
|
||||
| `morning_digest` | Scheduled | Daily briefing from email, calendar, health, news — with TTS audio |
|
||||
@@ -147,6 +120,13 @@ See the [User Guide](https://open-jarvis.github.io/OpenJarvis/user-guide/morning
|
||||
|
||||
Full documentation — including Docker deployment, cloud engines, development setup, and tutorials — at **[open-jarvis.github.io/OpenJarvis](https://open-jarvis.github.io/OpenJarvis/)**.
|
||||
|
||||
## Community
|
||||
|
||||
- **GitHub:** [github.com/open-jarvis/OpenJarvis](https://github.com/open-jarvis/OpenJarvis)
|
||||
- **Discord:** [discord.gg/CMVBmDQ5Fj](https://discord.gg/CMVBmDQ5Fj)
|
||||
- **X / Twitter:** [@OpenJarvisAI](https://x.com/OpenJarvisAI)
|
||||
- **Docs:** [open-jarvis.github.io/OpenJarvis](https://open-jarvis.github.io/OpenJarvis/)
|
||||
|
||||
## Contributing
|
||||
|
||||
We welcome contributions! See the [Contributing Guide](CONTRIBUTING.md) for incentives, contribution types, and the PR process.
|
||||
@@ -165,7 +145,7 @@ Browse the [Roadmap](https://open-jarvis.github.io/OpenJarvis/development/roadma
|
||||
|
||||
## About
|
||||
|
||||
OpenJarvis is part of [Intelligence Per Watt](https://www.intelligence-per-watt.ai/), a research initiative studying the efficiency of on-device AI systems. The project is developed at [Hazy Research](https://hazyresearch.stanford.edu/) and the [Scaling Intelligence Lab](https://scalingintelligence.stanford.edu/) at [Stanford SAIL](https://ai.stanford.edu/).
|
||||
OpenJarvis is part of [Intelligence Per Watt](https://www.intelligence-per-watt.ai/), a research initiative studying the intelligence efficiency of AI systems. The project is developed at [Hazy Research](https://hazyresearch.stanford.edu/) and the [Scaling Intelligence Lab](https://scalingintelligence.stanford.edu/) at [Stanford SAIL](https://ai.stanford.edu/).
|
||||
|
||||
## Sponsors
|
||||
|
||||
@@ -181,11 +161,14 @@ OpenJarvis is part of [Intelligence Per Watt](https://www.intelligence-per-watt.
|
||||
|
||||
## Citation
|
||||
```bibtex
|
||||
@misc{saadfalcon2026openjarvis,
|
||||
title={OpenJarvis: Personal AI, On Personal Devices},
|
||||
author={Jon Saad-Falcon and Avanika Narayan and Herumb Shandilya and Hakki Orhun Akengin and Robby Manihani and Gabriel Bo and John Hennessy and Christopher R\'{e} and Azalia Mirhoseini},
|
||||
year={2026},
|
||||
howpublished={\url{https://scalingintelligence.stanford.edu/blogs/openjarvis/}},
|
||||
@misc{saadfalcon2026openjarvispersonalaipersonal,
|
||||
title={OpenJarvis: Personal AI, On Personal Devices},
|
||||
author={Jon Saad-Falcon and Avanika Narayan and Robby Manihani and Tanvir Bhathal and Herumb Shandilya and Hakki Orhun Akengin and Gabriel Bo and Andrew Park and Matthew Hart and Caia Costello and Chuan Li and Christopher Ré and Azalia Mirhoseini},
|
||||
year={2026},
|
||||
eprint={2605.17172},
|
||||
archivePrefix={arXiv},
|
||||
primaryClass={cs.LG},
|
||||
url={https://arxiv.org/abs/2605.17172},
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
@@ -17,6 +17,7 @@ Check for logic errors, edge cases, and off-by-one errors. Pay particular attent
|
||||
- **Rust-Python bridge (PyO3) boundaries** — type conversions, error propagation, GIL handling
|
||||
- **Async/await patterns** — missing awaits, unclosed resources, blocking calls in async contexts
|
||||
- **Registry pattern compliance** — new components (engines, tools, agents, channels) must register via `ToolRegistry`, `EngineRegistry`, `AgentRegistry`, `ChannelRegistry`, etc. in `src/openjarvis/core/registry.py`
|
||||
- **Mining provider compliance** — new mining providers must register via `MinerRegistry` and expose an idempotent `ensure_registered()` for the autouse-clear test convention
|
||||
- **Event bus integration** — new lifecycle events should use `EventBus` from `src/openjarvis/core/events.py`
|
||||
|
||||
### 4. Testing
|
||||
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 4.5 MiB |
@@ -106,9 +106,11 @@ enabled = true # Record traces for analysis
|
||||
db_path = "~/.openjarvis/traces.db"
|
||||
|
||||
[server]
|
||||
host = "0.0.0.0"
|
||||
# Bind to loopback by default so the API is not exposed to the local network.
|
||||
# To serve other devices on your LAN, set host = "0.0.0.0" AND set an API key
|
||||
# (OPENJARVIS_API_KEY / `jarvis auth generate-key`) — startup refuses a
|
||||
# non-loopback bind without a key. The "server" security profile also flips
|
||||
# this to 0.0.0.0 intentionally.
|
||||
host = "127.0.0.1"
|
||||
port = 8000
|
||||
agent = "native_openhands"
|
||||
|
||||
[security]
|
||||
enabled = false # Disable for eval (no PII scanning overhead)
|
||||
|
||||
@@ -0,0 +1,76 @@
|
||||
# LLM-Guided Spec Search — quickstart configuration
|
||||
# Copy to ~/.openjarvis/config.toml and run:
|
||||
# python -m openjarvis_examples.spec_search_quickstart
|
||||
#
|
||||
# This config has two parts:
|
||||
# 1. The agent system being optimized (intelligence / engine / agent / tools).
|
||||
# Same schema as the other examples in this directory; parsed by
|
||||
# ``openjarvis.core.config.load_config``.
|
||||
# 2. ``[learning.spec_search]`` and its sub-tables — the search hyperparameters
|
||||
# consumed by ``SpecSearchOrchestrator.from_config`` and ``SpecSearchLoop``.
|
||||
#
|
||||
# Defaults below match the paper (Saad-Falcon et al., 2026):
|
||||
# - max_regression = 0.01 (epsilon in GateOK)
|
||||
# - stagnation_k = 5 (Algorithm 1 stopping rule)
|
||||
# - composite_reward weights (alpha, beta, gamma, delta) = (0.5, 0.1, 0.1, 0.3)
|
||||
#
|
||||
# Teacher API keys come from your environment / credentials store, not this file.
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Agent system being optimized
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
[engine]
|
||||
default = "ollama" # swap to "vllm" on H100/RTX 6000 / DGX Spark
|
||||
|
||||
[intelligence]
|
||||
default_model = "qwen3.5:9b" # the local student
|
||||
# default_model = "qwen3.5:27b-fp8" # workstation tier
|
||||
|
||||
[agent]
|
||||
default_agent = "orchestrator" # multi-turn, tool-using
|
||||
max_turns = 10
|
||||
|
||||
[tools]
|
||||
enabled = [
|
||||
"code_interpreter",
|
||||
"file_read",
|
||||
"web_search",
|
||||
"think",
|
||||
"calculator",
|
||||
]
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# LLM-guided spec search hyperparameters (paper §3.3, Algorithm 1)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
[learning.spec_search]
|
||||
enabled = true
|
||||
teacher_model = "claude-opus-4-6" # frontier proposer
|
||||
teacher_engine = "cloud" # CloudEngine registry key (uses LiteLLM)
|
||||
autonomy_mode = "tiered" # auto | tiered | manual
|
||||
|
||||
# Per-session bounds (one diagnose / plan / execute / record pass)
|
||||
min_traces = 20
|
||||
max_cost_per_session_usd = 5.0
|
||||
max_tool_calls_per_diagnosis = 30
|
||||
|
||||
# Multi-session loop (paper Algorithm 1 stopping)
|
||||
stagnation_k = 5 # stop after this many sessions with no gate-score gain
|
||||
stagnation_eps = 0.001 # delta below this counts as no improvement
|
||||
max_total_cost_usd = 50.0 # cumulative teacher-cost budget across all sessions
|
||||
|
||||
# GateOK predicate — accept iff target cluster improves AND every other cluster
|
||||
# regresses by at most max_regression (epsilon in the paper).
|
||||
max_regression = 0.01 # paper default: 1%
|
||||
min_improvement = 0.0
|
||||
benchmark_subsample_size = 50
|
||||
benchmark_version = "personal_v1"
|
||||
|
||||
# Composite reward (paper Eq. 1) — used only when an Intelligence edit triggers
|
||||
# LoRA / GRPO training inside an accepted edit. The held-out gate is unaffected.
|
||||
[learning.spec_search.composite_reward]
|
||||
alpha = 0.5 # accuracy weight
|
||||
beta = 0.1 # energy penalty
|
||||
gamma = 0.1 # latency penalty
|
||||
delta = 0.3 # cost penalty
|
||||
@@ -0,0 +1,7 @@
|
||||
# Copy to `.env` in this directory (deploy/docker/.env) before `docker compose up`.
|
||||
# docker-compose.yml requires this — the container binds 0.0.0.0, so the
|
||||
# server refuses to start without an API key.
|
||||
#
|
||||
# Generate a key with: jarvis auth generate-key
|
||||
# Then clients must send: Authorization: Bearer <key>
|
||||
OPENJARVIS_API_KEY=
|
||||
@@ -1,32 +1,83 @@
|
||||
# Base images are pinned to an immutable digest (in addition to a human-readable
|
||||
# tag) so every build resolves the exact same layers — reproducible builds and
|
||||
# safe rollbacks (#563).
|
||||
|
||||
# Stage 1: Build frontend SPA
|
||||
FROM node:22-slim AS frontend
|
||||
FROM node:22.23.0-slim@sha256:d9f850096136edbc402debdd8729579a288aac64574ada0ff4db26b6ae58b0b2 AS frontend
|
||||
# Public Supabase anon key for the savings leaderboard; empty by default so
|
||||
# the image's leaderboard stays disabled (#589). Pass --build-arg to enable.
|
||||
ARG OPENJARVIS_LEADERBOARD_PUBLIC_ANON=
|
||||
|
||||
WORKDIR /frontend
|
||||
COPY frontend/package.json frontend/package-lock.json* ./
|
||||
RUN npm ci --ignore-scripts 2>/dev/null || npm install
|
||||
COPY frontend/ .
|
||||
RUN npm run build
|
||||
RUN VITE_SUPABASE_ANON_KEY="${OPENJARVIS_LEADERBOARD_PUBLIC_ANON}" npm run build
|
||||
|
||||
# Stage 2: Build Python package
|
||||
FROM python:3.12-slim-bookworm AS builder
|
||||
FROM python:3.12.13-slim-bookworm@sha256:76d4b7b6305788c6b4c6a19d6a22a3921bf802e9af4d5e1e5bd771208dba74bf AS builder
|
||||
|
||||
RUN apt-get update && \
|
||||
apt-get install -y --no-install-recommends build-essential ca-certificates curl && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
ENV PATH="/root/.cargo/bin:${PATH}"
|
||||
|
||||
RUN curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | \
|
||||
sh -s -- -y --profile minimal --default-toolchain none && \
|
||||
rustup toolchain install 1.88 --profile minimal && \
|
||||
rustup default 1.88
|
||||
|
||||
WORKDIR /app
|
||||
COPY pyproject.toml README.md ./
|
||||
|
||||
# Install dependencies from the committed lockfile (#567). `uv export --frozen`
|
||||
# reads uv.lock as-is (no re-resolution) and emits a fully pinned, hash-verified
|
||||
# requirements set; `--no-deps` then installs exactly that set. This is a
|
||||
# separate layer from the source copy so dependency installs stay cached when
|
||||
# only application code changes.
|
||||
COPY pyproject.toml uv.lock README.md ./
|
||||
RUN pip install --no-cache-dir uv && \
|
||||
uv export --frozen --no-dev --extra server --no-emit-project > requirements.txt && \
|
||||
uv pip install --system --no-deps -r requirements.txt && \
|
||||
uv pip install --system --no-deps "maturin>=1.12.6,<2"
|
||||
|
||||
# Copy the source and the non-src force-include paths (see pyproject
|
||||
# [tool.hatch.build.targets.wheel.force-include]) before building the project.
|
||||
COPY src/ src/
|
||||
COPY rust/ rust/
|
||||
COPY scripts/install scripts/install
|
||||
COPY deploy/windows deploy/windows
|
||||
|
||||
# Copy built frontend into the server static directory
|
||||
COPY --from=frontend /src/openjarvis/server/static src/openjarvis/server/static/
|
||||
|
||||
RUN pip install --no-cache-dir uv && \
|
||||
uv pip install --system ".[server]"
|
||||
# Install the project itself without re-resolving dependencies.
|
||||
RUN uv pip install --system --no-deps . && \
|
||||
maturin build --release \
|
||||
--manifest-path rust/crates/openjarvis-python/Cargo.toml \
|
||||
--interpreter python3 \
|
||||
--out /tmp/openjarvis-rust-wheel && \
|
||||
uv pip install --system --no-deps /tmp/openjarvis-rust-wheel/*.whl && \
|
||||
python3 -c "import openjarvis_rust; print('openjarvis_rust ok')" && \
|
||||
python3 -m pip uninstall -y maturin && \
|
||||
rm -rf /tmp/openjarvis-rust-wheel rust
|
||||
|
||||
# Stage 3: Runtime
|
||||
FROM python:3.12-slim-bookworm
|
||||
FROM python:3.12.13-slim-bookworm@sha256:76d4b7b6305788c6b4c6a19d6a22a3921bf802e9af4d5e1e5bd771208dba74bf
|
||||
|
||||
COPY --from=builder /usr/local /usr/local
|
||||
COPY --from=builder /app /app
|
||||
WORKDIR /app
|
||||
|
||||
# Run as an unprivileged user — the server needs no root privileges, so dropping
|
||||
# them limits the blast radius of a compromise (#565). The app writes only to
|
||||
# $HOME (config/cache/state), which is owned by this user.
|
||||
RUN groupadd --system --gid 10001 openjarvis && \
|
||||
useradd --system --uid 10001 --gid openjarvis \
|
||||
--create-home --home-dir /home/openjarvis openjarvis
|
||||
ENV HOME=/home/openjarvis
|
||||
USER openjarvis
|
||||
|
||||
EXPOSE 8000
|
||||
|
||||
ENTRYPOINT ["jarvis"]
|
||||
|
||||
@@ -1,30 +1,69 @@
|
||||
# Base images are pinned to an immutable digest (in addition to a human-readable
|
||||
# tag) so every build resolves the exact same layers — reproducible builds and
|
||||
# safe rollbacks (#563).
|
||||
|
||||
# Stage 1: Build frontend SPA
|
||||
FROM node:22-slim AS frontend
|
||||
FROM node:22.23.0-slim@sha256:d9f850096136edbc402debdd8729579a288aac64574ada0ff4db26b6ae58b0b2 AS frontend
|
||||
# Public Supabase anon key for the savings leaderboard; empty by default so
|
||||
# the image's leaderboard stays disabled (#589). Pass --build-arg to enable.
|
||||
ARG OPENJARVIS_LEADERBOARD_PUBLIC_ANON=
|
||||
|
||||
WORKDIR /frontend
|
||||
COPY frontend/package.json frontend/package-lock.json* ./
|
||||
RUN npm ci --ignore-scripts 2>/dev/null || npm install
|
||||
COPY frontend/ .
|
||||
RUN npm run build
|
||||
RUN VITE_SUPABASE_ANON_KEY="${OPENJARVIS_LEADERBOARD_PUBLIC_ANON}" npm run build
|
||||
|
||||
# Stage 2: Build Python package (NVIDIA CUDA 12.4)
|
||||
FROM nvidia/cuda:12.4.0-runtime-ubuntu22.04 AS builder
|
||||
FROM nvidia/cuda:12.4.0-runtime-ubuntu22.04@sha256:af8bd179ed3bf69d4b63b19a763662a6141f0f62ef099283f68d0b14b4bab0e3 AS builder
|
||||
|
||||
RUN apt-get update && \
|
||||
apt-get install -y --no-install-recommends python3 python3-pip python3-venv && \
|
||||
apt-get install -y --no-install-recommends \
|
||||
build-essential \
|
||||
ca-certificates \
|
||||
curl \
|
||||
python3 \
|
||||
python3-dev \
|
||||
python3-pip \
|
||||
python3-venv && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
ENV PATH="/root/.cargo/bin:${PATH}"
|
||||
|
||||
RUN curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | \
|
||||
sh -s -- -y --profile minimal --default-toolchain none && \
|
||||
rustup toolchain install 1.88 --profile minimal && \
|
||||
rustup default 1.88
|
||||
|
||||
WORKDIR /app
|
||||
COPY pyproject.toml README.md ./
|
||||
|
||||
# Install dependencies from the committed lockfile (#567). See deploy/docker/Dockerfile
|
||||
# for the rationale behind the frozen export + --no-deps install.
|
||||
COPY pyproject.toml uv.lock README.md ./
|
||||
RUN pip install --no-cache-dir uv && \
|
||||
uv export --frozen --no-dev --extra server --no-emit-project > requirements.txt && \
|
||||
uv pip install --system --no-deps -r requirements.txt && \
|
||||
uv pip install --system --no-deps "maturin>=1.12.6,<2"
|
||||
|
||||
COPY src/ src/
|
||||
COPY rust/ rust/
|
||||
COPY scripts/install scripts/install
|
||||
COPY deploy/windows deploy/windows
|
||||
|
||||
COPY --from=frontend /src/openjarvis/server/static src/openjarvis/server/static/
|
||||
|
||||
RUN pip install --no-cache-dir uv && \
|
||||
uv pip install --system ".[server]"
|
||||
RUN uv pip install --system --no-deps . && \
|
||||
maturin build --release \
|
||||
--manifest-path rust/crates/openjarvis-python/Cargo.toml \
|
||||
--interpreter python3 \
|
||||
--out /tmp/openjarvis-rust-wheel && \
|
||||
uv pip install --system --no-deps /tmp/openjarvis-rust-wheel/*.whl && \
|
||||
python3 -c "import openjarvis_rust; print('openjarvis_rust ok')" && \
|
||||
python3 -m pip uninstall -y maturin && \
|
||||
rm -rf /tmp/openjarvis-rust-wheel rust
|
||||
|
||||
# Stage 3: Runtime
|
||||
FROM nvidia/cuda:12.4.0-runtime-ubuntu22.04
|
||||
FROM nvidia/cuda:12.4.0-runtime-ubuntu22.04@sha256:af8bd179ed3bf69d4b63b19a763662a6141f0f62ef099283f68d0b14b4bab0e3
|
||||
|
||||
RUN apt-get update && \
|
||||
apt-get install -y --no-install-recommends python3 python3-pip && \
|
||||
@@ -34,6 +73,14 @@ COPY --from=builder /usr/local /usr/local
|
||||
COPY --from=builder /app /app
|
||||
WORKDIR /app
|
||||
|
||||
# Run as an unprivileged user (#565). NVIDIA device nodes (/dev/nvidia*) are
|
||||
# world-accessible, so GPU workloads do not require root.
|
||||
RUN groupadd --system --gid 10001 openjarvis && \
|
||||
useradd --system --uid 10001 --gid openjarvis \
|
||||
--create-home --home-dir /home/openjarvis openjarvis
|
||||
ENV HOME=/home/openjarvis
|
||||
USER openjarvis
|
||||
|
||||
EXPOSE 8000
|
||||
|
||||
ENTRYPOINT ["jarvis"]
|
||||
|
||||
@@ -1,30 +1,69 @@
|
||||
# Base images are pinned to an immutable digest (in addition to a human-readable
|
||||
# tag) so every build resolves the exact same layers — reproducible builds and
|
||||
# safe rollbacks (#563).
|
||||
|
||||
# Stage 1: Build frontend SPA
|
||||
FROM node:22-slim AS frontend
|
||||
FROM node:22.23.0-slim@sha256:d9f850096136edbc402debdd8729579a288aac64574ada0ff4db26b6ae58b0b2 AS frontend
|
||||
# Public Supabase anon key for the savings leaderboard; empty by default so
|
||||
# the image's leaderboard stays disabled (#589). Pass --build-arg to enable.
|
||||
ARG OPENJARVIS_LEADERBOARD_PUBLIC_ANON=
|
||||
|
||||
WORKDIR /frontend
|
||||
COPY frontend/package.json frontend/package-lock.json* ./
|
||||
RUN npm ci --ignore-scripts 2>/dev/null || npm install
|
||||
COPY frontend/ .
|
||||
RUN npm run build
|
||||
RUN VITE_SUPABASE_ANON_KEY="${OPENJARVIS_LEADERBOARD_PUBLIC_ANON}" npm run build
|
||||
|
||||
# Stage 2: Build Python package (AMD ROCm 7.2)
|
||||
FROM rocm/dev-ubuntu-22.04:7.2 AS builder
|
||||
FROM rocm/dev-ubuntu-22.04:7.2@sha256:05af5f04a06b04676d4c7438997d0deadaeb7478961ad621376e199bf3aeb644 AS builder
|
||||
|
||||
RUN apt-get update && \
|
||||
apt-get install -y --no-install-recommends python3 python3-pip python3-venv && \
|
||||
apt-get install -y --no-install-recommends \
|
||||
build-essential \
|
||||
ca-certificates \
|
||||
curl \
|
||||
python3 \
|
||||
python3-dev \
|
||||
python3-pip \
|
||||
python3-venv && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
ENV PATH="/root/.cargo/bin:${PATH}"
|
||||
|
||||
RUN curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | \
|
||||
sh -s -- -y --profile minimal --default-toolchain none && \
|
||||
rustup toolchain install 1.88 --profile minimal && \
|
||||
rustup default 1.88
|
||||
|
||||
WORKDIR /app
|
||||
COPY pyproject.toml README.md ./
|
||||
|
||||
# Install dependencies from the committed lockfile (#567). See deploy/docker/Dockerfile
|
||||
# for the rationale behind the frozen export + --no-deps install.
|
||||
COPY pyproject.toml uv.lock README.md ./
|
||||
RUN pip install --no-cache-dir uv && \
|
||||
uv export --frozen --no-dev --extra server --no-emit-project > requirements.txt && \
|
||||
uv pip install --system --no-deps -r requirements.txt && \
|
||||
uv pip install --system --no-deps "maturin>=1.12.6,<2"
|
||||
|
||||
COPY src/ src/
|
||||
COPY rust/ rust/
|
||||
COPY scripts/install scripts/install
|
||||
COPY deploy/windows deploy/windows
|
||||
|
||||
COPY --from=frontend /src/openjarvis/server/static src/openjarvis/server/static/
|
||||
|
||||
RUN pip install --no-cache-dir uv && \
|
||||
uv pip install --system ".[server]"
|
||||
RUN uv pip install --system --no-deps . && \
|
||||
maturin build --release \
|
||||
--manifest-path rust/crates/openjarvis-python/Cargo.toml \
|
||||
--interpreter python3 \
|
||||
--out /tmp/openjarvis-rust-wheel && \
|
||||
uv pip install --system --no-deps /tmp/openjarvis-rust-wheel/*.whl && \
|
||||
python3 -c "import openjarvis_rust; print('openjarvis_rust ok')" && \
|
||||
python3 -m pip uninstall -y maturin && \
|
||||
rm -rf /tmp/openjarvis-rust-wheel rust
|
||||
|
||||
# Stage 3: Runtime
|
||||
FROM rocm/dev-ubuntu-22.04:7.2
|
||||
FROM rocm/dev-ubuntu-22.04:7.2@sha256:05af5f04a06b04676d4c7438997d0deadaeb7478961ad621376e199bf3aeb644
|
||||
|
||||
RUN apt-get update && \
|
||||
apt-get install -y --no-install-recommends python3 python3-pip && \
|
||||
@@ -34,6 +73,18 @@ COPY --from=builder /usr/local /usr/local
|
||||
COPY --from=builder /app /app
|
||||
WORKDIR /app
|
||||
|
||||
# Run as an unprivileged user (#565). ROCm GPU access is gated by the `video` and
|
||||
# `render` groups (see group_add in docker-compose.gpu.rocm.yml), so the user is
|
||||
# added to both; root is not required.
|
||||
RUN groupadd --system --gid 10001 openjarvis && \
|
||||
useradd --system --uid 10001 --gid openjarvis \
|
||||
--create-home --home-dir /home/openjarvis openjarvis && \
|
||||
(getent group video >/dev/null || groupadd --system video) && \
|
||||
(getent group render >/dev/null || groupadd --system render) && \
|
||||
usermod -aG video,render openjarvis
|
||||
ENV HOME=/home/openjarvis
|
||||
USER openjarvis
|
||||
|
||||
EXPOSE 8000
|
||||
|
||||
ENTRYPOINT ["jarvis"]
|
||||
|
||||
@@ -1,15 +1,69 @@
|
||||
FROM python:3.12-slim
|
||||
# Base images are pinned to an immutable digest (in addition to a human-readable
|
||||
# tag) so every build resolves the exact same layers (#563).
|
||||
|
||||
# Install Node.js 22
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
curl ca-certificates && \
|
||||
curl -fsSL https://deb.nodesource.com/setup_22.x | bash - && \
|
||||
apt-get install -y nodejs && \
|
||||
# Node.js is sourced from the official, digest-pinned image rather than piping a
|
||||
# remote setup script into bash (`curl ... | bash -`), which performed no
|
||||
# checksum or signature verification of the downloaded installer (#566). The
|
||||
# image digest is the integrity check, and the copy is architecture-agnostic.
|
||||
FROM node:22.23.0-slim@sha256:d9f850096136edbc402debdd8729579a288aac64574ada0ff4db26b6ae58b0b2 AS node
|
||||
|
||||
FROM python:3.12.13-slim-bookworm@sha256:76d4b7b6305788c6b4c6a19d6a22a3921bf802e9af4d5e1e5bd771208dba74bf AS builder
|
||||
|
||||
RUN apt-get update && \
|
||||
apt-get install -y --no-install-recommends build-essential ca-certificates curl && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
ENV PATH="/root/.cargo/bin:${PATH}"
|
||||
|
||||
RUN curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | \
|
||||
sh -s -- -y --profile minimal --default-toolchain none && \
|
||||
rustup toolchain install 1.88 --profile minimal && \
|
||||
rustup default 1.88
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
# Install dependencies from the committed lockfile (#567): `uv export --frozen`
|
||||
# reads uv.lock as-is and emits a pinned, hash-verified set installed with
|
||||
# --no-deps (no re-resolution). Copied first so this layer caches independently
|
||||
# of application source.
|
||||
COPY pyproject.toml uv.lock README.md ./
|
||||
RUN pip install --no-cache-dir uv && \
|
||||
uv export --frozen --no-dev --extra server --no-emit-project > requirements.txt && \
|
||||
uv pip install --system --no-deps -r requirements.txt && \
|
||||
uv pip install --system --no-deps "maturin>=1.12.6,<2"
|
||||
|
||||
COPY . .
|
||||
RUN pip install --no-cache-dir ".[server]"
|
||||
|
||||
# Install the project itself without re-resolving dependencies.
|
||||
RUN uv pip install --system --no-deps . && \
|
||||
maturin build --release \
|
||||
--manifest-path rust/crates/openjarvis-python/Cargo.toml \
|
||||
--interpreter python3 \
|
||||
--out /tmp/openjarvis-rust-wheel && \
|
||||
uv pip install --system --no-deps /tmp/openjarvis-rust-wheel/*.whl && \
|
||||
python3 -c "import openjarvis_rust; print('openjarvis_rust ok')" && \
|
||||
python3 -m pip uninstall -y maturin && \
|
||||
rm -rf /tmp/openjarvis-rust-wheel rust/target
|
||||
|
||||
FROM python:3.12.13-slim-bookworm@sha256:76d4b7b6305788c6b4c6a19d6a22a3921bf802e9af4d5e1e5bd771208dba74bf
|
||||
|
||||
# libstdc++6 + ca-certificates are the only runtime requirements of the Node
|
||||
# binary copied below (the python slim image already provides libc/libgcc).
|
||||
RUN apt-get update && \
|
||||
apt-get install -y --no-install-recommends ca-certificates libstdc++6 && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
COPY --from=builder /usr/local /usr/local
|
||||
COPY --from=builder /app /app
|
||||
|
||||
# Transplant the Node.js runtime from the official image. Both images are Debian
|
||||
# bookworm, so the glibc/libstdc++ ABI matches.
|
||||
COPY --from=node /usr/local/bin/node /usr/local/bin/node
|
||||
COPY --from=node /usr/local/lib/node_modules /usr/local/lib/node_modules
|
||||
RUN ln -sf /usr/local/lib/node_modules/npm/bin/npm-cli.js /usr/local/bin/npm && \
|
||||
ln -sf /usr/local/lib/node_modules/npm/bin/npx-cli.js /usr/local/bin/npx
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
LABEL openjarvis-sandbox=true
|
||||
|
||||
|
||||
@@ -18,7 +18,9 @@ services:
|
||||
capabilities: [gpu]
|
||||
|
||||
ollama:
|
||||
image: ollama/ollama:latest
|
||||
# Pinned to a fixed version + digest for reproducible deployments (#563);
|
||||
# must match the tag in docker-compose.yml.
|
||||
image: ollama/ollama:0.30.10@sha256:bfc9c6d53cc6989aa5131a6fde6b162b2802d4d337657f3253b5f69579bddeee
|
||||
environment:
|
||||
- NVIDIA_VISIBLE_DEVICES=all
|
||||
- NVIDIA_DRIVER_CAPABILITIES=compute,utility
|
||||
|
||||
@@ -8,13 +8,19 @@ services:
|
||||
environment:
|
||||
- OPENJARVIS_ENGINE_DEFAULT=ollama
|
||||
- OLLAMA_HOST=http://ollama:11434
|
||||
# The container binds 0.0.0.0, so an API key is REQUIRED. Compose fails
|
||||
# fast if OPENJARVIS_API_KEY is unset (set it in deploy/docker/.env —
|
||||
# see .env.example, or `export` it). Generate one: `jarvis auth generate-key`.
|
||||
- OPENJARVIS_API_KEY=${OPENJARVIS_API_KEY:?OPENJARVIS_API_KEY must be set (see deploy/docker/.env.example)}
|
||||
depends_on:
|
||||
ollama:
|
||||
condition: service_healthy
|
||||
restart: unless-stopped
|
||||
|
||||
ollama:
|
||||
image: ollama/ollama:latest
|
||||
# Pinned to a fixed version + digest for reproducible deployments and
|
||||
# predictable rollbacks (#563). Bump deliberately, not implicitly via :latest.
|
||||
image: ollama/ollama:0.30.10@sha256:bfc9c6d53cc6989aa5131a6fde6b162b2802d4d337657f3253b5f69579bddeee
|
||||
ports:
|
||||
- "11434:11434"
|
||||
volumes:
|
||||
|
||||
@@ -4,15 +4,27 @@
|
||||
<dict>
|
||||
<key>Label</key>
|
||||
<string>com.openjarvis</string>
|
||||
<!-- Binds loopback only: the personal-device default, reachable from this
|
||||
Mac but not the network, so no API key is required. To expose it on
|
||||
your LAN, change the host below to 0.0.0.0 AND uncomment the
|
||||
EnvironmentVariables block to set an API key (an unauthenticated
|
||||
0.0.0.0 server will refuse to start). -->
|
||||
<key>ProgramArguments</key>
|
||||
<array>
|
||||
<string>/usr/local/bin/jarvis</string>
|
||||
<string>serve</string>
|
||||
<string>--host</string>
|
||||
<string>0.0.0.0</string>
|
||||
<string>127.0.0.1</string>
|
||||
<string>--port</string>
|
||||
<string>8000</string>
|
||||
</array>
|
||||
<!--
|
||||
<key>EnvironmentVariables</key>
|
||||
<dict>
|
||||
<key>OPENJARVIS_API_KEY</key>
|
||||
<string>REPLACE_WITH_A_REAL_KEY</string>
|
||||
</dict>
|
||||
-->
|
||||
<key>RunAtLoad</key>
|
||||
<true/>
|
||||
<key>KeepAlive</key>
|
||||
|
||||
Executable
+142
@@ -0,0 +1,142 @@
|
||||
#!/usr/bin/env bash
|
||||
# posthog-hetzner-prep.sh — one-shot Hetzner Cloud server prep for
|
||||
# OpenJarvis's self-hosted PostHog analytics backend.
|
||||
#
|
||||
# Run this on a fresh Ubuntu 22.04+ box (Hetzner CCX23 in Ashburn, or
|
||||
# similar) after pointing the desired domain at it. Idempotent: safe
|
||||
# to re-run if a step fails partway.
|
||||
#
|
||||
# Usage:
|
||||
# sudo bash posthog-hetzner-prep.sh analytics.openjarvis.ai you@openjarvis.ai
|
||||
#
|
||||
# After it finishes:
|
||||
# 1. Visit https://<DOMAIN>/ and create the admin account.
|
||||
# 2. Create project "OpenJarvis".
|
||||
# 3. Settings → Project → grab the Project API Key (phc_…).
|
||||
# 4. Update src/openjarvis/core/config.py AnalyticsConfig defaults:
|
||||
# host = "https://<DOMAIN>"
|
||||
# key = "phc_<new>"
|
||||
# 5. Ship a release. Frontend + backend + install.sh all read those
|
||||
# same defaults via load_config().
|
||||
#
|
||||
# Cost: ~$35/mo on Hetzner CCX23 (4 dedicated vCPU / 16 GB / 240 GB
|
||||
# NVMe) in US-East. Add Hetzner Cloud Backups (+20%) for production.
|
||||
#
|
||||
# Retention: post-install, set 365 days in PostHog UI →
|
||||
# Settings → Data Management → Event ingestion → Data retention.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# ---- args ----
|
||||
if [[ $# -lt 2 ]]; then
|
||||
cat >&2 <<'USAGE'
|
||||
posthog-hetzner-prep.sh: missing arguments.
|
||||
|
||||
Usage:
|
||||
sudo bash posthog-hetzner-prep.sh <domain> <admin_email>
|
||||
|
||||
Examples:
|
||||
sudo bash posthog-hetzner-prep.sh analytics.openjarvis.ai team@openjarvis.ai
|
||||
|
||||
The domain must already resolve to this box (DNS A record) before
|
||||
the script runs — Let's Encrypt needs to reach this server on port 80.
|
||||
USAGE
|
||||
exit 2
|
||||
fi
|
||||
|
||||
DOMAIN="$1"
|
||||
ADMIN_EMAIL="$2"
|
||||
|
||||
if [[ $EUID -ne 0 ]]; then
|
||||
echo "posthog-hetzner-prep.sh: must be run as root (use sudo)." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# ---- step 1: system prep ----
|
||||
echo "[1/5] apt update + base packages..."
|
||||
export DEBIAN_FRONTEND=noninteractive
|
||||
apt-get update -y
|
||||
apt-get install -y --no-install-recommends \
|
||||
curl ufw ca-certificates gnupg \
|
||||
apt-transport-https software-properties-common
|
||||
|
||||
# ---- step 2: firewall ----
|
||||
echo "[2/5] firewall (UFW): 22/80/443 only..."
|
||||
ufw --force reset
|
||||
ufw default deny incoming
|
||||
ufw default allow outgoing
|
||||
ufw allow 22/tcp
|
||||
ufw allow 80/tcp
|
||||
ufw allow 443/tcp
|
||||
ufw --force enable
|
||||
|
||||
# ---- step 3: swap (helps ClickHouse under load spikes) ----
|
||||
echo "[3/5] swap..."
|
||||
if [[ ! -f /swapfile ]]; then
|
||||
fallocate -l 4G /swapfile
|
||||
chmod 600 /swapfile
|
||||
mkswap /swapfile
|
||||
swapon /swapfile
|
||||
if ! grep -q "/swapfile" /etc/fstab; then
|
||||
echo "/swapfile none swap sw 0 0" >> /etc/fstab
|
||||
fi
|
||||
else
|
||||
echo " /swapfile already present"
|
||||
fi
|
||||
|
||||
# ---- step 4: docker ----
|
||||
echo "[4/5] docker..."
|
||||
if ! command -v docker >/dev/null 2>&1; then
|
||||
curl -fsSL https://get.docker.com | sh
|
||||
else
|
||||
echo " docker already installed"
|
||||
fi
|
||||
systemctl enable --now docker
|
||||
|
||||
# ---- step 5: posthog hobby deploy ----
|
||||
echo "[5/5] PostHog Hobby Deploy..."
|
||||
echo
|
||||
echo " Domain: $DOMAIN"
|
||||
echo " Admin email: $ADMIN_EMAIL"
|
||||
echo " DNS A record: verify it points to $(curl -fsS -m 5 https://api.ipify.org 2>/dev/null || echo "<this server>")"
|
||||
echo
|
||||
|
||||
# PostHog's official one-liner. It writes a .env file with random
|
||||
# secrets, configures Caddy with Let's Encrypt TLS for the domain,
|
||||
# and brings up the full stack via docker-compose.
|
||||
/bin/bash -c "$(curl -fsSL https://raw.githubusercontent.com/posthog/posthog/HEAD/bin/deploy-hobby)" -s -- \
|
||||
--domain "$DOMAIN" \
|
||||
--email "$ADMIN_EMAIL" || {
|
||||
echo
|
||||
echo "PostHog deploy script exited with an error. Common causes:"
|
||||
echo " - DNS for $DOMAIN doesn't resolve to this server yet (wait + retry)"
|
||||
echo " - Port 80 not reachable from the public internet (firewall / cloud SG)"
|
||||
echo " - Out of disk on /var/lib/docker (need 20+ GB free)"
|
||||
echo
|
||||
exit 1
|
||||
}
|
||||
|
||||
cat <<EOF
|
||||
|
||||
================================================================
|
||||
PostHog is up at https://$DOMAIN/
|
||||
|
||||
Next steps:
|
||||
1. Open https://$DOMAIN/ in a browser.
|
||||
2. Create the first admin account (any email, your password).
|
||||
3. Create project "OpenJarvis".
|
||||
4. Settings → Project → Project API Key — copy the phc_… value.
|
||||
5. Update src/openjarvis/core/config.py AnalyticsConfig defaults:
|
||||
host = "https://$DOMAIN"
|
||||
key = "phc_<the-new-key>"
|
||||
6. Settings → Data Management → set retention to 365 days.
|
||||
7. Settings → Recordings → confirm Session Replay is OFF (default).
|
||||
8. Ship a release of OpenJarvis with the new config defaults.
|
||||
|
||||
Operational notes:
|
||||
- Updates: bash <(curl -fsSL https://raw.githubusercontent.com/posthog/posthog/HEAD/bin/upgrade-hobby)
|
||||
- Logs: docker compose -f /home/posthog/posthog/docker-compose.hobby.yml logs -f
|
||||
- Disk usage: df -h # bump VPS tier when /var/lib/docker > 70% full
|
||||
- Backups: enable Hetzner Cloud Backups in the Hetzner console
|
||||
================================================================
|
||||
EOF
|
||||
@@ -10,6 +10,31 @@ ExecStart=/opt/openjarvis/.venv/bin/jarvis serve --host 0.0.0.0 --port 8000
|
||||
Restart=on-failure
|
||||
RestartSec=5
|
||||
Environment=HOME=/opt/openjarvis
|
||||
# Binding 0.0.0.0 requires authentication. This file MUST exist and contain:
|
||||
# OPENJARVIS_API_KEY=<key> (generate one: `jarvis auth generate-key`)
|
||||
# It is not prefixed with "-", so the unit fails to start if the file is
|
||||
# missing — preventing an accidentally unauthenticated public server.
|
||||
# Keep secrets here (mode 0600, owned by root) rather than inline Environment=
|
||||
# lines, which leak into `systemctl show` and the journal.
|
||||
EnvironmentFile=/etc/openjarvis/env
|
||||
|
||||
# --- Sandboxing / hardening (#564) ---
|
||||
# Conservative set: tightens the unit without blocking the server's normal I/O
|
||||
# or local GPU inference. ProtectSystem=strict makes the whole filesystem
|
||||
# read-only except ReadWritePaths, so $HOME (config/cache/state under
|
||||
# /opt/openjarvis) stays writable.
|
||||
NoNewPrivileges=true
|
||||
ProtectSystem=strict
|
||||
ReadWritePaths=/opt/openjarvis
|
||||
ProtectHome=true
|
||||
PrivateTmp=true
|
||||
ProtectControlGroups=true
|
||||
ProtectKernelLogs=true
|
||||
ProtectKernelModules=true
|
||||
ProtectKernelTunables=true
|
||||
RestrictRealtime=true
|
||||
RestrictSUIDSGID=true
|
||||
LockPersonality=true
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
|
||||
@@ -0,0 +1,126 @@
|
||||
# OpenJarvis on native Windows
|
||||
|
||||
Phase-1 of the native-Windows-support RFC (#298). Mirrors the Linux
|
||||
(`deploy/systemd/`) and macOS (`deploy/launchd/`) deployments — but for
|
||||
PowerShell, without WSL2 or Docker.
|
||||
|
||||
## One-liner install
|
||||
|
||||
In an elevated-or-regular PowerShell:
|
||||
|
||||
```powershell
|
||||
irm https://open-jarvis.github.io/OpenJarvis/install.ps1 | iex
|
||||
```
|
||||
|
||||
What it does:
|
||||
|
||||
1. Refuses non-Windows hosts and Windows < 10 1809.
|
||||
2. Checks Python 3.10 – 3.13 (3.14 has no numpy wheels yet — see #432).
|
||||
3. Checks `git` on PATH.
|
||||
4. Installs `uv` (https://astral.sh/uv) if absent.
|
||||
5. Clones the OpenJarvis repository to `%LOCALAPPDATA%\OpenJarvis`
|
||||
(override with `$env:OPENJARVIS_HOME`).
|
||||
6. Runs `uv sync --extra desktop --group desktop-native` so the FastAPI server,
|
||||
speech backend, and native extension are importable.
|
||||
7. Optionally prompts to register a scheduled task that auto-starts the
|
||||
server at logon.
|
||||
|
||||
Flags (when invoked directly rather than via `irm | iex`):
|
||||
|
||||
| Flag | Effect |
|
||||
|------|--------|
|
||||
| `-Service` | Register the scheduled task without prompting |
|
||||
| `-SkipService` | Don't prompt; don't register |
|
||||
| `-Force` | Re-run all steps even if already done |
|
||||
|
||||
`irm | iex` can't pass `param()` args into a piped script string, so
|
||||
the same knobs are honored via env vars when the corresponding flag is
|
||||
absent:
|
||||
|
||||
```powershell
|
||||
$env:OPENJARVIS_SKIP_SERVICE = '1'
|
||||
irm https://open-jarvis.github.io/OpenJarvis/install.ps1 | iex
|
||||
```
|
||||
|
||||
The available env vars: `OPENJARVIS_SKIP_SERVICE`, `OPENJARVIS_SERVICE`,
|
||||
`OPENJARVIS_FORCE`. If you need richer control, save the script first
|
||||
(`irm ... -OutFile install.ps1; .\install.ps1 -Force`).
|
||||
|
||||
## Manual scheduled-task setup
|
||||
|
||||
If you skipped the prompt during install, you can register / inspect /
|
||||
remove the task with `jarvis-service.ps1`:
|
||||
|
||||
```powershell
|
||||
$srv = "$env:LOCALAPPDATA\OpenJarvis\src\deploy\windows\jarvis-service.ps1"
|
||||
|
||||
# install (idempotent — replaces existing)
|
||||
powershell -ExecutionPolicy Bypass -File $srv install
|
||||
|
||||
# status
|
||||
powershell -ExecutionPolicy Bypass -File $srv status
|
||||
|
||||
# remove
|
||||
powershell -ExecutionPolicy Bypass -File $srv uninstall
|
||||
```
|
||||
|
||||
The task runs as the current user with `LogonType=Interactive` and
|
||||
`RunLevel=Limited`. It restarts up to 3 times on failure (1-minute
|
||||
gap), has no execution-time limit, and starts when available (catches
|
||||
up if missed).
|
||||
|
||||
## Loopback vs LAN-exposed
|
||||
|
||||
By default the scheduled task binds `127.0.0.1` — reachable only from
|
||||
this machine, no API key required. This matches launchd parity (see
|
||||
`deploy/launchd/com.openjarvis.plist`).
|
||||
|
||||
To expose on your LAN:
|
||||
|
||||
```powershell
|
||||
# 1. Generate an API key. The server REFUSES to bind 0.0.0.0 without one.
|
||||
$env:OPENJARVIS_API_KEY = (uv run jarvis auth generate-key)
|
||||
|
||||
# 2. Re-register the task with -ListenHost 0.0.0.0.
|
||||
powershell -ExecutionPolicy Bypass -File $srv install -ListenHost 0.0.0.0
|
||||
```
|
||||
|
||||
`jarvis-service.ps1 install` refuses `-ListenHost 0.0.0.0` if
|
||||
`$env:OPENJARVIS_API_KEY` is unset — same guard as the systemd unit's
|
||||
`EnvironmentFile=/etc/openjarvis/env`.
|
||||
|
||||
## Parity table
|
||||
|
||||
| Concern | systemd | launchd | Windows |
|
||||
|---------|---------|---------|---------|
|
||||
| Service definition | `deploy/systemd/openjarvis.service` | `deploy/launchd/com.openjarvis.plist` | `deploy/windows/jarvis-service.ps1` (cmdlet-driven) |
|
||||
| Default bind | `0.0.0.0` (with API key) | `127.0.0.1` (no API key) | `127.0.0.1` (no API key) |
|
||||
| Restart on failure | `Restart=on-failure RestartSec=5` | `KeepAlive=true` | `RestartCount=3 RestartInterval=PT1M` |
|
||||
| Auto-start | `multi-user.target` | `RunAtLoad=true` | `AtLogOn` trigger |
|
||||
|
||||
## Updating
|
||||
|
||||
To pull the latest:
|
||||
|
||||
```powershell
|
||||
cd "$env:LOCALAPPDATA\OpenJarvis\src"
|
||||
git pull --ff-only
|
||||
uv sync --extra desktop --group desktop-native
|
||||
```
|
||||
|
||||
Or re-run the installer with `-Force`:
|
||||
|
||||
```powershell
|
||||
irm https://open-jarvis.github.io/OpenJarvis/install.ps1 | iex
|
||||
# (then re-run with the file directly, passing -Force)
|
||||
```
|
||||
|
||||
## Uninstall
|
||||
|
||||
```powershell
|
||||
powershell -ExecutionPolicy Bypass -File "$env:LOCALAPPDATA\OpenJarvis\src\deploy\windows\jarvis-service.ps1" uninstall
|
||||
Remove-Item -Recurse -Force "$env:LOCALAPPDATA\OpenJarvis"
|
||||
```
|
||||
|
||||
Uninstalling does NOT remove `uv` (it's a separate tool — you may have
|
||||
other Python projects using it).
|
||||
@@ -0,0 +1,529 @@
|
||||
<#
|
||||
.SYNOPSIS
|
||||
OpenJarvis native Windows installer.
|
||||
|
||||
.DESCRIPTION
|
||||
Phase-1 of the native-Windows-support RFC (#298). Mirrors the
|
||||
behavior of scripts/install/install.sh (the curl-pipe-bash installer
|
||||
for Linux/WSL2/macOS) but for native Windows PowerShell - no WSL,
|
||||
no Docker, no MSYS2.
|
||||
|
||||
Steps:
|
||||
1. Refuse non-Windows / Windows < 10.
|
||||
2. Check Python 3.10 - 3.13 on PATH (3.14 has no numpy wheels yet,
|
||||
see #432).
|
||||
3. Check git on PATH.
|
||||
4. Install uv (https://astral.sh/uv) if absent.
|
||||
5. Clone the OpenJarvis repository to $env:LOCALAPPDATA\OpenJarvis
|
||||
(override with $env:OPENJARVIS_HOME).
|
||||
6. Run `uv sync --extra desktop --group desktop-native` so the FastAPI
|
||||
server, speech backend, and native extension are importable.
|
||||
7. Optionally register the scheduled-task service (see
|
||||
deploy/windows/jarvis-service.ps1).
|
||||
|
||||
Usage (one-liner):
|
||||
irm https://open-jarvis.github.io/OpenJarvis/install.ps1 | iex
|
||||
|
||||
Usage (file invocation, supports flags):
|
||||
irm https://open-jarvis.github.io/OpenJarvis/install.ps1 -OutFile install.ps1
|
||||
.\install.ps1 -SkipService
|
||||
|
||||
Flags (when running the file directly):
|
||||
-SkipService Don't prompt for / install the scheduled task.
|
||||
-Service Install the scheduled task without prompting.
|
||||
-Force Re-run all steps even if already done.
|
||||
|
||||
Under `irm | iex` the param block is unreachable (Invoke-Expression
|
||||
can't pass named args into a piped script string), so the same knobs
|
||||
are honored via env vars when the corresponding flag is absent:
|
||||
$env:OPENJARVIS_SKIP_SERVICE = '1'
|
||||
$env:OPENJARVIS_SERVICE = '1'
|
||||
$env:OPENJARVIS_FORCE = '1'
|
||||
|
||||
.NOTES
|
||||
Loopback default: the scheduled-task service binds 127.0.0.1, so no
|
||||
API key is needed. To expose on the LAN, edit the registered task to
|
||||
pass `--host 0.0.0.0` AND set $env:OPENJARVIS_API_KEY (an
|
||||
unauthenticated 0.0.0.0 server refuses to start). See
|
||||
deploy/windows/README.md.
|
||||
#>
|
||||
|
||||
[CmdletBinding()]
|
||||
param(
|
||||
[switch] $SkipService,
|
||||
[switch] $Service,
|
||||
[switch] $Force
|
||||
)
|
||||
|
||||
$ErrorActionPreference = 'Stop'
|
||||
|
||||
# Env-var fallback for the `irm | iex` path, where the param block is
|
||||
# unreachable (see header comment). Any explicit -switch wins; env vars
|
||||
# only fill in the gaps.
|
||||
if (-not $SkipService -and $env:OPENJARVIS_SKIP_SERVICE) { $SkipService = $true }
|
||||
if (-not $Service -and $env:OPENJARVIS_SERVICE) { $Service = $true }
|
||||
if (-not $Force -and $env:OPENJARVIS_FORCE) { $Force = $true }
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Output helpers - coloured but plain enough for Constrained Language Mode.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
function Write-Info ($msg) { Write-Host "[info] $msg" -ForegroundColor Cyan }
|
||||
function Write-Ok ($msg) { Write-Host "[ok] $msg" -ForegroundColor Green }
|
||||
function Write-Warn2 ($msg) { Write-Host "[warn] $msg" -ForegroundColor Yellow }
|
||||
function Write-Fail ($msg) {
|
||||
Write-Host "[fail] $msg" -ForegroundColor Red
|
||||
exit 1
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Shared helpers - winget bootstrap + PATH refresh
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# Pull the latest Machine + User PATH from the registry into the current
|
||||
# PowerShell session. Tools installed by `winget install` (Python, git,
|
||||
# Ollama, etc.) update the User PATH, but the running process inherits
|
||||
# the parent shell's environment - so without this refresh the just-
|
||||
# installed tool stays invisible to subsequent `Get-Command` calls.
|
||||
#
|
||||
# CRITICAL: registry PATH entries can be REG_EXPAND_SZ (with literal
|
||||
# `%VAR%` placeholders); the Python.org installer in per-user mode adds
|
||||
# entries like `%LOCALAPPDATA%\Programs\Python\Python313\` unexpanded.
|
||||
# `GetEnvironmentVariable` returns the raw string and PowerShell does
|
||||
# NOT auto-expand on assignment to `$env:Path`, so `Get-Command python`
|
||||
# would miss the just-installed binary. Expand explicitly.
|
||||
function Update-PathFromRegistry {
|
||||
$machinePath = [System.Environment]::GetEnvironmentVariable('Path', 'Machine')
|
||||
$userPath = [System.Environment]::GetEnvironmentVariable('Path', 'User')
|
||||
$combined = "$machinePath;$userPath"
|
||||
$env:Path = [System.Environment]::ExpandEnvironmentVariables($combined)
|
||||
}
|
||||
|
||||
# Bootstrap a tool by winget id. Returns the resolved command source on
|
||||
# success, $null on failure. Caller decides whether failure is fatal.
|
||||
function Install-WithWinget {
|
||||
param(
|
||||
[string] $WingetId, # e.g. 'Python.Python.3.13'
|
||||
[string] $CommandName # e.g. 'python' or 'git'
|
||||
)
|
||||
if (-not (Get-Command winget -ErrorAction SilentlyContinue)) {
|
||||
# Windows 10 pre-2004 / Windows Server / locked-down corporate
|
||||
# images may not have winget. Fall back to the caller's manual
|
||||
# instructions.
|
||||
return $null
|
||||
}
|
||||
Write-Info " Installing $WingetId via winget (silent)..."
|
||||
& winget install --id $WingetId --silent --accept-source-agreements --accept-package-agreements 2>&1 | Out-Null
|
||||
if ($LASTEXITCODE -ne 0) {
|
||||
Write-Warn2 " winget install $WingetId exited $LASTEXITCODE"
|
||||
return $null
|
||||
}
|
||||
Update-PathFromRegistry
|
||||
$cmd = Get-Command $CommandName -ErrorAction SilentlyContinue
|
||||
if ($cmd) { return $cmd.Source }
|
||||
return $null
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 1. OS check
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
Write-Info "Checking OS..."
|
||||
if ($PSVersionTable.Platform -and $PSVersionTable.Platform -ne 'Win32NT') {
|
||||
Write-Fail "install.ps1 is for native Windows. On Linux/macOS use install.sh."
|
||||
}
|
||||
|
||||
# Build number 17763 = Windows 10 1809 (the oldest LTS we test against).
|
||||
$build = [System.Environment]::OSVersion.Version.Build
|
||||
if ($build -lt 17763) {
|
||||
Write-Fail "Windows 10 1809 (build 17763) or newer is required. Detected build $build."
|
||||
}
|
||||
Write-Ok "Windows build $build"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 2. Python check
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
function Get-PythonCommand {
|
||||
# Prefer `python3` (matches our cross-platform helper convention),
|
||||
# fall back to `python` (the Windows store / python.org default).
|
||||
foreach ($name in @('python3', 'python')) {
|
||||
$cmd = Get-Command $name -ErrorAction SilentlyContinue
|
||||
if ($cmd) { return $cmd.Source }
|
||||
}
|
||||
return $null
|
||||
}
|
||||
|
||||
Write-Info "Checking Python (3.10 - 3.13)..."
|
||||
$pythonExe = Get-PythonCommand
|
||||
if (-not $pythonExe) {
|
||||
Write-Info "Python not on PATH - attempting auto-install via winget..."
|
||||
$pythonExe = Install-WithWinget -WingetId 'Python.Python.3.13' -CommandName 'python'
|
||||
if (-not $pythonExe) {
|
||||
Write-Fail @"
|
||||
Python 3.10 - 3.13 not found and auto-install via winget failed.
|
||||
|
||||
Install manually from https://python.org (check 'Add python.exe to PATH'
|
||||
during install) or via winget:
|
||||
|
||||
winget install Python.Python.3.13
|
||||
|
||||
Then re-run this installer.
|
||||
"@
|
||||
}
|
||||
}
|
||||
|
||||
$verRaw = & $pythonExe --version 2>&1
|
||||
$verMatch = [regex]::Match($verRaw, '(\d+)\.(\d+)\.(\d+)')
|
||||
if (-not $verMatch.Success) {
|
||||
Write-Fail "Could not parse Python version from: $verRaw"
|
||||
}
|
||||
$pyMajor = [int]$verMatch.Groups[1].Value
|
||||
$pyMinor = [int]$verMatch.Groups[2].Value
|
||||
if ($pyMajor -ne 3 -or $pyMinor -lt 10 -or $pyMinor -gt 13) {
|
||||
Write-Fail @"
|
||||
Found Python $pyMajor.$pyMinor at $pythonExe, but OpenJarvis requires
|
||||
3.10 - 3.13. Python 3.14 has no numpy Windows wheels yet (#432, will
|
||||
re-open once numpy ships cp314).
|
||||
"@
|
||||
}
|
||||
Write-Ok "Python $pyMajor.$pyMinor ($pythonExe)"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 3. git check
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
Write-Info "Checking git..."
|
||||
$gitExe = (Get-Command git -ErrorAction SilentlyContinue).Source
|
||||
if (-not $gitExe) {
|
||||
Write-Info "git not on PATH - attempting auto-install via winget..."
|
||||
$gitExe = Install-WithWinget -WingetId 'Git.Git' -CommandName 'git'
|
||||
if (-not $gitExe) {
|
||||
Write-Fail @"
|
||||
git not found and auto-install via winget failed.
|
||||
|
||||
Install manually via winget:
|
||||
|
||||
winget install Git.Git
|
||||
|
||||
or download from https://git-scm.com, then re-run this installer.
|
||||
"@
|
||||
}
|
||||
}
|
||||
Write-Ok "git ($gitExe)"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 4. uv check / install
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
Write-Info "Checking uv..."
|
||||
$uvExe = (Get-Command uv -ErrorAction SilentlyContinue).Source
|
||||
if (-not $uvExe) {
|
||||
Write-Info "Installing uv via astral.sh/uv (official PowerShell installer)..."
|
||||
try {
|
||||
Invoke-RestMethod -Uri 'https://astral.sh/uv/install.ps1' -UseBasicParsing | Invoke-Expression
|
||||
} catch {
|
||||
Write-Fail "uv install failed: $($_.Exception.Message)"
|
||||
}
|
||||
# The astral installer puts uv at %USERPROFILE%\.local\bin\uv.exe and
|
||||
# adds that dir to the User PATH. The current process's PATH isn't
|
||||
# refreshed automatically - prepend the install dir so the rest of
|
||||
# this script picks it up.
|
||||
$uvDir = Join-Path $env:USERPROFILE '.local\bin'
|
||||
if (Test-Path (Join-Path $uvDir 'uv.exe')) {
|
||||
$env:Path = "$uvDir;$env:Path"
|
||||
}
|
||||
$uvExe = (Get-Command uv -ErrorAction SilentlyContinue).Source
|
||||
if (-not $uvExe) {
|
||||
Write-Fail "uv installed but isn't on PATH. Re-open a fresh PowerShell and re-run."
|
||||
}
|
||||
}
|
||||
Write-Ok "uv ($uvExe)"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 5. Clone the repo
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
$installRoot = if ($env:OPENJARVIS_HOME) {
|
||||
$env:OPENJARVIS_HOME
|
||||
} else {
|
||||
Join-Path $env:LOCALAPPDATA 'OpenJarvis'
|
||||
}
|
||||
$srcDir = Join-Path $installRoot 'src'
|
||||
|
||||
Write-Info "Install root: $installRoot"
|
||||
|
||||
if (-not (Test-Path $installRoot)) {
|
||||
New-Item -ItemType Directory -Path $installRoot | Out-Null
|
||||
}
|
||||
|
||||
$repoUrl = if ($env:OPENJARVIS_REPO_URL) {
|
||||
$env:OPENJARVIS_REPO_URL
|
||||
} else {
|
||||
'https://github.com/open-jarvis/OpenJarvis.git'
|
||||
}
|
||||
|
||||
if (Test-Path (Join-Path $srcDir '.git')) {
|
||||
if ($Force) {
|
||||
Write-Info "Force: pulling latest from $repoUrl..."
|
||||
& $gitExe -C $srcDir pull --ff-only
|
||||
if ($LASTEXITCODE -ne 0) { Write-Fail "git pull failed" }
|
||||
} else {
|
||||
Write-Ok "Repository already cloned (use -Force to update)"
|
||||
}
|
||||
} else {
|
||||
Write-Info "Cloning $repoUrl..."
|
||||
& $gitExe clone --depth 1 $repoUrl $srcDir
|
||||
if ($LASTEXITCODE -ne 0) { Write-Fail "git clone failed" }
|
||||
Write-Ok "Cloned to $srcDir"
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 6. uv sync --extra desktop --group desktop-native
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
Write-Info "Running 'uv sync --extra desktop --group desktop-native' in $srcDir (this can take a few minutes)..."
|
||||
Push-Location $srcDir
|
||||
try {
|
||||
& $uvExe sync --extra desktop --group desktop-native
|
||||
if ($LASTEXITCODE -ne 0) {
|
||||
Write-Fail "uv sync failed with exit code $LASTEXITCODE. Check the output above."
|
||||
}
|
||||
} finally {
|
||||
Pop-Location
|
||||
}
|
||||
Write-Ok "Dependencies installed"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 7. Ollama - install + start + wait for daemon
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
Write-Info "Checking Ollama..."
|
||||
$ollamaExe = (Get-Command ollama -ErrorAction SilentlyContinue).Source
|
||||
if (-not $ollamaExe) {
|
||||
Write-Info " Ollama not on PATH - downloading the official installer (~150 MB)..."
|
||||
$ollamaSetup = Join-Path $env:TEMP 'OllamaSetup.exe'
|
||||
# SilentlyContinue is load-bearing in PS 5.1: the default progress
|
||||
# bar renderer slows Invoke-WebRequest down 30x on large downloads
|
||||
# (a known PS5.1 issue), turning a 30s download into 15+ minutes.
|
||||
$prevProgress = $ProgressPreference
|
||||
$ProgressPreference = 'SilentlyContinue'
|
||||
try {
|
||||
Invoke-WebRequest `
|
||||
-Uri 'https://ollama.com/download/OllamaSetup.exe' `
|
||||
-OutFile $ollamaSetup `
|
||||
-UseBasicParsing
|
||||
} catch {
|
||||
Remove-Item $ollamaSetup -ErrorAction SilentlyContinue # clean up partial download
|
||||
$ProgressPreference = $prevProgress
|
||||
Write-Fail "Ollama download failed: $($_.Exception.Message)`nInstall manually from https://ollama.com, then re-run."
|
||||
} finally {
|
||||
$ProgressPreference = $prevProgress
|
||||
}
|
||||
# OllamaSetup.exe is built with NSIS, whose silent-install flag is
|
||||
# /S (uppercase). The Inno-Setup-style /silent would open the GUI
|
||||
# and hang `Start-Process -Wait` indefinitely.
|
||||
Write-Info " Running OllamaSetup.exe /S (this can take a minute)..."
|
||||
Start-Process -FilePath $ollamaSetup -ArgumentList '/S' -Wait
|
||||
Remove-Item $ollamaSetup -ErrorAction SilentlyContinue
|
||||
Update-PathFromRegistry
|
||||
$ollamaExe = (Get-Command ollama -ErrorAction SilentlyContinue).Source
|
||||
if (-not $ollamaExe) {
|
||||
Write-Fail "Ollama installer ran but 'ollama' isn't on PATH. Open a fresh PowerShell and re-run, or install manually from https://ollama.com."
|
||||
}
|
||||
}
|
||||
Write-Ok "Ollama ($ollamaExe)"
|
||||
|
||||
# Make sure the daemon is actually responsive before pulling. The Ollama
|
||||
# Windows installer launches the tray app at install time, but on a re-
|
||||
# run with an existing install the daemon may not be running yet.
|
||||
Write-Info "Waiting for Ollama daemon..."
|
||||
$ollamaReady = $false
|
||||
for ($i = 0; $i -lt 60; $i++) {
|
||||
# 'ollama list' writes to stderr until the daemon is reachable; under
|
||||
# $ErrorActionPreference='Stop' the 2>&1 merge surfaces that as a
|
||||
# terminating NativeCommandError that would abort the whole install on
|
||||
# the very first probe. Swallow it and rely on $LASTEXITCODE so the
|
||||
# Start-Process serve fallback below actually runs (issue #522).
|
||||
try { & $ollamaExe list 2>&1 | Out-Null } catch { }
|
||||
if ($LASTEXITCODE -eq 0) {
|
||||
$ollamaReady = $true
|
||||
break
|
||||
}
|
||||
if ($i -eq 5) {
|
||||
# Daemon clearly isn't auto-running - start it ourselves. Ollama
|
||||
# for Windows uses the tray app `ollama app.exe`; falling back to
|
||||
# `ollama serve` works headless.
|
||||
Start-Process -FilePath $ollamaExe -ArgumentList 'serve' -WindowStyle Hidden -ErrorAction SilentlyContinue
|
||||
}
|
||||
Start-Sleep -Seconds 1
|
||||
}
|
||||
if (-not $ollamaReady) {
|
||||
Write-Warn2 "Ollama daemon didn't become ready in 60s. Continuing - bg-orchestrator will retry later."
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 8. Pull a starter model (qwen3.5:2b - ~1.5 GB)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
$modelPullOk = $false
|
||||
if ($ollamaReady) {
|
||||
Write-Info "Pulling qwen3.5:2b (~1.5 GB) so 'jarvis' works on first run..."
|
||||
& $ollamaExe pull 'qwen3.5:2b'
|
||||
if ($LASTEXITCODE -eq 0) {
|
||||
$modelPullOk = $true
|
||||
Write-Ok "Starter model ready"
|
||||
} else {
|
||||
Write-Warn2 "ollama pull failed; the bg-orchestrator will retry once Ollama is reachable."
|
||||
}
|
||||
} else {
|
||||
Write-Warn2 "Skipping model pull - daemon wasn't ready."
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 9. jarvis.cmd shim - so bare `jarvis` works in any new PowerShell
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
$binDir = Join-Path $installRoot 'bin'
|
||||
$shimPath = Join-Path $binDir 'jarvis.cmd'
|
||||
|
||||
if (-not (Test-Path $binDir)) {
|
||||
New-Item -ItemType Directory -Path $binDir | Out-Null
|
||||
}
|
||||
|
||||
# %~dp0 in a .cmd file resolves to the directory containing the script,
|
||||
# so the shim is self-locating - moving %LOCALAPPDATA%\OpenJarvis won't
|
||||
# break it as long as the user moves the whole tree. `uv` is resolved
|
||||
# from PATH at runtime (astral installer adds it to User PATH); avoids
|
||||
# pinning to the install-time uv.exe path which can shift on uv updates.
|
||||
$shimContent = @"
|
||||
@echo off
|
||||
setlocal
|
||||
set "SRC=%~dp0..\src"
|
||||
uv run --project "%SRC%" jarvis %*
|
||||
"@
|
||||
Set-Content -Path $shimPath -Value $shimContent -Encoding ASCII
|
||||
|
||||
# Add %LOCALAPPDATA%\OpenJarvis\bin to User PATH if it isn't already
|
||||
# there. The current process won't see it until restart - handled in the
|
||||
# final banner.
|
||||
#
|
||||
# Compare against the EXPANDED form: a previous install may have written
|
||||
# the entry as `%LOCALAPPDATA%\OpenJarvis\bin` (unexpanded) into User
|
||||
# PATH, and a literal `-ieq` against the expanded `$binDir` would miss
|
||||
# it and append a duplicate every re-run.
|
||||
$userPath = [System.Environment]::GetEnvironmentVariable('Path', 'User')
|
||||
$pathOnUser = $false
|
||||
if ($userPath) {
|
||||
foreach ($entry in ($userPath -split ';')) {
|
||||
$expanded = [System.Environment]::ExpandEnvironmentVariables($entry)
|
||||
if ($expanded -ieq $binDir) { $pathOnUser = $true; break }
|
||||
}
|
||||
}
|
||||
$pathNeedsRefresh = $false
|
||||
if (-not $pathOnUser) {
|
||||
$newUserPath = if ($userPath) { "$userPath;$binDir" } else { $binDir }
|
||||
[System.Environment]::SetEnvironmentVariable('Path', $newUserPath, 'User')
|
||||
$pathNeedsRefresh = $true
|
||||
}
|
||||
Write-Ok "jarvis shim installed at $shimPath"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 10. Optional: register the scheduled-task service
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
$serviceScript = Join-Path $srcDir 'deploy\windows\jarvis-service.ps1'
|
||||
$shouldInstallService = $false
|
||||
|
||||
# Pre-check admin if the user wants the service - Register-ScheduledTask
|
||||
# requires elevation. We do this before the prompt so we don't ask "do
|
||||
# you want the service?" only to fail with Access Denied after they say
|
||||
# yes.
|
||||
$isAdmin = ([Security.Principal.WindowsPrincipal] `
|
||||
[Security.Principal.WindowsIdentity]::GetCurrent()
|
||||
).IsInRole([Security.Principal.WindowsBuiltInRole]::Administrator)
|
||||
|
||||
if ($Service -and -not $isAdmin) {
|
||||
Write-Fail "-Service was requested, but this PowerShell is not elevated. Register-ScheduledTask needs admin rights - re-run from an elevated PowerShell, or drop -Service."
|
||||
}
|
||||
if ($Service) {
|
||||
$shouldInstallService = $true
|
||||
} elseif ($SkipService) {
|
||||
$shouldInstallService = $false
|
||||
} elseif (-not $isAdmin) {
|
||||
# Default to skip-with-explanation when we can't elevate, rather
|
||||
# than prompting and then failing at Register-ScheduledTask.
|
||||
Write-Warn2 "Skipping scheduled-task setup - this PowerShell is not elevated."
|
||||
Write-Warn2 " Register-ScheduledTask requires admin. To install the service later:"
|
||||
Write-Warn2 " Right-click PowerShell -> Run as administrator, then run:"
|
||||
Write-Warn2 " powershell -ExecutionPolicy Bypass -File `"$serviceScript`" install"
|
||||
} else {
|
||||
# Interactive prompt only when there's a real user at the keyboard
|
||||
# AND stdin isn't piped. [Environment]::UserInteractive is the
|
||||
# canonical PowerShell idiom for "is this a user session" (false for
|
||||
# services, scheduled tasks, etc); we additionally guard against the
|
||||
# `irm | iex` case where stdin is redirected.
|
||||
$isInteractive = [Environment]::UserInteractive `
|
||||
-and -not [System.Console]::IsInputRedirected
|
||||
if ($isInteractive) {
|
||||
$reply = Read-Host "Register OpenJarvis as a Windows scheduled task (auto-start at logon, loopback only)? [y/N]"
|
||||
$shouldInstallService = ($reply -match '^[yY]')
|
||||
} else {
|
||||
Write-Warn2 "Non-interactive install - skipping scheduled-task setup."
|
||||
Write-Warn2 "To register the service later, run (from an elevated PowerShell):"
|
||||
Write-Warn2 " powershell -ExecutionPolicy Bypass -File `"$serviceScript`" install"
|
||||
}
|
||||
}
|
||||
|
||||
if ($shouldInstallService) {
|
||||
if (-not (Test-Path $serviceScript)) {
|
||||
Write-Fail "Service script not found at $serviceScript (the clone may be missing files; try -Force)."
|
||||
}
|
||||
Write-Info "Installing scheduled task..."
|
||||
& powershell -ExecutionPolicy Bypass -File $serviceScript install -InstallRoot $installRoot
|
||||
if ($LASTEXITCODE -ne 0) {
|
||||
Write-Fail "Scheduled task setup failed."
|
||||
}
|
||||
Write-Ok "Scheduled task 'OpenJarvis' registered (loopback default)."
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 8. Final message
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
Write-Host ""
|
||||
Write-Host " +----------------------------------+" -ForegroundColor Green
|
||||
Write-Host " | OpenJarvis install complete |" -ForegroundColor Green
|
||||
Write-Host " +----------------------------------+" -ForegroundColor Green
|
||||
Write-Host ""
|
||||
Write-Host " Repo: $srcDir"
|
||||
|
||||
# Tell the truth about what the user can run next, given (a) whether the
|
||||
# starter model finished pulling and (b) whether the User-PATH update
|
||||
# needs a fresh PowerShell to take effect.
|
||||
$nextCmd = if ($modelPullOk) { 'jarvis' } else { 'jarvis doctor' }
|
||||
|
||||
if ($pathNeedsRefresh) {
|
||||
Write-Host ""
|
||||
Write-Host " Run it: open a NEW PowerShell, then: $nextCmd" -ForegroundColor Yellow
|
||||
Write-Host " (the jarvis shim was added to your User PATH; the"
|
||||
Write-Host " current PowerShell won't see it until restart)"
|
||||
} else {
|
||||
Write-Host " Run it: $nextCmd"
|
||||
}
|
||||
|
||||
if (-not $modelPullOk) {
|
||||
Write-Host ""
|
||||
Write-Host " NOTE: the qwen3.5:2b model didn't finish downloading." -ForegroundColor Yellow
|
||||
Write-Host " Chat will fail until the bg-orchestrator finishes the retry."
|
||||
Write-Host " 'jarvis doctor' shows progress."
|
||||
}
|
||||
|
||||
if ($shouldInstallService) {
|
||||
Write-Host ""
|
||||
Write-Host " Service: schtasks /Query /TN OpenJarvis (status)"
|
||||
Write-Host " powershell -File `"$serviceScript`" uninstall (remove)"
|
||||
}
|
||||
Write-Host ""
|
||||
Write-Host " Docs: https://open-jarvis.github.io/OpenJarvis/"
|
||||
Write-Host ""
|
||||
@@ -0,0 +1,208 @@
|
||||
<#
|
||||
.SYNOPSIS
|
||||
Register / unregister the OpenJarvis Windows scheduled task.
|
||||
|
||||
.DESCRIPTION
|
||||
The Windows equivalent of deploy/systemd/openjarvis.service and
|
||||
deploy/launchd/com.openjarvis.plist.
|
||||
|
||||
Registers a per-user scheduled task named "OpenJarvis" that starts
|
||||
`jarvis serve` at logon and restarts on failure. Loopback default
|
||||
(127.0.0.1) so no API key is required — matches launchd parity.
|
||||
|
||||
Subcommands:
|
||||
install — create or replace the task
|
||||
uninstall — remove the task
|
||||
status — show task state
|
||||
|
||||
Arguments (install only):
|
||||
-InstallRoot <path> default: %LOCALAPPDATA%\OpenJarvis (matches
|
||||
install.ps1's default)
|
||||
-ListenHost <addr> default: 127.0.0.1 (loopback). Set to 0.0.0.0
|
||||
ONLY if you also set $env:OPENJARVIS_API_KEY
|
||||
— the server refuses to start unauthenticated
|
||||
on a non-loopback bind.
|
||||
-ListenPort <int> default: 8000
|
||||
|
||||
Usage:
|
||||
powershell -ExecutionPolicy Bypass -File jarvis-service.ps1 install
|
||||
powershell -ExecutionPolicy Bypass -File jarvis-service.ps1 uninstall
|
||||
powershell -ExecutionPolicy Bypass -File jarvis-service.ps1 status
|
||||
#>
|
||||
|
||||
[CmdletBinding()]
|
||||
param(
|
||||
[Parameter(Position = 0)]
|
||||
[ValidateSet('install', 'uninstall', 'status')]
|
||||
[string] $Command = 'status',
|
||||
|
||||
[string] $InstallRoot,
|
||||
[string] $ListenHost = '127.0.0.1',
|
||||
[int] $ListenPort = 8000
|
||||
)
|
||||
|
||||
$ErrorActionPreference = 'Stop'
|
||||
$TaskName = 'OpenJarvis'
|
||||
|
||||
function Write-Info ($msg) { Write-Host "[info] $msg" -ForegroundColor Cyan }
|
||||
function Write-Ok ($msg) { Write-Host "[ok] $msg" -ForegroundColor Green }
|
||||
function Write-Warn2 ($msg) { Write-Host "[warn] $msg" -ForegroundColor Yellow }
|
||||
function Write-Fail ($msg) {
|
||||
Write-Host "[fail] $msg" -ForegroundColor Red
|
||||
exit 1
|
||||
}
|
||||
|
||||
function Get-DefaultInstallRoot {
|
||||
# Use $script: prefix so this is robust to being called from any
|
||||
# function scope (PowerShell's default dynamic lookup would also
|
||||
# work today, but $script: is the explicit contract).
|
||||
if ($script:InstallRoot) { return $script:InstallRoot }
|
||||
if ($env:OPENJARVIS_HOME) { return $env:OPENJARVIS_HOME }
|
||||
return (Join-Path $env:LOCALAPPDATA 'OpenJarvis')
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# install
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
function Install-Task {
|
||||
$root = Get-DefaultInstallRoot
|
||||
$srcDir = Join-Path $root 'src'
|
||||
if (-not (Test-Path $srcDir)) {
|
||||
Write-Fail "OpenJarvis source not found at $srcDir. Run install.ps1 first."
|
||||
}
|
||||
|
||||
$uvCmd = Get-Command uv -ErrorAction SilentlyContinue
|
||||
if (-not $uvCmd) {
|
||||
$uvFallback = Join-Path $env:USERPROFILE '.local\bin\uv.exe'
|
||||
if (Test-Path $uvFallback) {
|
||||
$uvPath = $uvFallback
|
||||
} else {
|
||||
Write-Fail "uv.exe not found on PATH or at $uvFallback. Re-run install.ps1."
|
||||
}
|
||||
} else {
|
||||
$uvPath = $uvCmd.Source
|
||||
}
|
||||
|
||||
# Safety: refuse to register a non-loopback bind without an API key.
|
||||
# Mirrors deploy/systemd/openjarvis.service's EnvironmentFile guard.
|
||||
$isLoopback = ($ListenHost -eq '127.0.0.1' -or $ListenHost -eq 'localhost')
|
||||
if (-not $isLoopback -and -not $env:OPENJARVIS_API_KEY) {
|
||||
Write-Fail @"
|
||||
ListenHost is $ListenHost (non-loopback) but `$env:OPENJARVIS_API_KEY is
|
||||
not set. An unauthenticated non-loopback bind is refused by jarvis serve
|
||||
and would also create a security hole. Set the env var first:
|
||||
|
||||
`$env:OPENJARVIS_API_KEY = (uv run jarvis auth generate-key)
|
||||
|
||||
then re-run with -ListenHost 0.0.0.0.
|
||||
"@
|
||||
}
|
||||
|
||||
# CRITICAL: scheduled tasks do NOT inherit the registering session's
|
||||
# environment. If we registered the task now and stopped here, the
|
||||
# task would launch at logon with a clean env, find no API key, and
|
||||
# `jarvis serve` would refuse to bind 0.0.0.0 — failing silently every
|
||||
# logon. Persist the key to the User env scope so the task's logon
|
||||
# session picks it up. (Loopback path doesn't need the key, so this
|
||||
# only runs for the explicit LAN-exposed case.)
|
||||
if (-not $isLoopback) {
|
||||
Write-Info "Persisting OPENJARVIS_API_KEY to User environment so the scheduled task can read it at logon."
|
||||
[System.Environment]::SetEnvironmentVariable(
|
||||
'OPENJARVIS_API_KEY',
|
||||
$env:OPENJARVIS_API_KEY,
|
||||
'User'
|
||||
)
|
||||
}
|
||||
|
||||
Write-Info "Registering scheduled task '$TaskName'..."
|
||||
Write-Info " Working dir : $srcDir"
|
||||
Write-Info " Listen : $ListenHost`:$ListenPort"
|
||||
Write-Info " User : $env:USERNAME"
|
||||
|
||||
# If a previous task exists, remove it first (idempotent install).
|
||||
$existing = Get-ScheduledTask -TaskName $TaskName -ErrorAction SilentlyContinue
|
||||
if ($existing) {
|
||||
Write-Info "Existing task found — replacing."
|
||||
Unregister-ScheduledTask -TaskName $TaskName -Confirm:$false
|
||||
}
|
||||
|
||||
$action = New-ScheduledTaskAction `
|
||||
-Execute $uvPath `
|
||||
-Argument "run jarvis serve --host $ListenHost --port $ListenPort" `
|
||||
-WorkingDirectory $srcDir
|
||||
|
||||
$trigger = New-ScheduledTaskTrigger -AtLogOn -User $env:USERNAME
|
||||
|
||||
$settings = New-ScheduledTaskSettingsSet `
|
||||
-AllowStartIfOnBatteries `
|
||||
-DontStopIfGoingOnBatteries `
|
||||
-StartWhenAvailable `
|
||||
-RestartCount 3 `
|
||||
-RestartInterval (New-TimeSpan -Minutes 1) `
|
||||
-ExecutionTimeLimit (New-TimeSpan -Seconds 0)
|
||||
|
||||
$principal = New-ScheduledTaskPrincipal `
|
||||
-UserId $env:USERNAME `
|
||||
-LogonType Interactive `
|
||||
-RunLevel Limited
|
||||
|
||||
Register-ScheduledTask `
|
||||
-TaskName $TaskName `
|
||||
-Action $action `
|
||||
-Trigger $trigger `
|
||||
-Settings $settings `
|
||||
-Principal $principal `
|
||||
-Description 'OpenJarvis API server (loopback default — see deploy/windows/README.md)' | Out-Null
|
||||
|
||||
Write-Ok "Task '$TaskName' registered."
|
||||
Write-Info "It will start automatically at next logon."
|
||||
Write-Info "To start it now: Start-ScheduledTask -TaskName $TaskName"
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# uninstall
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
function Uninstall-Task {
|
||||
$existing = Get-ScheduledTask -TaskName $TaskName -ErrorAction SilentlyContinue
|
||||
if (-not $existing) {
|
||||
Write-Warn2 "Task '$TaskName' is not registered — nothing to remove."
|
||||
return
|
||||
}
|
||||
Write-Info "Stopping '$TaskName' (if running)..."
|
||||
Stop-ScheduledTask -TaskName $TaskName -ErrorAction SilentlyContinue
|
||||
Write-Info "Unregistering '$TaskName'..."
|
||||
Unregister-ScheduledTask -TaskName $TaskName -Confirm:$false
|
||||
Write-Ok "Task '$TaskName' removed."
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# status
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
function Show-Status {
|
||||
$task = Get-ScheduledTask -TaskName $TaskName -ErrorAction SilentlyContinue
|
||||
if (-not $task) {
|
||||
Write-Host "Task '$TaskName' is not registered."
|
||||
Write-Host "Install it with:"
|
||||
Write-Host " powershell -ExecutionPolicy Bypass -File `"$PSCommandPath`" install"
|
||||
return
|
||||
}
|
||||
$info = Get-ScheduledTaskInfo -TaskName $TaskName
|
||||
Write-Host "Task : $TaskName"
|
||||
Write-Host "State : $($task.State)"
|
||||
Write-Host "LastRun : $($info.LastRunTime)"
|
||||
Write-Host "LastRes : 0x$('{0:X8}' -f $info.LastTaskResult)"
|
||||
Write-Host "NextRun : $($info.NextRunTime)"
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# dispatch
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
switch ($Command) {
|
||||
'install' { Install-Task }
|
||||
'uninstall' { Uninstall-Task }
|
||||
'status' { Show-Status }
|
||||
}
|
||||
BIN
Binary file not shown.
@@ -116,7 +116,7 @@ All providers produce the same output format consumed by agents:
|
||||
| **LM Studio** | `lmstudio` | OpenAI-compatible | 1234 | No (GPU optional) | Desktop GUI, easy model management |
|
||||
| **Exo** | `exo` | OpenAI-compatible | 52415 | No (distributed) | Distributed inference across heterogeneous devices |
|
||||
| **Nexa** | `nexa` | OpenAI-compatible | 18181 | No (CPU/GPU) | On-device inference with GGUF models |
|
||||
| **Lemonade** | `lemonade` | OpenAI-compatible | 8000 | AMD GPU/NPU | AMD consumer GPUs (RDNA), Ryzen AI NPUs |
|
||||
| **Lemonade** | `lemonade` | OpenAI-compatible | 13305 | AMD GPU/NPU | AMD consumer GPUs (RDNA), Ryzen AI NPUs |
|
||||
| **Uzu** | `uzu` | OpenAI-compatible | 8000 | Varies | Uzu inference runtime |
|
||||
| **Apple FM** | `apple_fm` | OpenAI-compatible | 8079 | Apple Silicon | Apple Foundation Model on-device inference |
|
||||
| **LiteLLM** | `litellm` | OpenAI-compatible | — | No | Unified proxy to 100+ LLM providers |
|
||||
@@ -135,7 +135,7 @@ The Ollama backend communicates via Ollama's native HTTP API at `/api/chat` and
|
||||
|
||||
The vLLM backend uses the OpenAI-compatible `/v1/chat/completions` API. It is recommended for datacenter GPUs (NVIDIA A100, H100, L40, A10, A30 and AMD MI300, MI325, MI350, MI355).
|
||||
|
||||
- **Default host:** `http://localhost:8000`
|
||||
- **Default host:** `http://localhost:13305`
|
||||
- **Health check:** `GET /v1/models`
|
||||
- **Tool fallback:** If the server returns HTTP 400 when tools are included, the engine automatically retries without tools
|
||||
|
||||
@@ -204,7 +204,7 @@ The Nexa backend connects to the Nexa SDK on-device inference server via a FastA
|
||||
|
||||
The Lemonade backend connects to the [Lemonade](https://lemonade-server.ai/) inference server, which is optimized for AMD consumer GPUs (RDNA architecture) and Ryzen AI Neural Processing Units (NPUs). It uses the OpenAI-compatible `/v1/chat/completions` API.
|
||||
|
||||
- **Default host:** `http://localhost:8000`
|
||||
- **Default host:** `http://localhost:13305`
|
||||
- **Health check:** `GET /v1/models`
|
||||
- **Install:** Visit [lemonade-server.ai](https://lemonade-server.ai/) for platform-specific installation instructions
|
||||
- **Best for:** Ryzen AI GPUs and NPUs, and AMD-based desktop and laptop systems
|
||||
@@ -357,7 +357,7 @@ host = "http://localhost:30000"
|
||||
# binary_path = ""
|
||||
|
||||
# [engine.lemonade]
|
||||
# host = "http://localhost:8000"
|
||||
# host = "http://localhost:13305"
|
||||
```
|
||||
|
||||
The `EngineConfig` dataclass and its per-engine sub-dataclasses map these settings:
|
||||
@@ -370,7 +370,7 @@ The `EngineConfig` dataclass and its per-engine sub-dataclasses map these settin
|
||||
| `SGLangEngineConfig` | `host` | `http://localhost:30000` | SGLang server URL |
|
||||
| `LlamaCppEngineConfig` | `host` | `http://localhost:8080` | llama.cpp server URL |
|
||||
| `LlamaCppEngineConfig` | `binary_path` | `""` | Path to llama.cpp binary (for managed mode) |
|
||||
| `LemonadeEngineConfig` | `host` | `http://localhost:8000` | Lemonade server URL |
|
||||
| `LemonadeEngineConfig` | `host` | `http://localhost:13305` | Lemonade server URL |
|
||||
|
||||
!!! note "Backward compatibility"
|
||||
The old flat field names `ollama_host`, `vllm_host`, `llamacpp_host`, `llamacpp_path`, `sglang_host`, and `lemonade_host` under `[engine]` are still accepted as backward-compatible properties on `EngineConfig`. New configurations should use the nested sub-section format.
|
||||
|
||||
@@ -556,13 +556,13 @@ to Python.
|
||||
|
||||
---
|
||||
|
||||
## Distillation (Frontier-Driven Harness Learning)
|
||||
## LLM-Guided Spec Search (Frontier-Driven Harness Learning)
|
||||
|
||||
The distillation subsystem uses a frontier closed-source model (the "teacher") as a meta-engineer for the local student's full harness — not just its weights. Instead of pushing knowledge into a small model's weights, we push a frontier model's engineering judgement into the surrounding configuration: prompts, routing, agent class, tool availability, and tool descriptions.
|
||||
LLM-guided spec search uses a frontier closed-source model (the "teacher") as a meta-engineer for the local student's full harness — not just its weights. Instead of pushing knowledge into a small model's weights, we push a frontier model's engineering judgement into the surrounding configuration: prompts, routing, agent class, tool availability, and tool descriptions.
|
||||
|
||||
### Where it lives
|
||||
|
||||
`learning/distillation/` is the fifth subsystem within the Learning pillar, alongside `learning/routing/`, `learning/optimize/`, `learning/training/`, and `learning/intelligence/`.
|
||||
`learning/spec_search/` is the fifth subsystem within the Learning pillar, alongside `learning/routing/`, `learning/optimize/`, `learning/training/`, and `learning/intelligence/`.
|
||||
|
||||
### Four-phase loop
|
||||
|
||||
@@ -579,7 +579,7 @@ Trigger → Diagnose → Plan → Execute → Record
|
||||
|
||||
| Component | Module | Purpose |
|
||||
|-----------|--------|---------|
|
||||
| `DistillationOrchestrator` | `orchestrator.py` | Top-level session driver |
|
||||
| `SpecSearchOrchestrator` | `orchestrator.py` | Top-level session driver |
|
||||
| `TeacherAgent` | `diagnose/teacher_agent.py` | Frontier model tool-calling loop |
|
||||
| `DiagnosisRunner` | `diagnose/runner.py` | Phase 1 orchestration |
|
||||
| `LearningPlanner` | `plan/planner.py` | Diagnosis → typed LearningPlan |
|
||||
@@ -609,4 +609,4 @@ Every edit is assigned a tier from a deterministic lookup table:
|
||||
| `review` | System prompt edits, agent class, few-shot exemplars | Queue for user approval |
|
||||
| `manual` | LoRA fine-tuning (v2) | Never auto-apply |
|
||||
|
||||
See [Distillation user guide](../user-guide/learning-distillation.md) for CLI usage and configuration.
|
||||
See [LLM-guided spec search guide](../user-guide/llm-guided-spec-search.md) for the architecture and the building blocks.
|
||||
|
||||
@@ -45,7 +45,7 @@ The memory pipeline includes document ingestion, chunking, embedding generation,
|
||||
|
||||
The Learning system is the fifth primitive, connecting the other four through **trace-driven feedback**. Every agent interaction can produce a `Trace` capturing the full sequence of steps — routing decisions, memory retrieval, inference calls, tool invocations, and final responses. The `TraceAnalyzer` computes statistics from accumulated traces, and the `TraceDrivenPolicy` uses these statistics to learn which model/agent/tool combinations produce the best outcomes for different query types.
|
||||
|
||||
The learning system is configured through nested sub-sections in `config.toml`: `[learning.routing]` controls the router policy (heuristic, learned, sft, grpo), `[learning.intelligence]` controls the model-level learning policy, `[learning.agent]` controls agent advisor and ICL updater policies, and `[learning.metrics]` sets the composite reward function weights. The pillar also includes the distillation subsystem, a frontier-driven loop that improves the local harness — see [Learning architecture: Distillation](learning.md#distillation-frontier-driven-harness-learning).
|
||||
The learning system is configured through nested sub-sections in `config.toml`: `[learning.routing]` controls the router policy (heuristic, learned, sft, grpo), `[learning.intelligence]` controls the model-level learning policy, `[learning.agent]` controls agent advisor and ICL updater policies, and `[learning.metrics]` sets the composite reward function weights. The pillar also includes LLM-guided spec search, a frontier-driven loop that improves the local harness — see [Learning architecture: LLM-guided spec search](learning.md#llm-guided-spec-search-frontier-driven-harness-learning).
|
||||
|
||||
---
|
||||
|
||||
|
||||
@@ -0,0 +1,45 @@
|
||||
# Showcase screenshots
|
||||
|
||||
This directory holds the hero screenshot for each Showcase entry in `docs/showcase/`. Convention is one file per entry, named to match the entry's slug:
|
||||
|
||||
| Entry | Screenshot path |
|
||||
|---|---|
|
||||
| `docs/showcase/morning-brief.md` | `morning-brief.png` |
|
||||
| `docs/showcase/persistent-memory.md` | `persistent-memory.png` |
|
||||
| `docs/showcase/cost-savings.md` | `cost-savings.png` |
|
||||
| `docs/showcase/discord-companion.md` | `discord-companion.png` |
|
||||
| `docs/showcase/coding-assistant.md` | `coding-assistant.png` |
|
||||
|
||||
## Conventions
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| Format | PNG, sRGB, no alpha channel |
|
||||
| Size | 1600×1000 (4:2.5 — wider than 16:9, so screenshots don't get letterboxed in the docs grid) |
|
||||
| File size | Under 400 KB after `pngquant --quality 70-90 --speed 1` |
|
||||
| Loading | All `<img>` and `<figure>` tags in showcase pages use `loading=lazy` — these images are below the fold on the gallery page |
|
||||
|
||||
## What to redact
|
||||
|
||||
- Real email addresses
|
||||
- API keys, OAuth tokens, anything starting with `sk-`, `ghp_`, `xox`, `eyJ`
|
||||
- Personal phone numbers
|
||||
- Conversation partners' faces or full names (unless they've signed off)
|
||||
- File paths that include other people's home directories
|
||||
|
||||
## What to keep
|
||||
|
||||
- Model names ("llama3.1:8b", "qwen2.5:14b") — they're informative
|
||||
- Timestamps — proves the screenshot is recent
|
||||
- Dollar amounts on the leaderboard — the whole point
|
||||
- Emoji reactions, your own first name, your own avatar
|
||||
|
||||
## Placeholder PNGs
|
||||
|
||||
This directory ships with no images on the initial PR. The Showcase pages reference image paths that don't exist yet — MkDocs will render a broken-image placeholder, and the figcaption still conveys what should be there. Real screenshots arrive in follow-up PRs as Showcase entries are populated with each contributor's actual setup.
|
||||
|
||||
If you're contributing the first real entry, drop your PNG at `docs/assets/showcase/<your-slug>.png` in the same PR that adds your markdown page. The image filename must match the slug used in the showcase page's `<img>` reference.
|
||||
|
||||
## Regenerating screenshots in bulk
|
||||
|
||||
A future enhancement (tracked as PR #3 in the showcase-tier roadmap) will add `scripts/showcase/regen_screenshots.py` — a Playwright-driven pipeline that boots a demo `jarvis serve` against a sealed config and captures fresh screenshots for every showcase entry on each release tag. Until that lands, screenshots are contributed manually by each Showcase author.
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 6.2 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 6.2 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 6.2 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 6.2 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 6.2 KiB |
@@ -4,12 +4,25 @@ OpenJarvis provides Docker images for both CPU-only and GPU-accelerated deployme
|
||||
|
||||
## Quick Start
|
||||
|
||||
The fastest way to get OpenJarvis running in Docker is with Docker Compose, which starts both the API server and an Ollama backend:
|
||||
The container binds `0.0.0.0`, so an **API key is required** — the server
|
||||
refuses to start on a non-loopback address without one. Set it first:
|
||||
|
||||
```bash
|
||||
cd deploy/docker
|
||||
cp .env.example .env
|
||||
echo "OPENJARVIS_API_KEY=$(jarvis auth generate-key)" > .env # or paste your own
|
||||
```
|
||||
|
||||
Then start both the API server and an Ollama backend with Docker Compose:
|
||||
|
||||
```bash
|
||||
docker compose up -d
|
||||
```
|
||||
|
||||
`docker compose` reads `OPENJARVIS_API_KEY` from `.env` (or your shell
|
||||
environment) and fails fast if it is unset. Clients must then send
|
||||
`Authorization: Bearer <key>` on `/v1/*` and `/api/*` requests.
|
||||
|
||||
This brings up two services:
|
||||
|
||||
| Service | Port | Description |
|
||||
|
||||
@@ -24,6 +24,14 @@ launchctl load ~/Library/LaunchAgents/com.openjarvis.plist
|
||||
|
||||
The service starts immediately (due to `RunAtLoad`) and will automatically restart at each login.
|
||||
|
||||
!!! note "Binds loopback by default"
|
||||
The plist binds `127.0.0.1` — reachable from this Mac but not the network,
|
||||
the right default for a personal device, and no API key is needed. To
|
||||
expose it on your LAN, change the host to `0.0.0.0` **and** uncomment the
|
||||
`EnvironmentVariables` block to set `OPENJARVIS_API_KEY`
|
||||
(`jarvis auth generate-key`); an unauthenticated `0.0.0.0` server refuses
|
||||
to start.
|
||||
|
||||
Verify it is running:
|
||||
|
||||
```bash
|
||||
@@ -55,10 +63,17 @@ The provided plist file at `deploy/launchd/com.openjarvis.plist`:
|
||||
<string>/usr/local/bin/jarvis</string>
|
||||
<string>serve</string>
|
||||
<string>--host</string>
|
||||
<string>0.0.0.0</string>
|
||||
<string>127.0.0.1</string>
|
||||
<string>--port</string>
|
||||
<string>8000</string>
|
||||
</array>
|
||||
<!-- To expose on the LAN: set host to 0.0.0.0 and uncomment this block.
|
||||
<key>EnvironmentVariables</key>
|
||||
<dict>
|
||||
<key>OPENJARVIS_API_KEY</key>
|
||||
<string>REPLACE_WITH_A_REAL_KEY</string>
|
||||
</dict>
|
||||
-->
|
||||
<key>RunAtLoad</key>
|
||||
<true/>
|
||||
<key>KeepAlive</key>
|
||||
@@ -76,7 +91,7 @@ The provided plist file at `deploy/launchd/com.openjarvis.plist`:
|
||||
| Key | Value | Description |
|
||||
|----------------------|--------------------------------|------------------------------------------------------------------------------------------------------|
|
||||
| `Label` | `com.openjarvis` | Unique identifier for the service. Used with `launchctl` commands to manage the service. |
|
||||
| `ProgramArguments` | `["/usr/local/bin/jarvis", "serve", "--host", "0.0.0.0", "--port", "8000"]` | The command and arguments to execute. Each element of the command line is a separate string in the array. |
|
||||
| `ProgramArguments` | `["/usr/local/bin/jarvis", "serve", "--host", "127.0.0.1", "--port", "8000"]` | The command and arguments to execute. Binds loopback by default; see the note above to expose on the LAN with an API key. |
|
||||
| `RunAtLoad` | `true` | Start the service immediately when the plist is loaded (and on each login). |
|
||||
| `KeepAlive` | `true` | Automatically restart the service if it exits for any reason. launchd monitors the process and relaunches it. |
|
||||
| `StandardOutPath` | `/tmp/openjarvis.stdout.log` | File where standard output is written. Contains server startup messages and access logs. |
|
||||
|
||||
@@ -21,7 +21,17 @@ cd /opt/openjarvis/OpenJarvis && sudo -u openjarvis uv sync --extra server
|
||||
|
||||
## Installing the Service
|
||||
|
||||
Copy the unit file to the systemd directory, reload the daemon, and enable the service:
|
||||
The unit binds `0.0.0.0`, so an **API key is required** — and the unit
|
||||
declares `EnvironmentFile=/etc/openjarvis/env` (no `-` prefix), so it will
|
||||
**fail to start** until that file exists with a key. Create it first:
|
||||
|
||||
```bash
|
||||
sudo mkdir -p /etc/openjarvis
|
||||
echo "OPENJARVIS_API_KEY=$(jarvis auth generate-key)" | sudo tee /etc/openjarvis/env
|
||||
sudo chmod 600 /etc/openjarvis/env
|
||||
```
|
||||
|
||||
Then copy the unit file, reload the daemon, and enable the service:
|
||||
|
||||
```bash
|
||||
sudo cp deploy/systemd/openjarvis.service /etc/systemd/system/
|
||||
@@ -30,6 +40,10 @@ sudo systemctl enable openjarvis
|
||||
sudo systemctl start openjarvis
|
||||
```
|
||||
|
||||
Clients must send `Authorization: Bearer <key>` on `/v1/*` and `/api/*`
|
||||
requests. (If you instead bind to `127.0.0.1`, the key is optional and you
|
||||
can drop the `EnvironmentFile` line.)
|
||||
|
||||
Verify it is running:
|
||||
|
||||
```bash
|
||||
|
||||
@@ -0,0 +1,711 @@
|
||||
# Spec B — Apple Silicon enablement for Pearl mining
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| **Date** | 2026-05-05 |
|
||||
| **Status** | Design — Phase 0 investigation complete; v1 scope reduced (see §1.5) |
|
||||
| **Owner** | OpenJarvis team (parallel-agent friendly) |
|
||||
| **Companion spec** | [Spec A — vLLM-Pearl mining integration (v1)](2026-05-05-vllm-pearl-mining-integration-design.md) |
|
||||
| **Repos referenced** | `OpenJarvis`, `pearl-research-labs/pearl`, possibly upstream `ml-explore/mlx`, `ggerganov/llama.cpp` |
|
||||
|
||||
> **For agents picking this up cold:** read §1 ("Cold-start brief") first, then **§1.5 ("Phase 0 findings")** which substantially simplifies v1 scope. The original Phase 1–4 GPU-kernel plan in §5–§8 is preserved as the v2/v3 path; v1 ships using upstream Pearl's PyTorch reference and pure-Rust miner.
|
||||
|
||||
## 1. Cold-start brief
|
||||
|
||||
**One-paragraph problem statement.** OpenJarvis is adding a `mining` subsystem (Spec A) that lets users mine the Pearl PoUW blockchain through their local LLM inference. Pearl's reference miner is CUDA-only and bound to NVIDIA Hopper (`sm_90a`, H100/H200) — most OpenJarvis users on Apple Silicon are locked out at the protocol level. **This spec is the plan to unblock them.** The OJ-side integration work is small (a new `MiningProvider` implementation that drops into the existing `MinerRegistry` from Spec A); the substantial work is a Metal port of Pearl's `NoisyGEMM` kernel and a matching plugin into an Apple-native inference backend (MLX or llama.cpp Metal). The Pearl validation path is plonky2-STARK-based and **already hardware-neutral** — see §3 for the evidence — so a correct Metal implementation produces blocks Pearl validators accept without any consensus changes.
|
||||
|
||||
**Three things to know before doing any work.**
|
||||
|
||||
1. The Pearl validator (`pearl/zk-pow/src/api/verify.rs::verify_block`) operates on a STARK proof and references no hardware. **The protocol does not care what GPU produced the work** as long as the math is correct and the proof verifies. CUDA-only is a performance choice, not a consensus choice.
|
||||
2. Pearl's CUDA kernel (`pearl/miner/pearl-gemm/csrc/gemm/`) uses Hopper-only primitives (TMA, WGMMA, thread-block clusters, CUTLASS 3.x). A Metal port is **not a translation** — it's a from-scratch reimplementation against a different programming model. Plan effort accordingly.
|
||||
3. The OJ integration boundary is the `MiningProvider` ABC defined in Spec A §4.4. **Do not modify Spec A.** Add a new provider file (`mining/mlx_pearl.py` or `mining/llamacpp_pearl_metal.py`), implement the ABC, register via `MinerRegistry`, ship a new optional extra. Everything else in Spec A — sidecar shape, config schema, telemetry adapter contract, v2 fee/pool seams — applies unchanged.
|
||||
|
||||
## 1.5. Phase 0 findings — significant scope simplification (2026-05-05)
|
||||
|
||||
Phase 0 investigation produced four findings that reshape this spec. **The original §5–§8 plan (Metal NoisyGEMM kernel + custom MLX/llama.cpp plugin) is preserved as v2/v3, but is no longer required for v1.**
|
||||
|
||||
### 1.5.1 The validator is hardware-neutral (confirmed)
|
||||
|
||||
`zk-pow/src/api/verify.rs::verify_block` and `verify_plain_proof` are pure Rust + plonky2; no CUDA, no GPU paths, no hardware introspection. §3.1's claim is **verified by direct code reading** (see file paths in §11). The protocol accepts blocks from any implementation that produces correct math.
|
||||
|
||||
### 1.5.2 A complete hardware-neutral miner already exists upstream
|
||||
|
||||
`pearl/zk-pow/src/ffi/mine.rs::mine()` is a pure-Rust mining function:
|
||||
|
||||
- Generates random `i8` matrices `A` (m×k) and `B` (k×n) with values in `[-64, 64]`
|
||||
- Computes blake3-derived noise via `circuit/pearl_noise.rs::compute_noise_for_indices`
|
||||
- Performs the noised dot products in tile patterns (`PeriodicPattern.rows_pattern × cols_pattern`)
|
||||
- Hashes the jackpot tile and checks the difficulty target
|
||||
- Returns a `PlainProof` that `verify_plain_proof` accepts
|
||||
|
||||
It is exposed to Python via `py-pearl-mining` (`pearl_mining.mine`). Dependencies: pure Rust (`zk-pow`, `pearl-blake3`, `blake3`, `rayon`, `pyo3`, `tikv-jemallocator`). **No CUDA, no platform-specific code in the Cargo dependency tree.**
|
||||
|
||||
### 1.5.3 A PyTorch reference of the *production* NoisyGEMM also exists upstream
|
||||
|
||||
`pearl/miner/miner-base/src/miner_base/noisy_gemm.py::NoisyGemm` is a complete PyTorch reference of the same NoisyGEMM that vllm-miner accelerates with the H100 CUDA kernel:
|
||||
|
||||
- `noise_A`, `noise_B`, `gemm`, `noisy_gemm` methods replicating the kernel's math in `torch.matmul` calls
|
||||
- The denoising path produces **bit-exact results** versus a vanilla `torch.matmul(A.int32, B.int32)` — verified by the `assert torch.equal(result, expected)` test at `miner/miner-base/tests/test_noisy_gemm.py:92`. The "fp16 tolerance" budget I assumed in the original §5.2 is unnecessary for the int7×int7→int32 protocol path
|
||||
- Dependencies: `torch==2.11.0`, `blake3`, `numpy`, `pearl-gateway`, `py-pearl-mining` — all install on macOS arm64
|
||||
|
||||
### 1.5.4 Empirical confirmation (2026-05-05, on this hardware)
|
||||
|
||||
`py-pearl-mining` was built from the upstream source on the spec author's M2 Max. The build produced `py_pearl_mining-0.1.0-cp312-abi3-macosx_11_0_arm64.whl` in 56 seconds. End-to-end mining cycle:
|
||||
|
||||
```
|
||||
running mine(m=256, n=128, k=1024, rank=32) on Apple Silicon CPU…
|
||||
mine() returned a proof in 0.078s
|
||||
verify_plain_proof: ok=True, msg='Mining solution verified successfully' (0.3 ms)
|
||||
END-TO-END MINING ON APPLE SILICON SUCCEEDED
|
||||
```
|
||||
|
||||
The test difficulty here is `nbits=0x1D2FFFFF` (the test fixture from `py-pearl-mining/tests/test_python_api.py`), much lower than mainnet difficulty — so 78 ms is **per-share at the test difficulty**, not real expected wall-clock per share at the network's current difficulty. But the *correctness* of the path is proven.
|
||||
|
||||
### 1.5.5 Reframed v1 — what to build, what to defer
|
||||
|
||||
| Layer | Original plan (§5–§8) | New v1 plan |
|
||||
|---|---|---|
|
||||
| Reference oracle (Phase 0-B) | Build from scratch in PyTorch, validate against H100 CUDA | **Already exists upstream** — `miner-base.noisy_gemm` + `pearl_mining.mine`. OJ ships a thin wrapper, no reimplementation. |
|
||||
| Inference-backend plugin (Phase 2) | Custom MLX or llama.cpp Metal plugin doing NoisyGEMM | **Deferred to v2.** v1 is **decoupled mining** — mining runs as a separate process via the upstream Rust miner; user's existing inference (Ollama, MLX, llama.cpp) is unaffected. |
|
||||
| Metal NoisyGEMM kernel (Phase 1) | Months of GPU-kernel engineering | **Deferred to v3** as a perf optimization once v1 ships and demand is proven. Original §6.1 content preserved as the v3 plan. |
|
||||
| OJ provider integration (Phase 3) | New `MiningProvider` impl | **v1 ships this** — see §13. `MiningProvider` ABC from Spec A unchanged. |
|
||||
| Bringup + verification (Phase 4) | Hardware matrix + testnet | **v1 ships this** — see §14. Same hardware matrix, simpler scope. |
|
||||
| Pearl coordination (Phase 0-A) | Confirm upstream-vs-fork posture | Still needed (see §12) — but the bar is lower since v1 doesn't require any code from us in Pearl's tree. |
|
||||
|
||||
### 1.5.6 Honest performance expectations for v1
|
||||
|
||||
This is **not** competitive mining. The point of vllm-miner is that it amortizes mining work over LLM inference matmuls (the matmul you're already doing for inference *is* the mining work). v1 here decouples them: your CPU does mining, your GPU does inference. The hashrate will be low. **But it works today, and it ships with a credible upgrade path.** Document this transparently in `mine doctor` and the user guide.
|
||||
|
||||
The v2 (PyTorch-MPS NoisyGEMM coupled with MLX/llama.cpp inference) and v3 (native Metal kernel) work paths in §5–§8 remain the route to competitive Apple Silicon mining. They're explicitly not blocking v1.
|
||||
|
||||
## 2. Why this is its own spec
|
||||
|
||||
Spec A's scope is the v1 integration that ships today on the only working configuration (vLLM + sm90). Apple Silicon enablement is a separate, parallelizable workstream because:
|
||||
|
||||
- **Different ownership boundary.** Spec A is Python integration of an existing Pearl Docker image. Spec B is GPU-kernel engineering with potential upstream contribution to Pearl. These need different reviewers, different CI surfaces (no H100 needed, but Apple Silicon required), and different release cadence.
|
||||
- **Different timeline.** Spec A is weeks. Spec B is plausibly months for the kernel work alone.
|
||||
- **Different blast radius.** Spec A ships zero risk to non-mining users; even mining users who edit the wrong config get a clear error. Spec B carries protocol-correctness risk — a bug in NoisyGEMM produces invalid blocks that get rejected by validators.
|
||||
- **Parallel-agent ergonomics.** The user has explicitly asked for this spec to be picked up by a separate agent in parallel. Self-containment is a design goal.
|
||||
|
||||
## 3. Evidence: Apple Silicon support is possible
|
||||
|
||||
### 3.1 The validator is hardware-neutral
|
||||
|
||||
`pearl/zk-pow/src/api/verify.rs::verify_block`:
|
||||
|
||||
```rust
|
||||
pub fn verify_block(public_params: &PublicProofParams, proof: &ZKProof, cache: &mut CircuitCache) -> Result<()> {
|
||||
let (params, pis) = prepare_verification(public_params, proof, None)?;
|
||||
PearlRecursion::compile_circuits(params, cache, false)?;
|
||||
verify_with_cache(params, cache, &pis, proof)
|
||||
}
|
||||
```
|
||||
|
||||
Verification is `PearlRecursion::verify(params, cache, pis, &proof.plonky2_proof)` — a recursive plonky2 STARK check. No GPU code path, no CUDA dependency. Validator nodes run pure Rust.
|
||||
|
||||
The mining work consists of three things, all of which are mathematical specifications — not implementation specifications:
|
||||
|
||||
1. A NoisyGEMM result whose noise pattern is derived from blake3 of a per-block key
|
||||
2. A blake3 commitment hash over the noised matmul that meets a difficulty target
|
||||
3. A plonky2 STARK proof that ties the result to the commitment
|
||||
|
||||
Any implementation that produces matching outputs is acceptable to the network. **This is the design intent of PoUW** — the work has to be replayable and verifiable, but not hardware-bound.
|
||||
|
||||
### 3.2 The Pearl team explicitly anticipates non-CUDA plugins
|
||||
|
||||
From `pearl/miner/README.md`:
|
||||
|
||||
> "Currently only mining via vLLM is supported, in the future we hope to supply plugins for other LLM inference libraries, like SGLang, TensorRT-LLM, Ollama, ..."
|
||||
|
||||
Apple is not in their list, but the framing — "supply plugins for other LLM inference libraries" — implies the boundary is at the inference backend, not at the consensus protocol. Confirms the architectural read.
|
||||
|
||||
### 3.3 Reference implementation exists in py-pearl-mining
|
||||
|
||||
`pearl/py-pearl-mining/` is a PyO3 crate exposing Pearl mining primitives in Python. **Read it before designing the Metal port** — it likely contains the protocol-relevant constants in a hardware-neutral form, suitable as a reference oracle for Phase 0 testing (§5).
|
||||
|
||||
## 4. Scope
|
||||
|
||||
### 4.1 In scope
|
||||
|
||||
- **Phase 0** (§5): protocol-acceptance verification + Pearl-team coordination + Python reference oracle
|
||||
- **Phase 1** (§6.1): Metal NoisyGEMM kernel — the substantive engineering
|
||||
- **Phase 2** (§6.2): inference-backend plugin — MLX or llama.cpp Metal
|
||||
- **Phase 3** (§7): OJ provider integration — new `MiningProvider` impl, new optional extra, registry hookup
|
||||
- **Phase 4** (§8): verification matrix across Apple Silicon variants and bringup on Pearl testnet
|
||||
- Documentation deliverables and the upstream-contribution path
|
||||
|
||||
### 4.2 Out of scope
|
||||
|
||||
- Pool support and the 20% OJ fee (Spec A §8.5; that lives in a future v2 pool spec)
|
||||
- Custody, signing, or routing Pearl funds (Spec A anti-goal; same here)
|
||||
- AMD ROCm enablement (separate spec, parallel structure to this one)
|
||||
- Intel Arc / Mac Intel / older CUDA enablement (separate specs)
|
||||
- Modifying anything in Spec A. **Spec B is purely additive.**
|
||||
- Pearl protocol changes (none required; see §3.1)
|
||||
|
||||
### 4.3 Explicit non-goal: economic competitiveness
|
||||
|
||||
This spec does not promise that Apple Silicon mining will be **profitable**. The performance gap to a tuned H100 kernel is likely large (§6.1.5 discusses why). What this spec *does* promise: a correct, working Apple Silicon path that's enabled the day the kernel ships, with a transparent doctor surface that tells Mac users honestly what their hashrate looks like. Whether it's worth the electricity is a user decision.
|
||||
|
||||
## 5. Phase 0 — investigation, coordination, reference oracle
|
||||
|
||||
The phase that costs the least and prevents the most rework. Three workstreams in parallel.
|
||||
|
||||
### 5.1 Workstream P0-A: Pearl-side coordination
|
||||
|
||||
**Goal:** Confirm protocol acceptance in writing from Pearl maintainers; align on whether OJ contributes upstream or ships independently.
|
||||
|
||||
**Steps:**
|
||||
|
||||
1. Open a GitHub Discussion on `pearl-research-labs/pearl`: "Apple Silicon / Metal NoisyGEMM enablement — coordination". Reference Spec B URL.
|
||||
2. Get explicit confirmation from a Pearl maintainer that:
|
||||
- Validator path is hardware-neutral as believed (§3.1).
|
||||
- There is no Pearl-internal Metal port already in flight that would conflict.
|
||||
- LICENSE compatibility allows OJ-authored kernel code to be either contributed upstream (preferred) or distributed alongside OJ.
|
||||
3. Discuss the upstream-vs-fork question. Strong default: **contribute upstream into a new `pearl/miner/pearl-gemm-metal/` crate**, paralleling `pearl-gemm/`, so Pearl owns the kernel long-term and we benefit from their CI and review. Fork only if upstream contribution is blocked.
|
||||
|
||||
**Exit criteria:**
|
||||
|
||||
- [ ] Written confirmation of protocol acceptance
|
||||
- [ ] Agreement on contribution model (upstream / coordinated fork / independent)
|
||||
- [ ] No duplicate-effort risk
|
||||
|
||||
### 5.2 Workstream P0-B: build a reference oracle
|
||||
|
||||
**Goal:** A pure-Python (or pure-Rust) implementation of NoisyGEMM that produces bit-exact-or-fp16-tolerance-bounded outputs versus Pearl's CUDA reference. **Used as the test oracle for Phase 1** — without it, you can't verify the Metal kernel's correctness against a portable baseline.
|
||||
|
||||
**Steps:**
|
||||
|
||||
1. Read `pearl/miner/pearl-gemm/csrc/gemm/` end-to-end. Catalog the protocol-relevant constants in `pearl_gemm_constants.hpp`:
|
||||
- `kAxEBLScaleFactor = 1 << 14`
|
||||
- `kEARxBpEBScaleFactor = 1 << 12`
|
||||
- `kIntToFp16ScaleFactor = 1 << 12`
|
||||
- `kEBRScaleFactorDenoise`, `kEALScaleFactorDenoise`
|
||||
2. Read `pearl/py-pearl-mining/` to see what's already exposed in Python. If a reference impl already lives there, **use it**; do not duplicate.
|
||||
3. If gaps exist, build them in PyTorch (CPU). Mirror the structure of the CUDA kernels:
|
||||
- `noise_generation.cu` → `noise_generation.py` — derive `EAL`, `EAR`, `EBL`, `EBR` from blake3-of-key + seed
|
||||
- `pearl_gemm` (matmul + noise) → `pearl_gemm.py` — compute `Y_noisy = (A + EAL·EAR) × (B + EBL·EBR)` with the documented scaling
|
||||
- `inner_hash_kernel.cu` → `inner_hash.py` — blake3 commitment over the noised matmul
|
||||
- `denoise_converter.cu` → `denoise.py` — recover `Y_clean = A·B` from `Y_noisy` and the noise components
|
||||
- `pow_utils.hpp` → `pow_check.py` — difficulty target check
|
||||
4. Cross-check: run a corpus of 100+ inputs through the Pearl CUDA reference (on an H100 dev box; see §5.4) and through the Python reference. Assert outputs match within the documented tolerance — most likely **bit-exact for int paths and fp16-tolerance for the denoised result**.
|
||||
|
||||
**Exit criteria:**
|
||||
|
||||
- [ ] `tools/pearl-reference-oracle/` (in OJ repo, or separate repo) builds and tests pass
|
||||
- [ ] Parity confirmed against Pearl CUDA on ≥100 input sets
|
||||
- [ ] Constants table documented in this spec (replace the bullet list above with the verified values)
|
||||
|
||||
### 5.3 Workstream P0-C: Apple-side viability
|
||||
|
||||
**Goal:** Decide between MLX and llama.cpp Metal as the integration host before designing the kernel.
|
||||
|
||||
**Decision criteria:**
|
||||
|
||||
| Factor | MLX (`ml-explore/mlx`, `mlx-lm`) | llama.cpp Metal (`ggerganov/llama.cpp`) |
|
||||
|---|---|---|
|
||||
| Op-replacement hooks | Less mature; would likely require monkey-patching `mlx.nn.Linear` or upstream PR adding plugin hooks | More mature; `ggml` op tree is open and Metal backend has clear extension points (`ggml-metal.metal`) |
|
||||
| Apple-native quantization story | Excellent (4-bit, 8-bit native ops) | Good but not as native |
|
||||
| Inference-quality fidelity for OJ users today | High — MLX-LM is the de facto Mac LLM stack | High — also widely used |
|
||||
| Upstream-contribution complexity | Higher (smaller team, less plugin culture) | Lower (large open community, clear contributor flow) |
|
||||
| Ecosystem alignment with OJ engine map | OJ's `engine/` doesn't currently have an MLX engine; would need both | OJ already has llama.cpp via `engine/openai_compat_engines.py` |
|
||||
|
||||
**Recommendation:** **llama.cpp Metal first**, MLX as a fast-follow. Reasoning: ggml's op tree gives a cleaner extension path for a custom NoisyGEMM op; OJ already has llama.cpp engine wiring; and the upstream-contribution path is more navigable. MLX is a better long-term fit for Apple-native users but is currently a harder integration target.
|
||||
|
||||
**Steps:**
|
||||
|
||||
1. Spike: implement a no-op "custom op" passthrough in llama.cpp Metal. ~1-2 days work to confirm the integration mechanism is real and the build pipeline cooperates.
|
||||
2. Spike: same in MLX. Compare effort.
|
||||
3. Pick one. Document the decision in this spec.
|
||||
|
||||
**Exit criteria:**
|
||||
|
||||
- [ ] Decision made and documented in §6.2
|
||||
- [ ] Trivial plugin hook proven on the chosen backend
|
||||
|
||||
### 5.4 Hardware required for Phase 0
|
||||
|
||||
- One H100/H200 box (cloud rental fine — Lambda, RunPod, Crusoe). Used for: running Pearl's CUDA reference to capture parity test vectors, running the Pearl Docker miner end-to-end as a known-good baseline.
|
||||
- Apple Silicon dev machines: M2 Max minimum, M3/M4 Pro+ preferred. M-series Ultra ideal for any perf experiments.
|
||||
- Estimated cloud cost for Phase 0: < $200.
|
||||
|
||||
## 6. Phases 1 and 2 — kernel and plugin
|
||||
|
||||
### 6.1 Phase 1 — Metal NoisyGEMM kernel
|
||||
|
||||
**Goal:** A Metal compute-shader implementation of NoisyGEMM that produces outputs matching the Phase 0 reference oracle, performant enough to make Mac mining a real (if low-yield) feature.
|
||||
|
||||
#### 6.1.1 Implementation surface
|
||||
|
||||
Two viable targets, in order of preference:
|
||||
|
||||
**A. Direct Metal Shading Language (MSL) compute kernels.** Maximum control, maximum performance ceiling, maximum effort. The CUDA reference is highly tuned (TMA, WGMMA, multi-stage pipelines); a direct MSL port can lean on Apple's matmul intrinsics where they exist (`simdgroup_matrix` ops on M3+).
|
||||
|
||||
**B. Metal Performance Shaders Graph (MPSGraph).** Higher-level than raw MSL; uses Apple's tuned matmul kernels under the hood; limited control over the in-kernel commitment hash. Likely path: do the matmul via MPSGraph, do noise generation + commitment hashing as separate kernels, accept the perf hit from less fusion.
|
||||
|
||||
**Recommendation:** Start with B for correctness and shipping speed; profile; move hot paths to A only if economically justified. Apple's matmul intrinsics are fast enough that the perf gap to a fused implementation may be acceptable.
|
||||
|
||||
#### 6.1.2 Algorithm structure
|
||||
|
||||
Following the Pearl CUDA reference, end-to-end work performed for one mining attempt:
|
||||
|
||||
1. **Quantize inputs.** `A: fp16 → int8 + scale_A`, `B: fp16 → int8 + scale_B`. Match Pearl's `quantize_kernel.cu` semantics (per-row or per-channel scales — verify via Phase 0 oracle).
|
||||
2. **Generate noise tensors.** From `key_A`, `key_B` (per-block blake3-derived seeds), produce `EAL` (m, R), `EAR` (k, R), `EBL` (k, R), `EBR` (n, R) of int8. Scale factors per `pearl_gemm_constants.hpp`.
|
||||
3. **Noisy matmul.** Compute `Y_noisy = (A + EAL · EAR_T) × (B + EBL · EBR_T)`. Output is int32 then converted to fp16.
|
||||
4. **Inner-hash commitment.** blake3 over `Y_noisy` (or a row-tile of it) to produce the PoW target candidate. This is the hottest path — the noise + commitment loop runs at every share.
|
||||
5. **PoW check.** Compare commitment digest against the difficulty target (`make_pow_target_tensor` semantics from Pearl's Python interface).
|
||||
6. **On hit: denoise.** Compute `Y_clean = Y_noisy - (noise contributions)` to feed back into vLLM/MLX as the actual matmul output. Inference cannot be wrong.
|
||||
7. **Post-hit: STARK proof generation.** When a share meets the network difficulty target, the miner generates a plonky2 STARK proof tying the noisy matmul + commitment to the block. This proving step is **separate from the Metal kernel** — it runs in pure Rust via Pearl's existing `zk-pow/` and `py-pearl-mining` code paths and should work cross-platform unchanged. Cost: seconds-to-minutes of CPU per block. Confirm cross-platform builds during Phase 0-C and §7.5.
|
||||
|
||||
#### 6.1.3 Crate / package layout
|
||||
|
||||
Strong preference: **upstream contribution to Pearl** as `pearl/miner/pearl-gemm-metal/` paralleling the existing `pearl-gemm/`:
|
||||
|
||||
```
|
||||
pearl/miner/pearl-gemm-metal/
|
||||
Cargo.toml (or pyproject.toml + setup.py — match Pearl conventions)
|
||||
metal/ (.metal MSL source files)
|
||||
src/
|
||||
lib.rs (or src/pearl_gemm_metal/__init__.py)
|
||||
tests/
|
||||
```
|
||||
|
||||
If upstream contribution is blocked (Phase 0 outcome), fork with attribution into `OpenJarvis/vendor/pearl-gemm-metal/` and document the divergence policy in this spec.
|
||||
|
||||
#### 6.1.4 Testing
|
||||
|
||||
- **Parity tests.** Each kernel (noise gen, matmul, inner hash, denoise, PoW check) tested independently against the Phase 0 reference oracle. Bit-exact for int paths; fp16-tolerance bounded for fp paths (specific tolerance: TBD via Phase 0 measurement).
|
||||
- **End-to-end correctness.** Full mining attempt produces a candidate proof that the reference Rust prover (`zk-pow/`) accepts.
|
||||
- **Hardware fuzz.** Run on M1 Pro, M2 Max, M3 Max, M4 Max, and M-Ultra variants. Catch any silently-wrong hardware behavior (Metal feature variance across generations is real).
|
||||
|
||||
#### 6.1.5 Performance expectations
|
||||
|
||||
Honest baseline: **expect 0.05–0.2× the share rate of an H100** on a high-end M-Ultra, and proportionally less on smaller chips. Reasons:
|
||||
|
||||
- H100 has dedicated FP8/FP16 tensor cores with WGMMA throughput Apple Silicon does not match
|
||||
- Pearl's CUDA kernel is heavily fused (matmul + noise + commitment in one kernel via TMA pipelining); a Metal version will likely be less fused
|
||||
- 70B model bandwidth requirements stress unified memory
|
||||
|
||||
This is fine. Mac mining is a feature for Apple Silicon owners who want to participate, not a competitive yield product. Document it transparently in `mine doctor` and the user guide.
|
||||
|
||||
#### 6.1.6 Exit criteria for Phase 1
|
||||
|
||||
- [ ] Parity tests pass on M2 Max and M4 Max
|
||||
- [ ] End-to-end mining attempt produces a valid proof accepted by `zk-pow::verify_block`
|
||||
- [ ] Performance characterized and published (M-series matrix)
|
||||
- [ ] Code merged upstream OR forked-with-policy per Phase 0 outcome
|
||||
|
||||
### 6.2 Phase 2 — Inference-backend plugin
|
||||
|
||||
**Goal:** A llama.cpp Metal (or MLX, per Phase 0-C) plugin that swaps the standard quantized linear op for Phase 1's NoisyGEMM during inference, so a Mac running this plugin produces both correct LLM outputs and valid mining shares.
|
||||
|
||||
#### 6.2.1 Path: llama.cpp Metal (assuming Phase 0-C selected this)
|
||||
|
||||
- Add a custom `ggml` op `GGML_OP_PEARL_NOISY_GEMM` with a Metal backend implementation that calls Phase 1's kernels.
|
||||
- Plugin entry point: a small library that, when loaded, replaces the default linear op in the model graph during loading.
|
||||
- Build artifact: `libpearl_metal_plugin.dylib` (or static lib).
|
||||
|
||||
#### 6.2.2 Path: MLX (alternate)
|
||||
|
||||
- Define `mlx.NoisyLinear` as a subclass of `mlx.nn.Linear` that calls Phase 1's kernels via a custom Metal op binding.
|
||||
- Provide a model-loading shim: `from openjarvis.mining import patch_mlx_for_pearl; patch_mlx_for_pearl()` that monkey-patches `mlx.nn.Linear` instances at load time. Less elegant; works.
|
||||
|
||||
#### 6.2.3 Inference-quality regression tests
|
||||
|
||||
The plugin is correctness-critical: a noised model that doesn't fully denoise produces degraded responses. Test:
|
||||
|
||||
- Load a small reference model (e.g., a 1-3B parameter Pearl-blessed model if one exists for testing, otherwise the smallest model the protocol accepts).
|
||||
- Run a fixed prompt set through both noised+denoised and standard paths.
|
||||
- Assert outputs are bit-exact or within fp16 tolerance.
|
||||
- Run OJ's existing eval framework (`src/openjarvis/evals/`) on a small benchmark (e.g., an MMLU subset registered as a Pearl-mining-mode dataset). Assert no degradation > the tolerance budget. Falling back to `lm-eval-harness` is acceptable if OJ's eval surface for Mac is incomplete at the time.
|
||||
|
||||
#### 6.2.4 Exit criteria
|
||||
|
||||
- [ ] Plugin loads in chosen backend
|
||||
- [ ] End-to-end inference produces correct outputs (regression tests pass)
|
||||
- [ ] Mining shares are submitted to a Pearl testnet during inference
|
||||
- [ ] At least one block found on testnet from a Mac
|
||||
|
||||
## 7. Phase 3 — OpenJarvis provider integration
|
||||
|
||||
Where the OJ-side work is small. Inherits the entire `MiningProvider` ABC, registry, sidecar, config schema, telemetry adapter, and v2 seams from Spec A unchanged.
|
||||
|
||||
### 7.1 New files
|
||||
|
||||
```
|
||||
src/openjarvis/mining/
|
||||
llamacpp_pearl_metal.py # OR mlx_pearl.py — depending on Phase 2 path
|
||||
# @MinerRegistry.register("llamacpp-pearl-metal")
|
||||
# implements MiningProvider ABC from Spec A §4.4
|
||||
```
|
||||
|
||||
### 7.2 New optional extra
|
||||
|
||||
```toml
|
||||
# pyproject.toml
|
||||
mining-pearl-metal = [
|
||||
"pearl-metal-plugin>=0.1", # the Phase 2 plugin, however published
|
||||
# MLX path adds: "mlx>=0.X", "mlx-lm>=0.X"
|
||||
# llama.cpp path adds: "llama-cpp-python>=0.X" with Metal extras
|
||||
]
|
||||
```
|
||||
|
||||
### 7.3 Capability detection
|
||||
|
||||
```python
|
||||
# src/openjarvis/mining/llamacpp_pearl_metal.py
|
||||
class LlamaCppPearlMetalProvider(MiningProvider):
|
||||
provider_id = "llamacpp-pearl-metal"
|
||||
|
||||
@classmethod
|
||||
def detect(cls, hw: HardwareInfo, engine_id: str, model: str) -> MiningCapabilities:
|
||||
if hw.platform != "darwin":
|
||||
return MiningCapabilities(False, reason="Apple Silicon required (platform != darwin)")
|
||||
if hw.gpu is None or hw.gpu.vendor != "apple":
|
||||
return MiningCapabilities(False, reason="Apple Silicon GPU required")
|
||||
if engine_id not in {"llamacpp", "llama-cpp"}:
|
||||
return MiningCapabilities(False, reason=f"engine '{engine_id}' has no Pearl Metal plugin; use llamacpp")
|
||||
if not _pearl_metal_plugin_available():
|
||||
return MiningCapabilities(False, reason="install with `uv sync --extra mining-pearl-metal`")
|
||||
if not _model_has_pearl_variant(model):
|
||||
return MiningCapabilities(False, reason=f"model '{model}' has no Pearl-blessed variant")
|
||||
return MiningCapabilities(True, estimated_hashrate=_estimate_hashrate(hw))
|
||||
```
|
||||
|
||||
Each branch is exactly the kind of "why can't I mine" message Spec A's `mine doctor` surfaces verbatim.
|
||||
|
||||
### 7.4 Lifecycle
|
||||
|
||||
Unlike Spec A's vLLM provider which orchestrates a Docker container, the Apple provider runs **two coordinated subprocesses directly on the host**:
|
||||
|
||||
1. The inference server (llama.cpp server with the Pearl Metal plugin loaded, or MLX-LM server depending on Phase 0-C path)
|
||||
2. `pearl-gateway` as a sibling process — same one that runs inside the Docker container in Spec A, but here it runs natively on the Mac
|
||||
|
||||
Lifecycle:
|
||||
|
||||
- `start()`: spawn (1) with the Pearl Metal plugin pre-loaded (`DYLD_INSERT_LIBRARIES`-style or `--plugin` flag depending on chosen backend's invocation contract), then spawn (2) pointing at it. Write the same sidecar shape Spec A defines, with `gateway_url` pointing at the native pearl-gateway. Track both PIDs internally.
|
||||
- `stop()`: SIGTERM (2) first, then (1), with bounded waits and SIGKILL fallback. Remove sidecar.
|
||||
- `is_running()`, `stats()`: identical contract to vLLM provider; `stats()` reads from the native pearl-gateway's `:8339/metrics`.
|
||||
|
||||
**No Docker.** Apple Silicon Docker doesn't pass through Metal; running Pearl in a Mac Docker container would defeat the purpose. Document this explicitly in §7 of this spec; do not attempt a Docker path.
|
||||
|
||||
### 7.5 Pearl gateway on Mac
|
||||
|
||||
The Pearl `pearl-gateway` process is currently only documented as part of the Docker container. For Mac, we need it to run natively. Two options:
|
||||
|
||||
1. Build `pearl-gateway` from source via `uv sync --package pearl-gateway` — same workspace package the Docker image uses. Should work cross-platform since it's pure Python plus py-pearl-mining bindings. Verify.
|
||||
2. If (1) fails on Apple Silicon, work with Pearl maintainers (Phase 0-A) to port it — a small amount of work compared to the kernel.
|
||||
|
||||
**Phase 3 verifies (1).** This is a Phase 0-A coordination point.
|
||||
|
||||
### 7.6 Exit criteria
|
||||
|
||||
- [ ] `LlamaCppPearlMetalProvider` registered, detection matrix correct on M1/M2/M3/M4
|
||||
- [ ] `jarvis mine init` runs to completion on Apple Silicon
|
||||
- [ ] `jarvis mine start` launches subprocess + Pearl gateway on Mac
|
||||
- [ ] `jarvis mine status` returns valid `MiningStats` from a real Mac mining session
|
||||
- [ ] `jarvis mine doctor` produces honest, actionable output for Mac users
|
||||
|
||||
## 8. Phase 4 — Verification & bringup
|
||||
|
||||
### 8.1 Hardware matrix
|
||||
|
||||
| Chip | Test priority | Expected outcome |
|
||||
|---|---|---|
|
||||
| M1 / M1 Pro / M1 Max | low — generation 1 GPU may have feature gaps | works but slow |
|
||||
| M2 / M2 Pro / M2 Max | medium | works |
|
||||
| M2 Ultra | medium | best M2-class hashrate |
|
||||
| M3 / M3 Pro / M3 Max | high — first gen with `simdgroup_matrix` | works, meaningful share rate |
|
||||
| M4 / M4 Pro / M4 Max | high — current flagship | best non-M-Ultra hashrate |
|
||||
|
||||
For each chip in the matrix, run:
|
||||
|
||||
1. `jarvis mine init` end-to-end
|
||||
2. `jarvis mine start` and run for ≥4 h continuous
|
||||
3. Capture and publish: shares submitted, shares accepted, block-find time distribution, GPU temp, system load impact on normal use
|
||||
4. Run a parallel `lm-eval-harness` on the mining endpoint to assert inference quality is unaffected
|
||||
|
||||
### 8.2 Pearl testnet bringup
|
||||
|
||||
Before any mainnet recommendation:
|
||||
|
||||
- Mine on Pearl testnet for ≥7 continuous days from at least two Apple Silicon variants
|
||||
- Find at least one block on testnet from each variant
|
||||
- Verify all blocks accepted by `zk-pow::verify_block` on a reference validator node
|
||||
- Report results to Pearl maintainers; gate any mainnet announcement on their sign-off
|
||||
|
||||
### 8.3 Documentation deliverables
|
||||
|
||||
- `docs/user-guide/mining-apple-silicon.md` — user-facing: prerequisites, install flow, doctor reading guide, performance expectations table, links to share-rate calculators
|
||||
- `docs/development/mining-providers.md` — generalized "how to add a new provider" guide using this spec as the canonical worked example
|
||||
- An update to `docs/user-guide/mining.md` (Spec A) adding Apple Silicon to the supported-platforms list
|
||||
|
||||
### 8.4 Exit criteria
|
||||
|
||||
- [ ] Hardware matrix covered
|
||||
- [ ] Testnet bringup complete
|
||||
- [ ] Documentation merged
|
||||
- [ ] Pearl maintainer sign-off obtained
|
||||
- [ ] OJ release notes call out Apple Silicon mining as supported
|
||||
|
||||
## 9. Risks
|
||||
|
||||
| ID | Risk | Likelihood | Impact | Mitigation |
|
||||
|---|---|---|---|---|
|
||||
| R1 | Pearl validator rejects non-CUDA-mined blocks despite hardware-neutral validator code | low (validator code reviewed) | catastrophic (whole spec invalid) | Phase 0-A explicit confirmation; Phase 0-B oracle reduces likelihood of math drift |
|
||||
| R2 | Pearl ships their own Metal port, conflicts with OJ's | medium (depends on Pearl roadmap) | high (rework or fork) | Phase 0-A coordination; default to upstream contribution |
|
||||
| R3 | Metal NoisyGEMM is so slow that mining is uneconomical even for hobbyists | medium-high | medium (feature ships but unused) | §4.3 names this as a non-goal; transparency in `mine doctor`; consider M-Ultra-only-by-default in v1 of this spec |
|
||||
| R4 | NoisyGEMM correctness bug → invalid blocks → wasted user electricity | low if §6.1.4 testing rigorous | high (trust hit) | Strong parity testing against oracle; testnet bringup before mainnet |
|
||||
| R5 | Inference-quality regression — denoised path doesn't fully recover model fidelity | medium | high | §6.2.3 regression tests; eval-harness gate before ship |
|
||||
| R6 | Pearl protocol changes between Phase 0 and Phase 4 (multi-month) | medium | medium | Pin Phase 0 ref same as Spec A; renegotiate at each Pearl rev |
|
||||
| R7 | Apple changes Metal API in macOS update | low-medium | medium | Use stable MSL features; pin Xcode toolchain |
|
||||
| R8 | Upstream Pearl PR rejected | low (Pearl wants this) | medium (forced fork) | Phase 0-A negotiates upstream-vs-fork up front |
|
||||
| R9 | `pearl-gateway` doesn't build on Apple Silicon (§7.5) | medium (it's Python — should work, but py-pearl-mining has Rust deps) | low (small fix) | Phase 0 verifies builds; Phase 0-A coordination if not |
|
||||
|
||||
## 10. Open questions
|
||||
|
||||
Phase 0 answered most of these from the upstream code (annotations below). The remaining open items are ones that require either Pearl maintainer input or empirical measurement on real network conditions.
|
||||
|
||||
1. **(Open — coordination)** Is there a Pearl-blessed "small" model for testing? The reference miner uses a 70B model — too big for fast iteration. A 7B or 13B variant for development would dramatically speed up v2/v3 plugin work. *Less critical for v1, since v1 is decoupled from inference and runs `mine()` directly without a model.*
|
||||
2. **(Answered — N/A for v1)** ~~Documented fp16 tolerance budget for denoised matmul output~~ — `miner-base/tests/test_noisy_gemm.py:92` does `torch.equal(result, expected)`: the int7×int7→int32 path is **bit-exact**, no fp16 tolerance budget is needed. (May reappear in v3 Metal kernel work if int↔fp16 conversions are introduced for perf.)
|
||||
3. **(Answered — yes)** Does `py-pearl-mining` already expose enough of NoisyGEMM in Python that Phase 0-B becomes a thin wrapper? **Yes.** `pearl_mining.mine` runs the entire mining algorithm in pure Rust. Additionally, `miner-base.NoisyGemm` provides a PyTorch reference of the production NoisyGEMM. Phase 0-B's "build a reference oracle" deliverable is now a *thin OJ-side wrapper* that calls upstream — see §13 and `tools/pearl-reference-oracle/` (created in this session).
|
||||
4. **(Answered — likely yes; empirically verified for `py-pearl-mining`)** Is `pearl-gateway` cross-platform? Its `pyproject.toml` requires Python ≥ 3.10 and depends on `aiohttp`, `bitcoin-utils`, `blake3`, `numpy`, `prometheus-client`, `pybase64`, `pydantic`, `pyyaml`, `torch==2.11.0`, `py-pearl-mining` — all install on macOS arm64. Empirical install of the workspace was not run in this session; **action item for v1 implementation**: `uv sync` the workspace on macOS-15 and capture the build output.
|
||||
5. **(Open — coordination)** Does Pearl gate any difficulty / consensus parameters on hardware introspection? Code review found none. Confirm in Phase 0-A discussion.
|
||||
6. **(Open — measurement)** Minimum acceptable hashrate floor for `jarvis mine init` on Apple Silicon. The `MiningCapabilities.estimated_hashrate` field in Spec A §4.4 exists for this. v1 will populate from a calibration run during `mine init`. The *floor* is a policy decision, not a technical one — defer to user-research / community feedback once v1 ships.
|
||||
7. **(Open — coordination)** Upstream contribution / CLA / LICENSE. Pearl is ISC; OJ is Apache-2.0; both are permissive and combine cleanly. **CLA TBD via Phase 0-A discussion**, but for v1 this is moot — OJ contributes no code into Pearl's tree, only consumes their published Python packages.
|
||||
8. **(Answered — yes)** Apple Silicon CI on GitHub Actions: `macos-14` / `macos-15` runners are arm64 and can install `py-pearl-mining` via the wheel build verified in §1.5.4. OJ's CI can run mining unit tests. Mining the *real* network in CI is still out of scope.
|
||||
9. **(Answered — `"llamacpp"`)** OJ's llama.cpp engine_id is `"llamacpp"` (single token, no hyphen). Confirmed at `src/openjarvis/engine/openai_compat_engines.py:9` and `src/openjarvis/engine/_discovery.py:18`. Update §7.3 capability detection to use this key. *(For v1 in §13, this only matters if we add an "informational" mining-aware hint to the existing llamacpp engine — v1 does not require any plugin into the engine.)*
|
||||
10. **(Open — measurement, but de-risked)** plonky2 STARK proving latency on Apple Silicon CPU. Spec A §1 already notes proving is seconds-to-minutes of CPU per block (cross-platform, runs unchanged). For v1 the hashrate is so low that block-find latency is dominated by the search, not the proof. Empirical measurement still needed for v2/v3.
|
||||
11. **(New — v1 specific)** Does `bitcoin-utils>=0.7.0` (a `pearl-gateway` dependency) have C extensions that need Apple-specific build flags? Likely pure-Python; verify during the v1 install workstream.
|
||||
12. **(New — v1 specific)** Will `torch==2.11.0` (the version `miner-base` and `pearl-gateway` pin) install cleanly on macOS arm64? PyTorch generally has arm64 macOS wheels. Verify during v1 install.
|
||||
|
||||
## 11. Cross-references
|
||||
|
||||
- **[Spec A](2026-05-05-vllm-pearl-mining-integration-design.md)** — the v1 integration this extends. Read §4.4 (the `MiningProvider` ABC), §5.3 (sidecar shape), §8.1–8.2 (telemetry adapter contract), §8.5 (v2 fee/pool seams). All apply unchanged.
|
||||
- **Pearl coordination thread (P0-A draft):** [`2026-05-05-pearl-coordination-discussion-draft.md`](2026-05-05-pearl-coordination-discussion-draft.md) — content the user posts on `pearl-research-labs/pearl` to confirm protocol acceptance and align on contribution model.
|
||||
- **OJ-side Phase 0 deliverables (created this session):**
|
||||
- `tools/pearl-reference-oracle/` — thin Python wrapper around upstream Pearl bindings + smoke test, runnable on Apple Silicon
|
||||
- **Pearl repo paths read in Phase 0 (in priority order):**
|
||||
1. `pearl/zk-pow/src/api/verify.rs` — the validator. Pure Rust, no GPU. **Hardware-neutrality verified.**
|
||||
2. `pearl/zk-pow/src/api/proof.rs` — `PublicProofParams`, `ZKProof`, `PrivateProofParams`, `IncompleteBlockHeader`, `MiningConfiguration`, `MMAType`. Defines what the protocol commits to.
|
||||
3. `pearl/zk-pow/src/ffi/mine.rs` — **the entire hardware-neutral mining function**. Pure Rust. Already exposed to Python.
|
||||
4. `pearl/zk-pow/src/circuit/pearl_noise.rs` — noise generation: `compute_noise_for_indices`, `generate_uniform_random_matrix`, `generate_permutation_matrix`. Hardware-neutral.
|
||||
5. `pearl/py-pearl-mining/src/lib.rs` — PyO3 module. Re-exports `mine`, `verify_plain_proof`, `generate_proof`, `verify_proof`, `warmup_prove`. **Builds on macOS arm64, verified §1.5.4.**
|
||||
6. `pearl/py-pearl-mining/Cargo.toml` — pure Rust deps: `pearl-blake3`, `zk-pow`, `blake3`, `rayon`, `pyo3`, `lazy_static`, `tikv-jemallocator`. No CUDA in tree.
|
||||
7. `pearl/py-pearl-mining/tests/test_python_api.py` — the canonical end-to-end test. Use as the OJ smoke-test template.
|
||||
8. `pearl/miner/miner-base/src/miner_base/noisy_gemm.py` — PyTorch reference of the production NoisyGEMM. The "reference oracle" §5.2 wanted to build is here.
|
||||
9. `pearl/miner/miner-base/src/miner_base/noise_generation.py` — PyTorch noise generation matching `pearl_noise.rs`.
|
||||
10. `pearl/miner/miner-base/src/miner_base/inner_hash.py` — PyTorch inner-hash with XOR reduction.
|
||||
11. `pearl/miner/miner-base/tests/test_noisy_gemm.py` — bit-exact denoising verified at line 92.
|
||||
12. `pearl/miner/miner-base/pyproject.toml` — deps (`torch==2.11.0`, `blake3`, `numpy`, `pearl-gateway`, `py-pearl-mining`); **no platform markers** → installs on Apple Silicon.
|
||||
13. `pearl/miner/pearl-gateway/pyproject.toml` — deps (pure Python + py-pearl-mining + torch); **no platform markers**.
|
||||
14. `pearl/miner/vllm-miner/src/vllm_miner/register.py` — vLLM plugin registration via `vllm.general_plugins` entry point. The pattern Phase 2 (v2 plan) would mirror.
|
||||
15. `pearl/miner/pearl-gemm/csrc/gemm/pearl_gemm_constants.hpp` — protocol scale factors. Verified values:
|
||||
- `kAxEBLScaleFactor = 1<<14 = 16384`
|
||||
- `kEARxBpEBScaleFactor = 1<<12 = 4096`
|
||||
- `kIntToFp16ScaleFactor = 1<<12 = 4096`
|
||||
- `kEBRScaleFactorDenoise = -4` (= -kAxEBLScaleFactor / kIntToFp16ScaleFactor)
|
||||
- `kEALScaleFactorDenoise = -1` (= -kEARxBpEBScaleFactor / kIntToFp16ScaleFactor)
|
||||
16. `pearl/miner/pearl-gemm/setup.py:88` — `COMPUTE_CAPABILITY = "arch=compute_90a,code=sm_90a"`. Confirms CUDA kernel is Hopper-only.
|
||||
17. `pearl/Taskfile.yml` — `build:miner` task is gated to `platforms: [linux, windows]`. **The miner Python install path Pearl ships today is Linux/Windows-only**; OJ's v1 path uses the components that *do* install on macOS, sidestepping this gate.
|
||||
- **Pearl paper:** [Proof-of-Useful-Work via matrix multiplication (arXiv:2504.09971)](https://arxiv.org/abs/2504.09971) — read for the math formalization. Less critical now that the PyTorch reference exists upstream.
|
||||
- **Apple references (still relevant for v2/v3):**
|
||||
- [Metal Shading Language Specification](https://developer.apple.com/metal/Metal-Shading-Language-Specification.pdf)
|
||||
- [MPS / MPSGraph documentation](https://developer.apple.com/documentation/metalperformanceshadersgraph)
|
||||
- [MLX](https://github.com/ml-explore/mlx) — alternative v2 plugin host
|
||||
- [PyTorch MPS backend docs](https://pytorch.org/docs/stable/notes/mps.html) — relevant for v2 (MPS-accelerated decoupled mining)
|
||||
|
||||
## 12. Implementation plan
|
||||
|
||||
Two implementation plans now live alongside this spec:
|
||||
|
||||
- **v1 plan (decoupled CPU mining via upstream Pearl):** Written via `superpowers:writing-plans` after this Phase 0 update. Tracking issue: [`2026-05-05-apple-silicon-pearl-mining-plan-v1.md`](2026-05-05-apple-silicon-pearl-mining-plan-v1.md).
|
||||
- **v2 plan (PyTorch-MPS or MLX/llama.cpp coupled mining):** TBD. Written when v1 ships and we have empirical hashrate data justifying the next investment.
|
||||
- **v3 plan (native Metal NoisyGEMM kernel):** TBD. Written only if v2 measurements show the additional kernel work is economically justified.
|
||||
|
||||
The original §5–§8 content describing Phases 0–4 of the *kernel-first* approach is preserved as the v3 plan's reference. Do not delete it — when the time comes to write the v3 plan, that content is the starting point.
|
||||
|
||||
## 13. Apple Silicon v1 — minimal path
|
||||
|
||||
This section defines the v1 design that ships in weeks rather than months. v1 is **decoupled mining**: the user's existing inference workflow (Ollama, MLX-LM, llama.cpp, vLLM-on-CPU, anything) is untouched; mining runs as a separate process via the upstream Pearl miner.
|
||||
|
||||
### 13.1 Architecture
|
||||
|
||||
```
|
||||
OpenJarvis user (Apple Silicon)
|
||||
┌────────────────────────────────┐
|
||||
│ jarvis mine start │
|
||||
│ ↓ │
|
||||
│ CpuPearlProvider (this spec) │
|
||||
│ ↓ subprocess.Popen │
|
||||
│ ┌──────────────────────────┐ │
|
||||
│ │ pearl-gateway (Python) │ │
|
||||
│ │ ↑ JSON-RPC :8337 │ │
|
||||
│ │ pearl-mine-loop (Python) │ │ ← uses py-pearl-mining
|
||||
│ │ wraps pearl_mining.mine│ │ (pure Rust)
|
||||
│ └──────────────────────────┘ │
|
||||
│ │
|
||||
│ Inference (untouched) │
|
||||
│ ┌──────────────────────────┐ │
|
||||
│ │ Ollama / MLX / llamacpp │ │
|
||||
│ └──────────────────────────┘ │
|
||||
└────────────────────────────────┘
|
||||
↓
|
||||
pearld (BYO, same as Spec A)
|
||||
```
|
||||
|
||||
The mining loop wraps `pearl_mining.mine()` in a process that:
|
||||
|
||||
1. Polls `pearl-gateway` for current `IncompleteBlockHeader` + `MiningConfiguration`
|
||||
2. Calls `pearl_mining.mine(m, n, k, header, config)` to find a `PlainProof`
|
||||
3. Submits the proof back to `pearl-gateway`, which generates the ZK proof and forwards to `pearld`
|
||||
4. Loops
|
||||
|
||||
This is the *same* control flow vllm-miner runs — just without coupling the matmul to vLLM's inference. Pearl's existing `pearl-gateway` already does the orchestration we need; we just need a small mining loop that uses the CPU `mine()` instead of the CUDA path.
|
||||
|
||||
### 13.2 Module layout in OJ
|
||||
|
||||
```
|
||||
src/openjarvis/mining/
|
||||
cpu_pearl.py # @MinerRegistry.register("cpu-pearl") — v1 provider
|
||||
_pearl_subprocess.py # PearlSubprocessLauncher — gateway + miner subprocesses
|
||||
# Reused for future Apple-MPS / Metal providers
|
||||
src/openjarvis/cli/
|
||||
# mine_cmd.py is unchanged; cpu-pearl participates via the provider ABC
|
||||
|
||||
tests/mining/test_cpu_pearl.py
|
||||
tools/pearl-reference-oracle/
|
||||
README.md # documentation: oracle exists upstream
|
||||
smoke_test.py # end-to-end mine + verify smoke test (created in this session)
|
||||
```
|
||||
|
||||
### 13.3 Optional extra
|
||||
|
||||
```toml
|
||||
mining-pearl-cpu = [
|
||||
"py-pearl-mining>=0.1", # the wheel built in §1.5.4
|
||||
"miner-base>=0.1", # PyTorch reference (used for parity testing)
|
||||
"pearl-gateway>=0.1", # gateway service
|
||||
]
|
||||
```
|
||||
|
||||
When Pearl publishes these as PyPI wheels, the install is `uv sync --extra mining-pearl-cpu`. Until then, the spec for the implementation plan covers the local-build fallback (clone Pearl at the pinned ref, `maturin build` `py-pearl-mining`, `uv pip install` the workspace packages from local paths).
|
||||
|
||||
### 13.4 Capability detection
|
||||
|
||||
```python
|
||||
# src/openjarvis/mining/cpu_pearl.py
|
||||
class CpuPearlProvider(MiningProvider):
|
||||
provider_id = "cpu-pearl"
|
||||
|
||||
@classmethod
|
||||
def detect(cls, hw: HardwareInfo, engine_id: str, model: str) -> MiningCapabilities:
|
||||
# cpu-pearl is engine-independent — it doesn't plug into inference
|
||||
if not _pearl_mining_available():
|
||||
return MiningCapabilities(False, reason="install with `uv sync --extra mining-pearl-cpu`")
|
||||
if not _pearl_gateway_available():
|
||||
return MiningCapabilities(False, reason="pearl-gateway package not installed")
|
||||
if hw.platform not in {"darwin", "linux"}:
|
||||
return MiningCapabilities(False, reason=f"platform '{hw.platform}' not yet supported")
|
||||
# Optional: hardware-specific hashrate estimates
|
||||
return MiningCapabilities(True, estimated_hashrate=_estimate_cpu_hashrate(hw))
|
||||
```
|
||||
|
||||
The `engine_id` parameter is ignored because v1 is decoupled — mining works with **any** OJ engine, including no engine at all. (A future Apple-coupled provider would inspect `engine_id` to require `"llamacpp"` or `"mlx"`.)
|
||||
|
||||
### 13.5 Lifecycle (from Spec A's `MiningProvider` ABC)
|
||||
|
||||
- `start(config)`: spawn (1) `pearl-gateway` and (2) `pearl-mine-loop` subprocesses. Wait for gateway readiness on `:8339/metrics`. Write the standard sidecar JSON (Spec A §5.3) with `provider="cpu-pearl"`, gateway URL, and PIDs of both subprocesses.
|
||||
- `stop()`: SIGTERM mining loop, then gateway. Bounded waits, SIGKILL fallback.
|
||||
- `is_running()`: check sidecar + both PIDs.
|
||||
- `stats()`: read from `pearl-gateway`'s `:8339/metrics` exactly as Spec A §8.1 specifies. **Same metrics adapter contract.** No code changes in OJ's gateway-metrics adapter.
|
||||
|
||||
### 13.6 Configuration
|
||||
|
||||
Inherits Spec A's `[mining]` config schema unchanged. v1 uses:
|
||||
|
||||
```toml
|
||||
[mining]
|
||||
provider = "cpu-pearl" # NEW: was "vllm-pearl" in Spec A
|
||||
wallet_address = "prl1q..."
|
||||
submit_target = "solo"
|
||||
fee_bps = 0
|
||||
fee_payout_address = ""
|
||||
|
||||
[mining.extra]
|
||||
gateway_port = 8337
|
||||
metrics_port = 8339
|
||||
pearld_rpc_url = "http://localhost:44107"
|
||||
pearld_rpc_user = "rpcuser"
|
||||
pearld_rpc_password_env = "PEARLD_RPC_PASSWORD"
|
||||
# v1-specific: matmul shape for the search loop
|
||||
m = 256
|
||||
n = 128
|
||||
k = 1024
|
||||
rank = 32
|
||||
```
|
||||
|
||||
The `m / n / k / rank` shape can be tuned per Phase 0-A measurement (or per chip). Larger shapes search more space per call but use more memory.
|
||||
|
||||
### 13.7 Doctor surface (Apple Silicon)
|
||||
|
||||
```
|
||||
$ jarvis mine doctor
|
||||
Hardware
|
||||
GPU vendor apple ✓
|
||||
Apple chip M2 Max ✓
|
||||
Unified memory 96 GB ✓
|
||||
Pearl install
|
||||
py-pearl-mining 0.1.0 (cp312-abi3-macos-arm64) ✓
|
||||
miner-base 0.1.0 ✓
|
||||
pearl-gateway 0.1.0 ✓
|
||||
Pearl node
|
||||
RPC http://localhost:44107 ✓
|
||||
Auth ok ✓
|
||||
Block height 442107 (synced) ✓
|
||||
Wallet
|
||||
Address format prl1q... ✓
|
||||
Provider capability
|
||||
cpu-pearl SUPPORTED (est. 0.X share/h on M2 Max)
|
||||
Notes
|
||||
- This is decoupled mining: your normal LLM inference is unaffected
|
||||
- Hashrate is far below H100 mining; see docs/user-guide/mining-apple-silicon.md
|
||||
- Metal-accelerated mining: planned for v2; not available yet
|
||||
Session
|
||||
Sidecar absent (not running)
|
||||
```
|
||||
|
||||
Each row maps to a check function in `mining/_discovery.py`. The "est. share/h" line is populated from a one-time calibration during `mine init` — runs `pearl_mining.mine` in a 30-second loop and extrapolates.
|
||||
|
||||
### 13.8 v1 anti-goals
|
||||
|
||||
- **No coupling to inference.** The user's MLX-LM / Ollama / llama.cpp inference is untouched. v1 does not introduce a custom matmul. The "use AI = mine" narrative is **explicitly deferred to v2**.
|
||||
- **No Metal kernel.** All math is in upstream Rust + PyTorch + Python. Zero MSL written.
|
||||
- **No Pearl tree changes.** We consume their published packages; we contribute zero code into Pearl's repo for v1. (Phase 0-A discussion still happens — but it's lower-stakes since we're a downstream consumer in v1, not a contributor.)
|
||||
- **No upstream PRs blocking v1 ship.** v1 ships against the Pearl ref pinned in `mining/_constants.py` (Spec A §6) regardless of whether any of our coordination questions are answered.
|
||||
|
||||
### 13.9 v1 exit criteria
|
||||
|
||||
- [ ] `mining-pearl-cpu` extra installs cleanly on macOS arm64 (M1, M2, M3, M4 — at minimum the chip the spec author owns)
|
||||
- [ ] `jarvis mine init` completes successfully on macOS arm64
|
||||
- [ ] `jarvis mine start` launches gateway + miner subprocesses; sidecar valid; `mine status` reports live data
|
||||
- [ ] `mine doctor` produces honest, actionable output for Mac users
|
||||
- [ ] At least one block found on Pearl testnet from at least one Apple Silicon variant
|
||||
- [ ] User-facing doc `docs/user-guide/mining-apple-silicon.md` ships, including the honest hashrate caveat
|
||||
|
||||
### 13.10 Out of v1, into v2/v3
|
||||
|
||||
- **v2 (months):** Re-route `noisy_gemm` math to PyTorch-MPS for Apple Silicon GPU acceleration; integrate as a plugin into MLX-LM or `llama-cpp-python` so inference matmuls produce mining work (preserving the "use AI = mine" narrative). The original §5–§8 plan applies, swapped to use PyTorch MPS instead of raw MSL.
|
||||
- **v3 (months — optional, only if v2 perf is insufficient):** Native Metal Shading Language NoisyGEMM kernel as an upstream Pearl contribution. The original §5–§8 plan applies as written.
|
||||
|
||||
## 14. Phase 0 deliverables status (this session, 2026-05-05)
|
||||
|
||||
Tracking what was actually produced, against the §5 Phase 0 plan and the §1.5 reframing.
|
||||
|
||||
| Workstream | Original plan | Status | Deliverable |
|
||||
|---|---|---|---|
|
||||
| P0-A | Open Pearl GitHub Discussion, get protocol-acceptance confirmation | Draft written; user posts | `docs/design/2026-05-05-pearl-coordination-discussion-draft.md` |
|
||||
| P0-B | Build reference oracle from scratch in PyTorch, validate against H100 CUDA | **Reference oracle exists upstream.** Built thin OJ-side wrapper + verified empirically that `pearl_mining.mine` runs on Apple Silicon (78 ms / proof at test difficulty) | `tools/pearl-reference-oracle/` |
|
||||
| P0-C | Decide MLX vs llama.cpp Metal | **Deferred to v2.** v1 doesn't need either. | — |
|
||||
| Spec update | Capture findings | Done | This document, §1.5, §10–§14 |
|
||||
| v1 implementation plan | Plan written via `superpowers:writing-plans` after Phase 0 | Pending | `2026-05-05-apple-silicon-pearl-mining-plan-v1.md` (next deliverable) |
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,91 @@
|
||||
# Pearl coordination thread — draft
|
||||
|
||||
**For:** Posting on `pearl-research-labs/pearl` GitHub Discussions (Category: General / Q&A).
|
||||
**By:** OpenJarvis team (Stanford Hazy Research); contact: [user fills in].
|
||||
**Status:** Draft — review and edit before posting.
|
||||
|
||||
---
|
||||
|
||||
## Suggested title
|
||||
|
||||
> Apple Silicon support for Pearl mining — coordination & confirmation
|
||||
|
||||
## Suggested body
|
||||
|
||||
Hi Pearl team — we're [OpenJarvis](https://github.com/open-jarvis/OpenJarvis), a local-first personal AI agent framework from Stanford Hazy Research. We're working on a `mining` subsystem that lets OJ users mine Pearl through the agent framework. The first integration is the `vllm-miner`-on-H100/H200 path, which is straightforward. The second is Apple Silicon, where the situation is more interesting and we'd like to confirm a few things before we ship.
|
||||
|
||||
We have a v1 architecture that ships **today** using only your published Python packages (`py-pearl-mining`, `miner-base`, `pearl-gateway`) without any new code in your tree, plus an aspirational v2/v3 path that does involve potentially upstream contributions. Three asks below, plus a heads-up.
|
||||
|
||||
### What we built and verified locally (no protocol changes; all upstream code paths)
|
||||
|
||||
We read the Pearl source carefully — particularly:
|
||||
|
||||
- `zk-pow/src/api/verify.rs` — the validator
|
||||
- `zk-pow/src/ffi/mine.rs` — the pure-Rust `mine()` function
|
||||
- `zk-pow/src/circuit/pearl_noise.rs` — noise generation
|
||||
- `py-pearl-mining/` — the PyO3 bindings exposing the above to Python
|
||||
- `miner/miner-base/src/miner_base/noisy_gemm.py` — the PyTorch NoisyGEMM reference
|
||||
|
||||
…and then we built `py-pearl-mining` from source on an Apple Silicon M2 Max (macOS 26.4, Python 3.12, Rust 1.94). It produced `py_pearl_mining-0.1.0-cp312-abi3-macosx_11_0_arm64.whl` in ~56 seconds. We installed it and ran the `mine()` + `verify_plain_proof()` cycle from `tests/test_python_api.py`:
|
||||
|
||||
```
|
||||
running mine(m=256, n=128, k=1024, rank=32) on Apple Silicon CPU…
|
||||
mine() returned a proof in 0.078s
|
||||
verify_plain_proof: ok=True, msg='Mining solution verified successfully'
|
||||
```
|
||||
|
||||
So our v1 plan is: ship a CPU-mining mode for OJ users on Apple Silicon (and potentially other non-CUDA platforms) that wraps `pearl_mining.mine()` and your `pearl-gateway` as a subprocess. **We're not modifying anything in Pearl's tree for v1.** Just consuming what you've already published.
|
||||
|
||||
### Three asks
|
||||
|
||||
**1. Protocol acceptance confirmation.**
|
||||
|
||||
Reading the validator path, we believe `verify_block` and `verify_plain_proof` accept any `PlainProof` produced by a correct implementation, regardless of which hardware produced it. The plonky2 STARK and the difficulty check don't reference hardware.
|
||||
|
||||
**Could you confirm in writing that blocks mined via the pure-Rust `mine()` path (from a non-CUDA host like Apple Silicon) will be accepted by Pearl validators on testnet and mainnet?** We don't expect surprises here, but it's load-bearing for our spec and we want to record your sign-off before we ship.
|
||||
|
||||
**2. Heads-up: your `Taskfile.yml` restricts `build:miner` to `[linux, windows]`.**
|
||||
|
||||
That makes total sense for the GPU miner (CUDA + vLLM is Linux-only). But the `py-pearl-mining` and `miner-base` packages don't actually need that restriction — they install fine on macOS. We're working around the gate by installing the individual packages directly. Two questions:
|
||||
|
||||
- Is the `[linux, windows]` restriction load-bearing in some way we don't see (e.g., do you intend `py-pearl-mining` to remain a CUDA-bound dependency long-term)?
|
||||
- Would you be open to a small PR that splits `build:miner-cpu` (cross-platform) from `build:miner-gpu` (Linux + CUDA)? It would help downstream consumers like us — and any hobbyist who wants to experiment with `pearl_mining.mine()` on whatever hardware they own.
|
||||
|
||||
**3. PyPI publication of `py-pearl-mining` / `miner-base` / `pearl-gateway`.**
|
||||
|
||||
Do you have a roadmap for publishing these as PyPI wheels (`pip install py-pearl-mining` etc.)? Today we'd vendor a pinned commit and `maturin build` locally, which works but is brittle. If a 2026 PyPI publication is plausible, we'd defer the local-build code path; if it's not on the roadmap, we'll plan for the long-term local-build path.
|
||||
|
||||
### Aspirational (v2 / v3) — context only, no asks yet
|
||||
|
||||
Once v1 ships, we'd like to explore Apple-native acceleration:
|
||||
|
||||
- **v2:** Use PyTorch MPS to GPU-accelerate `miner-base.NoisyGemm` on Apple Silicon. Could potentially become a plugin into `mlx-lm` or `llama-cpp-python` so a Mac user's *inference* matmuls do mining work — same "useful work" framing as your vllm-miner. We don't need anything from Pearl for this; we'd build it on top of your existing PyTorch reference.
|
||||
- **v3 (only if v2 isn't enough):** A native Metal Shading Language port of NoisyGEMM, paralleling `pearl-gemm/`. That would be a real upstream contribution candidate (`pearl/miner/pearl-gemm-metal/`), and we'd want to coordinate with you before starting kernel work to avoid duplicate effort.
|
||||
|
||||
If you're already building Apple Silicon support internally (or have someone planning it), please tell us — we'd rather coordinate than duplicate.
|
||||
|
||||
### Logistics
|
||||
|
||||
- License compatibility: Pearl is ISC; OpenJarvis is Apache-2.0. We don't see any conflict for either consumption (v1) or contribution (v3), but please flag if you do.
|
||||
- CLA: do you require one for upstream contributions? Not blocking v1 — just want to know for v3.
|
||||
- Preferred coordination channel: this Discussion thread, a Discord, an email? We're happy to use whatever works for you.
|
||||
|
||||
Thanks for building this — Proof-of-Useful-Work via matmul is genuinely interesting and we're excited to bring more (slower!) hardware to the network.
|
||||
|
||||
— [user name], on behalf of OpenJarvis
|
||||
|
||||
---
|
||||
|
||||
## Notes for the user before posting
|
||||
|
||||
- Replace `[user fills in]` with your contact info, `[user name]` with your name.
|
||||
- The architecture/perf claims are all backed by code + an actual local build; you can stand behind them.
|
||||
- "Heads-up" framing on the `Taskfile.yml` is intentional — we're not asking them to *change* it, just flagging the friction point in case they want to.
|
||||
- Don't post until OJ Spec A is at least branch-pushed (which it is, PR #310) — it gives Pearl a way to see the broader integration we're building.
|
||||
- When their reply lands, update Spec B §10 (open questions 1, 5, 7) and §11 (cross-references → coordination thread URL).
|
||||
|
||||
## Possible Pearl responses to anticipate
|
||||
|
||||
- **Best case:** "Confirmed, looks great, we don't have an Apple Silicon plan, please do it." — proceed with §13.
|
||||
- **Middle case:** "Confirmed, but we have a Metal port in flight." — coordinate, share Spec B §6.1, decide upstream-vs-fork. v1 (CPU) is unaffected.
|
||||
- **Worst case:** "We'd prefer downstream non-CUDA mining stay disabled for now." — unlikely given their `pearl-gateway` README explicitly anticipates "plugins for other LLM inference libraries", but if it happens, this becomes a much harder problem and we'd need to revisit.
|
||||
@@ -0,0 +1,560 @@
|
||||
# Spec A — vLLM-Pearl mining integration (v1)
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| **Date** | 2026-05-05 |
|
||||
| **Status** | Design — pending implementation plan |
|
||||
| **Owner** | OpenJarvis team |
|
||||
| **Companion spec** | [Spec B — Apple Silicon enablement](2026-05-05-apple-silicon-pearl-mining-design.md) (separate effort, runs in parallel) |
|
||||
| **Repos referenced** | `OpenJarvis` (this repo), `pearl-research-labs/pearl` |
|
||||
|
||||
## 1. Summary
|
||||
|
||||
Add a new sibling subsystem `openjarvis.mining` that lets users run [Pearl](https://github.com/pearl-research-labs/pearl) Proof-of-Useful-Work mining as a property of their local LLM inference. v1 ships solo mining for users who already have an H100/H200 and run vLLM — the only configuration Pearl's reference miner currently supports. The architecture leaves three deliberate seams for v2 (pool support + a 20% OJ fee) and is engine-agnostic by construction so Apple Silicon, AMD, Ollama, llama.cpp, and MLX paths plug in via the registry without a rewrite when Pearl ships the matching plugins.
|
||||
|
||||
The narrative thesis: Pearl's `vllm-miner` is a vLLM plugin that swaps quantized linear ops with `NoisyGEMM`, a CUDA kernel that produces both the correct matmul output *and* a PoW commitment. Mining IS inference. For an OJ user already serving prompts on a powerful local GPU, this is a way to capture economic value from compute they were going to do anyway — directly aligned with OJ's Intelligence-Per-Watt thesis rather than against it.
|
||||
|
||||
## 2. Scope
|
||||
|
||||
### In scope (v1)
|
||||
- New `openjarvis.mining` subsystem with `MiningProvider` ABC, `MinerRegistry`, `MiningCapabilities` / `MiningConfig` / `MiningStats` dataclasses
|
||||
- `vllm-pearl` provider implementation: orchestrate Pearl's published `vllm-miner` Docker container
|
||||
- `[mining]` TOML section in OJ config; `MiningConfig` field in `JarvisConfig`
|
||||
- New CLI namespace: `jarvis mine init|start|stop|status|doctor|attach|logs`
|
||||
- Runtime sidecar at `~/.openjarvis/runtime/mining.json` for engine ↔ mining handoff
|
||||
- Hybrid Docker image acquisition: pull-if-published, otherwise build from a pinned Pearl ref
|
||||
- On-demand telemetry via Pearl gateway `:8339/metrics`; `mining_session_id` nullable column on telemetry inference rows
|
||||
- v2 seams: `submit_target` tagged-union parsing, zero-valued `fee_bps` / `fees_owed` plumbing, reserved `mining/pools/` location
|
||||
- Test strategy that doesn't require an H100 in CI
|
||||
- Documentation: `docs/user-guide/mining.md`, `docs/development/mining.md`, `CLAUDE.md` paragraph, `REVIEW.md` bullet
|
||||
|
||||
### Out of scope (v1) — deferred or owned elsewhere
|
||||
- Pool support and the 20% OJ fee mechanism (own spec, v2)
|
||||
- Custody, signing, or routing Pearl funds (anti-goal — must remain zero in v1)
|
||||
- Apple Silicon, AMD ROCm, sm89 (RTX 4090) NVIDIA, CPU, MLX, Ollama, llama.cpp, SGLang mining paths (Spec B for Apple; remaining hardware/engine paths blocked on Pearl)
|
||||
- Wallet generation, Oyster integration, key custody (paste-only address)
|
||||
- pearld lifecycle management (BYO node)
|
||||
- Background telemetry collection in OJ's gateway daemon (v1.x; the hook point is reserved)
|
||||
- Inference-quality drift detection (v1.x at earliest)
|
||||
- `mine doctor --fix` automatic remediation (v1.x stub)
|
||||
- Multi-GPU / multi-worker / multi-session per host (v2+)
|
||||
|
||||
## 3. Load-bearing decisions from brainstorming
|
||||
|
||||
These were the forks where the design could have gone several ways. Recorded so future-readers can audit reasoning rather than re-derive it.
|
||||
|
||||
| Decision | What we picked | Why |
|
||||
|---|---|---|
|
||||
| Target audience for v1 | H100/H200 owners running vLLM (Pearl's only working config today) | Anything broader is blocked on Pearl shipping non-CUDA / non-vLLM plugins. Power-user MVP ships in weeks; pool/fee/Apple are separate specs. |
|
||||
| Mining model | Co-located: every inference through the Pearl-flavored vLLM is mining work | Matches Pearl's `vllm-miner` plugin design and OJ's Intelligence-Per-Watt thesis. Side-car deferred until Pearl ships plugins for engines users care about for non-mining inference. |
|
||||
| Coupling to Pearl miner process | Wrap-and-launch via Docker | Pearl's Docker image (or Dockerfile) is the most stable contract they expose. (1) "BYO miner" is too thin to be a feature; (3) running Pearl's `uv` workspace natively couples us to their build system. |
|
||||
| Module placement | Sibling top-level subsystem `mining/` (peer to `engine/`, `agents/`) | Matches OJ's existing module pattern. `MinerRegistry` is a peer registry. Future non-vLLM providers slot in identically. |
|
||||
| Engine attachment | Runtime sidecar JSON at `~/.openjarvis/runtime/mining.json` | Existing vLLM engine class stays untouched. Sidecar is the single source of truth tying mining lifecycle to engine resolution. Inspectable via `cat`. |
|
||||
| Config shape | Flat top-level `[mining]` TOML section | Only one provider in v1; nested per-engine config can grow later if multi-provider becomes real. |
|
||||
| Wallet handling | Paste-only Pearl Taproot address | Keys are sensitive; Pearl's wallet RPC is unstable surface. v1.x can add Oyster integration once the contract stabilizes. |
|
||||
| pearld | BYO; user points OJ at their own node | OJ doesn't orchestrate L1 nodes. Doctor surfaces unreachable cleanly. |
|
||||
| Telemetry collection | On-demand reads in v1; persistent collector class shipped unwired (`MiningTelemetryCollector`) | Most users won't enable mining; daemon shouldn't grow surface for them. v1.x lights up the hook with zero API churn. |
|
||||
| v1 fee/pool seams | Three seams: `submit_target` parsed (one variant works), `fee_bps`/`fees_owed` plumbed at zero, `mining/pools/` reserved | Cheap to leave; painful to retrofit. Does not pre-decide the v2 API. |
|
||||
| Custody | **Anti-goal**: zero. v1 must not accept, sign, or route Pearl funds. | Avoids prematurely binding a legal/regulatory posture. v2 revisits as part of pool design. |
|
||||
| Apple Silicon support | Not in v1. Designed-for via the `MiningProvider` ABC + `MiningCapabilities.detect()`. Spec B documents the enablement work. | The Pearl `pearl-gemm` kernel is heavily Hopper-bound (`sm_90a`, WGMMA, TMA, cluster mode, CUTLASS 3.x). A Metal port is real GPU-kernel engineering, not a config flag. |
|
||||
|
||||
## 4. Architecture & module layout
|
||||
|
||||
### 4.1 New module tree
|
||||
|
||||
```
|
||||
src/openjarvis/mining/
|
||||
__init__.py # soft-imports providers (try/except ImportError)
|
||||
_stubs.py # MiningProvider ABC + dataclasses (MiningCapabilities, MiningConfig, MiningStats, SoloTarget, PoolTarget)
|
||||
_discovery.py # detect_providers(hardware, engine, model) -> list[MiningCapabilities]
|
||||
_docker.py # PearlDockerLauncher — shared Docker orchestration (image acquisition + container lifecycle)
|
||||
_collector.py # MiningTelemetryCollector class — defined but UNWIRED in v1; lit up in v1.x
|
||||
_constants.py # PEARL_REPO, PEARL_PINNED_REF, PEARL_IMAGE_TAG, OJ default tag
|
||||
vllm_pearl.py # @MinerRegistry.register("vllm-pearl") — only impl in v1
|
||||
pools/ # RESERVED for v2. Empty in v1 except for an __init__.py with a docstring saying so.
|
||||
|
||||
src/openjarvis/cli/
|
||||
mine_cmd.py # jarvis mine init|start|stop|status|doctor|attach|logs
|
||||
|
||||
tests/mining/
|
||||
__init__.py
|
||||
conftest.py # mining-specific fixtures (synthetic HardwareInfo, sample Prometheus output)
|
||||
fixtures/
|
||||
gateway_metrics_sample.txt # captured Prometheus output from a real Pearl run
|
||||
config_*.toml # golden TOML files
|
||||
test_stubs.py
|
||||
test_discovery.py
|
||||
test_docker.py
|
||||
test_collector.py
|
||||
test_vllm_pearl.py
|
||||
test_cli.py
|
||||
```
|
||||
|
||||
### 4.2 Registry additions
|
||||
|
||||
`MinerRegistry` added to `src/openjarvis/core/registry.py` as a peer to `EngineRegistry`, `AgentRegistry`, etc. `tests/conftest.py`'s autouse `_clean_registries` fixture is updated to include `MinerRegistry.clear()`.
|
||||
|
||||
`mining/vllm_pearl.py` exposes idempotent `ensure_registered()`:
|
||||
|
||||
```python
|
||||
def ensure_registered() -> None:
|
||||
if not MinerRegistry.contains("vllm-pearl"):
|
||||
MinerRegistry.register_value("vllm-pearl", VllmPearlProvider)
|
||||
```
|
||||
|
||||
`mining/__init__.py` soft-imports `vllm_pearl` inside `try / except ImportError` and calls `ensure_registered()`. Standard OJ pattern.
|
||||
|
||||
### 4.3 Optional-deps extras
|
||||
|
||||
```toml
|
||||
mining-pearl = ["docker>=7.0"] # v1 requires only the Docker SDK
|
||||
# mining-pearl-mlx = [...] # future, owned by Spec B
|
||||
# mining-pearl-rocm = [...] # future
|
||||
```
|
||||
|
||||
### 4.4 The central ABC
|
||||
|
||||
```python
|
||||
# src/openjarvis/mining/_stubs.py
|
||||
from abc import ABC, abstractmethod
|
||||
from dataclasses import dataclass, field
|
||||
from openjarvis.core.config import HardwareInfo
|
||||
|
||||
@dataclass(slots=True)
|
||||
class MiningCapabilities:
|
||||
supported: bool
|
||||
reason: str | None = None # human-readable: "needs sm90", "no Pearl plugin for engine ollama"
|
||||
estimated_hashrate: float | None = None
|
||||
|
||||
@dataclass(slots=True)
|
||||
class SoloTarget:
|
||||
pearld_rpc_url: str
|
||||
|
||||
@dataclass(slots=True)
|
||||
class PoolTarget:
|
||||
url: str
|
||||
worker_id: str | None = None
|
||||
|
||||
SubmitTarget = SoloTarget | PoolTarget
|
||||
|
||||
@dataclass(slots=True)
|
||||
class MiningConfig:
|
||||
provider: str # MinerRegistry key
|
||||
wallet_address: str
|
||||
submit_target: SubmitTarget # parsed from TOML "solo" / "pool:<url>"; v1 accepts only SoloTarget at runtime
|
||||
fee_bps: int = 0 # v1: 0; v2: 2000 (=20%)
|
||||
fee_payout_address: str | None = None # v1: ignored; v2: OJ's address
|
||||
extra: dict = field(default_factory=dict)
|
||||
|
||||
@dataclass(slots=True)
|
||||
class MiningStats:
|
||||
provider_id: str
|
||||
shares_submitted: int = 0
|
||||
shares_accepted: int = 0
|
||||
blocks_found: int = 0
|
||||
hashrate: float = 0.0
|
||||
uptime_seconds: float = 0.0
|
||||
last_share_at: float | None = None
|
||||
last_error: str | None = None
|
||||
payout_target: str = "solo" # v2 reporting; "solo" in v1
|
||||
fees_owed: int = 0 # v2 accounting hook; 0 in v1
|
||||
|
||||
class MiningProvider(ABC):
|
||||
provider_id: str
|
||||
|
||||
@classmethod
|
||||
@abstractmethod
|
||||
def detect(cls, hw: HardwareInfo, engine_id: str, model: str) -> MiningCapabilities: ...
|
||||
|
||||
@abstractmethod
|
||||
async def start(self, config: MiningConfig) -> None: ...
|
||||
@abstractmethod
|
||||
async def stop(self) -> None: ...
|
||||
@abstractmethod
|
||||
def is_running(self) -> bool: ...
|
||||
@abstractmethod
|
||||
def stats(self) -> MiningStats: ...
|
||||
```
|
||||
|
||||
## 5. Config schema & engine attachment
|
||||
|
||||
### 5.1 TOML schema
|
||||
|
||||
```toml
|
||||
[mining]
|
||||
provider = "vllm-pearl" # MinerRegistry key
|
||||
wallet_address = "prl1q..." # user's Pearl Taproot address (paste-only)
|
||||
submit_target = "solo" # v1: "solo" only; "pool:<url>" raises NotImplementedError at start()
|
||||
fee_bps = 0 # v1: 0; v2: 2000
|
||||
fee_payout_address = "" # v1: ignored; v2: OJ's address
|
||||
|
||||
[mining.extra]
|
||||
docker_image_tag = "openjarvis/pearl-miner:<pinned-ref>"
|
||||
model = "pearl-ai/Llama-3.3-70B-Instruct-pearl"
|
||||
gateway_port = 8337
|
||||
gateway_metrics_port = 8339
|
||||
vllm_port = 8000
|
||||
gpu_memory_utilization = 0.9
|
||||
max_model_len = 8192
|
||||
pearld_rpc_url = "http://localhost:44107"
|
||||
pearld_rpc_user = "rpcuser"
|
||||
pearld_rpc_password_env = "PEARLD_RPC_PASSWORD" # name of env var, not the secret
|
||||
hf_token_env = "HF_TOKEN" # name of env var
|
||||
```
|
||||
|
||||
Secrets: env-var *names*, never literal values. Matches OJ's existing convention for cloud API keys.
|
||||
|
||||
### 5.2 JarvisConfig field
|
||||
|
||||
`core/config.py` adds:
|
||||
|
||||
```python
|
||||
@dataclass(slots=True)
|
||||
class JarvisConfig:
|
||||
...
|
||||
mining: MiningConfig | None = None
|
||||
```
|
||||
|
||||
The TOML loader reads `[mining]`, parses `submit_target` into `SoloTarget | PoolTarget`, validates against the dataclass, surfaces unknown `extra` keys as warnings. Absent section → `mining = None` → zero behavior change.
|
||||
|
||||
### 5.3 Runtime sidecar
|
||||
|
||||
`~/.openjarvis/runtime/mining.json` (created on `mine start`, removed on `mine stop`):
|
||||
|
||||
```json
|
||||
{
|
||||
"provider": "vllm-pearl",
|
||||
"vllm_endpoint": "http://127.0.0.1:8000/v1",
|
||||
"model": "pearl-ai/Llama-3.3-70B-Instruct-pearl",
|
||||
"gateway_url": "http://127.0.0.1:8337",
|
||||
"gateway_metrics_url": "http://127.0.0.1:8339",
|
||||
"container_id": "abc123...",
|
||||
"wallet_address": "prl1q...",
|
||||
"started_at": 1714867200
|
||||
}
|
||||
```
|
||||
|
||||
Sidecar deliberately omits all secrets and process IDs. `container_id` is the authoritative handle (Docker is the source of truth for liveness); `wallet_address` is captured for drift-detection (config-vs-runtime).
|
||||
|
||||
### 5.4 Engine handoff flow
|
||||
|
||||
1. `jarvis mine start` → `MinerRegistry.get("vllm-pearl").start(config)`.
|
||||
2. `VllmPearlProvider.start()` calls `_docker.PearlDockerLauncher.start(config)` and writes the sidecar.
|
||||
3. `engine/_discovery.py` checks for `mining.json` on every engine lookup. When present, it auto-registers a `vllm` engine instance pointing at `vllm_endpoint`, named `vllm-pearl-mining`, marked default for mining-aware operations.
|
||||
4. `jarvis ask` and the SDK route to that endpoint transparently. The user's normal inference is the mining work.
|
||||
|
||||
The vLLM engine class itself (`engine/openai_compat_engines.py`) is **not modified**. The change to `engine/_discovery.py` is small and additive: it inspects for `mining.json` and registers a derived `vllm` instance pointing at the mining endpoint when the sidecar is present. Absent sidecar → unchanged discovery behavior.
|
||||
|
||||
### 5.5 Manual mode
|
||||
|
||||
Power users running their own Pearl container skip `jarvis mine start` and write the sidecar themselves via `jarvis mine attach --vllm-endpoint=... --gateway-url=...`. Decouples lifecycle from wiring.
|
||||
|
||||
## 6. CLI surface, lifecycle & daemon integration
|
||||
|
||||
### 6.1 Subcommands
|
||||
|
||||
| Command | Purpose |
|
||||
|---|---|
|
||||
| `jarvis mine init` | Interactive: hardware/Docker checks, prompt for wallet + pearld credentials, write `[mining]`, pull/build image. Does NOT start mining. Pre-checks `>=200 GB` free disk. |
|
||||
| `jarvis mine start` | Launch container via the registered provider, write sidecar, print endpoint info. Idempotent if running. |
|
||||
| `jarvis mine stop` | Stop container, remove sidecar. Idempotent if not running. |
|
||||
| `jarvis mine status` | Read sidecar + query gateway `:8339/metrics`. Print `MiningStats`. |
|
||||
| `jarvis mine doctor` | Capability matrix; every check ✓/✗ with reason. Works in any state. |
|
||||
| `jarvis mine attach` | Manual mode: write sidecar without launching. |
|
||||
| `jarvis mine logs [-f]` | Tail container logs through Docker SDK. |
|
||||
|
||||
### 6.2 Doctor output (canonical example)
|
||||
|
||||
```
|
||||
$ jarvis mine doctor
|
||||
Hardware
|
||||
GPU vendor nvidia ✓
|
||||
Compute capability sm_90a ✓
|
||||
VRAM 80 GB ✓ (need ≥ 70 GB for Pearl 70B)
|
||||
Docker
|
||||
Daemon running 24.0.7 ✓
|
||||
GPU runtime nvidia-container-toolkit ✓
|
||||
Disk
|
||||
Free in HF cache 312 GB ✓ (need ≥ 200 GB)
|
||||
Image
|
||||
openjarvis/pearl-miner:<ref> present (built 2026-04-30) ✓
|
||||
Pearl node
|
||||
RPC http://localhost:44107 ✓
|
||||
Auth ok ✓
|
||||
Block height 442107 (synced) ✓
|
||||
Wallet
|
||||
Address format prl1q... ✓
|
||||
Provider capability
|
||||
vllm-pearl SUPPORTED
|
||||
Session
|
||||
Sidecar absent (not running)
|
||||
Container —
|
||||
```
|
||||
|
||||
Each row maps to one check function in `mining/_discovery.py`. Failures print actionable reasons (e.g. `✗ reason: needs sm90, you have sm89 (RTX 4090)`).
|
||||
|
||||
### 6.3 Lifecycle states
|
||||
|
||||
```
|
||||
NOT_CONFIGURED → CONFIGURED → STARTING → RUNNING ⇄ STOPPING → STOPPED
|
||||
↘
|
||||
FAILED
|
||||
```
|
||||
|
||||
State derivation rules (no separate state file — derived from config + sidecar + container introspection):
|
||||
|
||||
- `NOT_CONFIGURED` — no `[mining]` in config
|
||||
- `CONFIGURED` — config present, no sidecar
|
||||
- `STARTING` — sidecar with `started_at` < ~30 s ago, container exists but gateway not yet healthy
|
||||
- `RUNNING` — sidecar present, container running, gateway responding
|
||||
- `FAILED` — sidecar present, but container exited or gateway failing > threshold
|
||||
- `STOPPING` — `mine stop` invoked, Docker stop in progress
|
||||
- `STOPPED` — `mine stop` complete, sidecar removed
|
||||
|
||||
### 6.4 Daemon integration: deliberately minimal in v1
|
||||
|
||||
- Docker handles container restart via `--restart=unless-stopped`. OJ does not babysit.
|
||||
- Existing `com.openjarvis.gateway` daemon is unchanged.
|
||||
- v1.x hook: `MiningTelemetryCollector` (already shipped in v1, unwired) can be added to the gateway as a 30-second-tick async task.
|
||||
- launchd/systemd installation surface (`jarvis daemon install`) untouched.
|
||||
|
||||
### 6.5 Concurrency
|
||||
|
||||
POSIX `flock` on `~/.openjarvis/runtime/mining.lock` prevents racing `mine start` invocations.
|
||||
|
||||
### 6.6 `jarvis ask` UX hint
|
||||
|
||||
When `[mining]` is configured but no sidecar exists, `cli/hints.py` emits one line: `"mining configured but not running — start it with \`jarvis mine start\`"`. One-line UX nudge, no new infrastructure.
|
||||
|
||||
## 7. Pearl Docker integration
|
||||
|
||||
### 7.1 Realities from inspecting Pearl's repo
|
||||
|
||||
- **Build context = entire Pearl monorepo.** Dockerfile copies root `pyproject.toml`/`uv.lock`, `miner/`, `pearl-blake3/`, `py-pearl-mining/`, `zk-pow/`, `plonky2/`. Building requires the full repo.
|
||||
- **Pearl publishes no registry image as of writing.** README documents only `docker buildx build -t vllm_miner . -f miner/vllm-miner/Dockerfile`.
|
||||
- **Single container, three ports.** `entrypoint.sh` launches `pearl-gateway` in the background, waits on `:8339/metrics`, then `exec`s `vllm serve`. Ports: `8000` (vLLM), `8337` (miner RPC), `8339` (gateway metrics).
|
||||
- **Pinned stack inside the image.** CUDA 12.9.1, vLLM 0.20.0+cu129, Python 3.12, `compute_90a/sm_90a`. Set by Pearl, not by us.
|
||||
- **First-launch cost.** vLLM pulls the 70 B model from HF on first serve (~140 GB). Build itself is 30–60 min on first init.
|
||||
|
||||
### 7.2 Hybrid image acquisition
|
||||
|
||||
| Mode | Behavior | When |
|
||||
|---|---|---|
|
||||
| **Pre-built pull** | OJ `docker pull`s the configured tag if it resolves in a registry | Default once Pearl publishes; users with private registry; CI |
|
||||
| **Build-from-pin** | OJ git-clones Pearl at a pinned ref into `~/.openjarvis/cache/pearl/`, then `docker buildx build` | v1 default (Pearl publishes nothing today) |
|
||||
| **BYO image** | User sets `mining.extra.docker_image_tag` to an image they built/pulled themselves | Power users, air-gapped envs |
|
||||
|
||||
Selection logic in `_docker.PearlDockerLauncher.ensure_image()`:
|
||||
1. If `docker_image_tag` resolves locally → use it.
|
||||
2. Else `docker pull <tag>` → on success, use it.
|
||||
3. Else if `tag == OJ_DEFAULT_TAG`, fall back to clone-and-build from `PEARL_PINNED_REF`.
|
||||
4. Else fail with a clear error pointing at `mine doctor`.
|
||||
|
||||
### 7.3 Pearl version pinning
|
||||
|
||||
`mining/_constants.py`:
|
||||
|
||||
```python
|
||||
PEARL_REPO = "https://github.com/pearl-research-labs/pearl.git"
|
||||
PEARL_PINNED_REF = "<sha-or-tag>" # bumped per OJ release after rev-testing
|
||||
PEARL_IMAGE_TAG = f"openjarvis/pearl-miner:{PEARL_PINNED_REF}"
|
||||
```
|
||||
|
||||
OJ release notes call out the Pearl ref shipped. Bumping the ref is its own PR with a documented Pearl-rev workflow.
|
||||
|
||||
### 7.4 Container launch shape
|
||||
|
||||
Via `docker>=7.0` SDK in `_docker.PearlDockerLauncher.start()`:
|
||||
|
||||
```python
|
||||
container = client.containers.run(
|
||||
image=PEARL_IMAGE_TAG,
|
||||
command=[
|
||||
config.extra["model"],
|
||||
"--host", "0.0.0.0",
|
||||
"--port", str(config.extra["vllm_port"]),
|
||||
"--gpu-memory-utilization", str(config.extra["gpu_memory_utilization"]),
|
||||
"--enforce-eager",
|
||||
"--max-model-len", str(config.extra.get("max_model_len", 8192)),
|
||||
],
|
||||
name="openjarvis-pearl-miner",
|
||||
detach=True,
|
||||
auto_remove=False,
|
||||
restart_policy={"Name": "unless-stopped"},
|
||||
device_requests=[ DeviceRequest(count=-1, capabilities=[["gpu"]]) ],
|
||||
shm_size="8g",
|
||||
network_mode="host",
|
||||
volumes={
|
||||
str(Path.home() / ".cache/huggingface"): {
|
||||
"bind": "/root/.cache/huggingface",
|
||||
"mode": "rw",
|
||||
},
|
||||
},
|
||||
environment={
|
||||
"PEARLD_RPC_URL": config.extra["pearld_rpc_url"],
|
||||
"PEARLD_RPC_USER": config.extra["pearld_rpc_user"],
|
||||
"PEARLD_RPC_PASSWORD": os.environ[config.extra["pearld_rpc_password_env"]],
|
||||
"PEARLD_MINING_ADDRESS": config.wallet_address,
|
||||
"HF_TOKEN": os.environ.get(config.extra.get("hf_token_env", "HF_TOKEN"), ""),
|
||||
"MINER_RPC_TRANSPORT": "tcp",
|
||||
},
|
||||
)
|
||||
```
|
||||
|
||||
### 7.5 Trade-offs called out
|
||||
|
||||
- **`network_mode="host"`** because pearld's RPC at `http://localhost:44107` lives on the host. A user-defined Docker network adds setup steps with no real isolation benefit on a single-tenant miner box. Pragmatism > purity. Note: host networking has Linux semantics; macOS/Windows Docker handle it differently. Acceptable for v1 since H100/H200 + nvidia-container-toolkit constrains the deployment to Linux anyway.
|
||||
- **`auto_remove=False`** so a crashed container stays around for `jarvis mine logs` post-mortem.
|
||||
- **HF cache mounted from host.** 140 GB weight download is one-time, survives container restarts, visible to other tools.
|
||||
- **Secrets via env-var names**, never persisted in the container image, the sidecar, or Docker labels.
|
||||
|
||||
### 7.6 Wallet handling boundary
|
||||
|
||||
OJ never sees Pearl mnemonic seeds, never imports Oyster keys, never signs Pearl transactions. Only Pearl-secret OJ touches is the pearld RPC password (passed through container env, sourced by name from host env). Mining address is public — fine in plaintext config.
|
||||
|
||||
### 7.7 Image lifecycle UX
|
||||
|
||||
- `jarvis mine init` triggers `ensure_image()`, streams build/pull output through the CLI with a clear time estimate (`"Building Pearl miner image — first run takes ~45 min on a fast machine"`).
|
||||
- `jarvis mine doctor` reports `image: present (tag, age, sha)` or `image: missing (run mine init)`.
|
||||
- `jarvis mine prune` (v1.x) cleans old `openjarvis/pearl-miner:*` tags. Manual `docker image rm` works in v1.
|
||||
|
||||
## 8. Telemetry hooks & v2 fee/pool seams
|
||||
|
||||
### 8.1 Telemetry — read surface
|
||||
|
||||
Pearl's container exposes `:8339/metrics` (Prometheus exposition format). v1 reads only this endpoint. Deeper RPC introspection via `:8337` deferred to v2.
|
||||
|
||||
### 8.2 Adapter and metric mapping
|
||||
|
||||
`mining/vllm_pearl.py::_parse_gateway_metrics()` translates Prometheus lines to `MiningStats`. Metric names are TBD on implementation — verified against captured fixture `tests/mining/fixtures/gateway_metrics_sample.txt`. Expected mapping (fallback: zero-fill any missing field, log a one-shot warning):
|
||||
|
||||
| `MiningStats` field | Likely Pearl metric (verify on implementation) |
|
||||
|---|---|
|
||||
| `shares_submitted` | `pearl_gateway_shares_submitted_total` |
|
||||
| `shares_accepted` | `pearl_gateway_shares_accepted_total` |
|
||||
| `blocks_found` | `pearl_gateway_blocks_found_total` |
|
||||
| `hashrate` | derived rate of `shares_submitted_total` |
|
||||
| `uptime_seconds` | `process_start_time_seconds` |
|
||||
| `last_share_at` | `pearl_gateway_last_share_timestamp` |
|
||||
| `last_error` | derived from `pearl_gateway_errors_total` deltas |
|
||||
|
||||
### 8.3 Collection cadence
|
||||
|
||||
- **v1: on-demand only.** `jarvis mine status` makes one HTTP GET per call (~10 ms). No background polling.
|
||||
- **v1.x: `MiningTelemetryCollector` lit up in the gateway daemon.** The class is shipped in v1 but unwired. v1.x adds a periodic asyncio task; same `MiningStats` schema, same gateway endpoint. Zero API churn.
|
||||
|
||||
### 8.4 Intelligence-Per-Watt extension
|
||||
|
||||
The `telemetry/store.py` schema gains a nullable `mining_session_id` column on inference rows:
|
||||
|
||||
- Tagged when an inference goes through the Pearl-mining endpoint; null otherwise.
|
||||
- Untagged rows behave exactly as today — zero impact on the non-mining path.
|
||||
- `jarvis telemetry stats --mining` (v1.x) joins to the latest `MiningStats` snapshot and reports `tokens / share`, `joules / share`, `est. PRL / kWh`.
|
||||
|
||||
v1 ships the column and the no-op join path. v1.x lights up the reporting. This is the metric the IPW thesis genuinely cares about.
|
||||
|
||||
### 8.5 v2 fee/pool seams (three concrete, no more)
|
||||
|
||||
**1. `submit_target` parsed into a tagged union; only one variant works.** `SoloTarget` accepted at runtime in v1; `PoolTarget` raises `NotImplementedError("pool support is v2 — track openjarvis#XYZ")`. Reachable only by users who edit their config to opt in.
|
||||
|
||||
**2. `fee_bps` / `fee_payout_address` plumbed; zero-valued in v1.** `MiningStats.fees_owed = 0` and `MiningStats.payout_target = "solo"` always in v1. Schema is real; values are zero. No migration in v2.
|
||||
|
||||
**3. `mining/pools/` module location reserved.** Empty in v1 except for an `__init__.py` whose docstring says the location is reserved for v2 `PoolClient` work. The v1 spec **does not** define a `PoolClient` ABC — predicting the v2 API precisely creates migration debt. The v2 spec writes against an empty slot.
|
||||
|
||||
### 8.6 What v1 deliberately does not lock in
|
||||
|
||||
- Pool protocol (PPLNS / PPS / SOLO+ / custom)
|
||||
- Custody model (escrow / trustless split-coinbase / settlement contract)
|
||||
- OJ pool URL, share format, share difficulty
|
||||
- KYC / TOS / payout thresholds
|
||||
|
||||
### 8.7 Custody anti-goal
|
||||
|
||||
v1 must not introduce any code path where OJ accepts custody of, signs, or routes Pearl funds. Closest v1 comes is reading `wallet_address` (public) and passing it through to the container. v2 revisits.
|
||||
|
||||
### 8.8 Single-session assumption (called out, not seamed)
|
||||
|
||||
v1 assumes one mining session per host (one sidecar). Multi-GPU / multi-worker fanout is v2+. Sidecar would become a list or directory.
|
||||
|
||||
## 9. Failure handling & test strategy
|
||||
|
||||
### 9.1 Principles
|
||||
|
||||
1. **Fail loud, don't auto-heal.** Docker handles container restarts; `mine doctor` surfaces what's wrong. OJ does not retry mining work, restart pearld, or paper over crashes.
|
||||
2. **`mine doctor` is the canonical failure surface.** Every failure mode below maps to one or more rows in doctor output.
|
||||
3. **Sidecar is authoritative; config is intent.** Drift surfaces as a warning, not a crash.
|
||||
|
||||
### 9.2 Failure mode matrix
|
||||
|
||||
| Failure | v1 behavior | Surface |
|
||||
|---|---|---|
|
||||
| Image missing | `mine start` errors with "run `mine init` to build/pull" | `mine doctor: image: missing` |
|
||||
| GPU not reachable in container | Docker error with `nvidia-container-toolkit` hint | `mine doctor: docker.gpu_runtime: ✗` |
|
||||
| Disk too low | `mine init` pre-checks `shutil.disk_usage`; errors if < 200 GB free | `mine doctor: disk_free: ✗` |
|
||||
| vLLM model load fails (HF auth, OOM, model not found) | Container exits; `mine status` reports `FAILED` with `last_error` from `docker logs` tail | `mine status` + `mine logs` |
|
||||
| pearl-gateway can't reach pearld | `:8339/metrics` exposes the error; adapter populates `MiningStats.last_error` | `mine status: last_error` |
|
||||
| Container crashes mid-run | Docker `--restart=unless-stopped` restarts; `mine status` shows brief `STARTING` → `RUNNING` | self-healing, logged |
|
||||
| Stale sidecar (container died, sidecar not cleaned) | `mine start` validates `container_id`; if Docker says it's gone, removes sidecar and proceeds | one-line warning |
|
||||
| Concurrent `mine start` | POSIX `flock` on `~/.openjarvis/runtime/mining.lock`; second invocation errors clearly | clear message |
|
||||
| Already-running `mine start` | Idempotent: detect via sidecar + container introspection, print status, exit 0 | informational |
|
||||
| Wallet/config drift | Sidecar carries wallet from start time; `mine status` cross-checks and warns on mismatch | warning, not auto-restart |
|
||||
| User edits `submit_target = "pool:..."` in v1 | `start()` raises `NotImplementedError` with tracking issue link | clear error |
|
||||
| Pearl protocol upgrade (block format / metric names change) | Adapter zero-fills with one-shot warning. `mine doctor` does a best-effort check: it reads the `image: openjarvis/pearl-miner:<ref>` Docker label and compares against `PEARL_PINNED_REF` baked into the OJ release; mismatch surfaces a warning. **OJ does not poll Pearl's GitHub at runtime.** | warning + spec'd Pearl-rev workflow |
|
||||
| Inference quality regression from NoisyGEMM | **Out of v1 scope to detect.** Documented risk; v1.x may add automated drift detection. | docs only |
|
||||
|
||||
### 9.3 Test strategy
|
||||
|
||||
Hard constraint: **OJ's CI has no H100, no GPU, no Pearl image, no pearld.** Almost everything must be testable without those.
|
||||
|
||||
| Layer | Pattern | Marker | Runs in CI? |
|
||||
|---|---|---|---|
|
||||
| `MiningCapabilities.detect()` matrix | Pure unit, parametrized over synthetic `HardwareInfo` | unmarked | yes |
|
||||
| `MiningConfig` parsing (TOML → dataclass, including `submit_target` tagged-union) | Unit, golden TOML fixtures | unmarked | yes |
|
||||
| Docker launch shape | `unittest.mock.patch("docker.from_env")`; assert `containers.run(...)` kwargs | unmarked | yes |
|
||||
| Gateway metrics adapter | `tests/mining/fixtures/gateway_metrics_sample.txt`, parse + assert `MiningStats` | unmarked | yes |
|
||||
| Sidecar lifecycle (write/read/stale-cleanup, `flock` acquisition) | `tmp_path`, real filesystem, real `flock` | unmarked | yes |
|
||||
| CLI smoke | Click `CliRunner`, mocked `MiningProvider` | unmarked | yes |
|
||||
| Container start/stop with real Docker daemon | Real Docker, swap Pearl image for tiny stub `alpine`-based image opening the right ports | new `docker` marker | optional in CI |
|
||||
| End-to-end mining (real container, real pearld, real shares) | Real H100 + pearld testnet + pinned Pearl image | `live and nvidia and slow` | **no** — manual pre-release smoke |
|
||||
|
||||
**New pytest marker.** `docker` registered alongside `live`, `cloud`, `nvidia`, etc. in `pyproject.toml`. CI matrix optionally runs `-m "docker and not live"` on a Docker-enabled runner.
|
||||
|
||||
**Conftest hygiene.** `tests/conftest.py`'s autouse fixture clears `MinerRegistry`. `mining/__init__.py`'s `ensure_registered()` survives the autouse clear via `MinerRegistry.contains(...)`.
|
||||
|
||||
**Captured Prometheus fixture.** Real metrics output from a Pearl gateway run, committed to the repo. Pins the metric-name assumptions and is the canary if Pearl renames metrics.
|
||||
|
||||
### 9.4 What v1 deliberately does not test
|
||||
|
||||
- Mining throughput/economics on a real H100 (Pearl's CI tests their kernels)
|
||||
- Inference quality drift from NoisyGEMM (out of v1 scope)
|
||||
- Pool share submission paths (v2 spec)
|
||||
- Apple Silicon paths (Spec B)
|
||||
|
||||
## 10. Documentation deliverables (part of this spec)
|
||||
|
||||
- `docs/user-guide/mining.md` — user-facing: prerequisites, init flow, doctor output reading guide, `mine status` interpretation, deliberately-unsupported list (Mac, AMD, sm89, non-vLLM engines)
|
||||
- `docs/development/mining.md` — for contributors: `MiningProvider` ABC, registry pattern, how to add a new provider (Spec B is the canonical worked example)
|
||||
- One paragraph in `CLAUDE.md` under "Architecture" pointing future-Claude at `mining/` as a sibling subsystem with its own optional-deps discipline
|
||||
- `REVIEW.md` gets a new bullet under registry compliance specifically calling out `MinerRegistry`
|
||||
|
||||
## 11. Open items to resolve at implementation time
|
||||
|
||||
1. **Pearl gateway metric names.** Verify the actual exposition labels by capturing `:8339/metrics` from a running Pearl gateway. Update the adapter mapping and commit the fixture.
|
||||
2. **`PEARL_PINNED_REF`.** Pick a specific commit/tag at the start of implementation. Document the rev-bump workflow.
|
||||
3. **The `pearl-ai/Llama-3.3-70B-Instruct-pearl` HF model.** Confirm it exists and is gated/ungated; document HF auth requirements.
|
||||
4. **Pearl Taproot address regex.** Confirm the prefix and length for `mine doctor`'s address-format check.
|
||||
5. **Pearl `:8337` miner RPC TCP port behavior.** Confirm `MINER_RPC_TRANSPORT=tcp` works as documented and binds to `0.0.0.0` not just `127.0.0.1` inside the host network namespace.
|
||||
6. **OJ default Docker image tag.** Decide whether to publish to GHCR/Docker Hub once we have a build, or leave users on build-from-pin. Likely v1.x.
|
||||
7. **Wallet generation hand-off (v1.x).** Decide whether `mine init` shells out to Pearl's `oyster` for users who want guidance, or stays paste-only.
|
||||
8. **Telemetry schema migration approach.** Adding the nullable `mining_session_id` column to `telemetry/store.py` is a SQLite schema change. Decide between (a) `ALTER TABLE` on first start with a guarded `PRAGMA user_version` bump, (b) per-query `try/except` on the column, or (c) creating a sidecar table joined on inference id. Confirm what convention OJ already uses for `telemetry/` schema evolution before picking; default lean is (a).
|
||||
|
||||
## 12. Cross-references
|
||||
|
||||
- **[Spec B — Apple Silicon enablement](2026-05-05-apple-silicon-pearl-mining-design.md)** — separate effort tracking the Pearl-side and OJ-side work to make Apple Silicon a registered `MiningProvider`. Spec A is engine-agnostic by design; Spec B drops in via `MinerRegistry` without modifying anything in this spec.
|
||||
- **Pearl repo:** [`pearl-research-labs/pearl`](https://github.com/pearl-research-labs/pearl) — referenced sub-paths: `miner/vllm-miner/`, `miner/pearl-gemm/`, `miner/pearl-gateway/`, `miner/vllm-miner/Dockerfile`, `miner/vllm-miner/entrypoint.sh`.
|
||||
- **Pearl paper:** [Proof-of-Useful-Work via matrix multiplication (arXiv:2504.09971)](https://arxiv.org/abs/2504.09971).
|
||||
- **OJ contributing guide:** `docs/development/contributing.md` — registry pattern, `_stubs.py` / `_discovery.py` conventions, `ensure_registered()` discipline, optional-deps soft-import pattern. All followed in this spec.
|
||||
|
||||
## 13. Implementation plan
|
||||
|
||||
The implementation plan for Spec A is a separate document, written via the `superpowers:writing-plans` skill after this design is approved by the user. It will decompose section 4–9 above into ordered, independently-reviewable steps and call out which steps can be parallelized.
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,112 @@
|
||||
# Desktop auto-update
|
||||
|
||||
The OpenJarvis desktop app ships with [Tauri's updater
|
||||
plugin](https://v2.tauri.app/plugin/updater/), which checks for new
|
||||
versions on launch and every 30 minutes. When a newer signed build is
|
||||
available, the app prompts the user to download and install it.
|
||||
|
||||
## How it works
|
||||
|
||||
```
|
||||
on launch / every 30 min
|
||||
│
|
||||
▼
|
||||
GET https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-latest/latest.json
|
||||
│
|
||||
▼
|
||||
Parse manifest: { "version": "X.Y.Z", "platforms": { ... } }
|
||||
│
|
||||
▼
|
||||
If manifest.version > installed_version:
|
||||
download signed .dmg / .deb / .msi from manifest.platforms[target].url
|
||||
verify against the minisign pubkey baked into the app
|
||||
prompt user to install
|
||||
```
|
||||
|
||||
The frontend code lives in
|
||||
[`frontend/src/components/Desktop/UpdateChecker.tsx`](../frontend/src/components/Desktop/UpdateChecker.tsx);
|
||||
the Tauri wiring is in
|
||||
[`frontend/src-tauri/tauri.conf.json`](../frontend/src-tauri/tauri.conf.json)
|
||||
under `plugins.updater`.
|
||||
|
||||
## How releases reach the update endpoint
|
||||
|
||||
The `Desktop Build & Release` GitHub Action
|
||||
([`.github/workflows/desktop.yml`](../.github/workflows/desktop.yml))
|
||||
builds signed binaries plus a `latest.json` manifest with the
|
||||
`tauri-action` step (`includeUpdaterJson: true` generates the manifest
|
||||
automatically). Where it publishes depends on the trigger.
|
||||
|
||||
Three release streams exist:
|
||||
|
||||
- **`desktop-latest`** (stable auto-update channel): **this is the
|
||||
channel the installed app polls.** It is *not* built directly —
|
||||
instead, when a stable `desktop-vX.Y.Z` release is published, the
|
||||
`refresh-stable-channel` job copies that release's `latest.json`
|
||||
into `desktop-latest`. So the app is only ever offered vetted stable
|
||||
builds, and `latest.json` here points at the current `desktop-v*`
|
||||
assets.
|
||||
- **`desktop-vX.Y.Z`** (tagged stable): created when someone pushes a
|
||||
`desktop-v*` git tag. The user-facing stable release with full
|
||||
installers; also the source of truth the stable channel mirrors.
|
||||
- **`desktop-edge`** (rolling pre-release): rebuilt on every push to
|
||||
`main` (via the `autotag` → `desktop.yml` dispatch) and on manual
|
||||
`workflow_dispatch`. Carries the most recent CI build for testers.
|
||||
The shipped app does **not** poll this stream, so dev builds never
|
||||
auto-install onto stable users.
|
||||
|
||||
This split means security and telemetry-policy fixes reach users on
|
||||
the next **stable** `desktop-v*` tag — cut one to ship an update.
|
||||
Edge builds are available for anyone who wants to test `main` ahead of
|
||||
a stable tag, without risking the stable population.
|
||||
|
||||
## Signing
|
||||
|
||||
Binaries are signed by `tauri-action` using the minisign key pair
|
||||
referenced via these GitHub Actions secrets:
|
||||
|
||||
| Secret | Purpose |
|
||||
|---|---|
|
||||
| `TAURI_SIGNING_PRIVATE_KEY` | Private key (PEM-formatted minisign) |
|
||||
| `TAURI_SIGNING_PRIVATE_KEY_PASSWORD` | Passphrase for the private key |
|
||||
|
||||
The matching public key is baked into the app at
|
||||
`tauri.conf.json:plugins.updater.pubkey`. If you ever need to rotate
|
||||
the key, replace the public key in the JSON file *and* update both
|
||||
secrets atomically — mismatched keys cause every update download to
|
||||
fail signature verification with no recovery path other than a manual
|
||||
reinstall.
|
||||
|
||||
## Disabling the updater locally
|
||||
|
||||
For frontend development, set `VITE_OPENJARVIS_NO_UPDATER=1` in your
|
||||
shell before running `npm run tauri dev`. Vite injects any
|
||||
`VITE_`-prefixed env var into `import.meta.env`, and the
|
||||
`UpdateChecker.tsx` component honors it to skip the 30-minute poll.
|
||||
|
||||
```bash
|
||||
export VITE_OPENJARVIS_NO_UPDATER=1
|
||||
npm run tauri dev
|
||||
```
|
||||
|
||||
This is purely a dev escape hatch — it has no effect on production
|
||||
builds (where `import.meta.env.VITE_OPENJARVIS_NO_UPDATER` will be
|
||||
`undefined` unless you explicitly set it at build time).
|
||||
|
||||
## Verifying a release manually
|
||||
|
||||
```bash
|
||||
# Download the latest manifest and confirm it parses cleanly
|
||||
curl -fsSL https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-latest/latest.json | jq .
|
||||
|
||||
# Fields:
|
||||
# version — semver string, must match the tag (without leading "v")
|
||||
# notes — release notes string
|
||||
# pub_date — RFC3339 timestamp
|
||||
# platforms — map keyed by "<target>-<arch>" e.g. "darwin-aarch64"
|
||||
# each entry has { signature: "...", url: "..." }
|
||||
```
|
||||
|
||||
A 404 on the manifest URL means the most recent desktop CI run
|
||||
didn't complete or didn't have signing secrets — check the
|
||||
`Desktop Build & Release` workflow logs.
|
||||
@@ -0,0 +1,289 @@
|
||||
# NVIDIA Pearl Mining Validation Runbook
|
||||
|
||||
This runbook is the release gate for the v1 `vllm-pearl` provider. Unit tests
|
||||
prove OpenJarvis wiring; this validates that a real H100/H200 host can mine
|
||||
through Pearl and serve inference through OpenJarvis.
|
||||
|
||||
## Required Host
|
||||
|
||||
Run this on a Linux machine with:
|
||||
|
||||
- NVIDIA H100 or H200 GPU, compute capability 9.0, at least 70 GB VRAM
|
||||
- Current NVIDIA driver with CUDA container support
|
||||
- Docker 24+ and `nvidia-container-toolkit`
|
||||
- At least 200 GB free disk
|
||||
- Reachable `pearld` JSON-RPC endpoint
|
||||
- Pearl payout address beginning with `prl1q` or `prl1p`
|
||||
- Hugging Face access to `pearl-ai/Llama-3.3-70B-Instruct-pearl`
|
||||
|
||||
The validated H100 configuration uses `gpu_memory_utilization = 0.96` with
|
||||
`max_model_len = 8192`. Lower memory utilization can fail during vLLM startup
|
||||
because the Pearl 70B mining model leaves too little KV cache at 8k context.
|
||||
|
||||
Do not run this on macOS, Apple Silicon, AMD, RTX 4090, or CPU-only hosts.
|
||||
Those are separate providers.
|
||||
|
||||
## Wallet Address Setup
|
||||
|
||||
Create the wallet from the Pearl repo root:
|
||||
|
||||
```bash
|
||||
./bin/oyster -u rpcuser -P rpcpass --create
|
||||
```
|
||||
|
||||
If you choose the optional public-data encryption prompt, Oyster will require
|
||||
that public passphrase on startup via `--walletpass`. Keep private and public
|
||||
passphrases out of shell history where possible.
|
||||
|
||||
Start Oyster:
|
||||
|
||||
```bash
|
||||
./bin/oyster \
|
||||
-u rpcuser \
|
||||
-P rpcpass \
|
||||
--walletpass '<public-wallet-passphrase-if-configured>' \
|
||||
&
|
||||
```
|
||||
|
||||
Then generate a mining address through the wallet RPC:
|
||||
|
||||
```bash
|
||||
./bin/prlctl \
|
||||
--wallet \
|
||||
--skipverify \
|
||||
-u rpcuser \
|
||||
-P rpcpass \
|
||||
-s localhost:44207 \
|
||||
getnewaddress
|
||||
```
|
||||
|
||||
Notes:
|
||||
|
||||
- `--wallet` is required. Without it, `prlctl` talks to `pearld` instead of
|
||||
Oyster and may look for `Pearld/pearld.conf`.
|
||||
- Use `-s localhost:44207`, not `-s https://localhost:44207`. `prlctl` expects
|
||||
host and port, not a URL.
|
||||
- `--skipverify` is acceptable for this local validation flow unless you have
|
||||
configured the Oyster RPC certificate path.
|
||||
- If a mnemonic has been pasted into logs, chat, or a PR, discard that wallet
|
||||
and create a fresh one before mining.
|
||||
|
||||
## Environment
|
||||
|
||||
```bash
|
||||
git checkout feat/mining-spec-a-only
|
||||
uv sync --extra dev --extra mining-pearl-vllm
|
||||
|
||||
export PEARLD_RPC_PASSWORD='<pearld-rpc-password>'
|
||||
export HF_TOKEN='<huggingface-token>'
|
||||
```
|
||||
|
||||
Confirm host prerequisites:
|
||||
|
||||
```bash
|
||||
nvidia-smi
|
||||
docker info
|
||||
docker run --rm --gpus all nvidia/cuda:12.9.1-base-ubuntu24.04 nvidia-smi
|
||||
df -h ~/.cache
|
||||
```
|
||||
|
||||
Expected:
|
||||
|
||||
- `nvidia-smi` shows H100 or H200.
|
||||
- Docker can run a CUDA container with GPU access.
|
||||
- `~/.cache` or the Hugging Face cache volume has at least 200 GB free.
|
||||
|
||||
On shared hosts, select only idle GPUs during `mine init`:
|
||||
|
||||
```bash
|
||||
uv run jarvis mine init --cuda-visible-devices 0
|
||||
```
|
||||
|
||||
This writes `[mining.extra].cuda_visible_devices`. `mine start` passes that
|
||||
device list to Docker and sets `CUDA_VISIBLE_DEVICES` /
|
||||
`NVIDIA_VISIBLE_DEVICES` inside the container. Omit the option only on a
|
||||
dedicated host where the miner may use all GPUs.
|
||||
|
||||
## Configure Mining
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
uv run jarvis mine doctor
|
||||
```
|
||||
|
||||
Before config exists, `doctor` should show hardware and Docker as OK, and
|
||||
Pearl node / wallet as unconfigured.
|
||||
|
||||
Then initialize:
|
||||
|
||||
```bash
|
||||
uv run jarvis mine init
|
||||
```
|
||||
|
||||
Use:
|
||||
|
||||
- Wallet: the user's Pearl `prl1q...` or `prl1p...` address
|
||||
- `pearld` URL: usually `http://localhost:44107`
|
||||
- RPC user: configured `pearld` user, often `rpcuser`
|
||||
- Password env: `PEARLD_RPC_PASSWORD`
|
||||
- Model: `pearl-ai/Llama-3.3-70B-Instruct-pearl`
|
||||
- Image: default unless validating a custom Pearl image
|
||||
- CUDA devices: an idle GPU ID such as `0` on shared hosts
|
||||
|
||||
Expected:
|
||||
|
||||
- `[mining]` and `[mining.extra]` are written to config.
|
||||
- Image resolves locally, pulls, or builds from the pinned Pearl ref.
|
||||
- First build may take 30-60 minutes.
|
||||
|
||||
Run `doctor` again:
|
||||
|
||||
```bash
|
||||
uv run jarvis mine doctor
|
||||
```
|
||||
|
||||
Expected:
|
||||
|
||||
- Hardware OK
|
||||
- Docker OK
|
||||
- Disk OK
|
||||
- Pearl node RPC OK and synced
|
||||
- Wallet format OK
|
||||
- `vllm-pearl SUPPORTED`
|
||||
- Sidecar absent
|
||||
|
||||
## Start Mining
|
||||
|
||||
```bash
|
||||
uv run jarvis mine start
|
||||
```
|
||||
|
||||
Expected:
|
||||
|
||||
- Docker container `openjarvis-pearl-miner` starts.
|
||||
- `~/.openjarvis/runtime/mining.json` is written.
|
||||
- Sidecar contains `vllm_endpoint`, `gateway_url`, `gateway_metrics_url`, and
|
||||
`container_id`.
|
||||
|
||||
Inspect:
|
||||
|
||||
```bash
|
||||
docker ps --filter name=openjarvis-pearl-miner
|
||||
cat ~/.openjarvis/runtime/mining.json
|
||||
uv run jarvis mine logs --tail 200
|
||||
uv run jarvis mine status
|
||||
```
|
||||
|
||||
Expected:
|
||||
|
||||
- Container is running.
|
||||
- vLLM is listening on the configured port, default `8000`.
|
||||
- Pearl gateway metrics are available on the configured metrics port, default
|
||||
`8339`.
|
||||
- `mine status` exits 0 and prints `provider: vllm-pearl`.
|
||||
|
||||
## Verify OpenJarvis Inference Uses Mining Endpoint
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
uv run jarvis mine doctor
|
||||
uv run jarvis ask "Say hello in one sentence."
|
||||
```
|
||||
|
||||
Expected:
|
||||
|
||||
- `doctor` shows sidecar present.
|
||||
- Engine discovery registers `vllm-pearl-mining`.
|
||||
- The prompt completes through the Pearl/vLLM endpoint.
|
||||
- Container logs show vLLM activity during the prompt.
|
||||
|
||||
If inference succeeds but mining stats stay zero, continue to the Pearl
|
||||
network checks below; vLLM serving alone is not enough to prove mining.
|
||||
|
||||
## Verify Pearl Network Submission
|
||||
|
||||
Check gateway metrics directly:
|
||||
|
||||
```bash
|
||||
curl -fsS http://127.0.0.1:8339/metrics | tee /tmp/pearl-gateway-metrics.txt
|
||||
uv run jarvis mine status
|
||||
```
|
||||
|
||||
Expected:
|
||||
|
||||
- Metrics endpoint returns Prometheus text.
|
||||
- If Pearl exposes share counters, `mine status` maps them correctly.
|
||||
- If metric names differ, attach `/tmp/pearl-gateway-metrics.txt` to the PR and
|
||||
update `src/openjarvis/mining/_metrics.py`.
|
||||
|
||||
Check `pearld` connectivity using the same RPC configuration used by mining:
|
||||
|
||||
```bash
|
||||
curl --user "rpcuser:${PEARLD_RPC_PASSWORD}" \
|
||||
--data-binary '{"jsonrpc":"1.0","id":"oj","method":"getblockchaininfo","params":[]}' \
|
||||
-H 'content-type: text/plain;' \
|
||||
http://127.0.0.1:44107
|
||||
```
|
||||
|
||||
Expected:
|
||||
|
||||
- `blocks` and `headers` are present.
|
||||
- Node is synced or close enough for mining validation.
|
||||
|
||||
Proof of actual earning requires a successful accepted share/block and wallet
|
||||
credit. Depending on Pearl network difficulty, this may take longer than the
|
||||
smoke test window. Record:
|
||||
|
||||
- Runtime duration
|
||||
- `mine status` before and after
|
||||
- Gateway metrics snapshot
|
||||
- Relevant container log tail
|
||||
- Wallet balance / transaction evidence if a reward lands
|
||||
|
||||
## Stop And Cleanup
|
||||
|
||||
```bash
|
||||
uv run jarvis mine stop
|
||||
docker ps --filter name=openjarvis-pearl-miner
|
||||
test ! -e ~/.openjarvis/runtime/mining.json
|
||||
```
|
||||
|
||||
Expected:
|
||||
|
||||
- Container stops.
|
||||
- Sidecar is removed.
|
||||
- `jarvis ask` no longer routes through `vllm-pearl-mining` unless another
|
||||
mining sidecar is attached.
|
||||
|
||||
## Pass Criteria
|
||||
|
||||
The NVIDIA provider is considered proven when all are true:
|
||||
|
||||
- `mine doctor` reports supported on H100/H200.
|
||||
- `mine init` resolves/builds the Pearl image.
|
||||
- `mine start` launches the container and writes the sidecar.
|
||||
- OpenJarvis inference succeeds through `vllm-pearl-mining`.
|
||||
- Pearl gateway metrics are reachable and `mine status` parses them.
|
||||
- `pearld` accepts the miner's network path.
|
||||
- At least one accepted share/block is observed, or a documented Pearl
|
||||
maintainer confirmation says the observed gateway state is sufficient proof
|
||||
of live mining.
|
||||
|
||||
## Failure Artifacts
|
||||
|
||||
For any failure, collect:
|
||||
|
||||
```bash
|
||||
uv run jarvis mine doctor
|
||||
uv run jarvis mine status || true
|
||||
uv run jarvis mine logs --tail 300 || true
|
||||
docker inspect openjarvis-pearl-miner || true
|
||||
curl -fsS http://127.0.0.1:8339/metrics || true
|
||||
nvidia-smi
|
||||
docker info
|
||||
```
|
||||
|
||||
Attach outputs to the implementation PR or follow-up issue. Do not paste
|
||||
`PEARLD_RPC_PASSWORD`, wallet seed material, or Hugging Face tokens.
|
||||
@@ -0,0 +1,84 @@
|
||||
# Adding a Mining Provider
|
||||
|
||||
The `openjarvis.mining` subsystem follows the same registry pattern as engines,
|
||||
agents, tools, memory, and channels. New mining paths should be provider
|
||||
modules, not special cases in the CLI or engine layer.
|
||||
|
||||
## Provider Contract
|
||||
|
||||
Every provider implements `openjarvis.mining.MiningProvider`:
|
||||
|
||||
- `detect(hw, engine_id, model)` is pure capability detection. It must not
|
||||
start subprocesses, hit the network, or mutate state.
|
||||
- `start(config)` owns provider lifecycle setup and writes the mining sidecar
|
||||
when it changes inference routing.
|
||||
- `stop()` tears down provider-owned processes or containers.
|
||||
- `is_running()` answers from provider-owned state.
|
||||
- `stats()` returns `MiningStats` using the provider's most stable telemetry
|
||||
surface.
|
||||
|
||||
Register providers through `MinerRegistry` and expose idempotent
|
||||
`ensure_registered()`:
|
||||
|
||||
```python
|
||||
from openjarvis.core.registry import MinerRegistry
|
||||
|
||||
|
||||
def ensure_registered() -> None:
|
||||
if not MinerRegistry.contains("my-provider"):
|
||||
MinerRegistry.register_value("my-provider", MyProvider)
|
||||
```
|
||||
|
||||
`tests/conftest.py` clears registries between tests, so test fixtures and CLI
|
||||
entry points should call `ensure_registered()` before relying on a provider.
|
||||
|
||||
## Optional Dependencies
|
||||
|
||||
Provider dependencies belong in scoped extras:
|
||||
|
||||
- `mining-pearl-vllm` for the NVIDIA/vLLM Docker provider
|
||||
- Future Apple work should use a separate extra such as `mining-pearl-metal`
|
||||
or `mining-pearl-cpu`
|
||||
|
||||
Avoid a generic `mining-pearl` extra until there is a shared dependency set
|
||||
that every provider actually needs.
|
||||
|
||||
## Sidecar Contract
|
||||
|
||||
The runtime sidecar lives at `~/.openjarvis/runtime/mining.json`. Engine
|
||||
handoff is data-driven:
|
||||
|
||||
- If the sidecar has `vllm_endpoint`, engine discovery registers
|
||||
`vllm-pearl-mining`.
|
||||
- If a future provider mines alongside the user's normal engine, it should omit
|
||||
`vllm_endpoint`; engine discovery will ignore it.
|
||||
|
||||
Do not branch on `provider == "vllm-pearl"` in generic code. Branch on sidecar
|
||||
shape or provider capability.
|
||||
|
||||
## Apple Silicon Handoff
|
||||
|
||||
The Apple Silicon effort should add its own provider module and reuse:
|
||||
|
||||
- `MiningProvider`
|
||||
- `MinerRegistry`
|
||||
- `MiningConfig`
|
||||
- `MiningStats`
|
||||
- `Sidecar`
|
||||
- `jarvis mine doctor` capability iteration
|
||||
|
||||
That work should not need to rewrite the NVIDIA provider, CLI group, telemetry
|
||||
collector, or engine sidecar handoff.
|
||||
|
||||
## NVIDIA Release Gate
|
||||
|
||||
The NVIDIA provider is not considered economically proven until the H100/H200
|
||||
runbook passes on real hardware. See
|
||||
[`mining-nvidia-validation.md`](./mining-nvidia-validation.md) for the required
|
||||
commands, artifacts, and pass criteria.
|
||||
|
||||
## Model Enablement
|
||||
|
||||
New Pearl-compatible language models are tracked separately from provider
|
||||
support. See [`pearl-model-enablement.md`](./pearl-model-enablement.md) for the
|
||||
conversion and validation checklist.
|
||||
@@ -0,0 +1,148 @@
|
||||
# Pearl Model Enablement
|
||||
|
||||
This page tracks the work required to make a new Hugging Face model mineable
|
||||
through Pearl's vLLM miner and OpenJarvis.
|
||||
|
||||
OpenJarvis can point `vllm-pearl` at a model id, but a raw Hugging Face model is
|
||||
not enough. The Pearl vLLM plugin expects a Pearl-compatible quantized model
|
||||
whose metadata marks mining layers for 7-bit NoisyGEMM and non-mining layers
|
||||
for the vanilla Pearl GEMM path.
|
||||
|
||||
## Supported Models
|
||||
|
||||
OpenJarvis only supports Pearl models published by the `pearl-ai` Hugging Face
|
||||
organization. Private staging artifacts and OpenJarvis-specific conversion
|
||||
repos are not user-facing supported mining models.
|
||||
|
||||
The current public support set is:
|
||||
|
||||
| Raw model | Pearl model | Status |
|
||||
|---|---|---|
|
||||
| `meta-llama/Llama-3.3-70B-Instruct` | `pearl-ai/Llama-3.3-70B-Instruct-pearl` | Validated default |
|
||||
| `google/gemma-4-31B-it` | `pearl-ai/Gemma-4-31B-it-pearl` | Planned until H100/H200 validation passes |
|
||||
| `meta-llama/Llama-3.1-8B-Instruct` | `pearl-ai/Llama-3.1-8B-Instruct-pearl` | Planned until H100/H200 validation passes |
|
||||
|
||||
## Current Validation Findings
|
||||
|
||||
The H100 smoke run validated the default Llama Pearl model end to end through
|
||||
`jarvis mine start`, vLLM `/v1/models`, OpenJarvis inference routing, Pearl
|
||||
gateway template refresh, and `jarvis mine validate-model`.
|
||||
|
||||
`pearl-ai/Gemma-4-31B-it-pearl` and
|
||||
`pearl-ai/Llama-3.1-8B-Instruct-pearl` are listed because they are public Pearl
|
||||
org artifacts. They remain `planned` in OpenJarvis until we have clean
|
||||
H100/H200 validation artifacts for the published repos.
|
||||
|
||||
## Enablement Checklist
|
||||
|
||||
1. Reproduce the current Llama Pearl model recipe.
|
||||
- Record the compressed-tensors config.
|
||||
- Record which linear layers are 7-bit mining layers.
|
||||
- Record which layers are 8-bit non-mining layers.
|
||||
- Record calibration data and SmoothQuant settings, if used.
|
||||
|
||||
2. Convert the target model.
|
||||
- Start with a model Pearl intends to publish under the `pearl-ai` org.
|
||||
- Generate Pearl-compatible quantized weights and metadata.
|
||||
- For Gemma4 artifacts, include the base model's processor metadata required
|
||||
by vLLM's Gemma4 multimodal profiler.
|
||||
- Publish under the planned `pearl-ai/*-pearl` id before enabling it in
|
||||
OpenJarvis.
|
||||
|
||||
OpenJarvis includes an experimental local converter for this work:
|
||||
|
||||
```bash
|
||||
python scripts/pearl/model_converter.py \
|
||||
meta-llama/Llama-3.1-8B-Instruct \
|
||||
/tmp/pearl-ai-Llama-3.1-8B-Instruct-pearl \
|
||||
--device cuda
|
||||
```
|
||||
|
||||
The converter copies Hugging Face metadata, emits
|
||||
`quantization_config.quant_method = "pearl"`, writes a safetensors index,
|
||||
converts attention q/k/v and MLP down projections to int8 non-mining layers,
|
||||
and converts the remaining text linear weights to int7 mining layers. Treat
|
||||
its output as a staging artifact until `jarvis mine inspect-model` and
|
||||
`jarvis mine validate-model` pass on H100/H200 hardware.
|
||||
|
||||
Local staging artifacts can be inspected before upload:
|
||||
|
||||
```bash
|
||||
jarvis mine inspect-model \
|
||||
--model /tmp/pearl-ai-Llama-3.1-8B-Instruct-pearl
|
||||
```
|
||||
|
||||
To run a local staging artifact through the Docker miner, keep `--model` as
|
||||
the intended served model name and point `--local-model-path` at the
|
||||
converted checkpoint directory:
|
||||
|
||||
```bash
|
||||
jarvis mine init \
|
||||
--provider vllm-pearl \
|
||||
--wallet-address <prl1...> \
|
||||
--model pearl-ai/Llama-3.1-8B-Instruct-pearl \
|
||||
--local-model-path /tmp/pearl-ai-Llama-3.1-8B-Instruct-pearl \
|
||||
--cuda-visible-devices 1 \
|
||||
--vllm-arg=--language-model-only \
|
||||
--vllm-arg=--skip-mm-profiling
|
||||
jarvis mine start
|
||||
```
|
||||
|
||||
3. Validate the Pearl vLLM plugin path.
|
||||
- Run `jarvis mine inspect-model --model <pearl-model-id>
|
||||
--allow-planned` before starting the miner.
|
||||
- Model loads in Pearl's `vllm-miner` container.
|
||||
- vLLM registers Pearl's quantization plugin.
|
||||
- Mining layers use int7 NoisyGEMM.
|
||||
- Non-mining layers use int8 vanilla Pearl GEMM.
|
||||
- Text generation works with mining enabled and disabled.
|
||||
|
||||
4. Validate chain integration.
|
||||
- `pearld` is reachable.
|
||||
- `pearl-gateway` receives work.
|
||||
- NoisyGEMM submits candidate proofs.
|
||||
- Gateway reports metrics.
|
||||
- `jarvis mine status` parses those metrics.
|
||||
|
||||
5. Promote the model in OpenJarvis.
|
||||
- Change its registry status from `planned` to `validated`.
|
||||
- Set measured VRAM and context defaults.
|
||||
- Add the model to user docs.
|
||||
- Attach validation logs to the PR.
|
||||
|
||||
## OpenJarvis Registry
|
||||
|
||||
Model support metadata lives in:
|
||||
|
||||
```text
|
||||
src/openjarvis/mining/_models.py
|
||||
```
|
||||
|
||||
`jarvis mine models` renders that registry. Planned models are visible to users
|
||||
but blocked by capability detection until the Pearl model artifact and H100/H200
|
||||
validation exist.
|
||||
|
||||
## Acceptance Criteria
|
||||
|
||||
A model is `validated` only when all of these pass on real hardware:
|
||||
|
||||
- `jarvis mine inspect-model --model <pearl-model-id> --allow-planned`
|
||||
- `jarvis mine init --model <pearl-model-id>`
|
||||
- `jarvis mine start`
|
||||
- `curl http://127.0.0.1:8000/v1/models`
|
||||
- `jarvis ask "Say hello in one sentence."`
|
||||
- `jarvis mine status`
|
||||
- `jarvis mine validate-model --model <pearl-model-id> --allow-planned --prompt
|
||||
"Say hello in one sentence." --output <artifact>.json`
|
||||
- Pearl gateway metrics show the mining path is active.
|
||||
- No block/share submission errors appear in gateway or miner logs.
|
||||
|
||||
Do not mark a model validated based only on vLLM load success. It must exercise
|
||||
Pearl's NoisyGEMM and submission path.
|
||||
|
||||
## Tracking
|
||||
|
||||
Use the `Pearl Model Validation` GitHub issue template for each candidate model.
|
||||
The issue should hold the quantization recipe, hardware details, command output,
|
||||
metrics excerpts, and the PR that changes the model status to `validated`.
|
||||
Attach the JSON artifact from `jarvis mine validate-model --output` to the issue.
|
||||
@@ -0,0 +1,52 @@
|
||||
# Release Checklist
|
||||
|
||||
Before tagging a release, run through this checklist on real machines (not just CI containers).
|
||||
|
||||
## Manual smoke tests (~30 min total)
|
||||
|
||||
For each platform below, start from a fresh user account / VM snapshot. Run the install one-liner and verify the steps in the table.
|
||||
|
||||
| Platform | One-liner | Verify |
|
||||
|---|---|---|
|
||||
| macOS Intel laptop | `curl -fsSL <url> \| bash` | (1)–(8) below |
|
||||
| macOS ARM laptop | same | (1)–(8) |
|
||||
| Ubuntu 22.04 fresh VM | same | (1)–(8) |
|
||||
| Fedora 40 fresh VM | same | (1)–(8) |
|
||||
| WSL2 Ubuntu on Windows | same | (1)–(8) |
|
||||
|
||||
### Verification steps
|
||||
|
||||
1. **Install completes ≤ 5 min** on typical broadband.
|
||||
2. **`jarvis` (no args)** drops into a chat session within 2 s.
|
||||
3. **First chat turn returns a response** from `qwen3.5:2b` via Ollama.
|
||||
4. **Banner shows background work** ("Setting up in background: …") while it's still going.
|
||||
5. **Completion notification fires** between turns when the bg work finishes (Rust extension or a model).
|
||||
6. **`jarvis doctor`** exits 0 once all bg work completes; shows the Background tasks table.
|
||||
7. **Re-run `curl … | bash`** on the same machine. It completes ≤ 30 s, says `[ok] step already done` for every step.
|
||||
8. **`jarvis-uninstall`** removes `~/.openjarvis/` and `~/.local/bin/jarvis*`. Verify with `ls`.
|
||||
|
||||
## Cloud quick-path verification
|
||||
|
||||
On any one platform:
|
||||
|
||||
```bash
|
||||
export ANTHROPIC_API_KEY=test-fake-key
|
||||
jarvis init --force
|
||||
```
|
||||
|
||||
Verify init proposes cloud (mentions "anthropic" in the prompt), and the resulting `config.toml` has `[intelligence] provider = "anthropic"`.
|
||||
|
||||
## Failure-mode spot checks
|
||||
|
||||
Run at least one failure scenario per release; rotate which one.
|
||||
|
||||
- Disconnect network mid-install — verify clear error and re-run completes.
|
||||
- Delete `~/.openjarvis/config.toml` — verify bare `jarvis` re-runs init.
|
||||
- Delete `~/.openjarvis/.venv` — verify re-running curl heals it.
|
||||
- `EUID=0 bash install.sh` — verify hard-fail with "don't run as root".
|
||||
|
||||
## CI gates (automated, no manual action)
|
||||
|
||||
- All pytest tests pass: `uv run pytest tests/`
|
||||
- All bats tests pass: see `.github/workflows/bash-tests.yml`
|
||||
- Container integration matrix is green: see `.github/workflows/installer-integration.yml`
|
||||
@@ -9,7 +9,7 @@ These are the areas where active development is happening and contributions are
|
||||
- **Energy-aware routing** — using power consumption data from telemetry to optimize for energy efficiency alongside latency and quality
|
||||
- **Plugin ecosystem** — community-contributed engines, tools, and agents distributed as Python packages
|
||||
- **Federated memory** — memory backends that synchronize across devices
|
||||
- **Distillation:** Frontier-driven harness learning — a frontier model analyzes your traces and proposes config improvements. See [user guide](../user-guide/learning-distillation.md) and [architecture](../architecture/learning.md#distillation-frontier-driven-harness-learning).
|
||||
- **LLM-guided spec search:** Frontier-driven harness learning — a frontier model analyzes your traces and proposes config improvements. See [user guide](../user-guide/llm-guided-spec-search.md) and [architecture](../architecture/learning.md#llm-guided-spec-search-frontier-driven-harness-learning).
|
||||
|
||||
---
|
||||
|
||||
|
||||
+7
-7
@@ -25,11 +25,11 @@ processing happens on your local machine — the app connects to the backend you
|
||||
|
||||
| Platform | Download | Notes |
|
||||
|----------|----------|-------|
|
||||
| macOS (Apple Silicon) | [:material-download: **OpenJarvis.dmg**](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-latest/OpenJarvis_0.1.0_aarch64.dmg) | M1/M2/M3/M4 Macs |
|
||||
| Windows (64-bit) | [:material-download: **OpenJarvis-setup.exe**](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-latest/OpenJarvis_0.1.0_x64-setup.exe) | Windows 10+ |
|
||||
| Linux (DEB) | [:material-download: **OpenJarvis.deb**](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-latest/OpenJarvis_0.1.0_amd64.deb) | Ubuntu, Debian |
|
||||
| Linux (RPM) | [:material-download: **OpenJarvis.rpm**](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-latest/OpenJarvis-0.1.0-1.x86_64.rpm) | Fedora, RHEL |
|
||||
| Linux (AppImage) | [:material-download: **OpenJarvis.AppImage**](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-latest/OpenJarvis_0.1.0_amd64.AppImage) | Any distro |
|
||||
| macOS (Universal) | [:material-download: **OpenJarvis.dmg**](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-v1.0.2/OpenJarvis_1.0.1_universal.dmg) | Apple Silicon + Intel |
|
||||
| Windows (64-bit) | [:material-download: **OpenJarvis-setup.exe**](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-v1.0.2/OpenJarvis_1.0.1_x64-setup.exe) | Windows 10+ |
|
||||
| Linux (DEB) | [:material-download: **OpenJarvis.deb**](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-v1.0.2/OpenJarvis_1.0.1_amd64.deb) | Ubuntu, Debian |
|
||||
| Linux (RPM) | [:material-download: **OpenJarvis.rpm**](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-v1.0.2/OpenJarvis-1.0.1-1.x86_64.rpm) | Fedora, RHEL |
|
||||
| Linux (AppImage) | [:material-download: **OpenJarvis.AppImage**](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-v1.0.2/OpenJarvis_1.0.1_amd64.AppImage) | Any distro |
|
||||
|
||||
!!! tip "All releases"
|
||||
Browse all versions on the [GitHub Releases](https://github.com/open-jarvis/OpenJarvis/releases) page.
|
||||
@@ -94,7 +94,7 @@ cd OpenJarvis
|
||||
|
||||
The script handles everything:
|
||||
|
||||
1. Checks for Python 3.10+ and Node.js 18+
|
||||
1. Checks for Python 3.10–3.13 and Node.js 18+
|
||||
2. Installs Ollama if not present and pulls a starter model
|
||||
3. Installs Python and frontend dependencies
|
||||
4. Starts the backend API server and frontend dev server
|
||||
@@ -109,7 +109,7 @@ If you prefer to run each step yourself:
|
||||
```bash
|
||||
git clone https://github.com/open-jarvis/OpenJarvis.git
|
||||
cd OpenJarvis
|
||||
uv sync --extra server
|
||||
uv sync --extra desktop
|
||||
cd frontend && npm install && cd ..
|
||||
```
|
||||
|
||||
|
||||
@@ -0,0 +1,30 @@
|
||||
"""Publish the canonical install scripts into the docs site.
|
||||
|
||||
Serves the installers at::
|
||||
|
||||
https://open-jarvis.github.io/OpenJarvis/install.sh (Linux / macOS / WSL2)
|
||||
https://open-jarvis.github.io/OpenJarvis/install.ps1 (native Windows)
|
||||
|
||||
so users have an HTTPS-valid, project-controlled install URL that does not
|
||||
depend on the externally-hosted ``openjarvis.ai`` domain — whose TLS config
|
||||
broke and which the project does not control (issue #337).
|
||||
|
||||
Single source of truth: the scripts live under ``scripts/install/`` and
|
||||
``deploy/windows/`` (also bundled into the wheel as ``_install_scripts/``).
|
||||
This copies them verbatim into the built site on every ``mkdocs build``,
|
||||
so the published copies can never drift from the canonical ones.
|
||||
"""
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import mkdocs_gen_files
|
||||
|
||||
# (source path, published URL path)
|
||||
_SCRIPTS = [
|
||||
(Path("scripts/install/install.sh"), "install.sh"),
|
||||
(Path("deploy/windows/install.ps1"), "install.ps1"),
|
||||
]
|
||||
|
||||
for src, dest in _SCRIPTS:
|
||||
with mkdocs_gen_files.open(dest, "wb") as out:
|
||||
out.write(src.read_bytes())
|
||||
@@ -17,6 +17,46 @@ The configuration file lives at:
|
||||
|
||||
OpenJarvis creates the `~/.openjarvis/` directory and populates it with a default config when you run `jarvis init`.
|
||||
|
||||
## Relocating the OpenJarvis directory
|
||||
|
||||
OpenJarvis keeps **all** of its state — config, databases, caches, logs,
|
||||
credentials, skills, recipes, connectors — under a **single root** so it never
|
||||
clutters your home directory beyond one folder. By default that root is
|
||||
`~/.openjarvis`, but you can move it.
|
||||
|
||||
The root is resolved in priority order:
|
||||
|
||||
1. **`$OPENJARVIS_HOME`** — explicit override. Honored by both the installer
|
||||
and the Python runtime.
|
||||
2. **`$XDG_DATA_HOME/openjarvis`** — used when `$XDG_DATA_HOME` is set (a single
|
||||
`openjarvis` directory nested under it, per the XDG Base Directory spec).
|
||||
3. **`~/.openjarvis`** — the default. With no environment variables set, the
|
||||
resolved path is exactly this, so existing installs are untouched.
|
||||
|
||||
```bash
|
||||
# Relocate the whole install + runtime tree at install time:
|
||||
OPENJARVIS_HOME=~/apps/openjarvis curl -fsSL https://open-jarvis.github.io/OpenJarvis/install.sh | bash
|
||||
|
||||
# Or for a single run / your shell profile:
|
||||
export OPENJARVIS_HOME=~/apps/openjarvis
|
||||
```
|
||||
|
||||
Confirm where your data lives with:
|
||||
|
||||
```bash
|
||||
jarvis config path
|
||||
```
|
||||
|
||||
!!! note "Migration"
|
||||
Because the default is unchanged, **no data migration is required** for
|
||||
existing installs. If you set `OPENJARVIS_HOME` (or `XDG_DATA_HOME`) on a
|
||||
machine that already has data in `~/.openjarvis`, OpenJarvis will look in
|
||||
the new location and not see your old data — move it yourself if you want
|
||||
to keep it: `mv ~/.openjarvis "$OPENJARVIS_HOME"`.
|
||||
|
||||
`$OPENJARVIS_CONFIG` still points at an explicit `config.toml` file
|
||||
independently of the root, if you need to override just the config file path.
|
||||
|
||||
## Generating Configuration
|
||||
|
||||
### First-Time Setup
|
||||
@@ -1060,21 +1100,21 @@ OpenJarvis respects the following environment variables:
|
||||
|
||||
---
|
||||
|
||||
## Learning & Distillation
|
||||
## Learning & spec search
|
||||
|
||||
The distillation subsystem uses a frontier model to automatically improve your local agent configuration. See the [user guide](../user-guide/learning-distillation.md) for a full walkthrough.
|
||||
LLM-guided spec search uses a frontier model to automatically improve your local agent configuration. See the [user guide](../user-guide/llm-guided-spec-search.md) for a full walkthrough.
|
||||
|
||||
### `[learning.distillation]`
|
||||
### `[learning.spec_search]`
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
| `enabled` | bool | `true` | Gate the entire distillation subsystem |
|
||||
| `enabled` | bool | `true` | Gate the entire spec-search subsystem |
|
||||
| `autonomy_mode` | string | `"tiered"` | `auto`, `tiered`, or `manual` |
|
||||
| `teacher_model` | string | `"claude-opus-4-6"` | Frontier model for diagnosis and planning |
|
||||
| `max_cost_per_session_usd` | float | `5.0` | Per-session teacher API budget |
|
||||
| `max_tool_calls_per_diagnosis` | int | `30` | Max teacher tool calls in diagnosis phase |
|
||||
|
||||
### `[learning.distillation.triggers]`
|
||||
### `[learning.spec_search.triggers]`
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
@@ -1086,7 +1126,7 @@ The distillation subsystem uses a frontier model to automatically improve your l
|
||||
| `cluster_min_size` | int | `5` | Minimum traces in a cluster |
|
||||
| `cluster_failure_threshold` | float | `0.3` | Feedback <= this counts as failure |
|
||||
|
||||
### `[learning.distillation.gate]`
|
||||
### `[learning.spec_search.gate]`
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
@@ -1095,7 +1135,7 @@ The distillation subsystem uses a frontier model to automatically improve your l
|
||||
| `benchmark_subsample_size` | int | `50` | Tasks per gate run |
|
||||
| `full_benchmark` | bool | `false` | Disable subsampling (slower, more accurate) |
|
||||
|
||||
### `[learning.distillation.benchmark]`
|
||||
### `[learning.spec_search.benchmark]`
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
@@ -1104,12 +1144,12 @@ The distillation subsystem uses a frontier model to automatically improve your l
|
||||
| `auto_refresh` | bool | `true` | Auto-mine new high-feedback traces |
|
||||
| `max_synthesis_cost_usd_per_refresh` | float | `2.0` | Cost cap per benchmark refresh |
|
||||
|
||||
### `[learning.distillation.tier_overrides]`
|
||||
### `[learning.spec_search.tier_overrides]`
|
||||
|
||||
Override the default risk tier for any operation. Keys are operation names, values are tier strings (`auto`, `review`, `manual`).
|
||||
|
||||
```toml
|
||||
[learning.distillation.tier_overrides]
|
||||
[learning.spec_search.tier_overrides]
|
||||
# patch_system_prompt = "auto" # promote to auto after trust
|
||||
# replace_system_prompt = "auto"
|
||||
```
|
||||
|
||||
@@ -0,0 +1,136 @@
|
||||
# Installation
|
||||
|
||||
## Platform-specific guides
|
||||
|
||||
| Platform | One-liner | Detailed guide |
|
||||
|---|---|---|
|
||||
| **macOS** | `curl -fsSL https://open-jarvis.github.io/OpenJarvis/install.sh \| bash` | [macOS install](macos.md) |
|
||||
| **Linux** | `curl -fsSL https://open-jarvis.github.io/OpenJarvis/install.sh \| bash` | [Linux install](linux.md) |
|
||||
| **WSL2 on Windows** | `curl -fsSL https://open-jarvis.github.io/OpenJarvis/install.sh \| bash` (run inside Ubuntu) | [WSL2 install](wsl2.md) |
|
||||
| **Native Windows** | `irm https://open-jarvis.github.io/OpenJarvis/install.ps1 \| iex` | [Native Windows install](windows-native.md) |
|
||||
| **Desktop GUI** | Download from the [latest release](https://github.com/open-jarvis/OpenJarvis/releases) | — |
|
||||
|
||||
The bash and PowerShell installers do the same thing on their respective hosts. The rest of this page documents the bash installer in detail; the [native Windows guide](windows-native.md) is the equivalent reference for PowerShell.
|
||||
|
||||
## Bash installer
|
||||
|
||||
```bash
|
||||
curl -fsSL https://open-jarvis.github.io/OpenJarvis/install.sh | bash
|
||||
```
|
||||
|
||||
The installer downloads everything for you — including [uv](https://docs.astral.sh/uv/)
|
||||
(the Python package manager), the Python venv, Ollama, and a small starter
|
||||
model. **You don't need to install uv or any other prerequisite first.**
|
||||
|
||||
!!! info "Install URL"
|
||||
This script is served straight from the project's own GitHub Pages site,
|
||||
so HTTPS always works. You may also see `https://openjarvis.ai/install.sh`
|
||||
referenced in older docs — that domain is community-operated and has had
|
||||
intermittent TLS issues ([#337](https://github.com/open-jarvis/OpenJarvis/issues/337)).
|
||||
The `open-jarvis.github.io` URL above is the canonical one.
|
||||
|
||||
About 3 minutes on a typical broadband connection. Type `jarvis` to start chatting.
|
||||
|
||||
## What the installer does
|
||||
|
||||
| Phase | Step | Where |
|
||||
|---|---|---|
|
||||
| Foreground | Install `uv` (Python package manager) | `~/.cargo/bin/` or `~/.local/bin/` |
|
||||
| Foreground | Clone OpenJarvis repo | `~/.openjarvis/src/` |
|
||||
| Foreground | Create Python 3.11 venv | `~/.openjarvis/.venv/` |
|
||||
| Foreground | `uv pip install -e .` (editable install) | venv |
|
||||
| Foreground | Install Ollama | system default |
|
||||
| Foreground | Start `ollama serve` | systemd-user / launchd / nohup |
|
||||
| Foreground | Pull `qwen3.5:2b` (~1.5 GB) | Ollama's model store |
|
||||
| Foreground | Write `config.toml` (auto-detected hardware + engine + model) | `~/.openjarvis/config.toml` |
|
||||
| Foreground | Symlink `jarvis` and `jarvis-uninstall` | `~/.local/bin/` |
|
||||
| Foreground | Add `~/.local/bin` to PATH if missing (with on-screen notice) | `~/.bashrc` or `~/.zshrc` |
|
||||
| Background | Install Rust toolchain via rustup | `~/.cargo/` |
|
||||
| Background | Build the maturin extension (memory + security features) | venv |
|
||||
| Background | Pull hardware-tier and tier+1 models | Ollama's model store |
|
||||
|
||||
## What the installer does NOT touch
|
||||
|
||||
- Your existing Python installations
|
||||
- Your `~/.bashrc` / `~/.zshrc` other than appending one PATH line (with on-screen notice)
|
||||
- Your existing Ollama models
|
||||
- Any other tool or dotfile
|
||||
|
||||
## Idempotent re-runs
|
||||
|
||||
Re-running the curl line is safe. The installer reads `~/.openjarvis/.state/install-state.json` and skips completed steps. If your venv got nuked, re-running heals it.
|
||||
|
||||
## Cloud quick-path
|
||||
|
||||
If any of these env vars are set when you install or run `jarvis init`, the installer/init proposes cloud as the default and writes the matching provider into `config.toml`:
|
||||
|
||||
- `OPENROUTER_API_KEY`
|
||||
- `ANTHROPIC_API_KEY`
|
||||
- `OPENAI_API_KEY`
|
||||
- `GOOGLE_API_KEY` (or `GEMINI_API_KEY`)
|
||||
|
||||
Local-first remains the default when no key is in env. Precedence is OpenRouter > Anthropic > OpenAI > Google.
|
||||
|
||||
## Flags
|
||||
|
||||
| Flag | Effect |
|
||||
|---|---|
|
||||
| `--minimal` | Skip the foreground model pull. First chat will need to wait for the bg pull to finish. |
|
||||
| `--no-bg-orchestrator` | Don't detach the background work pipeline. (Mostly for testing.) |
|
||||
| `--force` | Re-run all steps even if `install-state.json` says they're done. |
|
||||
|
||||
## Environment overrides
|
||||
|
||||
| Variable | Default | Purpose |
|
||||
|---|---|---|
|
||||
| `OPENJARVIS_HOME` | `$HOME/.openjarvis` | Install location. |
|
||||
| `OPENJARVIS_REPO_URL` | `https://github.com/open-jarvis/OpenJarvis.git` | Source repo for the clone step. |
|
||||
|
||||
## Uninstall
|
||||
|
||||
```bash
|
||||
jarvis-uninstall
|
||||
```
|
||||
|
||||
Removes `~/.openjarvis/`, `~/.local/bin/jarvis`, and `~/.local/bin/jarvis-uninstall`. Leaves Ollama, uv, and the Rust toolchain in place (they may be used by other tools); the script prints removal hints.
|
||||
|
||||
## Updating
|
||||
|
||||
```bash
|
||||
jarvis update
|
||||
```
|
||||
|
||||
Pulls the latest source, refreshes the editable install, and rebuilds the Rust extension in the background. Models are not touched.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### "command not found: jarvis"
|
||||
|
||||
`~/.local/bin` isn't on your PATH. Run `source ~/.bashrc` (or `~/.zshrc`) or open a new terminal.
|
||||
|
||||
### "memory features unavailable"
|
||||
|
||||
Rust extension hasn't finished building yet (or failed). Check status:
|
||||
|
||||
```bash
|
||||
jarvis doctor
|
||||
```
|
||||
|
||||
Manually retry:
|
||||
|
||||
```bash
|
||||
~/.openjarvis/.scripts/install-rust.sh && ~/.openjarvis/.scripts/build-extension.sh
|
||||
```
|
||||
|
||||
### A bigger model failed to download
|
||||
|
||||
Check status and retry:
|
||||
|
||||
```bash
|
||||
jarvis doctor
|
||||
~/.openjarvis/.scripts/pull-model.sh qwen3.5:9b
|
||||
```
|
||||
|
||||
### Behind a corporate proxy
|
||||
|
||||
Set `HTTPS_PROXY` and `CURL_CA_BUNDLE` in your environment before running the installer.
|
||||
@@ -41,7 +41,7 @@ If you prefer to run each step yourself:
|
||||
```bash
|
||||
git clone https://github.com/open-jarvis/OpenJarvis.git
|
||||
cd OpenJarvis
|
||||
uv sync --extra server
|
||||
uv sync --extra desktop
|
||||
uv run maturin develop -m rust/crates/openjarvis-python/Cargo.toml
|
||||
cd frontend && npm install && cd ..
|
||||
```
|
||||
@@ -94,11 +94,11 @@ cd OpenJarvis
|
||||
|
||||
| Platform | Download |
|
||||
|----------|----------|
|
||||
| macOS (Apple Silicon) | [:material-download: **OpenJarvis.dmg**](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-latest/OpenJarvis_0.1.0_aarch64.dmg) |
|
||||
| Windows (64-bit) | [:material-download: **OpenJarvis-setup.exe**](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-latest/OpenJarvis_0.1.0_x64-setup.exe) |
|
||||
| Linux (DEB) | [:material-download: **OpenJarvis.deb**](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-latest/OpenJarvis_0.1.0_amd64.deb) |
|
||||
| Linux (RPM) | [:material-download: **OpenJarvis.rpm**](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-latest/OpenJarvis-0.1.0-1.x86_64.rpm) |
|
||||
| Linux (AppImage) | [:material-download: **OpenJarvis.AppImage**](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-latest/OpenJarvis_0.1.0_amd64.AppImage) |
|
||||
| macOS (Universal) | [:material-download: **OpenJarvis.dmg**](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-v1.0.2/OpenJarvis_1.0.1_universal.dmg) |
|
||||
| Windows (64-bit) | [:material-download: **OpenJarvis-setup.exe**](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-v1.0.2/OpenJarvis_1.0.1_x64-setup.exe) |
|
||||
| Linux (DEB) | [:material-download: **OpenJarvis.deb**](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-v1.0.2/OpenJarvis_1.0.1_amd64.deb) |
|
||||
| Linux (RPM) | [:material-download: **OpenJarvis.rpm**](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-v1.0.2/OpenJarvis-1.0.1-1.x86_64.rpm) |
|
||||
| Linux (AppImage) | [:material-download: **OpenJarvis.AppImage**](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-v1.0.2/OpenJarvis_1.0.1_amd64.AppImage) |
|
||||
|
||||
The app connects to `http://localhost:8000` automatically.
|
||||
|
||||
@@ -238,7 +238,7 @@ See the [Python SDK guide](../user-guide/python-sdk.md) for the full API referen
|
||||
|
||||
| Requirement | Version | Install | Notes |
|
||||
|-------------|---------|---------|-------|
|
||||
| Python | 3.10+ | [python.org](https://www.python.org/downloads/) | Required |
|
||||
| Python | 3.10–3.13 | [python.org](https://www.python.org/downloads/) | Required. 3.14+ not yet supported (a core dependency lacks 3.14 wheels). |
|
||||
| uv | latest | `curl -LsSf https://astral.sh/uv/install.sh \| sh` or `brew install uv` (macOS) | Python package & project manager |
|
||||
| Git | any | [git-scm.com](https://git-scm.com/) or `brew install git` (macOS) | Required |
|
||||
| Rust | stable | `curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs \| sh` | Required for the Rust extension |
|
||||
@@ -278,6 +278,7 @@ OpenJarvis uses optional extras to keep the base installation lightweight.
|
||||
|
||||
| Extra | Install Command | Description |
|
||||
|-------|----------------|-------------|
|
||||
| `desktop` | `uv sync --extra desktop` | Desktop/API server plus local speech input |
|
||||
| `server` | `uv sync --extra server` | OpenAI-compatible API server (`jarvis serve`) |
|
||||
| `dev` | `uv sync --extra dev` | Development and testing tools |
|
||||
| `docs` | `uv sync --extra docs` | Documentation build tools |
|
||||
@@ -285,7 +286,7 @@ OpenJarvis uses optional extras to keep the base installation lightweight.
|
||||
Combine extras:
|
||||
|
||||
```bash
|
||||
uv sync --extra server --extra memory-faiss --extra inference-cloud
|
||||
uv sync --extra desktop --extra memory-faiss --extra inference-cloud
|
||||
```
|
||||
|
||||
## Setting Up an Inference Backend
|
||||
|
||||
@@ -0,0 +1,30 @@
|
||||
# Linux Install
|
||||
|
||||
```bash
|
||||
curl -fsSL https://open-jarvis.github.io/OpenJarvis/install.sh | bash
|
||||
```
|
||||
|
||||
Tested on: Ubuntu 22.04 / 24.04, Fedora 40, Debian 12, Arch.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
Most distros ship `git` and `curl`. If yours doesn't:
|
||||
|
||||
```bash
|
||||
# Debian / Ubuntu
|
||||
sudo apt install git curl
|
||||
|
||||
# Fedora / RHEL
|
||||
sudo dnf install git curl
|
||||
|
||||
# Arch
|
||||
sudo pacman -S git curl
|
||||
```
|
||||
|
||||
## NVIDIA / AMD GPU
|
||||
|
||||
The installer auto-detects via `nvidia-smi` / `rocm-smi`. Datacenter cards (A100, H100, MI300+) get vLLM as the recommended engine; consumer cards get Ollama (NVIDIA) or Lemonade (AMD).
|
||||
|
||||
## See also
|
||||
|
||||
- [Full installer reference](install.md)
|
||||
+12
-407
@@ -1,421 +1,26 @@
|
||||
---
|
||||
title: macOS Installation Guide
|
||||
description: Complete step-by-step guide to installing OpenJarvis on macOS with llama.cpp, including common pitfalls and fixes
|
||||
search:
|
||||
boost: 2
|
||||
---
|
||||
|
||||
# macOS Installation Guide
|
||||
|
||||
This guide walks through a complete OpenJarvis installation on macOS using **llama.cpp** as
|
||||
the inference engine. It covers every step from scratch — including pitfalls not documented
|
||||
elsewhere — and is suitable for both Apple Silicon and Intel Macs.
|
||||
|
||||
!!! tip "Prefer Ollama?"
|
||||
If you want the fastest possible setup, use [Ollama](installation.md#ollama-recommended)
|
||||
instead. This guide is for users who want to run GGUF models directly with llama.cpp,
|
||||
or who want a deeper understanding of the full stack.
|
||||
|
||||
---
|
||||
|
||||
## What You'll Install
|
||||
|
||||
| Tool | Purpose |
|
||||
|------|---------|
|
||||
| Homebrew | macOS package manager — installs everything else |
|
||||
| uv | Python version and dependency manager |
|
||||
| Git | Clones the OpenJarvis repo |
|
||||
| Node.js | Required for the browser UI |
|
||||
| Rust | Compiles the OpenJarvis security and memory extension |
|
||||
| llama.cpp | Local inference engine that runs GGUF model files |
|
||||
| OpenJarvis | The framework itself |
|
||||
| A GGUF model | The actual AI model (downloaded separately) |
|
||||
|
||||
---
|
||||
|
||||
## Step-by-Step Installation
|
||||
|
||||
### Step 1 — Install Homebrew
|
||||
|
||||
Homebrew is the standard macOS package manager. Everything else in this guide is installed
|
||||
through it.
|
||||
# macOS Install
|
||||
|
||||
```bash
|
||||
/bin/bash -c "$(curl -fsSL https://raw.githubusercontent.com/Homebrew/install/HEAD/install.sh)"
|
||||
curl -fsSL https://open-jarvis.github.io/OpenJarvis/install.sh | bash
|
||||
```
|
||||
|
||||
If you already have Homebrew, skip this step.
|
||||
Works on Intel and Apple Silicon. The installer auto-detects your CPU/GPU.
|
||||
|
||||
---
|
||||
## Prerequisites
|
||||
|
||||
### Step 2 — Install uv
|
||||
If you've never run `git` or `curl` on this Mac, macOS will prompt you to install the Xcode Command Line Tools the first time you run them. Accept the prompt; that gives you both.
|
||||
|
||||
`uv` replaces pip, virtualenv, and pyenv in one tool. OpenJarvis uses it to manage Python
|
||||
versions, virtual environments, and project dependencies.
|
||||
If you'd rather pre-install:
|
||||
|
||||
```bash
|
||||
brew install uv
|
||||
xcode-select --install
|
||||
```
|
||||
|
||||
---
|
||||
## Apple Silicon notes
|
||||
|
||||
### Step 3 — Install Git
|
||||
- The installer picks `mlx` as the recommended engine via the standard hardware-detect path, but the foreground default is still Ollama for compatibility. Switch later with `jarvis init --force` and pick `mlx` if you've installed `mlx-lm`.
|
||||
- Unified memory is reported as "VRAM" by the installer — that's intentional; on Apple Silicon, system RAM is what GPU-accelerated models can use.
|
||||
|
||||
Git is used to clone the OpenJarvis source code. It may already be present if you have
|
||||
Xcode Command Line Tools installed.
|
||||
## See also
|
||||
|
||||
```bash
|
||||
brew install git
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Step 4 — Install Node.js
|
||||
|
||||
Node.js is required to build and run the browser frontend. Without it you can still use
|
||||
the CLI, but not the web UI.
|
||||
|
||||
```bash
|
||||
brew install node
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Step 5 — Install Rust
|
||||
|
||||
OpenJarvis includes a Rust extension that provides security scanning, memory indexing,
|
||||
rate limiting, and tool execution. It must be compiled from source.
|
||||
|
||||
```bash
|
||||
curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh
|
||||
```
|
||||
|
||||
After the installer finishes, reload your shell so `rustc` is available:
|
||||
|
||||
```bash
|
||||
source "$HOME/.cargo/env"
|
||||
```
|
||||
|
||||
Verify:
|
||||
|
||||
```bash
|
||||
rustc --version
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Step 6 — Install llama.cpp
|
||||
|
||||
llama.cpp is the inference engine that loads and runs GGUF model files. It is not a model
|
||||
itself — think of it as a media player and the `.gguf` file as the content.
|
||||
|
||||
```bash
|
||||
brew install llama.cpp
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Step 7 — Clone the OpenJarvis repo
|
||||
|
||||
Run this from your home directory or any neutral parent folder.
|
||||
|
||||
```bash
|
||||
cd ~
|
||||
git clone https://github.com/open-jarvis/OpenJarvis.git
|
||||
cd OpenJarvis
|
||||
```
|
||||
|
||||
!!! warning "Do not clone from inside an existing OpenJarvis folder"
|
||||
A common mistake is running `git clone` while already inside the repo, creating deeply
|
||||
nested duplicates (`OpenJarvis/OpenJarvis/OpenJarvis`). Always clone from `~` or a
|
||||
neutral parent directory.
|
||||
|
||||
---
|
||||
|
||||
### Step 8 — Pin Python to 3.12
|
||||
|
||||
!!! warning "Critical step — do not skip"
|
||||
OpenJarvis requires Python 3.10–3.13. Its Rust extension uses PyO3, which does not yet
|
||||
support Python 3.14. If `uv` has Python 3.14 available, it will use it by default,
|
||||
causing the Rust extension build to fail silently and resulting in ~250 test failures
|
||||
with `ModuleNotFoundError: No module named 'openjarvis_rust'`.
|
||||
|
||||
Pin the project to Python 3.12:
|
||||
|
||||
```bash
|
||||
echo "3.12" > .python-version
|
||||
uv python install 3.12
|
||||
rm -rf .venv
|
||||
uv venv
|
||||
```
|
||||
|
||||
**Restart your terminal**, then verify:
|
||||
|
||||
```bash
|
||||
uv run python --version
|
||||
# Must show: Python 3.12.x
|
||||
```
|
||||
|
||||
!!! tip "Why restart the terminal?"
|
||||
Without restarting, the shell may still reference the old virtual environment. This is
|
||||
the most common reason the version pin appears not to work.
|
||||
|
||||
---
|
||||
|
||||
### Step 9 — Install Python dependencies
|
||||
|
||||
```bash
|
||||
uv sync --extra dev --extra server
|
||||
```
|
||||
|
||||
The `--extra server` flag adds the FastAPI backend required for the browser UI.
|
||||
|
||||
---
|
||||
|
||||
### Step 10 — Build the Rust extension
|
||||
|
||||
This compiles the Rust extension and installs it into the virtual environment. It provides
|
||||
security scanning, memory indexing, MCP tool execution, and rate limiting. This step takes
|
||||
a few minutes on first run.
|
||||
|
||||
```bash
|
||||
uv run maturin develop -m rust/crates/openjarvis-python/Cargo.toml
|
||||
```
|
||||
|
||||
Verify it built correctly:
|
||||
|
||||
```bash
|
||||
uv run python -c "import openjarvis_rust; print('Rust extension OK')"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Step 11 — Install frontend dependencies
|
||||
|
||||
```bash
|
||||
cd frontend && npm install && cd ..
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Step 12 — Download a model
|
||||
|
||||
OpenJarvis needs a GGUF model file to run inference. First install the Hugging Face CLI,
|
||||
then download your chosen model.
|
||||
|
||||
```bash
|
||||
uv tool install huggingface_hub
|
||||
```
|
||||
|
||||
!!! note "The CLI command is `hf`, not `huggingface-cli`"
|
||||
When installed via `uv tool`, the Hugging Face CLI is invoked as `hf`.
|
||||
|
||||
=== "Qwen3 4B (~2.5 GB)"
|
||||
|
||||
Faster, lower RAM requirement. Good for most everyday tasks.
|
||||
|
||||
```bash
|
||||
hf download bartowski/Qwen_Qwen3-4B-GGUF \
|
||||
--include "Qwen_Qwen3-4B-Q4_K_M.gguf" \
|
||||
--local-dir ~/models
|
||||
```
|
||||
|
||||
=== "Qwen3 8B (~4.7 GB)"
|
||||
|
||||
Better reasoning and instruction following. Requires more RAM.
|
||||
|
||||
```bash
|
||||
hf download bartowski/Qwen_Qwen3-8B-GGUF \
|
||||
--include "Qwen_Qwen3-8B-Q4_K_M.gguf" \
|
||||
--local-dir ~/models
|
||||
```
|
||||
|
||||
!!! warning "Use the `Qwen_` prefix"
|
||||
bartowski's Qwen3 repos use the `Qwen_` prefix (e.g. `Qwen_Qwen3-4B-GGUF`). Using
|
||||
the shorter name without the prefix returns a "repository not found" error.
|
||||
|
||||
!!! tip "Apple Silicon vs Intel"
|
||||
On Apple Silicon, both models benefit from Metal GPU acceleration when using the MLX
|
||||
engine. On Intel, inference runs on CPU — the 4B model is recommended for speed.
|
||||
|
||||
---
|
||||
|
||||
### Step 13 — Configure OpenJarvis
|
||||
|
||||
Run the init command to detect your hardware and generate a config file:
|
||||
|
||||
```bash
|
||||
uv run jarvis init
|
||||
```
|
||||
|
||||
Then open the config and set the default model to match the filename you downloaded:
|
||||
|
||||
```bash
|
||||
nano ~/.openjarvis/config.toml
|
||||
```
|
||||
|
||||
Find the `default_model` line and update it, for example:
|
||||
|
||||
```toml
|
||||
default_model = "Qwen_Qwen3-4B-Q4_K_M.gguf"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Step 14 — Verify the installation
|
||||
|
||||
```bash
|
||||
uv run jarvis doctor
|
||||
```
|
||||
|
||||
A healthy setup looks like this:
|
||||
|
||||
```
|
||||
✓ Python version 3.12.x
|
||||
✓ Config file ~/.openjarvis/config.toml
|
||||
✓ Config parsing Config loaded successfully
|
||||
✓ Engine: llamacpp Reachable
|
||||
✓ Models: llamacpp Qwen_Qwen3-4B-Q4_K_M.gguf
|
||||
✓ Default model Qwen_Qwen3-4B-Q4_K_M.gguf (on llamacpp)
|
||||
```
|
||||
|
||||
!!! note "Warnings for other engines are normal"
|
||||
The `!` warnings for engines like `ollama`, `vllm`, and `lmstudio` simply mean those
|
||||
backends are not running. You only need `llamacpp` to be reachable.
|
||||
|
||||
---
|
||||
|
||||
## Running OpenJarvis
|
||||
|
||||
### CLI
|
||||
|
||||
Start llama-server in one terminal, then run queries in another:
|
||||
|
||||
```bash
|
||||
# Terminal 1 — start the inference engine
|
||||
llama-server -m ~/models/Qwen_Qwen3-4B-Q4_K_M.gguf -c 4096 -t 8
|
||||
|
||||
# Terminal 2 — ask a question
|
||||
cd ~/OpenJarvis
|
||||
uv run jarvis ask "What is the capital of France?"
|
||||
```
|
||||
|
||||
### Browser UI
|
||||
|
||||
```bash
|
||||
# Terminal 1 — inference engine
|
||||
llama-server -m ~/models/Qwen_Qwen3-4B-Q4_K_M.gguf -c 4096 -t 8
|
||||
|
||||
# Terminal 2 — backend
|
||||
cd ~/OpenJarvis && uv run jarvis serve --port 8000
|
||||
|
||||
# Terminal 3 — frontend
|
||||
cd ~/OpenJarvis/frontend && npm run dev
|
||||
```
|
||||
|
||||
Then open [http://localhost:5173](http://localhost:5173).
|
||||
|
||||
### Skip typing `uv run` every time
|
||||
|
||||
Activate the virtual environment for your current terminal session:
|
||||
|
||||
```bash
|
||||
source ~/OpenJarvis/.venv/bin/activate
|
||||
```
|
||||
|
||||
Your prompt will show `(openjarvis)` when active, and you can type `jarvis ask "..."` directly.
|
||||
|
||||
---
|
||||
|
||||
## Performance Tips
|
||||
|
||||
These tips apply when using llama.cpp for CPU inference.
|
||||
|
||||
| Flag | Effect |
|
||||
|------|--------|
|
||||
| `-c 4096` | Reduces context window from the 32,768 default, freeing RAM for faster inference |
|
||||
| `-t 8` | Uses all available CPU threads (default is only 4) — adjust to your machine's thread count |
|
||||
| `Q4_K_M` quantization | Best balance of size, speed, and quality for CPU inference |
|
||||
|
||||
On Apple Silicon, switching to the [MLX engine](../architecture/engine.md) gives
|
||||
significantly better performance than llama.cpp for most models.
|
||||
|
||||
---
|
||||
|
||||
## Common Errors
|
||||
|
||||
### `No such file or directory` when loading model
|
||||
|
||||
The path `path/to/model.gguf` in examples is a placeholder. Replace it with your actual
|
||||
model path, e.g.:
|
||||
|
||||
```bash
|
||||
llama-server -m ~/models/Qwen_Qwen3-4B-Q4_K_M.gguf
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### `No module named 'openjarvis_rust'`
|
||||
|
||||
The Rust extension did not build correctly, or was built against the wrong Python version.
|
||||
|
||||
1. Confirm Python 3.12 is active: `uv run python --version`
|
||||
2. Rebuild: `uv run maturin develop -m rust/crates/openjarvis-python/Cargo.toml`
|
||||
|
||||
If the version shows 3.14, go back to [Step 8](#step-8--pin-python-to-312).
|
||||
|
||||
---
|
||||
|
||||
### `PyO3 version error — Python 3.14 too new`
|
||||
|
||||
```
|
||||
error: the configured Python interpreter version (3.14) is newer than
|
||||
PyO3's maximum supported version (3.13)
|
||||
```
|
||||
|
||||
PyO3 0.23.5 supports Python up to 3.13. Follow [Step 8](#step-8--pin-python-to-312) to
|
||||
pin to 3.12, then delete `.venv`, recreate it, and restart your terminal before retrying.
|
||||
|
||||
---
|
||||
|
||||
### `Repository not found` when downloading model
|
||||
|
||||
bartowski's Qwen3 repos use the `Qwen_` prefix. Use:
|
||||
|
||||
```
|
||||
bartowski/Qwen_Qwen3-4B-GGUF ✓
|
||||
bartowski/Qwen3-4B-GGUF ✗
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### `No inference engine available`
|
||||
|
||||
llama-server is not running. Start it in a separate terminal before running any `jarvis`
|
||||
commands, and wait until you see `model loaded` in the output.
|
||||
|
||||
---
|
||||
|
||||
### Python version still shows 3.14 after recreating the venv
|
||||
|
||||
Close the terminal completely and reopen it. The old venv path is cached in the shell
|
||||
environment and persists across commands until the session ends.
|
||||
|
||||
---
|
||||
|
||||
### `zsh: command not found: huggingface-cli`
|
||||
|
||||
When installed via `uv tool`, the CLI is invoked as `hf`, not `huggingface-cli`:
|
||||
|
||||
```bash
|
||||
hf download ... # ✓
|
||||
huggingface-cli download ... # ✗
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Next Steps
|
||||
|
||||
- [Quick Start](quickstart.md) — Run your first query and explore agents and tools
|
||||
- [Configuration](configuration.md) — Customize engine hosts, model routing, memory, and more
|
||||
- [Architecture](../architecture/overview.md) — Understand how OpenJarvis is structured
|
||||
- [Full installer reference](install.md)
|
||||
|
||||
@@ -7,6 +7,11 @@ search:
|
||||
|
||||
# Quick Start
|
||||
|
||||
!!! tip "Running `jarvis` commands"
|
||||
Every `jarvis ...` example below assumes you have either activated the project venv
|
||||
(`source .venv/bin/activate`) or are prefixing each command with `uv run`. A bare
|
||||
`jarvis init --preset ...` from a fresh clone will fail with `command not found`.
|
||||
|
||||
## What You Can Build
|
||||
|
||||
OpenJarvis is a modular AI assistant framework. Here's what developers build with it:
|
||||
@@ -14,7 +19,7 @@ OpenJarvis is a modular AI assistant framework. Here's what developers build wit
|
||||
=== "Chat with Any Model"
|
||||
|
||||
```bash
|
||||
jarvis ask "Explain quantum entanglement" -m qwen3:8b
|
||||
jarvis ask "Explain quantum entanglement" -m qwen3.5:4b # use qwen3.5:9b or larger on GPU
|
||||
```
|
||||
|
||||
=== "Agent + Tools"
|
||||
@@ -30,6 +35,13 @@ OpenJarvis is a modular AI assistant framework. Here's what developers build wit
|
||||
jarvis ask "How do I configure the engine?"
|
||||
```
|
||||
|
||||
!!! warning "Requires the Rust extension"
|
||||
`jarvis memory index` and `jarvis memory search` import `openjarvis_rust`. If you
|
||||
skipped the `uv run maturin develop -m rust/crates/openjarvis-python/Cargo.toml`
|
||||
step in [Installation](installation.md), these commands fail with
|
||||
`ModuleNotFoundError: No module named 'openjarvis_rust'`. Build the extension
|
||||
once and any preset (including `deep-research`) will work.
|
||||
|
||||
=== "5-Line Python SDK"
|
||||
|
||||
```python
|
||||
|
||||
@@ -0,0 +1,83 @@
|
||||
# Native Windows (advanced)
|
||||
|
||||
Phase-1 of the native-Windows-support RFC (#298). Mirrors the Linux
|
||||
(systemd) and macOS (launchd) deployments — but for PowerShell, without
|
||||
WSL2 or Docker. Choose this over [WSL2](wsl2.md) only if you want to
|
||||
avoid a Linux VM; WSL2 remains the smoother experience for most users.
|
||||
|
||||
## What you get
|
||||
|
||||
- A PowerShell installer that probes prerequisites, installs `uv`,
|
||||
clones the repo, and runs `uv sync --extra desktop --group desktop-native`.
|
||||
- An optional Windows scheduled-task service equivalent to the systemd
|
||||
unit and launchd plist.
|
||||
- Loopback default — the service binds `127.0.0.1` so no API key is
|
||||
required.
|
||||
|
||||
## What you need
|
||||
|
||||
- Windows 10 1809+ or Windows 11.
|
||||
- Python 3.10 – 3.13 (Python 3.14 has no numpy Windows wheels yet —
|
||||
see [#432](https://github.com/open-jarvis/OpenJarvis/issues/432)).
|
||||
- `git` on PATH.
|
||||
- ~5 GB free disk on `%LOCALAPPDATA%`.
|
||||
|
||||
## Install
|
||||
|
||||
In any PowerShell:
|
||||
|
||||
```powershell
|
||||
irm https://open-jarvis.github.io/OpenJarvis/install.ps1 | iex
|
||||
```
|
||||
|
||||
The installer will:
|
||||
|
||||
1. Refuse non-Windows hosts and old Windows builds.
|
||||
2. Confirm Python 3.10 – 3.13.
|
||||
3. Confirm `git`.
|
||||
4. Install `uv` if absent (via the official `astral.sh/uv` PowerShell
|
||||
installer).
|
||||
5. Clone the repo to `%LOCALAPPDATA%\OpenJarvis\src`.
|
||||
6. Run `uv sync --extra desktop --group desktop-native`.
|
||||
7. Prompt to register the scheduled-task service (skip with
|
||||
`-SkipService`).
|
||||
|
||||
## Run it
|
||||
|
||||
```powershell
|
||||
cd "$env:LOCALAPPDATA\OpenJarvis\src"
|
||||
uv run jarvis serve
|
||||
```
|
||||
|
||||
Open `http://127.0.0.1:8000/health` to verify.
|
||||
|
||||
## Scheduled-task service
|
||||
|
||||
If you skipped the prompt during install, register the auto-start task
|
||||
manually:
|
||||
|
||||
```powershell
|
||||
$srv = "$env:LOCALAPPDATA\OpenJarvis\src\deploy\windows\jarvis-service.ps1"
|
||||
powershell -ExecutionPolicy Bypass -File $srv install
|
||||
```
|
||||
|
||||
State:
|
||||
|
||||
```powershell
|
||||
powershell -ExecutionPolicy Bypass -File $srv status
|
||||
```
|
||||
|
||||
Remove:
|
||||
|
||||
```powershell
|
||||
powershell -ExecutionPolicy Bypass -File $srv uninstall
|
||||
```
|
||||
|
||||
See [`deploy/windows/README.md`](https://github.com/open-jarvis/OpenJarvis/blob/main/deploy/windows/README.md)
|
||||
for the LAN-exposed configuration and the parity table against
|
||||
systemd / launchd.
|
||||
|
||||
## See also
|
||||
|
||||
- [WSL2 install](wsl2.md) — the recommended Windows path.
|
||||
- [Full installer reference](install.md).
|
||||
@@ -0,0 +1,34 @@
|
||||
# WSL2 Install
|
||||
|
||||
OpenJarvis on Windows installs two ways: **WSL2** (this page — the
|
||||
recommended path; identical to native Linux) or **[native Windows
|
||||
(advanced)](windows-native.md)** (Phase-1; PowerShell installer, no
|
||||
WSL2 / no Docker). Pick WSL2 for the smoothest experience.
|
||||
|
||||
## One-time WSL setup
|
||||
|
||||
In an admin PowerShell:
|
||||
|
||||
```powershell
|
||||
wsl --install
|
||||
```
|
||||
|
||||
Then open the Ubuntu (or Debian) shell that gets installed.
|
||||
|
||||
## Install OpenJarvis
|
||||
|
||||
```bash
|
||||
curl -fsSL https://open-jarvis.github.io/OpenJarvis/install.sh | bash
|
||||
```
|
||||
|
||||
About 3 minutes. Type `jarvis` to start.
|
||||
|
||||
## WSL-specific notes
|
||||
|
||||
- The installer detects WSL via `/proc/sys/kernel/osrelease` and uses `nohup ollama serve &` instead of systemd to start the Ollama daemon (WSL2 doesn't ship systemd by default).
|
||||
- The first time you run `jarvis`, the WSL kernel may show a "process running in background" notification — that's the bg-orchestrator detaching. It's expected.
|
||||
- Models are stored in WSL's filesystem (`~/.openjarvis/`), not your Windows drive. To free up space later: `jarvis-uninstall` removes everything.
|
||||
|
||||
## See also
|
||||
|
||||
- [Full installer reference](install.md)
|
||||
+33
-14
@@ -14,13 +14,25 @@ OpenJarvis is a research framework for composable, on-device AI systems.
|
||||
Build personal AI that runs on your hardware. Cloud APIs are optional.
|
||||
</p>
|
||||
|
||||
<div class="grid cards" markdown>
|
||||
|
||||
- :material-image-multiple:{ .lg .middle } **See what people use it for**
|
||||
|
||||
---
|
||||
|
||||
A gallery of real setups — morning briefs that summarize your overnight Slack and email, a Discord companion that knows your calendar, a code reviewer that works at 30,000 feet. Outcome-first, with links to the docs that explain how to build each one.
|
||||
|
||||
[:octicons-arrow-right-24: Browse the Showcase](showcase/index.md)
|
||||
|
||||
</div>
|
||||
|
||||
---
|
||||
|
||||
## Why OpenJarvis?
|
||||
|
||||
Personal AI agents are exploding in popularity, but nearly all of them still route intelligence through cloud APIs. Your "personal" AI continues to depend on someone else's server. At the same time, our [Intelligence Per Watt](https://www.intelligence-per-watt.ai/) research showed that local language models already handle 88.7% of single-turn chat and reasoning queries, with intelligence efficiency improving 5.3× from 2023 to 2025. The models and hardware are increasingly ready. What has been missing is the software stack to make local-first personal AI practical.
|
||||
|
||||
OpenJarvis is that stack. It is an opinionated framework for local-first personal AI, built around three core ideas: shared primitives for building on-device agents; evaluations that treat energy, FLOPs, latency, and dollar cost as first-class constraints alongside accuracy; and a learning loop that improves models using local trace data. The goal is simple: make it possible to build personal AI agents that run locally by default, calling the cloud only when truly necessary. OpenJarvis aims to be both a research platform and a production foundation for local AI, in the spirit of PyTorch.
|
||||
OpenJarvis is that stack. It is a framework for local-first personal AI, built around three core ideas: shared primitives for building on-device agents; evaluations that treat energy, FLOPs, latency, and dollar cost as first-class constraints alongside accuracy; and a learning loop that improves models using local trace data. The goal is simple: make it possible to build personal AI agents that run locally by default, calling the cloud only when truly necessary. OpenJarvis aims to be both a research platform and a production foundation for local AI, in the spirit of PyTorch.
|
||||
|
||||
---
|
||||
|
||||
@@ -54,13 +66,15 @@ OpenJarvis is that stack. It is an opinionated framework for local-first persona
|
||||
|
||||
**Step 2.** Download and open the desktop app:
|
||||
|
||||
[Download for macOS](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-latest/OpenJarvis_0.1.0_universal.dmg){ .md-button .md-button--primary }
|
||||
[Download for macOS](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-v1.0.2/OpenJarvis_1.0.1_universal.dmg){ .md-button .md-button--primary }
|
||||
|
||||
Also available for [Windows](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-latest/OpenJarvis_0.1.0_x64-setup.exe), [Linux (DEB)](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-latest/OpenJarvis_0.1.0_amd64.deb), and [Linux (RPM)](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-latest/OpenJarvis-0.1.0-1.x86_64.rpm). See the [Downloads](downloads.md) page for details.
|
||||
Also available for [Windows](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-v1.0.2/OpenJarvis_1.0.1_x64-setup.exe), [Linux (DEB)](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-v1.0.2/OpenJarvis_1.0.1_amd64.deb), and [Linux (RPM)](https://github.com/open-jarvis/OpenJarvis/releases/download/desktop-v1.0.2/OpenJarvis-1.0.1-1.x86_64.rpm). See the [Downloads](downloads.md) page for details.
|
||||
|
||||
The app connects to `http://localhost:8000` automatically.
|
||||
|
||||
!!! warning "macOS: run `xattr -cr /Applications/OpenJarvis.app` if the app shows as \"damaged\"."
|
||||
!!! warning "macOS first launch"
|
||||
|
||||
Run `xattr -cr /Applications/OpenJarvis.app` if the app shows as "damaged".
|
||||
|
||||
=== "Python SDK"
|
||||
|
||||
@@ -104,9 +118,9 @@ OpenJarvis is that stack. It is an opinionated framework for local-first persona
|
||||
OpenJarvis is built around five composable layers. Each has a clean interface and can be swapped independently.
|
||||
|
||||
1. **Intelligence** — Pick a model, or let OpenJarvis pick one for your hardware. Manages the full catalog of local models across providers.
|
||||
2. **Agents** — Multi-step reasoning with tool use. Seven built-in agent types from simple chat to orchestrated workflows.
|
||||
3. **Tools** — Web search, calculator, file I/O, code interpreter, retrieval, and any external MCP server.
|
||||
4. **Engine** — The inference runtime: [Ollama](https://ollama.com), [vLLM](https://github.com/vllm-project/vllm), [SGLang](https://github.com/sgl-project/sglang), [llama.cpp](https://github.com/ggerganov/llama.cpp), cloud APIs, and more. Auto-detects your hardware and recommends the best fit.
|
||||
2. **Engine** — The inference runtime: [Ollama](https://ollama.com), [vLLM](https://github.com/vllm-project/vllm), [SGLang](https://github.com/sgl-project/sglang), [llama.cpp](https://github.com/ggerganov/llama.cpp), cloud APIs, and more. Auto-detects your hardware and recommends the best fit.
|
||||
3. **Agents** — Multi-step reasoning with tool use. Eight built-in agent types from simple chat to orchestrated workflows.
|
||||
4. **Tools & Memory** — Web search, calculator, file I/O, code interpreter, retrieval, persistent local state, and any external MCP server.
|
||||
5. **Learning** — Your AI gets better over time. Every interaction generates traces that drive automatic improvements to model weights, prompts, and agent behavior.
|
||||
|
||||
---
|
||||
@@ -169,7 +183,7 @@ OpenJarvis is built around five composable layers. Each has a clean interface an
|
||||
|
||||
---
|
||||
|
||||
CLI, Python SDK, and guides for [Morning Digest](user-guide/morning-digest.md), [Deep Research](user-guide/deep-research.md), [Code Assistant](user-guide/code-assistant.md), [Scheduled Monitor](user-guide/scheduled-monitor.md), [Simple Chat](user-guide/chat-simple.md), agents, memory, tools, and telemetry.
|
||||
CLI, Python SDK, and guides for [Morning Digest](user-guide/morning-digest.md), [Deep Research](user-guide/deep-research.md), [Code Assistant](user-guide/code-assistant.md), [Scheduled Monitor](user-guide/scheduled-monitor.md), [Simple Chat](user-guide/chat-simple.md), [Evaluations](user-guide/evaluations.md), agents, memory, tools, and telemetry.
|
||||
|
||||
- **[Architecture](architecture/overview.md)**
|
||||
|
||||
@@ -201,16 +215,19 @@ OpenJarvis is built around five composable layers. Each has a clean interface an
|
||||
|
||||
OpenJarvis is part of [Intelligence Per Watt](https://www.intelligence-per-watt.ai/), a research initiative studying the efficiency of on-device AI systems. Developed at [Hazy Research](https://hazyresearch.stanford.edu/) and the [Scaling Intelligence Lab](https://scalingintelligence.stanford.edu/) at [Stanford SAIL](https://ai.stanford.edu/).
|
||||
|
||||
Read the [blog post](https://scalingintelligence.stanford.edu/blogs/openjarvis/) for the full research motivation, architecture details, and experimental results.
|
||||
Read the [blog post](https://openjarvis.stanford.edu/) for the full research motivation, architecture details, and experimental results.
|
||||
|
||||
## Citation
|
||||
|
||||
```bibtex
|
||||
@misc{saadfalcon2026openjarvis,
|
||||
title={OpenJarvis: Personal AI, On Personal Devices},
|
||||
author={Jon Saad-Falcon and Avanika Narayan and Herumb Shandilya and Hakki Orhun Akengin and Robby Manihani and Gabriel Bo and John Hennessy and Christopher R\'{e} and Azalia Mirhoseini},
|
||||
year={2026},
|
||||
howpublished={\url{https://scalingintelligence.stanford.edu/blogs/openjarvis/}},
|
||||
@misc{saadfalcon2026openjarvispersonalaipersonal,
|
||||
title={OpenJarvis: Personal AI, On Personal Devices},
|
||||
author={Jon Saad-Falcon and Avanika Narayan and Robby Manihani and Tanvir Bhathal and Herumb Shandilya and Hakki Orhun Akengin and Gabriel Bo and Andrew Park and Matthew Hart and Caia Costello and Chuan Li and Christopher Ré and Azalia Mirhoseini},
|
||||
year={2026},
|
||||
eprint={2605.17172},
|
||||
archivePrefix={arXiv},
|
||||
primaryClass={cs.LG},
|
||||
url={https://arxiv.org/abs/2605.17172},
|
||||
}
|
||||
```
|
||||
|
||||
@@ -225,3 +242,5 @@ Read the [blog post](https://scalingintelligence.stanford.edu/blogs/openjarvis/)
|
||||
<a href="https://research.ibm.com/">IBM Research</a> •
|
||||
<a href="https://hai.stanford.edu/">Stanford HAI</a>
|
||||
</p>
|
||||
|
||||
Follow [@OpenJarvisAI](https://x.com/OpenJarvisAI) on X for updates.
|
||||
|
||||
@@ -0,0 +1,12 @@
|
||||
// Public Supabase config for the savings leaderboard.
|
||||
//
|
||||
// This file is loaded *before* leaderboard.js and supplies the anon key it
|
||||
// reads from `window.OPENJARVIS_SUPABASE_ANON_KEY`. The key is injected at
|
||||
// docs-build time from the VITE_SUPABASE_ANON_KEY repo secret (see
|
||||
// .github/workflows/docs.yml). It is intentionally empty here so that local
|
||||
// `mkdocs build` and fork pull requests — which have no secret — render the
|
||||
// graceful "Leaderboard not configured yet" message instead of failing.
|
||||
//
|
||||
// The anon key is public by design: Supabase Row-Level Security protects the
|
||||
// data, so shipping it in the public docs bundle is expected.
|
||||
window.OPENJARVIS_SUPABASE_ANON_KEY = "";
|
||||
@@ -1,9 +1,9 @@
|
||||
(function () {
|
||||
"use strict";
|
||||
|
||||
var SUPABASE_URL = "https://mtbtgpwzrbostweaanpr.supabase.co";
|
||||
var SUPABASE_ANON_KEY =
|
||||
"eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9.eyJpc3MiOiJzdXBhYmFzZSIsInJlZiI6Im10YnRncHd6cmJvc3R3ZWFhbnByIiwicm9sZSI6ImFub24iLCJpYXQiOjE3NzMxODk0OTQsImV4cCI6MjA4ODc2NTQ5NH0._xMlqCfljtXpwPj54H-ghxfLFO-jiq4W2WhpU8vVL1c";
|
||||
var SUPABASE_URL =
|
||||
window.OPENJARVIS_SUPABASE_URL || "https://mtbtgpwzrbostweaanpr.supabase.co";
|
||||
var SUPABASE_ANON_KEY = window.OPENJARVIS_SUPABASE_ANON_KEY || "";
|
||||
|
||||
var PAGE_SIZE = 50;
|
||||
var allRows = [];
|
||||
@@ -12,8 +12,14 @@
|
||||
// Outlier detection — hide entries with values that are physically
|
||||
// implausible relative to their token count. Thresholds are ~1000x
|
||||
// above legitimate per-token values to avoid false positives.
|
||||
var MAX_ENERGY_WH_PER_TOKEN = 10; // legit ≈ 0.001 Wh/tok
|
||||
var MAX_FLOPS_PER_TOKEN = 1e17; // legit ≈ 1e12 /tok
|
||||
// Outlier bounds. Set well above realistic upper limits but tight
|
||||
// enough to drop the pre-fix bimodal Group B (1-5 Wh/token, ~3e16
|
||||
// FLOPs/token) — see the leaderboard PR for the full diagnosis.
|
||||
// Realistic per-token rates on a consumer GPU + 10–30B local model:
|
||||
// ~0.001 Wh/token, ~1e10–1e11 FLOPs/token. We allow 500× and 10,000×
|
||||
// headroom respectively for inefficient hardware / larger models.
|
||||
var MAX_ENERGY_WH_PER_TOKEN = 0.5; // legit ≈ 0.001 Wh/tok
|
||||
var MAX_FLOPS_PER_TOKEN = 1e15; // legit ≈ 1e11 /tok
|
||||
var MAX_DOLLAR_PER_TOKEN = 25.0 / 1e6; // hard ceiling: $25/1M output
|
||||
|
||||
function isOutlier(row) {
|
||||
@@ -29,6 +35,25 @@
|
||||
);
|
||||
}
|
||||
|
||||
// Distinguish "user actually has zero work done" from "user's energy /
|
||||
// FLOPs telemetry never landed". The latter happens when the server
|
||||
// submits with valid dollar savings + token counts but the per-record
|
||||
// energy stamp was missing (pre-fix builds, GPU energy meter
|
||||
// unavailable, etc.). Without this check those rows show as "0.00 Wh"
|
||||
// and skew the rankings + headline totals.
|
||||
//
|
||||
// Threshold: 1000 tokens is well above any single chat-turn — if a
|
||||
// user has that many tokens recorded but no measured energy, the
|
||||
// telemetry is incomplete, not legitimately zero.
|
||||
var MIN_TOKENS_FOR_TELEMETRY = 1000;
|
||||
|
||||
function isMissingTelemetry(row) {
|
||||
var tokens = Number(row.total_tokens) || 0;
|
||||
var energy = Number(row.energy_wh_saved) || 0;
|
||||
var flops = Number(row.flops_saved) || 0;
|
||||
return tokens > MIN_TOKENS_FOR_TELEMETRY && energy === 0 && flops === 0;
|
||||
}
|
||||
|
||||
function escapeHtml(s) {
|
||||
var el = document.createElement("span");
|
||||
el.textContent = s;
|
||||
@@ -62,13 +87,24 @@
|
||||
var medal =
|
||||
rank === 1 ? "\uD83E\uDD47" : rank === 2 ? "\uD83E\uDD48" : rank === 3 ? "\uD83E\uDD49" : "";
|
||||
var row = pageRows[j];
|
||||
// Render "—" for energy / FLOPs columns when telemetry didn't
|
||||
// land (vs the user genuinely having 0). The dollar / request /
|
||||
// token columns are unaffected because those measurements landed
|
||||
// even when energy didn't.
|
||||
var missing = isMissingTelemetry(row);
|
||||
var energyCell = missing
|
||||
? '<td class="lb-number lb-missing" title="Energy telemetry missing for this entry">—</td>'
|
||||
: '<td class="lb-number">' + Number(row.energy_wh_saved || 0).toFixed(2) + "</td>";
|
||||
var flopsCell = missing
|
||||
? '<td class="lb-number lb-missing" title="FLOPs telemetry missing for this entry">—</td>'
|
||||
: '<td class="lb-number">' + fmtLarge(Number(row.flops_saved || 0)) + "</td>";
|
||||
html +=
|
||||
"<tr>" +
|
||||
'<td><span class="lb-rank' + rankClass + '">' + (medal || rank) + "</span></td>" +
|
||||
'<td class="lb-name">' + escapeHtml(row.display_name) + "</td>" +
|
||||
'<td class="lb-number">$' + Number(row.dollar_savings || 0).toFixed(4) + "</td>" +
|
||||
'<td class="lb-number">' + Number(row.energy_wh_saved || 0).toFixed(2) + "</td>" +
|
||||
'<td class="lb-number">' + fmtLarge(Number(row.flops_saved || 0)) + "</td>" +
|
||||
energyCell +
|
||||
flopsCell +
|
||||
'<td class="lb-number">' + Number(row.total_calls || 0).toLocaleString() + "</td>" +
|
||||
'<td class="lb-number">' + Number(row.total_tokens || 0).toLocaleString() + "</td>" +
|
||||
"</tr>";
|
||||
@@ -116,8 +152,15 @@
|
||||
}
|
||||
|
||||
fetch(
|
||||
// `methodology_version=gte.1` filter excludes rows that the
|
||||
// leaderboard-correctness migration quarantined (version 0). Rows
|
||||
// written by current and future clients carry version >= 1, so this
|
||||
// is forward-compatible — pre-fix corrupt rows hide at the query
|
||||
// level (fewer bytes over the wire than client-side outlier
|
||||
// filtering), and downstream client-side checks remain as a
|
||||
// belt-and-suspenders second line of defence.
|
||||
SUPABASE_URL +
|
||||
"/rest/v1/savings_entries?select=display_name,dollar_savings,energy_wh_saved,flops_saved,total_calls,total_tokens&order=dollar_savings.desc&limit=1000",
|
||||
"/rest/v1/savings_entries?select=display_name,dollar_savings,energy_wh_saved,flops_saved,total_calls,total_tokens&methodology_version=gte.1&order=dollar_savings.desc&limit=1000",
|
||||
{
|
||||
headers: {
|
||||
apikey: SUPABASE_ANON_KEY,
|
||||
|
||||
+1
-1
@@ -55,5 +55,5 @@ See how the OpenJarvis community saves money, energy, and compute by running AI
|
||||
<div id="leaderboard-pagination" class="lb-pagination"></div>
|
||||
|
||||
<p style="font-size:12px;opacity:0.6;margin-top:12px">
|
||||
*Dollar savings estimated vs. Claude Opus 4.6 API pricing ($5/1M input, $25/1M output tokens). Assumes local open-source models produce roughly the same number of tokens per request as cloud models.
|
||||
*Dollar savings estimated vs. Claude Fable 5 API pricing ($10/1M input, $50/1M output tokens). Assumes local open-source models produce roughly the same number of tokens per request as cloud models.
|
||||
</p>
|
||||
|
||||
@@ -0,0 +1,135 @@
|
||||
# ACE optimizer (Agentic Context Engineering)
|
||||
|
||||
OpenJarvis supports [ACE](https://github.com/ace-agent/ace) as a third
|
||||
optimizer alongside DSPy and GEPA. Where DSPy bootstraps few-shot
|
||||
examples and GEPA evolves prompts via reflective mutation, **ACE
|
||||
evolves a textual *playbook*** — annotated natural-language strategies
|
||||
the agent reads at inference time. The playbook is updated by a
|
||||
Generator / Reflector / Curator triad of LLM calls.
|
||||
|
||||
## When to pick ACE
|
||||
|
||||
| Task shape | DSPy | GEPA | ACE |
|
||||
|---|---|---|---|
|
||||
| Single-turn QA with crisp metric | strong | strong | weaker |
|
||||
| Long-running agent that should accumulate guidance | weak | medium | strong |
|
||||
| Open-domain where strategies matter more than templates | weak | medium | strong |
|
||||
| When you want to *read* what the optimizer learned | medium | medium | strong |
|
||||
|
||||
ACE's headline artifact is `final_playbook.txt` — a human-readable
|
||||
file like:
|
||||
|
||||
```
|
||||
## STRATEGIES & INSIGHTS
|
||||
[str-00001] helpful=5 harmful=0 :: When the user asks for unit
|
||||
conversion, prefer the exact
|
||||
rational form before rounding.
|
||||
[str-00002] helpful=3 harmful=1 :: Cite a primary source before
|
||||
stating a date claim.
|
||||
```
|
||||
|
||||
If reading those strategies feels like the form of "what learning
|
||||
should produce" for your task, ACE is the right choice.
|
||||
|
||||
## Setup
|
||||
|
||||
ACE is **not on PyPI** as of OpenJarvis v1.0.1, and the upstream
|
||||
repository is structured as a research codebase (multiple top-level
|
||||
directories) rather than a Python package. There's no `learning-ace`
|
||||
extra for that reason. Install ACE manually instead:
|
||||
|
||||
```bash
|
||||
# 1. Clone ACE somewhere outside your OpenJarvis checkout
|
||||
git clone https://github.com/ace-agent/ace.git ~/code/ace
|
||||
cd ~/code/ace
|
||||
curl -LsSf https://astral.sh/uv/install.sh | sh # if you don't have uv
|
||||
uv sync
|
||||
|
||||
# 2. Make ACE's src/ importable from your OpenJarvis venv
|
||||
echo "$HOME/code/ace/src" > \
|
||||
"$(python -c 'import site; print(site.getsitepackages()[0])')/ace.pth"
|
||||
|
||||
# 3. Set the API key for whichever provider ACE will call
|
||||
cp ~/code/ace/.env.example ~/code/ace/.env
|
||||
# Edit ~/code/ace/.env to set API_KEY for your chosen provider.
|
||||
|
||||
# 4. Verify the import resolves from OpenJarvis's venv
|
||||
python -c "from openjarvis.learning.agents.ace_optimizer import HAS_ACE; print(HAS_ACE)"
|
||||
# True
|
||||
```
|
||||
|
||||
If `HAS_ACE` prints `False`, the `.pth` file isn't being picked up —
|
||||
verify the path matches `site.getsitepackages()[0]` for the same
|
||||
Python interpreter you're using to run OpenJarvis.
|
||||
|
||||
## Configuration
|
||||
|
||||
ACE is configured under `[learning.agent.ace]` in your OpenJarvis
|
||||
config TOML:
|
||||
|
||||
```toml
|
||||
[learning.agent]
|
||||
policy = "ace"
|
||||
|
||||
[learning.agent.ace]
|
||||
# ACE's three roles. Empty = inherit from the intelligence primitive's
|
||||
# default cloud model.
|
||||
generator_model = "claude-opus-4-7"
|
||||
reflector_model = "claude-opus-4-7"
|
||||
curator_model = "claude-sonnet-4-6"
|
||||
|
||||
api_provider = "openai" # sambanova | together | openai | commonstack
|
||||
|
||||
num_epochs = 1
|
||||
max_num_rounds = 3
|
||||
playbook_token_budget = 80000
|
||||
max_tokens = 4096
|
||||
|
||||
task_name = "openjarvis"
|
||||
save_dir = "" # default: ~/.openjarvis/learning/ace/<task>/
|
||||
|
||||
min_traces = 20
|
||||
```
|
||||
|
||||
## Running
|
||||
|
||||
Once configured, the same orchestrator that runs DSPy / GEPA also runs
|
||||
ACE — pick it via the `policy` field above. To force a one-shot run:
|
||||
|
||||
```bash
|
||||
jarvis optimize agent --policy ace
|
||||
```
|
||||
|
||||
ACE writes intermediate state and the final playbook to `save_dir`.
|
||||
The OpenJarvis runtime will pick up the playbook on next agent start
|
||||
(via the same sidecar overlay mechanism the Skills System uses).
|
||||
|
||||
## Trace adapter behavior
|
||||
|
||||
OpenJarvis traces are adapted into ACE's `train_samples` /
|
||||
`val_samples` / `test_samples` format via a 70 / 15 / 15 split
|
||||
(order-preserving for reproducibility). Each trace becomes a
|
||||
`{question: trace.query, ground_truth_answer: trace.result}` sample.
|
||||
Traces with empty `query` or `result` are dropped before splitting.
|
||||
|
||||
The `DataProcessor` ACE expects is built from `_TraceDataProcessor`
|
||||
in `src/openjarvis/learning/agents/ace_optimizer.py` — it does a
|
||||
case-insensitive substring match for `answer_is_correct` and averages
|
||||
that for aggregate accuracy. If you're optimizing for a domain where
|
||||
substring matching is the wrong correctness signal (math problems,
|
||||
code, structured outputs), subclass `_TraceDataProcessor` and pass it
|
||||
through your own callsite to `ACEAgentOptimizer.optimize()`.
|
||||
|
||||
## Limitations in v1.0.1
|
||||
|
||||
- **No automatic install.** Document above is the only path.
|
||||
- **The trace adapter uses substring correctness.** Override for
|
||||
domain-specific scoring.
|
||||
- **Single provider per run.** ACE assigns the same `api_provider` to
|
||||
all three roles. To mix providers, run ACE outside OpenJarvis and
|
||||
hand-deliver the resulting playbook into `save_dir`.
|
||||
|
||||
These will get revisited once ACE publishes a PyPI package or stable
|
||||
provider interface — track
|
||||
[ace-agent/ace#issues](https://github.com/ace-agent/ace/issues) for
|
||||
upstream changes that would let us tighten the wrapper.
|
||||
@@ -11,4 +11,23 @@
|
||||
};
|
||||
</script>
|
||||
{% endif %}
|
||||
{% if config.extra.version and config.extra.version.default %}
|
||||
<script>
|
||||
document.addEventListener("DOMContentLoaded", function () {
|
||||
var header = document.querySelector(".md-header__inner");
|
||||
if (!header || header.querySelector(".md-version-badge")) return;
|
||||
var badge = document.createElement("a");
|
||||
badge.className = "md-version-badge";
|
||||
badge.textContent = "{{ config.extra.version.default }}";
|
||||
badge.href = "https://github.com/open-jarvis/OpenJarvis/releases";
|
||||
badge.rel = "noopener";
|
||||
var source = header.querySelector(".md-header__source");
|
||||
if (source) {
|
||||
header.insertBefore(badge, source);
|
||||
} else {
|
||||
header.appendChild(badge);
|
||||
}
|
||||
});
|
||||
</script>
|
||||
{% endif %}
|
||||
{% endblock %}
|
||||
|
||||
@@ -0,0 +1,104 @@
|
||||
---
|
||||
title: Contributing a Showcase Entry
|
||||
description: How to add your setup to the OpenJarvis Showcase
|
||||
---
|
||||
|
||||
# Contributing a Showcase Entry
|
||||
|
||||
The Showcase exists for one reason: to help a confused, curious, *non-technical* reader figure out whether OpenJarvis is worth their weekend. That goal sets every editorial choice on this page.
|
||||
|
||||
## The format
|
||||
|
||||
```markdown
|
||||
---
|
||||
title: <Your Title — short, capitalized>
|
||||
description: <One sentence. The hook a stranger sees in search results.>
|
||||
---
|
||||
|
||||
# <emoji> <One-sentence hook — what it does FOR you, in plain English>
|
||||
|
||||
<figure markdown>
|
||||
{ .showcase-screenshot loading=lazy }
|
||||
<figcaption>A one-sentence caption that adds context the image can't show on its own.</figcaption>
|
||||
</figure>
|
||||
|
||||
<2–3 short paragraphs of context: when do you use this, what changed for
|
||||
you, what the experience feels like. Concrete > abstract. "I read it on
|
||||
my phone before coffee" > "improves morning productivity."
|
||||
|
||||
A bulleted list of two or three CONCRETE OUTCOMES works well — your
|
||||
calendar, your inbox, your code. Specific verbs and proper nouns.>
|
||||
|
||||
## Why it's nice
|
||||
|
||||
- **<one-line benefit>.** <one or two sentences of evidence>
|
||||
- **<one-line benefit>.** <one or two sentences of evidence>
|
||||
- **<one-line benefit>.** <one or two sentences of evidence>
|
||||
|
||||
## How I set this up
|
||||
|
||||
→ **[Tutorial: <name>](../tutorials/<file>.md)** is the closest match.
|
||||
|
||||
→ **[Recipe: <name>](https://github.com/open-jarvis/OpenJarvis/tree/main/src/openjarvis/recipes/data)** if you want the exact config.
|
||||
|
||||
→ **[<one more related doc>](../<path>.md)** if the reader is going deeper.
|
||||
```
|
||||
|
||||
## Editorial conventions
|
||||
|
||||
These are guardrails, not rules. Break them if you have a reason.
|
||||
|
||||
### Lead with the outcome, not the technology
|
||||
|
||||
❌ "Multi-channel routing with MCP-backed memory and an orchestrator agent."<br>
|
||||
✅ "Jarvis answers my Discord messages while I sleep."
|
||||
|
||||
The reader doesn't know what an "orchestrator agent" is yet. They know what a Discord message is.
|
||||
|
||||
### Show one screenshot. Make it the headline.
|
||||
|
||||
A single, large, *interesting* screenshot beats five small ones. Crop it to show the result, not the UI chrome. If you can convey it in an image, don't write the paragraph.
|
||||
|
||||
**Screenshot specs:**
|
||||
|
||||
- 1600×1000 PNG, sRGB, no alpha
|
||||
- File path: `docs/assets/showcase/<your-slug>.png`
|
||||
- Redact: real email addresses, API keys, personal phone numbers, conversation partners' faces or full names (unless they've signed off)
|
||||
- Keep: model names, timestamps, dollar amounts, emoji reactions, your own first name
|
||||
|
||||
### Specific over impressive
|
||||
|
||||
❌ "Saves significant time every morning."<br>
|
||||
✅ "Cut my morning catch-up from 25 minutes to 2."
|
||||
|
||||
Numbers, durations, dollar amounts, and named tools build trust. Adjectives don't.
|
||||
|
||||
### Three paragraphs is plenty
|
||||
|
||||
A reader who wants more clicks the "How I set this up →" link at the bottom. Showcase pages are a funnel into the docs, not a replacement for them. If you find yourself explaining configuration in the showcase entry, that material belongs in the linked tutorial.
|
||||
|
||||
### "Why it's nice" is for the experience, not the architecture
|
||||
|
||||
The bullets under **Why it's nice** should answer "what's different *for you*?" — not "what's different about how the framework works?". Save the architecture talk for the linked docs.
|
||||
|
||||
❌ "Uses local SQLite for state with WAL mode for concurrent reads."<br>
|
||||
✅ "I can read my own memory file in a text editor. I can delete a line and the memory is gone."
|
||||
|
||||
### Every entry must end with at least one "How I set this up →" link
|
||||
|
||||
If there isn't a relevant tutorial yet, link to the closest [User Guide](../user-guide/cli.md) and open an issue noting that the tutorial is missing. We will write it.
|
||||
|
||||
## Submitting
|
||||
|
||||
1. **Fork** the repo and create a branch: `docs/showcase-<your-slug>`.
|
||||
2. **Add** your markdown file at `docs/showcase/<your-slug>.md` and screenshot at `docs/assets/showcase/<your-slug>.png`.
|
||||
3. **Add a tile** to the grid in `docs/showcase/index.md` (matches the existing pattern — emoji + title + 1-sentence summary + `[:octicons-arrow-right-24: See it](<your-slug>.md)`).
|
||||
4. **Open a PR** with the title `docs(showcase): <your title>`. Tag a maintainer if you'd like editorial feedback before merge.
|
||||
|
||||
## Where this goes after merge
|
||||
|
||||
Hannah and the docs team post merged showcase entries to **`#config-showcase`** in [the OpenJarvis Discord](https://discord.gg/openjarvis). You'll get tagged in the post — you don't have to do it yourself.
|
||||
|
||||
## Questions, drafts, half-finished ideas
|
||||
|
||||
Drop them in **`#config-showcase`** on Discord *before* opening a PR. Editorial feedback is faster on chat than in a PR review, and you'll save yourself a round of revisions.
|
||||
@@ -0,0 +1,36 @@
|
||||
---
|
||||
title: Offline Code Reviewer
|
||||
description: Review a pull request on a transatlantic flight, no internet required
|
||||
---
|
||||
|
||||
# 🛠️ Offline Code Reviewer — code review on an airplane
|
||||
|
||||
<figure markdown>
|
||||
{ .showcase-screenshot loading=lazy }
|
||||
<figcaption>Airplane mode in the menu bar. Jarvis reading a `git diff`, the surrounding files, and producing a code review at gate-level Wi-Fi (i.e., none).</figcaption>
|
||||
</figure>
|
||||
|
||||
Earlier this month I was on a flight from SFO to FRA — eleven hours, no usable Wi-Fi. I had a teammate's pull request open in VS Code. I asked Jarvis to review it. It read the diff, read the three files the diff touched, read the project's `CLAUDE.md` for conventions, and produced a review with five comments — two of which caught real bugs.
|
||||
|
||||
The review took about 40 seconds on the laptop's built-in GPU. No API call. No "you're offline" error. By the time we landed I'd dropped the comments into GitHub and the PR was merging.
|
||||
|
||||
The same setup handles:
|
||||
|
||||
- **Code review** — diff + context files + conventions, structured comments.
|
||||
- **Debugging** — paste a traceback, Jarvis reads the stack, opens the relevant files, suggests fixes.
|
||||
- **Test generation** — point at a function, get back a `pytest` file with edge cases.
|
||||
- **Documentation** — generate docstrings that actually match the code, because Jarvis has the file open.
|
||||
|
||||
## Why it's nice
|
||||
|
||||
- **It works on a plane.** Or a train, or a hotel with bad Wi-Fi, or your couch when Comcast is having a day. Same speed every time.
|
||||
- **It sees your repo, not a sanitized chunk.** Cloud coding assistants make you upload a context window. The local one just reads `git status` and the files you're working on.
|
||||
- **No "we trained on your code" question.** Your code never leaves your laptop. Period.
|
||||
|
||||
## How I set this up
|
||||
|
||||
→ **[Tutorial: Code Companion](../tutorials/code-companion.md)** walks through the ReAct-agent + git/file/shell tool stack this uses end-to-end.
|
||||
|
||||
→ **[User Guide: Code Assistant](../user-guide/code-assistant.md)** is the focused recipe walkthrough for daily-driver code review.
|
||||
|
||||
→ **[OpenAI-compatible server](../getting-started/quickstart.md)** — point your editor's existing AI integration (Cursor, Continue, Cody, Aider) at `localhost:8000`. They mostly don't know they're not talking to OpenAI.
|
||||
@@ -0,0 +1,38 @@
|
||||
---
|
||||
title: Track Your Savings
|
||||
description: A leaderboard that tells you exactly how much you saved by running locally
|
||||
---
|
||||
|
||||
# 💸 Track Your Savings — the leaderboard that makes local-first feel real
|
||||
|
||||
<figure markdown>
|
||||
{ .showcase-screenshot loading=lazy }
|
||||
<figcaption>The public leaderboard. The bar on the right is what a month of my Jarvis usage would have cost on the cloud — measured per-query, not estimated.</figcaption>
|
||||
</figure>
|
||||
|
||||
OpenJarvis tracks every inference call you make — the tokens, the latency, the GPU energy — and computes what that same call *would have cost* on OpenAI, Anthropic, Google, and Bedrock. There's a public leaderboard at **[/leaderboard](../leaderboard.md)** where anyone running Jarvis can opt in and watch their savings rack up.
|
||||
|
||||
My current month is roughly:
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| Local inference cost | **`$0.00`** |
|
||||
| Cloud-equivalent cost | **`$342.18`** (Claude Sonnet 4.6 baseline) |
|
||||
| Energy used | **`1.4 kWh`** (~12¢ of grid power) |
|
||||
| Prompts sent to a third party | **`0`** |
|
||||
|
||||
The dollar number is the hook. The bottom row is the actual reason I run Jarvis.
|
||||
|
||||
## Why it's nice
|
||||
|
||||
- **You can see what each query costs you.** Not estimated, not "roughly" — measured. Watt-hours per token, FLOPs per token, latency. Every primitive in OpenJarvis treats compute cost as a first-class quantity alongside accuracy.
|
||||
- **It makes "local-first" stop being abstract.** Watching a bar chart accumulate `$X` a week that *didn't* leave your hands is a different kind of motivating than "your data is private" claims that you can't verify.
|
||||
- **Privacy stops being an act of faith.** Every prompt I send to Jarvis can be traced through the codebase to local-only paths. No "cloud failover" hiding behind a switch.
|
||||
|
||||
## How I set this up
|
||||
|
||||
You don't, really — it's on by default. Every `jarvis ask`, `jarvis serve` request, and channel-routed message is metered by the [telemetry system](../telemetry.md). To opt your savings into the public leaderboard:
|
||||
|
||||
→ **[Leaderboard guide](../leaderboard.md)** — one command to opt in, one command to opt out. Telemetry is local-only by default.
|
||||
|
||||
→ **[Telemetry overview](../telemetry.md)** — what's measured, where it's stored, and how to inspect it yourself with `jarvis telemetry`.
|
||||
@@ -0,0 +1,34 @@
|
||||
---
|
||||
title: Discord Companion
|
||||
description: Jarvis answers questions in your private Discord while you sleep — reads your notes, checks your calendar, schedules things
|
||||
---
|
||||
|
||||
# 💬 Discord Companion — a personal assistant that lives in my Discord
|
||||
|
||||
<figure markdown>
|
||||
{ .showcase-screenshot loading=lazy }
|
||||
<figcaption>I DM'd Jarvis from my phone at midnight. It checked my Google Calendar, cross-referenced a note from last week, and answered — running on the Mac mini in my closet.</figcaption>
|
||||
</figure>
|
||||
|
||||
I have a private Discord server with two channels and one user (me). Jarvis lives there. I can DM it from my phone, my laptop, or my watch — anywhere Discord runs. Sample things I've asked it this week:
|
||||
|
||||
- "What's the address of the place I had that meeting last Tuesday?" → Jarvis searches my calendar + meeting notes, replies in 4 seconds.
|
||||
- "Reply to Mom's text from earlier saying I'll call tomorrow at 7." → drafts a reply, asks me to confirm, sends.
|
||||
- "Add 'Sam's birthday is March 12' to my long-term memory." → updates `MEMORY.md`, confirms.
|
||||
- "Summarize the last hour of conversation in `#deploys-prod`." → reads the Slack channel via MCP, summarizes.
|
||||
|
||||
I used to use my phone's voice assistant for this. The two differences that matter: **Jarvis answers in three sentences, not one,** and **it actually has my context** — my notes, my calendar, my projects, my history.
|
||||
|
||||
## Why it's nice
|
||||
|
||||
- **Latency feels like talking to a person.** Local inference on a modest GPU is 5–10× faster than round-tripping to a cloud API. Question to answer in 3 seconds.
|
||||
- **The Discord interface is multi-device for free.** Same conversation thread on my phone, laptop, watch — no special app to install.
|
||||
- **It's already private.** A Discord server I run, talking to a model on a machine I own. The data trail is two endpoints I control.
|
||||
|
||||
## How I set this up
|
||||
|
||||
→ **[Tutorial: Messaging Hub](../tutorials/messaging-hub.md)** is the closest match — same channel-adapter + orchestrator-agent pattern, with Discord substituted for Slack.
|
||||
|
||||
→ **[Channel docs](../user-guide/cli.md)** walks through Discord/Slack/Telegram/WhatsApp setup. Discord is two environment variables and a bot token.
|
||||
|
||||
→ **[MCP integration guide](../user-guide/cli.md)** if you want Jarvis to reach into Notion, Linear, Gmail, etc.
|
||||
@@ -0,0 +1,72 @@
|
||||
---
|
||||
title: Showcase
|
||||
description: What people actually do with OpenJarvis — outcomes first, scripts later
|
||||
---
|
||||
|
||||
# Showcase
|
||||
|
||||
These are stories from people who use OpenJarvis day to day. Each entry shows the **result** — a screenshot, a paragraph of context, and a short link to the docs that explain how to build it. If you're trying to figure out whether OpenJarvis is worth a weekend of your time, start here.
|
||||
|
||||
!!! tip "New here?"
|
||||
The Showcase answers *"what's possible?"*. When you find something you want for yourself, follow the **How I set this up** link at the bottom of each page — it lands on a [Tutorial](../tutorials/index.md) that walks through the build.
|
||||
|
||||
<div class="grid cards" markdown>
|
||||
|
||||
- :material-coffee:{ .lg .middle } **Morning Brief**
|
||||
|
||||
---
|
||||
|
||||
Slack, email, GitHub, and calendar — read overnight, summarized into 5 bullets in your phone by 7am. Cuts the daily "what did I miss" tax to zero.
|
||||
|
||||
[:octicons-arrow-right-24: See it](morning-brief.md)
|
||||
|
||||
- :material-brain:{ .lg .middle } **Memory That Doesn't Reset**
|
||||
|
||||
---
|
||||
|
||||
Tell Jarvis you're allergic to shellfish once. Three months later it brings it up when you're restaurant-planning. Plain markdown files, no vector-DB tricks.
|
||||
|
||||
[:octicons-arrow-right-24: See it](persistent-memory.md)
|
||||
|
||||
- :material-piggy-bank-outline:{ .lg .middle } **Track Your Savings**
|
||||
|
||||
---
|
||||
|
||||
A leaderboard that tells you exactly how much you saved by running locally — and reminds you that none of your prompts ever left your house.
|
||||
|
||||
[:octicons-arrow-right-24: See it](cost-savings.md)
|
||||
|
||||
- :material-message-text:{ .lg .middle } **Discord Companion**
|
||||
|
||||
---
|
||||
|
||||
Jarvis answers questions in your private Discord while you sleep. Reads your notes, checks your calendar, schedules things, replies in your voice.
|
||||
|
||||
[:octicons-arrow-right-24: See it](discord-companion.md)
|
||||
|
||||
- :material-code-tags-check:{ .lg .middle } **Offline Code Reviewer**
|
||||
|
||||
---
|
||||
|
||||
Review a pull request on a transatlantic flight. Jarvis reads the diff, the surrounding files, and the project conventions — without an internet connection.
|
||||
|
||||
[:octicons-arrow-right-24: See it](coding-assistant.md)
|
||||
|
||||
</div>
|
||||
|
||||
---
|
||||
|
||||
## Share your setup
|
||||
|
||||
The Showcase grows from real users. If you've built something interesting on top of OpenJarvis — or just have a configuration you're proud of — the format is simple and the bar is low:
|
||||
|
||||
1. **One-sentence hook**: what does this *do for you*?
|
||||
2. **A screenshot or 15-second screen recording**: the visible result.
|
||||
3. **2–3 short paragraphs**: when you use it, why it's nice (cost, privacy, speed, calm).
|
||||
4. **"How I set this up →"**: a link to the relevant [Tutorial](../tutorials/index.md), [User Guide](../user-guide/cli.md), or [Recipe](https://github.com/open-jarvis/OpenJarvis/tree/main/src/openjarvis/recipes/data).
|
||||
|
||||
See [Contributing a Showcase Entry](CONTRIBUTING.md) for the template and the editorial conventions (screenshot sizing, what to redact, tone).
|
||||
|
||||
## Want to talk to other people doing this?
|
||||
|
||||
The **`#config-showcase`** channel in the [OpenJarvis Discord](https://discord.gg/openjarvis) is where people post and discuss personal setups. Drop a screenshot, ask "how would I do X?", or browse what others have shared.
|
||||
@@ -0,0 +1,39 @@
|
||||
---
|
||||
title: Morning Brief
|
||||
description: Slack, email, GitHub, and calendar — summarized into a 5-bullet brief on your phone by 7am
|
||||
---
|
||||
|
||||
# ☕ Morning Brief — Jarvis reads everything overnight so I don't have to
|
||||
|
||||
<figure markdown>
|
||||
{ .showcase-screenshot loading=lazy }
|
||||
<figcaption>The 7am brief that arrives in my private Discord — 5 bullets, two minutes to read, written by an agent that ran on my desk while I slept.</figcaption>
|
||||
</figure>
|
||||
|
||||
Every morning at 7am, before my first coffee, a message appears in my private Discord with five bullets:
|
||||
|
||||
- what shipped at work overnight (GitHub releases + merged PRs)
|
||||
- the two emails I actually need to act on (with one-line summaries)
|
||||
- anything mentioned in my team's `#general` Slack channel
|
||||
- today's calendar with the next 24 hours of meetings
|
||||
- one thing I asked Jarvis to track for me ("did Tuesday's deploy roll out cleanly?")
|
||||
|
||||
It's the first thing I read on my phone, while I'm still in bed. The brief used to take me 25 minutes — opening four apps, scrolling, deciding what mattered. Now it's two minutes of reading and I'm done.
|
||||
|
||||
## Why it's nice
|
||||
|
||||
- **Costs me nothing per month.** It runs on a Mac mini in my closet. Same prompt-volume on the OpenAI API would be `~$18/month` based on the leaderboard's estimates.
|
||||
- **Nothing leaves my house.** My inbox, my Slack DMs, my calendar — Jarvis reads them locally and writes the digest locally. The only network call is the Discord webhook to my own private server.
|
||||
- **It learns my taste.** Over a few weeks Jarvis figured out that PR titles starting with `chore:` aren't worth surfacing and that I don't want to see calendar holds I created myself. The summarizer has a `MEMORY.md` it updates when I react with 👎 to a bullet.
|
||||
|
||||
## What you'd need
|
||||
|
||||
A laptop or mini-PC that stays on overnight, an inference engine (Ollama is the easy default), accounts on whichever surfaces you want summarized (Slack, Gmail, GitHub, Google Calendar), and a Discord (or Slack, or Telegram, or email) destination to post the brief to.
|
||||
|
||||
## How I set this up
|
||||
|
||||
→ **[Tutorial: Scheduled Personal Ops](../tutorials/scheduled-ops.md)** walks through the cron-scheduled agent pattern this uses. The morning-brief flavour is `orchestrator` agent + the channel adapters + the scheduler primitive — three primitives, one TOML recipe.
|
||||
|
||||
→ **[User Guide: Morning Digest](../user-guide/morning-digest.md)** is the focused recipe walkthrough if you only want this one workflow.
|
||||
|
||||
→ **[User Guide: Channels](../user-guide/cli.md)** for connecting Discord/Slack/Telegram as the destination.
|
||||
@@ -0,0 +1,34 @@
|
||||
---
|
||||
title: Memory That Doesn't Reset
|
||||
description: Tell Jarvis something once. It remembers — three months later, across every conversation
|
||||
---
|
||||
|
||||
# 🧠 Memory That Doesn't Reset — Jarvis actually knows me
|
||||
|
||||
<figure markdown>
|
||||
{ .showcase-screenshot loading=lazy }
|
||||
<figcaption>Three months after I mentioned the allergy in passing, Jarvis brings it up — unprompted — while helping me pick a birthday-dinner restaurant.</figcaption>
|
||||
</figure>
|
||||
|
||||
I mentioned to Jarvis once, in a throwaway sentence in April, that I'm allergic to shellfish. In July, when I asked it to help me pick a restaurant for my partner's birthday, it volunteered "you'll want to filter for menus that have non-shellfish options" — without being reminded, in a totally different conversation, on a different topic.
|
||||
|
||||
That's not magic. The trick is that Jarvis writes to three plain markdown files in my home directory whenever it learns something worth remembering:
|
||||
|
||||
- `SOUL.md` — how I want it to behave (tone, length, what to push back on)
|
||||
- `MEMORY.md` — facts about me, my projects, my preferences
|
||||
- `USER.md` — who I am: my role, my team, my context
|
||||
|
||||
Every new conversation starts by reading those three files. I can open them in any text editor. I can delete a line and the memory is gone. The whole thing is `~6 KB` of markdown. No vector DB, no embedding cache, no opaque "personalization layer."
|
||||
|
||||
## Why it's nice
|
||||
|
||||
- **It's auditable.** I can read what Jarvis "knows" about me in 30 seconds. Most personal-AI products literally can't tell you.
|
||||
- **It's portable.** I keep my three files in iCloud Drive. When I set up Jarvis on a new machine, my memory comes with me — without re-onboarding.
|
||||
- **It compounds.** After two weeks Jarvis stopped re-asking what my code style is. After six weeks it stopped re-asking who's on my team. The conversations get shorter because the context is already there.
|
||||
- **It can't drift.** Vector retrieval can confidently surface the wrong "memory" and you'd never know. Plain markdown that I can read can't lie about what it contains.
|
||||
|
||||
## How I set this up
|
||||
|
||||
→ **[User Guide: Agents](../user-guide/agents.md)** explains the persistent-agent pattern, including how `SOUL.md` / `MEMORY.md` / `USER.md` are loaded at conversation start.
|
||||
|
||||
→ **[Tutorial: Deep Research Assistant](../tutorials/deep-research.md)** uses the same persistent-memory primitive — a good place to see it in action with code.
|
||||
@@ -1,3 +1,33 @@
|
||||
/* ── Version badge (top-right of header) ─────────────────────────── */
|
||||
.md-version-badge {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
height: 1.6rem;
|
||||
padding: 0 0.55rem;
|
||||
margin: 0 0.4rem 0 0.2rem;
|
||||
font-family: var(--md-code-font);
|
||||
font-size: 0.72rem;
|
||||
font-weight: 500;
|
||||
line-height: 1;
|
||||
letter-spacing: 0.02em;
|
||||
color: var(--md-primary-bg-color, #ffffff);
|
||||
background-color: rgba(255, 255, 255, 0.12);
|
||||
border: 1px solid rgba(255, 255, 255, 0.25);
|
||||
border-radius: 0.6rem;
|
||||
text-decoration: none;
|
||||
white-space: nowrap;
|
||||
transition: background-color 120ms ease, border-color 120ms ease;
|
||||
}
|
||||
.md-version-badge:hover,
|
||||
.md-version-badge:focus {
|
||||
background-color: rgba(255, 255, 255, 0.22);
|
||||
border-color: rgba(255, 255, 255, 0.45);
|
||||
text-decoration: none;
|
||||
}
|
||||
@media screen and (max-width: 76.1875em) {
|
||||
.md-version-badge { display: none; }
|
||||
}
|
||||
|
||||
/* ── Fonts ─────────────────────────────────────────────────────────── */
|
||||
:root {
|
||||
--md-text-font: Georgia, "Times New Roman", serif;
|
||||
@@ -410,6 +440,13 @@
|
||||
font-family: var(--md-code-font-family, monospace);
|
||||
font-size: 13px;
|
||||
}
|
||||
/* Placeholder for rows where energy / FLOPs telemetry didn't land.
|
||||
Distinguishes "telemetry missing" from "user genuinely had 0 work
|
||||
done" without making the row visually pop more than data rows. */
|
||||
.lb-missing {
|
||||
color: var(--md-default-fg-color--light, #999);
|
||||
font-style: italic;
|
||||
}
|
||||
|
||||
/* ── DocSearch ───────────────────────────────────────────────────────── */
|
||||
#docsearch {
|
||||
@@ -532,3 +569,18 @@
|
||||
display: none !important;
|
||||
}
|
||||
}
|
||||
|
||||
/* ---------------------------------------------------------------------------
|
||||
* Showcase screenshots
|
||||
*
|
||||
* Hero images on docs/showcase/* pages. Constrains width on wide screens and
|
||||
* adds a subtle border so placeholder/broken-image states still look intentional
|
||||
* before community-contributed screenshots populate docs/assets/showcase/.
|
||||
* ------------------------------------------------------------------------- */
|
||||
.showcase-screenshot {
|
||||
max-width: 100%;
|
||||
height: auto;
|
||||
border-radius: 8px;
|
||||
border: 1px solid var(--md-default-fg-color--lightest, rgba(0, 0, 0, 0.08));
|
||||
box-shadow: 0 2px 8px rgba(0, 0, 0, 0.06);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,124 @@
|
||||
# Telemetry
|
||||
|
||||
OpenJarvis ships **anonymous usage telemetry** by default so the team can
|
||||
see where the product breaks, what features people actually use, and
|
||||
how to make it better. This page documents exactly what is and isn't
|
||||
collected, where the data goes, and how to opt out.
|
||||
|
||||
## TL;DR
|
||||
|
||||
- **On by default**, anonymous, no chat content.
|
||||
- **Anonymous** — one random UUID per install, no email, no name, no IP.
|
||||
- **No chat content, ever.** Only counts, timings, and feature names.
|
||||
- **Self-hosted backend** on the OpenJarvis team's PostHog instance —
|
||||
data is not sold or shared with third parties.
|
||||
- **365-day retention**, after which events are deleted automatically.
|
||||
|
||||
## What we collect
|
||||
|
||||
### Lifecycle events
|
||||
|
||||
| Event | Source | Why we send it |
|
||||
|---|---|---|
|
||||
| `install_started` | `install.sh` | Top of install funnel |
|
||||
| `install_stage_completed` | `install.sh` | Per-stage timing — where do people drop off? |
|
||||
| `install_completed` | `install.sh` | Did the install succeed? |
|
||||
| `install_failed` | `install.sh` | Which stage failed, and on what OS |
|
||||
| `app_opened` | Backend + frontend | DAU / WAU / MAU |
|
||||
| `setup_completed` | Frontend | First-run wizard finished |
|
||||
| `first_chat_sent` | Backend | First-ever message — activation |
|
||||
| `uninstall_started` | `uninstall.sh` (if user runs it) | Churn signal |
|
||||
|
||||
### Usage events
|
||||
|
||||
| Event | Why we send it |
|
||||
|---|---|
|
||||
| `chat_session_ended` | Aggregated per-session: turn count, tokens, latency, tool count |
|
||||
| `tool_first_used` | Which built-in tools are actually adopted |
|
||||
| `model_changed` | How often users switch models |
|
||||
| `feature_used` | Which features get traffic, which don't |
|
||||
| `connector_auth_completed` | Which connectors people set up |
|
||||
| `error_shown_to_user` | User-visible error class (not stack trace) |
|
||||
| `feedback_submitted` | Was a rating given? Was a comment included? |
|
||||
| `settings_changed` | Which settings get toggled |
|
||||
| `usage_daily_summary` | Once-per-day aggregated counts |
|
||||
|
||||
The canonical, authoritative list with every property name and its
|
||||
type validator lives in
|
||||
[`src/openjarvis/analytics/events.py`](../src/openjarvis/analytics/events.py).
|
||||
That file is the only place new events can be added — PR review is
|
||||
the gate.
|
||||
|
||||
## What we never collect
|
||||
|
||||
Hard guardrails, enforced by code:
|
||||
|
||||
- **Chat content** — prompts, model outputs, system messages, tool args.
|
||||
- **File paths** — anything matching `~/`, `$HOME`, `/Users/<name>`, `/home/<name>`, `file://`.
|
||||
- **Emails, names, phone numbers, addresses.**
|
||||
- **IP addresses** (IPv4 + IPv6). PostHog's IP geo lookup is disabled server-side too.
|
||||
- **MAC addresses, hardware serials, drive UUIDs.**
|
||||
- **Stack traces** — only error class enums.
|
||||
- **API keys, OAuth tokens, JWTs, bearer tokens, password assignments** —
|
||||
matched and dropped at value level.
|
||||
- **Hostnames** that look personal (e.g. `alice-macbook.local`).
|
||||
- **Lists, dicts, sets** — composite values are never sent so PII can't
|
||||
smuggle through inside containers.
|
||||
|
||||
Two independent filters run before every event leaves the machine:
|
||||
|
||||
1. [`src/openjarvis/analytics/redaction.py`](../src/openjarvis/analytics/redaction.py) — value-level pattern matching (20+ regexes for PII).
|
||||
2. [`src/openjarvis/analytics/events.py`](../src/openjarvis/analytics/events.py) — structural allowlist (event name + property name + type validator).
|
||||
|
||||
Any failure at either layer → the event or property is silently
|
||||
dropped. Tests covering the patterns: [`tests/analytics/test_redaction.py`](../tests/analytics/test_redaction.py).
|
||||
|
||||
## Where the data goes
|
||||
|
||||
- **Today** (alpha): PostHog Cloud (US region) free tier. Disclosed
|
||||
here for transparency.
|
||||
- **Production target**: A self-hosted PostHog instance at
|
||||
`analytics.openjarvis.ai`, Hetzner US-East. Single-tenant, operated
|
||||
by the OpenJarvis team.
|
||||
- **Never** sold, shared with advertisers, or used for anything other
|
||||
than improving OpenJarvis.
|
||||
|
||||
## Retention
|
||||
|
||||
- Default retention: **365 days**, then events are deleted by PostHog
|
||||
automatically.
|
||||
- `jarvis analytics reset-id` lets you orphan all of your past events
|
||||
by generating a fresh anonymous ID for future events.
|
||||
|
||||
## How identity works
|
||||
|
||||
A single UUID v4 is generated on first install and stored at
|
||||
`~/.openjarvis/anon_id`. The install script, backend, and frontend all
|
||||
read the same file so events across the full lifecycle tie to one
|
||||
person — without us ever knowing who that person is.
|
||||
|
||||
Delete the file (`rm ~/.openjarvis/anon_id`) and a fresh UUID will be
|
||||
generated next time the app runs. The previous UUID and its events
|
||||
are then orphaned.
|
||||
|
||||
## For researchers and contributors
|
||||
|
||||
- **Adding an event**: edit `src/openjarvis/analytics/events.py`,
|
||||
declare the spec, then update this page. PR review enforces both.
|
||||
- **Adding a PII pattern**: edit `src/openjarvis/analytics/redaction.py`
|
||||
and add a test case in `tests/analytics/test_redaction.py`.
|
||||
- **Inspecting what your install sends**: run with
|
||||
`OPENJARVIS_LOG_LEVEL=DEBUG` and grep for `Analytics`. You'll see
|
||||
every event name and (redacted) property dict before it ships.
|
||||
|
||||
## Related
|
||||
|
||||
- Local telemetry (FLOPs, energy, latency stored in
|
||||
`~/.openjarvis/telemetry.db`) is a **separate** subsystem documented
|
||||
in [`src/openjarvis/telemetry/`](../src/openjarvis/telemetry/). It
|
||||
never leaves the machine and is controlled by `[telemetry]` (not
|
||||
`[analytics]`) in `config.toml`.
|
||||
- The leaderboard / contest opt-in (`OptInModal.tsx`) is a separate,
|
||||
voluntary feature that publicly shares your energy and savings on
|
||||
the OpenJarvis leaderboard. It is **not** the same as analytics and
|
||||
requires explicit opt-in with a display name and email.
|
||||
@@ -13,11 +13,77 @@ Agents are the agentic logic layer of OpenJarvis. They determine how a query is
|
||||
| `RLMAgent` | `rlm` | Yes | Yes | Recursive LM with persistent REPL |
|
||||
| `OpenHandsAgent` | `openhands` | No | Yes | Wraps real openhands-sdk |
|
||||
| `ClaudeCodeAgent` | `claude_code` | No | Yes | Claude Agent SDK via Node.js subprocess |
|
||||
| `OpenCodeAgent` | `opencode` | No | Yes | [opencode](https://opencode.ai) coding agent on your local engine |
|
||||
| `OperativeAgent` | `operative` | Yes | Yes | Persistent scheduled agent with state management |
|
||||
| `MonitorOperativeAgent` | `monitor_operative` | Yes | Yes | Long-horizon agent with 4 configurable strategy axes |
|
||||
|
||||
---
|
||||
|
||||
## Persistent Persona: SOUL.md, MEMORY.md, USER.md
|
||||
|
||||
Every agent's system prompt is assembled at conversation start by the `SystemPromptBuilder`, which injects up to three optional Markdown files -- the **persistent persona**. They are plain text you own and edit, loaded at the start of each conversation. There is no vector database or embedding cache behind them.
|
||||
|
||||
| File | What it holds | Example line |
|
||||
|------|---------------|--------------|
|
||||
| `SOUL.md` | How the agent should behave -- tone, length, what to push back on | `Be concise. Challenge weak assumptions.` |
|
||||
| `MEMORY.md` | Facts about you, your projects, your preferences | `I deploy to Postgres, never MySQL.` |
|
||||
| `USER.md` | Who you are -- role, team, context | `Backend engineer at Acme, on the payments team.` |
|
||||
|
||||
This persona is distinct from the retrieval [memory backend](memory.md): the persona is always-on Markdown context loaded into the prompt, while the memory backend is searchable long-term storage the agent queries on demand.
|
||||
|
||||
### Where they live
|
||||
|
||||
By default the files are read from the config directory:
|
||||
|
||||
```
|
||||
~/.openjarvis/SOUL.md
|
||||
~/.openjarvis/MEMORY.md
|
||||
~/.openjarvis/USER.md
|
||||
```
|
||||
|
||||
(The config directory honors `$OPENJARVIS_HOME` / `$XDG_DATA_HOME` when set.) The paths are configurable under `[memory_files]`:
|
||||
|
||||
```toml
|
||||
[memory_files]
|
||||
soul_path = "~/.openjarvis/SOUL.md"
|
||||
memory_path = "~/.openjarvis/MEMORY.md"
|
||||
user_path = "~/.openjarvis/USER.md"
|
||||
persona_name = "" # optional named persona -- see below
|
||||
```
|
||||
|
||||
### How they're loaded
|
||||
|
||||
At the start of each conversation, `SystemPromptBuilder` reads each file as UTF-8 and adds its contents as a section of the system prompt, after the agent template and before the skill catalog:
|
||||
|
||||
- **All three are optional.** A missing or empty file is skipped, so any subset works and an install with no persona files behaves exactly as before.
|
||||
- **Edits apply to the next conversation.** The files are read once when a conversation's prompt is built, so there is no restart or re-indexing -- edit or delete a line and it takes effect the next time you start a conversation.
|
||||
- **Each section is length-capped.** Files are truncated to a per-section character budget so a large `MEMORY.md` cannot crowd out the rest of the prompt.
|
||||
|
||||
### Named personas
|
||||
|
||||
A single install can answer as different personas without changing global config. A named persona lives in its own directory:
|
||||
|
||||
```
|
||||
~/.openjarvis/personas/<name>/SOUL.md
|
||||
~/.openjarvis/personas/<name>/MEMORY.md
|
||||
~/.openjarvis/personas/<name>/USER.md
|
||||
```
|
||||
|
||||
Select one per invocation, or opt out entirely:
|
||||
|
||||
```bash
|
||||
jarvis ask --persona work "summarize my open PRs"
|
||||
jarvis ask --persona none "what is 2 + 2?" # inject no persona
|
||||
```
|
||||
|
||||
Set `persona_name` under `[memory_files]` to make a named persona the default. `persona_name = "none"` (equivalently `--persona none`) disables persona injection for that run.
|
||||
|
||||
### Editing them
|
||||
|
||||
`SOUL.md`, `MEMORY.md`, and `USER.md` are plain Markdown -- open them in any editor. `MEMORY.md` and `USER.md` can also be updated by the agent itself through the `memory_manage` and `user_profile_manage` tools when those are enabled, so the agent can record a new fact mid-conversation. These tools always target the default `MEMORY.md` and `USER.md` (under `~/.openjarvis/`), never a named persona's copies -- edit those by hand.
|
||||
|
||||
---
|
||||
|
||||
## BaseAgent ABC
|
||||
|
||||
All agents extend the abstract `BaseAgent` class.
|
||||
@@ -383,6 +449,64 @@ jarvis ask --agent claude_code "Refactor the tests to use pytest fixtures"
|
||||
|
||||
---
|
||||
|
||||
## OpenCodeAgent
|
||||
|
||||
The `OpenCodeAgent` delegates coding tasks to [opencode](https://opencode.ai), the open-source coding agent, running it **on your local engine**. opencode handles the agentic loop, file edits, and tool use; OpenJarvis supplies the model — keeping coding-agent work local-first.
|
||||
|
||||
!!! warning "Requirements"
|
||||
Requires the `opencode` binary on `PATH` (`npm i -g opencode-ai` or `brew install anomalyco/tap/opencode`). It is **not** bundled; `run()` returns a clear error if it is missing. No `ANTHROPIC_API_KEY` needed — inference goes through your OpenJarvis engine.
|
||||
|
||||
**How it works:**
|
||||
|
||||
1. Derives an OpenAI-compatible base URL from the `engine` (e.g. Ollama/vLLM/llama.cpp at `<host>/v1`) and writes an `opencode.json` in the workspace registering it as an `@ai-sdk/openai-compatible` provider (`openjarvis/<model>`).
|
||||
2. Spawns a headless `opencode serve` (loopback, random port) and waits for `/global/health`.
|
||||
3. Creates a session (`POST /session`) and sends the task (`POST /session/{id}/message`) with `model={providerID, modelID}` and the selected `agent` (`build` or `plan`).
|
||||
4. Parses the returned message `parts` — text parts → `content`, tool parts → `tool_results` — into an `AgentResult`.
|
||||
5. `close()` disposes the session/server.
|
||||
|
||||
**Constructor parameters (selected):**
|
||||
|
||||
| Parameter | Type | Default | Description |
|
||||
|---------------------|-------------------|------------------|----------------------------------------------------------|
|
||||
| `engine` | `InferenceEngine` | -- | Used to derive the local OpenAI-compatible provider URL |
|
||||
| `model` | `str` | -- | Model id served at the provider (e.g. `qwen3:8b`) |
|
||||
| `workspace` | `str` | `os.getcwd()` | Directory opencode operates in |
|
||||
| `agent` | `str` | `"build"` | opencode agent: `build` (full access) or `plan` (read-only) |
|
||||
| `provider_base_url` | `str` | derived | Override the engine-derived OpenAI base URL |
|
||||
| `provider_id` | `str` | `"openjarvis"` | opencode provider id to register/use |
|
||||
| `model_id` | `str` | `model` | Model id within the provider |
|
||||
| `server_password` | `str` | `$OPENCODE_SERVER_PASSWORD` | Optional basic-auth for the opencode server |
|
||||
| `timeout` | `int` | `600` | HTTP timeout in seconds |
|
||||
|
||||
```python
|
||||
from openjarvis.agents.opencode import OpenCodeAgent
|
||||
|
||||
agent = OpenCodeAgent(engine, "qwen3:8b", workspace="/path/to/project", agent="build")
|
||||
result = agent.run("Add type hints to utils.py and run the tests")
|
||||
print(result.content)
|
||||
agent.close()
|
||||
```
|
||||
|
||||
```bash
|
||||
# Via CLI (opencode must be installed)
|
||||
jarvis ask --agent opencode "Refactor the parser to use a state machine"
|
||||
```
|
||||
|
||||
!!! tip "Pass-through providers"
|
||||
If the `engine` has no derivable base URL, pass `model` as `provider/model` (e.g. `ollama/llama3`) and opencode resolves it from its own configuration — no `opencode.json` is written.
|
||||
|
||||
!!! warning "Model capability matters"
|
||||
opencode's agentic loop (planning + correct tool calls + multi-step
|
||||
follow-through) needs a reasonably capable model. In testing, a **27B**
|
||||
local model (Qwen3.5-27B served via vLLM) solved a 7-task coding suite
|
||||
cleanly (create / edit / bug-fix / implement-to-pass-tests / multi-file,
|
||||
verified by running the code and tests). An **8B** model (qwen3:8b) was
|
||||
unreliable — malformed tool calls, syntactically broken code, and
|
||||
half-finished tasks. Prefer a capable local model (or a cloud model) for
|
||||
real coding work.
|
||||
|
||||
---
|
||||
|
||||
## OperativeAgent
|
||||
|
||||
The `OperativeAgent` is a persistent, scheduled autonomous agent with built-in session persistence and state recall. Designed for "Operators" -- autonomous agents that run on a schedule with automatic state management between ticks. Extends `ToolUsingAgent`.
|
||||
|
||||
+74
-94
@@ -66,6 +66,8 @@ jarvis ask "What is the capital of France?"
|
||||
| `--no-context` | flag | off | Disable memory context injection |
|
||||
| `-a`, `--agent AGENT` | string | none | Agent to use (`simple`, `orchestrator`) |
|
||||
| `--tools TOOLS` | string | none | Comma-separated tool names to enable |
|
||||
| `-i`, `--image PATH` | path | none | Image file for a vision model (e.g. `gemma3:4b`); repeatable |
|
||||
| `-S`, `--screen` | flag | off | Capture the current screen and send it to the vision model |
|
||||
|
||||
### Direct Mode vs Agent Mode
|
||||
|
||||
@@ -105,6 +107,39 @@ jarvis ask --no-context "Tell me about Python"
|
||||
jarvis ask --max-tokens 2048 "Write a detailed essay about AI"
|
||||
```
|
||||
|
||||
### Vision Input
|
||||
|
||||
Vision-capable models (such as `gemma3:4b`) can read images alongside your
|
||||
text prompt. Attach one or more image files with `-i`/`--image`, or capture
|
||||
the current screen with `-S`/`--screen`:
|
||||
|
||||
```bash
|
||||
# Ask about a local image
|
||||
jarvis ask -i screenshot.png "What is shown in this image?"
|
||||
|
||||
# Send multiple images (the flag is repeatable)
|
||||
jarvis ask -i chart-a.png -i chart-b.png "Compare these two charts"
|
||||
|
||||
# Capture the current screen and ask about it
|
||||
jarvis ask --screen "Summarize what's on my screen"
|
||||
```
|
||||
|
||||
Vision runs in **direct mode** only. If you also pass `--agent`, the image is
|
||||
ignored and a note is printed — re-run with `--agent ""` to force direct mode.
|
||||
|
||||
The Ollama context window can be tuned for large images or long prompts with
|
||||
the `JARVIS_NUM_CTX` environment variable (default `16384`):
|
||||
|
||||
```bash
|
||||
JARVIS_NUM_CTX=8192 jarvis ask --screen "What's on my screen?"
|
||||
```
|
||||
|
||||
!!! note "Keep vision on-device"
|
||||
Images are sensitive. OpenJarvis prints a privacy warning before sending
|
||||
an image to a non-local engine, so a screenshot never leaves your machine
|
||||
unnoticed. Use a local engine (e.g. `ollama` with `gemma3:4b`) to keep
|
||||
vision fully local.
|
||||
|
||||
### JSON Output Format
|
||||
|
||||
When using `--json` in **direct mode**, the output includes:
|
||||
@@ -199,6 +234,35 @@ jarvis model pull qwen3:8b
|
||||
|
||||
---
|
||||
|
||||
## `jarvis pearl`
|
||||
|
||||
Access Pearl's native node, wallet, and RPC tools from the OpenJarvis CLI.
|
||||
|
||||
```bash
|
||||
jarvis pearl doctor
|
||||
jarvis pearl node -- <pearld args>
|
||||
jarvis pearl wallet -- <oyster args>
|
||||
jarvis pearl ctl -- <prlctl args>
|
||||
jarvis pearl address
|
||||
```
|
||||
|
||||
All Pearl wrapper commands use the `jarvis pearl <command>` shape. The
|
||||
pass-through commands map to Pearl's native binaries:
|
||||
|
||||
| OpenJarvis command | Pearl binary | Use |
|
||||
|--------------------|--------------|-----|
|
||||
| `jarvis pearl doctor` | n/a | Check whether `pearld`, `oyster`, and `prlctl` are discoverable |
|
||||
| `jarvis pearl node` | `pearld` | Run the Pearl full node |
|
||||
| `jarvis pearl wallet` | `oyster` | Run the Oyster wallet daemon |
|
||||
| `jarvis pearl ctl` | `prlctl` | Query Pearl node or wallet RPC |
|
||||
| `jarvis pearl address` | `prlctl --wallet getnewaddress` | Generate a wallet address from Oyster |
|
||||
|
||||
Use `PEARL_HOME=/path/to/pearl` or `--pearl-home /path/to/pearl` if Pearl's
|
||||
`bin/` directory is not on `PATH`. See the [Pearl CLI guide](pearl.md) for
|
||||
examples.
|
||||
|
||||
---
|
||||
|
||||
## `jarvis memory`
|
||||
|
||||
Manage the document memory store for retrieval-augmented generation.
|
||||
@@ -434,98 +498,14 @@ When an agent is configured (e.g., `--agent orchestrator`), non-streaming reques
|
||||
|
||||
---
|
||||
|
||||
## `jarvis learning`
|
||||
## LLM-guided spec search (no CLI yet)
|
||||
|
||||
Frontier-driven harness learning (distillation). Manages learning sessions, reviews pending edits, and controls the benchmark gate.
|
||||
|
||||
### `jarvis learning init`
|
||||
|
||||
Initialize the distillation checkpoint repo and directory layout.
|
||||
|
||||
```bash
|
||||
jarvis learning init
|
||||
```
|
||||
|
||||
### `jarvis learning run`
|
||||
|
||||
Run an on-demand learning session.
|
||||
|
||||
```bash
|
||||
jarvis learning run
|
||||
jarvis learning run --autonomy auto # auto-apply all edits
|
||||
jarvis learning run --autonomy manual # dry-run, everything goes to review
|
||||
```
|
||||
|
||||
| Flag | Default | Description |
|
||||
|------|---------|-------------|
|
||||
| `--autonomy` | `tiered` | `auto`, `tiered`, or `manual` |
|
||||
|
||||
### `jarvis learning history`
|
||||
|
||||
List past learning sessions.
|
||||
|
||||
```bash
|
||||
jarvis learning history
|
||||
jarvis learning history --limit 5
|
||||
```
|
||||
|
||||
### `jarvis learning show`
|
||||
|
||||
Show details of a learning session (diagnosis, plan, outcomes, cost).
|
||||
|
||||
```bash
|
||||
jarvis learning show <session-id>
|
||||
```
|
||||
|
||||
### `jarvis learning review`
|
||||
|
||||
List all pending edits awaiting approval.
|
||||
|
||||
```bash
|
||||
jarvis learning review
|
||||
```
|
||||
|
||||
### `jarvis learning approve`
|
||||
|
||||
Approve a pending edit (still goes through the benchmark gate).
|
||||
|
||||
```bash
|
||||
jarvis learning approve <edit-id>
|
||||
```
|
||||
|
||||
### `jarvis learning reject`
|
||||
|
||||
Reject a pending edit.
|
||||
|
||||
```bash
|
||||
jarvis learning reject <edit-id>
|
||||
jarvis learning reject <edit-id> --reason "too aggressive"
|
||||
```
|
||||
|
||||
### `jarvis learning rollback`
|
||||
|
||||
Rollback a session's committed edits (creates revert commits).
|
||||
|
||||
```bash
|
||||
jarvis learning rollback <session-id>
|
||||
jarvis learning rollback --last
|
||||
```
|
||||
|
||||
### `jarvis learning benchmark`
|
||||
|
||||
Personal benchmark management.
|
||||
|
||||
```bash
|
||||
jarvis learning benchmark show # current stats
|
||||
jarvis learning benchmark refresh # manual refresh
|
||||
```
|
||||
|
||||
### `jarvis learning daemon`
|
||||
|
||||
Background learning daemon.
|
||||
|
||||
```bash
|
||||
jarvis learning daemon start
|
||||
jarvis learning daemon stop
|
||||
jarvis learning daemon status
|
||||
```
|
||||
LLM-guided spec search (the frontier-driven harness-learning subsystem)
|
||||
is exposed as a Python library only — there is currently no top-level
|
||||
`jarvis` subcommand for it. Construct a `SpecSearchOrchestrator`
|
||||
directly from `openjarvis.learning.spec_search.orchestrator` and call
|
||||
`.run(trigger)` with a trigger from
|
||||
`openjarvis.learning.spec_search.triggers`. See
|
||||
[`docs/user-guide/llm-guided-spec-search.md`](llm-guided-spec-search.md)
|
||||
for the architecture and the building blocks
|
||||
(`splits.py`, external corpora, `external_adapter`).
|
||||
|
||||
+193
-56
@@ -1,39 +1,55 @@
|
||||
# Evaluations
|
||||
|
||||
The OpenJarvis evaluation framework (`openjarvis-evals`) measures model **correctness and accuracy** on academic datasets. It is a separate package from the main OpenJarvis library and is designed specifically for research workflows where you need reproducible, dataset-driven quality assessments.
|
||||
The OpenJarvis evaluation framework (`openjarvis.evals`) measures model **correctness and accuracy** on academic datasets. It ships inside the main `openjarvis` package (at `src/openjarvis/evals/`) and is designed specifically for research workflows where you need reproducible, dataset-driven quality assessments.
|
||||
|
||||
!!! info "Evals vs. Benchmarks"
|
||||
OpenJarvis has two distinct measurement systems that complement each other:
|
||||
|
||||
| System | Package | Measures | Entry Point |
|
||||
|--------|---------|----------|-------------|
|
||||
| **Evaluations** | `openjarvis-evals` | Correctness on academic datasets (accuracy, pass rate) | `openjarvis-eval` |
|
||||
| **Benchmarks** | `openjarvis` | Engine performance (latency, throughput) | `jarvis bench` |
|
||||
| System | Module | Measures | Entry Point |
|
||||
|--------|--------|----------|-------------|
|
||||
| **Evaluations** | `openjarvis.evals` | Correctness on academic datasets (accuracy, pass rate) | `jarvis eval` |
|
||||
| **Benchmarks** | `openjarvis.bench` | Engine performance (latency, throughput) | `jarvis bench` |
|
||||
|
||||
Use evaluations to answer "does this model get the right answer?" and benchmarks to answer "how fast does this model respond?". See the [Benchmarks guide](benchmarks.md) for the performance measurement system.
|
||||
|
||||
---
|
||||
|
||||
> **Tip:** The distillation system uses this same eval infrastructure to gate edits against your personal benchmark. See [Learning & Distillation](learning-distillation.md).
|
||||
> **Tip:** LLM-guided spec search uses this same eval infrastructure to gate edits against your personal benchmark. See [LLM-guided spec search](llm-guided-spec-search.md).
|
||||
|
||||
## Installation
|
||||
|
||||
The evaluation framework is a standalone package in the `evals/` directory. Install it alongside OpenJarvis:
|
||||
The evaluation framework is part of the main `openjarvis` package — no separate install or extra is required. The standard dev setup is enough:
|
||||
|
||||
```bash
|
||||
uv sync --extra eval
|
||||
uv sync --extra dev
|
||||
```
|
||||
|
||||
This installs the `openjarvis-eval` CLI entry point and all required dependencies (`datasets`, `huggingface-hub`, `tqdm`, `rich`).
|
||||
The framework's core dependencies (`click`, `datasets`, `rich`) are base dependencies of `openjarvis`. Two optional extras enable experiment tracking integrations:
|
||||
|
||||
```bash
|
||||
uv sync --extra dev --extra eval-wandb # Weights & Biases run tracking
|
||||
uv sync --extra dev --extra eval-sheets # Google Sheets results export
|
||||
```
|
||||
|
||||
!!! note "Python version requirement"
|
||||
Python 3.10 requires the `tomli` package for TOML config parsing. The `evals/pyproject.toml` includes this as a conditional dependency, so it is installed automatically.
|
||||
Python 3.10 requires the `tomli` package for TOML config parsing. `openjarvis` declares it as a conditional dependency, so it is installed automatically.
|
||||
|
||||
## Entry Points
|
||||
|
||||
Two equivalent entry points expose the framework:
|
||||
|
||||
| Command | Surface |
|
||||
|---------|---------|
|
||||
| `jarvis eval {list,run,compare,report}` | Canonical CLI. `run` covers the common options; `compare` and `report` post-process result files. |
|
||||
| `python -m openjarvis.evals {list,run,run-all,summarize,reparse-judge}` | Full research surface, including judge configuration, the agentic runner, and episode mode. |
|
||||
|
||||
The `openjarvis-eval` console script is an alias for `python -m openjarvis.evals` — same commands, same options. This guide uses `jarvis eval` wherever its option set suffices and the module form for research-only options.
|
||||
|
||||
---
|
||||
|
||||
## Datasets
|
||||
|
||||
The framework ships with **30+ datasets** covering academic reasoning, agentic tasks, retrieval, conversation quality, and practical use-case benchmarks. Datasets are grouped by category below.
|
||||
The framework ships with **40 registered benchmarks** covering academic reasoning, agentic tasks, coding, retrieval, conversation quality, and practical use-case benchmarks. Datasets are grouped by category below; `uv run python -m openjarvis.evals list` prints the authoritative registry.
|
||||
|
||||
### Use-Case Benchmarks
|
||||
|
||||
@@ -64,6 +80,7 @@ These benchmarks measure reasoning and knowledge on established academic dataset
|
||||
| **MATH-500** | `math500` | reasoning | Competition-level math problems |
|
||||
| **NaturalReasoning** | `natural-reasoning` | reasoning | Natural language reasoning |
|
||||
| **HLE** | `hle` | reasoning | Humanity's Last Exam hard challenges |
|
||||
| **LiveResearchBench** | `liveresearchbench` | reasoning | Recent research comprehension (Salesforce) |
|
||||
| **SimpleQA** | `simpleqa` | chat | Short-form factual question answering |
|
||||
| **IPW** | `ipw` | chat | Intelligence Per Watt mixed benchmark |
|
||||
|
||||
@@ -78,6 +95,12 @@ These benchmarks test multi-step agent capabilities including tool use, code gen
|
||||
| **SWEfficiency** | `swefficiency` | agentic | Software optimization tasks |
|
||||
| **TerminalBench** | `terminalbench` | agentic | Terminal-based task completion |
|
||||
| **TerminalBench Native** | `terminalbench-native` | agentic | TerminalBench with native Docker execution |
|
||||
| **TerminalBench V2.1** | `terminalbench-v2.1` | agentic | TB v2.1 Harbor-style Docker tasks |
|
||||
| **PinchBench** | `pinchbench` | agentic | Real-world agent tasks |
|
||||
| **TauBench** | `taubench` | agentic | Multi-turn customer service |
|
||||
| **DeepResearchBench** | `liveresearch` | agentic | Deep research report generation |
|
||||
| **DeepResearchBench (alias)** | `deepresearch` | agentic | Same benchmark as `liveresearch` |
|
||||
| **ToolCall-15** | `toolcall15` | agentic | Tool calling benchmark |
|
||||
| **LifelongAgent** | `lifelong-agent` | agentic | Sequential task learning across sessions |
|
||||
| **PaperArena** | `paperarena` | agentic | Scientific paper analysis |
|
||||
| **DeepPlanning** | `deepplanning` | agentic | Shopping constraint planning |
|
||||
@@ -86,6 +109,14 @@ These benchmarks test multi-step agent capabilities including tool use, code gen
|
||||
| **WebChoreArena** | `webchorearena` | agentic | Web chore tasks |
|
||||
| **WorkArena** | `workarena` | agentic | WorkArena++ enterprise workflows |
|
||||
|
||||
Both `liveresearch` and `deepresearch` are registered keys for the DeepResearchBench report-generation benchmark.
|
||||
|
||||
### Coding Benchmarks
|
||||
|
||||
| Dataset | Key | Category | Description |
|
||||
|---------|-----|----------|-------------|
|
||||
| **LiveCodeBench** | `livecodebench` | coding | Competitive programming |
|
||||
|
||||
### Retrieval Benchmarks
|
||||
|
||||
| Dataset | Key | Category | Description |
|
||||
@@ -122,7 +153,7 @@ The framework includes two pre-built configs for evaluating models on the five c
|
||||
### Cloud models
|
||||
|
||||
```bash
|
||||
uv run python -m openjarvis.evals --config src/openjarvis/evals/configs/use_case_v2_cloud.toml
|
||||
uv run jarvis eval run --config src/openjarvis/evals/configs/use_case_v2_cloud.toml
|
||||
```
|
||||
|
||||
This config evaluates **6 cloud models** (Claude Opus 4.6, Claude Haiku 4.5, Gemini 3.1 Pro, Gemini 3.1 Flash Lite, GPT-5.4, GPT-5 Mini) against all 5 use-case benchmarks with 30 samples each, producing a 6x5 = 30-run matrix. Results are written to `results/use-cases-v2-cloud/`.
|
||||
@@ -130,7 +161,7 @@ This config evaluates **6 cloud models** (Claude Opus 4.6, Claude Haiku 4.5, Gem
|
||||
### Local models
|
||||
|
||||
```bash
|
||||
uv run python -m openjarvis.evals --config src/openjarvis/evals/configs/use_case_v2_local.toml
|
||||
uv run jarvis eval run --config src/openjarvis/evals/configs/use_case_v2_local.toml
|
||||
```
|
||||
|
||||
This config evaluates **5 local models** via Ollama (Qwen3.5 122B-A10B, GPT-OSS 120B, GLM4, Qwen3.5 35B-A3B, GLM-4.7-Flash) against the same 5 benchmarks, producing a 5x5 = 25-run matrix. Uses 2 workers (suitable for single-GPU setups). Results are written to `results/use-cases-v2-local/`.
|
||||
@@ -142,15 +173,22 @@ This config evaluates **5 local models** via Ollama (Qwen3.5 122B-A10B, GPT-OSS
|
||||
|
||||
## Inference Backends
|
||||
|
||||
Every evaluation run routes model calls through one of two backends:
|
||||
Every evaluation run routes model calls through one of four backends:
|
||||
|
||||
| Backend | Key | Description |
|
||||
|---------|-----|-------------|
|
||||
| **jarvis-direct** | `jarvis-direct` | Engine-level inference via `SystemBuilder`. Works for local (Ollama, vLLM, llama.cpp) and cloud models. |
|
||||
| **jarvis-agent** | `jarvis-agent` | Agent-level inference with tool calling. Uses `JarvisSystem.ask()` with the specified agent and tools. |
|
||||
| **hermes** | `hermes` | Real Hermes Agent (Nous Research) via subprocess. Requires `--base-url` and `--api-key`. |
|
||||
| **openclaw** | `openclaw` | Real OpenClaw via Node subprocess. Requires `--base-url` and `--api-key`. |
|
||||
|
||||
Use `jarvis-direct` for most evaluations. Use `jarvis-agent` when the benchmark requires tool use — for example, GAIA tasks that reference files that must be read with `file_read`, or arithmetic tasks that benefit from `calculator`.
|
||||
|
||||
The `hermes` and `openclaw` backends shell out to external agent frameworks and need an OpenAI-compatible endpoint for their model calls: pass `--base-url`/`--api-key`, set the `JARVIS_BACKEND_BASE_URL`/`JARVIS_BACKEND_API_KEY` environment variables, or add a `[backend.external]` section to your config (see [Config Reference](#backendexternal)).
|
||||
|
||||
!!! note "TerminalBench Native"
|
||||
`jarvis eval run --backend` additionally accepts `terminalbench-native`, a Docker-based execution backend used by the TerminalBench Native benchmark.
|
||||
|
||||
---
|
||||
|
||||
## CLI Usage
|
||||
@@ -158,73 +196,106 @@ Use `jarvis-direct` for most evaluations. Use `jarvis-agent` when the benchmark
|
||||
### List available benchmarks and backends
|
||||
|
||||
```bash
|
||||
openjarvis-eval list
|
||||
uv run python -m openjarvis.evals list
|
||||
```
|
||||
|
||||
Output:
|
||||
Abridged output (40 benchmarks, 4 backends):
|
||||
|
||||
```
|
||||
Benchmarks:
|
||||
supergpqa [reasoning ] SuperGPQA multiple-choice
|
||||
gaia [agentic ] GAIA agentic benchmark
|
||||
frames [rag ] FRAMES multi-hop RAG
|
||||
wildchat [chat ] WildChat conversation quality
|
||||
|
||||
Backends:
|
||||
jarvis-direct Engine-level inference (local or cloud)
|
||||
jarvis-agent Agent-level inference with tool calling
|
||||
Available Benchmarks
|
||||
┌──────────────────────┬───────────┬───────────────────────────────────┐
|
||||
│ Name │ Category │ Description │
|
||||
├──────────────────────┼───────────┼───────────────────────────────────┤
|
||||
│ supergpqa │ reasoning │ SuperGPQA multiple-choice │
|
||||
│ gpqa │ reasoning │ GPQA graduate-level MCQ │
|
||||
│ ... │ ... │ ... │
|
||||
│ livecodebench │ coding │ LiveCodeBench competitive progr. │
|
||||
│ toolcall15 │ agentic │ ToolCall-15 tool calling benchmark│
|
||||
└──────────────────────┴───────────┴───────────────────────────────────┘
|
||||
Available Backends
|
||||
┌───────────────┬──────────────────────────────────────────────────┐
|
||||
│ jarvis-direct │ Engine-level inference (local or cloud) │
|
||||
│ jarvis-agent │ Agent-level inference with tool calling │
|
||||
│ hermes │ Real Hermes Agent (Nous Research) via subprocess │
|
||||
│ openclaw │ Real OpenClaw via Node subprocess │
|
||||
└───────────────┴──────────────────────────────────────────────────┘
|
||||
```
|
||||
|
||||
`jarvis eval list` prints a similar table but currently shows a curated subset of the registry; the module form above is the authoritative listing.
|
||||
|
||||
### Run a single benchmark
|
||||
|
||||
```bash
|
||||
# Evaluate qwen3:8b on SuperGPQA (engine-level, 10 samples default)
|
||||
openjarvis-eval run -b supergpqa -m qwen3:8b
|
||||
# Evaluate qwen3:8b on SuperGPQA (engine-level, 10 samples)
|
||||
uv run jarvis eval run -b supergpqa -m qwen3:8b -n 10
|
||||
|
||||
# Evaluate GPT-4o on GAIA using the agent backend with tools
|
||||
openjarvis-eval run -b gaia -m gpt-4o --backend jarvis-agent \
|
||||
# Evaluate GPT-5 Mini on GAIA using the agent backend with tools
|
||||
uv run jarvis eval run -b gaia -m gpt-5-mini --backend jarvis-agent \
|
||||
--agent orchestrator --tools calculator,file_read -n 50
|
||||
|
||||
# Run FRAMES with vLLM engine, write output to a file
|
||||
openjarvis-eval run -b frames -m llama3:70b -e vllm \
|
||||
# Run FRAMES with the vLLM engine, write output to a file
|
||||
uv run jarvis eval run -b frames -m llama3:70b -e vllm \
|
||||
-o results/frames_llama70b.jsonl
|
||||
|
||||
# Run WildChat with a higher temperature for chat quality
|
||||
openjarvis-eval run -b wildchat -m qwen3:8b --temperature 0.7 -n 100
|
||||
uv run jarvis eval run -b wildchat -m qwen3:8b --temperature 0.7 -n 100
|
||||
```
|
||||
|
||||
#### Full option reference
|
||||
#### `jarvis eval run` option reference
|
||||
|
||||
| Option | Short | Type | Default | Description |
|
||||
|--------|-------|------|---------|-------------|
|
||||
| `--config` | `-c` | path | — | TOML config file; when provided, `-b` and `-m` are not required |
|
||||
| `--benchmark` | `-b` | choice | required* | `supergpqa`, `gaia`, `frames`, or `wildchat` |
|
||||
| `--backend` | | choice | `jarvis-direct` | `jarvis-direct` or `jarvis-agent` |
|
||||
| `--model` | `-m` | str | required* | Model identifier (e.g., `qwen3:8b`, `gpt-4o`) |
|
||||
| `--engine` | `-e` | str | auto | Engine key (`ollama`, `vllm`, `cloud`, ...) |
|
||||
| `--agent` | | str | `orchestrator` | Agent name for `jarvis-agent` backend |
|
||||
| `--tools` | | str | `""` | Comma-separated tool names (e.g., `calculator,file_read`) |
|
||||
| `--benchmark` | `-b` | str | required* | Any registered benchmark key (see `... list`) |
|
||||
| `--model` | `-m` | str | required* | Model identifier (e.g., `qwen3:8b`, `gpt-5-mini`) |
|
||||
| `--max-samples` | `-n` | int | all | Limit the number of samples evaluated |
|
||||
| `--max-workers` | `-w` | int | `4` | Parallel evaluation workers |
|
||||
| `--judge-model` | | str | `gpt-4o` | LLM used for judge-based scoring |
|
||||
| `--output` | `-o` | path | auto-generated | Output JSONL file path |
|
||||
| `--backend` | | choice | `jarvis-direct` | `jarvis-direct`, `jarvis-agent`, `hermes`, `openclaw`, or `terminalbench-native` |
|
||||
| `--base-url` | | str | — | OpenAI-compatible endpoint URL (env: `JARVIS_BACKEND_BASE_URL`) |
|
||||
| `--api-key` | | str | — | API key for the endpoint (env: `JARVIS_BACKEND_API_KEY`) |
|
||||
| `--agent` | | str | — | Agent name for `jarvis-agent` backend (e.g., `orchestrator`) |
|
||||
| `--engine` | `-e` | str | auto | Engine key (`ollama`, `vllm`, `cloud`, ...) |
|
||||
| `--tools` | | str | `""` | Comma-separated tool names (e.g., `calculator,file_read`) |
|
||||
| `--telemetry/--no-telemetry` | | flag | off | Enable telemetry collection during eval |
|
||||
| `--gpu-metrics/--no-gpu-metrics` | | flag | off | Enable GPU metric polling |
|
||||
| `--seed` | | int | `42` | Random seed for dataset shuffling |
|
||||
| `--split` | | str | dataset default | Override the dataset split |
|
||||
| `--temperature` | | float | `0.0` | Generation temperature |
|
||||
| `--max-tokens` | | int | `2048` | Maximum output tokens |
|
||||
| `--model-filter` | | str | — | Filter models by name substring (multi-model configs) |
|
||||
| `--output` | `-o` | path | auto-generated | Output JSONL file path |
|
||||
| `--wandb-project` / `--wandb-entity` / `--wandb-tags` / `--wandb-group` | | str | `""` | Weights & Biases tracking (requires `eval-wandb` extra) |
|
||||
| `--sheets-id` / `--sheets-worksheet` / `--sheets-creds` | | str | `""` | Google Sheets export (requires `eval-sheets` extra) |
|
||||
| `--verbose` | `-v` | flag | off | Enable debug logging |
|
||||
|
||||
*Required when `--config` is not provided.
|
||||
|
||||
#### Research-only options (`python -m openjarvis.evals run`)
|
||||
|
||||
The module CLI accepts everything above plus research-grade options that `jarvis eval run` does not expose:
|
||||
|
||||
| Option | Short | Type | Default | Description |
|
||||
|--------|-------|------|---------|-------------|
|
||||
| `--max-workers` | `-w` | int | `4` | Parallel evaluation workers |
|
||||
| `--judge-model` | | str | `gpt-5-mini-2025-08-07` | LLM used for judge-based scoring (see `--help` for the current default) |
|
||||
| `--judge-engine` | | str | `cloud` | Engine key for the LLM judge; use `vllm` to judge locally |
|
||||
| `--split` | | str | dataset default | Override the dataset split |
|
||||
| `--compact` | | flag | off | Dense single-table output |
|
||||
| `--trace-detail` | | flag | off | Full per-step trace listing |
|
||||
| `--agentic` | | flag | off | Use `AgenticRunner` for multi-turn agent execution |
|
||||
| `--episode-mode` | | flag | off | Sequential episode processing with lifelong learning (required for `lifelong-agent` and similar benchmarks) |
|
||||
| `--concurrency` | | int | `1` | Parallel query execution (AgenticRunner only) |
|
||||
| `--query-timeout` | | float | — | Per-query wall-clock timeout in seconds (AgenticRunner only) |
|
||||
|
||||
Note: the module CLI's `--backend` choice covers `jarvis-direct`, `jarvis-agent`, `hermes`, and `openclaw`; `terminalbench-native` as a backend is available via `jarvis eval run` and TOML configs.
|
||||
|
||||
### Run all benchmarks at once
|
||||
|
||||
The `run-all` command evaluates a single model against all four benchmarks sequentially and writes results to an output directory:
|
||||
The `run-all` command (module CLI only) evaluates a single model against **every registered benchmark** sequentially and writes results to an output directory:
|
||||
|
||||
```bash
|
||||
openjarvis-eval run-all -m qwen3:8b
|
||||
uv run python -m openjarvis.evals run-all -m qwen3:8b
|
||||
|
||||
# With options
|
||||
openjarvis-eval run-all -m gpt-4o -n 100 --output-dir results/gpt4o/
|
||||
uv run python -m openjarvis.evals run-all -m gpt-5-mini -n 100 --output-dir results/gpt5mini/
|
||||
```
|
||||
|
||||
Output files are written as `{output_dir}/{benchmark}_{model-slug}.jsonl`. The model slug replaces `/` and `:` with `-`, so `qwen3:8b` becomes `qwen3-8b`.
|
||||
@@ -234,7 +305,7 @@ Output files are written as `{output_dir}/{benchmark}_{model-slug}.jsonl`. The m
|
||||
After a run, inspect a JSONL results file:
|
||||
|
||||
```bash
|
||||
openjarvis-eval summarize results/supergpqa_qwen3-8b.jsonl
|
||||
uv run python -m openjarvis.evals summarize results/supergpqa_qwen3-8b.jsonl
|
||||
```
|
||||
|
||||
Output:
|
||||
@@ -250,6 +321,55 @@ Accuracy: 0.7222
|
||||
Errors: 2
|
||||
```
|
||||
|
||||
The module CLI also provides `reparse-judge`, which re-parses stored judge output in a results file and recovers records whose judge verdicts initially failed to parse — useful after improving the judge-output parser without re-running inference.
|
||||
|
||||
### Compare and report
|
||||
|
||||
`jarvis eval` adds two post-processing commands for result files:
|
||||
|
||||
```bash
|
||||
# Side-by-side metric comparison across runs
|
||||
uv run jarvis eval compare results/supergpqa_qwen3-8b.jsonl results/supergpqa_gpt-5-mini.jsonl
|
||||
|
||||
# Detailed report (accuracy, latency, cost, per-subject breakdown) for one run
|
||||
uv run jarvis eval report results/supergpqa_qwen3-8b.jsonl
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Evaluating an Already-Running Endpoint
|
||||
|
||||
If you already have an OpenAI-compatible server running — `jarvis serve`, vLLM, SGLang, llama.cpp's server, or a hosted endpoint — point an eval directly at it with `--base-url` and `--api-key`:
|
||||
|
||||
```bash
|
||||
# A vLLM server is already serving Qwen/Qwen3-8B on a GPU node:
|
||||
# vllm serve Qwen/Qwen3-8B --port 8000
|
||||
uv run jarvis eval run -b supergpqa -m Qwen/Qwen3-8B \
|
||||
--base-url http://gpu-node:8000/v1 \
|
||||
--api-key local-key \
|
||||
-n 50
|
||||
```
|
||||
|
||||
The `-m` value must match a model id the server reports at `GET /v1/models`. Both flags fall back to the `JARVIS_BACKEND_BASE_URL` and `JARVIS_BACKEND_API_KEY` environment variables, so CI jobs can set them once:
|
||||
|
||||
```bash
|
||||
export JARVIS_BACKEND_BASE_URL=http://gpu-node:8000/v1
|
||||
export JARVIS_BACKEND_API_KEY=local-key
|
||||
uv run jarvis eval run -b gaia -m Qwen/Qwen3-8B --backend jarvis-agent -n 25
|
||||
```
|
||||
|
||||
For the external `hermes` and `openclaw` backends these values are **required** (the foreign frameworks need an endpoint to send model calls to).
|
||||
|
||||
!!! tip "Engine-level alternative for vLLM"
|
||||
The vLLM engine also honors the `VLLM_HOST` environment variable (default `http://localhost:8000`):
|
||||
|
||||
```bash
|
||||
VLLM_HOST=http://gpu-node:8000 uv run python -m openjarvis.evals run \
|
||||
-b supergpqa -m Qwen/Qwen3-8B -e vllm -n 50
|
||||
```
|
||||
|
||||
`VLLM_HOST` is process-global — if the candidate and the judge both use the `vllm` engine, they share the same endpoint. Prefer `--base-url` when you need them separate.
|
||||
|
||||
---
|
||||
|
||||
## TOML Config System
|
||||
@@ -259,7 +379,7 @@ For research workflows that compare multiple models across multiple benchmarks,
|
||||
### Running from a config
|
||||
|
||||
```bash
|
||||
openjarvis-eval run --config src/openjarvis/evals/configs/full-suite.toml
|
||||
uv run jarvis eval run --config src/openjarvis/evals/configs/full-suite.toml
|
||||
```
|
||||
|
||||
When `--config` is provided, the `-b`/`--benchmark` and `-m`/`--model` options are not required. All settings come from the config file. The CLI expands the matrix, prints a progress table, and writes results to the configured `output_dir`.
|
||||
@@ -268,7 +388,7 @@ When `--config` is provided, the `-b`/`--benchmark` and `-m`/`--model` options a
|
||||
|
||||
A config file has six sections: `[meta]`, `[defaults]`, `[judge]`, `[run]`, `[[models]]`, and `[[benchmarks]]`. Only `[[models]]` and `[[benchmarks]]` are required — all other sections are optional and fall back to built-in defaults.
|
||||
|
||||
```toml title="evals/configs/full-suite.toml"
|
||||
```toml title="src/openjarvis/evals/configs/full-suite.toml"
|
||||
# Suite-level metadata (optional)
|
||||
[meta]
|
||||
name = "full-suite-v1"
|
||||
@@ -352,7 +472,7 @@ For example, `temperature` is resolved as: use `[defaults].temperature` (0.0), t
|
||||
|
||||
A config requires only one `[[models]]` and one `[[benchmarks]]` entry:
|
||||
|
||||
```toml title="evals/configs/minimal.toml"
|
||||
```toml title="src/openjarvis/evals/configs/minimal.toml"
|
||||
[[models]]
|
||||
name = "qwen3:8b"
|
||||
|
||||
@@ -364,7 +484,7 @@ This runs SuperGPQA against qwen3:8b with all default settings. Use this as a st
|
||||
|
||||
### Single-run config with full options
|
||||
|
||||
```toml title="evals/configs/single-run.toml"
|
||||
```toml title="src/openjarvis/evals/configs/single-run.toml"
|
||||
[meta]
|
||||
name = "single-run-example"
|
||||
description = "Evaluate SuperGPQA with a single model and full configuration"
|
||||
@@ -424,7 +544,8 @@ Configuration for the LLM used as a judge in GAIA, FRAMES, and WildChat scoring.
|
||||
|
||||
| Field | Type | Default | Description |
|
||||
|-------|------|---------|-------------|
|
||||
| `model` | str | `"gpt-4o"` | Judge model identifier |
|
||||
| `model` | str | `"gpt-5-mini-2025-08-07"` | Judge model identifier |
|
||||
| `engine` | str | `None` | Engine key for the judge (e.g., `"vllm"` to judge locally; defaults to cloud) |
|
||||
| `provider` | str | `None` | Provider override (e.g., `"openai"`) |
|
||||
| `temperature` | float | `0.0` | Judge sampling temperature |
|
||||
| `max_tokens` | int | `1024` | Maximum judge output tokens |
|
||||
@@ -443,6 +564,20 @@ Execution settings that apply to the entire suite.
|
||||
| `seed` | int | `42` | Random seed for dataset shuffling |
|
||||
| `telemetry` | bool | `false` | Enable GPU telemetry capture (energy, power, utilization, throughput) |
|
||||
| `gpu_metrics` | bool | `false` | Enable GPU metric polling via `pynvml` (requires `pynvml` or `nvidia-ml-py`) |
|
||||
| `warmup_samples` | int | `0` | Untimed warmup samples before measurement |
|
||||
| `energy_vendor` | str | `""` | GPU energy vendor override |
|
||||
| `max_turns` | int | `None` | Maximum agent turns per query |
|
||||
| `wandb_project` / `wandb_entity` / `wandb_tags` / `wandb_group` | str | `""` | Weights & Biases tracking |
|
||||
| `sheets_spreadsheet_id` / `sheets_worksheet` / `sheets_credentials_path` | str | `""` / `"Results"` / `""` | Google Sheets export |
|
||||
|
||||
### `[backend.external]`
|
||||
|
||||
Endpoint settings for the `hermes` and `openclaw` backends. Environment variables override TOML values.
|
||||
|
||||
| Field | Type | Default | Description |
|
||||
|-------|------|---------|-------------|
|
||||
| `base_url` | str | `None` | OpenAI-compatible endpoint URL (env: `JARVIS_BACKEND_BASE_URL`) |
|
||||
| `api_key` | str | `None` | API key for the endpoint (env: `JARVIS_BACKEND_API_KEY`) |
|
||||
|
||||
### `[[models]]`
|
||||
|
||||
@@ -450,7 +585,7 @@ One block per model. The `name` field is required.
|
||||
|
||||
| Field | Type | Default | Description |
|
||||
|-------|------|---------|-------------|
|
||||
| `name` | str | required | Model identifier (e.g., `"qwen3:8b"`, `"gpt-4o"`) |
|
||||
| `name` | str | required | Model identifier (e.g., `"qwen3:8b"`, `"gpt-5-mini"`) |
|
||||
| `engine` | str | `None` | Engine key to use (`"ollama"`, `"vllm"`, `"cloud"`, ...) |
|
||||
| `provider` | str | `None` | Provider override for cloud models (e.g., `"openai"`) |
|
||||
| `temperature` | float | `None` | Override `[defaults].temperature` for this model |
|
||||
@@ -467,10 +602,12 @@ One block per benchmark. The `name` field is required.
|
||||
|
||||
| Field | Type | Default | Description |
|
||||
|-------|------|---------|-------------|
|
||||
| `name` | str | required | Benchmark key: `supergpqa`, `gaia`, `frames`, or `wildchat` |
|
||||
| `backend` | str | `"jarvis-direct"` | Inference backend: `jarvis-direct` or `jarvis-agent` |
|
||||
| `name` | str | required | Any registered benchmark key (see `uv run python -m openjarvis.evals list`) |
|
||||
| `backend` | str | `"jarvis-direct"` | `jarvis-direct`, `jarvis-agent`, `hermes`, `openclaw`, or `terminalbench-native` |
|
||||
| `max_samples` | int | `None` | Limit number of samples; `None` evaluates the full dataset |
|
||||
| `split` | str | `None` | Override the default dataset split |
|
||||
| `subset` | str | `None` | Dataset subset/variant (benchmark-specific) |
|
||||
| `record_ids` | list[str] | `None` | Evaluate only these record ids |
|
||||
| `agent` | str | `None` | Agent name for `jarvis-agent` backend (e.g., `"orchestrator"`) |
|
||||
| `tools` | list[str] | `[]` | Tool names for `jarvis-agent` backend |
|
||||
| `judge_model` | str | `None` | Override `[judge].model` for this benchmark only |
|
||||
@@ -646,7 +783,7 @@ The `EvalRunner` processes samples concurrently using a `ThreadPoolExecutor`. Re
|
||||
|
||||
```bash
|
||||
# Use more workers for faster evaluation (if the engine supports concurrent requests)
|
||||
openjarvis-eval run -b supergpqa -m qwen3:8b -w 8 -n 500
|
||||
uv run python -m openjarvis.evals run -b supergpqa -m qwen3:8b -w 8 -n 500
|
||||
```
|
||||
|
||||
!!! warning "Worker count and engine load"
|
||||
|
||||
@@ -1,251 +0,0 @@
|
||||
# Learning & Distillation
|
||||
|
||||
Use a frontier model as a meta-engineer to automatically improve your local agent's prompts, routing, and tools — reversibly, with benchmark-gated quality control.
|
||||
|
||||
## Quick Start
|
||||
|
||||
### 1. Initialize
|
||||
|
||||
```bash
|
||||
jarvis learning init
|
||||
```
|
||||
|
||||
This creates the distillation directory layout under `~/.openjarvis/learning/` and initializes a git checkpoint repo at `~/.openjarvis/.git` for tracking config changes.
|
||||
|
||||
### 2. Run your first session
|
||||
|
||||
Once you have at least 20 traces from regular use:
|
||||
|
||||
```bash
|
||||
jarvis learning run
|
||||
```
|
||||
|
||||
The system will:
|
||||
1. **Diagnose** — analyze your traces using a frontier model
|
||||
2. **Plan** — propose typed edits to your config
|
||||
3. **Execute** — apply edits that pass the benchmark gate
|
||||
4. **Record** — persist the session for history and rollback
|
||||
|
||||
### 3. Check results
|
||||
|
||||
```bash
|
||||
jarvis learning history
|
||||
jarvis learning show <session-id>
|
||||
```
|
||||
|
||||
## How a Learning Session Works
|
||||
|
||||
A learning session has four phases:
|
||||
|
||||
### Phase 1: Diagnose
|
||||
|
||||
A frontier model (the "teacher", default `claude-opus-4-6`) analyzes your recent traces using read-only diagnostic tools. It identifies **failure clusters** — groups of related failures with shared root causes. The teacher must actually re-run your student on sample tasks and compare outputs to populate failure rates. This forces evidence-based diagnosis.
|
||||
|
||||
**Output:** `diagnosis.md` with narrative analysis + structured failure clusters.
|
||||
|
||||
### Phase 2: Plan
|
||||
|
||||
A second teacher call converts the diagnosis into a typed `LearningPlan` — a list of `Edit` objects, each targeting a specific part of your configuration (model routing, system prompts, tool availability, etc.). The teacher cannot pick risk tiers — those are assigned deterministically from a lookup table.
|
||||
|
||||
**Output:** `plan.json` frozen and immutable.
|
||||
|
||||
### Phase 3: Execute
|
||||
|
||||
Each edit is applied through its registered `EditApplier`, then scored against your personal benchmark. Edits that improve the benchmark are committed; edits that cause regressions are rolled back. Edits in the `review` tier are queued for your approval instead of being auto-applied.
|
||||
|
||||
**Output:** Git commits in the checkpoint repo + `EditOutcome` records.
|
||||
|
||||
### Phase 4: Record
|
||||
|
||||
The session is persisted to `learning.db` (SQLite index) and `session.json` (authoritative artifact). You can query history, show details, and rollback any session.
|
||||
|
||||
## Configuration
|
||||
|
||||
Add to `~/.openjarvis/config.toml`:
|
||||
|
||||
```toml
|
||||
[learning.distillation]
|
||||
enabled = true # gate the entire subsystem
|
||||
autonomy_mode = "tiered" # auto | tiered | manual
|
||||
teacher_model = "claude-opus-4-6" # any CloudEngine-supported model
|
||||
max_cost_per_session_usd = 5.0 # per-session teacher API budget
|
||||
max_tool_calls_per_diagnosis = 30 # max teacher tool calls in diagnosis
|
||||
```
|
||||
|
||||
### Trigger configuration
|
||||
|
||||
```toml
|
||||
[learning.distillation.triggers]
|
||||
scheduled_enabled = true
|
||||
scheduled_cron = "0 3 * * *" # daily at 03:00 local
|
||||
scheduled_min_new_traces = 20 # minimum new traces to trigger
|
||||
|
||||
cluster_enabled = true
|
||||
cluster_check_interval_minutes = 60
|
||||
cluster_min_size = 5
|
||||
cluster_failure_threshold = 0.3 # feedback <= this counts as failure
|
||||
```
|
||||
|
||||
### Gate configuration
|
||||
|
||||
```toml
|
||||
[learning.distillation.gate]
|
||||
min_improvement = 0.0 # any improvement accepted (raise for margin)
|
||||
max_regression = 0.05 # max per-cluster score drop
|
||||
benchmark_subsample_size = 50 # tasks per gate run
|
||||
full_benchmark = false # set true to disable subsampling
|
||||
```
|
||||
|
||||
### Benchmark configuration
|
||||
|
||||
```toml
|
||||
[learning.distillation.benchmark]
|
||||
synthesis_feedback_threshold = 0.7 # min feedback for benchmark traces
|
||||
max_benchmark_size = 200 # max tasks in the benchmark
|
||||
auto_refresh = true # auto-mine new high-feedback traces
|
||||
max_synthesis_cost_usd_per_refresh = 2.0 # separate from session budget
|
||||
```
|
||||
|
||||
### Risk tier overrides
|
||||
|
||||
Power users can override the default tier for any operation:
|
||||
|
||||
```toml
|
||||
[learning.distillation.tier_overrides]
|
||||
# Promote prompt edits to auto-apply after trust is established:
|
||||
# patch_system_prompt = "auto"
|
||||
# replace_system_prompt = "auto"
|
||||
```
|
||||
|
||||
## Risk Tiers
|
||||
|
||||
Every edit is assigned a risk tier that controls how it's applied:
|
||||
|
||||
| Tier | Behavior | Default ops |
|
||||
|------|----------|-------------|
|
||||
| **auto** | Applied automatically if benchmark gate passes | Model routing, model params, tool add/remove/description, agent params |
|
||||
| **review** | Queued for user approval in `jarvis learning review` | System prompt edits, agent class changes, few-shot exemplars |
|
||||
| **manual** | Never auto-applied; requires explicit approval | LoRA fine-tuning (v2) |
|
||||
|
||||
The tier is assigned deterministically from the edit operation — the teacher cannot override it.
|
||||
|
||||
## Reviewing Edits
|
||||
|
||||
When edits land in the review queue:
|
||||
|
||||
```bash
|
||||
# List all pending edits
|
||||
jarvis learning review
|
||||
|
||||
# Approve an edit (still goes through the benchmark gate)
|
||||
jarvis learning approve <edit-id>
|
||||
|
||||
# Reject an edit with a reason
|
||||
jarvis learning reject <edit-id> --reason "prompt change too aggressive"
|
||||
```
|
||||
|
||||
Even approved edits are gated by the benchmark — approval means "try it", not "force it".
|
||||
|
||||
## Rollback and History
|
||||
|
||||
Every edit creates a git commit in the checkpoint repo at `~/.openjarvis/.git`. This is separate from your OpenJarvis source repo.
|
||||
|
||||
```bash
|
||||
# List past sessions
|
||||
jarvis learning history --limit 20
|
||||
|
||||
# Show session details (diagnosis, plan, outcomes, cost)
|
||||
jarvis learning show <session-id>
|
||||
|
||||
# Rollback a session (creates new revert commits, preserves history)
|
||||
jarvis learning rollback <session-id>
|
||||
jarvis learning rollback --last
|
||||
```
|
||||
|
||||
Rollback never rewrites git history — it creates new revert commits so the audit trail stays intact.
|
||||
|
||||
## Cost Controls
|
||||
|
||||
Three cost boundaries prevent runaway spending:
|
||||
|
||||
1. **`max_cost_per_session_usd`** (default $5.00) — caps the total teacher API cost per session (diagnosis + planning).
|
||||
2. **`max_synthesis_cost_usd_per_refresh`** (default $2.00) — caps the cost of generating gold answers for new benchmark tasks. Separate from the session budget.
|
||||
3. **`teacher_model`** — choose a cheaper model (e.g., `claude-sonnet-4-6`) to reduce per-token costs at the expense of diagnosis quality.
|
||||
|
||||
Cost is tracked on every `LearningSession` as `teacher_cost_usd` and surfaced in `jarvis learning show`.
|
||||
|
||||
## The Personal Benchmark
|
||||
|
||||
The benchmark is your acceptance gate's source of truth — a set of tasks distilled from your high-quality traces, scored by an LLM-as-judge against frontier gold answers.
|
||||
|
||||
**How it's built:**
|
||||
1. Traces with feedback >= 0.7 are candidates
|
||||
2. Tasks are grouped by query class and deduplicated
|
||||
3. For each task, the teacher generates a gold reference answer
|
||||
4. The benchmark is versioned (`personal_v1.json`, `personal_v2.json`, ...)
|
||||
|
||||
**Auto-refresh:** The benchmark grows over time as you accumulate more traces. New tasks are added automatically during background refresh cycles.
|
||||
|
||||
```bash
|
||||
# Manual refresh
|
||||
jarvis learning benchmark refresh
|
||||
|
||||
# Show stats
|
||||
jarvis learning benchmark show
|
||||
```
|
||||
|
||||
## Cold Start: What to Expect on Day One
|
||||
|
||||
The system needs real usage data before it can learn:
|
||||
|
||||
- **< 20 traces:** `jarvis learning run` returns "Not enough traces yet." Triggers are no-ops.
|
||||
- **20+ traces, < 10 high-feedback:** Enough for diagnosis, but no benchmark yet. Sessions will run diagnosis but can't gate edits.
|
||||
- **10+ high-feedback traces:** Bootstrap benchmark is created automatically (`personal_v1.json`). Full learning loop is available.
|
||||
|
||||
**Getting there faster:** Use OpenJarvis normally and provide feedback on results (thumbs up/down in the UI, or `jarvis feedback` in the CLI).
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
| Error | Cause | Fix |
|
||||
|-------|-------|-----|
|
||||
| "Not enough traces yet" | Fewer than 20 traces in the store | Use OpenJarvis more, provide feedback |
|
||||
| "Working tree dirty, cannot stage" | Manual edits to `~/.openjarvis/config.toml` during a session | Commit or revert manual changes first |
|
||||
| "All clusters dropped: insufficient evidence" | Teacher diagnosed clusters but couldn't reproduce failures | Check that the student is actually failing on the flagged tasks |
|
||||
| "ConfigurationError: distillation root inside source tree" | `OPENJARVIS_HOME` points inside the repo | Set `OPENJARVIS_HOME` to `~/.openjarvis` (default) or another external dir |
|
||||
| "Personal benchmark is empty" | Not enough high-feedback traces yet | Provide feedback on 10+ traces with score >= 0.7 |
|
||||
|
||||
## Where Artifacts Live
|
||||
|
||||
All distillation artifacts live under `~/.openjarvis/` (never inside the source repo):
|
||||
|
||||
```
|
||||
~/.openjarvis/
|
||||
├── config.toml # Your configuration (git-tracked by checkpoint)
|
||||
├── agents/ # Agent prompts (git-tracked)
|
||||
├── tools/ # Tool descriptions (git-tracked)
|
||||
├── .git/ # Checkpoint repo for rollback
|
||||
└── learning/
|
||||
├── learning.db # SQLite session index
|
||||
├── benchmarks/ # Personal benchmark versions + gold answers
|
||||
├── sessions/ # Per-session artifacts (diagnosis, plan, traces)
|
||||
└── pending_review/ # Edits awaiting user approval
|
||||
```
|
||||
|
||||
## Background Daemon
|
||||
|
||||
For continuous learning:
|
||||
|
||||
```bash
|
||||
jarvis learning daemon start # Start background watcher
|
||||
jarvis learning daemon status # Check if running
|
||||
jarvis learning daemon stop # Stop the daemon
|
||||
```
|
||||
|
||||
The daemon runs the scheduled trigger (default: daily at 03:00) and the cluster trigger (watches for failure patterns in real-time).
|
||||
|
||||
## See Also
|
||||
|
||||
- [Architecture: Learning](../architecture/learning.md#distillation-frontier-driven-harness-learning) — internal architecture of the distillation subsystem
|
||||
- [User Guide: Evaluations](evaluations.md) — the eval infrastructure that powers the benchmark gate
|
||||
- [User Guide: CLI](cli.md#jarvis-learning) — full CLI reference
|
||||
- [Getting Started: Configuration](../getting-started/configuration.md) — all config knobs
|
||||
@@ -0,0 +1,214 @@
|
||||
# LLM-Guided Spec Search
|
||||
|
||||
LLM-guided spec search (Saad-Falcon et al., 2026) is a local–cloud
|
||||
collaboration: a frontier cloud *teacher* reads traces from a deployed
|
||||
local agent and proposes typed edits across the agent's full
|
||||
configuration; the local hardware runs the resulting configuration with
|
||||
zero marginal API cost at inference time. A held-out *gate* accepts only
|
||||
edits that improve a target failure cluster without unacceptable
|
||||
regression elsewhere.
|
||||
|
||||
This page is a copy-paste tutorial. By the end you will have:
|
||||
|
||||
- a real `SpecSearchOrchestrator` running on your machine,
|
||||
- a multi-session loop with the paper's stagnation rule (Algorithm 1),
|
||||
- an understanding of which knobs to turn for production deployment.
|
||||
|
||||
## TL;DR — run it
|
||||
|
||||
```bash
|
||||
python examples/openjarvis/spec_search_quickstart.py
|
||||
```
|
||||
|
||||
The script is self-contained (no API key, no Ollama) — it wires up real
|
||||
orchestrator + multi-session loop + composite-reward modules with stub
|
||||
teacher/student/judge so you can see one full session and a stagnation
|
||||
loop terminate. Production swap-points are commented inline; see
|
||||
[Going to production](#going-to-production) below.
|
||||
|
||||
## How it works
|
||||
|
||||
A search session repeats four phases (paper §3.3):
|
||||
|
||||
| Phase | What happens |
|
||||
|---|---|
|
||||
| **Diagnose** | Teacher reads eligible traces and groups failures into clusters, each annotated with `(student_failure_rate, teacher_success_rate, skill_gap)`. |
|
||||
| **Plan** | Teacher proposes typed edits across the four editable primitives (Intelligence, Engine, Agents, Tools & Memory). One proposal can edit multiple slots at once. |
|
||||
| **Execute** | Each candidate edit is applied; the gate scores the resulting spec on a held-out subsample. Accepted iff `GateOK` holds (see below). |
|
||||
| **Record** | Accepted edits commit to the checkpoint store; rejected edits roll back. The session is persisted to `SessionStore`. |
|
||||
|
||||
`SpecSearchOrchestrator.run(trigger)` runs **one** session end-to-end.
|
||||
`SpecSearchLoop` (paper Algorithm 1) wraps the orchestrator and repeats
|
||||
sessions until either gate-score stagnation (default *k* = 5 sessions)
|
||||
or budget exhaustion.
|
||||
|
||||
### `GateOK` — the acceptance predicate
|
||||
|
||||
Let `G_c(S)` be the held-out gate score of spec `S` restricted to
|
||||
failure cluster `c`. For an edit `e` targeting cluster `c`, with
|
||||
`S' = apply(S, e)`:
|
||||
|
||||
```
|
||||
GateOK(S', S, c, eps) ⟺
|
||||
G_c(S') > G_c(S) # target cluster improves, AND
|
||||
G_c'(S') >= G_c'(S) − eps # every other cluster regresses by ≤ eps
|
||||
```
|
||||
|
||||
Default `eps = 0.01` (1 %) per the paper. The `BenchmarkGate` class
|
||||
implements this; the `max_regression` knob is `eps`.
|
||||
|
||||
### Composite reward (Intelligence-edit training only)
|
||||
|
||||
When an Intelligence edit triggers LoRA / GRPO training inside the
|
||||
execute phase, candidate responses `y` to query `q` are scored by
|
||||
(paper Eq. 1):
|
||||
|
||||
```
|
||||
R(q, y) = α · R_acc(q, y)
|
||||
− β · Ê(q, y) # energy
|
||||
− γ · L̂(q, y) # latency
|
||||
− δ · Ĉ(q, y) # cost
|
||||
```
|
||||
|
||||
Defaults `(α, β, γ, δ) = (0.5, 0.1, 0.1, 0.3)`. The efficiency
|
||||
quantities (E, L, C) are z-scored *within batch* before weighting, so
|
||||
the reward trades dimensionless deviations rather than raw joules /
|
||||
seconds / dollars (paper Appendix C.6). Implementation:
|
||||
`openjarvis.learning.spec_search.composite_reward.score_batch`.
|
||||
|
||||
The held-out gate evaluates the resulting spec end-to-end; it is
|
||||
unaffected by these weights.
|
||||
|
||||
## Configuration
|
||||
|
||||
The prebuilt config lives at
|
||||
`configs/openjarvis/examples/spec-search-quickstart.toml`. Copy it to
|
||||
`~/.openjarvis/config.toml` (or set `OPENJARVIS_CONFIG` to it) and the
|
||||
regular loader picks it up:
|
||||
|
||||
```python
|
||||
from openjarvis.core.config import load_config
|
||||
cfg = load_config().learning.spec_search # SpecSearchLearningConfig
|
||||
```
|
||||
|
||||
The `[learning.spec_search]` table maps 1:1 onto the
|
||||
`SpecSearchLearningConfig` dataclass and is read by both
|
||||
`SpecSearchOrchestrator.from_config` and `SpecSearchLoop`:
|
||||
|
||||
```toml
|
||||
[learning.spec_search]
|
||||
enabled = true
|
||||
teacher_model = "claude-opus-4-6"
|
||||
teacher_engine = "cloud"
|
||||
autonomy_mode = "tiered" # auto | tiered | manual
|
||||
|
||||
# Per-session bounds
|
||||
min_traces = 20
|
||||
max_cost_per_session_usd = 5.0
|
||||
max_tool_calls_per_diagnosis = 30
|
||||
|
||||
# Multi-session loop (paper Algorithm 1)
|
||||
stagnation_k = 5 # paper default
|
||||
stagnation_eps = 0.001
|
||||
max_total_cost_usd = 50.0
|
||||
|
||||
# Gate (GateOK)
|
||||
max_regression = 0.01 # paper default: epsilon = 1%
|
||||
min_improvement = 0.0
|
||||
benchmark_subsample_size = 50
|
||||
benchmark_version = "personal_v1"
|
||||
|
||||
[learning.spec_search.composite_reward]
|
||||
alpha = 0.5 # accuracy
|
||||
beta = 0.1 # energy
|
||||
gamma = 0.1 # latency
|
||||
delta = 0.3 # cost
|
||||
```
|
||||
|
||||
## Going to production
|
||||
|
||||
The quickstart uses fakes for the teacher engine, student runner, and
|
||||
judge so it runs without external services. To run a real session,
|
||||
swap each fake for the corresponding production component:
|
||||
|
||||
| Slot | Quickstart | Production |
|
||||
|---|---|---|
|
||||
| `teacher_engine` | `FakeTeacherEngine` | `EngineRegistry.get(cfg.teacher_engine)(model=cfg.teacher_model)` — set `ANTHROPIC_API_KEY` etc. |
|
||||
| `trace_store` | `MagicMock` | `openjarvis.traces.store.TraceStore(home / "traces.db")` |
|
||||
| `student_runner` | `MagicMock` | `openjarvis.learning.spec_search.student_runner.VLLMStudentRunner(host=..., model=...)` |
|
||||
| `judge` | `MagicMock` | `openjarvis.evals.core.scorer.LLMJudgeScorer(...)` (or a deterministic scorer if your benchmark provides one) |
|
||||
| `session_store` | `MagicMock` | `openjarvis.learning.spec_search.storage.session_store.SessionStore(home / "learning" / "sessions.db")` |
|
||||
| `checkpoint_store` | `MagicMock` | `openjarvis.learning.spec_search.checkpoint.store.CheckpointStore(home / "learning" / "checkpoints")` |
|
||||
| `scorer` | climbing-plateau fake | a real `Scorer` callable (typically a `BenchmarkGate.score` adapter) |
|
||||
|
||||
The orchestrator only depends on the *interface* of each slot, not the
|
||||
concrete class — anything implementing the corresponding protocol works.
|
||||
|
||||
## Adding a new external corpus
|
||||
|
||||
Diagnose phase can ingest records from a HuggingFace-backed external
|
||||
corpus. Three providers ship in-tree (`adp`, `toolorchestra`,
|
||||
`generalthoughts`); to add a new one:
|
||||
|
||||
1. Create `src/openjarvis/evals/datasets/<corpus>.py` implementing
|
||||
`DatasetProvider` (`adp.py` is a small reference). The provider's
|
||||
`load(max_samples, seed, split)` must respect `split` via
|
||||
`apply_split` from `openjarvis.evals.core.splits`.
|
||||
2. Register: `@DatasetRegistry.register("<corpus>")`.
|
||||
3. Feed it to the proposer via the trace store:
|
||||
|
||||
```python
|
||||
from openjarvis.evals.datasets.adp import ADPDataset
|
||||
from openjarvis.learning.spec_search.external_adapter import (
|
||||
write_external_records_as_traces,
|
||||
)
|
||||
from openjarvis.traces.store import TraceStore
|
||||
|
||||
records = list(ADPDataset().load(max_samples=200, seed=42, split="all"))
|
||||
store = TraceStore("~/.openjarvis/traces.db")
|
||||
n = write_external_records_as_traces(store, records, source_name="adp")
|
||||
# proposer can now filter on metadata["source"] == "adp"
|
||||
```
|
||||
|
||||
## What runs where
|
||||
|
||||
At inference time, the resulting spec runs entirely on-device — model
|
||||
inference, agent execution, tool invocation. Teacher API calls happen
|
||||
only at search time (diagnose + plan), and only **eligible scrubbed
|
||||
traces** are transmitted (per the trace-eligibility rules in your
|
||||
config).
|
||||
|
||||
Users requiring strict local-only operation can swap a larger local
|
||||
model in as the teacher; this trades search quality for zero cloud
|
||||
exposure.
|
||||
|
||||
## Bug fix bundled with this release
|
||||
|
||||
`src/openjarvis/evals/backends/jarvis_agent.py` previously hardcoded
|
||||
`builder.telemetry(telemetry).traces(True).build()`, ignoring the
|
||||
`telemetry` parameter. This silently caused every agent-backend
|
||||
evaluation to write to `~/.openjarvis/traces.db` regardless of caller
|
||||
intent. A corrupt traces.db then turned every agent eval into "database
|
||||
disk image is malformed" errors that the eval scorer dropped, producing
|
||||
fake high accuracies from a handful of successful samples.
|
||||
|
||||
The one-line fix:
|
||||
|
||||
```python
|
||||
self._system = builder.telemetry(telemetry).traces(telemetry).build()
|
||||
```
|
||||
|
||||
Callers that previously expected traces to always be written should pass
|
||||
`telemetry=True` explicitly.
|
||||
|
||||
## See also
|
||||
|
||||
- `examples/openjarvis/spec_search_quickstart.py` — runnable end-to-end demo.
|
||||
- `configs/openjarvis/examples/spec-search-quickstart.toml` — prebuilt config.
|
||||
- `src/openjarvis/learning/spec_search/orchestrator.py` — `SpecSearchOrchestrator` (single session).
|
||||
- `src/openjarvis/learning/spec_search/multi_session.py` — `SpecSearchLoop` (Algorithm 1).
|
||||
- `src/openjarvis/learning/spec_search/composite_reward.py` — paper Eq. 1.
|
||||
- `src/openjarvis/learning/spec_search/gate/benchmark_gate.py` — `GateOK` predicate.
|
||||
- `src/openjarvis/learning/spec_search/external_adapter.py` — corpus → trace adapter.
|
||||
- `tests/learning/spec_search/test_multi_session.py`, `test_composite_reward.py` — unit tests.
|
||||
- `tests/learning/spec_search/test_orchestrator.py` — full-session test with mocks.
|
||||
@@ -0,0 +1,162 @@
|
||||
# Mining Pearl on Apple Silicon (and other CPU hosts)
|
||||
|
||||
OpenJarvis can mine the [Pearl](https://github.com/pearl-research-labs/pearl) chain
|
||||
on Apple Silicon Macs (M1/M2/M3/M4) using the `cpu-pearl` provider. **This is
|
||||
v1**: decoupled CPU mining. Your existing local LLM workflow (Ollama, MLX-LM,
|
||||
llama.cpp, vLLM) is untouched; mining runs in the background as a separate
|
||||
process.
|
||||
|
||||
## Honest expectations
|
||||
|
||||
**Hashrate on Apple Silicon CPU is far below what an H100 produces with
|
||||
Pearl's `vllm-miner`.** A rough rule of thumb (subject to network difficulty):
|
||||
|
||||
- M2 Max / M4 Max: ≪ 1 share per second at typical mainnet difficulty
|
||||
- H100 with `vllm-miner`: meaningfully higher, plus the mining work is
|
||||
amortized over real LLM inference
|
||||
|
||||
If you want to mine for yield, this isn't the path. If you want to participate
|
||||
in the network from the hardware you own, with no special hardware purchase,
|
||||
this is the path.
|
||||
|
||||
An experimental `apple-mps-pearl` provider is available for developers. It
|
||||
uses PyTorch MPS for the NoisyGEMM matmuls, while transcript hashing and proof
|
||||
construction still run on CPU. This proves the Apple-GPU path can produce
|
||||
validator-accepted `PlainProof`s, but it is not yet the high-performance Metal
|
||||
kernel path.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- macOS arm64 (M1, M2, M3, M4) — or Linux x86_64 / aarch64
|
||||
- Python 3.12 (`brew install python@3.12` or use `uv venv --python 3.12`)
|
||||
- Rust toolchain (`brew install rust` or `curl https://sh.rustup.rs -sSf | sh`)
|
||||
- Your own running [`pearld`](https://github.com/pearl-research-labs/pearl#node)
|
||||
node, RPC reachable on `http://localhost:44107`
|
||||
- A Pearl Taproot wallet address from `oyster` (Pearl's wallet CLI)
|
||||
- ~1 GB free disk for the Pearl source clone and build artifacts
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
# from your OpenJarvis repo
|
||||
uv sync --extra mining-pearl-cpu
|
||||
```
|
||||
|
||||
If Pearl wheels are not yet on PyPI (still true as of 2026-05-05), `uv sync`
|
||||
succeeds but doesn't install the actual Pearl Python packages. Build/install
|
||||
them from a local Pearl checkout:
|
||||
|
||||
```bash
|
||||
cd /path/to/pearl/py-pearl-mining
|
||||
maturin build --release
|
||||
uv pip install target/wheels/py_pearl_mining-*.whl
|
||||
uv pip install ../miner/miner-utils ../miner/pearl-gateway ../miner/miner-base
|
||||
```
|
||||
|
||||
## Configure
|
||||
|
||||
Create a Pearl wallet and start a synced `pearld` separately using Pearl's
|
||||
README. Then write OpenJarvis' mining config:
|
||||
|
||||
```bash
|
||||
export PEARLD_RPC_PASSWORD="rpcpass"
|
||||
|
||||
jarvis mine init \
|
||||
--provider cpu-pearl \
|
||||
--wallet-address "<your-prl1...address>" \
|
||||
--pearld-rpc-url http://127.0.0.1:44107 \
|
||||
--pearld-rpc-user rpcuser \
|
||||
--pearld-rpc-password-env PEARLD_RPC_PASSWORD
|
||||
```
|
||||
|
||||
On Apple Silicon, `--provider auto` chooses `apple-mps-pearl`; use
|
||||
`--provider cpu-pearl` for the conservative CPU path. The MPS path is
|
||||
experimental and currently useful for validation/profiling, not revenue.
|
||||
|
||||
This writes:
|
||||
|
||||
```toml
|
||||
[mining]
|
||||
provider = "cpu-pearl"
|
||||
wallet_address = "prl1..."
|
||||
submit_target = "solo"
|
||||
fee_bps = 0
|
||||
|
||||
[mining.extra]
|
||||
pearld_rpc_url = "http://127.0.0.1:44107"
|
||||
pearld_rpc_user = "rpcuser"
|
||||
pearld_rpc_password_env = "PEARLD_RPC_PASSWORD"
|
||||
gateway_host = "127.0.0.1"
|
||||
gateway_port = 8337
|
||||
metrics_port = 9109
|
||||
```
|
||||
|
||||
## Run
|
||||
|
||||
```bash
|
||||
jarvis mine doctor # capability matrix
|
||||
jarvis mine start # launch gateway + miner-loop subprocesses
|
||||
jarvis mine status # check sidecar + gateway metrics
|
||||
jarvis mine logs -n 120 # print recent logs
|
||||
jarvis mine stop # stop mining subprocesses
|
||||
```
|
||||
|
||||
## Reading `mine doctor`
|
||||
|
||||
Each row is one check. `✓` means the check passed; `✗` shows the actionable fix.
|
||||
|
||||
```
|
||||
$ jarvis mine doctor
|
||||
Hardware
|
||||
GPU vendor apple ✓
|
||||
Apple chip M2 Max ✓
|
||||
Pearl install
|
||||
py-pearl-mining 0.1.0 (cp312-abi3-macos-arm64) ✓
|
||||
miner-base 0.1.0 ✓
|
||||
pearl-gateway 0.1.0 ✓
|
||||
Pearl node
|
||||
RPC http://localhost:44107 ✓
|
||||
Block height 442107 (synced) ✓
|
||||
Wallet
|
||||
Address format prl1q... ✓
|
||||
Provider capability
|
||||
cpu-pearl SUPPORTED (calibrated 0.X share/h on M2 Max)
|
||||
Notes
|
||||
- This is decoupled mining: your normal LLM inference is unaffected
|
||||
- Hashrate is far below H100 mining; see this doc above
|
||||
- MPS mining: available as experimental apple-mps-pearl
|
||||
Session
|
||||
Sidecar absent (not running)
|
||||
```
|
||||
|
||||
## Limitations
|
||||
|
||||
- **Windows is not supported in v1.** Pearl's pure-Rust miner builds on
|
||||
Windows in principle but the cross-platform install path is untested. Use
|
||||
WSL2 if you must.
|
||||
- **No coupling to inference yet.** v1 is a separate process; your CPU does
|
||||
mining, your GPU does inference. They don't share work. v2 changes this.
|
||||
- **Experimental PyTorch-MPS only.** `apple-mps-pearl` moves the NoisyGEMM
|
||||
matmuls to MPS but still has CPU readbacks for transcript hashing and proof
|
||||
construction. Use it for validation and profiling, not revenue expectations.
|
||||
- **No multi-host pool.** Solo mining only. The pool work is a separate spec.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
| Symptom | Likely cause | Fix |
|
||||
|---|---|---|
|
||||
| `mine doctor` says `Pearl Python packages not installed` | Wheels not built yet | Run `jarvis mine init` |
|
||||
| `pearl-gateway` log shows `connection refused` to `http://localhost:44107` | `pearld` not running | Start `pearld` per Pearl's README |
|
||||
| `mine status` shows `last_error: gateway metrics unreachable` | `pearl-gateway` crashed | Check `~/.openjarvis/logs/mining/pearl-gateway.log` |
|
||||
| Build fails with `error: linker 'cc' not found` | Xcode CLT not installed | `xcode-select --install` |
|
||||
| `maturin build` complains about `tikv-jemallocator` | macOS SDK too old | Update macOS / Xcode |
|
||||
|
||||
For anything not on this list, capture `~/.openjarvis/logs/mining/` and open
|
||||
an issue at https://github.com/open-jarvis/OpenJarvis/issues.
|
||||
|
||||
## What changes in v2 / v3
|
||||
|
||||
- **v2:** Optimize the current `apple-mps-pearl` path, then optionally plug it
|
||||
into MLX-LM or `llama-cpp-python` so inference matmuls become mining work.
|
||||
- **v3 (only if v2 perf is insufficient):** Native Metal kernel as a Pearl
|
||||
upstream contribution. No user-visible change other than higher hashrate.
|
||||
@@ -0,0 +1,141 @@
|
||||
# Pearl Mining
|
||||
|
||||
OpenJarvis can mine the Pearl Proof-of-Useful-Work chain through local LLM
|
||||
inference. The primary v1 path supports NVIDIA H100/H200 hosts running vLLM
|
||||
with Pearl's Docker miner. The consolidated Pearl integration also includes
|
||||
experimental Apple Silicon and CPU providers through the same `MiningProvider`
|
||||
registry.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
| Requirement | v1 expectation |
|
||||
|---|---|
|
||||
| GPU | NVIDIA H100 or H200, sm_90a class, at least 70 GB VRAM |
|
||||
| OS | Linux with `nvidia-container-toolkit` configured |
|
||||
| Docker | Docker 24+ with GPU runtime access |
|
||||
| Disk | At least 200 GB free for the 70B model and build cache |
|
||||
| Pearl node | Reachable `pearld` JSON-RPC endpoint, default `http://localhost:44107` |
|
||||
| Wallet | Pearl address beginning with `prl1q` or `prl1p` |
|
||||
|
||||
The default vLLM config uses `gpu_memory_utilization = 0.96` and
|
||||
`max_model_len = 8192` for the Pearl 70B mining model on H100/H200 80 GB GPUs.
|
||||
|
||||
To generate a wallet address with Pearl's Oyster wallet, run Pearl's wallet
|
||||
daemon and query it with `prlctl --wallet --skipverify -s localhost:44207
|
||||
getnewaddress`. Do not reuse a wallet whose mnemonic has been pasted into logs,
|
||||
chat, or issue trackers.
|
||||
|
||||
## Quick Start
|
||||
|
||||
```bash
|
||||
uv sync --extra mining-pearl-vllm
|
||||
export PEARLD_RPC_PASSWORD=<your-pearld-password>
|
||||
export HF_TOKEN=<your-huggingface-token>
|
||||
|
||||
uv run jarvis mine init
|
||||
uv run jarvis mine start
|
||||
uv run jarvis mine status
|
||||
```
|
||||
|
||||
`mine init` writes a `[mining]` config section and resolves the Pearl Docker
|
||||
image. If Pearl has not published a suitable image for the pinned ref,
|
||||
OpenJarvis falls back to building from the pinned Pearl source checkout. First
|
||||
builds can take 30-60 minutes.
|
||||
|
||||
On a shared NVIDIA host, restrict the miner to idle GPUs:
|
||||
|
||||
```bash
|
||||
uv run jarvis mine init --cuda-visible-devices 0
|
||||
```
|
||||
|
||||
This writes `[mining.extra].cuda_visible_devices`, which `mine start` passes to
|
||||
Docker instead of exposing every GPU on the machine.
|
||||
|
||||
## Commands
|
||||
|
||||
- `jarvis mine models` lists Pearl model support status.
|
||||
- `jarvis mine inspect-model` checks a Pearl model artifact before GPU launch.
|
||||
- `jarvis mine doctor` prints hardware, Docker, Pearl node, wallet, provider,
|
||||
and session checks.
|
||||
- `jarvis mine init` writes the local mining config and resolves the image.
|
||||
- `jarvis mine start` launches the Pearl miner container and writes the runtime
|
||||
sidecar.
|
||||
- `jarvis mine stop` stops the provider and removes the sidecar.
|
||||
- `jarvis mine status` reads live gateway metrics.
|
||||
- `jarvis mine attach` writes a sidecar for a miner you launched manually.
|
||||
- `jarvis mine logs` prints the Docker container log tail.
|
||||
- `jarvis mine validate-model` probes the active vLLM miner and gateway before
|
||||
promoting a planned Pearl model to validated.
|
||||
|
||||
## Model Support
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
jarvis mine models
|
||||
```
|
||||
|
||||
OpenJarvis only lists Pearl-compatible models published by the Pearl Research
|
||||
Labs Hugging Face org. Raw Hugging Face base models such as
|
||||
`meta-llama/Llama-3.3-70B-Instruct` or `google/gemma-4-31B-it` are not mineable
|
||||
by themselves; they need corresponding `pearl-ai/*-pearl` variants.
|
||||
|
||||
The supported Pearl model ids are:
|
||||
|
||||
```text
|
||||
pearl-ai/Llama-3.3-70B-Instruct-pearl
|
||||
pearl-ai/Gemma-4-31B-it-pearl
|
||||
pearl-ai/Llama-3.1-8B-Instruct-pearl
|
||||
```
|
||||
|
||||
`pearl-ai/Llama-3.3-70B-Instruct-pearl` is the default validated model.
|
||||
Additional public `pearl-ai/*` artifacts may remain marked `planned` until they
|
||||
pass the OpenJarvis H100/H200 validation run.
|
||||
|
||||
When validating a Pearl org model on a mining host, run:
|
||||
|
||||
```bash
|
||||
jarvis mine inspect-model \
|
||||
--model pearl-ai/Gemma-4-31B-it-pearl \
|
||||
--allow-planned
|
||||
|
||||
jarvis mine validate-model \
|
||||
--model pearl-ai/Gemma-4-31B-it-pearl \
|
||||
--allow-planned \
|
||||
--prompt "Say hello in one sentence." \
|
||||
--output gemma-4-31b-pearl-validation.json
|
||||
```
|
||||
|
||||
Attach the JSON artifact to the validation issue when promoting additional
|
||||
models.
|
||||
|
||||
## v1 Scope
|
||||
|
||||
v1 is solo mining only. OpenJarvis does not take fees, custody funds, generate
|
||||
wallet keys, run pools, or operate `pearld`. Users provide their own Pearl node
|
||||
and payout address.
|
||||
|
||||
Unsupported in this PR:
|
||||
|
||||
- Pool mining and the future 20% OpenJarvis fee model
|
||||
- AMD GPU mining and non-Pearl backends
|
||||
- RTX 4090 or other non-Hopper NVIDIA GPUs
|
||||
- Wallet generation or transaction signing inside OpenJarvis
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
uv run jarvis mine doctor
|
||||
```
|
||||
|
||||
Read the rows top-down. Fix the first failing dependency before retrying
|
||||
`mine start`. A Mac or AMD machine should fail honestly at provider capability;
|
||||
those paths are expected to land as separate providers.
|
||||
|
||||
## Production Readiness
|
||||
|
||||
The NVIDIA path requires one real H100/H200 validation run before it should be
|
||||
marketed as a proven earning path. The developer runbook is
|
||||
[`../development/mining-nvidia-validation.md`](../development/mining-nvidia-validation.md).
|
||||
@@ -0,0 +1,58 @@
|
||||
# Pearl CLI Integration
|
||||
|
||||
OpenJarvis includes a thin `jarvis pearl` wrapper for Pearl's native command
|
||||
line tools. It does not replace Pearl's node or wallet; it makes the common
|
||||
commands discoverable from the same CLI users use for mining.
|
||||
|
||||
## Binary Discovery
|
||||
|
||||
`jarvis pearl` looks for `pearld`, `oyster`, and `prlctl` on `PATH`, then under
|
||||
`$PEARL_HOME/bin`.
|
||||
|
||||
```bash
|
||||
export PEARL_HOME=/path/to/pearl
|
||||
jarvis pearl doctor
|
||||
```
|
||||
|
||||
## Native Pass-Through
|
||||
|
||||
Use pass-through commands when you need the full Pearl surface:
|
||||
|
||||
```bash
|
||||
jarvis pearl node -- --help
|
||||
jarvis pearl wallet -- --help
|
||||
jarvis pearl ctl -- --help
|
||||
```
|
||||
|
||||
These map directly to:
|
||||
|
||||
| OpenJarvis command | Pearl binary |
|
||||
|---|---|
|
||||
| `jarvis pearl node` | `pearld` |
|
||||
| `jarvis pearl wallet` | `oyster` |
|
||||
| `jarvis pearl ctl` | `prlctl` |
|
||||
|
||||
The command format is always `jarvis pearl <command>`. Pearl-native arguments
|
||||
go after that command. Use `--` before Pearl arguments when the arguments begin
|
||||
with dashes and you want to make the pass-through boundary explicit.
|
||||
|
||||
## Wallet Address Helper
|
||||
|
||||
If Oyster is already running, generate a mining address through wallet RPC:
|
||||
|
||||
```bash
|
||||
jarvis pearl address \
|
||||
-u rpcuser \
|
||||
-P rpcpass \
|
||||
-s localhost:44207
|
||||
```
|
||||
|
||||
The helper uses `prlctl --wallet` and defaults to `--notls`, which matches the
|
||||
local validation flow. Use `--tls --skipverify` if your Oyster RPC endpoint is
|
||||
serving TLS with a local certificate.
|
||||
|
||||
## Boundary
|
||||
|
||||
`jarvis mine` is the OpenJarvis mining lifecycle. `jarvis pearl` is an escape
|
||||
hatch to Pearl's native node, wallet, and RPC tools. For advanced node or
|
||||
wallet administration, Pearl's own help output is the source of truth.
|
||||
@@ -148,6 +148,14 @@ jarvis skill sync openclaw --search "web3|crypto"
|
||||
jarvis skill install github:user/repo/path/to/skill --url https://github.com/user/repo
|
||||
```
|
||||
|
||||
For example, install the Hermes Tweet skill when you want an agent to search
|
||||
Twitter/X, read tweet replies, monitor tweets, export followers, and run
|
||||
gated post, reply, or DM workflows:
|
||||
|
||||
```bash
|
||||
jarvis skill install github:Xquik-dev/hermes-tweet/skills/hermes-tweet --url https://github.com/Xquik-dev/hermes-tweet
|
||||
```
|
||||
|
||||
### Config-Driven Auto Import
|
||||
|
||||
Add sources to `~/.openjarvis/config.toml` for automatic syncing:
|
||||
|
||||
@@ -0,0 +1,306 @@
|
||||
"""LLM-Guided Spec Search — runnable quickstart.
|
||||
|
||||
This script wires up every primitive ``SpecSearchOrchestrator`` needs and runs
|
||||
one full session, then a multi-session loop with the paper's stagnation rule
|
||||
(Algorithm 1, k=5, epsilon=1%).
|
||||
|
||||
It is *self-contained*: by default the teacher engine and the student runner
|
||||
are local fakes, so you can ``python examples/openjarvis/spec_search_quickstart.py``
|
||||
with no API keys and no Ollama / vLLM running. Every "swap this for production"
|
||||
hookpoint is called out inline.
|
||||
|
||||
Configuration is read from ``configs/openjarvis/examples/spec-search-quickstart.toml``
|
||||
via the regular ``openjarvis.core.config.load_config`` machinery — you can copy
|
||||
that TOML to ``~/.openjarvis/config.toml`` and tune the gate / stagnation /
|
||||
reward knobs without editing this file.
|
||||
|
||||
Run:
|
||||
|
||||
OPENJARVIS_HOME=/tmp/openjarvis-spec-search-demo \\
|
||||
python examples/openjarvis/spec_search_quickstart.py
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import tempfile
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from openjarvis.core.config import SpecSearchLearningConfig
|
||||
from openjarvis.learning.spec_search.composite_reward import (
|
||||
RewardWeights,
|
||||
TrainingSample,
|
||||
score_batch,
|
||||
)
|
||||
from openjarvis.learning.spec_search.models import (
|
||||
BenchmarkSnapshot,
|
||||
FailureCluster,
|
||||
)
|
||||
from openjarvis.learning.spec_search.multi_session import SpecSearchLoop
|
||||
from openjarvis.learning.spec_search.orchestrator import SpecSearchOrchestrator
|
||||
from openjarvis.learning.spec_search.triggers import OnDemandTrigger
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Fakes — replace these with real production components.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@dataclass
|
||||
class FakeTeacherEngine:
|
||||
"""Stand-in for ``CloudEngine``.
|
||||
|
||||
For production, use the registry::
|
||||
|
||||
from openjarvis.core.registry import EngineRegistry
|
||||
engine_cls = EngineRegistry.get(cfg.teacher_engine) # "cloud"
|
||||
engine = engine_cls(model=cfg.teacher_model) # set ANTHROPIC_API_KEY
|
||||
|
||||
The orchestrator only calls ``engine.generate(...)`` so any object with
|
||||
that method will work.
|
||||
"""
|
||||
|
||||
call_count: int = 0
|
||||
|
||||
def generate(self, **_: Any) -> dict[str, Any]:
|
||||
self.call_count += 1
|
||||
# Propose one auto-tier Tools edit so the gate has something to score.
|
||||
return {
|
||||
"content": json.dumps(
|
||||
{
|
||||
"edits": [
|
||||
{
|
||||
"id": f"edit-{self.call_count:03d}",
|
||||
"pillar": "tools",
|
||||
"op": "edit_tool_description",
|
||||
"target": "tools.web_search",
|
||||
"payload": {
|
||||
"tool_name": "web_search",
|
||||
"new_description": (
|
||||
"Search the web for recent information. "
|
||||
"Prefer this for time-sensitive queries."
|
||||
),
|
||||
},
|
||||
"rationale": (
|
||||
"Student under-invokes web_search on multi-hop queries."
|
||||
),
|
||||
"expected_improvement": "c1",
|
||||
"risk_tier": "auto",
|
||||
"references": ["t-001"],
|
||||
}
|
||||
]
|
||||
}
|
||||
),
|
||||
"usage": {"total_tokens": 1200},
|
||||
"cost_usd": 0.04,
|
||||
"finish_reason": "stop",
|
||||
}
|
||||
|
||||
|
||||
def _fake_diagnosis() -> Any:
|
||||
"""A canned DiagnosisResult so the demo doesn't need a real teacher loop."""
|
||||
from openjarvis.learning.spec_search.diagnose.runner import DiagnosisResult
|
||||
|
||||
return DiagnosisResult(
|
||||
diagnosis_md=(
|
||||
"## Diagnosis\n\n"
|
||||
"Cluster `c1` (multi-hop research): student fails to invoke "
|
||||
"`web_search` on time-sensitive queries; teacher invokes it "
|
||||
"consistently. Likely cause: tool description does not signal "
|
||||
"freshness."
|
||||
),
|
||||
clusters=[
|
||||
FailureCluster(
|
||||
id="c1",
|
||||
description="Multi-hop research; web_search under-invocation",
|
||||
sample_trace_ids=["t-001", "t-002", "t-003"],
|
||||
student_failure_rate=0.7,
|
||||
teacher_success_rate=0.95,
|
||||
skill_gap="Student does not invoke web_search on multi-hop research.",
|
||||
)
|
||||
],
|
||||
cost_usd=0.05,
|
||||
tool_call_records=[],
|
||||
)
|
||||
|
||||
|
||||
def _fake_scorer_factory(start_score: float = 0.60, step: float = 0.06):
|
||||
"""Return a scorer whose overall score climbs by ``step`` on each call.
|
||||
|
||||
Mimics the loop the paper describes: each accepted edit lifts the gate
|
||||
score, and the multi-session loop stops once gains plateau.
|
||||
"""
|
||||
state = {"score": start_score - step, "calls": 0}
|
||||
|
||||
def scorer(**_: Any) -> BenchmarkSnapshot:
|
||||
state["calls"] += 1
|
||||
# Every other call (the "after" snapshot) bumps the score; gain
|
||||
# tapers after 3 sessions so the stagnation rule fires.
|
||||
if state["calls"] % 2 == 0:
|
||||
bump = step if state["calls"] // 2 <= 3 else 0.0
|
||||
state["score"] = min(1.0, state["score"] + bump)
|
||||
return BenchmarkSnapshot(
|
||||
benchmark_version="personal_v1",
|
||||
overall_score=state["score"],
|
||||
cluster_scores={"c1": state["score"]},
|
||||
task_count=10,
|
||||
elapsed_seconds=5.0,
|
||||
)
|
||||
|
||||
return scorer
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Wire-up
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def build_orchestrator(
|
||||
config: SpecSearchLearningConfig,
|
||||
home: Path,
|
||||
) -> SpecSearchOrchestrator:
|
||||
"""Construct a SpecSearchOrchestrator from a SpecSearchLearningConfig.
|
||||
|
||||
All five injected primitives below are demo fakes; the comments next to
|
||||
each one show the production replacement.
|
||||
"""
|
||||
return SpecSearchOrchestrator.from_config(
|
||||
config,
|
||||
# Teacher: replace with EngineRegistry.get("cloud")(model=cfg.teacher_model)
|
||||
teacher_engine=FakeTeacherEngine(),
|
||||
# TraceStore: production = TraceStore(home / "traces.db")
|
||||
trace_store=MagicMock(count=MagicMock(return_value=config.min_traces + 10)),
|
||||
benchmark_samples=[],
|
||||
# StudentRunner: production = VLLMStudentRunner(host=..., model=...)
|
||||
student_runner=MagicMock(),
|
||||
# Judge: production = openjarvis.evals.core.scorer.LLMJudgeScorer(...)
|
||||
judge=MagicMock(),
|
||||
# SessionStore + CheckpointStore: production = real on-disk stores
|
||||
session_store=MagicMock(),
|
||||
checkpoint_store=MagicMock(
|
||||
current_sha=MagicMock(return_value="demo-sha"),
|
||||
begin_stage=MagicMock(
|
||||
return_value=MagicMock(pre_stage_sha="demo-sha"),
|
||||
),
|
||||
),
|
||||
openjarvis_home=home,
|
||||
scorer=_fake_scorer_factory(),
|
||||
)
|
||||
|
||||
|
||||
def demo_composite_reward(weights: RewardWeights) -> None:
|
||||
"""Show how the paper Eq. 1 reward ranks Intelligence-edit candidates."""
|
||||
candidates = [
|
||||
TrainingSample(
|
||||
accuracy=1.0, energy_joules=200, latency_seconds=5.0, cost_usd=0.0
|
||||
),
|
||||
TrainingSample(
|
||||
accuracy=1.0, energy_joules=400, latency_seconds=8.0, cost_usd=0.0
|
||||
),
|
||||
TrainingSample(
|
||||
accuracy=0.0, energy_joules=100, latency_seconds=2.0, cost_usd=0.0
|
||||
),
|
||||
]
|
||||
rewards = score_batch(candidates, weights=weights)
|
||||
print("\nComposite reward (paper Eq. 1) — ranking 3 candidates:")
|
||||
print(
|
||||
f" weights = (alpha={weights.alpha}, beta={weights.beta}, "
|
||||
f"gamma={weights.gamma}, delta={weights.delta})"
|
||||
)
|
||||
for i, (c, r) in enumerate(zip(candidates, rewards)):
|
||||
print(
|
||||
f" candidate {i}: acc={c.accuracy} energy={c.energy_joules}J "
|
||||
f"latency={c.latency_seconds}s -> reward={r:+.3f}"
|
||||
)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
# In a real deployment, ``load_config()`` reads from ``~/.openjarvis/config.toml``.
|
||||
# For a self-contained demo we synthesize the spec-search config inline so the
|
||||
# script is runnable without copying any files. To use the prebuilt TOML:
|
||||
#
|
||||
# from openjarvis.core.config import load_config
|
||||
# cfg = load_config().learning.spec_search
|
||||
#
|
||||
# (after copying ``configs/openjarvis/examples/spec-search-quickstart.toml``
|
||||
# to ``~/.openjarvis/config.toml``)
|
||||
|
||||
cfg = SpecSearchLearningConfig(
|
||||
enabled=True,
|
||||
teacher_model="claude-opus-4-6",
|
||||
teacher_engine="cloud",
|
||||
autonomy_mode="auto",
|
||||
min_traces=20,
|
||||
max_cost_per_session_usd=5.0,
|
||||
max_tool_calls_per_diagnosis=30,
|
||||
stagnation_k=5, # paper default
|
||||
stagnation_eps=0.001,
|
||||
max_total_cost_usd=50.0,
|
||||
max_regression=0.01, # paper default: epsilon = 1%
|
||||
min_improvement=0.0,
|
||||
benchmark_subsample_size=10,
|
||||
)
|
||||
|
||||
home = Path(
|
||||
os.environ.get("OPENJARVIS_HOME")
|
||||
or tempfile.mkdtemp(prefix="openjarvis-spec-search-")
|
||||
)
|
||||
print(f"OPENJARVIS_HOME = {home}")
|
||||
|
||||
# ----- Single session ---------------------------------------------------
|
||||
orch = build_orchestrator(cfg, home)
|
||||
|
||||
# The orchestrator's diagnose phase calls a real DiagnosisRunner that
|
||||
# invokes the teacher. We swap it out for the canned diagnosis above so
|
||||
# the demo does not need an API key.
|
||||
from unittest.mock import patch
|
||||
|
||||
print("\n=== Single session (one diagnose / plan / execute / record) ===")
|
||||
with patch(
|
||||
"openjarvis.learning.spec_search.orchestrator.DiagnosisRunner"
|
||||
) as MockDiag:
|
||||
MockDiag.return_value.run.return_value = _fake_diagnosis()
|
||||
session = orch.run(OnDemandTrigger())
|
||||
|
||||
print(f" status = {session.status.value}")
|
||||
print(f" teacher_cost_usd = ${session.teacher_cost_usd:.4f}")
|
||||
print(f" edit_outcomes = {[(o.edit_id, o.status) for o in session.edit_outcomes]}")
|
||||
if session.benchmark_after is not None:
|
||||
print(
|
||||
f" before -> after = "
|
||||
f"{session.benchmark_before.overall_score:.3f} -> "
|
||||
f"{session.benchmark_after.overall_score:.3f}"
|
||||
)
|
||||
|
||||
# ----- Multi-session loop (paper Algorithm 1) ---------------------------
|
||||
orch = build_orchestrator(cfg, home) # fresh fakes for clean state
|
||||
loop = SpecSearchLoop(
|
||||
orch,
|
||||
stagnation_k=cfg.stagnation_k,
|
||||
stagnation_eps=cfg.stagnation_eps,
|
||||
max_total_cost_usd=cfg.max_total_cost_usd,
|
||||
)
|
||||
|
||||
print(
|
||||
"\n=== Multi-session loop (stagnation_k = "
|
||||
f"{cfg.stagnation_k}, max_total_cost = ${cfg.max_total_cost_usd}) ==="
|
||||
)
|
||||
with patch(
|
||||
"openjarvis.learning.spec_search.orchestrator.DiagnosisRunner"
|
||||
) as MockDiag:
|
||||
MockDiag.return_value.run.return_value = _fake_diagnosis()
|
||||
result = loop.run()
|
||||
|
||||
print(f" sessions = {len(result.sessions)}")
|
||||
print(f" stop_reason = {result.stop_reason}")
|
||||
print(f" total cost = ${result.total_cost_usd:.4f}")
|
||||
print(f" best score = {result.best_overall_score:.3f}")
|
||||
|
||||
demo_composite_reward(RewardWeights())
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Generated
+951
-56
File diff suppressed because it is too large
Load Diff
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "openjarvis-chat",
|
||||
"private": true,
|
||||
"version": "0.1.0",
|
||||
"version": "1.0.1",
|
||||
"type": "module",
|
||||
"engines": {
|
||||
"node": ">=20"
|
||||
@@ -11,13 +11,14 @@
|
||||
"build": "tsc -b && vite build",
|
||||
"build:tauri": "tsc -b && vite build --outDir dist",
|
||||
"preview": "vite preview",
|
||||
"tauri": "tauri"
|
||||
"tauri": "tauri",
|
||||
"test": "vitest run"
|
||||
},
|
||||
"dependencies": {
|
||||
"@base-ui/react": "^1.3.0",
|
||||
"@fontsource-variable/geist": "^5.2.8",
|
||||
"@tailwindcss/vite": "^4.2.1",
|
||||
"@tauri-apps/api": "^2",
|
||||
"@tauri-apps/api": "^2.11.1",
|
||||
"@tauri-apps/plugin-autostart": "^2",
|
||||
"@tauri-apps/plugin-dialog": "^2.7.0",
|
||||
"@tauri-apps/plugin-global-shortcut": "^2",
|
||||
@@ -30,6 +31,7 @@
|
||||
"katex": "^0.16.38",
|
||||
"lucide-react": "^0.576.0",
|
||||
"motion": "^12.38.0",
|
||||
"posthog-js": "^1.373.2",
|
||||
"react": "^19.0.0",
|
||||
"react-dom": "^19.0.0",
|
||||
"react-markdown": "^10.1.0",
|
||||
@@ -47,12 +49,13 @@
|
||||
"zustand": "^5.0.11"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@tauri-apps/cli": "^2",
|
||||
"@tauri-apps/cli": "^2.11.4",
|
||||
"@types/react": "^19.0.0",
|
||||
"@types/react-dom": "^19.0.0",
|
||||
"@vitejs/plugin-react": "^4.3.4",
|
||||
"typescript": "~5.7.0",
|
||||
"vite": "^6.0.0",
|
||||
"vite-plugin-pwa": "^1.2.0"
|
||||
"vite-plugin-pwa": "^1.2.0",
|
||||
"vitest": "^3.2.6"
|
||||
}
|
||||
}
|
||||
|
||||
Generated
+1222
-1010
File diff suppressed because it is too large
Load Diff
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "openjarvis-desktop"
|
||||
version = "0.1.0"
|
||||
version = "1.0.1"
|
||||
description = "OpenJarvis Desktop — Native AI assistant with energy monitoring, trace debugging, and learning visualization"
|
||||
edition = "2021"
|
||||
license = "MIT"
|
||||
@@ -9,6 +9,7 @@ license = "MIT"
|
||||
tauri-build = { version = "2", features = [] }
|
||||
|
||||
[dependencies]
|
||||
toml_edit = "0.22"
|
||||
tauri = { version = "2", features = ["tray-icon"] }
|
||||
tauri-plugin-notification = "2"
|
||||
tauri-plugin-shell = "2"
|
||||
@@ -23,9 +24,22 @@ serde_json = "1"
|
||||
reqwest = { version = "0.12", features = ["json", "multipart"] }
|
||||
tokio = { version = "1", features = ["full"] }
|
||||
|
||||
# Cloud API keys are stored in the OS credential store via `keyring`. keyring v3
|
||||
# enables NO backend by default — without an explicit per-platform feature it
|
||||
# silently falls back to a non-persistent in-memory mock, so keys would not
|
||||
# survive an app restart. Each desktop target opts into its native store.
|
||||
[target.'cfg(target_os = "macos")'.dependencies]
|
||||
objc = "0.2"
|
||||
dispatch = "0.2"
|
||||
keyring = { version = "3", features = ["apple-native"] }
|
||||
|
||||
[target.'cfg(target_os = "windows")'.dependencies]
|
||||
keyring = { version = "3", features = ["windows-native"] }
|
||||
|
||||
[target.'cfg(target_os = "linux")'.dependencies]
|
||||
# Blocking Secret Service backend (no internal async runtime, so it is safe to
|
||||
# call from the tokio-driven Tauri commands). Needs libdbus-1-dev at build time.
|
||||
keyring = { version = "3", features = ["sync-secret-service", "crypto-rust"] }
|
||||
|
||||
[features]
|
||||
default = ["custom-protocol"]
|
||||
|
||||
+1829
-258
File diff suppressed because it is too large
Load Diff
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user