jevkit-runtime 0.1.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {jevkit_runtime-0.1.0 → jevkit_runtime-0.3.0}/.github/workflows/downstream.yml +0 -2
- jevkit_runtime-0.3.0/PKG-INFO +112 -0
- jevkit_runtime-0.3.0/README.md +97 -0
- jevkit_runtime-0.3.0/docs/diffusiongemma.md +86 -0
- jevkit_runtime-0.3.0/docs/laya.md +80 -0
- {jevkit_runtime-0.1.0 → jevkit_runtime-0.3.0}/pyproject.toml +2 -2
- {jevkit_runtime-0.1.0 → jevkit_runtime-0.3.0}/scripts/dev.py +48 -50
- jevkit_runtime-0.3.0/scripts/laya_server.py +116 -0
- jevkit_runtime-0.3.0/src/jevkit_runtime/__init__.py +65 -0
- jevkit_runtime-0.3.0/src/jevkit_runtime/client.py +222 -0
- jevkit_runtime-0.3.0/src/jevkit_runtime/errors.py +43 -0
- jevkit_runtime-0.3.0/src/jevkit_runtime/meter.py +67 -0
- jevkit_runtime-0.3.0/src/jevkit_runtime/protocol.py +160 -0
- jevkit_runtime-0.3.0/src/jevkit_runtime/providers.py +183 -0
- jevkit_runtime-0.3.0/src/jevkit_runtime/settings.py +64 -0
- jevkit_runtime-0.3.0/src/jevkit_runtime/store.py +90 -0
- jevkit_runtime-0.3.0/src/jevkit_runtime/transport.py +90 -0
- jevkit_runtime-0.3.0/tests/test_client.py +267 -0
- jevkit_runtime-0.3.0/tests/test_dev_runner.py +60 -0
- jevkit_runtime-0.3.0/tests/test_protocol.py +117 -0
- jevkit_runtime-0.3.0/tests/test_settings_and_providers.py +139 -0
- jevkit_runtime-0.3.0/tests/test_store.py +77 -0
- jevkit_runtime-0.3.0/tests/test_transport.py +159 -0
- {jevkit_runtime-0.1.0 → jevkit_runtime-0.3.0}/uv.lock +1 -1
- jevkit_runtime-0.1.0/PKG-INFO +0 -179
- jevkit_runtime-0.1.0/README.md +0 -164
- jevkit_runtime-0.1.0/TESTING.md +0 -95
- jevkit_runtime-0.1.0/consumer-baselines.json +0 -7
- jevkit_runtime-0.1.0/scripts/probe_consumer.py +0 -232
- jevkit_runtime-0.1.0/src/jevkit_core/__init__.py +0 -48
- jevkit_runtime-0.1.0/src/jevkit_core/backends.py +0 -142
- jevkit_runtime-0.1.0/src/jevkit_core/cache.py +0 -92
- jevkit_runtime-0.1.0/src/jevkit_core/client.py +0 -109
- jevkit_runtime-0.1.0/src/jevkit_core/errors.py +0 -22
- jevkit_runtime-0.1.0/src/jevkit_core/provenance.py +0 -18
- jevkit_runtime-0.1.0/src/jevkit_core/transport.py +0 -119
- jevkit_runtime-0.1.0/src/jevkit_core/usage.py +0 -88
- jevkit_runtime-0.1.0/tests/test_accounting.py +0 -62
- jevkit_runtime-0.1.0/tests/test_backends.py +0 -82
- jevkit_runtime-0.1.0/tests/test_cache_and_usage.py +0 -91
- jevkit_runtime-0.1.0/tests/test_shared_requests.py +0 -122
- jevkit_runtime-0.1.0/tests/test_transport.py +0 -136
- {jevkit_runtime-0.1.0 → jevkit_runtime-0.3.0}/.github/workflows/publish.yml +0 -0
- {jevkit_runtime-0.1.0 → jevkit_runtime-0.3.0}/.github/workflows/test.yml +0 -0
- {jevkit_runtime-0.1.0 → jevkit_runtime-0.3.0}/.gitignore +0 -0
- {jevkit_runtime-0.1.0 → jevkit_runtime-0.3.0}/LICENSE +0 -0
- {jevkit_runtime-0.1.0 → jevkit_runtime-0.3.0}/scripts/offline/sitecustomize.py +0 -0
|
@@ -14,8 +14,6 @@ permissions:
|
|
|
14
14
|
contents: read
|
|
15
15
|
jobs:
|
|
16
16
|
consumers:
|
|
17
|
-
# Enabled after all five migration PRs land; manual runs remain available.
|
|
18
|
-
if: vars.JEVKIT_CONSUMERS_READY == 'true' || github.event_name == 'workflow_dispatch'
|
|
19
17
|
runs-on: ubuntu-latest
|
|
20
18
|
strategy:
|
|
21
19
|
fail-fast: false
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: jevkit-runtime
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: Shared transport, configuration, caching, and accounting for JevKit tools
|
|
5
|
+
Project-URL: Homepage, https://github.com/keltokhy/jevkit-core
|
|
6
|
+
Project-URL: Issues, https://github.com/keltokhy/jevkit-core/issues
|
|
7
|
+
Author: Khaled Eltokhy
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Requires-Python: >=3.10
|
|
11
|
+
Requires-Dist: httpx>=0.27
|
|
12
|
+
Provides-Extra: http2
|
|
13
|
+
Requires-Dist: httpx[http2]>=0.27; extra == 'http2'
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
|
|
16
|
+
# JevKit core
|
|
17
|
+
|
|
18
|
+
Distribution **`jevkit-runtime`**, import **`jevkit_runtime`**. The PyPI name `jevkit-core` belongs
|
|
19
|
+
to a different project.
|
|
20
|
+
|
|
21
|
+
One request pipeline, one answer store, one provider catalog for jgrep, jsort, jlink, jselect,
|
|
22
|
+
and jcol. Each tool remains its own package and repository; the core imports none of them, and a
|
|
23
|
+
tool's adapter is a few lines naming which providers it offers.
|
|
24
|
+
|
|
25
|
+
## What a tool gets
|
|
26
|
+
|
|
27
|
+
```python
|
|
28
|
+
from jevkit_runtime import AnswerStore, Client, catalog, resolve
|
|
29
|
+
|
|
30
|
+
PROVIDERS = catalog("typesafe", "openrouter", "gateway")
|
|
31
|
+
backend = resolve(PROVIDERS, name=None, model=None) # or JEV_API / JEV_MODEL, else the first configured
|
|
32
|
+
async with Client(backend, store=AnswerStore()) as client:
|
|
33
|
+
answers = await client.ask(state, {"q": {"type": "noul", "instructions": "..."}})
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
`Client.ask` does the whole thing: computes each question's identity, serves what the store already
|
|
37
|
+
knows, joins an identical request already in flight, sends only the misses, validates the entire
|
|
38
|
+
response before storing any of it, and meters the call before validation so a billed but malformed
|
|
39
|
+
answer still counts. It returns `Answers`, a dict by question id whose `origins` say who answered
|
|
40
|
+
each one and whether it came from the API, the store, or a shared call. Per-call policy is keyword
|
|
41
|
+
arguments: `allow_paid=False` for cache-only runs, `on_cost` for the caller who should be charged,
|
|
42
|
+
`hedge_after` to resend a slow call, and `keys` for callers whose reuse unit is not the request.
|
|
43
|
+
HTTP/2 is used whenever the `http2` extra is installed.
|
|
44
|
+
|
|
45
|
+
| Module | Owns |
|
|
46
|
+
|---|---|
|
|
47
|
+
| `settings.py` | Every environment and filesystem convention, read in one place: `XDG_*`, `JEV_API`, `JEV_URL`, `JEV_MODEL`, `JEV_PRICE_PER_MTOK`, provider keys and URL files |
|
|
48
|
+
| `providers.py` | The catalog (`Provider`), a tool's selection of it or its own entries, and `resolve()` to one `Backend`: endpoint, model, key |
|
|
49
|
+
| `protocol.py` | Request bodies, typed answer validation (`noul`, `choice`, `score`), usage parsing, answer identity, provenance |
|
|
50
|
+
| `transport.py` | One HTTP call with a total deadline, retries with backoff and `Retry-After`, structured status errors |
|
|
51
|
+
| `store.py` | SQLite answers with their provenance in one row, one versioned schema |
|
|
52
|
+
| `client.py` | The pipeline above, request sharing, hedging |
|
|
53
|
+
| `meter.py` | Calls, cache hits, retries, hedges, tokens, cost, and which models actually answered |
|
|
54
|
+
| `errors.py` | `JevError`, `JevFatal`, `JevBudgetExceeded`, `RequestExhausted`, `ProviderError`, `ProviderFatal` |
|
|
55
|
+
|
|
56
|
+
## Conventions every tool shares
|
|
57
|
+
|
|
58
|
+
- **Answer identity** is `answer_key(backend, state, question)`: provider, endpoint, model, state and
|
|
59
|
+
question. An answer from one provider or model is never served for another.
|
|
60
|
+
- **The store** lives at `$XDG_CACHE_HOME/jev/answers.sqlite` (default `~/.cache/jev`), is created
|
|
61
|
+
private to the user, and resets itself when it finds an older schema. Version 0.2 cannot read
|
|
62
|
+
caches written by 0.1 tools; the first run after upgrading re-asks.
|
|
63
|
+
- **Credentials** come from the provider's variable, then `$XDG_CONFIG_HOME/jev/<provider>.key`.
|
|
64
|
+
Gateways take their URL from `JEV_GATEWAY_URL` or `<provider>.url`. `JEV_URL` overrides any endpoint.
|
|
65
|
+
- **Metering** refuses malformed usage rather than under-counting; a response without a reported
|
|
66
|
+
cost is priced from its tokens at the provider's price, zero for local servers, or the list price.
|
|
67
|
+
`JEV_PRICE_PER_MTOK` overrides both.
|
|
68
|
+
- **Errors** keep their wording across tools: a fatal status reads `PROVIDER said 401: detail`, a
|
|
69
|
+
bad request reads `HTTP 400: detail`, and exhaustion reads `gave up after 15s (last failure)`.
|
|
70
|
+
Both status errors carry `provider`, `status` and `detail` for tools that word or redact them.
|
|
71
|
+
|
|
72
|
+
## Local servers
|
|
73
|
+
|
|
74
|
+
Two catalog entries point at System One servers on your own machine: `diffusiongemma`, an
|
|
75
|
+
[OpenJev](https://github.com/razorback16/openjev) server on port 8080, and `laya`, a
|
|
76
|
+
[laya-mlx](https://github.com/mizorewww/laya-mlx) server on port 8081. Every JevKit tool names
|
|
77
|
+
them in its catalog, so `--api laya` or `JEV_API=laya` works everywhere. They are never chosen
|
|
78
|
+
in place of a configured hosted provider, need no key, and are metered at zero API fees unless
|
|
79
|
+
`JEV_PRICE_PER_MTOK` says otherwise. `JEV_LAYA_URL` and `JEV_DIFFUSIONGEMMA_URL`, or the matching
|
|
80
|
+
`.url` files, point at a server elsewhere.
|
|
81
|
+
|
|
82
|
+
DiffusionGemma reads every question in a batch together, so the runtime keys each of its answers
|
|
83
|
+
on the whole ordered batch and re-sends a batch whole when any slot is missing.
|
|
84
|
+
|
|
85
|
+
No package ships the models. [docs/diffusiongemma.md](docs/diffusiongemma.md) and
|
|
86
|
+
[docs/laya.md](docs/laya.md) explain how to run the servers, and `scripts/laya_server.py` is the
|
|
87
|
+
adapter the Laya guide starts.
|
|
88
|
+
|
|
89
|
+
## Development
|
|
90
|
+
|
|
91
|
+
Keep the six checkouts as siblings. Each consumer depends on `jevkit-runtime>=0.2.0,<0.3.0` and
|
|
92
|
+
overrides it for development with `jevkit-runtime = { path = "../jevkit-core", editable = true }`
|
|
93
|
+
under `[tool.uv.sources]`.
|
|
94
|
+
|
|
95
|
+
```bash
|
|
96
|
+
python3 scripts/dev.py setup # uv sync every checkout, fetch jselect's tokenizer data
|
|
97
|
+
python3 scripts/dev.py check # core and consumer suites, credentials stripped, sockets blocked
|
|
98
|
+
python3 scripts/dev.py wheel-check # build and exercise real wheel installs in temporary environments
|
|
99
|
+
python3 scripts/dev.py run jgrep -- --help
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
`check` also proves each consumer imports this exact source tree; `wheel-check` proves the installed
|
|
103
|
+
wheel, not the checkout. `--tool NAME` limits either to one consumer and `--suffix` selects
|
|
104
|
+
alternatively named checkouts. The GitHub workflows run the core suite on Python 3.10 and 3.13 and
|
|
105
|
+
the downstream matrix against each consumer's main branch.
|
|
106
|
+
|
|
107
|
+
## Releasing
|
|
108
|
+
|
|
109
|
+
Tag the verified core `vX.Y.Z` and dispatch the publish workflow with that tag; the workflow checks
|
|
110
|
+
the tag matches the package version and publishes through PyPI Trusted Publishing. Then release each
|
|
111
|
+
consumer through its own process, bumping its supported core range, lockfile, and CI core reference
|
|
112
|
+
together.
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
# JevKit core
|
|
2
|
+
|
|
3
|
+
Distribution **`jevkit-runtime`**, import **`jevkit_runtime`**. The PyPI name `jevkit-core` belongs
|
|
4
|
+
to a different project.
|
|
5
|
+
|
|
6
|
+
One request pipeline, one answer store, one provider catalog for jgrep, jsort, jlink, jselect,
|
|
7
|
+
and jcol. Each tool remains its own package and repository; the core imports none of them, and a
|
|
8
|
+
tool's adapter is a few lines naming which providers it offers.
|
|
9
|
+
|
|
10
|
+
## What a tool gets
|
|
11
|
+
|
|
12
|
+
```python
|
|
13
|
+
from jevkit_runtime import AnswerStore, Client, catalog, resolve
|
|
14
|
+
|
|
15
|
+
PROVIDERS = catalog("typesafe", "openrouter", "gateway")
|
|
16
|
+
backend = resolve(PROVIDERS, name=None, model=None) # or JEV_API / JEV_MODEL, else the first configured
|
|
17
|
+
async with Client(backend, store=AnswerStore()) as client:
|
|
18
|
+
answers = await client.ask(state, {"q": {"type": "noul", "instructions": "..."}})
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
`Client.ask` does the whole thing: computes each question's identity, serves what the store already
|
|
22
|
+
knows, joins an identical request already in flight, sends only the misses, validates the entire
|
|
23
|
+
response before storing any of it, and meters the call before validation so a billed but malformed
|
|
24
|
+
answer still counts. It returns `Answers`, a dict by question id whose `origins` say who answered
|
|
25
|
+
each one and whether it came from the API, the store, or a shared call. Per-call policy is keyword
|
|
26
|
+
arguments: `allow_paid=False` for cache-only runs, `on_cost` for the caller who should be charged,
|
|
27
|
+
`hedge_after` to resend a slow call, and `keys` for callers whose reuse unit is not the request.
|
|
28
|
+
HTTP/2 is used whenever the `http2` extra is installed.
|
|
29
|
+
|
|
30
|
+
| Module | Owns |
|
|
31
|
+
|---|---|
|
|
32
|
+
| `settings.py` | Every environment and filesystem convention, read in one place: `XDG_*`, `JEV_API`, `JEV_URL`, `JEV_MODEL`, `JEV_PRICE_PER_MTOK`, provider keys and URL files |
|
|
33
|
+
| `providers.py` | The catalog (`Provider`), a tool's selection of it or its own entries, and `resolve()` to one `Backend`: endpoint, model, key |
|
|
34
|
+
| `protocol.py` | Request bodies, typed answer validation (`noul`, `choice`, `score`), usage parsing, answer identity, provenance |
|
|
35
|
+
| `transport.py` | One HTTP call with a total deadline, retries with backoff and `Retry-After`, structured status errors |
|
|
36
|
+
| `store.py` | SQLite answers with their provenance in one row, one versioned schema |
|
|
37
|
+
| `client.py` | The pipeline above, request sharing, hedging |
|
|
38
|
+
| `meter.py` | Calls, cache hits, retries, hedges, tokens, cost, and which models actually answered |
|
|
39
|
+
| `errors.py` | `JevError`, `JevFatal`, `JevBudgetExceeded`, `RequestExhausted`, `ProviderError`, `ProviderFatal` |
|
|
40
|
+
|
|
41
|
+
## Conventions every tool shares
|
|
42
|
+
|
|
43
|
+
- **Answer identity** is `answer_key(backend, state, question)`: provider, endpoint, model, state and
|
|
44
|
+
question. An answer from one provider or model is never served for another.
|
|
45
|
+
- **The store** lives at `$XDG_CACHE_HOME/jev/answers.sqlite` (default `~/.cache/jev`), is created
|
|
46
|
+
private to the user, and resets itself when it finds an older schema. Version 0.2 cannot read
|
|
47
|
+
caches written by 0.1 tools; the first run after upgrading re-asks.
|
|
48
|
+
- **Credentials** come from the provider's variable, then `$XDG_CONFIG_HOME/jev/<provider>.key`.
|
|
49
|
+
Gateways take their URL from `JEV_GATEWAY_URL` or `<provider>.url`. `JEV_URL` overrides any endpoint.
|
|
50
|
+
- **Metering** refuses malformed usage rather than under-counting; a response without a reported
|
|
51
|
+
cost is priced from its tokens at the provider's price, zero for local servers, or the list price.
|
|
52
|
+
`JEV_PRICE_PER_MTOK` overrides both.
|
|
53
|
+
- **Errors** keep their wording across tools: a fatal status reads `PROVIDER said 401: detail`, a
|
|
54
|
+
bad request reads `HTTP 400: detail`, and exhaustion reads `gave up after 15s (last failure)`.
|
|
55
|
+
Both status errors carry `provider`, `status` and `detail` for tools that word or redact them.
|
|
56
|
+
|
|
57
|
+
## Local servers
|
|
58
|
+
|
|
59
|
+
Two catalog entries point at System One servers on your own machine: `diffusiongemma`, an
|
|
60
|
+
[OpenJev](https://github.com/razorback16/openjev) server on port 8080, and `laya`, a
|
|
61
|
+
[laya-mlx](https://github.com/mizorewww/laya-mlx) server on port 8081. Every JevKit tool names
|
|
62
|
+
them in its catalog, so `--api laya` or `JEV_API=laya` works everywhere. They are never chosen
|
|
63
|
+
in place of a configured hosted provider, need no key, and are metered at zero API fees unless
|
|
64
|
+
`JEV_PRICE_PER_MTOK` says otherwise. `JEV_LAYA_URL` and `JEV_DIFFUSIONGEMMA_URL`, or the matching
|
|
65
|
+
`.url` files, point at a server elsewhere.
|
|
66
|
+
|
|
67
|
+
DiffusionGemma reads every question in a batch together, so the runtime keys each of its answers
|
|
68
|
+
on the whole ordered batch and re-sends a batch whole when any slot is missing.
|
|
69
|
+
|
|
70
|
+
No package ships the models. [docs/diffusiongemma.md](docs/diffusiongemma.md) and
|
|
71
|
+
[docs/laya.md](docs/laya.md) explain how to run the servers, and `scripts/laya_server.py` is the
|
|
72
|
+
adapter the Laya guide starts.
|
|
73
|
+
|
|
74
|
+
## Development
|
|
75
|
+
|
|
76
|
+
Keep the six checkouts as siblings. Each consumer depends on `jevkit-runtime>=0.2.0,<0.3.0` and
|
|
77
|
+
overrides it for development with `jevkit-runtime = { path = "../jevkit-core", editable = true }`
|
|
78
|
+
under `[tool.uv.sources]`.
|
|
79
|
+
|
|
80
|
+
```bash
|
|
81
|
+
python3 scripts/dev.py setup # uv sync every checkout, fetch jselect's tokenizer data
|
|
82
|
+
python3 scripts/dev.py check # core and consumer suites, credentials stripped, sockets blocked
|
|
83
|
+
python3 scripts/dev.py wheel-check # build and exercise real wheel installs in temporary environments
|
|
84
|
+
python3 scripts/dev.py run jgrep -- --help
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
`check` also proves each consumer imports this exact source tree; `wheel-check` proves the installed
|
|
88
|
+
wheel, not the checkout. `--tool NAME` limits either to one consumer and `--suffix` selects
|
|
89
|
+
alternatively named checkouts. The GitHub workflows run the core suite on Python 3.10 and 3.13 and
|
|
90
|
+
the downstream matrix against each consumer's main branch.
|
|
91
|
+
|
|
92
|
+
## Releasing
|
|
93
|
+
|
|
94
|
+
Tag the verified core `vX.Y.Z` and dispatch the publish workflow with that tag; the workflow checks
|
|
95
|
+
the tag matches the package version and publishes through PyPI Trusted Publishing. Then release each
|
|
96
|
+
consumer through its own process, bumping its supported core range, lockfile, and CI core reference
|
|
97
|
+
together.
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
# DiffusionGemma (local, experimental)
|
|
2
|
+
|
|
3
|
+
Any JevKit tool can send its decisions to a DiffusionGemma decision server on your own machine
|
|
4
|
+
instead of a hosted provider. The server is chosen only by name, never automatically, and its
|
|
5
|
+
calls are metered at zero API fees:
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
jgrep --api diffusiongemma -j 1 "a complaint about noise" complaints.txt
|
|
9
|
+
jcol --api diffusiongemma run table.csv codebook.json
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
No JevKit package ships the model; it runs in its own environment and process. Model quality
|
|
13
|
+
and useful probability thresholds need evaluation on your own inputs.
|
|
14
|
+
|
|
15
|
+
## Apple silicon
|
|
16
|
+
|
|
17
|
+
[OpenJev](https://github.com/razorback16/openjev) supplies an MLX implementation of the
|
|
18
|
+
structured-read approach from [vLLM PR #57250](https://github.com/vllm-project/vllm/pull/57250).
|
|
19
|
+
Its 4-bit checkpoint is roughly 16.6 GB to download and needs about 16 GB of model memory plus
|
|
20
|
+
working memory. Install the server in its own directory and environment:
|
|
21
|
+
|
|
22
|
+
```bash
|
|
23
|
+
git clone https://github.com/razorback16/openjev.git
|
|
24
|
+
cd openjev
|
|
25
|
+
git checkout e04794ab36e4f7e6040c2547baecdb2737ce2e79
|
|
26
|
+
uv sync --python 3.12 --extra mlx
|
|
27
|
+
OPENJEV_BACKEND=mlx uv run --extra mlx python -m openjev
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
The server downloads `mlx-community/diffusiongemma-26B-A4B-it-4bit` on first start. Wait for
|
|
31
|
+
`Application startup complete` before querying it. `OPENJEV_MLX_MODEL` can point at a downloaded
|
|
32
|
+
snapshot to pin the weights; the revision used in the original experiment was
|
|
33
|
+
`a7a81407613811e8ba63af92ac0d852b809e191f`. Then, in another terminal:
|
|
34
|
+
|
|
35
|
+
```bash
|
|
36
|
+
printf '%s\n' 'The music next door keeps me awake.' 'The elevator is broken.' |
|
|
37
|
+
jgrep --api diffusiongemma --model openjev-0.1 -j 1 --timeout 120 --stats -o 'a complaint about noise'
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
Start with one request at a time: the MLX backend executes model work serially, and a deep
|
|
41
|
+
queue can exceed a tool's per-request deadline. The longer timeout covers the first inference.
|
|
42
|
+
Raise concurrency from measured throughput once the model is warm.
|
|
43
|
+
|
|
44
|
+
## NVIDIA / vLLM
|
|
45
|
+
|
|
46
|
+
Run OpenJev's documented vLLM deployment, or the prototype `structured_server.py` from
|
|
47
|
+
[PR #57250](https://github.com/vllm-project/vllm/pull/57250). That PR was unmerged at the time of
|
|
48
|
+
writing, so a released vLLM is not enough on its own; follow the server's pinned build
|
|
49
|
+
instructions. Its `/v1/systemone` adapter sits in front of vLLM. For the PR example server's
|
|
50
|
+
default port:
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
JEV_DIFFUSIONGEMMA_URL=http://127.0.0.1:8011/v1/systemone \
|
|
54
|
+
jgrep --api diffusiongemma --model jev-latest 'a stack trace' build.log
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
## Configuration
|
|
58
|
+
|
|
59
|
+
| Setting | Meaning |
|
|
60
|
+
|---|---|
|
|
61
|
+
| `--api diffusiongemma` or `JEV_API=diffusiongemma` | Select it; it is never picked by discovery |
|
|
62
|
+
| `JEV_DIFFUSIONGEMMA_URL` or `~/.config/jev/diffusiongemma.url` | Full endpoint, default `http://127.0.0.1:8080/v1/systemone` |
|
|
63
|
+
| `JEV_DIFFUSIONGEMMA_API_KEY` or `~/.config/jev/diffusiongemma.key` | Optional bearer token; no `Authorization` header when absent |
|
|
64
|
+
| `--model` or `JEV_MODEL` | Default `openjev-latest`; `openjev-0.1` pins the server's decision model |
|
|
65
|
+
| `JEV_PRICE_PER_MTOK` | Overrides the zero price for `--stats`, `--estimate`, and dollar budgets |
|
|
66
|
+
|
|
67
|
+
`JEV_URL` overrides every endpoint. A cost the server reports always wins over the price.
|
|
68
|
+
Zero means no API fee, not zero compute. Configure a price before using a dollar budget against
|
|
69
|
+
a metered remote server; an offline estimate is a byte-based approximation and cannot predict
|
|
70
|
+
the server's adaptive re-reads.
|
|
71
|
+
|
|
72
|
+
## Joint reads and the cache
|
|
73
|
+
|
|
74
|
+
A diffusion read answers every question in a batch in the light of the others, so the runtime
|
|
75
|
+
keys each of this provider's answers on the provider, endpoint, model, state, and the entire
|
|
76
|
+
ordered batch of questions, IDs included. A batch is reused only whole: if any slot is missing
|
|
77
|
+
from the cache, the whole batch is sent again. Hosted providers and Laya keep per-question
|
|
78
|
+
caching. Pin both server and weights; after changing weights or inference settings behind the
|
|
79
|
+
same URL and model, use `--no-cache` or a separate `XDG_CACHE_HOME`.
|
|
80
|
+
|
|
81
|
+
## What to expect
|
|
82
|
+
|
|
83
|
+
A September 2026 jgrep comparison over 10,000 public-data decisions found news classification
|
|
84
|
+
close to hosted Jev but more false positives on SMS spam at a 0.9 cutoff. For code, judge whole
|
|
85
|
+
diff hunks or functions; isolated diff lines recalled poorly on a small synthetic fixture. Keep
|
|
86
|
+
it experimental and evaluate its cutoff on your own data.
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
# Laya (local, experimental)
|
|
2
|
+
|
|
3
|
+
Any JevKit tool can send its decisions to a Laya server on your own machine instead of a hosted
|
|
4
|
+
provider. The server is chosen only by name, never automatically, and its calls are metered at
|
|
5
|
+
zero API fees:
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
jgrep --api laya "a complaint about noise" complaints.txt
|
|
9
|
+
jsort --api laya "more urgent" tickets.txt
|
|
10
|
+
JEV_API=laya jlink link left.csv right.csv --on name
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
This guide runs the general English 421M-parameter [Laya](https://github.com/NandhaKishorM/laya)
|
|
14
|
+
checkpoint through the independent [laya-mlx port](https://github.com/mizorewww/laya-mlx) on
|
|
15
|
+
Apple silicon. No JevKit package ships the model; it runs in its own environment and process.
|
|
16
|
+
The multilingual and newer typed-decision checkpoints have not been tried.
|
|
17
|
+
|
|
18
|
+
## Setup (Apple silicon)
|
|
19
|
+
|
|
20
|
+
Keep the model dependencies in a separate environment. From a directory outside this repository:
|
|
21
|
+
|
|
22
|
+
```bash
|
|
23
|
+
git clone https://github.com/mizorewww/laya-mlx.git
|
|
24
|
+
cd laya-mlx
|
|
25
|
+
git checkout fc1df62828a3fedf4d8229fdac1cbd85f1cdf337
|
|
26
|
+
uv sync --python 3.12
|
|
27
|
+
uv pip install --python .venv/bin/python fastapi==0.141.1 uvicorn==0.53.0
|
|
28
|
+
.venv/bin/python -c 'from huggingface_hub import snapshot_download; print(snapshot_download("aac6fef/laya-mlx", revision="047678560251f28113ee8f5df4be82102c7bf336"))'
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
Use the printed snapshot path as `--checkpoint`, then start the adapter from this repository
|
|
32
|
+
with that environment's Python:
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
/path/to/laya-mlx/.venv/bin/python scripts/laya_server.py \
|
|
36
|
+
--checkpoint /path/to/downloaded/snapshot \
|
|
37
|
+
--audit /path/to/laya-audit.jsonl
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
It binds to loopback port 8081, loads FP16 weights, warms the model, and serves `/v1/systemone`.
|
|
41
|
+
`/health` reports the checkpoint, context budget, and truncation policy. Then, in another terminal:
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
printf '%s\n' 'The music next door keeps me awake.' 'The elevator is broken.' |
|
|
45
|
+
jgrep --api laya -j 4 --timeout 120 --stats -o 'a complaint about noise'
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
## Context limit
|
|
49
|
+
|
|
50
|
+
The model's 512-token window includes the question's instructions and options, leaving less
|
|
51
|
+
room for the record. The adapter checks the remaining budget separately for every question and
|
|
52
|
+
answers HTTP 422 if any state would be cropped; the tool reports an error rather than a negative
|
|
53
|
+
answer. `--allow-truncation` enables the native runtime's prefix retention instead, and every
|
|
54
|
+
request then records the available and dropped state tokens per question in the audit log.
|
|
55
|
+
Ordinary use should keep rejection on. Long question instructions can also be shortened by the
|
|
56
|
+
runtime's own question budget; the adapter audits state truncation, not question truncation.
|
|
57
|
+
|
|
58
|
+
## Configuration
|
|
59
|
+
|
|
60
|
+
| Setting | Meaning |
|
|
61
|
+
|---|---|
|
|
62
|
+
| `--api laya` or `JEV_API=laya` | Select it; it is never picked by discovery |
|
|
63
|
+
| `JEV_LAYA_URL` or `~/.config/jev/laya.url` | Full endpoint, default `http://127.0.0.1:8081/v1/systemone` |
|
|
64
|
+
| `JEV_LAYA_API_KEY` or `~/.config/jev/laya.key` | Optional bearer token for a server that authenticates; the supplied adapter does not |
|
|
65
|
+
| `--model` or `JEV_MODEL` | `laya-421m`; the supplied adapter rejects other IDs |
|
|
66
|
+
| `JEV_PRICE_PER_MTOK` | Overrides the zero price for `--stats`, `--estimate`, and dollar budgets |
|
|
67
|
+
|
|
68
|
+
`JEV_URL` overrides every endpoint. Zero API fees exclude hardware and electricity.
|
|
69
|
+
|
|
70
|
+
Laya reads each question on its own, so its answers share the ordinary per-question cache with
|
|
71
|
+
the hosted providers, keyed by provider, endpoint, model, state, and question. Pin the server
|
|
72
|
+
and the weights; after changing the model behind the same URL and model ID, use `--no-cache`
|
|
73
|
+
or a separate `XDG_CACHE_HOME`.
|
|
74
|
+
|
|
75
|
+
## What to expect
|
|
76
|
+
|
|
77
|
+
A September 2026 jgrep experiment on an M3 Ultra found fast, strong short-text classification
|
|
78
|
+
(92% top-1 on a news set) but weak precision on spam at the default 0.5 cutoff (53%, rising to
|
|
79
|
+
79% at 0.9) and weak code judgments. Treat it as a fast classifier for short texts, and evaluate
|
|
80
|
+
thresholds on your own labeled inputs before relying on the scores.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "jevkit-runtime"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.3.0"
|
|
4
4
|
description = "Shared transport, configuration, caching, and accounting for JevKit tools"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "MIT"
|
|
@@ -23,7 +23,7 @@ requires = ["hatchling"]
|
|
|
23
23
|
build-backend = "hatchling.build"
|
|
24
24
|
|
|
25
25
|
[tool.hatch.build.targets.wheel]
|
|
26
|
-
packages = ["src/
|
|
26
|
+
packages = ["src/jevkit_runtime"]
|
|
27
27
|
|
|
28
28
|
[tool.pytest.ini_options]
|
|
29
29
|
testpaths = ["tests"]
|
|
@@ -12,6 +12,10 @@ from pathlib import Path
|
|
|
12
12
|
|
|
13
13
|
CORE = Path(__file__).resolve().parents[1]
|
|
14
14
|
TOOLS = ("jgrep", "jsort", "jlink", "jselect", "jcol")
|
|
15
|
+
PACKAGED_ASSETS = {
|
|
16
|
+
"jcol": ["static/index.html"],
|
|
17
|
+
"jlink": ["assets/review.html", "assets/review.js", "assets/review.css"],
|
|
18
|
+
}
|
|
15
19
|
|
|
16
20
|
|
|
17
21
|
def command(argv, *, cwd=CORE, env=None):
|
|
@@ -39,10 +43,34 @@ def python(repo):
|
|
|
39
43
|
return repo / ".venv" / ("Scripts/python.exe" if os.name == "nt" else "bin/python")
|
|
40
44
|
|
|
41
45
|
|
|
46
|
+
def runtime_environment(executable, env=None):
|
|
47
|
+
"""Use the selected environment for child CLI processes as well as Python."""
|
|
48
|
+
result = dict(os.environ if env is None else env)
|
|
49
|
+
result["PATH"] = os.pathsep.join((str(executable.parent), result.get("PATH", os.defpath)))
|
|
50
|
+
result["VIRTUAL_ENV"] = str(executable.parent.parent)
|
|
51
|
+
return result
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def expect_core(executable, location, *, cwd, env):
|
|
55
|
+
"""The consumer's interpreter must import the core from `location`, not a stray install."""
|
|
56
|
+
command(
|
|
57
|
+
[
|
|
58
|
+
executable,
|
|
59
|
+
"-c",
|
|
60
|
+
"import sys, pathlib, jevkit_runtime; "
|
|
61
|
+
"here = pathlib.Path(jevkit_runtime.__file__).resolve().parent; "
|
|
62
|
+
"assert here == pathlib.Path(sys.argv[1]).resolve(), here",
|
|
63
|
+
location,
|
|
64
|
+
],
|
|
65
|
+
cwd=cwd,
|
|
66
|
+
env=env,
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
|
|
42
70
|
def main():
|
|
43
71
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
44
72
|
parser.add_argument("--repos-root", type=Path, default=CORE.parent)
|
|
45
|
-
parser.add_argument("--suffix", default="", help="e.g. --suffix=-
|
|
73
|
+
parser.add_argument("--suffix", default="", help="checkout name suffix, e.g. --suffix=-wip")
|
|
46
74
|
parser.add_argument("--tool", choices=TOOLS, help="operate on only one consumer")
|
|
47
75
|
parser.add_argument(
|
|
48
76
|
"--tokenizer-cache", type=Path, default=Path(tempfile.gettempdir()) / "data-gym-cache"
|
|
@@ -63,7 +91,11 @@ def main():
|
|
|
63
91
|
if args.action == "run":
|
|
64
92
|
repo = root / (args.name + args.suffix)
|
|
65
93
|
arguments = args.arguments[1:] if args.arguments[:1] == ["--"] else args.arguments
|
|
66
|
-
command(
|
|
94
|
+
command(
|
|
95
|
+
[python(repo), "-m", args.name, *arguments],
|
|
96
|
+
cwd=Path.cwd(),
|
|
97
|
+
env=runtime_environment(python(repo)),
|
|
98
|
+
)
|
|
67
99
|
return
|
|
68
100
|
for repo in repos.values():
|
|
69
101
|
if not (repo / "pyproject.toml").is_file():
|
|
@@ -89,25 +121,11 @@ def main():
|
|
|
89
121
|
temp = Path(temporary)
|
|
90
122
|
env = environment(temp, args.tokenizer_cache)
|
|
91
123
|
if args.action == "check":
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
python(repo),
|
|
98
|
-
CORE / "scripts/probe_consumer.py",
|
|
99
|
-
name,
|
|
100
|
-
"--expect-core",
|
|
101
|
-
CORE / "src/jevkit_core",
|
|
102
|
-
"--baseline-repo",
|
|
103
|
-
repo,
|
|
104
|
-
"--baseline-ref",
|
|
105
|
-
baselines[name],
|
|
106
|
-
],
|
|
107
|
-
cwd=temp,
|
|
108
|
-
env=env,
|
|
109
|
-
)
|
|
110
|
-
command([python(repo), "-m", "pytest", "-q"], cwd=repo, env=env)
|
|
124
|
+
command([python(CORE), "-m", "pytest", "-q"], env=runtime_environment(python(CORE), env))
|
|
125
|
+
for repo in repos.values():
|
|
126
|
+
consumer_env = runtime_environment(python(repo), env)
|
|
127
|
+
expect_core(python(repo), CORE / "src/jevkit_runtime", cwd=temp, env=consumer_env)
|
|
128
|
+
command([python(repo), "-m", "pytest", "-q"], cwd=repo, env=consumer_env)
|
|
111
129
|
return
|
|
112
130
|
wheels = temp / "wheels"
|
|
113
131
|
command(["uv", "build", "--no-sources", "--out-dir", wheels])
|
|
@@ -119,47 +137,27 @@ def main():
|
|
|
119
137
|
venv = temp / name
|
|
120
138
|
command(["uv", "venv", "--python", python(repo), venv])
|
|
121
139
|
executable = venv / ("Scripts/python.exe" if os.name == "nt" else "bin/python")
|
|
140
|
+
wheel_env = runtime_environment(executable, env)
|
|
122
141
|
command(["uv", "pip", "install", "--python", executable, core_wheel, package])
|
|
123
142
|
site = subprocess.check_output(
|
|
124
143
|
[str(executable), "-c", "import sysconfig; print(sysconfig.get_path('purelib'))"], text=True
|
|
125
144
|
).strip()
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
CORE / "scripts/probe_consumer.py",
|
|
130
|
-
name,
|
|
131
|
-
"--expect-core",
|
|
132
|
-
Path(site) / "jevkit_core",
|
|
133
|
-
],
|
|
134
|
-
cwd=temp,
|
|
135
|
-
env=env,
|
|
136
|
-
)
|
|
137
|
-
command([executable, "-m", name, "--version"], cwd=temp, env=env)
|
|
138
|
-
if name == "jcol":
|
|
139
|
-
cli = venv / ("Scripts/jcol.exe" if os.name == "nt" else "bin/jcol")
|
|
140
|
-
command([executable, repo / "tests/test_process.py", cli], cwd=temp, env=env)
|
|
145
|
+
expect_core(executable, Path(site) / "jevkit_runtime", cwd=temp, env=wheel_env)
|
|
146
|
+
command([executable, "-m", name, "--version"], cwd=temp, env=wheel_env)
|
|
147
|
+
for asset in PACKAGED_ASSETS.get(name, []):
|
|
141
148
|
command(
|
|
142
149
|
[
|
|
143
150
|
executable,
|
|
144
151
|
"-c",
|
|
145
152
|
"from importlib.resources import files; "
|
|
146
|
-
"assert files(
|
|
153
|
+
f"assert files({name!r}).joinpath({asset!r}).is_file()",
|
|
147
154
|
],
|
|
148
155
|
cwd=temp,
|
|
149
|
-
env=
|
|
150
|
-
)
|
|
151
|
-
if name == "jlink":
|
|
152
|
-
command(
|
|
153
|
-
[
|
|
154
|
-
executable,
|
|
155
|
-
"-c",
|
|
156
|
-
"from importlib.resources import files; "
|
|
157
|
-
"assert all(files('jlink').joinpath('assets', f).is_file() "
|
|
158
|
-
"for f in ('review.html', 'review.js', 'review.css'))",
|
|
159
|
-
],
|
|
160
|
-
cwd=temp,
|
|
161
|
-
env=env,
|
|
156
|
+
env=wheel_env,
|
|
162
157
|
)
|
|
158
|
+
if name == "jcol":
|
|
159
|
+
cli = venv / ("Scripts/jcol.exe" if os.name == "nt" else "bin/jcol")
|
|
160
|
+
command([executable, repo / "tests/test_process.py", cli], cwd=temp, env=wheel_env)
|
|
163
161
|
print(json.dumps({"wheel_checks": names, "status": "passed"}))
|
|
164
162
|
|
|
165
163
|
|