jevkit-runtime 0.1.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. {jevkit_runtime-0.1.0 → jevkit_runtime-0.3.0}/.github/workflows/downstream.yml +0 -2
  2. jevkit_runtime-0.3.0/PKG-INFO +112 -0
  3. jevkit_runtime-0.3.0/README.md +97 -0
  4. jevkit_runtime-0.3.0/docs/diffusiongemma.md +86 -0
  5. jevkit_runtime-0.3.0/docs/laya.md +80 -0
  6. {jevkit_runtime-0.1.0 → jevkit_runtime-0.3.0}/pyproject.toml +2 -2
  7. {jevkit_runtime-0.1.0 → jevkit_runtime-0.3.0}/scripts/dev.py +48 -50
  8. jevkit_runtime-0.3.0/scripts/laya_server.py +116 -0
  9. jevkit_runtime-0.3.0/src/jevkit_runtime/__init__.py +65 -0
  10. jevkit_runtime-0.3.0/src/jevkit_runtime/client.py +222 -0
  11. jevkit_runtime-0.3.0/src/jevkit_runtime/errors.py +43 -0
  12. jevkit_runtime-0.3.0/src/jevkit_runtime/meter.py +67 -0
  13. jevkit_runtime-0.3.0/src/jevkit_runtime/protocol.py +160 -0
  14. jevkit_runtime-0.3.0/src/jevkit_runtime/providers.py +183 -0
  15. jevkit_runtime-0.3.0/src/jevkit_runtime/settings.py +64 -0
  16. jevkit_runtime-0.3.0/src/jevkit_runtime/store.py +90 -0
  17. jevkit_runtime-0.3.0/src/jevkit_runtime/transport.py +90 -0
  18. jevkit_runtime-0.3.0/tests/test_client.py +267 -0
  19. jevkit_runtime-0.3.0/tests/test_dev_runner.py +60 -0
  20. jevkit_runtime-0.3.0/tests/test_protocol.py +117 -0
  21. jevkit_runtime-0.3.0/tests/test_settings_and_providers.py +139 -0
  22. jevkit_runtime-0.3.0/tests/test_store.py +77 -0
  23. jevkit_runtime-0.3.0/tests/test_transport.py +159 -0
  24. {jevkit_runtime-0.1.0 → jevkit_runtime-0.3.0}/uv.lock +1 -1
  25. jevkit_runtime-0.1.0/PKG-INFO +0 -179
  26. jevkit_runtime-0.1.0/README.md +0 -164
  27. jevkit_runtime-0.1.0/TESTING.md +0 -95
  28. jevkit_runtime-0.1.0/consumer-baselines.json +0 -7
  29. jevkit_runtime-0.1.0/scripts/probe_consumer.py +0 -232
  30. jevkit_runtime-0.1.0/src/jevkit_core/__init__.py +0 -48
  31. jevkit_runtime-0.1.0/src/jevkit_core/backends.py +0 -142
  32. jevkit_runtime-0.1.0/src/jevkit_core/cache.py +0 -92
  33. jevkit_runtime-0.1.0/src/jevkit_core/client.py +0 -109
  34. jevkit_runtime-0.1.0/src/jevkit_core/errors.py +0 -22
  35. jevkit_runtime-0.1.0/src/jevkit_core/provenance.py +0 -18
  36. jevkit_runtime-0.1.0/src/jevkit_core/transport.py +0 -119
  37. jevkit_runtime-0.1.0/src/jevkit_core/usage.py +0 -88
  38. jevkit_runtime-0.1.0/tests/test_accounting.py +0 -62
  39. jevkit_runtime-0.1.0/tests/test_backends.py +0 -82
  40. jevkit_runtime-0.1.0/tests/test_cache_and_usage.py +0 -91
  41. jevkit_runtime-0.1.0/tests/test_shared_requests.py +0 -122
  42. jevkit_runtime-0.1.0/tests/test_transport.py +0 -136
  43. {jevkit_runtime-0.1.0 → jevkit_runtime-0.3.0}/.github/workflows/publish.yml +0 -0
  44. {jevkit_runtime-0.1.0 → jevkit_runtime-0.3.0}/.github/workflows/test.yml +0 -0
  45. {jevkit_runtime-0.1.0 → jevkit_runtime-0.3.0}/.gitignore +0 -0
  46. {jevkit_runtime-0.1.0 → jevkit_runtime-0.3.0}/LICENSE +0 -0
  47. {jevkit_runtime-0.1.0 → jevkit_runtime-0.3.0}/scripts/offline/sitecustomize.py +0 -0
@@ -14,8 +14,6 @@ permissions:
14
14
  contents: read
15
15
  jobs:
16
16
  consumers:
17
- # Enabled after all five migration PRs land; manual runs remain available.
18
- if: vars.JEVKIT_CONSUMERS_READY == 'true' || github.event_name == 'workflow_dispatch'
19
17
  runs-on: ubuntu-latest
20
18
  strategy:
21
19
  fail-fast: false
@@ -0,0 +1,112 @@
1
+ Metadata-Version: 2.5
2
+ Name: jevkit-runtime
3
+ Version: 0.3.0
4
+ Summary: Shared transport, configuration, caching, and accounting for JevKit tools
5
+ Project-URL: Homepage, https://github.com/keltokhy/jevkit-core
6
+ Project-URL: Issues, https://github.com/keltokhy/jevkit-core/issues
7
+ Author: Khaled Eltokhy
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Requires-Python: >=3.10
11
+ Requires-Dist: httpx>=0.27
12
+ Provides-Extra: http2
13
+ Requires-Dist: httpx[http2]>=0.27; extra == 'http2'
14
+ Description-Content-Type: text/markdown
15
+
16
+ # JevKit core
17
+
18
+ Distribution **`jevkit-runtime`**, import **`jevkit_runtime`**. The PyPI name `jevkit-core` belongs
19
+ to a different project.
20
+
21
+ One request pipeline, one answer store, one provider catalog for jgrep, jsort, jlink, jselect,
22
+ and jcol. Each tool remains its own package and repository; the core imports none of them, and a
23
+ tool's adapter is a few lines naming which providers it offers.
24
+
25
+ ## What a tool gets
26
+
27
+ ```python
28
+ from jevkit_runtime import AnswerStore, Client, catalog, resolve
29
+
30
+ PROVIDERS = catalog("typesafe", "openrouter", "gateway")
31
+ backend = resolve(PROVIDERS, name=None, model=None) # or JEV_API / JEV_MODEL, else the first configured
32
+ async with Client(backend, store=AnswerStore()) as client:
33
+ answers = await client.ask(state, {"q": {"type": "noul", "instructions": "..."}})
34
+ ```
35
+
36
+ `Client.ask` does the whole thing: computes each question's identity, serves what the store already
37
+ knows, joins an identical request already in flight, sends only the misses, validates the entire
38
+ response before storing any of it, and meters the call before validation so a billed but malformed
39
+ answer still counts. It returns `Answers`, a dict by question id whose `origins` say who answered
40
+ each one and whether it came from the API, the store, or a shared call. Per-call policy is keyword
41
+ arguments: `allow_paid=False` for cache-only runs, `on_cost` for the caller who should be charged,
42
+ `hedge_after` to resend a slow call, and `keys` for callers whose reuse unit is not the request.
43
+ HTTP/2 is used whenever the `http2` extra is installed.
44
+
45
+ | Module | Owns |
46
+ |---|---|
47
+ | `settings.py` | Every environment and filesystem convention, read in one place: `XDG_*`, `JEV_API`, `JEV_URL`, `JEV_MODEL`, `JEV_PRICE_PER_MTOK`, provider keys and URL files |
48
+ | `providers.py` | The catalog (`Provider`), a tool's selection of it or its own entries, and `resolve()` to one `Backend`: endpoint, model, key |
49
+ | `protocol.py` | Request bodies, typed answer validation (`noul`, `choice`, `score`), usage parsing, answer identity, provenance |
50
+ | `transport.py` | One HTTP call with a total deadline, retries with backoff and `Retry-After`, structured status errors |
51
+ | `store.py` | SQLite answers with their provenance in one row, one versioned schema |
52
+ | `client.py` | The pipeline above, request sharing, hedging |
53
+ | `meter.py` | Calls, cache hits, retries, hedges, tokens, cost, and which models actually answered |
54
+ | `errors.py` | `JevError`, `JevFatal`, `JevBudgetExceeded`, `RequestExhausted`, `ProviderError`, `ProviderFatal` |
55
+
56
+ ## Conventions every tool shares
57
+
58
+ - **Answer identity** is `answer_key(backend, state, question)`: provider, endpoint, model, state and
59
+ question. An answer from one provider or model is never served for another.
60
+ - **The store** lives at `$XDG_CACHE_HOME/jev/answers.sqlite` (default `~/.cache/jev`), is created
61
+ private to the user, and resets itself when it finds an older schema. Version 0.2 cannot read
62
+ caches written by 0.1 tools; the first run after upgrading re-asks.
63
+ - **Credentials** come from the provider's variable, then `$XDG_CONFIG_HOME/jev/<provider>.key`.
64
+ Gateways take their URL from `JEV_GATEWAY_URL` or `<provider>.url`. `JEV_URL` overrides any endpoint.
65
+ - **Metering** refuses malformed usage rather than under-counting; a response without a reported
66
+ cost is priced from its tokens at the provider's price, zero for local servers, or the list price.
67
+ `JEV_PRICE_PER_MTOK` overrides both.
68
+ - **Errors** keep their wording across tools: a fatal status reads `PROVIDER said 401: detail`, a
69
+ bad request reads `HTTP 400: detail`, and exhaustion reads `gave up after 15s (last failure)`.
70
+ Both status errors carry `provider`, `status` and `detail` for tools that word or redact them.
71
+
72
+ ## Local servers
73
+
74
+ Two catalog entries point at System One servers on your own machine: `diffusiongemma`, an
75
+ [OpenJev](https://github.com/razorback16/openjev) server on port 8080, and `laya`, a
76
+ [laya-mlx](https://github.com/mizorewww/laya-mlx) server on port 8081. Every JevKit tool names
77
+ them in its catalog, so `--api laya` or `JEV_API=laya` works everywhere. They are never chosen
78
+ in place of a configured hosted provider, need no key, and are metered at zero API fees unless
79
+ `JEV_PRICE_PER_MTOK` says otherwise. `JEV_LAYA_URL` and `JEV_DIFFUSIONGEMMA_URL`, or the matching
80
+ `.url` files, point at a server elsewhere.
81
+
82
+ DiffusionGemma reads every question in a batch together, so the runtime keys each of its answers
83
+ on the whole ordered batch and re-sends a batch whole when any slot is missing.
84
+
85
+ No package ships the models. [docs/diffusiongemma.md](docs/diffusiongemma.md) and
86
+ [docs/laya.md](docs/laya.md) explain how to run the servers, and `scripts/laya_server.py` is the
87
+ adapter the Laya guide starts.
88
+
89
+ ## Development
90
+
91
+ Keep the six checkouts as siblings. Each consumer depends on `jevkit-runtime>=0.2.0,<0.3.0` and
92
+ overrides it for development with `jevkit-runtime = { path = "../jevkit-core", editable = true }`
93
+ under `[tool.uv.sources]`.
94
+
95
+ ```bash
96
+ python3 scripts/dev.py setup # uv sync every checkout, fetch jselect's tokenizer data
97
+ python3 scripts/dev.py check # core and consumer suites, credentials stripped, sockets blocked
98
+ python3 scripts/dev.py wheel-check # build and exercise real wheel installs in temporary environments
99
+ python3 scripts/dev.py run jgrep -- --help
100
+ ```
101
+
102
+ `check` also proves each consumer imports this exact source tree; `wheel-check` proves the installed
103
+ wheel, not the checkout. `--tool NAME` limits either to one consumer and `--suffix` selects
104
+ alternatively named checkouts. The GitHub workflows run the core suite on Python 3.10 and 3.13 and
105
+ the downstream matrix against each consumer's main branch.
106
+
107
+ ## Releasing
108
+
109
+ Tag the verified core `vX.Y.Z` and dispatch the publish workflow with that tag; the workflow checks
110
+ the tag matches the package version and publishes through PyPI Trusted Publishing. Then release each
111
+ consumer through its own process, bumping its supported core range, lockfile, and CI core reference
112
+ together.
@@ -0,0 +1,97 @@
1
+ # JevKit core
2
+
3
+ Distribution **`jevkit-runtime`**, import **`jevkit_runtime`**. The PyPI name `jevkit-core` belongs
4
+ to a different project.
5
+
6
+ One request pipeline, one answer store, one provider catalog for jgrep, jsort, jlink, jselect,
7
+ and jcol. Each tool remains its own package and repository; the core imports none of them, and a
8
+ tool's adapter is a few lines naming which providers it offers.
9
+
10
+ ## What a tool gets
11
+
12
+ ```python
13
+ from jevkit_runtime import AnswerStore, Client, catalog, resolve
14
+
15
+ PROVIDERS = catalog("typesafe", "openrouter", "gateway")
16
+ backend = resolve(PROVIDERS, name=None, model=None) # or JEV_API / JEV_MODEL, else the first configured
17
+ async with Client(backend, store=AnswerStore()) as client:
18
+ answers = await client.ask(state, {"q": {"type": "noul", "instructions": "..."}})
19
+ ```
20
+
21
+ `Client.ask` does the whole thing: computes each question's identity, serves what the store already
22
+ knows, joins an identical request already in flight, sends only the misses, validates the entire
23
+ response before storing any of it, and meters the call before validation so a billed but malformed
24
+ answer still counts. It returns `Answers`, a dict by question id whose `origins` say who answered
25
+ each one and whether it came from the API, the store, or a shared call. Per-call policy is keyword
26
+ arguments: `allow_paid=False` for cache-only runs, `on_cost` for the caller who should be charged,
27
+ `hedge_after` to resend a slow call, and `keys` for callers whose reuse unit is not the request.
28
+ HTTP/2 is used whenever the `http2` extra is installed.
29
+
30
+ | Module | Owns |
31
+ |---|---|
32
+ | `settings.py` | Every environment and filesystem convention, read in one place: `XDG_*`, `JEV_API`, `JEV_URL`, `JEV_MODEL`, `JEV_PRICE_PER_MTOK`, provider keys and URL files |
33
+ | `providers.py` | The catalog (`Provider`), a tool's selection of it or its own entries, and `resolve()` to one `Backend`: endpoint, model, key |
34
+ | `protocol.py` | Request bodies, typed answer validation (`noul`, `choice`, `score`), usage parsing, answer identity, provenance |
35
+ | `transport.py` | One HTTP call with a total deadline, retries with backoff and `Retry-After`, structured status errors |
36
+ | `store.py` | SQLite answers with their provenance in one row, one versioned schema |
37
+ | `client.py` | The pipeline above, request sharing, hedging |
38
+ | `meter.py` | Calls, cache hits, retries, hedges, tokens, cost, and which models actually answered |
39
+ | `errors.py` | `JevError`, `JevFatal`, `JevBudgetExceeded`, `RequestExhausted`, `ProviderError`, `ProviderFatal` |
40
+
41
+ ## Conventions every tool shares
42
+
43
+ - **Answer identity** is `answer_key(backend, state, question)`: provider, endpoint, model, state and
44
+ question. An answer from one provider or model is never served for another.
45
+ - **The store** lives at `$XDG_CACHE_HOME/jev/answers.sqlite` (default `~/.cache/jev`), is created
46
+ private to the user, and resets itself when it finds an older schema. Version 0.2 cannot read
47
+ caches written by 0.1 tools; the first run after upgrading re-asks.
48
+ - **Credentials** come from the provider's variable, then `$XDG_CONFIG_HOME/jev/<provider>.key`.
49
+ Gateways take their URL from `JEV_GATEWAY_URL` or `<provider>.url`. `JEV_URL` overrides any endpoint.
50
+ - **Metering** refuses malformed usage rather than under-counting; a response without a reported
51
+ cost is priced from its tokens at the provider's price, zero for local servers, or the list price.
52
+ `JEV_PRICE_PER_MTOK` overrides both.
53
+ - **Errors** keep their wording across tools: a fatal status reads `PROVIDER said 401: detail`, a
54
+ bad request reads `HTTP 400: detail`, and exhaustion reads `gave up after 15s (last failure)`.
55
+ Both status errors carry `provider`, `status` and `detail` for tools that word or redact them.
56
+
57
+ ## Local servers
58
+
59
+ Two catalog entries point at System One servers on your own machine: `diffusiongemma`, an
60
+ [OpenJev](https://github.com/razorback16/openjev) server on port 8080, and `laya`, a
61
+ [laya-mlx](https://github.com/mizorewww/laya-mlx) server on port 8081. Every JevKit tool names
62
+ them in its catalog, so `--api laya` or `JEV_API=laya` works everywhere. They are never chosen
63
+ in place of a configured hosted provider, need no key, and are metered at zero API fees unless
64
+ `JEV_PRICE_PER_MTOK` says otherwise. `JEV_LAYA_URL` and `JEV_DIFFUSIONGEMMA_URL`, or the matching
65
+ `.url` files, point at a server elsewhere.
66
+
67
+ DiffusionGemma reads every question in a batch together, so the runtime keys each of its answers
68
+ on the whole ordered batch and re-sends a batch whole when any slot is missing.
69
+
70
+ No package ships the models. [docs/diffusiongemma.md](docs/diffusiongemma.md) and
71
+ [docs/laya.md](docs/laya.md) explain how to run the servers, and `scripts/laya_server.py` is the
72
+ adapter the Laya guide starts.
73
+
74
+ ## Development
75
+
76
+ Keep the six checkouts as siblings. Each consumer depends on `jevkit-runtime>=0.2.0,<0.3.0` and
77
+ overrides it for development with `jevkit-runtime = { path = "../jevkit-core", editable = true }`
78
+ under `[tool.uv.sources]`.
79
+
80
+ ```bash
81
+ python3 scripts/dev.py setup # uv sync every checkout, fetch jselect's tokenizer data
82
+ python3 scripts/dev.py check # core and consumer suites, credentials stripped, sockets blocked
83
+ python3 scripts/dev.py wheel-check # build and exercise real wheel installs in temporary environments
84
+ python3 scripts/dev.py run jgrep -- --help
85
+ ```
86
+
87
+ `check` also proves each consumer imports this exact source tree; `wheel-check` proves the installed
88
+ wheel, not the checkout. `--tool NAME` limits either to one consumer and `--suffix` selects
89
+ alternatively named checkouts. The GitHub workflows run the core suite on Python 3.10 and 3.13 and
90
+ the downstream matrix against each consumer's main branch.
91
+
92
+ ## Releasing
93
+
94
+ Tag the verified core `vX.Y.Z` and dispatch the publish workflow with that tag; the workflow checks
95
+ the tag matches the package version and publishes through PyPI Trusted Publishing. Then release each
96
+ consumer through its own process, bumping its supported core range, lockfile, and CI core reference
97
+ together.
@@ -0,0 +1,86 @@
1
+ # DiffusionGemma (local, experimental)
2
+
3
+ Any JevKit tool can send its decisions to a DiffusionGemma decision server on your own machine
4
+ instead of a hosted provider. The server is chosen only by name, never automatically, and its
5
+ calls are metered at zero API fees:
6
+
7
+ ```bash
8
+ jgrep --api diffusiongemma -j 1 "a complaint about noise" complaints.txt
9
+ jcol --api diffusiongemma run table.csv codebook.json
10
+ ```
11
+
12
+ No JevKit package ships the model; it runs in its own environment and process. Model quality
13
+ and useful probability thresholds need evaluation on your own inputs.
14
+
15
+ ## Apple silicon
16
+
17
+ [OpenJev](https://github.com/razorback16/openjev) supplies an MLX implementation of the
18
+ structured-read approach from [vLLM PR #57250](https://github.com/vllm-project/vllm/pull/57250).
19
+ Its 4-bit checkpoint is roughly 16.6 GB to download and needs about 16 GB of model memory plus
20
+ working memory. Install the server in its own directory and environment:
21
+
22
+ ```bash
23
+ git clone https://github.com/razorback16/openjev.git
24
+ cd openjev
25
+ git checkout e04794ab36e4f7e6040c2547baecdb2737ce2e79
26
+ uv sync --python 3.12 --extra mlx
27
+ OPENJEV_BACKEND=mlx uv run --extra mlx python -m openjev
28
+ ```
29
+
30
+ The server downloads `mlx-community/diffusiongemma-26B-A4B-it-4bit` on first start. Wait for
31
+ `Application startup complete` before querying it. `OPENJEV_MLX_MODEL` can point at a downloaded
32
+ snapshot to pin the weights; the revision used in the original experiment was
33
+ `a7a81407613811e8ba63af92ac0d852b809e191f`. Then, in another terminal:
34
+
35
+ ```bash
36
+ printf '%s\n' 'The music next door keeps me awake.' 'The elevator is broken.' |
37
+ jgrep --api diffusiongemma --model openjev-0.1 -j 1 --timeout 120 --stats -o 'a complaint about noise'
38
+ ```
39
+
40
+ Start with one request at a time: the MLX backend executes model work serially, and a deep
41
+ queue can exceed a tool's per-request deadline. The longer timeout covers the first inference.
42
+ Raise concurrency from measured throughput once the model is warm.
43
+
44
+ ## NVIDIA / vLLM
45
+
46
+ Run OpenJev's documented vLLM deployment, or the prototype `structured_server.py` from
47
+ [PR #57250](https://github.com/vllm-project/vllm/pull/57250). That PR was unmerged at the time of
48
+ writing, so a released vLLM is not enough on its own; follow the server's pinned build
49
+ instructions. Its `/v1/systemone` adapter sits in front of vLLM. For the PR example server's
50
+ default port:
51
+
52
+ ```bash
53
+ JEV_DIFFUSIONGEMMA_URL=http://127.0.0.1:8011/v1/systemone \
54
+ jgrep --api diffusiongemma --model jev-latest 'a stack trace' build.log
55
+ ```
56
+
57
+ ## Configuration
58
+
59
+ | Setting | Meaning |
60
+ |---|---|
61
+ | `--api diffusiongemma` or `JEV_API=diffusiongemma` | Select it; it is never picked by discovery |
62
+ | `JEV_DIFFUSIONGEMMA_URL` or `~/.config/jev/diffusiongemma.url` | Full endpoint, default `http://127.0.0.1:8080/v1/systemone` |
63
+ | `JEV_DIFFUSIONGEMMA_API_KEY` or `~/.config/jev/diffusiongemma.key` | Optional bearer token; no `Authorization` header when absent |
64
+ | `--model` or `JEV_MODEL` | Default `openjev-latest`; `openjev-0.1` pins the server's decision model |
65
+ | `JEV_PRICE_PER_MTOK` | Overrides the zero price for `--stats`, `--estimate`, and dollar budgets |
66
+
67
+ `JEV_URL` overrides every endpoint. A cost the server reports always wins over the price.
68
+ Zero means no API fee, not zero compute. Configure a price before using a dollar budget against
69
+ a metered remote server; an offline estimate is a byte-based approximation and cannot predict
70
+ the server's adaptive re-reads.
71
+
72
+ ## Joint reads and the cache
73
+
74
+ A diffusion read answers every question in a batch in the light of the others, so the runtime
75
+ keys each of this provider's answers on the provider, endpoint, model, state, and the entire
76
+ ordered batch of questions, IDs included. A batch is reused only whole: if any slot is missing
77
+ from the cache, the whole batch is sent again. Hosted providers and Laya keep per-question
78
+ caching. Pin both server and weights; after changing weights or inference settings behind the
79
+ same URL and model, use `--no-cache` or a separate `XDG_CACHE_HOME`.
80
+
81
+ ## What to expect
82
+
83
+ A September 2026 jgrep comparison over 10,000 public-data decisions found news classification
84
+ close to hosted Jev but more false positives on SMS spam at a 0.9 cutoff. For code, judge whole
85
+ diff hunks or functions; isolated diff lines recalled poorly on a small synthetic fixture. Keep
86
+ it experimental and evaluate its cutoff on your own data.
@@ -0,0 +1,80 @@
1
+ # Laya (local, experimental)
2
+
3
+ Any JevKit tool can send its decisions to a Laya server on your own machine instead of a hosted
4
+ provider. The server is chosen only by name, never automatically, and its calls are metered at
5
+ zero API fees:
6
+
7
+ ```bash
8
+ jgrep --api laya "a complaint about noise" complaints.txt
9
+ jsort --api laya "more urgent" tickets.txt
10
+ JEV_API=laya jlink link left.csv right.csv --on name
11
+ ```
12
+
13
+ This guide runs the general English 421M-parameter [Laya](https://github.com/NandhaKishorM/laya)
14
+ checkpoint through the independent [laya-mlx port](https://github.com/mizorewww/laya-mlx) on
15
+ Apple silicon. No JevKit package ships the model; it runs in its own environment and process.
16
+ The multilingual and newer typed-decision checkpoints have not been tried.
17
+
18
+ ## Setup (Apple silicon)
19
+
20
+ Keep the model dependencies in a separate environment. From a directory outside this repository:
21
+
22
+ ```bash
23
+ git clone https://github.com/mizorewww/laya-mlx.git
24
+ cd laya-mlx
25
+ git checkout fc1df62828a3fedf4d8229fdac1cbd85f1cdf337
26
+ uv sync --python 3.12
27
+ uv pip install --python .venv/bin/python fastapi==0.141.1 uvicorn==0.53.0
28
+ .venv/bin/python -c 'from huggingface_hub import snapshot_download; print(snapshot_download("aac6fef/laya-mlx", revision="047678560251f28113ee8f5df4be82102c7bf336"))'
29
+ ```
30
+
31
+ Use the printed snapshot path as `--checkpoint`, then start the adapter from this repository
32
+ with that environment's Python:
33
+
34
+ ```bash
35
+ /path/to/laya-mlx/.venv/bin/python scripts/laya_server.py \
36
+ --checkpoint /path/to/downloaded/snapshot \
37
+ --audit /path/to/laya-audit.jsonl
38
+ ```
39
+
40
+ It binds to loopback port 8081, loads FP16 weights, warms the model, and serves `/v1/systemone`.
41
+ `/health` reports the checkpoint, context budget, and truncation policy. Then, in another terminal:
42
+
43
+ ```bash
44
+ printf '%s\n' 'The music next door keeps me awake.' 'The elevator is broken.' |
45
+ jgrep --api laya -j 4 --timeout 120 --stats -o 'a complaint about noise'
46
+ ```
47
+
48
+ ## Context limit
49
+
50
+ The model's 512-token window includes the question's instructions and options, leaving less
51
+ room for the record. The adapter checks the remaining budget separately for every question and
52
+ answers HTTP 422 if any state would be cropped; the tool reports an error rather than a negative
53
+ answer. `--allow-truncation` enables the native runtime's prefix retention instead, and every
54
+ request then records the available and dropped state tokens per question in the audit log.
55
+ Ordinary use should keep rejection on. Long question instructions can also be shortened by the
56
+ runtime's own question budget; the adapter audits state truncation, not question truncation.
57
+
58
+ ## Configuration
59
+
60
+ | Setting | Meaning |
61
+ |---|---|
62
+ | `--api laya` or `JEV_API=laya` | Select it; it is never picked by discovery |
63
+ | `JEV_LAYA_URL` or `~/.config/jev/laya.url` | Full endpoint, default `http://127.0.0.1:8081/v1/systemone` |
64
+ | `JEV_LAYA_API_KEY` or `~/.config/jev/laya.key` | Optional bearer token for a server that authenticates; the supplied adapter does not |
65
+ | `--model` or `JEV_MODEL` | `laya-421m`; the supplied adapter rejects other IDs |
66
+ | `JEV_PRICE_PER_MTOK` | Overrides the zero price for `--stats`, `--estimate`, and dollar budgets |
67
+
68
+ `JEV_URL` overrides every endpoint. Zero API fees exclude hardware and electricity.
69
+
70
+ Laya reads each question on its own, so its answers share the ordinary per-question cache with
71
+ the hosted providers, keyed by provider, endpoint, model, state, and question. Pin the server
72
+ and the weights; after changing the model behind the same URL and model ID, use `--no-cache`
73
+ or a separate `XDG_CACHE_HOME`.
74
+
75
+ ## What to expect
76
+
77
+ A September 2026 jgrep experiment on an M3 Ultra found fast, strong short-text classification
78
+ (92% top-1 on a news set) but weak precision on spam at the default 0.5 cutoff (53%, rising to
79
+ 79% at 0.9) and weak code judgments. Treat it as a fast classifier for short texts, and evaluate
80
+ thresholds on your own labeled inputs before relying on the scores.
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "jevkit-runtime"
3
- version = "0.1.0"
3
+ version = "0.3.0"
4
4
  description = "Shared transport, configuration, caching, and accounting for JevKit tools"
5
5
  readme = "README.md"
6
6
  license = "MIT"
@@ -23,7 +23,7 @@ requires = ["hatchling"]
23
23
  build-backend = "hatchling.build"
24
24
 
25
25
  [tool.hatch.build.targets.wheel]
26
- packages = ["src/jevkit_core"]
26
+ packages = ["src/jevkit_runtime"]
27
27
 
28
28
  [tool.pytest.ini_options]
29
29
  testpaths = ["tests"]
@@ -12,6 +12,10 @@ from pathlib import Path
12
12
 
13
13
  CORE = Path(__file__).resolve().parents[1]
14
14
  TOOLS = ("jgrep", "jsort", "jlink", "jselect", "jcol")
15
+ PACKAGED_ASSETS = {
16
+ "jcol": ["static/index.html"],
17
+ "jlink": ["assets/review.html", "assets/review.js", "assets/review.css"],
18
+ }
15
19
 
16
20
 
17
21
  def command(argv, *, cwd=CORE, env=None):
@@ -39,10 +43,34 @@ def python(repo):
39
43
  return repo / ".venv" / ("Scripts/python.exe" if os.name == "nt" else "bin/python")
40
44
 
41
45
 
46
+ def runtime_environment(executable, env=None):
47
+ """Use the selected environment for child CLI processes as well as Python."""
48
+ result = dict(os.environ if env is None else env)
49
+ result["PATH"] = os.pathsep.join((str(executable.parent), result.get("PATH", os.defpath)))
50
+ result["VIRTUAL_ENV"] = str(executable.parent.parent)
51
+ return result
52
+
53
+
54
+ def expect_core(executable, location, *, cwd, env):
55
+ """The consumer's interpreter must import the core from `location`, not a stray install."""
56
+ command(
57
+ [
58
+ executable,
59
+ "-c",
60
+ "import sys, pathlib, jevkit_runtime; "
61
+ "here = pathlib.Path(jevkit_runtime.__file__).resolve().parent; "
62
+ "assert here == pathlib.Path(sys.argv[1]).resolve(), here",
63
+ location,
64
+ ],
65
+ cwd=cwd,
66
+ env=env,
67
+ )
68
+
69
+
42
70
  def main():
43
71
  parser = argparse.ArgumentParser(description=__doc__)
44
72
  parser.add_argument("--repos-root", type=Path, default=CORE.parent)
45
- parser.add_argument("--suffix", default="", help="e.g. --suffix=-jevkit for isolated migration worktrees")
73
+ parser.add_argument("--suffix", default="", help="checkout name suffix, e.g. --suffix=-wip")
46
74
  parser.add_argument("--tool", choices=TOOLS, help="operate on only one consumer")
47
75
  parser.add_argument(
48
76
  "--tokenizer-cache", type=Path, default=Path(tempfile.gettempdir()) / "data-gym-cache"
@@ -63,7 +91,11 @@ def main():
63
91
  if args.action == "run":
64
92
  repo = root / (args.name + args.suffix)
65
93
  arguments = args.arguments[1:] if args.arguments[:1] == ["--"] else args.arguments
66
- command([python(repo), "-m", args.name, *arguments], cwd=Path.cwd())
94
+ command(
95
+ [python(repo), "-m", args.name, *arguments],
96
+ cwd=Path.cwd(),
97
+ env=runtime_environment(python(repo)),
98
+ )
67
99
  return
68
100
  for repo in repos.values():
69
101
  if not (repo / "pyproject.toml").is_file():
@@ -89,25 +121,11 @@ def main():
89
121
  temp = Path(temporary)
90
122
  env = environment(temp, args.tokenizer_cache)
91
123
  if args.action == "check":
92
- baselines = json.loads((CORE / "consumer-baselines.json").read_text())
93
- command([python(CORE), "-m", "pytest", "-q"], env=env)
94
- for name, repo in repos.items():
95
- command(
96
- [
97
- python(repo),
98
- CORE / "scripts/probe_consumer.py",
99
- name,
100
- "--expect-core",
101
- CORE / "src/jevkit_core",
102
- "--baseline-repo",
103
- repo,
104
- "--baseline-ref",
105
- baselines[name],
106
- ],
107
- cwd=temp,
108
- env=env,
109
- )
110
- command([python(repo), "-m", "pytest", "-q"], cwd=repo, env=env)
124
+ command([python(CORE), "-m", "pytest", "-q"], env=runtime_environment(python(CORE), env))
125
+ for repo in repos.values():
126
+ consumer_env = runtime_environment(python(repo), env)
127
+ expect_core(python(repo), CORE / "src/jevkit_runtime", cwd=temp, env=consumer_env)
128
+ command([python(repo), "-m", "pytest", "-q"], cwd=repo, env=consumer_env)
111
129
  return
112
130
  wheels = temp / "wheels"
113
131
  command(["uv", "build", "--no-sources", "--out-dir", wheels])
@@ -119,47 +137,27 @@ def main():
119
137
  venv = temp / name
120
138
  command(["uv", "venv", "--python", python(repo), venv])
121
139
  executable = venv / ("Scripts/python.exe" if os.name == "nt" else "bin/python")
140
+ wheel_env = runtime_environment(executable, env)
122
141
  command(["uv", "pip", "install", "--python", executable, core_wheel, package])
123
142
  site = subprocess.check_output(
124
143
  [str(executable), "-c", "import sysconfig; print(sysconfig.get_path('purelib'))"], text=True
125
144
  ).strip()
126
- command(
127
- [
128
- executable,
129
- CORE / "scripts/probe_consumer.py",
130
- name,
131
- "--expect-core",
132
- Path(site) / "jevkit_core",
133
- ],
134
- cwd=temp,
135
- env=env,
136
- )
137
- command([executable, "-m", name, "--version"], cwd=temp, env=env)
138
- if name == "jcol":
139
- cli = venv / ("Scripts/jcol.exe" if os.name == "nt" else "bin/jcol")
140
- command([executable, repo / "tests/test_process.py", cli], cwd=temp, env=env)
145
+ expect_core(executable, Path(site) / "jevkit_runtime", cwd=temp, env=wheel_env)
146
+ command([executable, "-m", name, "--version"], cwd=temp, env=wheel_env)
147
+ for asset in PACKAGED_ASSETS.get(name, []):
141
148
  command(
142
149
  [
143
150
  executable,
144
151
  "-c",
145
152
  "from importlib.resources import files; "
146
- "assert files('jcol').joinpath('static/index.html').is_file()",
153
+ f"assert files({name!r}).joinpath({asset!r}).is_file()",
147
154
  ],
148
155
  cwd=temp,
149
- env=env,
150
- )
151
- if name == "jlink":
152
- command(
153
- [
154
- executable,
155
- "-c",
156
- "from importlib.resources import files; "
157
- "assert all(files('jlink').joinpath('assets', f).is_file() "
158
- "for f in ('review.html', 'review.js', 'review.css'))",
159
- ],
160
- cwd=temp,
161
- env=env,
156
+ env=wheel_env,
162
157
  )
158
+ if name == "jcol":
159
+ cli = venv / ("Scripts/jcol.exe" if os.name == "nt" else "bin/jcol")
160
+ command([executable, repo / "tests/test_process.py", cli], cwd=temp, env=wheel_env)
163
161
  print(json.dumps({"wheel_checks": names, "status": "passed"}))
164
162
 
165
163