llms2jev 0.3.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llms2jev-0.3.3/.gitignore +9 -0
- llms2jev-0.3.3/CONTRIBUTING.md +66 -0
- llms2jev-0.3.3/LICENSE +21 -0
- llms2jev-0.3.3/PKG-INFO +278 -0
- llms2jev-0.3.3/README.md +244 -0
- llms2jev-0.3.3/benchmarks/README.md +45 -0
- llms2jev-0.3.3/benchmarks/live_demo.py +49 -0
- llms2jev-0.3.3/benchmarks/local_inference.py +157 -0
- llms2jev-0.3.3/benchmarks/plot_results.py +51 -0
- llms2jev-0.3.3/benchmarks/record_terminal.py +80 -0
- llms2jev-0.3.3/benchmarks/refund_cases.json +14 -0
- llms2jev-0.3.3/benchmarks/refund_workload.py +27 -0
- llms2jev-0.3.3/benchmarks/results/m4-qwen2.5-0.5b-configured-strict-json.json +2650 -0
- llms2jev-0.3.3/benchmarks/results/m4-qwen2.5-0.5b-configured.json +2724 -0
- llms2jev-0.3.3/benchmarks/results/m4-qwen2.5-0.5b.json +2644 -0
- llms2jev-0.3.3/benchmarks/results/m4-qwen3-0.6b-after-optimization.json +964 -0
- llms2jev-0.3.3/benchmarks/results/m4-qwen3-0.6b-before-optimization.json +1737 -0
- llms2jev-0.3.3/benchmarks/results/ollama-smoke.json +47 -0
- llms2jev-0.3.3/benchmarks/results/refund-configuration.json +50 -0
- llms2jev-0.3.3/docs/api.md +99 -0
- llms2jev-0.3.3/docs/architecture.md +152 -0
- llms2jev-0.3.3/docs/assets/benchmark-local.png +0 -0
- llms2jev-0.3.3/docs/assets/benchmark-qwen3.png +0 -0
- llms2jev-0.3.3/docs/assets/local-inference-poster.png +0 -0
- llms2jev-0.3.3/docs/assets/local-inference.json +110 -0
- llms2jev-0.3.3/docs/assets/local-inference.mp4 +0 -0
- llms2jev-0.3.3/docs/assets/readme-hero.png +0 -0
- llms2jev-0.3.3/docs/binary-question-design.md +25 -0
- llms2jev-0.3.3/docs/concepts.md +57 -0
- llms2jev-0.3.3/docs/decisions/component-contracts.md +59 -0
- llms2jev-0.3.3/docs/decisions/evidence-api.md +27 -0
- llms2jev-0.3.3/docs/decisions/project-review.md +111 -0
- llms2jev-0.3.3/docs/decisions/python38-components.md +55 -0
- llms2jev-0.3.3/docs/decisions/release-0.2.0.md +27 -0
- llms2jev-0.3.3/docs/decisions/review-resolution.md +22 -0
- llms2jev-0.3.3/docs/evaluation.md +58 -0
- llms2jev-0.3.3/docs/guides/cli-mcp.md +257 -0
- llms2jev-0.3.3/docs/guides/http.md +32 -0
- llms2jev-0.3.3/docs/guides/ollama.md +47 -0
- llms2jev-0.3.3/docs/guides/python.md +74 -0
- llms2jev-0.3.3/docs/implementation-plan.md +89 -0
- llms2jev-0.3.3/docs/index.md +20 -0
- llms2jev-0.3.3/docs/privacy.md +30 -0
- llms2jev-0.3.3/docs/reports/archive-comparison-after.json +518 -0
- llms2jev-0.3.3/docs/reports/archive-comparison-before.json +1595 -0
- llms2jev-0.3.3/docs/reports/cli-mcp-local.json +99 -0
- llms2jev-0.3.3/docs/reports/cli-mcp-validation.md +62 -0
- llms2jev-0.3.3/examples/bound_evaluation.py +34 -0
- llms2jev-0.3.3/examples/mcp_evaluate.py +48 -0
- llms2jev-0.3.3/examples/ollama_inference.py +43 -0
- llms2jev-0.3.3/examples/transformers_inference.py +37 -0
- llms2jev-0.3.3/examples/triage-request.json +24 -0
- llms2jev-0.3.3/pyproject.toml +79 -0
- llms2jev-0.3.3/src/llm2jev/__init__.py +80 -0
- llms2jev-0.3.3/src/llm2jev/application/__init__.py +8 -0
- llms2jev-0.3.3/src/llm2jev/application/bound.py +103 -0
- llms2jev-0.3.3/src/llm2jev/application/pipeline.py +119 -0
- llms2jev-0.3.3/src/llm2jev/application/service.py +120 -0
- llms2jev-0.3.3/src/llm2jev/contracts.py +135 -0
- llms2jev-0.3.3/src/llm2jev/core/__init__.py +26 -0
- llms2jev-0.3.3/src/llm2jev/core/answers.py +120 -0
- llms2jev-0.3.3/src/llm2jev/core/questions.py +114 -0
- llms2jev-0.3.3/src/llm2jev/core/request.py +41 -0
- llms2jev-0.3.3/src/llm2jev/core/response.py +122 -0
- llms2jev-0.3.3/src/llm2jev/core/types.py +18 -0
- llms2jev-0.3.3/src/llm2jev/core/validation.py +55 -0
- llms2jev-0.3.3/src/llm2jev/inference/__init__.py +13 -0
- llms2jev-0.3.3/src/llm2jev/inference/compilation/__init__.py +3 -0
- llms2jev-0.3.3/src/llm2jev/inference/compilation/compiler.py +70 -0
- llms2jev-0.3.3/src/llm2jev/inference/compilation/plan.py +80 -0
- llms2jev-0.3.3/src/llm2jev/inference/distribution.py +31 -0
- llms2jev-0.3.3/src/llm2jev/inference/encoding.py +99 -0
- llms2jev-0.3.3/src/llm2jev/inference/probes.py +94 -0
- llms2jev-0.3.3/src/llm2jev/inference/projection.py +134 -0
- llms2jev-0.3.3/src/llm2jev/py.typed +0 -0
- llms2jev-0.3.3/src/llm2jev/runtime/__init__.py +17 -0
- llms2jev-0.3.3/src/llm2jev/runtime/lifecycle.py +147 -0
- llms2jev-0.3.3/src/llm2jev/runtime/ollama/__init__.py +8 -0
- llms2jev-0.3.3/src/llm2jev/runtime/ollama/configuration.py +28 -0
- llms2jev-0.3.3/src/llm2jev/runtime/ollama/protocol.py +154 -0
- llms2jev-0.3.3/src/llm2jev/runtime/ollama/runtime.py +112 -0
- llms2jev-0.3.3/src/llm2jev/runtime/transformers/__init__.py +7 -0
- llms2jev-0.3.3/src/llm2jev/runtime/transformers/runtime.py +181 -0
- llms2jev-0.3.3/src/llm2jev/runtime/transformers/tokenization.py +28 -0
- llms2jev-0.3.3/src/llm2jev/serving/__init__.py +3 -0
- llms2jev-0.3.3/src/llm2jev/serving/app.py +85 -0
- llms2jev-0.3.3/src/llm2jev/serving/cli.py +43 -0
- llms2jev-0.3.3/src/llm2jev/serving/client_cli.py +62 -0
- llms2jev-0.3.3/src/llm2jev/serving/mcp.py +95 -0
- llms2jev-0.3.3/src/llm2jev/transport/__init__.py +3 -0
- llms2jev-0.3.3/src/llm2jev/transport/client.py +102 -0
- llms2jev-0.3.3/src/llm2jev/transport/http.py +81 -0
- llms2jev-0.3.3/src/llm2jev/transport/parsing.py +47 -0
- llms2jev-0.3.3/src/llm2jev/utils/__init__.py +4 -0
- llms2jev-0.3.3/src/llm2jev/utils/json.py +85 -0
- llms2jev-0.3.3/src/llm2jev/utils/probability.py +48 -0
- llms2jev-0.3.3/tests/application/__init__.py +4 -0
- llms2jev-0.3.3/tests/application/test_evaluation.py +78 -0
- llms2jev-0.3.3/tests/application/test_privacy.py +51 -0
- llms2jev-0.3.3/tests/application/test_service.py +372 -0
- llms2jev-0.3.3/tests/architecture/__init__.py +4 -0
- llms2jev-0.3.3/tests/architecture/test_architecture.py +125 -0
- llms2jev-0.3.3/tests/architecture/test_documentation.py +49 -0
- llms2jev-0.3.3/tests/architecture/test_python_compatibility.py +50 -0
- llms2jev-0.3.3/tests/architecture/test_source_comparison.py +55 -0
- llms2jev-0.3.3/tests/core/__init__.py +4 -0
- llms2jev-0.3.3/tests/core/test_answers.py +67 -0
- llms2jev-0.3.3/tests/core/test_questions.py +77 -0
- llms2jev-0.3.3/tests/core/test_request.py +56 -0
- llms2jev-0.3.3/tests/core/test_response.py +63 -0
- llms2jev-0.3.3/tests/inference/__init__.py +4 -0
- llms2jev-0.3.3/tests/inference/test_distribution.py +35 -0
- llms2jev-0.3.3/tests/inference/test_encoding.py +60 -0
- llms2jev-0.3.3/tests/inference/test_projection.py +106 -0
- llms2jev-0.3.3/tests/inference/test_rules.py +93 -0
- llms2jev-0.3.3/tests/runtime/__init__.py +4 -0
- llms2jev-0.3.3/tests/runtime/test_lifecycle.py +216 -0
- llms2jev-0.3.3/tests/runtime/test_ollama_runtime.py +276 -0
- llms2jev-0.3.3/tests/runtime/test_transformers_positions.py +82 -0
- llms2jev-0.3.3/tests/runtime/test_transformers_runtime.py +42 -0
- llms2jev-0.3.3/tests/runtime/test_transformers_scoring.py +101 -0
- llms2jev-0.3.3/tests/serving/__init__.py +4 -0
- llms2jev-0.3.3/tests/serving/test_client_cli.py +50 -0
- llms2jev-0.3.3/tests/serving/test_mcp.py +71 -0
- llms2jev-0.3.3/tests/serving/test_server.py +106 -0
- llms2jev-0.3.3/tests/transport/__init__.py +4 -0
- llms2jev-0.3.3/tests/transport/test_client.py +80 -0
- llms2jev-0.3.3/tests/transport/test_http.py +139 -0
- llms2jev-0.3.3/tests/transport/test_http_parsing.py +61 -0
- llms2jev-0.3.3/tests/utils/__init__.py +4 -0
- llms2jev-0.3.3/tests/utils/test_json.py +56 -0
- llms2jev-0.3.3/tests/utils/test_probability.py +45 -0
- llms2jev-0.3.3/tools/compare_archive.py +139 -0
- llms2jev-0.3.3/uv.lock +4196 -0
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
# Contributing to LLM2Jev
|
|
2
|
+
|
|
3
|
+
PRs are welcome. Help make bounded decisions reliable and measurable: fix a failing
|
|
4
|
+
case, improve documentation, add a reproducible workload, or improve a runtime.
|
|
5
|
+
Submit issues and PRs to https://github.com/Qingbolan/llms2jev-releases.
|
|
6
|
+
|
|
7
|
+
## Propose a focused change
|
|
8
|
+
|
|
9
|
+
Explain the concrete use case, current behavior, and intended result. Small fixes
|
|
10
|
+
can go directly to a PR. Discuss public API changes, new dependencies, providers,
|
|
11
|
+
and architecture changes in an issue first. Avoid unrelated cleanup in a bug fix.
|
|
12
|
+
|
|
13
|
+
Credit sources of adapted ideas and code, and retain required license notices.
|
|
14
|
+
The public decision vocabulary and state/questions design are inspired by
|
|
15
|
+
[TypeSafe’s Jev](https://typesafe.ai/blog/introducing-system-one-models-and-jev)
|
|
16
|
+
and its [API primitives](https://docs.typesafe.ai/introduction).
|
|
17
|
+
Do not present local speed measurements as Jev results or calibrated accuracy.
|
|
18
|
+
|
|
19
|
+
## Development and verification
|
|
20
|
+
|
|
21
|
+
Use Python 3.8-compatible syntax and type annotations. CI checks Python 3.8, 3.10,
|
|
22
|
+
and 3.12; provider tensor checks run on 3.8 and 3.12. From a source checkout:
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
uv sync --python 3.8 --group dev
|
|
26
|
+
uv run python -m unittest discover -s tests -v
|
|
27
|
+
uv run python -m compileall -q src tests examples benchmarks tools
|
|
28
|
+
uv run mypy src/llm2jev
|
|
29
|
+
uv run python examples/bound_evaluation.py
|
|
30
|
+
uv build
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
For model adapter or tensor changes, install `uv sync --extra transformers --group dev`
|
|
34
|
+
and run the suite again. Repeat compatibility checks with Python 3.12. Deterministic
|
|
35
|
+
tests must run without network access or model downloads. Model-backed measurements
|
|
36
|
+
are separate evidence; report skipped checks and environment limitations explicitly.
|
|
37
|
+
|
|
38
|
+
## Implementation rules
|
|
39
|
+
|
|
40
|
+
- Keep public imports deliberate and preserve documented public and wire behavior;
|
|
41
|
+
discuss intentional breaking changes and document migration before merging.
|
|
42
|
+
- Keep domain validation in `core`, scoring contracts in `contracts.py`, orchestration
|
|
43
|
+
in `application`, algorithms in `inference`, and model/resource ownership in `runtime`.
|
|
44
|
+
Providers must not import each other or inference implementations.
|
|
45
|
+
- Consolidate shared invariants instead of adding duplicate checks or legacy fallbacks.
|
|
46
|
+
Use explicit lifecycle states for resources and keep diagnostic errors free of input content.
|
|
47
|
+
- Explain each Python module’s architecture in its docstring and each named function’s
|
|
48
|
+
contract in a preceding comment. Describe ownership, invariants, and failure behavior.
|
|
49
|
+
- Add regression tests for behavior changes and valid/invalid cases for validation rules.
|
|
50
|
+
Update affected examples and documentation together with the implementation.
|
|
51
|
+
|
|
52
|
+
See [architecture](docs/architecture.md), [privacy](docs/privacy.md), and
|
|
53
|
+
[evaluation requirements](docs/evaluation.md) for the full contracts.
|
|
54
|
+
|
|
55
|
+
## PR checklist
|
|
56
|
+
|
|
57
|
+
- Describe the problem and resulting behavior; link the relevant issue if there is one.
|
|
58
|
+
- List verification commands and results, including skips or checks not run.
|
|
59
|
+
- Identify API, probability, dependency, and lifecycle assumptions that changed.
|
|
60
|
+
- Include example JSON when changing a wire contract and migration instructions for breaking changes.
|
|
61
|
+
- For performance claims, provide hardware, model revision, dependency versions, raw
|
|
62
|
+
observations, timing boundaries, a minimal baseline, and correctness results.
|
|
63
|
+
- Exclude credentials, private inputs, downloaded model weights, and generated build artifacts.
|
|
64
|
+
|
|
65
|
+
Keep commits focused with concise imperative subjects. Maintainer review checks
|
|
66
|
+
behavior, architecture, attribution, and evidence before a change is merged.
|
llms2jev-0.3.3/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 LLM2Jev contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
llms2jev-0.3.3/PKG-INFO
ADDED
|
@@ -0,0 +1,278 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: llms2jev
|
|
3
|
+
Version: 0.3.3
|
|
4
|
+
Summary: Inspired by Jev. Turn LLMs into decision engines—score choices, skip the chat.
|
|
5
|
+
Project-URL: Homepage, https://github.com/Qingbolan/llms2jev-releases
|
|
6
|
+
Project-URL: Repository, https://github.com/Qingbolan/llms2jev-releases
|
|
7
|
+
Project-URL: Documentation, https://github.com/Qingbolan/llms2jev-releases/tree/v0.3.3/docs
|
|
8
|
+
Project-URL: Benchmarks, https://github.com/Qingbolan/llms2jev-releases/tree/v0.3.3/benchmarks
|
|
9
|
+
Author-email: "Silan.Hu" <silan.hu@u.nus.edu>
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: candidate-scoring,decision-runtime,semantic-operators
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
15
|
+
Classifier: Typing :: Typed
|
|
16
|
+
Requires-Python: >=3.8
|
|
17
|
+
Provides-Extra: client
|
|
18
|
+
Requires-Dist: httpx>=0.27; extra == 'client'
|
|
19
|
+
Provides-Extra: mcp
|
|
20
|
+
Requires-Dist: httpx>=0.27; extra == 'mcp'
|
|
21
|
+
Requires-Dist: mcp<3,>=2.2; (python_version >= '3.10') and extra == 'mcp'
|
|
22
|
+
Provides-Extra: ollama
|
|
23
|
+
Requires-Dist: httpx>=0.27; extra == 'ollama'
|
|
24
|
+
Provides-Extra: server
|
|
25
|
+
Requires-Dist: fastapi>=0.115; extra == 'server'
|
|
26
|
+
Requires-Dist: httpx>=0.27; extra == 'server'
|
|
27
|
+
Requires-Dist: uvicorn>=0.30; extra == 'server'
|
|
28
|
+
Provides-Extra: transformers
|
|
29
|
+
Requires-Dist: torch<2.5,>=2.0; (python_version < '3.9') and extra == 'transformers'
|
|
30
|
+
Requires-Dist: torch>=2.0; (python_version >= '3.9') and extra == 'transformers'
|
|
31
|
+
Requires-Dist: transformers<4.47,>=4.46.3; (python_version < '3.9') and extra == 'transformers'
|
|
32
|
+
Requires-Dist: transformers>=4.51; (python_version >= '3.9') and extra == 'transformers'
|
|
33
|
+
Description-Content-Type: text/markdown
|
|
34
|
+
|
|
35
|
+

|
|
36
|
+
|
|
37
|
+
# LLM2Jev — Less generation. Faster decisions.
|
|
38
|
+
|
|
39
|
+
**Classify. Filter. Score. Route.** Jev-style decisions from local language models.
|
|
40
|
+
|
|
41
|
+
An email pipeline often needs a category, a document pipeline needs a relevance decision, and an agent router needs a destination. Generating an explanation and a JSON object can add work that these applications never use. LLM2Jev tests a narrower execution path: define the possible answers, read binary model scores, and construct the result in Python.
|
|
42
|
+
|
|
43
|
+
The goal is to reduce decoding and parsing overhead in repeated semantic decisions. **Local measurements show a modest speedup in one configured workload; quality and application-level savings remain workload-dependent.** Candidate scoring repeats input processing, so the approach can also be slower than generating a short label. See the measured results below, including incorrect decisions.
|
|
44
|
+
|
|
45
|
+
## The idea from Jev
|
|
46
|
+
|
|
47
|
+
LLM2Jev’s public API design is inspired by [TypeSafe’s Jev](https://typesafe.ai/blog/introducing-system-one-models-and-jev). The `state` plus typed `questions` interface and the `Choice`, `Score`, and `Noul` vocabulary follow [TypeSafe’s published API primitives](https://docs.typesafe.ai/introduction). Credit for these interface concepts belongs to TypeSafe; they are not new primitives introduced by LLM2Jev.
|
|
48
|
+
|
|
49
|
+
[TypeSafe's Jev](https://typesafe.ai/blog/introducing-system-one-models-and-jev) makes typed decisions over supplied state. It motivates a useful application question: how much work can software avoid when it needs a bounded judgment rather than generated text?
|
|
50
|
+
|
|
51
|
+
LLM2Jev explores that question using existing language models. It implements Jev-style `Choice`, `Score`, and `Noul` responses through next-token binary scoring. It does not reproduce Jev's architecture, training, calibrated probabilities, or parallel sampler, and external API conformance has not been established. This is an independent project.
|
|
52
|
+
|
|
53
|
+
TypeSafe publishes large speed and cost improvements for its own workloads and service. Those measurements do not transfer to this adapter. Here, Transformers performs a forward pass with **zero generated tokens**; Ollama requests **one generated token per candidate** to obtain logprobs and reports actual usage. Neither path parses generated text into an answer.
|
|
54
|
+
|
|
55
|
+
## Concrete workloads
|
|
56
|
+
|
|
57
|
+
The strongest starting hypothesis is short text, a small fixed answer space, repeated decisions, and an application that consumes the result directly.
|
|
58
|
+
|
|
59
|
+
| Application | Unstructured input and decision | How the application uses it | Work to account for |
|
|
60
|
+
| --- | --- | --- | --- |
|
|
61
|
+
| Email or support triage | Message → `Choice` among billing, delivery, technical, and other | Assign a queue; count categories in ordinary code | 500 messages × 4 categories = 2,000 candidate evaluations; no email ingestion or aggregation is bundled |
|
|
62
|
+
| Semantic filtering | Passage → `Noul`: does it describe a customer requesting a refund? | Keep matching passages before an expensive downstream analysis | One candidate per passage with instructions-only Noul; choose a threshold against labeled data |
|
|
63
|
+
| Model routing | Request → `Choice` among application-defined nano, balanced, and frontier tiers | Application dispatches to the selected model | Three candidates plus the downstream call; routing errors and retries can erase savings |
|
|
64
|
+
| Rubric scoring | Ticket → `Score` over explicit urgency levels | Prioritize a review queue | One candidate per level; output is an expected level, not a verified fact |
|
|
65
|
+
| Browser action selection | Textual page state plus a short list of known actions → `Choice` | Browser controller validates and executes an action | Each action is a candidate; DOM extraction, planning, execution, and success checks remain outside this library |
|
|
66
|
+
|
|
67
|
+
These are integration patterns; the refund predicate has a small measured smoke experiment below, while the other workloads remain unmeasured. There is no bundled reproduction of a 500-email speed benchmark or a 7.1-second flight-search agent. For routing, describe operational task requirements in the criteria; model names alone do not teach the scorer which downstream model will succeed.
|
|
68
|
+
|
|
69
|
+
In a data pipeline, `Noul` can supply a **semantic filter**, `Choice` a **bounded semantic map**, and `Score` a **rubric-based ranking signal**. Applications retain record IDs, apply thresholds, sort, group, and count. LLM2Jev currently provides the per-record decision primitive; it has no dataframe/SQL integration, relational optimizer, or dataset-level operator API. Comparing document pairs is possible by supplying both in `state`, but a naive semantic join still needs one evaluation per pair.
|
|
70
|
+
|
|
71
|
+
## Install from PyPI
|
|
72
|
+
|
|
73
|
+
The distribution is now named **`llms2jev`**. Python imports remain `llm2jev`, and
|
|
74
|
+
the commands remain `llm2jev`, `llm2jev-serve`, and `llm2jev-mcp`. When migrating an
|
|
75
|
+
existing environment, uninstall `llm2jev` before installing `llms2jev`: both
|
|
76
|
+
distributions provide the same import package and should not be installed together.
|
|
77
|
+
|
|
78
|
+
```bash
|
|
79
|
+
python -m pip uninstall llm2jev
|
|
80
|
+
python -m pip install 'llms2jev[server,mcp]==0.3.3'
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
LLM2Jev is available on [PyPI](https://pypi.org/project/llms2jev/). Install the core Python package directly:
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
python -m pip install llms2jev
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
To install the published version used by this README with the dependencies for your runtime:
|
|
90
|
+
|
|
91
|
+
| Use case | Installation command |
|
|
92
|
+
| --- | --- |
|
|
93
|
+
| Core Python API and custom runtimes | `python -m pip install 'llms2jev==0.3.3'` |
|
|
94
|
+
| Local Transformers inference | `python -m pip install 'llms2jev[transformers]==0.3.3'` |
|
|
95
|
+
| Connect to Ollama | `python -m pip install 'llms2jev[ollama]==0.3.3'` |
|
|
96
|
+
| Run the HTTP server | `python -m pip install 'llms2jev[server]==0.3.3'` |
|
|
97
|
+
|
|
98
|
+
Wheel and source archives are also available from the [PyPI release files](https://pypi.org/project/llms2jev/0.3.3/#files).
|
|
99
|
+
|
|
100
|
+
## Try a decision
|
|
101
|
+
|
|
102
|
+
The core package has no mandatory third-party runtime dependencies. Install the Transformers extra for local inference; model weights are supplied separately. See the [runtime compatibility details](https://github.com/Qingbolan/llms2jev-releases/blob/v0.3.3/docs/architecture.md#python-38-compatibility) when choosing an interpreter and model family:
|
|
103
|
+
|
|
104
|
+
```bash
|
|
105
|
+
pip install 'llms2jev[transformers]==0.3.3'
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
Supply downloaded model weights; the runtime loads local files only. On Apple Silicon, select `device="mps"` explicitly; the example below uses CPU for portability. The [measured refund example](https://github.com/Qingbolan/llms2jev-releases/blob/v0.3.3/benchmarks/README.md) includes the pinned model download, explicit encoder, and `Yes`/`No` label configuration used in the benchmark.
|
|
109
|
+
|
|
110
|
+
From a source checkout, start with the deterministic, model-free example:
|
|
111
|
+
|
|
112
|
+
```bash
|
|
113
|
+
uv sync --group dev
|
|
114
|
+
uv run python examples/bound_evaluation.py
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
That example checks the application path with synthetic scores. For source development, install `uv sync --extra transformers --group dev`. The following is a general API walkthrough with a compatible local model, not a validated model/encoder configuration:
|
|
118
|
+
|
|
119
|
+
```python
|
|
120
|
+
from llm2jev import Choice, LLM2Jev, Noul, TransformersRuntime
|
|
121
|
+
|
|
122
|
+
with TransformersRuntime("/path/to/local/model", device="cpu") as runtime:
|
|
123
|
+
triage = LLM2Jev(runtime=runtime, model_identity=runtime.identity).bind(
|
|
124
|
+
model=runtime.identity.name,
|
|
125
|
+
questions={
|
|
126
|
+
"department": Choice(
|
|
127
|
+
instructions="Which queue should handle this customer message?",
|
|
128
|
+
criteria={
|
|
129
|
+
"billing": "Payments, invoices, or refunds",
|
|
130
|
+
"delivery": "Shipping, tracking, or missing packages",
|
|
131
|
+
"other": "Requests outside billing and delivery",
|
|
132
|
+
},
|
|
133
|
+
),
|
|
134
|
+
"refund_requested": Noul(
|
|
135
|
+
instructions="Does the customer explicitly request a refund?",
|
|
136
|
+
),
|
|
137
|
+
},
|
|
138
|
+
)
|
|
139
|
+
response = triage.evaluate(
|
|
140
|
+
state="My package never arrived. Please refund my payment.",
|
|
141
|
+
)
|
|
142
|
+
print(response.choices["department"].choice)
|
|
143
|
+
print(response.nouls["refund_requested"].noul)
|
|
144
|
+
print(response.to_json(indent=2))
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
This executes four binary candidates. Actual values depend on the model and prompt; the message intentionally contains both delivery and billing evidence. Binding snapshots reusable rules once. Each evaluation supplies one record, and `iter_evaluate()` processes records lazily and sequentially. Passing a whole list of emails as one `state` asks questions about that list; it does not classify each email separately. Question IDs are result keys: put the predicate in `instructions`, rather than relying on a name such as `refund_requested`.
|
|
148
|
+
|
|
149
|
+
| Result | Meaning |
|
|
150
|
+
| --- | --- |
|
|
151
|
+
| `Choice` | Selected option and a normalized candidate distribution |
|
|
152
|
+
| `Score` | Expected rubric level, using equally spaced indices from 0, plus a distribution |
|
|
153
|
+
| `Noul` | Binary probability conditioned on the selected label pair; explicit true/false criteria use two candidates |
|
|
154
|
+
|
|
155
|
+
`Choice` and `Score` report `confidence` as distribution concentration, **not calibrated correctness**. For example, raw candidate support `[0.009, 0.001]` and `[0.9, 0.1]` both become `[0.9, 0.1]` with confidence `0.8`. The API does not automatically abstain. Select an application policy on held-out data; an `other` option can help represent scope but does not guarantee rejection. [Probability semantics](https://github.com/Qingbolan/llms2jev-releases/blob/v0.3.3/docs/concepts.md) · [Python workflow](https://github.com/Qingbolan/llms2jev-releases/blob/v0.3.3/docs/guides/python.md).
|
|
156
|
+
|
|
157
|
+
## Does it actually make local inference faster?
|
|
158
|
+
|
|
159
|
+
**In the measured configuration, yes—by a modest amount against one-token generation. It is not a general acceleration switch, and the tested small model is not accurate enough for unattended refund decisions.**
|
|
160
|
+
|
|
161
|
+
Historical measurement of wheel `llms2jev==0.1.0`; these timings are not a fresh benchmark of 0.2.0. Version 0.2.0 has separately passed installed-wheel inference with Qwen2.5 on Python 3.8. See the [component validation record](https://github.com/Qingbolan/llms2jev-releases/blob/v0.3.3/docs/decisions/component-contracts.md). Benchmark hardware: Apple M4, 24 GB RAM; Qwen2.5-0.5B-Instruct; MPS, float32; 12 hand-authored English records × 3 repetitions. The configured predicate uses `Yes`/`No` and the explicit example renderer. The one-token baseline uses the same prompt. Timings include tokenization, model execution, synchronization, and output handling; model loading is excluded.
|
|
162
|
+
|
|
163
|
+
| Method | Median / p95 per record | Correct unique records | Generated tokens / record |
|
|
164
|
+
| --- | ---: | ---: | ---: |
|
|
165
|
+
| LLM2Jev, native tail logits | **89.3 / 126.6 ms** | 8/12 | **0** |
|
|
166
|
+
| LLM2Jev, original full logits | 109.1 / 220.4 ms | 8/12 | 0 |
|
|
167
|
+
| Greedy one-token label | 103.4 / 134.9 ms | 8/12 | 1 |
|
|
168
|
+
| Generated JSON, fence-aware parser | 890.1 / 1,438.4 ms | 6/12 | 13 |
|
|
169
|
+
|
|
170
|
+
Tail projection reduced median scoring time by **18.1%** versus full logits (1.22×), and this scoring path took **13.7% less time** than one-token generation (1.16×). JSON was slower **and less accurate** here; that comparison does not establish equal-quality application savings. It is prompted JSON, not grammar-constrained decoding.
|
|
171
|
+
|
|
172
|
+

|
|
173
|
+
|
|
174
|
+
These are smoke measurements, not a held-out quality study. The configured binary path missed two of six refund requests and incorrectly accepted two of six negative records. Default rendering with lowercase labels scored every record positive with both tested small Qwen models. No calibration, downstream savings, energy reduction, or universal speedup is claimed. [Raw observations, configuration failures, and reproduction commands](https://github.com/Qingbolan/llms2jev-releases/tree/v0.3.3/benchmarks).
|
|
175
|
+
|
|
176
|
+
## Watch the actual run
|
|
177
|
+
|
|
178
|
+
[](https://github.com/Qingbolan/llms2jev-releases/raw/refs/tags/v0.2.0/docs/assets/local-inference.mp4)
|
|
179
|
+
|
|
180
|
+
[Play/download the 13-second MP4](https://github.com/Qingbolan/llms2jev-releases/raw/refs/tags/v0.2.0/docs/assets/local-inference.mp4). This historical 0.1.x recording shows a real subprocess at wall-clock speed, including model loading, real probabilities, measured per-call times, and expected versus predicted decisions. Its three demo timings are a separate run from the repeated benchmark. [Timestamped output](https://github.com/Qingbolan/llms2jev-releases/blob/v0.3.3/docs/assets/local-inference.json).
|
|
181
|
+
|
|
182
|
+
## Where the latency and cost can go
|
|
183
|
+
|
|
184
|
+
```text
|
|
185
|
+
rules ── bind once ──> candidate plan
|
|
186
|
+
+ one record
|
|
187
|
+
↓
|
|
188
|
+
C binary prompts
|
|
189
|
+
↓
|
|
190
|
+
next-token yes/no scores
|
|
191
|
+
↓
|
|
192
|
+
typed decision results
|
|
193
|
+
↓
|
|
194
|
+
application filter / route / count
|
|
195
|
+
```
|
|
196
|
+
|
|
197
|
+
For one record, `C` is the sum of Choice options, Score levels, and Noul candidates. Each prompt contains the state. Transformers batches candidate prompts, but recomputes their input representations; binding is not a KV or prefix cache. Ollama sends candidates sequentially, including through its async adapter.
|
|
198
|
+
|
|
199
|
+
A useful break-even comparison is **C candidate prefills and their serving overhead** versus **one prefill plus the baseline's output decoding**. Batching can reduce wall-clock time without eliminating input work. Compare against a minimal generated label or constrained structured output, not only a verbose reasoning response. Input length, candidate count, label availability, hardware, and downstream mistakes all matter. [Evaluation protocol](https://github.com/Qingbolan/llms2jev-releases/blob/v0.3.3/docs/evaluation.md).
|
|
200
|
+
|
|
201
|
+
## Failure workloads and limits
|
|
202
|
+
|
|
203
|
+
| Workload | Why it can fail or lose its advantage | Application or evaluation response |
|
|
204
|
+
| --- | --- | --- |
|
|
205
|
+
| Long documents × many categories | Repeated prefills dominate; models without native logit selection still materialize sequence-wide vocabulary logits | Measure input-length/candidate sweeps and peak memory; check runtime context limits |
|
|
206
|
+
| Short inputs with a one-token baseline | There is little decoding to remove, while multiple binary prompts add work | Include this baseline; a speedup is not assumed |
|
|
207
|
+
| Large semantic joins or hundreds of browser actions | Candidate/pair counts grow quickly; no retrieval or query planner reduces them | Retrieve a shortlist and measure shortlist recall as well as final quality |
|
|
208
|
+
| Ambiguous, overlapping, or out-of-scope categories | Independent scores are renormalized into a forced choice, even if all candidates have weak support | Define useful criteria, test out-of-scope records, and evaluate rejection policies |
|
|
209
|
+
| Decisions that depend on each other | Questions and candidates are evaluated independently; constraints are not jointly solved | Enforce workflow dependencies and action preconditions in application code |
|
|
210
|
+
| Arithmetic, multi-step planning, open-ended extraction, or explanation | Removing decoding does not preserve every reasoning capability; outputs are restricted to supplied alternatives | Evaluate a reasoning/generative baseline or use deterministic code where appropriate |
|
|
211
|
+
| Ollama missing either exact label, or overlong context | Missing labels cause an explicit error; upstream context truncation is not currently detected by this adapter | Measure capability failures and enforce a deployment-specific input budget |
|
|
212
|
+
| Adversarial text or domain shift | Typed output ensures shape, not correct interpretation or immunity to prompt injection | Include these records in the held-out workload; validate actions separately |
|
|
213
|
+
|
|
214
|
+
The adapters accept text/JSON state; they do not provide OCR, audio understanding, browser control, or evidence retrieval. Deployment and correctness findings are recorded in the [architecture review](https://github.com/Qingbolan/llms2jev-releases/blob/v0.3.3/docs/decisions/review-resolution.md).
|
|
215
|
+
|
|
216
|
+
## Existing techniques and this project's contribution
|
|
217
|
+
|
|
218
|
+
Binary next-token scoring is established: [Qwen3-Reranker](https://huggingface.co/Qwen/Qwen3-Reranker-0.6B) uses yes/no logits for relevance scoring. Semantic data operators are also established in [LOTUS](https://github.com/lotus-data/lotus). TypeSafe publishes a [System One Adapter](https://github.com/typesafe-ai/system-one-adapter-python) for obtaining its decision interface from language models. LLM2Jev does not claim to invent these ideas.
|
|
219
|
+
|
|
220
|
+
The current contribution is their integration into a small decision runtime: immutable rule compilation, per-execution state snapshots, typed probability assembly, shared Python/HTTP execution, and explicit runtime ownership and capability failures. Unlike the generated-answer adapter, this runtime reads binary token scores and assembles answers without parsing generated probabilities. Its practical value must be established by showing useful decisions at lower total cost or latency at a fixed error budget. No new model or training method is claimed. The measured execution advantage below is limited to its stated configuration and does not establish general application savings. See the [workload evaluation protocol](https://github.com/Qingbolan/llms2jev-releases/blob/v0.3.3/docs/evaluation.md).
|
|
221
|
+
|
|
222
|
+
## Execution and development
|
|
223
|
+
|
|
224
|
+
| Adapter | Execution contract | Guide |
|
|
225
|
+
| --- | --- | --- |
|
|
226
|
+
| `TransformersRuntime` | Local next-token logits; prefill-only; candidate batches | [Python](https://github.com/Qingbolan/llms2jev-releases/blob/v0.3.3/docs/guides/python.md) |
|
|
227
|
+
| `OllamaRuntime` | One-token requests; both exact label logprobs required | [Ollama](https://github.com/Qingbolan/llms2jev-releases/blob/v0.3.3/docs/guides/ollama.md) |
|
|
228
|
+
| `AsyncOllamaRuntime` | Same sequential scoring policy with async I/O | [Ollama](https://github.com/Qingbolan/llms2jev-releases/blob/v0.3.3/docs/guides/ollama.md) |
|
|
229
|
+
| `llm2jev-serve` | Ollama-backed `POST /v1/systemone`, model listing, liveness, optional bearer authentication | [HTTP](https://github.com/Qingbolan/llms2jev-releases/blob/v0.3.3/docs/guides/http.md) |
|
|
230
|
+
|
|
231
|
+
The root `llm2jev` exports are the public Python API. `application/` orchestrates evaluation through scoring contracts; `runtime/` owns model execution and resources; `inference/` owns the scoring-independent algorithms. `transport/` parses and routes HTTP, while `serving/` composes these parts and owns process lifespan. Runtimes do not import application or inference implementations. [Architecture](https://github.com/Qingbolan/llms2jev-releases/blob/v0.3.3/docs/architecture.md) · [API](https://github.com/Qingbolan/llms2jev-releases/blob/v0.3.3/docs/api.md) · [Privacy](https://github.com/Qingbolan/llms2jev-releases/blob/v0.3.3/docs/privacy.md) · [Documentation index](https://github.com/Qingbolan/llms2jev-releases/blob/v0.3.3/docs/index.md).
|
|
232
|
+
|
|
233
|
+
```text
|
|
234
|
+
src/llm2jev/
|
|
235
|
+
├── core/ Domain values and wire serialization
|
|
236
|
+
├── contracts.py Runtime protocols, labels, identities, BinaryScores
|
|
237
|
+
├── application/ Services, bound evaluators, per-call preparation
|
|
238
|
+
├── inference/ Compilation, rendering, normalization, assembly
|
|
239
|
+
├── runtime/
|
|
240
|
+
│ ├── ollama/ HTTP execution, wire policy, configuration
|
|
241
|
+
│ ├── transformers/ Tensor execution and provider tokenization
|
|
242
|
+
│ └── lifecycle.py Shared resource state machines
|
|
243
|
+
├── transport/ Framework-free parsing and HTTP routes
|
|
244
|
+
├── serving/ App composition, authentication, lifespan, CLI
|
|
245
|
+
└── utils/ Internal JSON and probability primitives
|
|
246
|
+
```
|
|
247
|
+
|
|
248
|
+
Customize prompts with `ProbeEncoder.encode(CandidateProbe)` and `encoder=`, and probability policies with `distribution=`. See the [API and packaging decision](docs/decisions/evidence-api.md).
|
|
249
|
+
|
|
250
|
+
Use `LLM2Jev(runtime=...)` to inject model execution. Custom runtimes return `BinaryScores`; question/answer JSON contracts are defined in the API reference.
|
|
251
|
+
|
|
252
|
+
```bash
|
|
253
|
+
uv run python -m unittest discover -s tests -v
|
|
254
|
+
uv run python -m compileall -q src tests
|
|
255
|
+
uv run mypy src/llm2jev
|
|
256
|
+
uv build
|
|
257
|
+
```
|
|
258
|
+
|
|
259
|
+
Tests cover deterministic scoring, state isolation, lifecycle, HTTP, and diagnostic privacy. CPU tensor tests need the optional PyTorch dependency and otherwise skip. These checks establish software contracts; the small local-model experiment establishes bounded execution evidence; held-out accuracy, calibration, scaling, and downstream cost remain open evaluation work in the [implementation plan](https://github.com/Qingbolan/llms2jev-releases/blob/v0.3.3/docs/implementation-plan.md).
|
|
260
|
+
|
|
261
|
+
## CLI and MCP
|
|
262
|
+
|
|
263
|
+
Version 0.3.3 provides `llm2jev health`, `llm2jev models`, `llm2jev evaluate`, and
|
|
264
|
+
`llm2jev-mcp` (stdio). Install them with `pip install 'llms2jev[server,mcp]==0.3.3'`. They share
|
|
265
|
+
one HTTP client and call the existing Ollama-backed `llm2jev-serve`; MCP needs
|
|
266
|
+
Python 3.10+ and the optional SDK 2.x dependency. See the
|
|
267
|
+
[complete CLI/MCP tutorial](https://github.com/Qingbolan/llms2jev-releases/blob/v0.3.3/docs/guides/cli-mcp.md)
|
|
268
|
+
for source installation, JSON examples, client configuration, and real-run evidence.
|
|
269
|
+
|
|
270
|
+
## Contributions — PRs welcome
|
|
271
|
+
|
|
272
|
+
PRs are welcome: bug fixes, clearer documentation, reproducible evaluations, and runtime improvements tied to a concrete use case. For architectural or public API changes, open an issue first to discuss the behavior and tradeoffs.
|
|
273
|
+
|
|
274
|
+
Describe the problem, the resulting behavior, and the checks you ran. Keep changes focused, preserve documented contracts, and include regression tests for behavior changes. Performance claims need reproducible measurements with hardware, dependency versions, baselines, and correctness results. Retain source attribution when adapting ideas or code.
|
|
275
|
+
|
|
276
|
+
See the [contribution guide](https://github.com/Qingbolan/llms2jev-releases/blob/v0.3.3/CONTRIBUTING.md) for setup, architecture rules, and the PR checklist. Submit PRs to [llms2jev-releases](https://github.com/Qingbolan/llms2jev-releases/pulls).
|
|
277
|
+
|
|
278
|
+
Licensed under [MIT](https://github.com/Qingbolan/llms2jev-releases/blob/v0.3.3/LICENSE).
|