llms2jev 0.3.3__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm2jev/__init__.py +80 -0
- llm2jev/application/__init__.py +8 -0
- llm2jev/application/bound.py +103 -0
- llm2jev/application/pipeline.py +119 -0
- llm2jev/application/service.py +120 -0
- llm2jev/contracts.py +135 -0
- llm2jev/core/__init__.py +26 -0
- llm2jev/core/answers.py +120 -0
- llm2jev/core/questions.py +114 -0
- llm2jev/core/request.py +41 -0
- llm2jev/core/response.py +122 -0
- llm2jev/core/types.py +18 -0
- llm2jev/core/validation.py +55 -0
- llm2jev/inference/__init__.py +13 -0
- llm2jev/inference/compilation/__init__.py +3 -0
- llm2jev/inference/compilation/compiler.py +70 -0
- llm2jev/inference/compilation/plan.py +80 -0
- llm2jev/inference/distribution.py +31 -0
- llm2jev/inference/encoding.py +99 -0
- llm2jev/inference/probes.py +94 -0
- llm2jev/inference/projection.py +134 -0
- llm2jev/py.typed +0 -0
- llm2jev/runtime/__init__.py +17 -0
- llm2jev/runtime/lifecycle.py +147 -0
- llm2jev/runtime/ollama/__init__.py +8 -0
- llm2jev/runtime/ollama/configuration.py +28 -0
- llm2jev/runtime/ollama/protocol.py +154 -0
- llm2jev/runtime/ollama/runtime.py +112 -0
- llm2jev/runtime/transformers/__init__.py +7 -0
- llm2jev/runtime/transformers/runtime.py +181 -0
- llm2jev/runtime/transformers/tokenization.py +28 -0
- llm2jev/serving/__init__.py +3 -0
- llm2jev/serving/app.py +85 -0
- llm2jev/serving/cli.py +43 -0
- llm2jev/serving/client_cli.py +62 -0
- llm2jev/serving/mcp.py +95 -0
- llm2jev/transport/__init__.py +3 -0
- llm2jev/transport/client.py +102 -0
- llm2jev/transport/http.py +81 -0
- llm2jev/transport/parsing.py +47 -0
- llm2jev/utils/__init__.py +4 -0
- llm2jev/utils/json.py +85 -0
- llm2jev/utils/probability.py +48 -0
- llms2jev-0.3.3.dist-info/METADATA +278 -0
- llms2jev-0.3.3.dist-info/RECORD +48 -0
- llms2jev-0.3.3.dist-info/WHEEL +4 -0
- llms2jev-0.3.3.dist-info/entry_points.txt +4 -0
- llms2jev-0.3.3.dist-info/licenses/LICENSE +21 -0
llm2jev/__init__.py
ADDED
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
"""Public composition facade for decision definitions, services, and runtime ports. Exports
|
|
2
|
+
form the supported import contract; concrete provider dependencies remain lazily loaded
|
|
3
|
+
by their constructors. Internal compilation, JSON, and projection helpers are
|
|
4
|
+
intentionally not re-exported.
|
|
5
|
+
"""
|
|
6
|
+
from .application import AsyncBoundEvaluator, AsyncLLM2Jev, BoundEvaluator, LLM2Jev
|
|
7
|
+
from .contracts import (
|
|
8
|
+
AsyncBinaryRuntime,
|
|
9
|
+
BinaryLabels,
|
|
10
|
+
BinaryRuntime,
|
|
11
|
+
BinaryScores,
|
|
12
|
+
ChatMessage,
|
|
13
|
+
ChatPrompt,
|
|
14
|
+
InvalidRequestError,
|
|
15
|
+
ModelIdentity,
|
|
16
|
+
ModelNotFoundError,
|
|
17
|
+
)
|
|
18
|
+
from .core import (
|
|
19
|
+
Answer,
|
|
20
|
+
Choice,
|
|
21
|
+
ChoiceAnswer,
|
|
22
|
+
JSONContent,
|
|
23
|
+
JSONValue,
|
|
24
|
+
JevRequest,
|
|
25
|
+
JevResponse,
|
|
26
|
+
Noul,
|
|
27
|
+
NoulAnswer,
|
|
28
|
+
Question,
|
|
29
|
+
Score,
|
|
30
|
+
ScoreAnswer,
|
|
31
|
+
State,
|
|
32
|
+
Usage,
|
|
33
|
+
)
|
|
34
|
+
from .inference import (
|
|
35
|
+
CandidateProbe,
|
|
36
|
+
StructuredProbeEncoder,
|
|
37
|
+
DistributionPolicy,
|
|
38
|
+
ProbeEncoder,
|
|
39
|
+
RelativeSupport,
|
|
40
|
+
)
|
|
41
|
+
from .runtime import AsyncOllamaRuntime, OllamaRuntime, OllamaScoringError, TransformersRuntime
|
|
42
|
+
|
|
43
|
+
__all__ = [
|
|
44
|
+
"Answer",
|
|
45
|
+
"AsyncBinaryRuntime",
|
|
46
|
+
"AsyncBoundEvaluator",
|
|
47
|
+
"AsyncLLM2Jev",
|
|
48
|
+
"AsyncOllamaRuntime",
|
|
49
|
+
"BinaryLabels",
|
|
50
|
+
"BinaryRuntime",
|
|
51
|
+
"BinaryScores",
|
|
52
|
+
"BoundEvaluator",
|
|
53
|
+
"CandidateProbe",
|
|
54
|
+
"ChatMessage",
|
|
55
|
+
"ChatPrompt",
|
|
56
|
+
"Choice",
|
|
57
|
+
"ChoiceAnswer",
|
|
58
|
+
"DistributionPolicy",
|
|
59
|
+
"InvalidRequestError",
|
|
60
|
+
"JSONContent",
|
|
61
|
+
"JSONValue",
|
|
62
|
+
"JevRequest",
|
|
63
|
+
"JevResponse",
|
|
64
|
+
"LLM2Jev",
|
|
65
|
+
"ModelIdentity",
|
|
66
|
+
"ModelNotFoundError",
|
|
67
|
+
"Noul",
|
|
68
|
+
"NoulAnswer",
|
|
69
|
+
"OllamaRuntime",
|
|
70
|
+
"OllamaScoringError",
|
|
71
|
+
"ProbeEncoder",
|
|
72
|
+
"Question",
|
|
73
|
+
"RelativeSupport",
|
|
74
|
+
"Score",
|
|
75
|
+
"ScoreAnswer",
|
|
76
|
+
"State",
|
|
77
|
+
"StructuredProbeEncoder",
|
|
78
|
+
"TransformersRuntime",
|
|
79
|
+
"Usage",
|
|
80
|
+
]
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""Application entry points expose services and bound evaluators. This facade exports
|
|
2
|
+
execution handles without allocating, closing, or owning concrete model resources.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from .bound import AsyncBoundEvaluator, BoundEvaluator
|
|
6
|
+
from .service import AsyncLLM2Jev, LLM2Jev
|
|
7
|
+
|
|
8
|
+
__all__ = ["AsyncBoundEvaluator", "AsyncLLM2Jev", "BoundEvaluator", "LLM2Jev"]
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
"""Execution handles for a captured rule plan and a borrowed scoring runtime. Each call
|
|
2
|
+
owns preparation and response projection, and iterators consume records lazily in source
|
|
3
|
+
order. Cancellation and failures propagate without closing a runtime that another
|
|
4
|
+
evaluator may still need.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from typing import AsyncIterable, AsyncIterator, Iterable, Iterator, Union
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
|
|
12
|
+
from ..contracts import AsyncBinaryRuntime, BinaryRuntime
|
|
13
|
+
from ..core.response import JevResponse
|
|
14
|
+
from ..core.types import State, JSONValue
|
|
15
|
+
from ..inference.compilation.plan import DecisionPlan
|
|
16
|
+
from .pipeline import EvaluationPipeline
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass(frozen=True, init=False)
|
|
20
|
+
class BoundEvaluator:
|
|
21
|
+
"""A validated rule snapshot borrowing a synchronous runtime."""
|
|
22
|
+
|
|
23
|
+
_runtime: BinaryRuntime = field(repr=False)
|
|
24
|
+
_pipeline: EvaluationPipeline = field(repr=False)
|
|
25
|
+
_plan: DecisionPlan = field(repr=False)
|
|
26
|
+
model: str
|
|
27
|
+
|
|
28
|
+
# Retain the validated plan and borrowed collaborators without capturing any per-record input.
|
|
29
|
+
def __init__(
|
|
30
|
+
self,
|
|
31
|
+
*,
|
|
32
|
+
_runtime: BinaryRuntime,
|
|
33
|
+
_pipeline: EvaluationPipeline,
|
|
34
|
+
_plan: DecisionPlan,
|
|
35
|
+
model: str,
|
|
36
|
+
) -> None:
|
|
37
|
+
object.__setattr__(self, "_runtime", _runtime)
|
|
38
|
+
object.__setattr__(self, "_pipeline", _pipeline)
|
|
39
|
+
object.__setattr__(self, "_plan", _plan)
|
|
40
|
+
object.__setattr__(self, "model", model)
|
|
41
|
+
|
|
42
|
+
# Prepare an isolated record, score its ordered candidates, then project against that same
|
|
43
|
+
# plan.
|
|
44
|
+
def evaluate(self, *, state: State) -> JevResponse:
|
|
45
|
+
prepared = self._pipeline.prepare(plan=self._plan, model=self.model, state=state)
|
|
46
|
+
output = self._runtime.score(model=prepared.model, prompts=prepared.prompts)
|
|
47
|
+
return self._pipeline.finish(prepared, output)
|
|
48
|
+
|
|
49
|
+
# Pull one record at a time and stop on failure; reject strings that would otherwise stream
|
|
50
|
+
# characters.
|
|
51
|
+
def iter_evaluate(self, states: Iterable[State]) -> Iterator[JevResponse]:
|
|
52
|
+
"""Consume states lazily in order, stopping on the first error."""
|
|
53
|
+
if isinstance(states, (str, bytes, bytearray)):
|
|
54
|
+
raise ValueError("states must be an iterable of state values, not a string")
|
|
55
|
+
for state in states:
|
|
56
|
+
yield self.evaluate(state=state)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@dataclass(frozen=True, init=False)
|
|
60
|
+
class AsyncBoundEvaluator:
|
|
61
|
+
"""A validated rule snapshot borrowing a native asynchronous runtime."""
|
|
62
|
+
|
|
63
|
+
_runtime: AsyncBinaryRuntime = field(repr=False)
|
|
64
|
+
_pipeline: EvaluationPipeline = field(repr=False)
|
|
65
|
+
_plan: DecisionPlan = field(repr=False)
|
|
66
|
+
model: str
|
|
67
|
+
|
|
68
|
+
# Store binding-time policy separately from the state snapshots created by later awaited
|
|
69
|
+
# calls.
|
|
70
|
+
def __init__(
|
|
71
|
+
self,
|
|
72
|
+
*,
|
|
73
|
+
_runtime: AsyncBinaryRuntime,
|
|
74
|
+
_pipeline: EvaluationPipeline,
|
|
75
|
+
_plan: DecisionPlan,
|
|
76
|
+
model: str,
|
|
77
|
+
) -> None:
|
|
78
|
+
object.__setattr__(self, "_runtime", _runtime)
|
|
79
|
+
object.__setattr__(self, "_pipeline", _pipeline)
|
|
80
|
+
object.__setattr__(self, "_plan", _plan)
|
|
81
|
+
object.__setattr__(self, "model", model)
|
|
82
|
+
|
|
83
|
+
# Keep preparation immutable across the await boundary so caller edits cannot change response
|
|
84
|
+
# assembly.
|
|
85
|
+
async def evaluate(self, *, state: State) -> JevResponse:
|
|
86
|
+
prepared = self._pipeline.prepare(plan=self._plan, model=self.model, state=state)
|
|
87
|
+
output = await self._runtime.score(model=prepared.model, prompts=prepared.prompts)
|
|
88
|
+
return self._pipeline.finish(prepared, output)
|
|
89
|
+
|
|
90
|
+
# Consume either source protocol sequentially; no hidden buffering, retries, or parallel model
|
|
91
|
+
# calls occur.
|
|
92
|
+
async def iter_evaluate(
|
|
93
|
+
self, states: Union[Iterable[State], AsyncIterable[State]],
|
|
94
|
+
) -> AsyncIterator[JevResponse]:
|
|
95
|
+
"""Consume a sync or async source sequentially without buffering it."""
|
|
96
|
+
if isinstance(states, (str, bytes, bytearray)):
|
|
97
|
+
raise ValueError("states must be an iterable of state values, not a string")
|
|
98
|
+
if isinstance(states, AsyncIterable):
|
|
99
|
+
async for state in states:
|
|
100
|
+
yield await self.evaluate(state=state)
|
|
101
|
+
else:
|
|
102
|
+
for state in states:
|
|
103
|
+
yield await self.evaluate(state=state)
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
"""Shared preparation and completion boundary for both execution modes. An evaluation
|
|
2
|
+
freezes one state tree and attaches it to the precompiled candidates before encoding.
|
|
3
|
+
Runtime calls occur outside this module; completion interprets evidence only against the
|
|
4
|
+
captured plan, never mutable source rules.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from typing import Optional, Tuple
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
|
|
12
|
+
from ..contracts import BinaryLabels, BinaryScores, ChatPrompt, InvalidRequestError, ModelIdentity
|
|
13
|
+
from ..core.response import JevResponse
|
|
14
|
+
from ..core.types import State, JSONValue
|
|
15
|
+
from ..core.validation import document_snapshot, model_name
|
|
16
|
+
from ..inference.projection import DecisionProjector
|
|
17
|
+
from ..inference.probes import CandidateProbe
|
|
18
|
+
from ..inference.compilation.plan import DecisionPlan
|
|
19
|
+
from ..inference.distribution import DistributionPolicy
|
|
20
|
+
from ..inference.encoding import StructuredProbeEncoder, ProbeEncoder
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass(frozen=True, init=False)
|
|
24
|
+
class PreparedEvaluation:
|
|
25
|
+
plan: DecisionPlan = field(repr=False)
|
|
26
|
+
model: str
|
|
27
|
+
state: State = field(repr=False)
|
|
28
|
+
prompts: Tuple[ChatPrompt, ...] = field(repr=False)
|
|
29
|
+
|
|
30
|
+
# Keep plan, frozen state, and rendered prompts together for one evaluation's alignment and
|
|
31
|
+
# lifetime.
|
|
32
|
+
def __init__(
|
|
33
|
+
self,
|
|
34
|
+
*,
|
|
35
|
+
plan: DecisionPlan,
|
|
36
|
+
model: str,
|
|
37
|
+
state: State,
|
|
38
|
+
prompts: Tuple[ChatPrompt, ...],
|
|
39
|
+
) -> None:
|
|
40
|
+
object.__setattr__(self, "plan", plan)
|
|
41
|
+
object.__setattr__(self, "model", model)
|
|
42
|
+
object.__setattr__(self, "state", state)
|
|
43
|
+
object.__setattr__(self, "prompts", prompts)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass(frozen=True, init=False)
|
|
47
|
+
class EvaluationPipeline:
|
|
48
|
+
"""The shared, model-independent preparation and assembly boundary."""
|
|
49
|
+
|
|
50
|
+
encoder: ProbeEncoder = field(repr=False)
|
|
51
|
+
projector: DecisionProjector = field(repr=False)
|
|
52
|
+
model_identity: Optional[ModelIdentity] = None
|
|
53
|
+
|
|
54
|
+
# Capture encoding, projection, and optional identity policy without importing a concrete
|
|
55
|
+
# provider.
|
|
56
|
+
def __init__(
|
|
57
|
+
self,
|
|
58
|
+
*,
|
|
59
|
+
encoder: ProbeEncoder,
|
|
60
|
+
projector: DecisionProjector,
|
|
61
|
+
model_identity: Optional[ModelIdentity] = None,
|
|
62
|
+
) -> None:
|
|
63
|
+
object.__setattr__(self, "encoder", encoder)
|
|
64
|
+
object.__setattr__(self, "projector", projector)
|
|
65
|
+
object.__setattr__(self, "model_identity", model_identity)
|
|
66
|
+
|
|
67
|
+
# Reject invalid or unserved names before candidate materialization or runtime access.
|
|
68
|
+
def validate_model(self, model: str) -> None:
|
|
69
|
+
try:
|
|
70
|
+
model_name(model)
|
|
71
|
+
except ValueError as error:
|
|
72
|
+
raise InvalidRequestError(str(error)) from error
|
|
73
|
+
if self.model_identity is not None:
|
|
74
|
+
self.model_identity.validate(model)
|
|
75
|
+
|
|
76
|
+
# Freeze one state tree, share it across candidate views, and materialize prompts in plan
|
|
77
|
+
# order.
|
|
78
|
+
def prepare(self, *, plan: DecisionPlan, model: str, state: State) -> PreparedEvaluation:
|
|
79
|
+
self.validate_model(model)
|
|
80
|
+
try:
|
|
81
|
+
snapshot = document_snapshot(state, "state", immutable=True)
|
|
82
|
+
except ValueError as error:
|
|
83
|
+
raise InvalidRequestError(str(error)) from error
|
|
84
|
+
prompts = tuple(self.encoder.encode(CandidateProbe._from_snapshot(
|
|
85
|
+
candidate=candidate, state=snapshot,
|
|
86
|
+
)) for candidate in plan.candidates)
|
|
87
|
+
return PreparedEvaluation(plan=plan, model=model, state=snapshot, prompts=prompts)
|
|
88
|
+
|
|
89
|
+
# Interpret runtime evidence using the prepared plan, never by rereading caller-owned question
|
|
90
|
+
# mappings.
|
|
91
|
+
def finish(self, prepared: PreparedEvaluation, output: BinaryScores) -> JevResponse:
|
|
92
|
+
return self.projector.project(
|
|
93
|
+
plan=prepared.plan, model=prepared.model, scores=output,
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
# Resolve default labels once and check extension compatibility before creating shared execution
|
|
98
|
+
# policy.
|
|
99
|
+
def create_pipeline(
|
|
100
|
+
*, runtime: object, encoder: Optional[ProbeEncoder],
|
|
101
|
+
distribution: DistributionPolicy, model_identity: Optional[ModelIdentity],
|
|
102
|
+
) -> EvaluationPipeline:
|
|
103
|
+
# Label metadata is optional for third-party implementations of the score protocol.
|
|
104
|
+
labels = getattr(runtime, "labels", None)
|
|
105
|
+
if labels is not None and not isinstance(labels, BinaryLabels):
|
|
106
|
+
raise TypeError("runtime labels must be BinaryLabels")
|
|
107
|
+
selected = encoder if encoder is not None else StructuredProbeEncoder(
|
|
108
|
+
labels=labels if labels is not None else BinaryLabels(),
|
|
109
|
+
)
|
|
110
|
+
if not callable(getattr(selected, "encode", None)):
|
|
111
|
+
raise TypeError("encoder must implement encode")
|
|
112
|
+
if isinstance(selected, StructuredProbeEncoder) and labels is not None and selected.labels != labels:
|
|
113
|
+
raise ValueError("encoder and runtime must use the same binary labels")
|
|
114
|
+
if model_identity is not None and not isinstance(model_identity, ModelIdentity):
|
|
115
|
+
raise TypeError("model_identity must be ModelIdentity")
|
|
116
|
+
return EvaluationPipeline(
|
|
117
|
+
encoder=selected, projector=DecisionProjector(distribution=distribution),
|
|
118
|
+
model_identity=model_identity,
|
|
119
|
+
)
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
"""Configure decision evaluation and bind reusable definitions to scoring ports. Sync and
|
|
2
|
+
async entry points share validation and plan compilation, while their execution modes
|
|
3
|
+
remain explicit. Services borrow runtimes; changing service configuration affects future
|
|
4
|
+
bindings rather than existing evaluator snapshots.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from typing import Mapping, Optional, Tuple, Union
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
from ..contracts import AsyncBinaryRuntime, BinaryRuntime, ModelIdentity
|
|
12
|
+
from ..core.request import JevRequest, Question
|
|
13
|
+
from ..core.response import JevResponse
|
|
14
|
+
from ..inference.distribution import DistributionPolicy, RelativeSupport
|
|
15
|
+
from ..inference.compilation.compiler import RuleCompiler
|
|
16
|
+
from ..inference.compilation.plan import DecisionPlan
|
|
17
|
+
from ..inference.encoding import ProbeEncoder
|
|
18
|
+
from .bound import AsyncBoundEvaluator, BoundEvaluator
|
|
19
|
+
from .pipeline import EvaluationPipeline, create_pipeline
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class _EvaluationService:
|
|
23
|
+
"""One configuration and binding policy for both execution modes."""
|
|
24
|
+
|
|
25
|
+
# Capture extension configuration and validate it before accepting bindings; runtime ownership
|
|
26
|
+
# remains external.
|
|
27
|
+
def __init__(
|
|
28
|
+
self,
|
|
29
|
+
*,
|
|
30
|
+
runtime: Union[BinaryRuntime, AsyncBinaryRuntime],
|
|
31
|
+
encoder: Optional[ProbeEncoder] = None,
|
|
32
|
+
distribution: DistributionPolicy = RelativeSupport(),
|
|
33
|
+
model_identity: Optional[ModelIdentity] = None,
|
|
34
|
+
) -> None:
|
|
35
|
+
self.runtime = runtime
|
|
36
|
+
self.encoder = encoder
|
|
37
|
+
self.distribution = distribution
|
|
38
|
+
self.model_identity = model_identity
|
|
39
|
+
self._compiler = RuleCompiler()
|
|
40
|
+
self.encoder = self._pipeline().encoder
|
|
41
|
+
|
|
42
|
+
# Build a validated policy snapshot so future service edits do not change already-bound
|
|
43
|
+
# evaluators.
|
|
44
|
+
def _pipeline(self) -> EvaluationPipeline:
|
|
45
|
+
return create_pipeline(
|
|
46
|
+
runtime=self.runtime, encoder=self.encoder,
|
|
47
|
+
distribution=self.distribution, model_identity=self.model_identity,
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
# Validate the model before compiling rules, preventing unsupported identities from reaching
|
|
51
|
+
# any scorer.
|
|
52
|
+
def _binding(
|
|
53
|
+
self, model: str, questions: Mapping[str, Question],
|
|
54
|
+
) -> Tuple[EvaluationPipeline, DecisionPlan]:
|
|
55
|
+
pipeline = self._pipeline()
|
|
56
|
+
pipeline.validate_model(model)
|
|
57
|
+
return pipeline, self._compiler.compile(questions)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class LLM2Jev(_EvaluationService):
|
|
61
|
+
"""Evaluate dynamic requests or bind reusable rules to a scoring runtime."""
|
|
62
|
+
|
|
63
|
+
runtime: BinaryRuntime
|
|
64
|
+
|
|
65
|
+
# Keep a concrete keyword-only signature for Python 3.8 introspection while sharing
|
|
66
|
+
# configuration policy.
|
|
67
|
+
def __init__(
|
|
68
|
+
self, *, runtime: BinaryRuntime, encoder: Optional[ProbeEncoder] = None,
|
|
69
|
+
distribution: DistributionPolicy = RelativeSupport(),
|
|
70
|
+
model_identity: Optional[ModelIdentity] = None,
|
|
71
|
+
) -> None:
|
|
72
|
+
super().__init__(
|
|
73
|
+
runtime=runtime, encoder=encoder, distribution=distribution,
|
|
74
|
+
model_identity=model_identity,
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
# Capture one rule plan and policy set; returned evaluators borrow rather than close the
|
|
78
|
+
# runtime.
|
|
79
|
+
def bind(self, *, model: str, questions: Mapping[str, Question]) -> BoundEvaluator:
|
|
80
|
+
pipeline, plan = self._binding(model, questions)
|
|
81
|
+
return BoundEvaluator(
|
|
82
|
+
_runtime=self.runtime, _pipeline=pipeline,
|
|
83
|
+
_plan=plan, model=model,
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
# Route one-shot requests through binding so dynamic and reusable evaluation have identical
|
|
87
|
+
# semantics.
|
|
88
|
+
def evaluate(self, request: JevRequest) -> JevResponse:
|
|
89
|
+
return self.bind(model=request.model, questions=request.questions).evaluate(state=request.state)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
class AsyncLLM2Jev(_EvaluationService):
|
|
93
|
+
"""The native async entry point using the same compiler and pipeline."""
|
|
94
|
+
|
|
95
|
+
runtime: AsyncBinaryRuntime
|
|
96
|
+
|
|
97
|
+
# Expose the asynchronous port explicitly without creating an event loop or acquiring model
|
|
98
|
+
# resources.
|
|
99
|
+
def __init__(
|
|
100
|
+
self, *, runtime: AsyncBinaryRuntime, encoder: Optional[ProbeEncoder] = None,
|
|
101
|
+
distribution: DistributionPolicy = RelativeSupport(),
|
|
102
|
+
model_identity: Optional[ModelIdentity] = None,
|
|
103
|
+
) -> None:
|
|
104
|
+
super().__init__(
|
|
105
|
+
runtime=runtime, encoder=encoder, distribution=distribution,
|
|
106
|
+
model_identity=model_identity,
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
# Binding is synchronous rule work; only the returned evaluator's model execution needs
|
|
110
|
+
# awaiting.
|
|
111
|
+
def bind(self, *, model: str, questions: Mapping[str, Question]) -> AsyncBoundEvaluator:
|
|
112
|
+
pipeline, plan = self._binding(model, questions)
|
|
113
|
+
return AsyncBoundEvaluator(
|
|
114
|
+
_runtime=self.runtime, _pipeline=pipeline,
|
|
115
|
+
_plan=plan, model=model,
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
# Reuse the binding path and propagate cancellation through the asynchronous scoring call.
|
|
119
|
+
async def evaluate(self, request: JevRequest) -> JevResponse:
|
|
120
|
+
return await self.bind(model=request.model, questions=request.questions).evaluate(state=request.state)
|
llm2jev/contracts.py
ADDED
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""Model-independent ports connecting application execution to binary scorers. Ordered
|
|
2
|
+
scores, label pairs, identity metadata, and error categories cross this boundary; model
|
|
3
|
+
libraries and inference implementations do not. A runtime supplies evidence and measured
|
|
4
|
+
usage, never assembled application answers.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from typing import Literal, Protocol, Sequence, Tuple, TypedDict
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
|
|
12
|
+
from .core.response import Usage
|
|
13
|
+
from .core.validation import model_name
|
|
14
|
+
from .utils.probability import support_values
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class ChatMessage(TypedDict):
|
|
18
|
+
role: Literal['system', 'user']
|
|
19
|
+
content: str
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
ChatPrompt = Tuple[ChatMessage, ...]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass(frozen=True, init=False)
|
|
26
|
+
class BinaryLabels:
|
|
27
|
+
"""The two answer labels shared by a encoder and a scoring runtime."""
|
|
28
|
+
|
|
29
|
+
yes: str = "yes"
|
|
30
|
+
no: str = "no"
|
|
31
|
+
|
|
32
|
+
# Construct an explicit pair of answer tokens; provider-specific tokenization is checked
|
|
33
|
+
# later.
|
|
34
|
+
def __init__(
|
|
35
|
+
self,
|
|
36
|
+
*,
|
|
37
|
+
yes: str = 'yes',
|
|
38
|
+
no: str = 'no',
|
|
39
|
+
) -> None:
|
|
40
|
+
object.__setattr__(self, "yes", yes)
|
|
41
|
+
object.__setattr__(self, "no", no)
|
|
42
|
+
self.__post_init__()
|
|
43
|
+
|
|
44
|
+
# Distinct nonblank labels are required to define two separate outcomes of the same
|
|
45
|
+
# experiment.
|
|
46
|
+
def __post_init__(self) -> None:
|
|
47
|
+
if any(not isinstance(value, str) or not value.strip() for value in (self.yes, self.no)):
|
|
48
|
+
raise ValueError("binary labels must be non-empty strings")
|
|
49
|
+
if self.yes == self.no:
|
|
50
|
+
raise ValueError("binary labels must be different")
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@dataclass(frozen=True, init=False)
|
|
54
|
+
class ModelIdentity:
|
|
55
|
+
"""A loaded model and the explicit request names accepted for it."""
|
|
56
|
+
|
|
57
|
+
name: str
|
|
58
|
+
aliases: Sequence[str] = ()
|
|
59
|
+
|
|
60
|
+
# Capture the loaded model name and accepted request aliases without implicitly routing
|
|
61
|
+
# between models.
|
|
62
|
+
def __init__(
|
|
63
|
+
self,
|
|
64
|
+
*,
|
|
65
|
+
name: str,
|
|
66
|
+
aliases: Sequence[str] = (),
|
|
67
|
+
) -> None:
|
|
68
|
+
object.__setattr__(self, "name", name)
|
|
69
|
+
object.__setattr__(self, "aliases", aliases)
|
|
70
|
+
self.__post_init__()
|
|
71
|
+
|
|
72
|
+
# Detach alias storage and reject malformed names before any identity comparison is attempted.
|
|
73
|
+
def __post_init__(self) -> None:
|
|
74
|
+
model_name(self.name)
|
|
75
|
+
if not isinstance(self.aliases, Sequence) or isinstance(self.aliases, (str, bytes)):
|
|
76
|
+
raise ValueError("model aliases must be a sequence of names")
|
|
77
|
+
aliases = tuple(self.aliases)
|
|
78
|
+
if any(not isinstance(alias, str) or not alias.strip() for alias in aliases):
|
|
79
|
+
raise ValueError("model aliases must be non-empty strings")
|
|
80
|
+
object.__setattr__(self, "aliases", aliases)
|
|
81
|
+
|
|
82
|
+
# Match only explicitly accepted names; the HTTP layer sanitizes this typed failure for public
|
|
83
|
+
# responses.
|
|
84
|
+
def validate(self, model: str) -> None:
|
|
85
|
+
if model != self.name and model not in self.aliases:
|
|
86
|
+
raise ModelNotFoundError(f"The model {model!r} does not exist")
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
class InvalidRequestError(ValueError):
|
|
90
|
+
"""Input cannot be evaluated; no scoring call has been made."""
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
class ModelNotFoundError(ValueError):
|
|
94
|
+
"""A request names a model outside an explicitly configured identity."""
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
@dataclass(frozen=True, init=False)
|
|
98
|
+
class BinaryScores:
|
|
99
|
+
"""One ordered P(yes) per prompt, conditional on the two labels."""
|
|
100
|
+
|
|
101
|
+
yes_probabilities: Sequence[float]
|
|
102
|
+
usage: Usage = field(default_factory=Usage)
|
|
103
|
+
|
|
104
|
+
# Pair candidate-ordered support with measured usage, allowing unknown counters to remain
|
|
105
|
+
# unknown.
|
|
106
|
+
def __init__(
|
|
107
|
+
self,
|
|
108
|
+
*,
|
|
109
|
+
yes_probabilities: Sequence[float],
|
|
110
|
+
usage: Usage = Usage(),
|
|
111
|
+
) -> None:
|
|
112
|
+
object.__setattr__(self, "yes_probabilities", yes_probabilities)
|
|
113
|
+
object.__setattr__(self, "usage", usage)
|
|
114
|
+
self.__post_init__()
|
|
115
|
+
|
|
116
|
+
# Copy and validate independent supports without imposing categorical unit-mass normalization.
|
|
117
|
+
def __post_init__(self) -> None:
|
|
118
|
+
probabilities = support_values(self.yes_probabilities)
|
|
119
|
+
if not isinstance(self.usage, Usage):
|
|
120
|
+
raise ValueError("usage must be a Usage instance")
|
|
121
|
+
object.__setattr__(self, "yes_probabilities", probabilities)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
class BinaryRuntime(Protocol):
|
|
125
|
+
# Implementations must return one label-conditioned support per prompt in the same order, or
|
|
126
|
+
# raise.
|
|
127
|
+
def score(self, *, model: str, prompts: Sequence[ChatPrompt]) -> BinaryScores:
|
|
128
|
+
"""Return next-token binary probabilities in input order."""
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
class AsyncBinaryRuntime(Protocol):
|
|
132
|
+
# Awaitable implementations preserve ordering and cancellation; they return evidence rather
|
|
133
|
+
# than answers.
|
|
134
|
+
async def score(self, *, model: str, prompts: Sequence[ChatPrompt]) -> BinaryScores:
|
|
135
|
+
"""Await ordered binary scores without blocking the event loop."""
|
llm2jev/core/__init__.py
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""Domain facade gathers question, answer, request, response, and JSON vocabulary. Imports
|
|
2
|
+
remain independent of runtimes and web frameworks so protocol values work in minimal
|
|
3
|
+
installations.
|
|
4
|
+
"""
|
|
5
|
+
from .answers import Answer, ChoiceAnswer, NoulAnswer, ScoreAnswer
|
|
6
|
+
from .questions import Choice, Noul, Score
|
|
7
|
+
from .request import JevRequest, Question
|
|
8
|
+
from .response import JevResponse, Usage
|
|
9
|
+
from .types import JSONContent, JSONValue, State
|
|
10
|
+
|
|
11
|
+
__all__ = [
|
|
12
|
+
"Answer",
|
|
13
|
+
"Choice",
|
|
14
|
+
"ChoiceAnswer",
|
|
15
|
+
"JSONContent",
|
|
16
|
+
"JSONValue",
|
|
17
|
+
"JevRequest",
|
|
18
|
+
"Question",
|
|
19
|
+
"JevResponse",
|
|
20
|
+
"Noul",
|
|
21
|
+
"NoulAnswer",
|
|
22
|
+
"Score",
|
|
23
|
+
"ScoreAnswer",
|
|
24
|
+
"State",
|
|
25
|
+
"Usage",
|
|
26
|
+
]
|