dev-double 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dev_double/__init__.py +3 -0
- dev_double/apps/__init__.py +1 -0
- dev_double/apps/cli/__init__.py +5 -0
- dev_double/apps/cli/main.py +122 -0
- dev_double/apps/client/__init__.py +5 -0
- dev_double/apps/client/http_client.py +102 -0
- dev_double/apps/composition.py +71 -0
- dev_double/apps/config.py +36 -0
- dev_double/apps/server/__init__.py +5 -0
- dev_double/apps/server/app.py +122 -0
- dev_double/apps/server/recorder.py +62 -0
- dev_double/apps/server/transport/__init__.py +39 -0
- dev_double/apps/server/transport/choice_use_cases.py +121 -0
- dev_double/apps/server/transport/common.py +37 -0
- dev_double/apps/server/transport/decide.py +131 -0
- dev_double/apps/server/transport/extract.py +93 -0
- dev_double/apps/server/transport/guard_judge.py +97 -0
- dev_double/apps/server/transport/rerank.py +55 -0
- dev_double/client.py +5 -0
- dev_double/core/__init__.py +1 -0
- dev_double/core/decision/__init__.py +62 -0
- dev_double/core/decision/answer_shaping.py +48 -0
- dev_double/core/decision/confidence.py +19 -0
- dev_double/core/decision/decider_basic_impl.py +65 -0
- dev_double/core/decision/decision_service_basic_impl.py +141 -0
- dev_double/core/decision/defaults.py +40 -0
- dev_double/core/decision/errors.py +7 -0
- dev_double/core/decision/extract_questions.py +40 -0
- dev_double/core/decision/field_extraction.py +53 -0
- dev_double/core/decision/i_clock.py +11 -0
- dev_double/core/decision/i_decider.py +36 -0
- dev_double/core/decision/i_decision_service.py +40 -0
- dev_double/core/decision/i_engine.py +27 -0
- dev_double/core/decision/i_extractor.py +19 -0
- dev_double/core/decision/i_generator.py +19 -0
- dev_double/core/decision/i_id_provider.py +9 -0
- dev_double/core/decision/i_record_reader.py +25 -0
- dev_double/core/decision/i_reranker.py +18 -0
- dev_double/core/decision/label_scoring.py +89 -0
- dev_double/core/decision/prompts.py +130 -0
- dev_double/core/decision/record_parsing.py +153 -0
- dev_double/core/decision/t_answer.py +32 -0
- dev_double/core/decision/t_classify.py +42 -0
- dev_double/core/decision/t_decide.py +26 -0
- dev_double/core/decision/t_extract.py +89 -0
- dev_double/core/decision/t_gate.py +36 -0
- dev_double/core/decision/t_generate.py +37 -0
- dev_double/core/decision/t_guard.py +38 -0
- dev_double/core/decision/t_input.py +8 -0
- dev_double/core/decision/t_judge.py +35 -0
- dev_double/core/decision/t_label_query.py +36 -0
- dev_double/core/decision/t_meta.py +17 -0
- dev_double/core/decision/t_question.py +70 -0
- dev_double/core/decision/t_rerank.py +36 -0
- dev_double/core/decision/t_route.py +28 -0
- dev_double/core/decision/t_usage.py +15 -0
- dev_double/core/decision/tracker.py +27 -0
- dev_double/core/decision/use_case_questions.py +80 -0
- dev_double/core/decision/value_parsing.py +109 -0
- dev_double/providers/__init__.py +1 -0
- dev_double/providers/mock/__init__.py +1 -0
- dev_double/providers/mock/decision/__init__.py +5 -0
- dev_double/providers/mock/decision/clock_mock_impl.py +19 -0
- dev_double/providers/mock/decision/engine_mock_impl.py +78 -0
- dev_double/providers/mock/decision/id_provider_mock_impl.py +15 -0
- dev_double/providers/needle/__init__.py +1 -0
- dev_double/providers/needle/decision/__init__.py +4 -0
- dev_double/providers/needle/decision/decider_needle_impl.py +117 -0
- dev_double/providers/needle/decision/record_tool.py +75 -0
- dev_double/providers/openai/__init__.py +1 -0
- dev_double/providers/openai/decision/__init__.py +3 -0
- dev_double/providers/openai/decision/engine_openai_impl.py +154 -0
- dev_double/providers/std/__init__.py +1 -0
- dev_double/providers/std/decision/__init__.py +4 -0
- dev_double/providers/std/decision/clock_std_impl.py +12 -0
- dev_double/providers/std/decision/id_provider_std_impl.py +12 -0
- dev_double/providers/systemone/__init__.py +1 -0
- dev_double/providers/systemone/decision/__init__.py +3 -0
- dev_double/providers/systemone/decision/decider_system_one_impl.py +100 -0
- dev_double-0.1.0.dist-info/METADATA +354 -0
- dev_double-0.1.0.dist-info/RECORD +84 -0
- dev_double-0.1.0.dist-info/WHEEL +4 -0
- dev_double-0.1.0.dist-info/entry_points.txt +2 -0
- dev_double-0.1.0.dist-info/licenses/LICENSE +21 -0
dev_double/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""App layer: composition roots (server, cli, client). The only place that picks implementations."""
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
"""Command line: `dev-double serve` and `dev-double doctor`."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import asyncio
|
|
7
|
+
import sys
|
|
8
|
+
|
|
9
|
+
from ... import __version__
|
|
10
|
+
from ...core.decision.errors import EngineError
|
|
11
|
+
from ...core.decision.t_answer import TBinaryAnswer, TChoiceAnswer
|
|
12
|
+
from ...core.decision.t_question import TBinaryQuestion, TChoiceQuestion
|
|
13
|
+
from ...core.decision.tracker import Tracker
|
|
14
|
+
from ...providers.std.decision.clock_std_impl import ClockStdImpl
|
|
15
|
+
from ..composition import build_decider
|
|
16
|
+
from ..config import ENGINES, Settings
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _add_engine_args(p: argparse.ArgumentParser) -> None:
|
|
20
|
+
p.add_argument("--engine", choices=list(ENGINES), help="Backend engine (default: openai).")
|
|
21
|
+
p.add_argument("--base-url", help="OpenAI-compatible base URL (default: Ollama on localhost).")
|
|
22
|
+
p.add_argument("--model", help="Model name on that server (default: qwen2.5:7b).")
|
|
23
|
+
p.add_argument("--api-key", help="API key for the model server, if it needs one.")
|
|
24
|
+
p.add_argument("--max-concurrency", type=int, help="Parallel model calls (default: 4).")
|
|
25
|
+
p.add_argument(
|
|
26
|
+
"--systemone-url", help="Kev/Laya server for --engine systemone (default: http://localhost:8000)."
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _settings(args: argparse.Namespace) -> Settings:
|
|
31
|
+
s = Settings.from_env()
|
|
32
|
+
for name in ("engine", "base_url", "model", "api_key", "max_concurrency", "record", "systemone_url"):
|
|
33
|
+
value = getattr(args, name, None)
|
|
34
|
+
if value is not None:
|
|
35
|
+
setattr(s, name, value)
|
|
36
|
+
return s
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def serve(args: argparse.Namespace) -> int:
|
|
40
|
+
import uvicorn
|
|
41
|
+
|
|
42
|
+
from ..server.app import create_app
|
|
43
|
+
|
|
44
|
+
settings = _settings(args)
|
|
45
|
+
print(f"dev-double {__version__}: engine={settings.engine} model={settings.model}")
|
|
46
|
+
if settings.engine == "openai":
|
|
47
|
+
print(f" model server: {settings.base_url}")
|
|
48
|
+
if settings.engine == "systemone":
|
|
49
|
+
print(f" System 1 server: {settings.systemone_url}")
|
|
50
|
+
if settings.record:
|
|
51
|
+
print(f" recording calls to: {settings.record}")
|
|
52
|
+
uvicorn.run(create_app(settings), host=args.host, port=args.port, log_level="info")
|
|
53
|
+
return 0
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
async def _doctor(settings: Settings) -> int:
|
|
57
|
+
settings.max_concurrency = 1
|
|
58
|
+
decider = build_decider(settings)
|
|
59
|
+
print(f"Checking engine={decider.name} model={decider.model} ...")
|
|
60
|
+
ok = True
|
|
61
|
+
t = Tracker(ClockStdImpl())
|
|
62
|
+
try:
|
|
63
|
+
yes = await decider.ask(
|
|
64
|
+
"The package arrived broken and I want my money back.",
|
|
65
|
+
TBinaryQuestion(question="Is the customer asking for a refund?"),
|
|
66
|
+
t,
|
|
67
|
+
)
|
|
68
|
+
pick = await decider.ask(
|
|
69
|
+
"My flight was cancelled, please put me on the next one.",
|
|
70
|
+
TChoiceQuestion(
|
|
71
|
+
question="What does the customer want?",
|
|
72
|
+
options={
|
|
73
|
+
"refund": "Money returned.",
|
|
74
|
+
"rebooking": "A replacement flight.",
|
|
75
|
+
"information": "Only information.",
|
|
76
|
+
},
|
|
77
|
+
),
|
|
78
|
+
t,
|
|
79
|
+
)
|
|
80
|
+
except EngineError as exc:
|
|
81
|
+
print(f"FAIL {exc}")
|
|
82
|
+
return 1
|
|
83
|
+
finally:
|
|
84
|
+
await decider.aclose()
|
|
85
|
+
|
|
86
|
+
assert isinstance(yes, TBinaryAnswer) and isinstance(pick, TChoiceAnswer)
|
|
87
|
+
meta = t.meta(decider.name, decider.model)
|
|
88
|
+
print(f" binary: refund? p(yes)={yes.probability} (expected high)")
|
|
89
|
+
print(f" choice: {pick.value} {pick.probabilities} (expected rebooking)")
|
|
90
|
+
print(f" latency: {meta.latency_ms} ms for 2 questions")
|
|
91
|
+
for w in meta.warnings:
|
|
92
|
+
ok = False
|
|
93
|
+
print(f"WARN {w}")
|
|
94
|
+
if yes.probability < 0.5 or pick.value != "rebooking":
|
|
95
|
+
ok = False
|
|
96
|
+
print("WARN Answers look wrong; this model may be too small for your use case.")
|
|
97
|
+
print("OK" if ok else "Finished with warnings.")
|
|
98
|
+
return 0 if ok else 2
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def main(argv: list[str] | None = None) -> int:
|
|
102
|
+
parser = argparse.ArgumentParser(prog="dev-double", description=__doc__)
|
|
103
|
+
parser.add_argument("--version", action="version", version=__version__)
|
|
104
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
105
|
+
|
|
106
|
+
p_serve = sub.add_parser("serve", help="Run the HTTP server.")
|
|
107
|
+
_add_engine_args(p_serve)
|
|
108
|
+
p_serve.add_argument("--host", default="127.0.0.1")
|
|
109
|
+
p_serve.add_argument("--port", type=int, default=8787)
|
|
110
|
+
p_serve.add_argument("--record", help="Append every call to this JSONL file.")
|
|
111
|
+
p_serve.set_defaults(func=serve)
|
|
112
|
+
|
|
113
|
+
p_doc = sub.add_parser("doctor", help="Check that the model server works and returns logprobs.")
|
|
114
|
+
_add_engine_args(p_doc)
|
|
115
|
+
p_doc.set_defaults(func=lambda a: asyncio.run(_doctor(_settings(a))))
|
|
116
|
+
|
|
117
|
+
args = parser.parse_args(argv)
|
|
118
|
+
return args.func(args)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
if __name__ == "__main__":
|
|
122
|
+
sys.exit(main())
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"""A thin synchronous client for the dev-double server.
|
|
2
|
+
|
|
3
|
+
Keep this client behind your own interface in your codebase. When the real
|
|
4
|
+
decision engine is approved, write a second implementation of that interface
|
|
5
|
+
against the vendor's SDK and switch over; the rest of your harness stays.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
import httpx
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class DevDouble:
|
|
16
|
+
def __init__(self, base_url: str = "http://127.0.0.1:8787", timeout: float = 120.0) -> None:
|
|
17
|
+
self._http = httpx.Client(base_url=base_url.rstrip("/"), timeout=timeout)
|
|
18
|
+
|
|
19
|
+
def close(self) -> None:
|
|
20
|
+
self._http.close()
|
|
21
|
+
|
|
22
|
+
def __enter__(self) -> "DevDouble":
|
|
23
|
+
return self
|
|
24
|
+
|
|
25
|
+
def __exit__(self, *exc: object) -> None:
|
|
26
|
+
self.close()
|
|
27
|
+
|
|
28
|
+
def _post(self, path: str, body: dict[str, Any]) -> dict[str, Any]:
|
|
29
|
+
response = self._http.post(path, json={k: v for k, v in body.items() if v is not None})
|
|
30
|
+
response.raise_for_status()
|
|
31
|
+
return response.json()
|
|
32
|
+
|
|
33
|
+
def decide(self, input: Any, questions: dict[str, dict[str, Any]]) -> dict[str, Any]:
|
|
34
|
+
return self._post("/v1/decide", {"input": input, "questions": questions})
|
|
35
|
+
|
|
36
|
+
def route(self, input: Any, routes: dict[str, str] | None = None) -> dict[str, Any]:
|
|
37
|
+
return self._post("/v1/route", {"input": input, "routes": routes})
|
|
38
|
+
|
|
39
|
+
def guard(
|
|
40
|
+
self,
|
|
41
|
+
input: Any,
|
|
42
|
+
policies: dict[str, str] | None = None,
|
|
43
|
+
scope: str | None = None,
|
|
44
|
+
threshold: float | None = None,
|
|
45
|
+
) -> dict[str, Any]:
|
|
46
|
+
return self._post(
|
|
47
|
+
"/v1/guard", {"input": input, "policies": policies, "scope": scope, "threshold": threshold}
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
def gate(
|
|
51
|
+
self,
|
|
52
|
+
tool: str,
|
|
53
|
+
arguments: dict[str, Any] | str | None = None,
|
|
54
|
+
context: Any = None,
|
|
55
|
+
policy: str | None = None,
|
|
56
|
+
outcomes: dict[str, str] | None = None,
|
|
57
|
+
) -> dict[str, Any]:
|
|
58
|
+
return self._post(
|
|
59
|
+
"/v1/gate",
|
|
60
|
+
{
|
|
61
|
+
"tool_call": {"name": tool, "arguments": arguments or {}},
|
|
62
|
+
"context": context,
|
|
63
|
+
"policy": policy,
|
|
64
|
+
"outcomes": outcomes,
|
|
65
|
+
},
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
def classify(
|
|
69
|
+
self,
|
|
70
|
+
labels: dict[str, str],
|
|
71
|
+
input: Any = None,
|
|
72
|
+
inputs: list[Any] | None = None,
|
|
73
|
+
question: str | None = None,
|
|
74
|
+
) -> dict[str, Any]:
|
|
75
|
+
return self._post(
|
|
76
|
+
"/v1/classify", {"labels": labels, "input": input, "inputs": inputs, "question": question}
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
def judge(
|
|
80
|
+
self,
|
|
81
|
+
output: Any,
|
|
82
|
+
input: Any = None,
|
|
83
|
+
reference: Any = None,
|
|
84
|
+
criteria: str | None = None,
|
|
85
|
+
levels: list[str] | None = None,
|
|
86
|
+
) -> dict[str, Any]:
|
|
87
|
+
return self._post(
|
|
88
|
+
"/v1/judge",
|
|
89
|
+
{"output": output, "input": input, "reference": reference, "criteria": criteria, "levels": levels},
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
def rerank(
|
|
93
|
+
self, query: str, documents: list[str], top_n: int | None = None, return_documents: bool = False
|
|
94
|
+
) -> dict[str, Any]:
|
|
95
|
+
return self._post(
|
|
96
|
+
"/v1/rerank",
|
|
97
|
+
{"query": query, "documents": documents, "top_n": top_n, "return_documents": return_documents},
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
def extract(self, input: Any, fields: dict[str, dict[str, Any]]) -> dict[str, Any]:
|
|
101
|
+
"""``fields``: name -> {"type": string|number|integer|boolean|enum, "description"?, "options"?}."""
|
|
102
|
+
return self._post("/v1/extract", {"input": input, "fields": fields})
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
"""The composition root: the only place that picks implementations.
|
|
2
|
+
|
|
3
|
+
To add an engine: implement ``IEngine`` (a model that scores labels) or
|
|
4
|
+
``IDecider`` (a model that answers typed questions natively) under
|
|
5
|
+
``providers/<name>/decision/``, then add a branch here and to ``ENGINES``.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import Optional
|
|
11
|
+
|
|
12
|
+
from ..core.decision.decider_basic_impl import DeciderBasicImpl
|
|
13
|
+
from ..core.decision.decision_service_basic_impl import DecisionServiceBasicImpl
|
|
14
|
+
from ..core.decision.i_decider import IDecider
|
|
15
|
+
from ..core.decision.i_engine import IEngine
|
|
16
|
+
from ..providers.std.decision.clock_std_impl import ClockStdImpl
|
|
17
|
+
from ..providers.std.decision.id_provider_std_impl import IdProviderStdImpl
|
|
18
|
+
from .config import ENGINES, Settings
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _unknown(engine: str) -> ValueError:
|
|
22
|
+
return ValueError(f"Unknown engine {engine!r}; expected one of: {', '.join(ENGINES)}.")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def build_engine(settings: Settings) -> IEngine:
|
|
26
|
+
"""The label-scoring engine for prompt-based engines (openai, mock)."""
|
|
27
|
+
if settings.engine == "mock":
|
|
28
|
+
from ..providers.mock.decision.engine_mock_impl import EngineMockImpl
|
|
29
|
+
|
|
30
|
+
return EngineMockImpl()
|
|
31
|
+
if settings.engine == "openai":
|
|
32
|
+
from ..providers.openai.decision.engine_openai_impl import EngineOpenAIImpl
|
|
33
|
+
|
|
34
|
+
return EngineOpenAIImpl(
|
|
35
|
+
base_url=settings.base_url,
|
|
36
|
+
model=settings.model,
|
|
37
|
+
api_key=settings.api_key,
|
|
38
|
+
timeout=settings.timeout,
|
|
39
|
+
)
|
|
40
|
+
raise _unknown(settings.engine)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def build_decider(settings: Settings, engine: Optional[IEngine] = None) -> IDecider:
|
|
44
|
+
if engine is not None or settings.engine in ("openai", "mock"):
|
|
45
|
+
return DeciderBasicImpl(engine or build_engine(settings), settings.max_concurrency)
|
|
46
|
+
if settings.engine == "systemone":
|
|
47
|
+
from ..providers.systemone.decision.decider_system_one_impl import DeciderSystemOneImpl
|
|
48
|
+
|
|
49
|
+
return DeciderSystemOneImpl(
|
|
50
|
+
"systemone",
|
|
51
|
+
settings.systemone_url,
|
|
52
|
+
settings.api_key,
|
|
53
|
+
max_concurrency=settings.max_concurrency,
|
|
54
|
+
timeout=settings.timeout,
|
|
55
|
+
)
|
|
56
|
+
if settings.engine == "needle":
|
|
57
|
+
from ..providers.needle.decision.decider_needle_impl import DeciderNeedleImpl
|
|
58
|
+
|
|
59
|
+
return DeciderNeedleImpl()
|
|
60
|
+
raise _unknown(settings.engine)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def build_service(
|
|
64
|
+
settings: Settings,
|
|
65
|
+
engine: Optional[IEngine] = None,
|
|
66
|
+
decider: Optional[IDecider] = None,
|
|
67
|
+
) -> DecisionServiceBasicImpl:
|
|
68
|
+
"""The use-case service. Pass ``engine`` or ``decider`` to override the configured one."""
|
|
69
|
+
return DecisionServiceBasicImpl(
|
|
70
|
+
decider or build_decider(settings, engine), ClockStdImpl(), IdProviderStdImpl()
|
|
71
|
+
)
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""Settings from environment variables (CLI flags override them)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from typing import Optional
|
|
8
|
+
|
|
9
|
+
ENV_PREFIX = "DEV_DOUBLE_"
|
|
10
|
+
ENGINES = ("openai", "mock", "systemone", "needle")
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass
|
|
14
|
+
class Settings:
|
|
15
|
+
engine: str = "openai" # one of ENGINES
|
|
16
|
+
base_url: str = "http://localhost:11434/v1" # OpenAI-compatible server (engine=openai)
|
|
17
|
+
model: str = "qwen2.5:7b"
|
|
18
|
+
api_key: Optional[str] = None
|
|
19
|
+
max_concurrency: int = 4
|
|
20
|
+
timeout: float = 120.0
|
|
21
|
+
record: Optional[str] = None
|
|
22
|
+
systemone_url: str = "http://localhost:8000" # Kev / Laya server (engine=systemone)
|
|
23
|
+
|
|
24
|
+
@classmethod
|
|
25
|
+
def from_env(cls) -> "Settings":
|
|
26
|
+
s = cls()
|
|
27
|
+
env = os.environ
|
|
28
|
+
s.engine = env.get(ENV_PREFIX + "ENGINE", s.engine)
|
|
29
|
+
s.base_url = env.get(ENV_PREFIX + "BASE_URL", s.base_url)
|
|
30
|
+
s.model = env.get(ENV_PREFIX + "MODEL", s.model)
|
|
31
|
+
s.api_key = env.get(ENV_PREFIX + "API_KEY", s.api_key)
|
|
32
|
+
s.max_concurrency = int(env.get(ENV_PREFIX + "MAX_CONCURRENCY", s.max_concurrency))
|
|
33
|
+
s.timeout = float(env.get(ENV_PREFIX + "TIMEOUT", s.timeout))
|
|
34
|
+
s.record = env.get(ENV_PREFIX + "RECORD", s.record)
|
|
35
|
+
s.systemone_url = env.get(ENV_PREFIX + "SYSTEMONE_URL", s.systemone_url)
|
|
36
|
+
return s
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
"""The HTTP server: maps transport schemas to the core use-case service."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from contextlib import asynccontextmanager
|
|
6
|
+
from typing import Optional
|
|
7
|
+
|
|
8
|
+
from fastapi import FastAPI, Request
|
|
9
|
+
from fastapi.responses import JSONResponse
|
|
10
|
+
|
|
11
|
+
from ... import __version__
|
|
12
|
+
from ...core.decision.errors import EngineError
|
|
13
|
+
from ...core.decision.i_decider import IDecider
|
|
14
|
+
from ...core.decision.i_decision_service import IDecisionService
|
|
15
|
+
from ...core.decision.i_engine import IEngine
|
|
16
|
+
from ..composition import build_service
|
|
17
|
+
from ..config import Settings
|
|
18
|
+
from .recorder import install_recorder
|
|
19
|
+
from .transport import (
|
|
20
|
+
ClassifyRequest,
|
|
21
|
+
ClassifyResponse,
|
|
22
|
+
DecideRequest,
|
|
23
|
+
DecideResponse,
|
|
24
|
+
ExtractRequest,
|
|
25
|
+
ExtractResponse,
|
|
26
|
+
GateRequest,
|
|
27
|
+
GateResponse,
|
|
28
|
+
GuardRequest,
|
|
29
|
+
GuardResponse,
|
|
30
|
+
JudgeRequest,
|
|
31
|
+
JudgeResponse,
|
|
32
|
+
RerankRequest,
|
|
33
|
+
RerankResponse,
|
|
34
|
+
RouteRequest,
|
|
35
|
+
RouteResponse,
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def create_app(
|
|
40
|
+
settings: Optional[Settings] = None,
|
|
41
|
+
engine: Optional[IEngine] = None,
|
|
42
|
+
decider: Optional[IDecider] = None,
|
|
43
|
+
) -> FastAPI:
|
|
44
|
+
"""``engine`` / ``decider`` override the one ``settings`` would build."""
|
|
45
|
+
settings = settings or Settings.from_env()
|
|
46
|
+
|
|
47
|
+
@asynccontextmanager
|
|
48
|
+
async def lifespan(app: FastAPI): # type: ignore[no-untyped-def]
|
|
49
|
+
svc = build_service(settings, engine=engine, decider=decider)
|
|
50
|
+
app.state.service = svc
|
|
51
|
+
try:
|
|
52
|
+
yield
|
|
53
|
+
finally:
|
|
54
|
+
await svc.aclose()
|
|
55
|
+
|
|
56
|
+
app = FastAPI(
|
|
57
|
+
title="dev-double",
|
|
58
|
+
version=__version__,
|
|
59
|
+
description=(
|
|
60
|
+
"A local stand-in for fast decision-model and reranking APIs. "
|
|
61
|
+
"Build against it now; point your client at the real engine later."
|
|
62
|
+
),
|
|
63
|
+
lifespan=lifespan,
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
@app.exception_handler(EngineError)
|
|
67
|
+
async def _engine_error(_: Request, exc: EngineError) -> JSONResponse:
|
|
68
|
+
return JSONResponse(status_code=502, content={"error": "engine_error", "detail": str(exc)})
|
|
69
|
+
|
|
70
|
+
if settings.record:
|
|
71
|
+
install_recorder(app, settings.record)
|
|
72
|
+
|
|
73
|
+
def service(request: Request) -> IDecisionService:
|
|
74
|
+
return request.app.state.service
|
|
75
|
+
|
|
76
|
+
@app.get("/health")
|
|
77
|
+
async def health(request: Request) -> dict:
|
|
78
|
+
svc = service(request)
|
|
79
|
+
return {"status": "ok", "engine": svc.name, "model": svc.model}
|
|
80
|
+
|
|
81
|
+
@app.post("/v1/decide", response_model=DecideResponse, response_model_exclude_none=True)
|
|
82
|
+
async def decide(req: DecideRequest, request: Request): # type: ignore[no-untyped-def]
|
|
83
|
+
"""Generic decision: binary, choice, and scale questions about one input."""
|
|
84
|
+
return DecideResponse.from_core(await service(request).decide(req.to_core()))
|
|
85
|
+
|
|
86
|
+
@app.post("/v1/route", response_model=RouteResponse)
|
|
87
|
+
async def route(req: RouteRequest, request: Request): # type: ignore[no-untyped-def]
|
|
88
|
+
"""Pick which model (or handler) should take a request."""
|
|
89
|
+
return RouteResponse.from_core(await service(request).route(req.to_core()))
|
|
90
|
+
|
|
91
|
+
@app.post("/v1/guard", response_model=GuardResponse)
|
|
92
|
+
async def guard(req: GuardRequest, request: Request): # type: ignore[no-untyped-def]
|
|
93
|
+
"""Screen input for injection, abuse, or off-topic use before it reaches an LLM."""
|
|
94
|
+
return GuardResponse.from_core(await service(request).guard(req.to_core()))
|
|
95
|
+
|
|
96
|
+
@app.post("/v1/gate", response_model=GateResponse)
|
|
97
|
+
async def gate(req: GateRequest, request: Request): # type: ignore[no-untyped-def]
|
|
98
|
+
"""Decide whether an agent's tool call is allowed, needs confirmation, or is denied."""
|
|
99
|
+
return GateResponse.from_core(await service(request).gate(req.to_core()))
|
|
100
|
+
|
|
101
|
+
@app.post("/v1/classify", response_model=ClassifyResponse)
|
|
102
|
+
async def classify(req: ClassifyRequest, request: Request): # type: ignore[no-untyped-def]
|
|
103
|
+
"""Label one item or a batch: inbox triage, intent detection, bulk labeling."""
|
|
104
|
+
return ClassifyResponse.from_core(await service(request).classify(req.to_core()))
|
|
105
|
+
|
|
106
|
+
@app.post("/v1/judge", response_model=JudgeResponse)
|
|
107
|
+
async def judge(req: JudgeRequest, request: Request): # type: ignore[no-untyped-def]
|
|
108
|
+
"""Score an LLM output against a rubric."""
|
|
109
|
+
return JudgeResponse.from_core(await service(request).judge(req.to_core()))
|
|
110
|
+
|
|
111
|
+
@app.post("/v1/extract", response_model=ExtractResponse, response_model_exclude_none=True)
|
|
112
|
+
async def extract(req: ExtractRequest, request: Request): # type: ignore[no-untyped-def]
|
|
113
|
+
"""Pull typed fields out of text: invoices, tickets, bookings, tool-call arguments."""
|
|
114
|
+
return ExtractResponse.from_core(await service(request).extract(req.to_core()))
|
|
115
|
+
|
|
116
|
+
@app.post("/v1/rerank", response_model=RerankResponse, response_model_exclude_none=True)
|
|
117
|
+
@app.post("/v2/rerank", response_model=RerankResponse, response_model_exclude_none=True)
|
|
118
|
+
async def rerank(req: RerankRequest, request: Request): # type: ignore[no-untyped-def]
|
|
119
|
+
"""Order documents by relevance to a query. Cohere-style wire format."""
|
|
120
|
+
return RerankResponse.from_core(await service(request).rerank(req.to_core()))
|
|
121
|
+
|
|
122
|
+
return app
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""Appends every API call to a JSONL file, so you can replay the same
|
|
2
|
+
traffic against the real engine later and compare."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import json
|
|
7
|
+
import threading
|
|
8
|
+
import time
|
|
9
|
+
from datetime import datetime, timezone
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from fastapi import FastAPI, Request
|
|
13
|
+
from fastapi.responses import Response
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class Recorder:
|
|
17
|
+
def __init__(self, path: str) -> None:
|
|
18
|
+
self._path = path
|
|
19
|
+
self._lock = threading.Lock()
|
|
20
|
+
|
|
21
|
+
def write(self, entry: dict) -> None:
|
|
22
|
+
line = json.dumps(entry, ensure_ascii=False)
|
|
23
|
+
with self._lock, open(self._path, "a", encoding="utf-8") as f:
|
|
24
|
+
f.write(line + "\n")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def install_recorder(app: FastAPI, path: str) -> None:
|
|
28
|
+
"""Record every POST: request, response, status, and latency."""
|
|
29
|
+
recorder = Recorder(path)
|
|
30
|
+
|
|
31
|
+
@app.middleware("http")
|
|
32
|
+
async def _record(request: Request, call_next): # type: ignore[no-untyped-def]
|
|
33
|
+
if request.method != "POST":
|
|
34
|
+
return await call_next(request)
|
|
35
|
+
started = time.perf_counter()
|
|
36
|
+
body = await request.body()
|
|
37
|
+
response = await call_next(request)
|
|
38
|
+
chunks = [chunk async for chunk in response.body_iterator]
|
|
39
|
+
raw = b"".join(chunks)
|
|
40
|
+
recorder.write(
|
|
41
|
+
{
|
|
42
|
+
"ts": datetime.now(timezone.utc).isoformat(),
|
|
43
|
+
"path": request.url.path,
|
|
44
|
+
"status": response.status_code,
|
|
45
|
+
"latency_ms": round((time.perf_counter() - started) * 1000),
|
|
46
|
+
"request": _json_or_text(body),
|
|
47
|
+
"response": _json_or_text(raw),
|
|
48
|
+
}
|
|
49
|
+
)
|
|
50
|
+
return Response(
|
|
51
|
+
content=raw,
|
|
52
|
+
status_code=response.status_code,
|
|
53
|
+
headers=dict(response.headers),
|
|
54
|
+
media_type=response.media_type,
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _json_or_text(raw: bytes) -> Any:
|
|
59
|
+
try:
|
|
60
|
+
return json.loads(raw)
|
|
61
|
+
except ValueError:
|
|
62
|
+
return raw.decode("utf-8", errors="replace")
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""HTTP transport schemas: pydantic models that validate requests, describe the
|
|
2
|
+
OpenAPI schema, and map to and from the core T-types. All serialization lives here."""
|
|
3
|
+
|
|
4
|
+
from .choice_use_cases import (
|
|
5
|
+
Classification,
|
|
6
|
+
ClassifyRequest,
|
|
7
|
+
ClassifyResponse,
|
|
8
|
+
GateRequest,
|
|
9
|
+
GateResponse,
|
|
10
|
+
RouteRequest,
|
|
11
|
+
RouteResponse,
|
|
12
|
+
ToolCall,
|
|
13
|
+
)
|
|
14
|
+
from .common import InputValue, Meta, Usage
|
|
15
|
+
from .decide import (
|
|
16
|
+
Answer,
|
|
17
|
+
BinaryAnswer,
|
|
18
|
+
BinaryQuestion,
|
|
19
|
+
ChoiceAnswer,
|
|
20
|
+
ChoiceQuestion,
|
|
21
|
+
DecideRequest,
|
|
22
|
+
DecideResponse,
|
|
23
|
+
Question,
|
|
24
|
+
ScaleAnswer,
|
|
25
|
+
ScaleQuestion,
|
|
26
|
+
)
|
|
27
|
+
from .extract import ExtractField, ExtractRequest, ExtractResponse, FieldValue
|
|
28
|
+
from .guard_judge import GuardCheck, GuardRequest, GuardResponse, JudgeRequest, JudgeResponse
|
|
29
|
+
from .rerank import RerankDocument, RerankRequest, RerankResponse, RerankResult
|
|
30
|
+
|
|
31
|
+
__all__ = [
|
|
32
|
+
"Answer", "BinaryAnswer", "BinaryQuestion", "ChoiceAnswer", "ChoiceQuestion",
|
|
33
|
+
"Classification", "ClassifyRequest", "ClassifyResponse", "DecideRequest", "DecideResponse",
|
|
34
|
+
"ExtractField", "ExtractRequest", "ExtractResponse", "FieldValue",
|
|
35
|
+
"GateRequest", "GateResponse", "GuardCheck", "GuardRequest", "GuardResponse", "InputValue",
|
|
36
|
+
"JudgeRequest", "JudgeResponse", "Meta", "Question", "RerankDocument", "RerankRequest",
|
|
37
|
+
"RerankResponse", "RerankResult", "RouteRequest", "RouteResponse", "ScaleAnswer",
|
|
38
|
+
"ScaleQuestion", "ToolCall", "Usage",
|
|
39
|
+
]
|