dev-double 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. dev_double/__init__.py +3 -0
  2. dev_double/apps/__init__.py +1 -0
  3. dev_double/apps/cli/__init__.py +5 -0
  4. dev_double/apps/cli/main.py +122 -0
  5. dev_double/apps/client/__init__.py +5 -0
  6. dev_double/apps/client/http_client.py +102 -0
  7. dev_double/apps/composition.py +71 -0
  8. dev_double/apps/config.py +36 -0
  9. dev_double/apps/server/__init__.py +5 -0
  10. dev_double/apps/server/app.py +122 -0
  11. dev_double/apps/server/recorder.py +62 -0
  12. dev_double/apps/server/transport/__init__.py +39 -0
  13. dev_double/apps/server/transport/choice_use_cases.py +121 -0
  14. dev_double/apps/server/transport/common.py +37 -0
  15. dev_double/apps/server/transport/decide.py +131 -0
  16. dev_double/apps/server/transport/extract.py +93 -0
  17. dev_double/apps/server/transport/guard_judge.py +97 -0
  18. dev_double/apps/server/transport/rerank.py +55 -0
  19. dev_double/client.py +5 -0
  20. dev_double/core/__init__.py +1 -0
  21. dev_double/core/decision/__init__.py +62 -0
  22. dev_double/core/decision/answer_shaping.py +48 -0
  23. dev_double/core/decision/confidence.py +19 -0
  24. dev_double/core/decision/decider_basic_impl.py +65 -0
  25. dev_double/core/decision/decision_service_basic_impl.py +141 -0
  26. dev_double/core/decision/defaults.py +40 -0
  27. dev_double/core/decision/errors.py +7 -0
  28. dev_double/core/decision/extract_questions.py +40 -0
  29. dev_double/core/decision/field_extraction.py +53 -0
  30. dev_double/core/decision/i_clock.py +11 -0
  31. dev_double/core/decision/i_decider.py +36 -0
  32. dev_double/core/decision/i_decision_service.py +40 -0
  33. dev_double/core/decision/i_engine.py +27 -0
  34. dev_double/core/decision/i_extractor.py +19 -0
  35. dev_double/core/decision/i_generator.py +19 -0
  36. dev_double/core/decision/i_id_provider.py +9 -0
  37. dev_double/core/decision/i_record_reader.py +25 -0
  38. dev_double/core/decision/i_reranker.py +18 -0
  39. dev_double/core/decision/label_scoring.py +89 -0
  40. dev_double/core/decision/prompts.py +130 -0
  41. dev_double/core/decision/record_parsing.py +153 -0
  42. dev_double/core/decision/t_answer.py +32 -0
  43. dev_double/core/decision/t_classify.py +42 -0
  44. dev_double/core/decision/t_decide.py +26 -0
  45. dev_double/core/decision/t_extract.py +89 -0
  46. dev_double/core/decision/t_gate.py +36 -0
  47. dev_double/core/decision/t_generate.py +37 -0
  48. dev_double/core/decision/t_guard.py +38 -0
  49. dev_double/core/decision/t_input.py +8 -0
  50. dev_double/core/decision/t_judge.py +35 -0
  51. dev_double/core/decision/t_label_query.py +36 -0
  52. dev_double/core/decision/t_meta.py +17 -0
  53. dev_double/core/decision/t_question.py +70 -0
  54. dev_double/core/decision/t_rerank.py +36 -0
  55. dev_double/core/decision/t_route.py +28 -0
  56. dev_double/core/decision/t_usage.py +15 -0
  57. dev_double/core/decision/tracker.py +27 -0
  58. dev_double/core/decision/use_case_questions.py +80 -0
  59. dev_double/core/decision/value_parsing.py +109 -0
  60. dev_double/providers/__init__.py +1 -0
  61. dev_double/providers/mock/__init__.py +1 -0
  62. dev_double/providers/mock/decision/__init__.py +5 -0
  63. dev_double/providers/mock/decision/clock_mock_impl.py +19 -0
  64. dev_double/providers/mock/decision/engine_mock_impl.py +78 -0
  65. dev_double/providers/mock/decision/id_provider_mock_impl.py +15 -0
  66. dev_double/providers/needle/__init__.py +1 -0
  67. dev_double/providers/needle/decision/__init__.py +4 -0
  68. dev_double/providers/needle/decision/decider_needle_impl.py +117 -0
  69. dev_double/providers/needle/decision/record_tool.py +75 -0
  70. dev_double/providers/openai/__init__.py +1 -0
  71. dev_double/providers/openai/decision/__init__.py +3 -0
  72. dev_double/providers/openai/decision/engine_openai_impl.py +154 -0
  73. dev_double/providers/std/__init__.py +1 -0
  74. dev_double/providers/std/decision/__init__.py +4 -0
  75. dev_double/providers/std/decision/clock_std_impl.py +12 -0
  76. dev_double/providers/std/decision/id_provider_std_impl.py +12 -0
  77. dev_double/providers/systemone/__init__.py +1 -0
  78. dev_double/providers/systemone/decision/__init__.py +3 -0
  79. dev_double/providers/systemone/decision/decider_system_one_impl.py +100 -0
  80. dev_double-0.1.0.dist-info/METADATA +354 -0
  81. dev_double-0.1.0.dist-info/RECORD +84 -0
  82. dev_double-0.1.0.dist-info/WHEEL +4 -0
  83. dev_double-0.1.0.dist-info/entry_points.txt +2 -0
  84. dev_double-0.1.0.dist-info/licenses/LICENSE +21 -0
dev_double/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """dev-double: a local stand-in for fast decision-model and reranking APIs."""
2
+
3
+ __version__ = "0.1.0"
@@ -0,0 +1 @@
1
+ """App layer: composition roots (server, cli, client). The only place that picks implementations."""
@@ -0,0 +1,5 @@
1
+ """Command line entry point."""
2
+
3
+ from .main import main
4
+
5
+ __all__ = ["main"]
@@ -0,0 +1,122 @@
1
+ """Command line: `dev-double serve` and `dev-double doctor`."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import asyncio
7
+ import sys
8
+
9
+ from ... import __version__
10
+ from ...core.decision.errors import EngineError
11
+ from ...core.decision.t_answer import TBinaryAnswer, TChoiceAnswer
12
+ from ...core.decision.t_question import TBinaryQuestion, TChoiceQuestion
13
+ from ...core.decision.tracker import Tracker
14
+ from ...providers.std.decision.clock_std_impl import ClockStdImpl
15
+ from ..composition import build_decider
16
+ from ..config import ENGINES, Settings
17
+
18
+
19
+ def _add_engine_args(p: argparse.ArgumentParser) -> None:
20
+ p.add_argument("--engine", choices=list(ENGINES), help="Backend engine (default: openai).")
21
+ p.add_argument("--base-url", help="OpenAI-compatible base URL (default: Ollama on localhost).")
22
+ p.add_argument("--model", help="Model name on that server (default: qwen2.5:7b).")
23
+ p.add_argument("--api-key", help="API key for the model server, if it needs one.")
24
+ p.add_argument("--max-concurrency", type=int, help="Parallel model calls (default: 4).")
25
+ p.add_argument(
26
+ "--systemone-url", help="Kev/Laya server for --engine systemone (default: http://localhost:8000)."
27
+ )
28
+
29
+
30
+ def _settings(args: argparse.Namespace) -> Settings:
31
+ s = Settings.from_env()
32
+ for name in ("engine", "base_url", "model", "api_key", "max_concurrency", "record", "systemone_url"):
33
+ value = getattr(args, name, None)
34
+ if value is not None:
35
+ setattr(s, name, value)
36
+ return s
37
+
38
+
39
+ def serve(args: argparse.Namespace) -> int:
40
+ import uvicorn
41
+
42
+ from ..server.app import create_app
43
+
44
+ settings = _settings(args)
45
+ print(f"dev-double {__version__}: engine={settings.engine} model={settings.model}")
46
+ if settings.engine == "openai":
47
+ print(f" model server: {settings.base_url}")
48
+ if settings.engine == "systemone":
49
+ print(f" System 1 server: {settings.systemone_url}")
50
+ if settings.record:
51
+ print(f" recording calls to: {settings.record}")
52
+ uvicorn.run(create_app(settings), host=args.host, port=args.port, log_level="info")
53
+ return 0
54
+
55
+
56
+ async def _doctor(settings: Settings) -> int:
57
+ settings.max_concurrency = 1
58
+ decider = build_decider(settings)
59
+ print(f"Checking engine={decider.name} model={decider.model} ...")
60
+ ok = True
61
+ t = Tracker(ClockStdImpl())
62
+ try:
63
+ yes = await decider.ask(
64
+ "The package arrived broken and I want my money back.",
65
+ TBinaryQuestion(question="Is the customer asking for a refund?"),
66
+ t,
67
+ )
68
+ pick = await decider.ask(
69
+ "My flight was cancelled, please put me on the next one.",
70
+ TChoiceQuestion(
71
+ question="What does the customer want?",
72
+ options={
73
+ "refund": "Money returned.",
74
+ "rebooking": "A replacement flight.",
75
+ "information": "Only information.",
76
+ },
77
+ ),
78
+ t,
79
+ )
80
+ except EngineError as exc:
81
+ print(f"FAIL {exc}")
82
+ return 1
83
+ finally:
84
+ await decider.aclose()
85
+
86
+ assert isinstance(yes, TBinaryAnswer) and isinstance(pick, TChoiceAnswer)
87
+ meta = t.meta(decider.name, decider.model)
88
+ print(f" binary: refund? p(yes)={yes.probability} (expected high)")
89
+ print(f" choice: {pick.value} {pick.probabilities} (expected rebooking)")
90
+ print(f" latency: {meta.latency_ms} ms for 2 questions")
91
+ for w in meta.warnings:
92
+ ok = False
93
+ print(f"WARN {w}")
94
+ if yes.probability < 0.5 or pick.value != "rebooking":
95
+ ok = False
96
+ print("WARN Answers look wrong; this model may be too small for your use case.")
97
+ print("OK" if ok else "Finished with warnings.")
98
+ return 0 if ok else 2
99
+
100
+
101
+ def main(argv: list[str] | None = None) -> int:
102
+ parser = argparse.ArgumentParser(prog="dev-double", description=__doc__)
103
+ parser.add_argument("--version", action="version", version=__version__)
104
+ sub = parser.add_subparsers(dest="command", required=True)
105
+
106
+ p_serve = sub.add_parser("serve", help="Run the HTTP server.")
107
+ _add_engine_args(p_serve)
108
+ p_serve.add_argument("--host", default="127.0.0.1")
109
+ p_serve.add_argument("--port", type=int, default=8787)
110
+ p_serve.add_argument("--record", help="Append every call to this JSONL file.")
111
+ p_serve.set_defaults(func=serve)
112
+
113
+ p_doc = sub.add_parser("doctor", help="Check that the model server works and returns logprobs.")
114
+ _add_engine_args(p_doc)
115
+ p_doc.set_defaults(func=lambda a: asyncio.run(_doctor(_settings(a))))
116
+
117
+ args = parser.parse_args(argv)
118
+ return args.func(args)
119
+
120
+
121
+ if __name__ == "__main__":
122
+ sys.exit(main())
@@ -0,0 +1,5 @@
1
+ """The sync HTTP SDK client."""
2
+
3
+ from .http_client import DevDouble
4
+
5
+ __all__ = ["DevDouble"]
@@ -0,0 +1,102 @@
1
+ """A thin synchronous client for the dev-double server.
2
+
3
+ Keep this client behind your own interface in your codebase. When the real
4
+ decision engine is approved, write a second implementation of that interface
5
+ against the vendor's SDK and switch over; the rest of your harness stays.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from typing import Any
11
+
12
+ import httpx
13
+
14
+
15
+ class DevDouble:
16
+ def __init__(self, base_url: str = "http://127.0.0.1:8787", timeout: float = 120.0) -> None:
17
+ self._http = httpx.Client(base_url=base_url.rstrip("/"), timeout=timeout)
18
+
19
+ def close(self) -> None:
20
+ self._http.close()
21
+
22
+ def __enter__(self) -> "DevDouble":
23
+ return self
24
+
25
+ def __exit__(self, *exc: object) -> None:
26
+ self.close()
27
+
28
+ def _post(self, path: str, body: dict[str, Any]) -> dict[str, Any]:
29
+ response = self._http.post(path, json={k: v for k, v in body.items() if v is not None})
30
+ response.raise_for_status()
31
+ return response.json()
32
+
33
+ def decide(self, input: Any, questions: dict[str, dict[str, Any]]) -> dict[str, Any]:
34
+ return self._post("/v1/decide", {"input": input, "questions": questions})
35
+
36
+ def route(self, input: Any, routes: dict[str, str] | None = None) -> dict[str, Any]:
37
+ return self._post("/v1/route", {"input": input, "routes": routes})
38
+
39
+ def guard(
40
+ self,
41
+ input: Any,
42
+ policies: dict[str, str] | None = None,
43
+ scope: str | None = None,
44
+ threshold: float | None = None,
45
+ ) -> dict[str, Any]:
46
+ return self._post(
47
+ "/v1/guard", {"input": input, "policies": policies, "scope": scope, "threshold": threshold}
48
+ )
49
+
50
+ def gate(
51
+ self,
52
+ tool: str,
53
+ arguments: dict[str, Any] | str | None = None,
54
+ context: Any = None,
55
+ policy: str | None = None,
56
+ outcomes: dict[str, str] | None = None,
57
+ ) -> dict[str, Any]:
58
+ return self._post(
59
+ "/v1/gate",
60
+ {
61
+ "tool_call": {"name": tool, "arguments": arguments or {}},
62
+ "context": context,
63
+ "policy": policy,
64
+ "outcomes": outcomes,
65
+ },
66
+ )
67
+
68
+ def classify(
69
+ self,
70
+ labels: dict[str, str],
71
+ input: Any = None,
72
+ inputs: list[Any] | None = None,
73
+ question: str | None = None,
74
+ ) -> dict[str, Any]:
75
+ return self._post(
76
+ "/v1/classify", {"labels": labels, "input": input, "inputs": inputs, "question": question}
77
+ )
78
+
79
+ def judge(
80
+ self,
81
+ output: Any,
82
+ input: Any = None,
83
+ reference: Any = None,
84
+ criteria: str | None = None,
85
+ levels: list[str] | None = None,
86
+ ) -> dict[str, Any]:
87
+ return self._post(
88
+ "/v1/judge",
89
+ {"output": output, "input": input, "reference": reference, "criteria": criteria, "levels": levels},
90
+ )
91
+
92
+ def rerank(
93
+ self, query: str, documents: list[str], top_n: int | None = None, return_documents: bool = False
94
+ ) -> dict[str, Any]:
95
+ return self._post(
96
+ "/v1/rerank",
97
+ {"query": query, "documents": documents, "top_n": top_n, "return_documents": return_documents},
98
+ )
99
+
100
+ def extract(self, input: Any, fields: dict[str, dict[str, Any]]) -> dict[str, Any]:
101
+ """``fields``: name -> {"type": string|number|integer|boolean|enum, "description"?, "options"?}."""
102
+ return self._post("/v1/extract", {"input": input, "fields": fields})
@@ -0,0 +1,71 @@
1
+ """The composition root: the only place that picks implementations.
2
+
3
+ To add an engine: implement ``IEngine`` (a model that scores labels) or
4
+ ``IDecider`` (a model that answers typed questions natively) under
5
+ ``providers/<name>/decision/``, then add a branch here and to ``ENGINES``.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from typing import Optional
11
+
12
+ from ..core.decision.decider_basic_impl import DeciderBasicImpl
13
+ from ..core.decision.decision_service_basic_impl import DecisionServiceBasicImpl
14
+ from ..core.decision.i_decider import IDecider
15
+ from ..core.decision.i_engine import IEngine
16
+ from ..providers.std.decision.clock_std_impl import ClockStdImpl
17
+ from ..providers.std.decision.id_provider_std_impl import IdProviderStdImpl
18
+ from .config import ENGINES, Settings
19
+
20
+
21
+ def _unknown(engine: str) -> ValueError:
22
+ return ValueError(f"Unknown engine {engine!r}; expected one of: {', '.join(ENGINES)}.")
23
+
24
+
25
+ def build_engine(settings: Settings) -> IEngine:
26
+ """The label-scoring engine for prompt-based engines (openai, mock)."""
27
+ if settings.engine == "mock":
28
+ from ..providers.mock.decision.engine_mock_impl import EngineMockImpl
29
+
30
+ return EngineMockImpl()
31
+ if settings.engine == "openai":
32
+ from ..providers.openai.decision.engine_openai_impl import EngineOpenAIImpl
33
+
34
+ return EngineOpenAIImpl(
35
+ base_url=settings.base_url,
36
+ model=settings.model,
37
+ api_key=settings.api_key,
38
+ timeout=settings.timeout,
39
+ )
40
+ raise _unknown(settings.engine)
41
+
42
+
43
+ def build_decider(settings: Settings, engine: Optional[IEngine] = None) -> IDecider:
44
+ if engine is not None or settings.engine in ("openai", "mock"):
45
+ return DeciderBasicImpl(engine or build_engine(settings), settings.max_concurrency)
46
+ if settings.engine == "systemone":
47
+ from ..providers.systemone.decision.decider_system_one_impl import DeciderSystemOneImpl
48
+
49
+ return DeciderSystemOneImpl(
50
+ "systemone",
51
+ settings.systemone_url,
52
+ settings.api_key,
53
+ max_concurrency=settings.max_concurrency,
54
+ timeout=settings.timeout,
55
+ )
56
+ if settings.engine == "needle":
57
+ from ..providers.needle.decision.decider_needle_impl import DeciderNeedleImpl
58
+
59
+ return DeciderNeedleImpl()
60
+ raise _unknown(settings.engine)
61
+
62
+
63
+ def build_service(
64
+ settings: Settings,
65
+ engine: Optional[IEngine] = None,
66
+ decider: Optional[IDecider] = None,
67
+ ) -> DecisionServiceBasicImpl:
68
+ """The use-case service. Pass ``engine`` or ``decider`` to override the configured one."""
69
+ return DecisionServiceBasicImpl(
70
+ decider or build_decider(settings, engine), ClockStdImpl(), IdProviderStdImpl()
71
+ )
@@ -0,0 +1,36 @@
1
+ """Settings from environment variables (CLI flags override them)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import os
6
+ from dataclasses import dataclass
7
+ from typing import Optional
8
+
9
+ ENV_PREFIX = "DEV_DOUBLE_"
10
+ ENGINES = ("openai", "mock", "systemone", "needle")
11
+
12
+
13
+ @dataclass
14
+ class Settings:
15
+ engine: str = "openai" # one of ENGINES
16
+ base_url: str = "http://localhost:11434/v1" # OpenAI-compatible server (engine=openai)
17
+ model: str = "qwen2.5:7b"
18
+ api_key: Optional[str] = None
19
+ max_concurrency: int = 4
20
+ timeout: float = 120.0
21
+ record: Optional[str] = None
22
+ systemone_url: str = "http://localhost:8000" # Kev / Laya server (engine=systemone)
23
+
24
+ @classmethod
25
+ def from_env(cls) -> "Settings":
26
+ s = cls()
27
+ env = os.environ
28
+ s.engine = env.get(ENV_PREFIX + "ENGINE", s.engine)
29
+ s.base_url = env.get(ENV_PREFIX + "BASE_URL", s.base_url)
30
+ s.model = env.get(ENV_PREFIX + "MODEL", s.model)
31
+ s.api_key = env.get(ENV_PREFIX + "API_KEY", s.api_key)
32
+ s.max_concurrency = int(env.get(ENV_PREFIX + "MAX_CONCURRENCY", s.max_concurrency))
33
+ s.timeout = float(env.get(ENV_PREFIX + "TIMEOUT", s.timeout))
34
+ s.record = env.get(ENV_PREFIX + "RECORD", s.record)
35
+ s.systemone_url = env.get(ENV_PREFIX + "SYSTEMONE_URL", s.systemone_url)
36
+ return s
@@ -0,0 +1,5 @@
1
+ """The HTTP server."""
2
+
3
+ from .app import create_app
4
+
5
+ __all__ = ["create_app"]
@@ -0,0 +1,122 @@
1
+ """The HTTP server: maps transport schemas to the core use-case service."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from contextlib import asynccontextmanager
6
+ from typing import Optional
7
+
8
+ from fastapi import FastAPI, Request
9
+ from fastapi.responses import JSONResponse
10
+
11
+ from ... import __version__
12
+ from ...core.decision.errors import EngineError
13
+ from ...core.decision.i_decider import IDecider
14
+ from ...core.decision.i_decision_service import IDecisionService
15
+ from ...core.decision.i_engine import IEngine
16
+ from ..composition import build_service
17
+ from ..config import Settings
18
+ from .recorder import install_recorder
19
+ from .transport import (
20
+ ClassifyRequest,
21
+ ClassifyResponse,
22
+ DecideRequest,
23
+ DecideResponse,
24
+ ExtractRequest,
25
+ ExtractResponse,
26
+ GateRequest,
27
+ GateResponse,
28
+ GuardRequest,
29
+ GuardResponse,
30
+ JudgeRequest,
31
+ JudgeResponse,
32
+ RerankRequest,
33
+ RerankResponse,
34
+ RouteRequest,
35
+ RouteResponse,
36
+ )
37
+
38
+
39
+ def create_app(
40
+ settings: Optional[Settings] = None,
41
+ engine: Optional[IEngine] = None,
42
+ decider: Optional[IDecider] = None,
43
+ ) -> FastAPI:
44
+ """``engine`` / ``decider`` override the one ``settings`` would build."""
45
+ settings = settings or Settings.from_env()
46
+
47
+ @asynccontextmanager
48
+ async def lifespan(app: FastAPI): # type: ignore[no-untyped-def]
49
+ svc = build_service(settings, engine=engine, decider=decider)
50
+ app.state.service = svc
51
+ try:
52
+ yield
53
+ finally:
54
+ await svc.aclose()
55
+
56
+ app = FastAPI(
57
+ title="dev-double",
58
+ version=__version__,
59
+ description=(
60
+ "A local stand-in for fast decision-model and reranking APIs. "
61
+ "Build against it now; point your client at the real engine later."
62
+ ),
63
+ lifespan=lifespan,
64
+ )
65
+
66
+ @app.exception_handler(EngineError)
67
+ async def _engine_error(_: Request, exc: EngineError) -> JSONResponse:
68
+ return JSONResponse(status_code=502, content={"error": "engine_error", "detail": str(exc)})
69
+
70
+ if settings.record:
71
+ install_recorder(app, settings.record)
72
+
73
+ def service(request: Request) -> IDecisionService:
74
+ return request.app.state.service
75
+
76
+ @app.get("/health")
77
+ async def health(request: Request) -> dict:
78
+ svc = service(request)
79
+ return {"status": "ok", "engine": svc.name, "model": svc.model}
80
+
81
+ @app.post("/v1/decide", response_model=DecideResponse, response_model_exclude_none=True)
82
+ async def decide(req: DecideRequest, request: Request): # type: ignore[no-untyped-def]
83
+ """Generic decision: binary, choice, and scale questions about one input."""
84
+ return DecideResponse.from_core(await service(request).decide(req.to_core()))
85
+
86
+ @app.post("/v1/route", response_model=RouteResponse)
87
+ async def route(req: RouteRequest, request: Request): # type: ignore[no-untyped-def]
88
+ """Pick which model (or handler) should take a request."""
89
+ return RouteResponse.from_core(await service(request).route(req.to_core()))
90
+
91
+ @app.post("/v1/guard", response_model=GuardResponse)
92
+ async def guard(req: GuardRequest, request: Request): # type: ignore[no-untyped-def]
93
+ """Screen input for injection, abuse, or off-topic use before it reaches an LLM."""
94
+ return GuardResponse.from_core(await service(request).guard(req.to_core()))
95
+
96
+ @app.post("/v1/gate", response_model=GateResponse)
97
+ async def gate(req: GateRequest, request: Request): # type: ignore[no-untyped-def]
98
+ """Decide whether an agent's tool call is allowed, needs confirmation, or is denied."""
99
+ return GateResponse.from_core(await service(request).gate(req.to_core()))
100
+
101
+ @app.post("/v1/classify", response_model=ClassifyResponse)
102
+ async def classify(req: ClassifyRequest, request: Request): # type: ignore[no-untyped-def]
103
+ """Label one item or a batch: inbox triage, intent detection, bulk labeling."""
104
+ return ClassifyResponse.from_core(await service(request).classify(req.to_core()))
105
+
106
+ @app.post("/v1/judge", response_model=JudgeResponse)
107
+ async def judge(req: JudgeRequest, request: Request): # type: ignore[no-untyped-def]
108
+ """Score an LLM output against a rubric."""
109
+ return JudgeResponse.from_core(await service(request).judge(req.to_core()))
110
+
111
+ @app.post("/v1/extract", response_model=ExtractResponse, response_model_exclude_none=True)
112
+ async def extract(req: ExtractRequest, request: Request): # type: ignore[no-untyped-def]
113
+ """Pull typed fields out of text: invoices, tickets, bookings, tool-call arguments."""
114
+ return ExtractResponse.from_core(await service(request).extract(req.to_core()))
115
+
116
+ @app.post("/v1/rerank", response_model=RerankResponse, response_model_exclude_none=True)
117
+ @app.post("/v2/rerank", response_model=RerankResponse, response_model_exclude_none=True)
118
+ async def rerank(req: RerankRequest, request: Request): # type: ignore[no-untyped-def]
119
+ """Order documents by relevance to a query. Cohere-style wire format."""
120
+ return RerankResponse.from_core(await service(request).rerank(req.to_core()))
121
+
122
+ return app
@@ -0,0 +1,62 @@
1
+ """Appends every API call to a JSONL file, so you can replay the same
2
+ traffic against the real engine later and compare."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import json
7
+ import threading
8
+ import time
9
+ from datetime import datetime, timezone
10
+ from typing import Any
11
+
12
+ from fastapi import FastAPI, Request
13
+ from fastapi.responses import Response
14
+
15
+
16
+ class Recorder:
17
+ def __init__(self, path: str) -> None:
18
+ self._path = path
19
+ self._lock = threading.Lock()
20
+
21
+ def write(self, entry: dict) -> None:
22
+ line = json.dumps(entry, ensure_ascii=False)
23
+ with self._lock, open(self._path, "a", encoding="utf-8") as f:
24
+ f.write(line + "\n")
25
+
26
+
27
+ def install_recorder(app: FastAPI, path: str) -> None:
28
+ """Record every POST: request, response, status, and latency."""
29
+ recorder = Recorder(path)
30
+
31
+ @app.middleware("http")
32
+ async def _record(request: Request, call_next): # type: ignore[no-untyped-def]
33
+ if request.method != "POST":
34
+ return await call_next(request)
35
+ started = time.perf_counter()
36
+ body = await request.body()
37
+ response = await call_next(request)
38
+ chunks = [chunk async for chunk in response.body_iterator]
39
+ raw = b"".join(chunks)
40
+ recorder.write(
41
+ {
42
+ "ts": datetime.now(timezone.utc).isoformat(),
43
+ "path": request.url.path,
44
+ "status": response.status_code,
45
+ "latency_ms": round((time.perf_counter() - started) * 1000),
46
+ "request": _json_or_text(body),
47
+ "response": _json_or_text(raw),
48
+ }
49
+ )
50
+ return Response(
51
+ content=raw,
52
+ status_code=response.status_code,
53
+ headers=dict(response.headers),
54
+ media_type=response.media_type,
55
+ )
56
+
57
+
58
+ def _json_or_text(raw: bytes) -> Any:
59
+ try:
60
+ return json.loads(raw)
61
+ except ValueError:
62
+ return raw.decode("utf-8", errors="replace")
@@ -0,0 +1,39 @@
1
+ """HTTP transport schemas: pydantic models that validate requests, describe the
2
+ OpenAPI schema, and map to and from the core T-types. All serialization lives here."""
3
+
4
+ from .choice_use_cases import (
5
+ Classification,
6
+ ClassifyRequest,
7
+ ClassifyResponse,
8
+ GateRequest,
9
+ GateResponse,
10
+ RouteRequest,
11
+ RouteResponse,
12
+ ToolCall,
13
+ )
14
+ from .common import InputValue, Meta, Usage
15
+ from .decide import (
16
+ Answer,
17
+ BinaryAnswer,
18
+ BinaryQuestion,
19
+ ChoiceAnswer,
20
+ ChoiceQuestion,
21
+ DecideRequest,
22
+ DecideResponse,
23
+ Question,
24
+ ScaleAnswer,
25
+ ScaleQuestion,
26
+ )
27
+ from .extract import ExtractField, ExtractRequest, ExtractResponse, FieldValue
28
+ from .guard_judge import GuardCheck, GuardRequest, GuardResponse, JudgeRequest, JudgeResponse
29
+ from .rerank import RerankDocument, RerankRequest, RerankResponse, RerankResult
30
+
31
+ __all__ = [
32
+ "Answer", "BinaryAnswer", "BinaryQuestion", "ChoiceAnswer", "ChoiceQuestion",
33
+ "Classification", "ClassifyRequest", "ClassifyResponse", "DecideRequest", "DecideResponse",
34
+ "ExtractField", "ExtractRequest", "ExtractResponse", "FieldValue",
35
+ "GateRequest", "GateResponse", "GuardCheck", "GuardRequest", "GuardResponse", "InputValue",
36
+ "JudgeRequest", "JudgeResponse", "Meta", "Question", "RerankDocument", "RerankRequest",
37
+ "RerankResponse", "RerankResult", "RouteRequest", "RouteResponse", "ScaleAnswer",
38
+ "ScaleQuestion", "ToolCall", "Usage",
39
+ ]