sengol 0.8.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (197) hide show
  1. sengol/__init__.py +44 -0
  2. sengol/adapters/__init__.py +38 -0
  3. sengol/adapters/base.py +125 -0
  4. sengol/adapters/jsonl.py +139 -0
  5. sengol/adapters/mlflow.py +229 -0
  6. sengol/adapters/registry.py +174 -0
  7. sengol/analysis/__init__.py +27 -0
  8. sengol/analysis/sampler.py +154 -0
  9. sengol/analysis/trace_analyzer.py +373 -0
  10. sengol/analysis/transition_matrix.py +234 -0
  11. sengol/api/CLAUDE.md +17 -0
  12. sengol/api/__init__.py +25 -0
  13. sengol/api/agent_call_builder.py +160 -0
  14. sengol/api/anchor.py +315 -0
  15. sengol/api/app.py +1322 -0
  16. sengol/api/auth_oidc.py +335 -0
  17. sengol/api/countersign.py +148 -0
  18. sengol/api/drift_events.py +97 -0
  19. sengol/api/dynamodb_audit_store.py +628 -0
  20. sengol/api/eval_run_sink.py +174 -0
  21. sengol/api/gold_score_sink.py +138 -0
  22. sengol/api/middleware/__init__.py +0 -0
  23. sengol/api/middleware/auth.py +227 -0
  24. sengol/api/middleware/rate_limit.py +208 -0
  25. sengol/api/openapi.yaml +4390 -0
  26. sengol/api/otlp_ingest.py +360 -0
  27. sengol/api/postgres_audit_store.py +532 -0
  28. sengol/api/postgres_certification_store.py +260 -0
  29. sengol/api/postgres_controlbook_store.py +117 -0
  30. sengol/api/postgres_eval_run_store.py +250 -0
  31. sengol/api/postgres_evidence_store.py +279 -0
  32. sengol/api/postgres_gold_score_store.py +250 -0
  33. sengol/api/postgres_human_review_store.py +213 -0
  34. sengol/api/postgres_report_store.py +224 -0
  35. sengol/api/postgres_saturation_store.py +240 -0
  36. sengol/api/s3_audit_store.py +465 -0
  37. sengol/api/saturation_sink.py +129 -0
  38. sengol/api/signing_kms.py +203 -0
  39. sengol/api/sink.py +598 -0
  40. sengol/api/sse_broker.py +74 -0
  41. sengol/api/v1/__init__.py +0 -0
  42. sengol/api/v1/agents.py +182 -0
  43. sengol/api/v1/audit.py +124 -0
  44. sengol/api/v1/auth.py +431 -0
  45. sengol/api/v1/console.py +826 -0
  46. sengol/api/v1/console_audit.py +253 -0
  47. sengol/api/v1/controlbook.py +82 -0
  48. sengol/api/v1/github_oidc.py +186 -0
  49. sengol/api/v1/keys.py +401 -0
  50. sengol/api/v1/lifecycle.py +200 -0
  51. sengol/api/v1/regulations.py +211 -0
  52. sengol/api/v1/reports.py +242 -0
  53. sengol/api/v1/review.py +201 -0
  54. sengol/api/v1/token_issuance_log.py +129 -0
  55. sengol/api/v1/users.py +581 -0
  56. sengol/build/__init__.py +15 -0
  57. sengol/build/gate.py +256 -0
  58. sengol/build/regression.py +129 -0
  59. sengol/build/suite.py +617 -0
  60. sengol/capture/__init__.py +26 -0
  61. sengol/capture/config.py +166 -0
  62. sengol/capture/redaction.py +44 -0
  63. sengol/capture/store.py +174 -0
  64. sengol/capture/types.py +36 -0
  65. sengol/cicd/__init__.py +10 -0
  66. sengol/cicd/_render.py +273 -0
  67. sengol/cicd/azure_pipelines.py +321 -0
  68. sengol/cicd/github.py +525 -0
  69. sengol/cli/__init__.py +0 -0
  70. sengol/cli/main.py +3859 -0
  71. sengol/core/__init__.py +7 -0
  72. sengol/core/drift.py +148 -0
  73. sengol/core/evaluator.py +106 -0
  74. sengol/core/judge.py +71 -0
  75. sengol/core/judge_validation.py +196 -0
  76. sengol/core/keyring.py +60 -0
  77. sengol/core/merkle.py +43 -0
  78. sengol/core/payload_registry.py +322 -0
  79. sengol/core/registry.py +231 -0
  80. sengol/core/severity.py +33 -0
  81. sengol/core/signing.py +89 -0
  82. sengol/core/types.py +840 -0
  83. sengol/dataset/__init__.py +10 -0
  84. sengol/dataset/_hash.py +96 -0
  85. sengol/dataset/loader.py +258 -0
  86. sengol/dataset/registry.py +277 -0
  87. sengol/evaluators/CLAUDE.md +17 -0
  88. sengol/evaluators/__init__.py +151 -0
  89. sengol/evaluators/deterministic/__init__.py +33 -0
  90. sengol/evaluators/deterministic/disclosure.py +148 -0
  91. sengol/evaluators/deterministic/drift_monitor.py +60 -0
  92. sengol/evaluators/deterministic/exact_match.py +68 -0
  93. sengol/evaluators/deterministic/keyword.py +103 -0
  94. sengol/evaluators/deterministic/pii.py +180 -0
  95. sengol/evaluators/deterministic/regex.py +66 -0
  96. sengol/evaluators/deterministic/response_length.py +78 -0
  97. sengol/evaluators/deterministic/schema_validation.py +82 -0
  98. sengol/evaluators/deterministic/trajectory.py +244 -0
  99. sengol/evaluators/deterministic/unauthorized_delegation.py +68 -0
  100. sengol/evaluators/ensemble.py +243 -0
  101. sengol/evaluators/llm/__init__.py +37 -0
  102. sengol/evaluators/llm/adversarial_robustness.py +112 -0
  103. sengol/evaluators/llm/explainability.py +127 -0
  104. sengol/evaluators/llm/faithfulness.py +83 -0
  105. sengol/evaluators/llm/groundedness.py +178 -0
  106. sengol/evaluators/llm/hallucination.py +85 -0
  107. sengol/evaluators/llm/pii.py +72 -0
  108. sengol/evaluators/llm/refusal.py +125 -0
  109. sengol/evaluators/llm/relevance.py +137 -0
  110. sengol/evaluators/llm/toxicity.py +159 -0
  111. sengol/evaluators/llm/trajectory.py +295 -0
  112. sengol/evaluators/model/__init__.py +1 -0
  113. sengol/evaluators/model/ab_evaluator.py +208 -0
  114. sengol/evaluators/model/auc.py +146 -0
  115. sengol/evaluators/model/champion_challenger.py +154 -0
  116. sengol/evaluators/model/fairness.py +236 -0
  117. sengol/evaluators/model/outcome_monitor.py +166 -0
  118. sengol/evaluators/model/psi.py +175 -0
  119. sengol/evaluators/redteam/__init__.py +17 -0
  120. sengol/evaluators/redteam/base.py +155 -0
  121. sengol/evaluators/redteam/garak_backend.py +127 -0
  122. sengol/governance/CLAUDE.md +13 -0
  123. sengol/governance/__init__.py +0 -0
  124. sengol/governance/_license.py +86 -0
  125. sengol/governance/audit_store.py +329 -0
  126. sengol/governance/certification_store.py +111 -0
  127. sengol/governance/controlbook_store.py +56 -0
  128. sengol/governance/controlbook_validator.py +291 -0
  129. sengol/governance/drift_evidence.py +602 -0
  130. sengol/governance/eval_run_store.py +117 -0
  131. sengol/governance/evidence_store.py +150 -0
  132. sengol/governance/gold_score.py +207 -0
  133. sengol/governance/gold_score_store.py +171 -0
  134. sengol/governance/human_review_store.py +168 -0
  135. sengol/governance/offline_verify.py +407 -0
  136. sengol/governance/report_base.py +780 -0
  137. sengol/governance/report_enterprise.py +780 -0
  138. sengol/governance/report_store.py +173 -0
  139. sengol/governance/saturation_store.py +175 -0
  140. sengol/instrument/__init__.py +73 -0
  141. sengol/instrument/_runner.py +342 -0
  142. sengol/instrument/anthropic_patch.py +108 -0
  143. sengol/instrument/config.py +52 -0
  144. sengol/instrument/exceptions.py +16 -0
  145. sengol/instrument/extractor.py +113 -0
  146. sengol/instrument/openai_patch.py +103 -0
  147. sengol/instrument/registry.py +58 -0
  148. sengol/judges/__init__.py +22 -0
  149. sengol/judges/_base.py +147 -0
  150. sengol/judges/_exceptions.py +11 -0
  151. sengol/judges/_prompt.py +73 -0
  152. sengol/judges/anthropic.py +78 -0
  153. sengol/judges/openai_compatible.py +84 -0
  154. sengol/logging.py +76 -0
  155. sengol/observability/__init__.py +36 -0
  156. sengol/observability/config.py +156 -0
  157. sengol/observability/spans.py +128 -0
  158. sengol/observability/tracer.py +135 -0
  159. sengol/policies/__init__.py +27 -0
  160. sengol/policies/catalog.yaml +385 -0
  161. sengol/policies/compiled.py +177 -0
  162. sengol/policies/evaluator_routing.yaml +375 -0
  163. sengol/policies/evaluator_translations.yaml +14 -0
  164. sengol/policies/loader.py +358 -0
  165. sengol/policies/taxonomy.py +194 -0
  166. sengol/policies/taxonomy.yaml +105 -0
  167. sengol/policies/taxonomy_crosswalk.yaml +68 -0
  168. sengol/runtime/__init__.py +26 -0
  169. sengol/runtime/async_worker.py +216 -0
  170. sengol/runtime/drift.py +147 -0
  171. sengol/runtime/guardrails.py +695 -0
  172. sengol/runtime/inline_evaluator.py +129 -0
  173. sengol/runtime/sampling/__init__.py +6 -0
  174. sengol/runtime/sampling/clustering.py +168 -0
  175. sengol/runtime/sampling/feedback.py +22 -0
  176. sengol/runtime/sampling/random.py +41 -0
  177. sengol/runtime/sqs_eval_worker.py +292 -0
  178. sengol/runtime/webhook.py +108 -0
  179. sengol/sinks/__init__.py +48 -0
  180. sengol/sinks/base.py +121 -0
  181. sengol/sinks/dispatch.py +186 -0
  182. sengol/sinks/langfuse.py +171 -0
  183. sengol/sinks/otlp.py +277 -0
  184. sengol/sinks/registry.py +164 -0
  185. sengol/sinks/semconv.py +51 -0
  186. sengol/synthetic/__init__.py +26 -0
  187. sengol/synthetic/config.py +55 -0
  188. sengol/synthetic/generator.py +210 -0
  189. sengol/synthetic/trace_extractor.py +167 -0
  190. sengol/trajectory/__init__.py +24 -0
  191. sengol/trajectory/mapping.py +355 -0
  192. sengol/trajectory/types.py +42 -0
  193. sengol-0.8.0.dist-info/METADATA +80 -0
  194. sengol-0.8.0.dist-info/RECORD +197 -0
  195. sengol-0.8.0.dist-info/WHEEL +4 -0
  196. sengol-0.8.0.dist-info/entry_points.txt +6 -0
  197. sengol-0.8.0.dist-info/licenses/LICENSE +201 -0
sengol/__init__.py ADDED
@@ -0,0 +1,44 @@
1
+ """sengol — AI governance SDK."""
2
+
3
+ from sengol.build.gate import PromotionGate
4
+ from sengol.runtime.guardrails import GuardrailPolicy, GuardrailRuntime
5
+ from sengol.runtime.sampling import FeedbackSampler, RandomSampler
6
+
7
+ __version__ = "0.8.0"
8
+
9
+
10
+ # Auto-intercept (ADR-0084) — lazy imports to keep anthropic/openai optional
11
+ def configure(**kwargs):
12
+ """Configure sengol auto-intercept. See sengol.instrument.configure for args."""
13
+ from sengol.instrument import configure as _configure
14
+
15
+ return _configure(**kwargs)
16
+
17
+
18
+ def instrument():
19
+ """Patch Anthropic and OpenAI clients with sengol guardrails."""
20
+ from sengol.instrument import instrument as _instrument
21
+
22
+ return _instrument()
23
+
24
+
25
+ def uninstrument():
26
+ """Reverse patches applied by instrument()."""
27
+ from sengol.instrument import uninstrument as _uninstrument
28
+
29
+ return _uninstrument()
30
+
31
+
32
+ from sengol.instrument.exceptions import BlockedError # noqa: E402
33
+
34
+ __all__ = [
35
+ "GuardrailPolicy",
36
+ "GuardrailRuntime",
37
+ "PromotionGate",
38
+ "RandomSampler",
39
+ "FeedbackSampler",
40
+ "configure",
41
+ "instrument",
42
+ "uninstrument",
43
+ "BlockedError",
44
+ ]
@@ -0,0 +1,38 @@
1
+ """Backend export adapters — push Sengol evaluation results to external systems.
2
+
3
+ Adapters here are **optional, bank-configurable export targets**. They are never on
4
+ the agent code path and never the source of truth — the Postgres ``AuditStore``
5
+ (HMAC-signed, append-only) always is. An adapter is a one-way exporter the operator
6
+ opts into; removing one changes nothing about what Sengol records.
7
+
8
+ Each adapter soft-imports its vendor SDK so the dependency is required **only** when
9
+ the adapter is used (see the matching ``[project.optional-dependencies]`` extra in
10
+ ``pyproject.toml``). Importing this package never imports a vendor SDK.
11
+
12
+ Free / Apache 2.0.
13
+ """
14
+
15
+ from sengol.adapters.base import BaseExporter, ResultExporter
16
+ from sengol.adapters.registry import (
17
+ DISABLE_PLUGINS_ENV,
18
+ ENTRY_POINT_GROUP,
19
+ ExporterRegistry,
20
+ register_exporter,
21
+ )
22
+
23
+ # Importing these modules registers the built-in exporters (via the
24
+ # @register_exporter decorator) — keep them last so the registry symbols above are
25
+ # defined when the decorators run.
26
+ from sengol.adapters.jsonl import JsonlFileExporter
27
+ from sengol.adapters.mlflow import MLflowExporter
28
+
29
+ __all__ = [
30
+ "BaseExporter",
31
+ "ResultExporter",
32
+ "ExporterRegistry",
33
+ "register_exporter",
34
+ "ENTRY_POINT_GROUP",
35
+ "DISABLE_PLUGINS_ENV",
36
+ "MLflowExporter",
37
+ "JsonlFileExporter",
38
+ ]
@@ -0,0 +1,125 @@
1
+ """Exporter plugin contract — the structural type and base class third parties extend.
2
+
3
+ An *exporter* is an optional, downstream sink for evaluation results: it pushes an
4
+ :class:`~sengol.core.types.EvalResult` to a system of the operator's choosing (MLflow,
5
+ Weights & Biases, Datadog, an in-house warehouse, …). Exporters are **never** on the
6
+ agent code path and **never** the source of truth — the Postgres ``AuditStore`` always
7
+ is. Removing or breaking an exporter changes nothing about what Sengol records.
8
+
9
+ Two surfaces:
10
+
11
+ * :class:`ResultExporter` — a ``runtime_checkable`` Protocol describing the minimal shape
12
+ the run-suite exporter loop depends on (``name`` + ``export``). Type against this.
13
+ * :class:`BaseExporter` — the ABC built-ins and third-party plugins subclass. It supplies
14
+ ``from_config`` (build from a ``sengol.yaml`` options mapping) and a default
15
+ ``export_batch``, so a plugin only has to implement ``export`` and set ``name``.
16
+
17
+ Register a subclass with :func:`sengol.adapters.registry.register_exporter` (built-ins) or
18
+ declare it under the ``sengol.exporters`` entry-point group (third-party packages) — see
19
+ :mod:`sengol.adapters.registry`.
20
+
21
+ Free / Apache 2.0 — no enterprise gate on this code path.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ from abc import ABC, abstractmethod
27
+ from typing import (
28
+ Any,
29
+ ClassVar,
30
+ Mapping,
31
+ Optional,
32
+ Protocol,
33
+ Sequence,
34
+ runtime_checkable,
35
+ )
36
+
37
+ from sengol.core.types import EvalResult
38
+
39
+
40
+ @runtime_checkable
41
+ class ResultExporter(Protocol):
42
+ """Structural type the exporter loop depends on.
43
+
44
+ Anything carrying a ``name`` and an
45
+ ``export(result, *, gold_score=None, tenant_id=None)`` method satisfies it —
46
+ subclassing :class:`BaseExporter` is the supported way to get there, but the
47
+ Protocol keeps the run-suite loop decoupled from the base class.
48
+
49
+ ``export`` is **synchronous by design** even though most targets are network
50
+ I/O. The run-suite loop runs each exporter in a worker thread with a timeout
51
+ (so a slow/unreachable backend can't stall the run), which keeps the plugin
52
+ contract trivial for authors — no event loop, no ``async def``. An exporter
53
+ that wants concurrency internally is free to use it.
54
+ """
55
+
56
+ name: str
57
+
58
+ def export(
59
+ self,
60
+ result: EvalResult,
61
+ *,
62
+ gold_score: Optional[float] = None,
63
+ tenant_id: Optional[str] = None,
64
+ ) -> Optional[str]: ...
65
+
66
+
67
+ class BaseExporter(ABC):
68
+ """Base class for built-in and third-party exporters.
69
+
70
+ Subclasses set the class-level :attr:`name` and implement :meth:`export`. The
71
+ default :meth:`from_config` maps a ``sengol.yaml`` options mapping straight to
72
+ keyword arguments; override it when the exporter needs custom parsing/validation.
73
+ """
74
+
75
+ #: Stable short name used to select the exporter in ``sengol.yaml`` / ``--export``.
76
+ #: Must be non-empty and unique across registered exporters.
77
+ name: ClassVar[str] = ""
78
+
79
+ @classmethod
80
+ def from_config(cls, config: Mapping[str, Any]) -> "BaseExporter":
81
+ """Build an instance from a ``sengol.yaml`` ``exporters.<name>`` options mapping.
82
+
83
+ The default forwards the mapping as keyword arguments, so an exporter whose
84
+ ``__init__`` keywords match the config keys needs no override. Raise
85
+ ``TypeError``/``ValueError`` for invalid options — the run-suite loop reports
86
+ it and skips the exporter rather than failing the run.
87
+ """
88
+ return cls(**dict(config))
89
+
90
+ @abstractmethod
91
+ def export(
92
+ self,
93
+ result: EvalResult,
94
+ *,
95
+ gold_score: Optional[float] = None,
96
+ tenant_id: Optional[str] = None,
97
+ ) -> Optional[str]:
98
+ """Push one result downstream. Return an opaque id (e.g. a run id) or ``None``.
99
+
100
+ ``tenant_id`` is the run's tenant (``EvalResult`` itself carries none) so the
101
+ exported record stays attributable in a multi-tenant deployment — log it as a
102
+ tag/label/column. ``gold_score`` is the period score when the caller computed
103
+ one (the ``run-suite`` path does not, so it is ``None`` there).
104
+
105
+ Must not raise for *expected* downstream conditions the caller can do nothing
106
+ about — but any exception is caught by the run-suite loop and logged, never
107
+ propagated into the gate verdict.
108
+ """
109
+ raise NotImplementedError
110
+
111
+ def export_batch(
112
+ self,
113
+ results: Sequence[EvalResult],
114
+ *,
115
+ gold_score: Optional[float] = None,
116
+ tenant_id: Optional[str] = None,
117
+ ) -> list[Optional[str]]:
118
+ """Export each result in order. Override for a more efficient bulk path
119
+ (e.g. grouping the batch under one parent run)."""
120
+ return [
121
+ self.export(r, gold_score=gold_score, tenant_id=tenant_id) for r in results
122
+ ]
123
+
124
+
125
+ __all__ = ["ResultExporter", "BaseExporter"]
@@ -0,0 +1,139 @@
1
+ """JSONL file exporter — the lightweight, dependency-free built-in (and worked example).
2
+
3
+ Writes one JSON line per :class:`~sengol.core.types.EvalResult` to a local file. It needs
4
+ no extra dependency (unlike ``[mlflow]``), so it is the cheapest way to get results out of
5
+ Sengol — useful on its own and as the **reference implementation** a third party copies to
6
+ build their own exporter.
7
+
8
+ Unlike :class:`sengol.adapters.mlflow.MLflowExporter` (content-free by construction), this
9
+ exporter *can* emit free text — evaluator ``reason`` strings — which may carry regulated
10
+ content. It therefore demonstrates the **redaction duty of care**: pass ``redact: true`` (or
11
+ custom ``redact_patterns``) and every emitted ``reason`` is scrubbed through a
12
+ :class:`~sengol.capture.redaction.RegexRedactor` before it touches disk. Exporters that ship
13
+ content to a *third-party SaaS* should always redact.
14
+
15
+ Free / Apache 2.0.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import json
21
+ import os
22
+ from pathlib import Path
23
+ from typing import Any, ClassVar, Mapping, Optional
24
+
25
+ from sengol.adapters.base import BaseExporter
26
+ from sengol.adapters.registry import register_exporter
27
+ from sengol.capture.redaction import NoOpRedactor, Redactor, RegexRedactor
28
+ from sengol.core.types import EvalResult
29
+
30
+ # Default PII patterns applied when ``redact: true`` is set without custom patterns.
31
+ # Conservative, illustrative set — emails and US-style SSNs. Operators with stricter
32
+ # needs pass their own ``redact_patterns``.
33
+ _DEFAULT_PII_PATTERNS = [
34
+ r"[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}", # email
35
+ r"\b\d{3}-\d{2}-\d{4}\b", # SSN
36
+ ]
37
+
38
+
39
+ @register_exporter
40
+ class JsonlFileExporter(BaseExporter):
41
+ """Append each result as a JSON line to ``{path}``.
42
+
43
+ Parameters
44
+ ----------
45
+ path:
46
+ Output file. Parent directories are created. Lines are appended, so reruns
47
+ accumulate (rotate/clear it yourself if you want a fresh file per run).
48
+ include_reasons:
49
+ When ``True`` (default), each score's ``reason`` is included. Set ``False`` to
50
+ emit only structured fields (ids, pass/fail) — the safest, content-free shape.
51
+ redactor:
52
+ A :class:`~sengol.capture.redaction.Redactor` applied to every emitted ``reason``.
53
+ Defaults to :class:`NoOpRedactor`. Build one from config with ``redact`` /
54
+ ``redact_patterns`` (see :meth:`from_config`).
55
+ """
56
+
57
+ name: ClassVar[str] = "jsonl"
58
+
59
+ @classmethod
60
+ def from_config(cls, config: Mapping[str, Any]) -> "JsonlFileExporter":
61
+ """Build from a ``sengol.yaml`` ``exporters.jsonl`` mapping.
62
+
63
+ Keys: ``path`` (required), ``include_reasons`` (bool, default True),
64
+ ``redact`` (bool — turn on the default PII patterns) and/or ``redact_patterns``
65
+ (list[str] — custom regexes; implies redaction). Any pattern list takes
66
+ precedence over the default set.
67
+ """
68
+ path = config.get("path")
69
+ if not path:
70
+ raise ValueError("jsonl exporter requires a `path` option")
71
+ patterns = config.get("redact_patterns")
72
+ redactor: Redactor
73
+ if patterns:
74
+ redactor = RegexRedactor(list(patterns))
75
+ elif config.get("redact"):
76
+ redactor = RegexRedactor(_DEFAULT_PII_PATTERNS)
77
+ else:
78
+ redactor = NoOpRedactor()
79
+ return cls(
80
+ path=path,
81
+ include_reasons=bool(config.get("include_reasons", True)),
82
+ redactor=redactor,
83
+ )
84
+
85
+ def __init__(
86
+ self,
87
+ *,
88
+ path: str | os.PathLike[str],
89
+ include_reasons: bool = True,
90
+ redactor: Optional[Redactor] = None,
91
+ ) -> None:
92
+ self._path = Path(path)
93
+ self._include_reasons = include_reasons
94
+ self._redactor: Redactor = redactor if redactor is not None else NoOpRedactor()
95
+
96
+ def export(
97
+ self,
98
+ result: EvalResult,
99
+ *,
100
+ gold_score: Optional[float] = None,
101
+ tenant_id: Optional[str] = None,
102
+ ) -> str:
103
+ """Append one JSON line for ``result``; return the path written to."""
104
+ self._path.parent.mkdir(parents=True, exist_ok=True)
105
+ line = json.dumps(self._row(result, gold_score, tenant_id), sort_keys=True)
106
+ with self._path.open("a", encoding="utf-8") as fh:
107
+ fh.write(line + "\n")
108
+ return str(self._path)
109
+
110
+ def _row(
111
+ self,
112
+ result: EvalResult,
113
+ gold_score: Optional[float],
114
+ tenant_id: Optional[str],
115
+ ) -> dict[str, Any]:
116
+ scores = []
117
+ for s in result.scores:
118
+ entry: dict[str, Any] = {"evaluator": s.evaluator, "passed": s.passed}
119
+ if s.failure_mode is not None:
120
+ entry["failure_mode"] = s.failure_mode
121
+ if self._include_reasons and s.reason:
122
+ # The one free-text field — scrub it before it lands on disk.
123
+ entry["reason"] = self._redactor.redact(s.reason)
124
+ scores.append(entry)
125
+ row: dict[str, Any] = {
126
+ "result_id": str(result.result_id),
127
+ "agent_id": result.agent_id,
128
+ "agent_version": result.agent_version,
129
+ "tenant_id": tenant_id,
130
+ "overall_passed": result.overall_passed,
131
+ "policies": list(result.policies),
132
+ "scores": scores,
133
+ }
134
+ if gold_score is not None:
135
+ row["gold_score"] = gold_score
136
+ return row
137
+
138
+
139
+ __all__ = ["JsonlFileExporter"]
@@ -0,0 +1,229 @@
1
+ """MLflow backend adapter — export Sengol eval results to MLflow tracking.
2
+
3
+ MLflow is reachable two ways in Sengol: implicitly over OTLP (ADR-0059, Session 5c)
4
+ when a collector fans spans out to it, and explicitly through *this* adapter. This is
5
+ the explicit path — a deliberate ``adapter.export(result)`` call that logs an MLflow run
6
+ with the eval's params, metrics, and tags, for teams that track model/agent quality in
7
+ MLflow experiments alongside their training runs.
8
+
9
+ The adapter is an **optional export target**, never on the agent code path and never the
10
+ source of truth — the Postgres ``AuditStore`` always is. ``mlflow`` is a soft import
11
+ (``pip install 'sengol[mlflow]'`` / ``uv sync --extra mlflow``): importing this module
12
+ costs nothing, and the SDK is only required when an adapter is actually constructed.
13
+
14
+ **Authentication.** This adapter brokers no credentials. ``tracking_uri`` is the endpoint
15
+ only; auth is handled by the mlflow SDK's own environment-variable mechanism in the
16
+ runtime — ``MLFLOW_TRACKING_USERNAME``/``MLFLOW_TRACKING_PASSWORD`` (basic),
17
+ ``MLFLOW_TRACKING_TOKEN`` (bearer), or ``DATABRICKS_HOST``/``DATABRICKS_TOKEN``. Keep
18
+ secrets in the environment, not in ``sengol.yaml`` (see ``docs/reference/env.md``).
19
+
20
+ Free / Apache 2.0 — no enterprise gate on this code path.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ import re
26
+ from typing import Any, ClassVar, Mapping, Optional
27
+
28
+ from sengol.adapters.base import BaseExporter
29
+ from sengol.adapters.registry import register_exporter
30
+ from sengol.core.types import EvalResult
31
+
32
+ # Message shown when the adapter is used without the optional dependency. Mirrors
33
+ # the soft-import contract used by the ``otlp`` and ``s3`` extras.
34
+ _MISSING_MLFLOW = (
35
+ "MLflowExporter requires the 'mlflow' optional dependency. "
36
+ "Install it with `pip install 'sengol[mlflow]'` or `uv sync --extra mlflow`."
37
+ )
38
+
39
+
40
+ def _import_mlflow() -> Any:
41
+ try:
42
+ import mlflow # type: ignore
43
+ except ImportError as exc: # pragma: no cover - exercised via injected module
44
+ raise ImportError(_MISSING_MLFLOW) from exc
45
+ return mlflow
46
+
47
+
48
+ def _pass_rate(result: EvalResult) -> float:
49
+ """Fraction of this result's scores that passed. No scores → 0.0."""
50
+ if not result.scores:
51
+ return 0.0
52
+ return sum(1 for s in result.scores if s.passed) / len(result.scores)
53
+
54
+
55
+ # MLflow metric keys allow only these characters; anything else raises on log.
56
+ _METRIC_KEY_OK = re.compile(r"[^A-Za-z0-9_\-./ ]")
57
+
58
+
59
+ def _sanitize_metric_key(name: str) -> str:
60
+ """Replace characters MLflow forbids in a metric key with ``_``.
61
+
62
+ Keeps the key stable and unique-enough for the common case (most evaluator
63
+ names are already valid); a name that is *only* invalid chars collapses to
64
+ underscores, which is acceptable for a derived per-evaluator metric.
65
+ """
66
+ return _METRIC_KEY_OK.sub("_", name)
67
+
68
+
69
+ @register_exporter
70
+ class MLflowExporter(BaseExporter):
71
+ """Export :class:`EvalResult` batches to an MLflow experiment.
72
+
73
+ Registered as the built-in ``mlflow`` exporter — select it via the ``sengol.yaml``
74
+ ``exporters:`` block or ``sengol run-suite --export mlflow``.
75
+
76
+ **Content-free by construction.** This exporter logs only *structured, non-content*
77
+ fields — ids, counts, pass/fail metrics, policy ids. It never logs prompts, responses,
78
+ or evaluator ``reason`` free text, so it cannot leak regulated content to an external
79
+ tracking server. An exporter that *does* need to emit free text must scrub it first
80
+ (see :class:`sengol.adapters.jsonl.JsonlFileExporter` and its ``redact`` option for the
81
+ reference pattern).
82
+
83
+ Parameters
84
+ ----------
85
+ experiment:
86
+ MLflow experiment name. Created on first use by ``set_experiment``.
87
+ tracking_uri:
88
+ Optional MLflow tracking URI (``http://…``, ``databricks``, a local
89
+ path, …). When omitted, MLflow's own default/config resolution applies.
90
+ Authentication is handled by the mlflow SDK's environment variables
91
+ (``MLFLOW_TRACKING_*`` / ``DATABRICKS_*``) — never passed here.
92
+ mlflow_module:
93
+ Injection seam for tests — pass a stand-in exposing
94
+ ``set_tracking_uri`` / ``set_experiment`` / ``start_run`` /
95
+ ``log_params`` / ``log_metrics`` / ``set_tags``. Production callers
96
+ leave this ``None`` and the real ``mlflow`` SDK is soft-imported.
97
+ """
98
+
99
+ name: ClassVar[str] = "mlflow"
100
+
101
+ @classmethod
102
+ def from_config(cls, config: Mapping[str, Any]) -> "MLflowExporter":
103
+ """Build from a ``sengol.yaml`` ``exporters.mlflow`` mapping.
104
+
105
+ Recognized keys: ``experiment``, ``tracking_uri``. The ``mlflow_module``
106
+ test seam is intentionally not configurable from YAML.
107
+ """
108
+ return cls(
109
+ experiment=config.get("experiment", "sengol"),
110
+ tracking_uri=config.get("tracking_uri"),
111
+ )
112
+
113
+ def __init__(
114
+ self,
115
+ *,
116
+ experiment: str = "sengol",
117
+ tracking_uri: Optional[str] = None,
118
+ mlflow_module: Any = None,
119
+ ) -> None:
120
+ self._mlflow = mlflow_module if mlflow_module is not None else _import_mlflow()
121
+ if tracking_uri:
122
+ self._mlflow.set_tracking_uri(tracking_uri)
123
+ self._experiment = experiment
124
+ self._mlflow.set_experiment(experiment)
125
+
126
+ def export(
127
+ self,
128
+ result: EvalResult,
129
+ *,
130
+ gold_score: Optional[float] = None,
131
+ tenant_id: Optional[str] = None,
132
+ run_name: Optional[str] = None,
133
+ nested: bool = False,
134
+ ) -> str:
135
+ """Log one MLflow run for ``result`` and return its ``run_id``.
136
+
137
+ Params capture the run's identity (agent, version, policies, tenant); metrics
138
+ capture the verdict (pass-rate, per-evaluator pass/fail, latency/token/cost
139
+ rollups, optional ``gold_score``); tags carry cross-references back to the Sengol
140
+ record. ``gold_score`` is logged only when supplied — the ``run-suite`` path does
141
+ not compute one. ``nested`` opens the run under the active parent run (used by
142
+ :meth:`export_batch`).
143
+ """
144
+ run_ctx = self._mlflow.start_run(
145
+ run_name=run_name or result.agent_id, nested=nested
146
+ )
147
+ with run_ctx as run:
148
+ self._mlflow.log_params(self._params(result, tenant_id))
149
+ self._mlflow.log_metrics(self._metrics(result, gold_score))
150
+ self._mlflow.set_tags(self._tags(result, tenant_id))
151
+ return run.info.run_id
152
+
153
+ def export_batch(
154
+ self,
155
+ results,
156
+ *,
157
+ gold_score: Optional[float] = None,
158
+ tenant_id: Optional[str] = None,
159
+ ) -> list[Optional[str]]:
160
+ """Group the batch under one parent MLflow run, each result a nested child.
161
+
162
+ Keeps a multi-input suite legible in the MLflow UI: one parent run per
163
+ ``export_batch`` call, ``len(results)`` nested children. Empty input → no run.
164
+ """
165
+ results = list(results)
166
+ if not results:
167
+ return []
168
+ parent_name = f"sengol-suite-{results[0].agent_id}"
169
+ with self._mlflow.start_run(run_name=parent_name):
170
+ if tenant_id is not None:
171
+ self._mlflow.set_tags({"sengol.tenant_id": tenant_id})
172
+ return [
173
+ self.export(r, gold_score=gold_score, tenant_id=tenant_id, nested=True)
174
+ for r in results
175
+ ]
176
+
177
+ # ------------------------------------------------------------------ #
178
+ # Field mapping (pure — no SDK calls, so it is unit-testable directly)
179
+ # ------------------------------------------------------------------ #
180
+
181
+ @staticmethod
182
+ def _params(result: EvalResult, tenant_id: Optional[str]) -> dict[str, Any]:
183
+ params: dict[str, Any] = {
184
+ "agent_id": result.agent_id,
185
+ "agent_version": result.agent_version,
186
+ "policies": ",".join(result.policies),
187
+ "result_id": str(result.result_id),
188
+ "evaluator_count": len(result.scores),
189
+ "saturation_warning": result.saturation_warning,
190
+ }
191
+ if tenant_id is not None:
192
+ params["tenant_id"] = tenant_id
193
+ return params
194
+
195
+ @staticmethod
196
+ def _metrics(result: EvalResult, gold_score: Optional[float]) -> dict[str, float]:
197
+ metrics: dict[str, float] = {
198
+ "pass_rate": _pass_rate(result),
199
+ "overall_passed": 1.0 if result.overall_passed else 0.0,
200
+ "total_latency_ms": sum(s.latency_ms for s in result.scores),
201
+ "total_tokens": float(sum(s.tokens_used for s in result.scores)),
202
+ "total_cost_usd": sum(s.cost_usd for s in result.scores),
203
+ }
204
+ for s in result.scores:
205
+ # MLflow metric keys allow only alnum + _-./ space. The evaluator
206
+ # registry permits any non-empty name, so a third-party evaluator may
207
+ # carry chars MLflow rejects (e.g. ':') — sanitize so one such name
208
+ # can't raise and abort the whole export.
209
+ metrics[f"eval.{_sanitize_metric_key(s.evaluator)}.passed"] = (
210
+ 1.0 if s.passed else 0.0
211
+ )
212
+ if gold_score is not None:
213
+ metrics["gold_score"] = gold_score
214
+ return metrics
215
+
216
+ @staticmethod
217
+ def _tags(result: EvalResult, tenant_id: Optional[str]) -> dict[str, str]:
218
+ tags = {
219
+ "sengol.source": "sengol",
220
+ "sengol.result_id": str(result.result_id),
221
+ "sengol.agent_id": result.agent_id,
222
+ "sengol.agent_version": result.agent_version,
223
+ }
224
+ if tenant_id is not None:
225
+ tags["sengol.tenant_id"] = tenant_id
226
+ return tags
227
+
228
+
229
+ __all__ = ["MLflowExporter"]