sengol 0.8.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sengol/__init__.py +44 -0
- sengol/adapters/__init__.py +38 -0
- sengol/adapters/base.py +125 -0
- sengol/adapters/jsonl.py +139 -0
- sengol/adapters/mlflow.py +229 -0
- sengol/adapters/registry.py +174 -0
- sengol/analysis/__init__.py +27 -0
- sengol/analysis/sampler.py +154 -0
- sengol/analysis/trace_analyzer.py +373 -0
- sengol/analysis/transition_matrix.py +234 -0
- sengol/api/CLAUDE.md +17 -0
- sengol/api/__init__.py +25 -0
- sengol/api/agent_call_builder.py +160 -0
- sengol/api/anchor.py +315 -0
- sengol/api/app.py +1322 -0
- sengol/api/auth_oidc.py +335 -0
- sengol/api/countersign.py +148 -0
- sengol/api/drift_events.py +97 -0
- sengol/api/dynamodb_audit_store.py +628 -0
- sengol/api/eval_run_sink.py +174 -0
- sengol/api/gold_score_sink.py +138 -0
- sengol/api/middleware/__init__.py +0 -0
- sengol/api/middleware/auth.py +227 -0
- sengol/api/middleware/rate_limit.py +208 -0
- sengol/api/openapi.yaml +4390 -0
- sengol/api/otlp_ingest.py +360 -0
- sengol/api/postgres_audit_store.py +532 -0
- sengol/api/postgres_certification_store.py +260 -0
- sengol/api/postgres_controlbook_store.py +117 -0
- sengol/api/postgres_eval_run_store.py +250 -0
- sengol/api/postgres_evidence_store.py +279 -0
- sengol/api/postgres_gold_score_store.py +250 -0
- sengol/api/postgres_human_review_store.py +213 -0
- sengol/api/postgres_report_store.py +224 -0
- sengol/api/postgres_saturation_store.py +240 -0
- sengol/api/s3_audit_store.py +465 -0
- sengol/api/saturation_sink.py +129 -0
- sengol/api/signing_kms.py +203 -0
- sengol/api/sink.py +598 -0
- sengol/api/sse_broker.py +74 -0
- sengol/api/v1/__init__.py +0 -0
- sengol/api/v1/agents.py +182 -0
- sengol/api/v1/audit.py +124 -0
- sengol/api/v1/auth.py +431 -0
- sengol/api/v1/console.py +826 -0
- sengol/api/v1/console_audit.py +253 -0
- sengol/api/v1/controlbook.py +82 -0
- sengol/api/v1/github_oidc.py +186 -0
- sengol/api/v1/keys.py +401 -0
- sengol/api/v1/lifecycle.py +200 -0
- sengol/api/v1/regulations.py +211 -0
- sengol/api/v1/reports.py +242 -0
- sengol/api/v1/review.py +201 -0
- sengol/api/v1/token_issuance_log.py +129 -0
- sengol/api/v1/users.py +581 -0
- sengol/build/__init__.py +15 -0
- sengol/build/gate.py +256 -0
- sengol/build/regression.py +129 -0
- sengol/build/suite.py +617 -0
- sengol/capture/__init__.py +26 -0
- sengol/capture/config.py +166 -0
- sengol/capture/redaction.py +44 -0
- sengol/capture/store.py +174 -0
- sengol/capture/types.py +36 -0
- sengol/cicd/__init__.py +10 -0
- sengol/cicd/_render.py +273 -0
- sengol/cicd/azure_pipelines.py +321 -0
- sengol/cicd/github.py +525 -0
- sengol/cli/__init__.py +0 -0
- sengol/cli/main.py +3859 -0
- sengol/core/__init__.py +7 -0
- sengol/core/drift.py +148 -0
- sengol/core/evaluator.py +106 -0
- sengol/core/judge.py +71 -0
- sengol/core/judge_validation.py +196 -0
- sengol/core/keyring.py +60 -0
- sengol/core/merkle.py +43 -0
- sengol/core/payload_registry.py +322 -0
- sengol/core/registry.py +231 -0
- sengol/core/severity.py +33 -0
- sengol/core/signing.py +89 -0
- sengol/core/types.py +840 -0
- sengol/dataset/__init__.py +10 -0
- sengol/dataset/_hash.py +96 -0
- sengol/dataset/loader.py +258 -0
- sengol/dataset/registry.py +277 -0
- sengol/evaluators/CLAUDE.md +17 -0
- sengol/evaluators/__init__.py +151 -0
- sengol/evaluators/deterministic/__init__.py +33 -0
- sengol/evaluators/deterministic/disclosure.py +148 -0
- sengol/evaluators/deterministic/drift_monitor.py +60 -0
- sengol/evaluators/deterministic/exact_match.py +68 -0
- sengol/evaluators/deterministic/keyword.py +103 -0
- sengol/evaluators/deterministic/pii.py +180 -0
- sengol/evaluators/deterministic/regex.py +66 -0
- sengol/evaluators/deterministic/response_length.py +78 -0
- sengol/evaluators/deterministic/schema_validation.py +82 -0
- sengol/evaluators/deterministic/trajectory.py +244 -0
- sengol/evaluators/deterministic/unauthorized_delegation.py +68 -0
- sengol/evaluators/ensemble.py +243 -0
- sengol/evaluators/llm/__init__.py +37 -0
- sengol/evaluators/llm/adversarial_robustness.py +112 -0
- sengol/evaluators/llm/explainability.py +127 -0
- sengol/evaluators/llm/faithfulness.py +83 -0
- sengol/evaluators/llm/groundedness.py +178 -0
- sengol/evaluators/llm/hallucination.py +85 -0
- sengol/evaluators/llm/pii.py +72 -0
- sengol/evaluators/llm/refusal.py +125 -0
- sengol/evaluators/llm/relevance.py +137 -0
- sengol/evaluators/llm/toxicity.py +159 -0
- sengol/evaluators/llm/trajectory.py +295 -0
- sengol/evaluators/model/__init__.py +1 -0
- sengol/evaluators/model/ab_evaluator.py +208 -0
- sengol/evaluators/model/auc.py +146 -0
- sengol/evaluators/model/champion_challenger.py +154 -0
- sengol/evaluators/model/fairness.py +236 -0
- sengol/evaluators/model/outcome_monitor.py +166 -0
- sengol/evaluators/model/psi.py +175 -0
- sengol/evaluators/redteam/__init__.py +17 -0
- sengol/evaluators/redteam/base.py +155 -0
- sengol/evaluators/redteam/garak_backend.py +127 -0
- sengol/governance/CLAUDE.md +13 -0
- sengol/governance/__init__.py +0 -0
- sengol/governance/_license.py +86 -0
- sengol/governance/audit_store.py +329 -0
- sengol/governance/certification_store.py +111 -0
- sengol/governance/controlbook_store.py +56 -0
- sengol/governance/controlbook_validator.py +291 -0
- sengol/governance/drift_evidence.py +602 -0
- sengol/governance/eval_run_store.py +117 -0
- sengol/governance/evidence_store.py +150 -0
- sengol/governance/gold_score.py +207 -0
- sengol/governance/gold_score_store.py +171 -0
- sengol/governance/human_review_store.py +168 -0
- sengol/governance/offline_verify.py +407 -0
- sengol/governance/report_base.py +780 -0
- sengol/governance/report_enterprise.py +780 -0
- sengol/governance/report_store.py +173 -0
- sengol/governance/saturation_store.py +175 -0
- sengol/instrument/__init__.py +73 -0
- sengol/instrument/_runner.py +342 -0
- sengol/instrument/anthropic_patch.py +108 -0
- sengol/instrument/config.py +52 -0
- sengol/instrument/exceptions.py +16 -0
- sengol/instrument/extractor.py +113 -0
- sengol/instrument/openai_patch.py +103 -0
- sengol/instrument/registry.py +58 -0
- sengol/judges/__init__.py +22 -0
- sengol/judges/_base.py +147 -0
- sengol/judges/_exceptions.py +11 -0
- sengol/judges/_prompt.py +73 -0
- sengol/judges/anthropic.py +78 -0
- sengol/judges/openai_compatible.py +84 -0
- sengol/logging.py +76 -0
- sengol/observability/__init__.py +36 -0
- sengol/observability/config.py +156 -0
- sengol/observability/spans.py +128 -0
- sengol/observability/tracer.py +135 -0
- sengol/policies/__init__.py +27 -0
- sengol/policies/catalog.yaml +385 -0
- sengol/policies/compiled.py +177 -0
- sengol/policies/evaluator_routing.yaml +375 -0
- sengol/policies/evaluator_translations.yaml +14 -0
- sengol/policies/loader.py +358 -0
- sengol/policies/taxonomy.py +194 -0
- sengol/policies/taxonomy.yaml +105 -0
- sengol/policies/taxonomy_crosswalk.yaml +68 -0
- sengol/runtime/__init__.py +26 -0
- sengol/runtime/async_worker.py +216 -0
- sengol/runtime/drift.py +147 -0
- sengol/runtime/guardrails.py +695 -0
- sengol/runtime/inline_evaluator.py +129 -0
- sengol/runtime/sampling/__init__.py +6 -0
- sengol/runtime/sampling/clustering.py +168 -0
- sengol/runtime/sampling/feedback.py +22 -0
- sengol/runtime/sampling/random.py +41 -0
- sengol/runtime/sqs_eval_worker.py +292 -0
- sengol/runtime/webhook.py +108 -0
- sengol/sinks/__init__.py +48 -0
- sengol/sinks/base.py +121 -0
- sengol/sinks/dispatch.py +186 -0
- sengol/sinks/langfuse.py +171 -0
- sengol/sinks/otlp.py +277 -0
- sengol/sinks/registry.py +164 -0
- sengol/sinks/semconv.py +51 -0
- sengol/synthetic/__init__.py +26 -0
- sengol/synthetic/config.py +55 -0
- sengol/synthetic/generator.py +210 -0
- sengol/synthetic/trace_extractor.py +167 -0
- sengol/trajectory/__init__.py +24 -0
- sengol/trajectory/mapping.py +355 -0
- sengol/trajectory/types.py +42 -0
- sengol-0.8.0.dist-info/METADATA +80 -0
- sengol-0.8.0.dist-info/RECORD +197 -0
- sengol-0.8.0.dist-info/WHEEL +4 -0
- sengol-0.8.0.dist-info/entry_points.txt +6 -0
- sengol-0.8.0.dist-info/licenses/LICENSE +201 -0
sengol/__init__.py
ADDED
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""sengol — AI governance SDK."""
|
|
2
|
+
|
|
3
|
+
from sengol.build.gate import PromotionGate
|
|
4
|
+
from sengol.runtime.guardrails import GuardrailPolicy, GuardrailRuntime
|
|
5
|
+
from sengol.runtime.sampling import FeedbackSampler, RandomSampler
|
|
6
|
+
|
|
7
|
+
__version__ = "0.8.0"
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
# Auto-intercept (ADR-0084) — lazy imports to keep anthropic/openai optional
|
|
11
|
+
def configure(**kwargs):
|
|
12
|
+
"""Configure sengol auto-intercept. See sengol.instrument.configure for args."""
|
|
13
|
+
from sengol.instrument import configure as _configure
|
|
14
|
+
|
|
15
|
+
return _configure(**kwargs)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def instrument():
|
|
19
|
+
"""Patch Anthropic and OpenAI clients with sengol guardrails."""
|
|
20
|
+
from sengol.instrument import instrument as _instrument
|
|
21
|
+
|
|
22
|
+
return _instrument()
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def uninstrument():
|
|
26
|
+
"""Reverse patches applied by instrument()."""
|
|
27
|
+
from sengol.instrument import uninstrument as _uninstrument
|
|
28
|
+
|
|
29
|
+
return _uninstrument()
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
from sengol.instrument.exceptions import BlockedError # noqa: E402
|
|
33
|
+
|
|
34
|
+
__all__ = [
|
|
35
|
+
"GuardrailPolicy",
|
|
36
|
+
"GuardrailRuntime",
|
|
37
|
+
"PromotionGate",
|
|
38
|
+
"RandomSampler",
|
|
39
|
+
"FeedbackSampler",
|
|
40
|
+
"configure",
|
|
41
|
+
"instrument",
|
|
42
|
+
"uninstrument",
|
|
43
|
+
"BlockedError",
|
|
44
|
+
]
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""Backend export adapters — push Sengol evaluation results to external systems.
|
|
2
|
+
|
|
3
|
+
Adapters here are **optional, bank-configurable export targets**. They are never on
|
|
4
|
+
the agent code path and never the source of truth — the Postgres ``AuditStore``
|
|
5
|
+
(HMAC-signed, append-only) always is. An adapter is a one-way exporter the operator
|
|
6
|
+
opts into; removing one changes nothing about what Sengol records.
|
|
7
|
+
|
|
8
|
+
Each adapter soft-imports its vendor SDK so the dependency is required **only** when
|
|
9
|
+
the adapter is used (see the matching ``[project.optional-dependencies]`` extra in
|
|
10
|
+
``pyproject.toml``). Importing this package never imports a vendor SDK.
|
|
11
|
+
|
|
12
|
+
Free / Apache 2.0.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from sengol.adapters.base import BaseExporter, ResultExporter
|
|
16
|
+
from sengol.adapters.registry import (
|
|
17
|
+
DISABLE_PLUGINS_ENV,
|
|
18
|
+
ENTRY_POINT_GROUP,
|
|
19
|
+
ExporterRegistry,
|
|
20
|
+
register_exporter,
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
# Importing these modules registers the built-in exporters (via the
|
|
24
|
+
# @register_exporter decorator) — keep them last so the registry symbols above are
|
|
25
|
+
# defined when the decorators run.
|
|
26
|
+
from sengol.adapters.jsonl import JsonlFileExporter
|
|
27
|
+
from sengol.adapters.mlflow import MLflowExporter
|
|
28
|
+
|
|
29
|
+
__all__ = [
|
|
30
|
+
"BaseExporter",
|
|
31
|
+
"ResultExporter",
|
|
32
|
+
"ExporterRegistry",
|
|
33
|
+
"register_exporter",
|
|
34
|
+
"ENTRY_POINT_GROUP",
|
|
35
|
+
"DISABLE_PLUGINS_ENV",
|
|
36
|
+
"MLflowExporter",
|
|
37
|
+
"JsonlFileExporter",
|
|
38
|
+
]
|
sengol/adapters/base.py
ADDED
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
"""Exporter plugin contract — the structural type and base class third parties extend.
|
|
2
|
+
|
|
3
|
+
An *exporter* is an optional, downstream sink for evaluation results: it pushes an
|
|
4
|
+
:class:`~sengol.core.types.EvalResult` to a system of the operator's choosing (MLflow,
|
|
5
|
+
Weights & Biases, Datadog, an in-house warehouse, …). Exporters are **never** on the
|
|
6
|
+
agent code path and **never** the source of truth — the Postgres ``AuditStore`` always
|
|
7
|
+
is. Removing or breaking an exporter changes nothing about what Sengol records.
|
|
8
|
+
|
|
9
|
+
Two surfaces:
|
|
10
|
+
|
|
11
|
+
* :class:`ResultExporter` — a ``runtime_checkable`` Protocol describing the minimal shape
|
|
12
|
+
the run-suite exporter loop depends on (``name`` + ``export``). Type against this.
|
|
13
|
+
* :class:`BaseExporter` — the ABC built-ins and third-party plugins subclass. It supplies
|
|
14
|
+
``from_config`` (build from a ``sengol.yaml`` options mapping) and a default
|
|
15
|
+
``export_batch``, so a plugin only has to implement ``export`` and set ``name``.
|
|
16
|
+
|
|
17
|
+
Register a subclass with :func:`sengol.adapters.registry.register_exporter` (built-ins) or
|
|
18
|
+
declare it under the ``sengol.exporters`` entry-point group (third-party packages) — see
|
|
19
|
+
:mod:`sengol.adapters.registry`.
|
|
20
|
+
|
|
21
|
+
Free / Apache 2.0 — no enterprise gate on this code path.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
from abc import ABC, abstractmethod
|
|
27
|
+
from typing import (
|
|
28
|
+
Any,
|
|
29
|
+
ClassVar,
|
|
30
|
+
Mapping,
|
|
31
|
+
Optional,
|
|
32
|
+
Protocol,
|
|
33
|
+
Sequence,
|
|
34
|
+
runtime_checkable,
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
from sengol.core.types import EvalResult
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@runtime_checkable
|
|
41
|
+
class ResultExporter(Protocol):
|
|
42
|
+
"""Structural type the exporter loop depends on.
|
|
43
|
+
|
|
44
|
+
Anything carrying a ``name`` and an
|
|
45
|
+
``export(result, *, gold_score=None, tenant_id=None)`` method satisfies it —
|
|
46
|
+
subclassing :class:`BaseExporter` is the supported way to get there, but the
|
|
47
|
+
Protocol keeps the run-suite loop decoupled from the base class.
|
|
48
|
+
|
|
49
|
+
``export`` is **synchronous by design** even though most targets are network
|
|
50
|
+
I/O. The run-suite loop runs each exporter in a worker thread with a timeout
|
|
51
|
+
(so a slow/unreachable backend can't stall the run), which keeps the plugin
|
|
52
|
+
contract trivial for authors — no event loop, no ``async def``. An exporter
|
|
53
|
+
that wants concurrency internally is free to use it.
|
|
54
|
+
"""
|
|
55
|
+
|
|
56
|
+
name: str
|
|
57
|
+
|
|
58
|
+
def export(
|
|
59
|
+
self,
|
|
60
|
+
result: EvalResult,
|
|
61
|
+
*,
|
|
62
|
+
gold_score: Optional[float] = None,
|
|
63
|
+
tenant_id: Optional[str] = None,
|
|
64
|
+
) -> Optional[str]: ...
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class BaseExporter(ABC):
|
|
68
|
+
"""Base class for built-in and third-party exporters.
|
|
69
|
+
|
|
70
|
+
Subclasses set the class-level :attr:`name` and implement :meth:`export`. The
|
|
71
|
+
default :meth:`from_config` maps a ``sengol.yaml`` options mapping straight to
|
|
72
|
+
keyword arguments; override it when the exporter needs custom parsing/validation.
|
|
73
|
+
"""
|
|
74
|
+
|
|
75
|
+
#: Stable short name used to select the exporter in ``sengol.yaml`` / ``--export``.
|
|
76
|
+
#: Must be non-empty and unique across registered exporters.
|
|
77
|
+
name: ClassVar[str] = ""
|
|
78
|
+
|
|
79
|
+
@classmethod
|
|
80
|
+
def from_config(cls, config: Mapping[str, Any]) -> "BaseExporter":
|
|
81
|
+
"""Build an instance from a ``sengol.yaml`` ``exporters.<name>`` options mapping.
|
|
82
|
+
|
|
83
|
+
The default forwards the mapping as keyword arguments, so an exporter whose
|
|
84
|
+
``__init__`` keywords match the config keys needs no override. Raise
|
|
85
|
+
``TypeError``/``ValueError`` for invalid options — the run-suite loop reports
|
|
86
|
+
it and skips the exporter rather than failing the run.
|
|
87
|
+
"""
|
|
88
|
+
return cls(**dict(config))
|
|
89
|
+
|
|
90
|
+
@abstractmethod
|
|
91
|
+
def export(
|
|
92
|
+
self,
|
|
93
|
+
result: EvalResult,
|
|
94
|
+
*,
|
|
95
|
+
gold_score: Optional[float] = None,
|
|
96
|
+
tenant_id: Optional[str] = None,
|
|
97
|
+
) -> Optional[str]:
|
|
98
|
+
"""Push one result downstream. Return an opaque id (e.g. a run id) or ``None``.
|
|
99
|
+
|
|
100
|
+
``tenant_id`` is the run's tenant (``EvalResult`` itself carries none) so the
|
|
101
|
+
exported record stays attributable in a multi-tenant deployment — log it as a
|
|
102
|
+
tag/label/column. ``gold_score`` is the period score when the caller computed
|
|
103
|
+
one (the ``run-suite`` path does not, so it is ``None`` there).
|
|
104
|
+
|
|
105
|
+
Must not raise for *expected* downstream conditions the caller can do nothing
|
|
106
|
+
about — but any exception is caught by the run-suite loop and logged, never
|
|
107
|
+
propagated into the gate verdict.
|
|
108
|
+
"""
|
|
109
|
+
raise NotImplementedError
|
|
110
|
+
|
|
111
|
+
def export_batch(
|
|
112
|
+
self,
|
|
113
|
+
results: Sequence[EvalResult],
|
|
114
|
+
*,
|
|
115
|
+
gold_score: Optional[float] = None,
|
|
116
|
+
tenant_id: Optional[str] = None,
|
|
117
|
+
) -> list[Optional[str]]:
|
|
118
|
+
"""Export each result in order. Override for a more efficient bulk path
|
|
119
|
+
(e.g. grouping the batch under one parent run)."""
|
|
120
|
+
return [
|
|
121
|
+
self.export(r, gold_score=gold_score, tenant_id=tenant_id) for r in results
|
|
122
|
+
]
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
__all__ = ["ResultExporter", "BaseExporter"]
|
sengol/adapters/jsonl.py
ADDED
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
"""JSONL file exporter — the lightweight, dependency-free built-in (and worked example).
|
|
2
|
+
|
|
3
|
+
Writes one JSON line per :class:`~sengol.core.types.EvalResult` to a local file. It needs
|
|
4
|
+
no extra dependency (unlike ``[mlflow]``), so it is the cheapest way to get results out of
|
|
5
|
+
Sengol — useful on its own and as the **reference implementation** a third party copies to
|
|
6
|
+
build their own exporter.
|
|
7
|
+
|
|
8
|
+
Unlike :class:`sengol.adapters.mlflow.MLflowExporter` (content-free by construction), this
|
|
9
|
+
exporter *can* emit free text — evaluator ``reason`` strings — which may carry regulated
|
|
10
|
+
content. It therefore demonstrates the **redaction duty of care**: pass ``redact: true`` (or
|
|
11
|
+
custom ``redact_patterns``) and every emitted ``reason`` is scrubbed through a
|
|
12
|
+
:class:`~sengol.capture.redaction.RegexRedactor` before it touches disk. Exporters that ship
|
|
13
|
+
content to a *third-party SaaS* should always redact.
|
|
14
|
+
|
|
15
|
+
Free / Apache 2.0.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import json
|
|
21
|
+
import os
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
from typing import Any, ClassVar, Mapping, Optional
|
|
24
|
+
|
|
25
|
+
from sengol.adapters.base import BaseExporter
|
|
26
|
+
from sengol.adapters.registry import register_exporter
|
|
27
|
+
from sengol.capture.redaction import NoOpRedactor, Redactor, RegexRedactor
|
|
28
|
+
from sengol.core.types import EvalResult
|
|
29
|
+
|
|
30
|
+
# Default PII patterns applied when ``redact: true`` is set without custom patterns.
|
|
31
|
+
# Conservative, illustrative set — emails and US-style SSNs. Operators with stricter
|
|
32
|
+
# needs pass their own ``redact_patterns``.
|
|
33
|
+
_DEFAULT_PII_PATTERNS = [
|
|
34
|
+
r"[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}", # email
|
|
35
|
+
r"\b\d{3}-\d{2}-\d{4}\b", # SSN
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@register_exporter
|
|
40
|
+
class JsonlFileExporter(BaseExporter):
|
|
41
|
+
"""Append each result as a JSON line to ``{path}``.
|
|
42
|
+
|
|
43
|
+
Parameters
|
|
44
|
+
----------
|
|
45
|
+
path:
|
|
46
|
+
Output file. Parent directories are created. Lines are appended, so reruns
|
|
47
|
+
accumulate (rotate/clear it yourself if you want a fresh file per run).
|
|
48
|
+
include_reasons:
|
|
49
|
+
When ``True`` (default), each score's ``reason`` is included. Set ``False`` to
|
|
50
|
+
emit only structured fields (ids, pass/fail) — the safest, content-free shape.
|
|
51
|
+
redactor:
|
|
52
|
+
A :class:`~sengol.capture.redaction.Redactor` applied to every emitted ``reason``.
|
|
53
|
+
Defaults to :class:`NoOpRedactor`. Build one from config with ``redact`` /
|
|
54
|
+
``redact_patterns`` (see :meth:`from_config`).
|
|
55
|
+
"""
|
|
56
|
+
|
|
57
|
+
name: ClassVar[str] = "jsonl"
|
|
58
|
+
|
|
59
|
+
@classmethod
|
|
60
|
+
def from_config(cls, config: Mapping[str, Any]) -> "JsonlFileExporter":
|
|
61
|
+
"""Build from a ``sengol.yaml`` ``exporters.jsonl`` mapping.
|
|
62
|
+
|
|
63
|
+
Keys: ``path`` (required), ``include_reasons`` (bool, default True),
|
|
64
|
+
``redact`` (bool — turn on the default PII patterns) and/or ``redact_patterns``
|
|
65
|
+
(list[str] — custom regexes; implies redaction). Any pattern list takes
|
|
66
|
+
precedence over the default set.
|
|
67
|
+
"""
|
|
68
|
+
path = config.get("path")
|
|
69
|
+
if not path:
|
|
70
|
+
raise ValueError("jsonl exporter requires a `path` option")
|
|
71
|
+
patterns = config.get("redact_patterns")
|
|
72
|
+
redactor: Redactor
|
|
73
|
+
if patterns:
|
|
74
|
+
redactor = RegexRedactor(list(patterns))
|
|
75
|
+
elif config.get("redact"):
|
|
76
|
+
redactor = RegexRedactor(_DEFAULT_PII_PATTERNS)
|
|
77
|
+
else:
|
|
78
|
+
redactor = NoOpRedactor()
|
|
79
|
+
return cls(
|
|
80
|
+
path=path,
|
|
81
|
+
include_reasons=bool(config.get("include_reasons", True)),
|
|
82
|
+
redactor=redactor,
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
def __init__(
|
|
86
|
+
self,
|
|
87
|
+
*,
|
|
88
|
+
path: str | os.PathLike[str],
|
|
89
|
+
include_reasons: bool = True,
|
|
90
|
+
redactor: Optional[Redactor] = None,
|
|
91
|
+
) -> None:
|
|
92
|
+
self._path = Path(path)
|
|
93
|
+
self._include_reasons = include_reasons
|
|
94
|
+
self._redactor: Redactor = redactor if redactor is not None else NoOpRedactor()
|
|
95
|
+
|
|
96
|
+
def export(
|
|
97
|
+
self,
|
|
98
|
+
result: EvalResult,
|
|
99
|
+
*,
|
|
100
|
+
gold_score: Optional[float] = None,
|
|
101
|
+
tenant_id: Optional[str] = None,
|
|
102
|
+
) -> str:
|
|
103
|
+
"""Append one JSON line for ``result``; return the path written to."""
|
|
104
|
+
self._path.parent.mkdir(parents=True, exist_ok=True)
|
|
105
|
+
line = json.dumps(self._row(result, gold_score, tenant_id), sort_keys=True)
|
|
106
|
+
with self._path.open("a", encoding="utf-8") as fh:
|
|
107
|
+
fh.write(line + "\n")
|
|
108
|
+
return str(self._path)
|
|
109
|
+
|
|
110
|
+
def _row(
|
|
111
|
+
self,
|
|
112
|
+
result: EvalResult,
|
|
113
|
+
gold_score: Optional[float],
|
|
114
|
+
tenant_id: Optional[str],
|
|
115
|
+
) -> dict[str, Any]:
|
|
116
|
+
scores = []
|
|
117
|
+
for s in result.scores:
|
|
118
|
+
entry: dict[str, Any] = {"evaluator": s.evaluator, "passed": s.passed}
|
|
119
|
+
if s.failure_mode is not None:
|
|
120
|
+
entry["failure_mode"] = s.failure_mode
|
|
121
|
+
if self._include_reasons and s.reason:
|
|
122
|
+
# The one free-text field — scrub it before it lands on disk.
|
|
123
|
+
entry["reason"] = self._redactor.redact(s.reason)
|
|
124
|
+
scores.append(entry)
|
|
125
|
+
row: dict[str, Any] = {
|
|
126
|
+
"result_id": str(result.result_id),
|
|
127
|
+
"agent_id": result.agent_id,
|
|
128
|
+
"agent_version": result.agent_version,
|
|
129
|
+
"tenant_id": tenant_id,
|
|
130
|
+
"overall_passed": result.overall_passed,
|
|
131
|
+
"policies": list(result.policies),
|
|
132
|
+
"scores": scores,
|
|
133
|
+
}
|
|
134
|
+
if gold_score is not None:
|
|
135
|
+
row["gold_score"] = gold_score
|
|
136
|
+
return row
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
__all__ = ["JsonlFileExporter"]
|
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
"""MLflow backend adapter — export Sengol eval results to MLflow tracking.
|
|
2
|
+
|
|
3
|
+
MLflow is reachable two ways in Sengol: implicitly over OTLP (ADR-0059, Session 5c)
|
|
4
|
+
when a collector fans spans out to it, and explicitly through *this* adapter. This is
|
|
5
|
+
the explicit path — a deliberate ``adapter.export(result)`` call that logs an MLflow run
|
|
6
|
+
with the eval's params, metrics, and tags, for teams that track model/agent quality in
|
|
7
|
+
MLflow experiments alongside their training runs.
|
|
8
|
+
|
|
9
|
+
The adapter is an **optional export target**, never on the agent code path and never the
|
|
10
|
+
source of truth — the Postgres ``AuditStore`` always is. ``mlflow`` is a soft import
|
|
11
|
+
(``pip install 'sengol[mlflow]'`` / ``uv sync --extra mlflow``): importing this module
|
|
12
|
+
costs nothing, and the SDK is only required when an adapter is actually constructed.
|
|
13
|
+
|
|
14
|
+
**Authentication.** This adapter brokers no credentials. ``tracking_uri`` is the endpoint
|
|
15
|
+
only; auth is handled by the mlflow SDK's own environment-variable mechanism in the
|
|
16
|
+
runtime — ``MLFLOW_TRACKING_USERNAME``/``MLFLOW_TRACKING_PASSWORD`` (basic),
|
|
17
|
+
``MLFLOW_TRACKING_TOKEN`` (bearer), or ``DATABRICKS_HOST``/``DATABRICKS_TOKEN``. Keep
|
|
18
|
+
secrets in the environment, not in ``sengol.yaml`` (see ``docs/reference/env.md``).
|
|
19
|
+
|
|
20
|
+
Free / Apache 2.0 — no enterprise gate on this code path.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import re
|
|
26
|
+
from typing import Any, ClassVar, Mapping, Optional
|
|
27
|
+
|
|
28
|
+
from sengol.adapters.base import BaseExporter
|
|
29
|
+
from sengol.adapters.registry import register_exporter
|
|
30
|
+
from sengol.core.types import EvalResult
|
|
31
|
+
|
|
32
|
+
# Message shown when the adapter is used without the optional dependency. Mirrors
|
|
33
|
+
# the soft-import contract used by the ``otlp`` and ``s3`` extras.
|
|
34
|
+
_MISSING_MLFLOW = (
|
|
35
|
+
"MLflowExporter requires the 'mlflow' optional dependency. "
|
|
36
|
+
"Install it with `pip install 'sengol[mlflow]'` or `uv sync --extra mlflow`."
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _import_mlflow() -> Any:
|
|
41
|
+
try:
|
|
42
|
+
import mlflow # type: ignore
|
|
43
|
+
except ImportError as exc: # pragma: no cover - exercised via injected module
|
|
44
|
+
raise ImportError(_MISSING_MLFLOW) from exc
|
|
45
|
+
return mlflow
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _pass_rate(result: EvalResult) -> float:
|
|
49
|
+
"""Fraction of this result's scores that passed. No scores → 0.0."""
|
|
50
|
+
if not result.scores:
|
|
51
|
+
return 0.0
|
|
52
|
+
return sum(1 for s in result.scores if s.passed) / len(result.scores)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
# MLflow metric keys allow only these characters; anything else raises on log.
|
|
56
|
+
_METRIC_KEY_OK = re.compile(r"[^A-Za-z0-9_\-./ ]")
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _sanitize_metric_key(name: str) -> str:
|
|
60
|
+
"""Replace characters MLflow forbids in a metric key with ``_``.
|
|
61
|
+
|
|
62
|
+
Keeps the key stable and unique-enough for the common case (most evaluator
|
|
63
|
+
names are already valid); a name that is *only* invalid chars collapses to
|
|
64
|
+
underscores, which is acceptable for a derived per-evaluator metric.
|
|
65
|
+
"""
|
|
66
|
+
return _METRIC_KEY_OK.sub("_", name)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
@register_exporter
|
|
70
|
+
class MLflowExporter(BaseExporter):
|
|
71
|
+
"""Export :class:`EvalResult` batches to an MLflow experiment.
|
|
72
|
+
|
|
73
|
+
Registered as the built-in ``mlflow`` exporter — select it via the ``sengol.yaml``
|
|
74
|
+
``exporters:`` block or ``sengol run-suite --export mlflow``.
|
|
75
|
+
|
|
76
|
+
**Content-free by construction.** This exporter logs only *structured, non-content*
|
|
77
|
+
fields — ids, counts, pass/fail metrics, policy ids. It never logs prompts, responses,
|
|
78
|
+
or evaluator ``reason`` free text, so it cannot leak regulated content to an external
|
|
79
|
+
tracking server. An exporter that *does* need to emit free text must scrub it first
|
|
80
|
+
(see :class:`sengol.adapters.jsonl.JsonlFileExporter` and its ``redact`` option for the
|
|
81
|
+
reference pattern).
|
|
82
|
+
|
|
83
|
+
Parameters
|
|
84
|
+
----------
|
|
85
|
+
experiment:
|
|
86
|
+
MLflow experiment name. Created on first use by ``set_experiment``.
|
|
87
|
+
tracking_uri:
|
|
88
|
+
Optional MLflow tracking URI (``http://…``, ``databricks``, a local
|
|
89
|
+
path, …). When omitted, MLflow's own default/config resolution applies.
|
|
90
|
+
Authentication is handled by the mlflow SDK's environment variables
|
|
91
|
+
(``MLFLOW_TRACKING_*`` / ``DATABRICKS_*``) — never passed here.
|
|
92
|
+
mlflow_module:
|
|
93
|
+
Injection seam for tests — pass a stand-in exposing
|
|
94
|
+
``set_tracking_uri`` / ``set_experiment`` / ``start_run`` /
|
|
95
|
+
``log_params`` / ``log_metrics`` / ``set_tags``. Production callers
|
|
96
|
+
leave this ``None`` and the real ``mlflow`` SDK is soft-imported.
|
|
97
|
+
"""
|
|
98
|
+
|
|
99
|
+
name: ClassVar[str] = "mlflow"
|
|
100
|
+
|
|
101
|
+
@classmethod
|
|
102
|
+
def from_config(cls, config: Mapping[str, Any]) -> "MLflowExporter":
|
|
103
|
+
"""Build from a ``sengol.yaml`` ``exporters.mlflow`` mapping.
|
|
104
|
+
|
|
105
|
+
Recognized keys: ``experiment``, ``tracking_uri``. The ``mlflow_module``
|
|
106
|
+
test seam is intentionally not configurable from YAML.
|
|
107
|
+
"""
|
|
108
|
+
return cls(
|
|
109
|
+
experiment=config.get("experiment", "sengol"),
|
|
110
|
+
tracking_uri=config.get("tracking_uri"),
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
def __init__(
|
|
114
|
+
self,
|
|
115
|
+
*,
|
|
116
|
+
experiment: str = "sengol",
|
|
117
|
+
tracking_uri: Optional[str] = None,
|
|
118
|
+
mlflow_module: Any = None,
|
|
119
|
+
) -> None:
|
|
120
|
+
self._mlflow = mlflow_module if mlflow_module is not None else _import_mlflow()
|
|
121
|
+
if tracking_uri:
|
|
122
|
+
self._mlflow.set_tracking_uri(tracking_uri)
|
|
123
|
+
self._experiment = experiment
|
|
124
|
+
self._mlflow.set_experiment(experiment)
|
|
125
|
+
|
|
126
|
+
def export(
|
|
127
|
+
self,
|
|
128
|
+
result: EvalResult,
|
|
129
|
+
*,
|
|
130
|
+
gold_score: Optional[float] = None,
|
|
131
|
+
tenant_id: Optional[str] = None,
|
|
132
|
+
run_name: Optional[str] = None,
|
|
133
|
+
nested: bool = False,
|
|
134
|
+
) -> str:
|
|
135
|
+
"""Log one MLflow run for ``result`` and return its ``run_id``.
|
|
136
|
+
|
|
137
|
+
Params capture the run's identity (agent, version, policies, tenant); metrics
|
|
138
|
+
capture the verdict (pass-rate, per-evaluator pass/fail, latency/token/cost
|
|
139
|
+
rollups, optional ``gold_score``); tags carry cross-references back to the Sengol
|
|
140
|
+
record. ``gold_score`` is logged only when supplied — the ``run-suite`` path does
|
|
141
|
+
not compute one. ``nested`` opens the run under the active parent run (used by
|
|
142
|
+
:meth:`export_batch`).
|
|
143
|
+
"""
|
|
144
|
+
run_ctx = self._mlflow.start_run(
|
|
145
|
+
run_name=run_name or result.agent_id, nested=nested
|
|
146
|
+
)
|
|
147
|
+
with run_ctx as run:
|
|
148
|
+
self._mlflow.log_params(self._params(result, tenant_id))
|
|
149
|
+
self._mlflow.log_metrics(self._metrics(result, gold_score))
|
|
150
|
+
self._mlflow.set_tags(self._tags(result, tenant_id))
|
|
151
|
+
return run.info.run_id
|
|
152
|
+
|
|
153
|
+
def export_batch(
|
|
154
|
+
self,
|
|
155
|
+
results,
|
|
156
|
+
*,
|
|
157
|
+
gold_score: Optional[float] = None,
|
|
158
|
+
tenant_id: Optional[str] = None,
|
|
159
|
+
) -> list[Optional[str]]:
|
|
160
|
+
"""Group the batch under one parent MLflow run, each result a nested child.
|
|
161
|
+
|
|
162
|
+
Keeps a multi-input suite legible in the MLflow UI: one parent run per
|
|
163
|
+
``export_batch`` call, ``len(results)`` nested children. Empty input → no run.
|
|
164
|
+
"""
|
|
165
|
+
results = list(results)
|
|
166
|
+
if not results:
|
|
167
|
+
return []
|
|
168
|
+
parent_name = f"sengol-suite-{results[0].agent_id}"
|
|
169
|
+
with self._mlflow.start_run(run_name=parent_name):
|
|
170
|
+
if tenant_id is not None:
|
|
171
|
+
self._mlflow.set_tags({"sengol.tenant_id": tenant_id})
|
|
172
|
+
return [
|
|
173
|
+
self.export(r, gold_score=gold_score, tenant_id=tenant_id, nested=True)
|
|
174
|
+
for r in results
|
|
175
|
+
]
|
|
176
|
+
|
|
177
|
+
# ------------------------------------------------------------------ #
|
|
178
|
+
# Field mapping (pure — no SDK calls, so it is unit-testable directly)
|
|
179
|
+
# ------------------------------------------------------------------ #
|
|
180
|
+
|
|
181
|
+
@staticmethod
|
|
182
|
+
def _params(result: EvalResult, tenant_id: Optional[str]) -> dict[str, Any]:
|
|
183
|
+
params: dict[str, Any] = {
|
|
184
|
+
"agent_id": result.agent_id,
|
|
185
|
+
"agent_version": result.agent_version,
|
|
186
|
+
"policies": ",".join(result.policies),
|
|
187
|
+
"result_id": str(result.result_id),
|
|
188
|
+
"evaluator_count": len(result.scores),
|
|
189
|
+
"saturation_warning": result.saturation_warning,
|
|
190
|
+
}
|
|
191
|
+
if tenant_id is not None:
|
|
192
|
+
params["tenant_id"] = tenant_id
|
|
193
|
+
return params
|
|
194
|
+
|
|
195
|
+
@staticmethod
|
|
196
|
+
def _metrics(result: EvalResult, gold_score: Optional[float]) -> dict[str, float]:
|
|
197
|
+
metrics: dict[str, float] = {
|
|
198
|
+
"pass_rate": _pass_rate(result),
|
|
199
|
+
"overall_passed": 1.0 if result.overall_passed else 0.0,
|
|
200
|
+
"total_latency_ms": sum(s.latency_ms for s in result.scores),
|
|
201
|
+
"total_tokens": float(sum(s.tokens_used for s in result.scores)),
|
|
202
|
+
"total_cost_usd": sum(s.cost_usd for s in result.scores),
|
|
203
|
+
}
|
|
204
|
+
for s in result.scores:
|
|
205
|
+
# MLflow metric keys allow only alnum + _-./ space. The evaluator
|
|
206
|
+
# registry permits any non-empty name, so a third-party evaluator may
|
|
207
|
+
# carry chars MLflow rejects (e.g. ':') — sanitize so one such name
|
|
208
|
+
# can't raise and abort the whole export.
|
|
209
|
+
metrics[f"eval.{_sanitize_metric_key(s.evaluator)}.passed"] = (
|
|
210
|
+
1.0 if s.passed else 0.0
|
|
211
|
+
)
|
|
212
|
+
if gold_score is not None:
|
|
213
|
+
metrics["gold_score"] = gold_score
|
|
214
|
+
return metrics
|
|
215
|
+
|
|
216
|
+
@staticmethod
|
|
217
|
+
def _tags(result: EvalResult, tenant_id: Optional[str]) -> dict[str, str]:
|
|
218
|
+
tags = {
|
|
219
|
+
"sengol.source": "sengol",
|
|
220
|
+
"sengol.result_id": str(result.result_id),
|
|
221
|
+
"sengol.agent_id": result.agent_id,
|
|
222
|
+
"sengol.agent_version": result.agent_version,
|
|
223
|
+
}
|
|
224
|
+
if tenant_id is not None:
|
|
225
|
+
tags["sengol.tenant_id"] = tenant_id
|
|
226
|
+
return tags
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
__all__ = ["MLflowExporter"]
|