baseltest 0.22.0__py3-none-win_amd64.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- baseltest/__init__.py +36 -0
- baseltest/_version.py +11 -0
- baseltest/baseline/__init__.py +46 -0
- baseltest/baseline/reader.py +290 -0
- baseltest/baseline/record.py +239 -0
- baseltest/baseline/writer.py +252 -0
- baseltest/contract/__init__.py +126 -0
- baseltest/contract/errors.py +21 -0
- baseltest/contract/evaluation.py +419 -0
- baseltest/contract/model.py +489 -0
- baseltest/contract/postconditions.py +530 -0
- baseltest/contract/reply.py +37 -0
- baseltest/declarative/__init__.py +66 -0
- baseltest/declarative/_cli.py +685 -0
- baseltest/declarative/_errors.py +13 -0
- baseltest/declarative/_instantiate/__init__.py +34 -0
- baseltest/declarative/_instantiate/_baseline.py +185 -0
- baseltest/declarative/_instantiate/_compose.py +158 -0
- baseltest/declarative/_instantiate/_explore.py +147 -0
- baseltest/declarative/_instantiate/_latency.py +108 -0
- baseltest/declarative/_instantiate/_optimize_point.py +122 -0
- baseltest/declarative/_instantiate/_postconditions.py +279 -0
- baseltest/declarative/_instantiate/_service.py +129 -0
- baseltest/declarative/_instantiate/_sizing_policy.py +90 -0
- baseltest/declarative/_instantiate/_views.py +79 -0
- baseltest/declarative/_materialise.py +191 -0
- baseltest/declarative/_optimize.py +334 -0
- baseltest/declarative/_parser/__init__.py +40 -0
- baseltest/declarative/_parser/_contract.py +160 -0
- baseltest/declarative/_parser/_criteria.py +162 -0
- baseltest/declarative/_parser/_forms.py +308 -0
- baseltest/declarative/_parser/_inputs.py +173 -0
- baseltest/declarative/_parser/_latency.py +112 -0
- baseltest/declarative/_parser/_model.py +128 -0
- baseltest/declarative/_parser/_shape.py +47 -0
- baseltest/declarative/_parser/_structure.py +89 -0
- baseltest/declarative/_providers/__init__.py +296 -0
- baseltest/declarative/_providers/_anthropic.py +143 -0
- baseltest/declarative/_providers/_apertus.py +34 -0
- baseltest/declarative/_providers/_litellm.py +94 -0
- baseltest/declarative/_providers/_media.py +93 -0
- baseltest/declarative/_providers/_mistral.py +28 -0
- baseltest/declarative/_providers/_ollama.py +79 -0
- baseltest/declarative/_providers/_openai.py +45 -0
- baseltest/declarative/_providers/_protocol.py +234 -0
- baseltest/declarative/_registrations.py +62 -0
- baseltest/declarative/_registry/__init__.py +31 -0
- baseltest/declarative/_registry/_bindings.py +102 -0
- baseltest/declarative/_registry/_core.py +355 -0
- baseltest/declarative/_registry/_guards.py +56 -0
- baseltest/declarative/_registry/_service_types.py +165 -0
- baseltest/declarative/_registry/_transform.py +82 -0
- baseltest/declarative/_report.py +142 -0
- baseltest/declarative/_roots.py +167 -0
- baseltest/declarative/_runner/__init__.py +38 -0
- baseltest/declarative/_runner/_check.py +146 -0
- baseltest/declarative/_runner/_explore.py +231 -0
- baseltest/declarative/_runner/_load.py +45 -0
- baseltest/declarative/_runner/_optimize.py +120 -0
- baseltest/declarative/_runner/_optimize_loop.py +297 -0
- baseltest/declarative/_runner/_run.py +215 -0
- baseltest/declarative/_runner/_shared.py +46 -0
- baseltest/declarative/_schema_walk.py +355 -0
- baseltest/declarative/_services/__init__.py +49 -0
- baseltest/declarative/_services/_language_model.py +368 -0
- baseltest/declarative/_services/_model.py +97 -0
- baseltest/declarative/_services/_parse.py +343 -0
- baseltest/declarative/_signatures.py +42 -0
- baseltest/declarative/_sizing/__init__.py +34 -0
- baseltest/declarative/_sizing/_criteria.py +117 -0
- baseltest/declarative/_sizing/_flags.py +79 -0
- baseltest/declarative/_sizing/_model.py +62 -0
- baseltest/declarative/_sizing/_modes.py +195 -0
- baseltest/declarative/_sizing/_pricing.py +66 -0
- baseltest/declarative/_sizing/_prompts.py +92 -0
- baseltest/declarative/_sizing/_rates.py +43 -0
- baseltest/declarative/_sizing/_render.py +108 -0
- baseltest/declarative/_sizing/_resolve.py +161 -0
- baseltest/declarative/_steppers/__init__.py +74 -0
- baseltest/declarative/_steppers/_builtins.py +34 -0
- baseltest/declarative/_steppers/_context.py +114 -0
- baseltest/declarative/_steppers/_contract.py +146 -0
- baseltest/declarative/_steppers/_linear_sweep.py +43 -0
- baseltest/declarative/_steppers/_numeric.py +30 -0
- baseltest/declarative/_steppers/_prompt_engineer.py +117 -0
- baseltest/declarative/_steppers/_refining_grid.py +240 -0
- baseltest/declarative/_structured.py +352 -0
- baseltest/declarative/_types.py +108 -0
- baseltest/engine/__init__.py +65 -0
- baseltest/engine/artefact.py +68 -0
- baseltest/engine/defect.py +68 -0
- baseltest/engine/latency.py +226 -0
- baseltest/engine/naming.py +62 -0
- baseltest/engine/run/__init__.py +36 -0
- baseltest/engine/run/attainment.py +48 -0
- baseltest/engine/run/execute.py +319 -0
- baseltest/engine/run/feasibility.py +77 -0
- baseltest/engine/run/identity.py +44 -0
- baseltest/engine/run/judge.py +29 -0
- baseltest/engine/run/model.py +210 -0
- baseltest/engine/run/sample.py +218 -0
- baseltest/exploration/__init__.py +23 -0
- baseltest/exploration/writer.py +191 -0
- baseltest/observation/__init__.py +20 -0
- baseltest/observation/emit.py +162 -0
- baseltest/observation/record.py +261 -0
- baseltest/optimization/__init__.py +29 -0
- baseltest/optimization/record.py +114 -0
- baseltest/optimization/writer.py +99 -0
- baseltest/py.typed +0 -0
- baseltest/reporting/__init__.py +53 -0
- baseltest/reporting/console.py +390 -0
- baseltest/reporting/run_design.py +86 -0
- baseltest/reporting/verdict_reader.py +253 -0
- baseltest/reporting/verdict_xml.py +266 -0
- baseltest/statistics/__init__.py +94 -0
- baseltest/statistics/_constants.py +19 -0
- baseltest/statistics/_validation.py +32 -0
- baseltest/statistics/feasibility.py +93 -0
- baseltest/statistics/latency.py +166 -0
- baseltest/statistics/power.py +111 -0
- baseltest/statistics/proportion.py +33 -0
- baseltest/statistics/sizing.py +210 -0
- baseltest/statistics/summary.py +74 -0
- baseltest/statistics/threshold.py +269 -0
- baseltest/statistics/verdict.py +168 -0
- baseltest/statistics/wilson.py +168 -0
- baseltest-0.22.0.data/scripts/mavai +0 -0
- baseltest-0.22.0.dist-info/METADATA +142 -0
- baseltest-0.22.0.dist-info/RECORD +134 -0
- baseltest-0.22.0.dist-info/WHEEL +4 -0
- baseltest-0.22.0.dist-info/entry_points.txt +2 -0
- baseltest-0.22.0.dist-info/licenses/LICENSE +202 -0
- baseltest-0.22.0.dist-info/licenses/NOTICE +11 -0
baseltest/__init__.py
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""baseltest: probabilistic testing for stochastic services.
|
|
2
|
+
|
|
3
|
+
Python-native counterpart to punit (Java) and feotest (Rust) in the
|
|
4
|
+
mavai framework family — statistical inference over repeated samples,
|
|
5
|
+
not a single pass/fail assertion.
|
|
6
|
+
|
|
7
|
+
The common entry points are re-exported here: construct a :class:`Bindings`,
|
|
8
|
+
then call :func:`run`, :func:`explore`, :func:`optimize`, or
|
|
9
|
+
:func:`check_contract` on a contract file. These are the declarative
|
|
10
|
+
authoring surface (:mod:`baseltest.declarative`), promoted to the package
|
|
11
|
+
root for convenience; that module also carries the narrower stepper/scorer
|
|
12
|
+
context types, and :mod:`baseltest.contract` is the surface for authoring a
|
|
13
|
+
service contract in Python directly.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from baseltest._version import __version__
|
|
17
|
+
from baseltest.contract import FileInput, MessageParts, Reply
|
|
18
|
+
from baseltest.declarative import (
|
|
19
|
+
Bindings,
|
|
20
|
+
check_contract,
|
|
21
|
+
explore,
|
|
22
|
+
optimize,
|
|
23
|
+
run,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
__all__ = [
|
|
27
|
+
"Bindings",
|
|
28
|
+
"FileInput",
|
|
29
|
+
"MessageParts",
|
|
30
|
+
"Reply",
|
|
31
|
+
"__version__",
|
|
32
|
+
"check_contract",
|
|
33
|
+
"explore",
|
|
34
|
+
"optimize",
|
|
35
|
+
"run",
|
|
36
|
+
]
|
baseltest/_version.py
ADDED
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
"""Single source of the package version, importable without the API surface.
|
|
2
|
+
|
|
3
|
+
Lives apart from the package ``__init__`` so a lower layer (the reporting
|
|
4
|
+
renderers stamp the version into the verdict record) can read the version
|
|
5
|
+
without importing the top-level authoring surface that ``__init__``
|
|
6
|
+
re-exports — which would make a lower layer depend on a higher one.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from importlib import metadata
|
|
10
|
+
|
|
11
|
+
__version__ = metadata.version("baseltest")
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
"""The baseline artefact: the durable record of a measurement run.
|
|
2
|
+
|
|
3
|
+
A measurement persists what was observed -- per-criterion counts and rates,
|
|
4
|
+
identity, and provenance -- as a YAML artefact other tooling can later read.
|
|
5
|
+
This package owns the artefact's schema and is its **single writer**: the
|
|
6
|
+
serialisation lives here and nowhere else. The writer accepts
|
|
7
|
+
characterisation data as input, so any caller that has measured something
|
|
8
|
+
-- this framework's own measure runs, or an external characteriser -- hands
|
|
9
|
+
its data over rather than emitting the format itself.
|
|
10
|
+
|
|
11
|
+
Nothing in this package reads baselines back for threshold derivation; the
|
|
12
|
+
artefact is a durable record whose consumption is a deliberately separate,
|
|
13
|
+
later capability.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from .reader import (
|
|
17
|
+
BaselineResolution,
|
|
18
|
+
StoredBaseline,
|
|
19
|
+
StoredCriterion,
|
|
20
|
+
StoredLatency,
|
|
21
|
+
read_baseline,
|
|
22
|
+
resolve_baseline,
|
|
23
|
+
)
|
|
24
|
+
from .record import (
|
|
25
|
+
BaselineRecord,
|
|
26
|
+
CriterionCharacterisation,
|
|
27
|
+
JudgementState,
|
|
28
|
+
NormativeJudgement,
|
|
29
|
+
)
|
|
30
|
+
from .writer import baseline_filename, render_baseline, write_baseline
|
|
31
|
+
|
|
32
|
+
__all__ = [
|
|
33
|
+
"BaselineRecord",
|
|
34
|
+
"BaselineResolution",
|
|
35
|
+
"CriterionCharacterisation",
|
|
36
|
+
"JudgementState",
|
|
37
|
+
"NormativeJudgement",
|
|
38
|
+
"StoredBaseline",
|
|
39
|
+
"StoredCriterion",
|
|
40
|
+
"StoredLatency",
|
|
41
|
+
"baseline_filename",
|
|
42
|
+
"read_baseline",
|
|
43
|
+
"render_baseline",
|
|
44
|
+
"resolve_baseline",
|
|
45
|
+
"write_baseline",
|
|
46
|
+
]
|
|
@@ -0,0 +1,290 @@
|
|
|
1
|
+
"""Read-side of the baseline artefact: parsing and resolution.
|
|
2
|
+
|
|
3
|
+
The single-writer rule is untouched — reading is not writing. Because the
|
|
4
|
+
artefact is emitted by exactly one writer (:mod:`.writer`), the parser here
|
|
5
|
+
accepts precisely that emission grammar: two-space indentation, one
|
|
6
|
+
``key: value`` per line or one ``- item`` per line under a list key, every
|
|
7
|
+
string JSON-quoted. No third-party dependency; the scalars are JSON, the
|
|
8
|
+
structure is indentation.
|
|
9
|
+
|
|
10
|
+
Both artefact generations read back: ``baseltest-baseline-2`` (current)
|
|
11
|
+
and ``baseltest-baseline-1`` (no ``latency:`` block — a version-1 artefact
|
|
12
|
+
simply characterises the functional dimension only).
|
|
13
|
+
|
|
14
|
+
Resolution is strict identity of what was measured: same contract, same
|
|
15
|
+
inputs fingerprint, same covariates (the recorded provenance, minus the
|
|
16
|
+
volatile keys). A near-miss is reported with the reason it did not match —
|
|
17
|
+
a config drift must never silently downgrade a judged criterion.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
import hashlib
|
|
21
|
+
import json
|
|
22
|
+
from dataclasses import dataclass, field
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
from typing import Any
|
|
25
|
+
|
|
26
|
+
from baseltest.engine import LatencyBasis
|
|
27
|
+
|
|
28
|
+
from .writer import FINGERPRINT_KEY, SCHEMA_VERSION, baseline_filename_for
|
|
29
|
+
|
|
30
|
+
_READABLE_SCHEMAS = frozenset({SCHEMA_VERSION})
|
|
31
|
+
|
|
32
|
+
# Provenance keys that legitimately differ between the measure run and a
|
|
33
|
+
# later test run: they identify the run, not the thing measured.
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass(frozen=True, slots=True)
|
|
37
|
+
class StoredCriterion:
|
|
38
|
+
"""One criterion's recorded evidence, as read back from the artefact."""
|
|
39
|
+
|
|
40
|
+
successes: int
|
|
41
|
+
trials: int
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@dataclass(frozen=True, slots=True)
|
|
45
|
+
class StoredLatency:
|
|
46
|
+
"""The artefact's latency block, read back for bound derivation.
|
|
47
|
+
|
|
48
|
+
The sorted vector is the payload a later test derives its bound from;
|
|
49
|
+
the percentiles are the measurement run's descriptive summary.
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
basis: LatencyBasis
|
|
53
|
+
contributing_samples: int
|
|
54
|
+
total_samples: int
|
|
55
|
+
percentiles: tuple[tuple[str, int], ...]
|
|
56
|
+
sorted_passing_latencies_ms: tuple[int, ...]
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@dataclass(frozen=True, slots=True)
|
|
60
|
+
class StoredBaseline:
|
|
61
|
+
"""The artefact's content, read back for resolution and judgement."""
|
|
62
|
+
|
|
63
|
+
path: Path
|
|
64
|
+
contract_id: str
|
|
65
|
+
service_name: str
|
|
66
|
+
sample_count: int
|
|
67
|
+
inputs_identity: str
|
|
68
|
+
generated_at: str
|
|
69
|
+
covariate_profile: dict[str, str]
|
|
70
|
+
factor_record: dict[str, str]
|
|
71
|
+
criteria: dict[str, StoredCriterion]
|
|
72
|
+
latency: StoredLatency | None = None
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@dataclass(frozen=True, slots=True)
|
|
76
|
+
class BaselineResolution:
|
|
77
|
+
"""The outcome of looking for a matching baseline.
|
|
78
|
+
|
|
79
|
+
Exactly one of ``baseline`` / ``reason`` is meaningful: a match carries
|
|
80
|
+
the stored baseline; a non-match carries the honest reason.
|
|
81
|
+
"""
|
|
82
|
+
|
|
83
|
+
baseline: StoredBaseline | None = None
|
|
84
|
+
reason: str | None = None
|
|
85
|
+
mismatched_keys: tuple[str, ...] = field(default=())
|
|
86
|
+
|
|
87
|
+
@property
|
|
88
|
+
def matched(self) -> bool:
|
|
89
|
+
return self.baseline is not None
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _parse_lines(lines: list[str]) -> dict[str, Any]:
|
|
93
|
+
"""Parse the writer's emission grammar into nested mappings and lists.
|
|
94
|
+
|
|
95
|
+
A ``key:`` line opens a nested container that starts as a mapping and
|
|
96
|
+
becomes a list on its first ``- item`` line — the writer only ever
|
|
97
|
+
emits homogeneous containers, so the switch is unambiguous.
|
|
98
|
+
"""
|
|
99
|
+
root: dict[str, Any] = {}
|
|
100
|
+
# (indent, container, parent, key-in-parent); the root has no parent.
|
|
101
|
+
stack: list[tuple[int, Any, dict[str, Any] | None, str | None]] = [(0, root, None, None)]
|
|
102
|
+
for raw in lines:
|
|
103
|
+
if not raw.strip():
|
|
104
|
+
continue
|
|
105
|
+
indent = len(raw) - len(raw.lstrip(" "))
|
|
106
|
+
if indent % 2 != 0:
|
|
107
|
+
raise ValueError(f"malformed indentation: {raw!r}")
|
|
108
|
+
line = raw.strip()
|
|
109
|
+
while stack and stack[-1][0] > indent:
|
|
110
|
+
stack.pop()
|
|
111
|
+
top_indent, container, parent, parent_key = stack[-1]
|
|
112
|
+
if line.startswith("- "):
|
|
113
|
+
if isinstance(container, dict):
|
|
114
|
+
if container or parent is None or parent_key is None:
|
|
115
|
+
raise ValueError(f"malformed list item: {raw!r}")
|
|
116
|
+
container = []
|
|
117
|
+
parent[parent_key] = container
|
|
118
|
+
stack[-1] = (top_indent, container, parent, parent_key)
|
|
119
|
+
container.append(json.loads(line[2:]))
|
|
120
|
+
elif isinstance(container, list):
|
|
121
|
+
raise ValueError(f"malformed line inside a list: {raw!r}")
|
|
122
|
+
else:
|
|
123
|
+
key, value_text = _split_entry(line)
|
|
124
|
+
if value_text is None:
|
|
125
|
+
child: dict[str, Any] = {}
|
|
126
|
+
container[key] = child
|
|
127
|
+
stack.append((indent + 2, child, container, key))
|
|
128
|
+
else:
|
|
129
|
+
container[key] = json.loads(value_text)
|
|
130
|
+
return root
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
_DECODER = json.JSONDecoder()
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _split_entry(line: str) -> tuple[str, str | None]:
|
|
137
|
+
"""One emitted mapping line into its key and value text.
|
|
138
|
+
|
|
139
|
+
Returns ``(key, None)`` for a block-opening ``key:`` line, otherwise
|
|
140
|
+
``(key, value_text)``. A JSON-quoted key is decoded, never searched
|
|
141
|
+
for a separator: the key itself may contain ``": "`` (a
|
|
142
|
+
failure-reason string quoting a regex, a covariate value).
|
|
143
|
+
"""
|
|
144
|
+
if line.startswith('"'):
|
|
145
|
+
try:
|
|
146
|
+
key, end = _DECODER.raw_decode(line)
|
|
147
|
+
except json.JSONDecodeError as error:
|
|
148
|
+
raise ValueError(f"malformed line: {line!r}") from error
|
|
149
|
+
rest = line[end:]
|
|
150
|
+
if rest == ":":
|
|
151
|
+
return str(key), None
|
|
152
|
+
if rest.startswith(": ") and rest[2:]:
|
|
153
|
+
return str(key), rest[2:]
|
|
154
|
+
raise ValueError(f"malformed line: {line!r}")
|
|
155
|
+
if line.endswith(":"):
|
|
156
|
+
return line[:-1], None
|
|
157
|
+
key, _, value_text = line.partition(": ")
|
|
158
|
+
if not value_text:
|
|
159
|
+
raise ValueError(f"malformed line: {line!r}")
|
|
160
|
+
return key, value_text
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _parse_latency(body: dict[str, Any] | None) -> StoredLatency | None:
|
|
164
|
+
if body is None:
|
|
165
|
+
return None
|
|
166
|
+
percentiles = tuple(
|
|
167
|
+
(key, int(value))
|
|
168
|
+
for key, value in body.items()
|
|
169
|
+
if key.startswith("p") and key.endswith("Ms")
|
|
170
|
+
)
|
|
171
|
+
vector = tuple(int(v) for v in body.get("sortedPassingLatenciesMs", []))
|
|
172
|
+
return StoredLatency(
|
|
173
|
+
basis=LatencyBasis(body["basis"]),
|
|
174
|
+
contributing_samples=int(body["contributingSamples"]),
|
|
175
|
+
total_samples=int(body["totalSamples"]),
|
|
176
|
+
percentiles=percentiles,
|
|
177
|
+
sorted_passing_latencies_ms=vector,
|
|
178
|
+
)
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def read_baseline(path: Path) -> StoredBaseline:
|
|
182
|
+
"""Read one artefact back.
|
|
183
|
+
|
|
184
|
+
Raises:
|
|
185
|
+
ValueError: The file is not a readable baseline artefact (unknown
|
|
186
|
+
schema generation, malformed emission).
|
|
187
|
+
OSError: The file cannot be read.
|
|
188
|
+
"""
|
|
189
|
+
text = path.read_text(encoding="utf-8")
|
|
190
|
+
data = _parse_lines(text.splitlines())
|
|
191
|
+
schema = data.get("schemaVersion")
|
|
192
|
+
if schema != SCHEMA_VERSION:
|
|
193
|
+
# The whole migration experience is this sentence: what was found,
|
|
194
|
+
# what is expected, and the verb that regenerates the artefact.
|
|
195
|
+
raise ValueError(
|
|
196
|
+
f"{path.name}: schema {schema!r} is not {SCHEMA_VERSION!r}. "
|
|
197
|
+
"Baselines written by an earlier baseltest are not read: re-run "
|
|
198
|
+
"`basel measure` against this contract to regenerate it."
|
|
199
|
+
)
|
|
200
|
+
_verify_fingerprint(path, text, data)
|
|
201
|
+
criteria: dict[str, StoredCriterion] = {}
|
|
202
|
+
for name, body in data.get("criteria", {}).items():
|
|
203
|
+
criteria[name] = StoredCriterion(
|
|
204
|
+
successes=int(body["successes"]), trials=int(body["trials"])
|
|
205
|
+
)
|
|
206
|
+
execution = data.get("execution", {})
|
|
207
|
+
return StoredBaseline(
|
|
208
|
+
path=path,
|
|
209
|
+
contract_id=str(data["serviceContractId"]),
|
|
210
|
+
service_name=str(data["serviceName"]),
|
|
211
|
+
sample_count=int(execution["samplesExecuted"]),
|
|
212
|
+
inputs_identity=str(data["inputsIdentity"]),
|
|
213
|
+
generated_at=str(data.get("generatedAt", "")),
|
|
214
|
+
covariate_profile={str(k): str(v) for k, v in data.get("covariateProfile", {}).items()},
|
|
215
|
+
factor_record={str(k): str(v) for k, v in data.get("factorRecord", {}).items()},
|
|
216
|
+
criteria=criteria,
|
|
217
|
+
latency=_parse_latency(data.get("latency")),
|
|
218
|
+
)
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def _verify_fingerprint(path: Path, text: str, data: dict[str, object]) -> None:
|
|
222
|
+
"""Refuse a record whose content does not match its stated fingerprint.
|
|
223
|
+
|
|
224
|
+
The fingerprint covers the document with its own line absent, which is
|
|
225
|
+
the emitted body up to that final line.
|
|
226
|
+
"""
|
|
227
|
+
stated = data.get(FINGERPRINT_KEY)
|
|
228
|
+
if not stated:
|
|
229
|
+
raise ValueError(f"{path.name}: no {FINGERPRINT_KEY} — the record cannot be trusted")
|
|
230
|
+
marker = f"\n{FINGERPRINT_KEY}:"
|
|
231
|
+
body = text[: text.rindex(marker) + 1] if marker in text else text
|
|
232
|
+
actual = "sha256:" + hashlib.sha256(body.encode("utf-8")).hexdigest()
|
|
233
|
+
if actual != stated:
|
|
234
|
+
raise ValueError(
|
|
235
|
+
f"{path.name}: {FINGERPRINT_KEY} does not match the record's content "
|
|
236
|
+
"— the file has been edited or truncated since it was written"
|
|
237
|
+
)
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def resolve_baseline(
|
|
241
|
+
baseline_dir: Path,
|
|
242
|
+
contract_id: str,
|
|
243
|
+
service_name: str,
|
|
244
|
+
inputs_identity: str,
|
|
245
|
+
covariate_profile: dict[str, str],
|
|
246
|
+
) -> BaselineResolution:
|
|
247
|
+
"""Find the baseline matching what this run would measure.
|
|
248
|
+
|
|
249
|
+
Matching is strict identity over the tuple (contract, service, inputs,
|
|
250
|
+
covariate profile). The deterministic filename locates the candidate and
|
|
251
|
+
the loaded record's own statement decides — the body is authoritative,
|
|
252
|
+
never the path. Any difference is a non-match, and the resolution says
|
|
253
|
+
which keys differed: a drifted configuration is surfaced, never silently
|
|
254
|
+
treated as "no baseline".
|
|
255
|
+
|
|
256
|
+
The covariate profile is compared as stated, rather than reconstructed
|
|
257
|
+
by subtracting volatile keys from provenance: identity and provenance
|
|
258
|
+
are separate fields in the artefact, so they no longer have to be
|
|
259
|
+
separated again on the way back in.
|
|
260
|
+
"""
|
|
261
|
+
candidate = baseline_dir / baseline_filename_for(contract_id, service_name, inputs_identity)
|
|
262
|
+
if not candidate.is_file():
|
|
263
|
+
return BaselineResolution(reason=f"no baseline found (expected {candidate.as_posix()})")
|
|
264
|
+
try:
|
|
265
|
+
stored = read_baseline(candidate)
|
|
266
|
+
except (ValueError, OSError, json.JSONDecodeError) as error:
|
|
267
|
+
return BaselineResolution(reason=f"baseline {candidate.name} is unreadable: {error}")
|
|
268
|
+
if stored.inputs_identity != inputs_identity:
|
|
269
|
+
return BaselineResolution(reason=f"baseline {candidate.name} records different inputs")
|
|
270
|
+
if stored.service_name != service_name:
|
|
271
|
+
return BaselineResolution(
|
|
272
|
+
reason=(
|
|
273
|
+
f"baseline {candidate.name} records service {stored.service_name!r}, "
|
|
274
|
+
f"not {service_name!r}"
|
|
275
|
+
)
|
|
276
|
+
)
|
|
277
|
+
theirs = dict(stored.covariate_profile)
|
|
278
|
+
ours = dict(covariate_profile)
|
|
279
|
+
if theirs != ours:
|
|
280
|
+
differing = sorted(
|
|
281
|
+
key for key in set(theirs) | set(ours) if theirs.get(key) != ours.get(key)
|
|
282
|
+
)
|
|
283
|
+
return BaselineResolution(
|
|
284
|
+
reason=(
|
|
285
|
+
f"baseline {candidate.name} was measured under a different "
|
|
286
|
+
f"configuration (differing: {', '.join(differing)})"
|
|
287
|
+
),
|
|
288
|
+
mismatched_keys=tuple(differing),
|
|
289
|
+
)
|
|
290
|
+
return BaselineResolution(baseline=stored)
|
|
@@ -0,0 +1,239 @@
|
|
|
1
|
+
"""The baseline record: what a measurement run durably states about a service."""
|
|
2
|
+
|
|
3
|
+
from collections.abc import Mapping
|
|
4
|
+
from dataclasses import dataclass, field
|
|
5
|
+
from datetime import datetime
|
|
6
|
+
from enum import StrEnum
|
|
7
|
+
from types import MappingProxyType
|
|
8
|
+
|
|
9
|
+
from baseltest.contract import DeliveryCause, FailureAxis, PostconditionStanding
|
|
10
|
+
from baseltest.engine import LatencyBlock, RunResult, latency_block
|
|
11
|
+
from baseltest.engine.naming import bounded_excerpt
|
|
12
|
+
from baseltest.statistics import DEFAULT_CONFIDENCE_LEVEL, wilson_lower_bound
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class CriterionMode(StrEnum):
|
|
16
|
+
"""Whether a criterion estimates a proportion (companion §1.5).
|
|
17
|
+
|
|
18
|
+
An observational criterion states no rate and no bound: it estimates no
|
|
19
|
+
proportion at all, passing iff no failure was observed.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
INFERENTIAL = "inferential"
|
|
23
|
+
OBSERVATIONAL = "observational"
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class CriterionProcedure(StrEnum):
|
|
27
|
+
"""The inferential procedure a criterion is judged under."""
|
|
28
|
+
|
|
29
|
+
REGRESSION = "REGRESSION"
|
|
30
|
+
COMPLIANCE = "COMPLIANCE"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class JudgementState(StrEnum):
|
|
34
|
+
"""A measurement-time normative judgement's outcome.
|
|
35
|
+
|
|
36
|
+
The schema also reserves ``unsupportable`` for callers whose sample size
|
|
37
|
+
was not validated up front; baseltest validates every run's size before
|
|
38
|
+
sampling, so it emits only ``met`` or ``failed``.
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
MET = "met"
|
|
42
|
+
FAILED = "failed"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@dataclass(frozen=True, slots=True)
|
|
46
|
+
class NormativeJudgement:
|
|
47
|
+
"""The measurement-time judgement of one criterion against its declared threshold.
|
|
48
|
+
|
|
49
|
+
Purely documentary: a later reader sees not only what was measured but
|
|
50
|
+
how the measurement stood relative to a bar in force at measurement
|
|
51
|
+
time. It never affects how the artefact is consumed.
|
|
52
|
+
|
|
53
|
+
Attributes:
|
|
54
|
+
state: The :class:`JudgementState` reached against the bar.
|
|
55
|
+
stipulated_threshold: The declared threshold judged against.
|
|
56
|
+
confidence: The confidence level of the judgement.
|
|
57
|
+
"""
|
|
58
|
+
|
|
59
|
+
state: JudgementState
|
|
60
|
+
stipulated_threshold: float
|
|
61
|
+
confidence: float
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
@dataclass(frozen=True, slots=True)
|
|
65
|
+
class CriterionCharacterisation:
|
|
66
|
+
"""One criterion's measured characterisation.
|
|
67
|
+
|
|
68
|
+
Attributes:
|
|
69
|
+
successes: Passing trials.
|
|
70
|
+
trials: Total trials.
|
|
71
|
+
failure_distribution: Failure reasons and their counts; empty when
|
|
72
|
+
every trial passed.
|
|
73
|
+
failure_axes: The companion's diagnostic axis per observed reason.
|
|
74
|
+
delivery_causes: The delivery cause per observed reason that was a
|
|
75
|
+
failed delivery — what the artefact states as that entry's
|
|
76
|
+
identity, since the reason itself is an interpolated message.
|
|
77
|
+
judgement: The measurement-time judgement, when the criterion
|
|
78
|
+
declared a threshold; ``None`` otherwise.
|
|
79
|
+
standings: The criterion's descriptive per-postcondition tally —
|
|
80
|
+
per ``(input, check)``, passed/failed/skipped counts and the
|
|
81
|
+
observed fraction, each row carrying its check's optional
|
|
82
|
+
flag. Triage data, additive in the artefact schema; never an
|
|
83
|
+
interval or a per-check verdict.
|
|
84
|
+
optional_slack: The criterion's declared optional-check failure
|
|
85
|
+
budget, verbatim as authored (``None`` when undeclared —
|
|
86
|
+
never ``"0"``). Additive in the artefact schema.
|
|
87
|
+
"""
|
|
88
|
+
|
|
89
|
+
successes: int
|
|
90
|
+
trials: int
|
|
91
|
+
failure_distribution: Mapping[str, int] = field(default_factory=dict)
|
|
92
|
+
failure_axes: Mapping[str, FailureAxis] = field(default_factory=dict)
|
|
93
|
+
delivery_causes: Mapping[str, DeliveryCause] = field(default_factory=dict)
|
|
94
|
+
judgement: NormativeJudgement | None = None
|
|
95
|
+
standings: tuple[PostconditionStanding, ...] = ()
|
|
96
|
+
optional_slack: str | None = None
|
|
97
|
+
mode: CriterionMode = CriterionMode.INFERENTIAL
|
|
98
|
+
procedure: CriterionProcedure | None = CriterionProcedure.REGRESSION
|
|
99
|
+
wilson_lower_bound: float | None = None
|
|
100
|
+
|
|
101
|
+
def __post_init__(self) -> None:
|
|
102
|
+
object.__setattr__(
|
|
103
|
+
self, "failure_distribution", MappingProxyType(dict(self.failure_distribution))
|
|
104
|
+
)
|
|
105
|
+
object.__setattr__(self, "failure_axes", MappingProxyType(dict(self.failure_axes)))
|
|
106
|
+
object.__setattr__(self, "delivery_causes", MappingProxyType(dict(self.delivery_causes)))
|
|
107
|
+
|
|
108
|
+
@property
|
|
109
|
+
def observed_rate(self) -> float:
|
|
110
|
+
"""The observed pass rate. A recorded characterisation has at least
|
|
111
|
+
one trial."""
|
|
112
|
+
return self.successes / self.trials
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
@dataclass(frozen=True, slots=True)
|
|
116
|
+
class BaselineRecord:
|
|
117
|
+
"""Everything the baseline artefact states.
|
|
118
|
+
|
|
119
|
+
Attributes:
|
|
120
|
+
service_contract_id: The measured service contract's identity.
|
|
121
|
+
service_name: The name of the service that was invoked. Identity,
|
|
122
|
+
not provenance: a contract may exercise several services and a
|
|
123
|
+
service may be the subject of several contracts, so neither
|
|
124
|
+
names a record alone.
|
|
125
|
+
generated_at: Measurement time, UTC.
|
|
126
|
+
confidence_level: The level every ``wilson_lower_bound`` in this
|
|
127
|
+
record was computed at. Distinct from a criterion's stipulated
|
|
128
|
+
``NormativeJudgement.confidence``, which need not agree.
|
|
129
|
+
inputs_identity: Order-insensitive fingerprint of the input list.
|
|
130
|
+
samples_planned: Trials asked for.
|
|
131
|
+
samples_executed: Trials actually run. Equal to ``samples_planned``
|
|
132
|
+
until early termination exists; sourced from the run either way,
|
|
133
|
+
never assumed by the writer.
|
|
134
|
+
termination_reason: Why the run ended.
|
|
135
|
+
covariate_profile: The resolved covariate values — identity.
|
|
136
|
+
factor_record: Run provenance. Never compared by resolution.
|
|
137
|
+
criteria: Per-criterion characterisations, keyed by criterion name,
|
|
138
|
+
in declaration order.
|
|
139
|
+
latency: The gated aggregate-latency summary, carrying the full
|
|
140
|
+
ascending vector of passing-sample durations — the raw material
|
|
141
|
+
a later test needs to derive its own bound at its own
|
|
142
|
+
confidence. ``None`` when no sample passed or no per-sample
|
|
143
|
+
observations were recorded.
|
|
144
|
+
views: Descriptive fingerprints of declared view output schemas
|
|
145
|
+
that are NOT covariates, keyed by view name — visible and
|
|
146
|
+
diffable in the artefact, never compared by baseline
|
|
147
|
+
resolution (covariate fingerprints travel in ``provenance``
|
|
148
|
+
instead). Additive, optional field of the artefact schema.
|
|
149
|
+
"""
|
|
150
|
+
|
|
151
|
+
service_contract_id: str
|
|
152
|
+
service_name: str
|
|
153
|
+
generated_at: datetime
|
|
154
|
+
confidence_level: float
|
|
155
|
+
inputs_identity: str
|
|
156
|
+
samples_planned: int
|
|
157
|
+
samples_executed: int
|
|
158
|
+
criteria: Mapping[str, CriterionCharacterisation]
|
|
159
|
+
termination_reason: str = "COMPLETED"
|
|
160
|
+
covariate_profile: Mapping[str, str] = field(default_factory=dict)
|
|
161
|
+
factor_record: Mapping[str, str] = field(default_factory=dict)
|
|
162
|
+
#: How each input the measurement drove presents itself, in input
|
|
163
|
+
#: order. Informational: identity stays ``inputs_identity``.
|
|
164
|
+
inputs: tuple[str, ...] = ()
|
|
165
|
+
latency: LatencyBlock | None = None
|
|
166
|
+
views: Mapping[str, str] = field(default_factory=dict)
|
|
167
|
+
|
|
168
|
+
def __post_init__(self) -> None:
|
|
169
|
+
object.__setattr__(self, "views", MappingProxyType(dict(self.views)))
|
|
170
|
+
|
|
171
|
+
@staticmethod
|
|
172
|
+
def from_run_result(
|
|
173
|
+
result: RunResult,
|
|
174
|
+
service_name: str,
|
|
175
|
+
covariate_profile: Mapping[str, str] | None = None,
|
|
176
|
+
factor_record: Mapping[str, str] | None = None,
|
|
177
|
+
views: Mapping[str, str] | None = None,
|
|
178
|
+
confidence_level: float = DEFAULT_CONFIDENCE_LEVEL,
|
|
179
|
+
) -> "BaselineRecord":
|
|
180
|
+
"""Build a record from a completed run.
|
|
181
|
+
|
|
182
|
+
Thresholded criteria carry their measurement-time judgement
|
|
183
|
+
(met/failed from the run's verdict); unthresholded criteria are
|
|
184
|
+
characterised without one.
|
|
185
|
+
|
|
186
|
+
The Wilson lower bound is computed here, where the counts and the
|
|
187
|
+
confidence level are both in hand — never in the writer, which
|
|
188
|
+
states what the record holds and derives nothing.
|
|
189
|
+
"""
|
|
190
|
+
criteria: dict[str, CriterionCharacterisation] = {}
|
|
191
|
+
for criterion_result in result.criterion_results:
|
|
192
|
+
judgement = None
|
|
193
|
+
if criterion_result.verdict is not None:
|
|
194
|
+
criterion = criterion_result.criterion
|
|
195
|
+
assert criterion.threshold is not None
|
|
196
|
+
judgement = NormativeJudgement(
|
|
197
|
+
state=JudgementState.MET
|
|
198
|
+
if criterion_result.verdict.value == "pass"
|
|
199
|
+
else JudgementState.FAILED,
|
|
200
|
+
stipulated_threshold=criterion.threshold,
|
|
201
|
+
confidence=criterion.confidence,
|
|
202
|
+
)
|
|
203
|
+
tally = criterion_result.tally
|
|
204
|
+
slack = criterion_result.criterion.optional_slack
|
|
205
|
+
criteria[criterion_result.name] = CriterionCharacterisation(
|
|
206
|
+
successes=tally.successes,
|
|
207
|
+
trials=tally.trials,
|
|
208
|
+
failure_distribution=dict(tally.failure_reasons),
|
|
209
|
+
failure_axes=dict(tally.failure_axes),
|
|
210
|
+
delivery_causes=dict(tally.delivery_causes),
|
|
211
|
+
judgement=judgement,
|
|
212
|
+
standings=criterion_result.standings,
|
|
213
|
+
optional_slack=slack.declared if slack is not None else None,
|
|
214
|
+
wilson_lower_bound=(
|
|
215
|
+
wilson_lower_bound(tally.successes, tally.trials, confidence_level)
|
|
216
|
+
if tally.trials
|
|
217
|
+
else None
|
|
218
|
+
),
|
|
219
|
+
)
|
|
220
|
+
# Every criterion sees every sample, so any criterion's trial count
|
|
221
|
+
# is the run's executed count (companion §1.4.5a). Sourced rather
|
|
222
|
+
# than assumed equal to the plan: when early termination lands, this
|
|
223
|
+
# already states the truth.
|
|
224
|
+
executed = max((r.tally.trials for r in result.criterion_results), default=0)
|
|
225
|
+
return BaselineRecord(
|
|
226
|
+
service_contract_id=result.contract_id,
|
|
227
|
+
service_name=service_name,
|
|
228
|
+
generated_at=result.finished_at,
|
|
229
|
+
confidence_level=confidence_level,
|
|
230
|
+
inputs_identity=result.inputs_identity,
|
|
231
|
+
samples_planned=result.plan.samples,
|
|
232
|
+
samples_executed=executed,
|
|
233
|
+
criteria=criteria,
|
|
234
|
+
covariate_profile=dict(covariate_profile or {}),
|
|
235
|
+
factor_record=dict(factor_record or {}),
|
|
236
|
+
inputs=tuple(bounded_excerpt(str(v)) for v in result.plan.inputs),
|
|
237
|
+
latency=latency_block(result.samples),
|
|
238
|
+
views=dict(views or {}),
|
|
239
|
+
)
|