baseltest 0.22.0__py3-none-win_amd64.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (134) hide show
  1. baseltest/__init__.py +36 -0
  2. baseltest/_version.py +11 -0
  3. baseltest/baseline/__init__.py +46 -0
  4. baseltest/baseline/reader.py +290 -0
  5. baseltest/baseline/record.py +239 -0
  6. baseltest/baseline/writer.py +252 -0
  7. baseltest/contract/__init__.py +126 -0
  8. baseltest/contract/errors.py +21 -0
  9. baseltest/contract/evaluation.py +419 -0
  10. baseltest/contract/model.py +489 -0
  11. baseltest/contract/postconditions.py +530 -0
  12. baseltest/contract/reply.py +37 -0
  13. baseltest/declarative/__init__.py +66 -0
  14. baseltest/declarative/_cli.py +685 -0
  15. baseltest/declarative/_errors.py +13 -0
  16. baseltest/declarative/_instantiate/__init__.py +34 -0
  17. baseltest/declarative/_instantiate/_baseline.py +185 -0
  18. baseltest/declarative/_instantiate/_compose.py +158 -0
  19. baseltest/declarative/_instantiate/_explore.py +147 -0
  20. baseltest/declarative/_instantiate/_latency.py +108 -0
  21. baseltest/declarative/_instantiate/_optimize_point.py +122 -0
  22. baseltest/declarative/_instantiate/_postconditions.py +279 -0
  23. baseltest/declarative/_instantiate/_service.py +129 -0
  24. baseltest/declarative/_instantiate/_sizing_policy.py +90 -0
  25. baseltest/declarative/_instantiate/_views.py +79 -0
  26. baseltest/declarative/_materialise.py +191 -0
  27. baseltest/declarative/_optimize.py +334 -0
  28. baseltest/declarative/_parser/__init__.py +40 -0
  29. baseltest/declarative/_parser/_contract.py +160 -0
  30. baseltest/declarative/_parser/_criteria.py +162 -0
  31. baseltest/declarative/_parser/_forms.py +308 -0
  32. baseltest/declarative/_parser/_inputs.py +173 -0
  33. baseltest/declarative/_parser/_latency.py +112 -0
  34. baseltest/declarative/_parser/_model.py +128 -0
  35. baseltest/declarative/_parser/_shape.py +47 -0
  36. baseltest/declarative/_parser/_structure.py +89 -0
  37. baseltest/declarative/_providers/__init__.py +296 -0
  38. baseltest/declarative/_providers/_anthropic.py +143 -0
  39. baseltest/declarative/_providers/_apertus.py +34 -0
  40. baseltest/declarative/_providers/_litellm.py +94 -0
  41. baseltest/declarative/_providers/_media.py +93 -0
  42. baseltest/declarative/_providers/_mistral.py +28 -0
  43. baseltest/declarative/_providers/_ollama.py +79 -0
  44. baseltest/declarative/_providers/_openai.py +45 -0
  45. baseltest/declarative/_providers/_protocol.py +234 -0
  46. baseltest/declarative/_registrations.py +62 -0
  47. baseltest/declarative/_registry/__init__.py +31 -0
  48. baseltest/declarative/_registry/_bindings.py +102 -0
  49. baseltest/declarative/_registry/_core.py +355 -0
  50. baseltest/declarative/_registry/_guards.py +56 -0
  51. baseltest/declarative/_registry/_service_types.py +165 -0
  52. baseltest/declarative/_registry/_transform.py +82 -0
  53. baseltest/declarative/_report.py +142 -0
  54. baseltest/declarative/_roots.py +167 -0
  55. baseltest/declarative/_runner/__init__.py +38 -0
  56. baseltest/declarative/_runner/_check.py +146 -0
  57. baseltest/declarative/_runner/_explore.py +231 -0
  58. baseltest/declarative/_runner/_load.py +45 -0
  59. baseltest/declarative/_runner/_optimize.py +120 -0
  60. baseltest/declarative/_runner/_optimize_loop.py +297 -0
  61. baseltest/declarative/_runner/_run.py +215 -0
  62. baseltest/declarative/_runner/_shared.py +46 -0
  63. baseltest/declarative/_schema_walk.py +355 -0
  64. baseltest/declarative/_services/__init__.py +49 -0
  65. baseltest/declarative/_services/_language_model.py +368 -0
  66. baseltest/declarative/_services/_model.py +97 -0
  67. baseltest/declarative/_services/_parse.py +343 -0
  68. baseltest/declarative/_signatures.py +42 -0
  69. baseltest/declarative/_sizing/__init__.py +34 -0
  70. baseltest/declarative/_sizing/_criteria.py +117 -0
  71. baseltest/declarative/_sizing/_flags.py +79 -0
  72. baseltest/declarative/_sizing/_model.py +62 -0
  73. baseltest/declarative/_sizing/_modes.py +195 -0
  74. baseltest/declarative/_sizing/_pricing.py +66 -0
  75. baseltest/declarative/_sizing/_prompts.py +92 -0
  76. baseltest/declarative/_sizing/_rates.py +43 -0
  77. baseltest/declarative/_sizing/_render.py +108 -0
  78. baseltest/declarative/_sizing/_resolve.py +161 -0
  79. baseltest/declarative/_steppers/__init__.py +74 -0
  80. baseltest/declarative/_steppers/_builtins.py +34 -0
  81. baseltest/declarative/_steppers/_context.py +114 -0
  82. baseltest/declarative/_steppers/_contract.py +146 -0
  83. baseltest/declarative/_steppers/_linear_sweep.py +43 -0
  84. baseltest/declarative/_steppers/_numeric.py +30 -0
  85. baseltest/declarative/_steppers/_prompt_engineer.py +117 -0
  86. baseltest/declarative/_steppers/_refining_grid.py +240 -0
  87. baseltest/declarative/_structured.py +352 -0
  88. baseltest/declarative/_types.py +108 -0
  89. baseltest/engine/__init__.py +65 -0
  90. baseltest/engine/artefact.py +68 -0
  91. baseltest/engine/defect.py +68 -0
  92. baseltest/engine/latency.py +226 -0
  93. baseltest/engine/naming.py +62 -0
  94. baseltest/engine/run/__init__.py +36 -0
  95. baseltest/engine/run/attainment.py +48 -0
  96. baseltest/engine/run/execute.py +319 -0
  97. baseltest/engine/run/feasibility.py +77 -0
  98. baseltest/engine/run/identity.py +44 -0
  99. baseltest/engine/run/judge.py +29 -0
  100. baseltest/engine/run/model.py +210 -0
  101. baseltest/engine/run/sample.py +218 -0
  102. baseltest/exploration/__init__.py +23 -0
  103. baseltest/exploration/writer.py +191 -0
  104. baseltest/observation/__init__.py +20 -0
  105. baseltest/observation/emit.py +162 -0
  106. baseltest/observation/record.py +261 -0
  107. baseltest/optimization/__init__.py +29 -0
  108. baseltest/optimization/record.py +114 -0
  109. baseltest/optimization/writer.py +99 -0
  110. baseltest/py.typed +0 -0
  111. baseltest/reporting/__init__.py +53 -0
  112. baseltest/reporting/console.py +390 -0
  113. baseltest/reporting/run_design.py +86 -0
  114. baseltest/reporting/verdict_reader.py +253 -0
  115. baseltest/reporting/verdict_xml.py +266 -0
  116. baseltest/statistics/__init__.py +94 -0
  117. baseltest/statistics/_constants.py +19 -0
  118. baseltest/statistics/_validation.py +32 -0
  119. baseltest/statistics/feasibility.py +93 -0
  120. baseltest/statistics/latency.py +166 -0
  121. baseltest/statistics/power.py +111 -0
  122. baseltest/statistics/proportion.py +33 -0
  123. baseltest/statistics/sizing.py +210 -0
  124. baseltest/statistics/summary.py +74 -0
  125. baseltest/statistics/threshold.py +269 -0
  126. baseltest/statistics/verdict.py +168 -0
  127. baseltest/statistics/wilson.py +168 -0
  128. baseltest-0.22.0.data/scripts/mavai +0 -0
  129. baseltest-0.22.0.dist-info/METADATA +142 -0
  130. baseltest-0.22.0.dist-info/RECORD +134 -0
  131. baseltest-0.22.0.dist-info/WHEEL +4 -0
  132. baseltest-0.22.0.dist-info/entry_points.txt +2 -0
  133. baseltest-0.22.0.dist-info/licenses/LICENSE +202 -0
  134. baseltest-0.22.0.dist-info/licenses/NOTICE +11 -0
baseltest/__init__.py ADDED
@@ -0,0 +1,36 @@
1
+ """baseltest: probabilistic testing for stochastic services.
2
+
3
+ Python-native counterpart to punit (Java) and feotest (Rust) in the
4
+ mavai framework family — statistical inference over repeated samples,
5
+ not a single pass/fail assertion.
6
+
7
+ The common entry points are re-exported here: construct a :class:`Bindings`,
8
+ then call :func:`run`, :func:`explore`, :func:`optimize`, or
9
+ :func:`check_contract` on a contract file. These are the declarative
10
+ authoring surface (:mod:`baseltest.declarative`), promoted to the package
11
+ root for convenience; that module also carries the narrower stepper/scorer
12
+ context types, and :mod:`baseltest.contract` is the surface for authoring a
13
+ service contract in Python directly.
14
+ """
15
+
16
+ from baseltest._version import __version__
17
+ from baseltest.contract import FileInput, MessageParts, Reply
18
+ from baseltest.declarative import (
19
+ Bindings,
20
+ check_contract,
21
+ explore,
22
+ optimize,
23
+ run,
24
+ )
25
+
26
+ __all__ = [
27
+ "Bindings",
28
+ "FileInput",
29
+ "MessageParts",
30
+ "Reply",
31
+ "__version__",
32
+ "check_contract",
33
+ "explore",
34
+ "optimize",
35
+ "run",
36
+ ]
baseltest/_version.py ADDED
@@ -0,0 +1,11 @@
1
+ """Single source of the package version, importable without the API surface.
2
+
3
+ Lives apart from the package ``__init__`` so a lower layer (the reporting
4
+ renderers stamp the version into the verdict record) can read the version
5
+ without importing the top-level authoring surface that ``__init__``
6
+ re-exports — which would make a lower layer depend on a higher one.
7
+ """
8
+
9
+ from importlib import metadata
10
+
11
+ __version__ = metadata.version("baseltest")
@@ -0,0 +1,46 @@
1
+ """The baseline artefact: the durable record of a measurement run.
2
+
3
+ A measurement persists what was observed -- per-criterion counts and rates,
4
+ identity, and provenance -- as a YAML artefact other tooling can later read.
5
+ This package owns the artefact's schema and is its **single writer**: the
6
+ serialisation lives here and nowhere else. The writer accepts
7
+ characterisation data as input, so any caller that has measured something
8
+ -- this framework's own measure runs, or an external characteriser -- hands
9
+ its data over rather than emitting the format itself.
10
+
11
+ Nothing in this package reads baselines back for threshold derivation; the
12
+ artefact is a durable record whose consumption is a deliberately separate,
13
+ later capability.
14
+ """
15
+
16
+ from .reader import (
17
+ BaselineResolution,
18
+ StoredBaseline,
19
+ StoredCriterion,
20
+ StoredLatency,
21
+ read_baseline,
22
+ resolve_baseline,
23
+ )
24
+ from .record import (
25
+ BaselineRecord,
26
+ CriterionCharacterisation,
27
+ JudgementState,
28
+ NormativeJudgement,
29
+ )
30
+ from .writer import baseline_filename, render_baseline, write_baseline
31
+
32
+ __all__ = [
33
+ "BaselineRecord",
34
+ "BaselineResolution",
35
+ "CriterionCharacterisation",
36
+ "JudgementState",
37
+ "NormativeJudgement",
38
+ "StoredBaseline",
39
+ "StoredCriterion",
40
+ "StoredLatency",
41
+ "baseline_filename",
42
+ "read_baseline",
43
+ "render_baseline",
44
+ "resolve_baseline",
45
+ "write_baseline",
46
+ ]
@@ -0,0 +1,290 @@
1
+ """Read-side of the baseline artefact: parsing and resolution.
2
+
3
+ The single-writer rule is untouched — reading is not writing. Because the
4
+ artefact is emitted by exactly one writer (:mod:`.writer`), the parser here
5
+ accepts precisely that emission grammar: two-space indentation, one
6
+ ``key: value`` per line or one ``- item`` per line under a list key, every
7
+ string JSON-quoted. No third-party dependency; the scalars are JSON, the
8
+ structure is indentation.
9
+
10
+ Both artefact generations read back: ``baseltest-baseline-2`` (current)
11
+ and ``baseltest-baseline-1`` (no ``latency:`` block — a version-1 artefact
12
+ simply characterises the functional dimension only).
13
+
14
+ Resolution is strict identity of what was measured: same contract, same
15
+ inputs fingerprint, same covariates (the recorded provenance, minus the
16
+ volatile keys). A near-miss is reported with the reason it did not match —
17
+ a config drift must never silently downgrade a judged criterion.
18
+ """
19
+
20
+ import hashlib
21
+ import json
22
+ from dataclasses import dataclass, field
23
+ from pathlib import Path
24
+ from typing import Any
25
+
26
+ from baseltest.engine import LatencyBasis
27
+
28
+ from .writer import FINGERPRINT_KEY, SCHEMA_VERSION, baseline_filename_for
29
+
30
+ _READABLE_SCHEMAS = frozenset({SCHEMA_VERSION})
31
+
32
+ # Provenance keys that legitimately differ between the measure run and a
33
+ # later test run: they identify the run, not the thing measured.
34
+
35
+
36
+ @dataclass(frozen=True, slots=True)
37
+ class StoredCriterion:
38
+ """One criterion's recorded evidence, as read back from the artefact."""
39
+
40
+ successes: int
41
+ trials: int
42
+
43
+
44
+ @dataclass(frozen=True, slots=True)
45
+ class StoredLatency:
46
+ """The artefact's latency block, read back for bound derivation.
47
+
48
+ The sorted vector is the payload a later test derives its bound from;
49
+ the percentiles are the measurement run's descriptive summary.
50
+ """
51
+
52
+ basis: LatencyBasis
53
+ contributing_samples: int
54
+ total_samples: int
55
+ percentiles: tuple[tuple[str, int], ...]
56
+ sorted_passing_latencies_ms: tuple[int, ...]
57
+
58
+
59
+ @dataclass(frozen=True, slots=True)
60
+ class StoredBaseline:
61
+ """The artefact's content, read back for resolution and judgement."""
62
+
63
+ path: Path
64
+ contract_id: str
65
+ service_name: str
66
+ sample_count: int
67
+ inputs_identity: str
68
+ generated_at: str
69
+ covariate_profile: dict[str, str]
70
+ factor_record: dict[str, str]
71
+ criteria: dict[str, StoredCriterion]
72
+ latency: StoredLatency | None = None
73
+
74
+
75
+ @dataclass(frozen=True, slots=True)
76
+ class BaselineResolution:
77
+ """The outcome of looking for a matching baseline.
78
+
79
+ Exactly one of ``baseline`` / ``reason`` is meaningful: a match carries
80
+ the stored baseline; a non-match carries the honest reason.
81
+ """
82
+
83
+ baseline: StoredBaseline | None = None
84
+ reason: str | None = None
85
+ mismatched_keys: tuple[str, ...] = field(default=())
86
+
87
+ @property
88
+ def matched(self) -> bool:
89
+ return self.baseline is not None
90
+
91
+
92
+ def _parse_lines(lines: list[str]) -> dict[str, Any]:
93
+ """Parse the writer's emission grammar into nested mappings and lists.
94
+
95
+ A ``key:`` line opens a nested container that starts as a mapping and
96
+ becomes a list on its first ``- item`` line — the writer only ever
97
+ emits homogeneous containers, so the switch is unambiguous.
98
+ """
99
+ root: dict[str, Any] = {}
100
+ # (indent, container, parent, key-in-parent); the root has no parent.
101
+ stack: list[tuple[int, Any, dict[str, Any] | None, str | None]] = [(0, root, None, None)]
102
+ for raw in lines:
103
+ if not raw.strip():
104
+ continue
105
+ indent = len(raw) - len(raw.lstrip(" "))
106
+ if indent % 2 != 0:
107
+ raise ValueError(f"malformed indentation: {raw!r}")
108
+ line = raw.strip()
109
+ while stack and stack[-1][0] > indent:
110
+ stack.pop()
111
+ top_indent, container, parent, parent_key = stack[-1]
112
+ if line.startswith("- "):
113
+ if isinstance(container, dict):
114
+ if container or parent is None or parent_key is None:
115
+ raise ValueError(f"malformed list item: {raw!r}")
116
+ container = []
117
+ parent[parent_key] = container
118
+ stack[-1] = (top_indent, container, parent, parent_key)
119
+ container.append(json.loads(line[2:]))
120
+ elif isinstance(container, list):
121
+ raise ValueError(f"malformed line inside a list: {raw!r}")
122
+ else:
123
+ key, value_text = _split_entry(line)
124
+ if value_text is None:
125
+ child: dict[str, Any] = {}
126
+ container[key] = child
127
+ stack.append((indent + 2, child, container, key))
128
+ else:
129
+ container[key] = json.loads(value_text)
130
+ return root
131
+
132
+
133
+ _DECODER = json.JSONDecoder()
134
+
135
+
136
+ def _split_entry(line: str) -> tuple[str, str | None]:
137
+ """One emitted mapping line into its key and value text.
138
+
139
+ Returns ``(key, None)`` for a block-opening ``key:`` line, otherwise
140
+ ``(key, value_text)``. A JSON-quoted key is decoded, never searched
141
+ for a separator: the key itself may contain ``": "`` (a
142
+ failure-reason string quoting a regex, a covariate value).
143
+ """
144
+ if line.startswith('"'):
145
+ try:
146
+ key, end = _DECODER.raw_decode(line)
147
+ except json.JSONDecodeError as error:
148
+ raise ValueError(f"malformed line: {line!r}") from error
149
+ rest = line[end:]
150
+ if rest == ":":
151
+ return str(key), None
152
+ if rest.startswith(": ") and rest[2:]:
153
+ return str(key), rest[2:]
154
+ raise ValueError(f"malformed line: {line!r}")
155
+ if line.endswith(":"):
156
+ return line[:-1], None
157
+ key, _, value_text = line.partition(": ")
158
+ if not value_text:
159
+ raise ValueError(f"malformed line: {line!r}")
160
+ return key, value_text
161
+
162
+
163
+ def _parse_latency(body: dict[str, Any] | None) -> StoredLatency | None:
164
+ if body is None:
165
+ return None
166
+ percentiles = tuple(
167
+ (key, int(value))
168
+ for key, value in body.items()
169
+ if key.startswith("p") and key.endswith("Ms")
170
+ )
171
+ vector = tuple(int(v) for v in body.get("sortedPassingLatenciesMs", []))
172
+ return StoredLatency(
173
+ basis=LatencyBasis(body["basis"]),
174
+ contributing_samples=int(body["contributingSamples"]),
175
+ total_samples=int(body["totalSamples"]),
176
+ percentiles=percentiles,
177
+ sorted_passing_latencies_ms=vector,
178
+ )
179
+
180
+
181
+ def read_baseline(path: Path) -> StoredBaseline:
182
+ """Read one artefact back.
183
+
184
+ Raises:
185
+ ValueError: The file is not a readable baseline artefact (unknown
186
+ schema generation, malformed emission).
187
+ OSError: The file cannot be read.
188
+ """
189
+ text = path.read_text(encoding="utf-8")
190
+ data = _parse_lines(text.splitlines())
191
+ schema = data.get("schemaVersion")
192
+ if schema != SCHEMA_VERSION:
193
+ # The whole migration experience is this sentence: what was found,
194
+ # what is expected, and the verb that regenerates the artefact.
195
+ raise ValueError(
196
+ f"{path.name}: schema {schema!r} is not {SCHEMA_VERSION!r}. "
197
+ "Baselines written by an earlier baseltest are not read: re-run "
198
+ "`basel measure` against this contract to regenerate it."
199
+ )
200
+ _verify_fingerprint(path, text, data)
201
+ criteria: dict[str, StoredCriterion] = {}
202
+ for name, body in data.get("criteria", {}).items():
203
+ criteria[name] = StoredCriterion(
204
+ successes=int(body["successes"]), trials=int(body["trials"])
205
+ )
206
+ execution = data.get("execution", {})
207
+ return StoredBaseline(
208
+ path=path,
209
+ contract_id=str(data["serviceContractId"]),
210
+ service_name=str(data["serviceName"]),
211
+ sample_count=int(execution["samplesExecuted"]),
212
+ inputs_identity=str(data["inputsIdentity"]),
213
+ generated_at=str(data.get("generatedAt", "")),
214
+ covariate_profile={str(k): str(v) for k, v in data.get("covariateProfile", {}).items()},
215
+ factor_record={str(k): str(v) for k, v in data.get("factorRecord", {}).items()},
216
+ criteria=criteria,
217
+ latency=_parse_latency(data.get("latency")),
218
+ )
219
+
220
+
221
+ def _verify_fingerprint(path: Path, text: str, data: dict[str, object]) -> None:
222
+ """Refuse a record whose content does not match its stated fingerprint.
223
+
224
+ The fingerprint covers the document with its own line absent, which is
225
+ the emitted body up to that final line.
226
+ """
227
+ stated = data.get(FINGERPRINT_KEY)
228
+ if not stated:
229
+ raise ValueError(f"{path.name}: no {FINGERPRINT_KEY} — the record cannot be trusted")
230
+ marker = f"\n{FINGERPRINT_KEY}:"
231
+ body = text[: text.rindex(marker) + 1] if marker in text else text
232
+ actual = "sha256:" + hashlib.sha256(body.encode("utf-8")).hexdigest()
233
+ if actual != stated:
234
+ raise ValueError(
235
+ f"{path.name}: {FINGERPRINT_KEY} does not match the record's content "
236
+ "— the file has been edited or truncated since it was written"
237
+ )
238
+
239
+
240
+ def resolve_baseline(
241
+ baseline_dir: Path,
242
+ contract_id: str,
243
+ service_name: str,
244
+ inputs_identity: str,
245
+ covariate_profile: dict[str, str],
246
+ ) -> BaselineResolution:
247
+ """Find the baseline matching what this run would measure.
248
+
249
+ Matching is strict identity over the tuple (contract, service, inputs,
250
+ covariate profile). The deterministic filename locates the candidate and
251
+ the loaded record's own statement decides — the body is authoritative,
252
+ never the path. Any difference is a non-match, and the resolution says
253
+ which keys differed: a drifted configuration is surfaced, never silently
254
+ treated as "no baseline".
255
+
256
+ The covariate profile is compared as stated, rather than reconstructed
257
+ by subtracting volatile keys from provenance: identity and provenance
258
+ are separate fields in the artefact, so they no longer have to be
259
+ separated again on the way back in.
260
+ """
261
+ candidate = baseline_dir / baseline_filename_for(contract_id, service_name, inputs_identity)
262
+ if not candidate.is_file():
263
+ return BaselineResolution(reason=f"no baseline found (expected {candidate.as_posix()})")
264
+ try:
265
+ stored = read_baseline(candidate)
266
+ except (ValueError, OSError, json.JSONDecodeError) as error:
267
+ return BaselineResolution(reason=f"baseline {candidate.name} is unreadable: {error}")
268
+ if stored.inputs_identity != inputs_identity:
269
+ return BaselineResolution(reason=f"baseline {candidate.name} records different inputs")
270
+ if stored.service_name != service_name:
271
+ return BaselineResolution(
272
+ reason=(
273
+ f"baseline {candidate.name} records service {stored.service_name!r}, "
274
+ f"not {service_name!r}"
275
+ )
276
+ )
277
+ theirs = dict(stored.covariate_profile)
278
+ ours = dict(covariate_profile)
279
+ if theirs != ours:
280
+ differing = sorted(
281
+ key for key in set(theirs) | set(ours) if theirs.get(key) != ours.get(key)
282
+ )
283
+ return BaselineResolution(
284
+ reason=(
285
+ f"baseline {candidate.name} was measured under a different "
286
+ f"configuration (differing: {', '.join(differing)})"
287
+ ),
288
+ mismatched_keys=tuple(differing),
289
+ )
290
+ return BaselineResolution(baseline=stored)
@@ -0,0 +1,239 @@
1
+ """The baseline record: what a measurement run durably states about a service."""
2
+
3
+ from collections.abc import Mapping
4
+ from dataclasses import dataclass, field
5
+ from datetime import datetime
6
+ from enum import StrEnum
7
+ from types import MappingProxyType
8
+
9
+ from baseltest.contract import DeliveryCause, FailureAxis, PostconditionStanding
10
+ from baseltest.engine import LatencyBlock, RunResult, latency_block
11
+ from baseltest.engine.naming import bounded_excerpt
12
+ from baseltest.statistics import DEFAULT_CONFIDENCE_LEVEL, wilson_lower_bound
13
+
14
+
15
+ class CriterionMode(StrEnum):
16
+ """Whether a criterion estimates a proportion (companion §1.5).
17
+
18
+ An observational criterion states no rate and no bound: it estimates no
19
+ proportion at all, passing iff no failure was observed.
20
+ """
21
+
22
+ INFERENTIAL = "inferential"
23
+ OBSERVATIONAL = "observational"
24
+
25
+
26
+ class CriterionProcedure(StrEnum):
27
+ """The inferential procedure a criterion is judged under."""
28
+
29
+ REGRESSION = "REGRESSION"
30
+ COMPLIANCE = "COMPLIANCE"
31
+
32
+
33
+ class JudgementState(StrEnum):
34
+ """A measurement-time normative judgement's outcome.
35
+
36
+ The schema also reserves ``unsupportable`` for callers whose sample size
37
+ was not validated up front; baseltest validates every run's size before
38
+ sampling, so it emits only ``met`` or ``failed``.
39
+ """
40
+
41
+ MET = "met"
42
+ FAILED = "failed"
43
+
44
+
45
+ @dataclass(frozen=True, slots=True)
46
+ class NormativeJudgement:
47
+ """The measurement-time judgement of one criterion against its declared threshold.
48
+
49
+ Purely documentary: a later reader sees not only what was measured but
50
+ how the measurement stood relative to a bar in force at measurement
51
+ time. It never affects how the artefact is consumed.
52
+
53
+ Attributes:
54
+ state: The :class:`JudgementState` reached against the bar.
55
+ stipulated_threshold: The declared threshold judged against.
56
+ confidence: The confidence level of the judgement.
57
+ """
58
+
59
+ state: JudgementState
60
+ stipulated_threshold: float
61
+ confidence: float
62
+
63
+
64
+ @dataclass(frozen=True, slots=True)
65
+ class CriterionCharacterisation:
66
+ """One criterion's measured characterisation.
67
+
68
+ Attributes:
69
+ successes: Passing trials.
70
+ trials: Total trials.
71
+ failure_distribution: Failure reasons and their counts; empty when
72
+ every trial passed.
73
+ failure_axes: The companion's diagnostic axis per observed reason.
74
+ delivery_causes: The delivery cause per observed reason that was a
75
+ failed delivery — what the artefact states as that entry's
76
+ identity, since the reason itself is an interpolated message.
77
+ judgement: The measurement-time judgement, when the criterion
78
+ declared a threshold; ``None`` otherwise.
79
+ standings: The criterion's descriptive per-postcondition tally —
80
+ per ``(input, check)``, passed/failed/skipped counts and the
81
+ observed fraction, each row carrying its check's optional
82
+ flag. Triage data, additive in the artefact schema; never an
83
+ interval or a per-check verdict.
84
+ optional_slack: The criterion's declared optional-check failure
85
+ budget, verbatim as authored (``None`` when undeclared —
86
+ never ``"0"``). Additive in the artefact schema.
87
+ """
88
+
89
+ successes: int
90
+ trials: int
91
+ failure_distribution: Mapping[str, int] = field(default_factory=dict)
92
+ failure_axes: Mapping[str, FailureAxis] = field(default_factory=dict)
93
+ delivery_causes: Mapping[str, DeliveryCause] = field(default_factory=dict)
94
+ judgement: NormativeJudgement | None = None
95
+ standings: tuple[PostconditionStanding, ...] = ()
96
+ optional_slack: str | None = None
97
+ mode: CriterionMode = CriterionMode.INFERENTIAL
98
+ procedure: CriterionProcedure | None = CriterionProcedure.REGRESSION
99
+ wilson_lower_bound: float | None = None
100
+
101
+ def __post_init__(self) -> None:
102
+ object.__setattr__(
103
+ self, "failure_distribution", MappingProxyType(dict(self.failure_distribution))
104
+ )
105
+ object.__setattr__(self, "failure_axes", MappingProxyType(dict(self.failure_axes)))
106
+ object.__setattr__(self, "delivery_causes", MappingProxyType(dict(self.delivery_causes)))
107
+
108
+ @property
109
+ def observed_rate(self) -> float:
110
+ """The observed pass rate. A recorded characterisation has at least
111
+ one trial."""
112
+ return self.successes / self.trials
113
+
114
+
115
+ @dataclass(frozen=True, slots=True)
116
+ class BaselineRecord:
117
+ """Everything the baseline artefact states.
118
+
119
+ Attributes:
120
+ service_contract_id: The measured service contract's identity.
121
+ service_name: The name of the service that was invoked. Identity,
122
+ not provenance: a contract may exercise several services and a
123
+ service may be the subject of several contracts, so neither
124
+ names a record alone.
125
+ generated_at: Measurement time, UTC.
126
+ confidence_level: The level every ``wilson_lower_bound`` in this
127
+ record was computed at. Distinct from a criterion's stipulated
128
+ ``NormativeJudgement.confidence``, which need not agree.
129
+ inputs_identity: Order-insensitive fingerprint of the input list.
130
+ samples_planned: Trials asked for.
131
+ samples_executed: Trials actually run. Equal to ``samples_planned``
132
+ until early termination exists; sourced from the run either way,
133
+ never assumed by the writer.
134
+ termination_reason: Why the run ended.
135
+ covariate_profile: The resolved covariate values — identity.
136
+ factor_record: Run provenance. Never compared by resolution.
137
+ criteria: Per-criterion characterisations, keyed by criterion name,
138
+ in declaration order.
139
+ latency: The gated aggregate-latency summary, carrying the full
140
+ ascending vector of passing-sample durations — the raw material
141
+ a later test needs to derive its own bound at its own
142
+ confidence. ``None`` when no sample passed or no per-sample
143
+ observations were recorded.
144
+ views: Descriptive fingerprints of declared view output schemas
145
+ that are NOT covariates, keyed by view name — visible and
146
+ diffable in the artefact, never compared by baseline
147
+ resolution (covariate fingerprints travel in ``provenance``
148
+ instead). Additive, optional field of the artefact schema.
149
+ """
150
+
151
+ service_contract_id: str
152
+ service_name: str
153
+ generated_at: datetime
154
+ confidence_level: float
155
+ inputs_identity: str
156
+ samples_planned: int
157
+ samples_executed: int
158
+ criteria: Mapping[str, CriterionCharacterisation]
159
+ termination_reason: str = "COMPLETED"
160
+ covariate_profile: Mapping[str, str] = field(default_factory=dict)
161
+ factor_record: Mapping[str, str] = field(default_factory=dict)
162
+ #: How each input the measurement drove presents itself, in input
163
+ #: order. Informational: identity stays ``inputs_identity``.
164
+ inputs: tuple[str, ...] = ()
165
+ latency: LatencyBlock | None = None
166
+ views: Mapping[str, str] = field(default_factory=dict)
167
+
168
+ def __post_init__(self) -> None:
169
+ object.__setattr__(self, "views", MappingProxyType(dict(self.views)))
170
+
171
+ @staticmethod
172
+ def from_run_result(
173
+ result: RunResult,
174
+ service_name: str,
175
+ covariate_profile: Mapping[str, str] | None = None,
176
+ factor_record: Mapping[str, str] | None = None,
177
+ views: Mapping[str, str] | None = None,
178
+ confidence_level: float = DEFAULT_CONFIDENCE_LEVEL,
179
+ ) -> "BaselineRecord":
180
+ """Build a record from a completed run.
181
+
182
+ Thresholded criteria carry their measurement-time judgement
183
+ (met/failed from the run's verdict); unthresholded criteria are
184
+ characterised without one.
185
+
186
+ The Wilson lower bound is computed here, where the counts and the
187
+ confidence level are both in hand — never in the writer, which
188
+ states what the record holds and derives nothing.
189
+ """
190
+ criteria: dict[str, CriterionCharacterisation] = {}
191
+ for criterion_result in result.criterion_results:
192
+ judgement = None
193
+ if criterion_result.verdict is not None:
194
+ criterion = criterion_result.criterion
195
+ assert criterion.threshold is not None
196
+ judgement = NormativeJudgement(
197
+ state=JudgementState.MET
198
+ if criterion_result.verdict.value == "pass"
199
+ else JudgementState.FAILED,
200
+ stipulated_threshold=criterion.threshold,
201
+ confidence=criterion.confidence,
202
+ )
203
+ tally = criterion_result.tally
204
+ slack = criterion_result.criterion.optional_slack
205
+ criteria[criterion_result.name] = CriterionCharacterisation(
206
+ successes=tally.successes,
207
+ trials=tally.trials,
208
+ failure_distribution=dict(tally.failure_reasons),
209
+ failure_axes=dict(tally.failure_axes),
210
+ delivery_causes=dict(tally.delivery_causes),
211
+ judgement=judgement,
212
+ standings=criterion_result.standings,
213
+ optional_slack=slack.declared if slack is not None else None,
214
+ wilson_lower_bound=(
215
+ wilson_lower_bound(tally.successes, tally.trials, confidence_level)
216
+ if tally.trials
217
+ else None
218
+ ),
219
+ )
220
+ # Every criterion sees every sample, so any criterion's trial count
221
+ # is the run's executed count (companion §1.4.5a). Sourced rather
222
+ # than assumed equal to the plan: when early termination lands, this
223
+ # already states the truth.
224
+ executed = max((r.tally.trials for r in result.criterion_results), default=0)
225
+ return BaselineRecord(
226
+ service_contract_id=result.contract_id,
227
+ service_name=service_name,
228
+ generated_at=result.finished_at,
229
+ confidence_level=confidence_level,
230
+ inputs_identity=result.inputs_identity,
231
+ samples_planned=result.plan.samples,
232
+ samples_executed=executed,
233
+ criteria=criteria,
234
+ covariate_profile=dict(covariate_profile or {}),
235
+ factor_record=dict(factor_record or {}),
236
+ inputs=tuple(bounded_excerpt(str(v)) for v in result.plan.inputs),
237
+ latency=latency_block(result.samples),
238
+ views=dict(views or {}),
239
+ )