phaseprobe 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- phaseprobe/__init__.py +30 -0
- phaseprobe/__main__.py +5 -0
- phaseprobe/adapters/__init__.py +4 -0
- phaseprobe/adapters/loader.py +51 -0
- phaseprobe/adapters/scipy.py +539 -0
- phaseprobe/api.py +64 -0
- phaseprobe/artifacts.py +124 -0
- phaseprobe/cli.py +242 -0
- phaseprobe/config.py +132 -0
- phaseprobe/data/__init__.py +1 -0
- phaseprobe/data/examples/__init__.py +1 -0
- phaseprobe/data/examples/logistic-negative.json +27 -0
- phaseprobe/data/examples/logistic-scan.json +27 -0
- phaseprobe/data/examples/lorenz-negative.json +30 -0
- phaseprobe/data/examples/lorenz-perturb.json +30 -0
- phaseprobe/data/examples/predator-prey-check.json +25 -0
- phaseprobe/data/examples/predator-prey-negative.json +25 -0
- phaseprobe/data/examples/toggle-negative.json +30 -0
- phaseprobe/data/examples/toggle-perturb.json +31 -0
- phaseprobe/engine.py +810 -0
- phaseprobe/errors.py +31 -0
- phaseprobe/examples/__init__.py +1 -0
- phaseprobe/examples/scipy_models.py +151 -0
- phaseprobe/generate.py +72 -0
- phaseprobe/models/__init__.py +36 -0
- phaseprobe/models/_common.py +56 -0
- phaseprobe/models/logistic.py +63 -0
- phaseprobe/models/lorenz.py +64 -0
- phaseprobe/models/predator_prey.py +86 -0
- phaseprobe/models/toggle.py +73 -0
- phaseprobe/replay.py +496 -0
- phaseprobe/reporting.py +173 -0
- phaseprobe/types.py +131 -0
- phaseprobe-0.2.0.dist-info/METADATA +275 -0
- phaseprobe-0.2.0.dist-info/RECORD +38 -0
- phaseprobe-0.2.0.dist-info/WHEEL +4 -0
- phaseprobe-0.2.0.dist-info/entry_points.txt +2 -0
- phaseprobe-0.2.0.dist-info/licenses/LICENSE +201 -0
phaseprobe/replay.py
ADDED
|
@@ -0,0 +1,496 @@
|
|
|
1
|
+
"""Versioned exact/tolerance replay with integrity-protected evidence."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
import math
|
|
8
|
+
from collections.abc import Mapping, Sequence
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import Any, cast
|
|
12
|
+
|
|
13
|
+
from phaseprobe import __version__
|
|
14
|
+
from phaseprobe.config import ProbeConfig, canonical_json, parse_config
|
|
15
|
+
from phaseprobe.engine import ProbeOutcome, SimulationResult, simulate
|
|
16
|
+
from phaseprobe.errors import ConfigurationError, IntegrityError
|
|
17
|
+
from phaseprobe.types import InvariantResult, TracePoint
|
|
18
|
+
|
|
19
|
+
REPLAY_SCHEMA_VERSION = "2.0"
|
|
20
|
+
SUPPORTED_REPLAY_SCHEMA_VERSIONS = frozenset({"1.0", REPLAY_SCHEMA_VERSION})
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _invariant_payload(result: InvariantResult) -> dict[str, object]:
|
|
24
|
+
return {
|
|
25
|
+
"name": result.name,
|
|
26
|
+
"passed": result.passed,
|
|
27
|
+
"measured": result.measured,
|
|
28
|
+
"tolerance": result.tolerance,
|
|
29
|
+
"detail": result.detail,
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _point_payload(point: TracePoint) -> dict[str, object]:
|
|
34
|
+
return {
|
|
35
|
+
"step": point.step,
|
|
36
|
+
"time": point.time,
|
|
37
|
+
"state": list(point.state),
|
|
38
|
+
"observations": dict(point.observations),
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _execution_payload(result: SimulationResult) -> dict[str, object]:
|
|
43
|
+
payload: dict[str, object] = {
|
|
44
|
+
"parameters": dict(result.parameters),
|
|
45
|
+
"initial_state": list(result.initial_state),
|
|
46
|
+
"final_state": list(result.final_state),
|
|
47
|
+
"classification": result.classification,
|
|
48
|
+
"observations": dict(result.observations),
|
|
49
|
+
"invariants": [_invariant_payload(item) for item in result.invariants],
|
|
50
|
+
"trace_sha256": result.trace_sha256,
|
|
51
|
+
"invariant_violations": result.invariant_violations,
|
|
52
|
+
"solver_success": bool(result.execution_metadata.get("solver_success", True)),
|
|
53
|
+
"execution_metadata": dict(result.execution_metadata),
|
|
54
|
+
}
|
|
55
|
+
if result.replay_mode == "tolerance":
|
|
56
|
+
payload["trace"] = [_point_payload(point) for point in result.trace]
|
|
57
|
+
return payload
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _number(values: Mapping[str, object], name: str) -> float:
|
|
61
|
+
raw = values.get(name)
|
|
62
|
+
if not isinstance(raw, int | float) or isinstance(raw, bool):
|
|
63
|
+
raise ConfigurationError(f"replay.{name} must be numeric")
|
|
64
|
+
result = float(raw)
|
|
65
|
+
if not math.isfinite(result) or result < 0.0:
|
|
66
|
+
raise ConfigurationError(f"replay.{name} must be finite and non-negative")
|
|
67
|
+
return result
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _tolerance_policy(config: ProbeConfig) -> dict[str, object]:
|
|
71
|
+
values = config.section("replay")
|
|
72
|
+
if values.get("mode") != "tolerance":
|
|
73
|
+
raise ConfigurationError(
|
|
74
|
+
"trajectory adapters require replay.mode='tolerance'; exact adaptive replay is not claimed"
|
|
75
|
+
)
|
|
76
|
+
observable_raw = values.get("observable_atol")
|
|
77
|
+
if not isinstance(observable_raw, dict):
|
|
78
|
+
raise ConfigurationError("replay.observable_atol must be an object")
|
|
79
|
+
observable_atol: dict[str, float] = {}
|
|
80
|
+
for name, raw in observable_raw.items():
|
|
81
|
+
if not isinstance(raw, int | float) or isinstance(raw, bool):
|
|
82
|
+
raise ConfigurationError(f"replay.observable_atol.{name} must be numeric")
|
|
83
|
+
parsed = float(raw)
|
|
84
|
+
if not math.isfinite(parsed) or parsed < 0.0:
|
|
85
|
+
raise ConfigurationError(
|
|
86
|
+
f"replay.observable_atol.{name} must be finite and non-negative"
|
|
87
|
+
)
|
|
88
|
+
observable_atol[name] = parsed
|
|
89
|
+
max_unmatched = values.get("max_unmatched_points")
|
|
90
|
+
if not isinstance(max_unmatched, int) or isinstance(max_unmatched, bool) or max_unmatched < 0:
|
|
91
|
+
raise ConfigurationError("replay.max_unmatched_points must be a non-negative integer")
|
|
92
|
+
expected_success = values.get("expected_solver_success")
|
|
93
|
+
require_classifier = values.get("require_classifier")
|
|
94
|
+
require_invariants = values.get("require_invariants")
|
|
95
|
+
if not all(
|
|
96
|
+
isinstance(item, bool)
|
|
97
|
+
for item in (expected_success, require_classifier, require_invariants)
|
|
98
|
+
):
|
|
99
|
+
raise ConfigurationError(
|
|
100
|
+
"replay expected_solver_success, require_classifier, and require_invariants must be booleans"
|
|
101
|
+
)
|
|
102
|
+
return {
|
|
103
|
+
"mode": "tolerance",
|
|
104
|
+
"state_atol": _number(values, "state_atol"),
|
|
105
|
+
"state_rtol": _number(values, "state_rtol"),
|
|
106
|
+
"observable_atol": observable_atol,
|
|
107
|
+
"invariant_measure_atol": _number(values, "invariant_measure_atol"),
|
|
108
|
+
"endpoint_time_atol": _number(values, "endpoint_time_atol"),
|
|
109
|
+
"event_time_atol": _number(values, "event_time_atol"),
|
|
110
|
+
"retained_grid_time_atol": _number(values, "retained_grid_time_atol"),
|
|
111
|
+
"max_unmatched_points": max_unmatched,
|
|
112
|
+
"expected_solver_success": expected_success,
|
|
113
|
+
"require_classifier": require_classifier,
|
|
114
|
+
"require_invariants": require_invariants,
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def fixture_payload(outcome: ProbeOutcome) -> dict[str, object]:
|
|
119
|
+
"""Create an integrity-protected replay fixture from validated evidence."""
|
|
120
|
+
|
|
121
|
+
modes = {outcome.baseline.replay_mode}
|
|
122
|
+
if outcome.changed is not None:
|
|
123
|
+
modes.add(outcome.changed.replay_mode)
|
|
124
|
+
if len(modes) != 1:
|
|
125
|
+
raise ConfigurationError("baseline and changed executions use different replay modes")
|
|
126
|
+
mode = next(iter(modes))
|
|
127
|
+
comparison = {"mode": "exact"} if mode == "exact" else _tolerance_policy(outcome.config)
|
|
128
|
+
payload: dict[str, object] = {
|
|
129
|
+
"schema_version": REPLAY_SCHEMA_VERSION,
|
|
130
|
+
"created_by": f"phaseprobe {__version__}",
|
|
131
|
+
"model": outcome.baseline.model,
|
|
132
|
+
"model_identity": outcome.baseline.model_identity,
|
|
133
|
+
"seed": outcome.baseline.seed,
|
|
134
|
+
"configuration": dict(outcome.config.data),
|
|
135
|
+
"comparison": comparison,
|
|
136
|
+
"baseline": _execution_payload(outcome.baseline),
|
|
137
|
+
"changed": _execution_payload(outcome.changed) if outcome.changed is not None else None,
|
|
138
|
+
"finding": dict(outcome.finding) if outcome.finding is not None else None,
|
|
139
|
+
"reproducible": outcome.reproducible,
|
|
140
|
+
}
|
|
141
|
+
integrity = hashlib.sha256(canonical_json(payload).encode("utf-8")).hexdigest()
|
|
142
|
+
payload["integrity_sha256"] = integrity
|
|
143
|
+
return payload
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _load_fixture(path: Path) -> dict[str, object]:
|
|
147
|
+
try:
|
|
148
|
+
parsed: Any = json.loads(path.read_text(encoding="utf-8"))
|
|
149
|
+
except OSError as exc:
|
|
150
|
+
raise ConfigurationError(f"cannot read replay fixture {path}: {exc}") from exc
|
|
151
|
+
except json.JSONDecodeError as exc:
|
|
152
|
+
raise IntegrityError(f"invalid replay JSON in {path}: {exc}") from exc
|
|
153
|
+
if not isinstance(parsed, dict):
|
|
154
|
+
raise IntegrityError("replay fixture must be a JSON object")
|
|
155
|
+
return cast(dict[str, object], parsed)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def validate_fixture(path: Path) -> dict[str, object]:
|
|
159
|
+
"""Validate a v1 exact or v2 exact/tolerance fixture before execution."""
|
|
160
|
+
|
|
161
|
+
payload = _load_fixture(path)
|
|
162
|
+
version = payload.get("schema_version")
|
|
163
|
+
if version not in SUPPORTED_REPLAY_SCHEMA_VERSIONS:
|
|
164
|
+
raise IntegrityError(
|
|
165
|
+
f"unsupported replay schema {version!r}; supported versions are "
|
|
166
|
+
f"{sorted(SUPPORTED_REPLAY_SCHEMA_VERSIONS)!r}"
|
|
167
|
+
)
|
|
168
|
+
expected = payload.get("integrity_sha256")
|
|
169
|
+
if not isinstance(expected, str):
|
|
170
|
+
raise IntegrityError("replay fixture has no integrity_sha256")
|
|
171
|
+
unsigned = dict(payload)
|
|
172
|
+
del unsigned["integrity_sha256"]
|
|
173
|
+
actual = hashlib.sha256(canonical_json(unsigned).encode("utf-8")).hexdigest()
|
|
174
|
+
if actual != expected:
|
|
175
|
+
raise IntegrityError(
|
|
176
|
+
f"replay fixture integrity mismatch: expected {expected}, got {actual}"
|
|
177
|
+
)
|
|
178
|
+
return payload
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def _mapping(value: object, context: str) -> Mapping[str, object]:
|
|
182
|
+
if not isinstance(value, dict):
|
|
183
|
+
raise IntegrityError(f"replay {context} must be an object")
|
|
184
|
+
return cast(Mapping[str, object], value)
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def _sequence(value: object, context: str) -> Sequence[object]:
|
|
188
|
+
if not isinstance(value, list):
|
|
189
|
+
raise IntegrityError(f"replay {context} must be an array")
|
|
190
|
+
return cast(Sequence[object], value)
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _float_mapping(value: object, context: str) -> dict[str, float]:
|
|
194
|
+
values = _mapping(value, context)
|
|
195
|
+
result: dict[str, float] = {}
|
|
196
|
+
for name, raw in values.items():
|
|
197
|
+
if not isinstance(raw, int | float) or isinstance(raw, bool):
|
|
198
|
+
raise IntegrityError(f"replay {context}.{name} must be numeric")
|
|
199
|
+
result[name] = float(raw)
|
|
200
|
+
return result
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def _state(value: object, context: str) -> tuple[float, ...]:
|
|
204
|
+
raw_values = _sequence(value, context)
|
|
205
|
+
state: list[float] = []
|
|
206
|
+
for raw in raw_values:
|
|
207
|
+
if not isinstance(raw, int | float) or isinstance(raw, bool):
|
|
208
|
+
raise IntegrityError(f"replay {context} contains a non-number")
|
|
209
|
+
state.append(float(raw))
|
|
210
|
+
return tuple(state)
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _policy_number(policy: Mapping[str, object], name: str) -> float:
|
|
214
|
+
raw = policy.get(name)
|
|
215
|
+
if not isinstance(raw, int | float) or isinstance(raw, bool):
|
|
216
|
+
raise IntegrityError(f"replay comparison.{name} must be numeric")
|
|
217
|
+
result = float(raw)
|
|
218
|
+
if not math.isfinite(result) or result < 0.0:
|
|
219
|
+
raise IntegrityError(f"replay comparison.{name} must be finite and non-negative")
|
|
220
|
+
return result
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def _close(left: float, right: float, *, atol: float, rtol: float) -> bool:
|
|
224
|
+
return math.isclose(left, right, abs_tol=atol, rel_tol=rtol)
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _observations_match(
|
|
228
|
+
expected: object,
|
|
229
|
+
actual: Mapping[str, object],
|
|
230
|
+
policy: Mapping[str, object],
|
|
231
|
+
) -> bool:
|
|
232
|
+
expected_values = _mapping(expected, "trace observations")
|
|
233
|
+
tolerances = _mapping(policy.get("observable_atol"), "comparison.observable_atol")
|
|
234
|
+
if set(expected_values) != set(actual):
|
|
235
|
+
return False
|
|
236
|
+
default_atol = _policy_number(policy, "state_atol")
|
|
237
|
+
rtol = _policy_number(policy, "state_rtol")
|
|
238
|
+
for name, expected_value in expected_values.items():
|
|
239
|
+
actual_value = actual[name]
|
|
240
|
+
if isinstance(expected_value, int | float) and not isinstance(expected_value, bool):
|
|
241
|
+
if not isinstance(actual_value, int | float) or isinstance(actual_value, bool):
|
|
242
|
+
return False
|
|
243
|
+
configured = tolerances.get(name, default_atol)
|
|
244
|
+
if not isinstance(configured, int | float) or isinstance(configured, bool):
|
|
245
|
+
raise IntegrityError(f"replay observable tolerance for {name!r} is not numeric")
|
|
246
|
+
if not _close(
|
|
247
|
+
float(expected_value), float(actual_value), atol=float(configured), rtol=rtol
|
|
248
|
+
):
|
|
249
|
+
return False
|
|
250
|
+
elif expected_value != actual_value:
|
|
251
|
+
return False
|
|
252
|
+
return True
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def _invariants_match(
|
|
256
|
+
expected: object, actual: tuple[InvariantResult, ...], policy: Mapping[str, object]
|
|
257
|
+
) -> tuple[bool, bool]:
|
|
258
|
+
expected_items = _sequence(expected, "invariants")
|
|
259
|
+
if len(expected_items) != len(actual):
|
|
260
|
+
return False, False
|
|
261
|
+
outcome_match = True
|
|
262
|
+
threshold_match = True
|
|
263
|
+
atol = _policy_number(policy, "invariant_measure_atol")
|
|
264
|
+
rtol = _policy_number(policy, "state_rtol")
|
|
265
|
+
for raw, observed in zip(expected_items, actual, strict=True):
|
|
266
|
+
item = _mapping(raw, "invariant")
|
|
267
|
+
outcome_match = outcome_match and item.get("name") == observed.name
|
|
268
|
+
outcome_match = outcome_match and item.get("passed") is observed.passed
|
|
269
|
+
measured = item.get("measured")
|
|
270
|
+
tolerance = item.get("tolerance")
|
|
271
|
+
if not isinstance(measured, int | float) or isinstance(measured, bool):
|
|
272
|
+
raise IntegrityError("replay invariant measured value must be numeric")
|
|
273
|
+
if not isinstance(tolerance, int | float) or isinstance(tolerance, bool):
|
|
274
|
+
raise IntegrityError("replay invariant tolerance must be numeric")
|
|
275
|
+
outcome_match = outcome_match and _close(
|
|
276
|
+
float(measured), observed.measured, atol=atol, rtol=rtol
|
|
277
|
+
)
|
|
278
|
+
threshold_match = threshold_match and float(tolerance) == observed.tolerance
|
|
279
|
+
return outcome_match, threshold_match
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def _event_match(
|
|
283
|
+
expected_metadata: Mapping[str, object],
|
|
284
|
+
actual_metadata: Mapping[str, object],
|
|
285
|
+
policy: Mapping[str, object],
|
|
286
|
+
) -> bool:
|
|
287
|
+
expected_events = _sequence(expected_metadata.get("events", []), "execution events")
|
|
288
|
+
actual_events = _sequence(actual_metadata.get("events", []), "actual execution events")
|
|
289
|
+
if len(expected_events) != len(actual_events):
|
|
290
|
+
return False
|
|
291
|
+
time_atol = _policy_number(policy, "event_time_atol")
|
|
292
|
+
state_atol = _policy_number(policy, "state_atol")
|
|
293
|
+
state_rtol = _policy_number(policy, "state_rtol")
|
|
294
|
+
for expected_raw, actual_raw in zip(expected_events, actual_events, strict=True):
|
|
295
|
+
expected = _mapping(expected_raw, "expected event")
|
|
296
|
+
actual = _mapping(actual_raw, "actual event")
|
|
297
|
+
if expected.get("name") != actual.get("name"):
|
|
298
|
+
return False
|
|
299
|
+
expected_times = _sequence(expected.get("times"), "expected event times")
|
|
300
|
+
actual_times = _sequence(actual.get("times"), "actual event times")
|
|
301
|
+
expected_states = _sequence(expected.get("states"), "expected event states")
|
|
302
|
+
actual_states = _sequence(actual.get("states"), "actual event states")
|
|
303
|
+
if len(expected_times) != len(actual_times) or len(expected_states) != len(actual_states):
|
|
304
|
+
return False
|
|
305
|
+
for expected_time, actual_time in zip(expected_times, actual_times, strict=True):
|
|
306
|
+
if not isinstance(expected_time, int | float) or not isinstance(
|
|
307
|
+
actual_time, int | float
|
|
308
|
+
):
|
|
309
|
+
raise IntegrityError("replay event times must be numeric")
|
|
310
|
+
if not _close(float(expected_time), float(actual_time), atol=time_atol, rtol=0.0):
|
|
311
|
+
return False
|
|
312
|
+
for expected_state, actual_state in zip(expected_states, actual_states, strict=True):
|
|
313
|
+
left = _state(expected_state, "expected event state")
|
|
314
|
+
right = _state(actual_state, "actual event state")
|
|
315
|
+
if len(left) != len(right) or not all(
|
|
316
|
+
_close(a, b, atol=state_atol, rtol=state_rtol)
|
|
317
|
+
for a, b in zip(left, right, strict=True)
|
|
318
|
+
):
|
|
319
|
+
return False
|
|
320
|
+
return True
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def _tolerance_comparison(
|
|
324
|
+
label: str,
|
|
325
|
+
expected: Mapping[str, object],
|
|
326
|
+
run: SimulationResult,
|
|
327
|
+
expected_identity: object,
|
|
328
|
+
policy: Mapping[str, object],
|
|
329
|
+
) -> Mapping[str, object]:
|
|
330
|
+
expected_trace = _sequence(expected.get("trace"), f"{label}.trace")
|
|
331
|
+
actual_trace = run.trace
|
|
332
|
+
max_unmatched = policy.get("max_unmatched_points")
|
|
333
|
+
if not isinstance(max_unmatched, int) or isinstance(max_unmatched, bool):
|
|
334
|
+
raise IntegrityError("replay comparison.max_unmatched_points must be an integer")
|
|
335
|
+
count_difference = abs(len(expected_trace) - len(actual_trace))
|
|
336
|
+
grid_match = count_difference <= max_unmatched
|
|
337
|
+
state_match = True
|
|
338
|
+
observations_match = True
|
|
339
|
+
time_atol = _policy_number(policy, "retained_grid_time_atol")
|
|
340
|
+
state_atol = _policy_number(policy, "state_atol")
|
|
341
|
+
state_rtol = _policy_number(policy, "state_rtol")
|
|
342
|
+
for raw, point in zip(expected_trace, actual_trace, strict=False):
|
|
343
|
+
expected_point = _mapping(raw, f"{label}.trace point")
|
|
344
|
+
expected_time = expected_point.get("time")
|
|
345
|
+
if not isinstance(expected_time, int | float) or isinstance(expected_time, bool):
|
|
346
|
+
raise IntegrityError(f"replay {label}.trace point time must be numeric")
|
|
347
|
+
grid_match = grid_match and _close(
|
|
348
|
+
float(expected_time), point.time, atol=time_atol, rtol=0.0
|
|
349
|
+
)
|
|
350
|
+
expected_state = _state(expected_point.get("state"), f"{label}.trace state")
|
|
351
|
+
state_match = (
|
|
352
|
+
state_match
|
|
353
|
+
and len(expected_state) == len(point.state)
|
|
354
|
+
and all(
|
|
355
|
+
_close(a, b, atol=state_atol, rtol=state_rtol)
|
|
356
|
+
for a, b in zip(expected_state, point.state, strict=False)
|
|
357
|
+
)
|
|
358
|
+
)
|
|
359
|
+
observations_match = observations_match and _observations_match(
|
|
360
|
+
expected_point.get("observations"), point.observations, policy
|
|
361
|
+
)
|
|
362
|
+
expected_final = _state(expected.get("final_state"), f"{label}.final_state")
|
|
363
|
+
endpoint_state_match = len(expected_final) == len(run.final_state) and all(
|
|
364
|
+
_close(a, b, atol=state_atol, rtol=state_rtol)
|
|
365
|
+
for a, b in zip(expected_final, run.final_state, strict=False)
|
|
366
|
+
)
|
|
367
|
+
expected_metadata = _mapping(expected.get("execution_metadata"), "execution metadata")
|
|
368
|
+
expected_endpoint = expected_metadata.get("termination_time")
|
|
369
|
+
actual_endpoint = run.execution_metadata.get("termination_time")
|
|
370
|
+
if not isinstance(expected_endpoint, int | float) or not isinstance(
|
|
371
|
+
actual_endpoint, int | float
|
|
372
|
+
):
|
|
373
|
+
raise IntegrityError("tolerance replay requires numeric termination_time evidence")
|
|
374
|
+
endpoint_time_match = _close(
|
|
375
|
+
float(expected_endpoint),
|
|
376
|
+
float(actual_endpoint),
|
|
377
|
+
atol=_policy_number(policy, "endpoint_time_atol"),
|
|
378
|
+
rtol=0.0,
|
|
379
|
+
)
|
|
380
|
+
invariant_match, invariant_threshold_match = _invariants_match(
|
|
381
|
+
expected.get("invariants"), run.invariants, policy
|
|
382
|
+
)
|
|
383
|
+
require_classifier = policy.get("require_classifier") is True
|
|
384
|
+
require_invariants = policy.get("require_invariants") is True
|
|
385
|
+
expected_success = policy.get("expected_solver_success")
|
|
386
|
+
actual_success = bool(run.execution_metadata.get("solver_success", True))
|
|
387
|
+
return {
|
|
388
|
+
"series": label,
|
|
389
|
+
"mode": "tolerance",
|
|
390
|
+
"classification_match": (run.classification == expected.get("classification"))
|
|
391
|
+
if require_classifier
|
|
392
|
+
else True,
|
|
393
|
+
"model_identity_match": run.model_identity == expected_identity,
|
|
394
|
+
"solver_success_match": actual_success is expected_success,
|
|
395
|
+
"retained_grid_match": grid_match,
|
|
396
|
+
"state_tolerance_match": state_match,
|
|
397
|
+
"observable_tolerance_match": observations_match,
|
|
398
|
+
"endpoint_state_match": endpoint_state_match,
|
|
399
|
+
"endpoint_time_match": endpoint_time_match,
|
|
400
|
+
"event_tolerance_match": _event_match(expected_metadata, run.execution_metadata, policy),
|
|
401
|
+
"invariant_result_match": invariant_match if require_invariants else True,
|
|
402
|
+
"invariant_threshold_match": invariant_threshold_match if require_invariants else True,
|
|
403
|
+
"artifact_trace_sha256_equal": run.trace_sha256 == expected.get("trace_sha256"),
|
|
404
|
+
"expected_environment": expected_metadata,
|
|
405
|
+
"actual_environment": dict(run.execution_metadata),
|
|
406
|
+
}
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
@dataclass(frozen=True, slots=True)
|
|
410
|
+
class ReplayVerification:
|
|
411
|
+
"""Replay verdict with exact or declared-tolerance comparisons."""
|
|
412
|
+
|
|
413
|
+
ok: bool
|
|
414
|
+
mode: str
|
|
415
|
+
model: str
|
|
416
|
+
baseline: SimulationResult
|
|
417
|
+
changed: SimulationResult | None
|
|
418
|
+
comparisons: tuple[Mapping[str, object], ...]
|
|
419
|
+
|
|
420
|
+
def as_dict(self) -> dict[str, object]:
|
|
421
|
+
return {
|
|
422
|
+
"schema_version": "2.0",
|
|
423
|
+
"status": "REPLAY VERIFIED" if self.ok else "REPLAY MISMATCH",
|
|
424
|
+
"comparison_mode": self.mode,
|
|
425
|
+
"model": self.model,
|
|
426
|
+
"ok": self.ok,
|
|
427
|
+
"comparisons": [dict(item) for item in self.comparisons],
|
|
428
|
+
"baseline": self.baseline.as_dict(),
|
|
429
|
+
"changed": self.changed.as_dict() if self.changed is not None else None,
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
|
|
433
|
+
def verify_replay(path: Path) -> ReplayVerification:
|
|
434
|
+
"""Re-execute a v1/v2 fixture and apply its explicit comparison policy."""
|
|
435
|
+
|
|
436
|
+
fixture = validate_fixture(path)
|
|
437
|
+
configuration = fixture.get("configuration")
|
|
438
|
+
if not isinstance(configuration, dict):
|
|
439
|
+
raise IntegrityError("replay configuration must be an object")
|
|
440
|
+
config = parse_config(canonical_json(configuration), f"replay fixture {path.name}")
|
|
441
|
+
expected_model = fixture.get("model")
|
|
442
|
+
if config.model != expected_model:
|
|
443
|
+
raise IntegrityError("replay model does not match embedded configuration")
|
|
444
|
+
expected_identity = fixture.get("model_identity")
|
|
445
|
+
if fixture.get("schema_version") == "1.0":
|
|
446
|
+
mode = "exact"
|
|
447
|
+
policy: Mapping[str, object] = {"mode": mode}
|
|
448
|
+
else:
|
|
449
|
+
policy = _mapping(fixture.get("comparison"), "comparison")
|
|
450
|
+
raw_mode = policy.get("mode")
|
|
451
|
+
if raw_mode not in {"exact", "tolerance"}:
|
|
452
|
+
raise IntegrityError("replay comparison.mode must be exact or tolerance")
|
|
453
|
+
mode = cast(str, raw_mode)
|
|
454
|
+
|
|
455
|
+
comparisons: list[Mapping[str, object]] = []
|
|
456
|
+
|
|
457
|
+
def execute(label: str, raw: object) -> SimulationResult:
|
|
458
|
+
expected = _mapping(raw, label)
|
|
459
|
+
run = simulate(
|
|
460
|
+
config,
|
|
461
|
+
parameters_override=_float_mapping(expected.get("parameters"), f"{label}.parameters"),
|
|
462
|
+
initial_override=_state(expected.get("initial_state"), f"{label}.initial_state"),
|
|
463
|
+
)
|
|
464
|
+
if mode == "exact":
|
|
465
|
+
comparisons.append(
|
|
466
|
+
{
|
|
467
|
+
"series": label,
|
|
468
|
+
"mode": "exact",
|
|
469
|
+
"classification_match": run.classification == expected.get("classification"),
|
|
470
|
+
"trace_hash_match": run.trace_sha256 == expected.get("trace_sha256"),
|
|
471
|
+
"model_identity_match": run.model_identity == expected_identity,
|
|
472
|
+
"expected_trace_sha256": expected.get("trace_sha256"),
|
|
473
|
+
"actual_trace_sha256": run.trace_sha256,
|
|
474
|
+
}
|
|
475
|
+
)
|
|
476
|
+
else:
|
|
477
|
+
comparisons.append(
|
|
478
|
+
_tolerance_comparison(label, expected, run, expected_identity, policy)
|
|
479
|
+
)
|
|
480
|
+
return run
|
|
481
|
+
|
|
482
|
+
baseline = execute("baseline", fixture.get("baseline"))
|
|
483
|
+
changed_raw = fixture.get("changed")
|
|
484
|
+
changed = execute("changed", changed_raw) if changed_raw is not None else None
|
|
485
|
+
ok = all(
|
|
486
|
+
all(value is True for key, value in item.items() if key.endswith("_match"))
|
|
487
|
+
for item in comparisons
|
|
488
|
+
)
|
|
489
|
+
return ReplayVerification(
|
|
490
|
+
ok=ok,
|
|
491
|
+
mode=mode,
|
|
492
|
+
model=config.model,
|
|
493
|
+
baseline=baseline,
|
|
494
|
+
changed=changed,
|
|
495
|
+
comparisons=tuple(comparisons),
|
|
496
|
+
)
|
phaseprobe/reporting.py
ADDED
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
"""Offline terminal, JSON, and self-contained HTML evidence reports."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import html
|
|
6
|
+
import json
|
|
7
|
+
from collections.abc import Mapping, Sequence
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import cast
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _mapping(value: object) -> Mapping[str, object]:
|
|
13
|
+
return cast(Mapping[str, object], value) if isinstance(value, dict) else {}
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _sequence(value: object) -> Sequence[object]:
|
|
17
|
+
return cast(Sequence[object], value) if isinstance(value, list) else ()
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _format_float(value: object) -> str:
|
|
21
|
+
return f"{value:.10g}" if isinstance(value, float) else str(value)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _int_value(value: object) -> int:
|
|
25
|
+
return value if isinstance(value, int) and not isinstance(value, bool) else 0
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def terminal_report(
|
|
29
|
+
data: Mapping[str, object],
|
|
30
|
+
*,
|
|
31
|
+
replay: str | None = None,
|
|
32
|
+
generated_test: str | None = None,
|
|
33
|
+
) -> str:
|
|
34
|
+
"""Lead with the result and present compact, scientifically qualified evidence."""
|
|
35
|
+
|
|
36
|
+
status = str(data.get("status", "PHASEPROBE RESULT"))
|
|
37
|
+
model = str(data.get("model", "unknown"))
|
|
38
|
+
baseline = _mapping(data.get("baseline"))
|
|
39
|
+
changed = _mapping(data.get("changed"))
|
|
40
|
+
finding = _mapping(data.get("finding"))
|
|
41
|
+
lines = [status, "", f"Model: {model}"]
|
|
42
|
+
kind = finding.get("kind")
|
|
43
|
+
if kind is not None:
|
|
44
|
+
lines.append(f"Evidence: {kind}")
|
|
45
|
+
parameter = finding.get("parameter") or finding.get("dimension")
|
|
46
|
+
if parameter is not None:
|
|
47
|
+
lines.append(f"Search dimension: {parameter}")
|
|
48
|
+
bracket = _sequence(finding.get("stable_bracket"))
|
|
49
|
+
if len(bracket) == 2:
|
|
50
|
+
lines.append(f"Stable bracket: {_format_float(bracket[0])} .. {_format_float(bracket[1])}")
|
|
51
|
+
smallest = finding.get("smallest_reproducible_change_found")
|
|
52
|
+
if smallest is not None:
|
|
53
|
+
lines.append(f"Smallest reproducible change found: {_format_float(smallest)}")
|
|
54
|
+
lines.extend(
|
|
55
|
+
[
|
|
56
|
+
f"Baseline regime: {baseline.get('classification', 'n/a')}",
|
|
57
|
+
f"Changed regime: {changed.get('classification', 'n/a') if changed else 'n/a'}",
|
|
58
|
+
]
|
|
59
|
+
)
|
|
60
|
+
violations = _int_value(baseline.get("invariant_violations", 0))
|
|
61
|
+
if changed:
|
|
62
|
+
violations += _int_value(changed.get("invariant_violations", 0))
|
|
63
|
+
lines.append(f"Invariant violations: {violations}")
|
|
64
|
+
lines.append(f"Repeatable: {str(bool(data.get('reproducible', False))).lower()}")
|
|
65
|
+
if replay is not None:
|
|
66
|
+
lines.append(f"Replay: {replay}")
|
|
67
|
+
if generated_test is not None:
|
|
68
|
+
lines.append(f"Generated test: {generated_test}")
|
|
69
|
+
minimality = finding.get("minimality_statement")
|
|
70
|
+
if minimality is not None:
|
|
71
|
+
lines.extend(["", f"Scope: {minimality}"])
|
|
72
|
+
return "\n".join(lines)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def json_report(data: Mapping[str, object]) -> str:
|
|
76
|
+
"""Render the complete versioned JSON evidence document."""
|
|
77
|
+
|
|
78
|
+
return json.dumps(data, allow_nan=False, indent=2, sort_keys=True) + "\n"
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def html_report(data: Mapping[str, object]) -> str:
|
|
82
|
+
"""Render a standalone offline report with no scripts, fonts, or CDN assets."""
|
|
83
|
+
|
|
84
|
+
status = html.escape(str(data.get("status", "PhaseProbe result")))
|
|
85
|
+
model = html.escape(str(data.get("model", "unknown")))
|
|
86
|
+
baseline = _mapping(data.get("baseline"))
|
|
87
|
+
changed = _mapping(data.get("changed"))
|
|
88
|
+
finding = _mapping(data.get("finding"))
|
|
89
|
+
history = _sequence(data.get("history"))
|
|
90
|
+
config = _mapping(data.get("configuration"))
|
|
91
|
+
replay_mode = html.escape(str(baseline.get("replay_mode", "exact")))
|
|
92
|
+
|
|
93
|
+
history_rows: list[str] = []
|
|
94
|
+
for item in history:
|
|
95
|
+
row = _mapping(item)
|
|
96
|
+
phase = html.escape(str(row.get("phase", "")))
|
|
97
|
+
value = row.get("value", row.get("delta", ""))
|
|
98
|
+
classification = html.escape(str(row.get("classification", "")))
|
|
99
|
+
interval = ""
|
|
100
|
+
if "low" in row and "high" in row:
|
|
101
|
+
interval = f"{_format_float(row['low'])} .. {_format_float(row['high'])}"
|
|
102
|
+
history_rows.append(
|
|
103
|
+
"<tr>"
|
|
104
|
+
f"<td>{phase}</td><td>{html.escape(_format_float(value))}</td>"
|
|
105
|
+
f"<td>{classification}</td><td>{html.escape(interval)}</td>"
|
|
106
|
+
"</tr>"
|
|
107
|
+
)
|
|
108
|
+
if not history_rows:
|
|
109
|
+
history_rows.append('<tr><td colspan="4">No refinement history for this check.</td></tr>')
|
|
110
|
+
|
|
111
|
+
finding_json = html.escape(json.dumps(finding, allow_nan=False, indent=2, sort_keys=True))
|
|
112
|
+
config_json = html.escape(json.dumps(config, allow_nan=False, indent=2, sort_keys=True))
|
|
113
|
+
limitations = (
|
|
114
|
+
"This report contains bounded numerical evidence. It does not prove global minimality, "
|
|
115
|
+
"an exact bifurcation point, chaos, a formal Lyapunov exponent, or scientific validity "
|
|
116
|
+
"outside the declared model, integration settings, search space, and tolerances."
|
|
117
|
+
)
|
|
118
|
+
return f"""<!doctype html>
|
|
119
|
+
<html lang="en">
|
|
120
|
+
<head>
|
|
121
|
+
<meta charset="utf-8">
|
|
122
|
+
<meta name="viewport" content="width=device-width, initial-scale=1">
|
|
123
|
+
<title>PhaseProbe — {status}</title>
|
|
124
|
+
<style>
|
|
125
|
+
:root {{ color-scheme: light dark; --ink:#16202a; --paper:#f7f4ed; --accent:#ef5b35; --cool:#167d8d; }}
|
|
126
|
+
* {{ box-sizing:border-box; }}
|
|
127
|
+
body {{ margin:0; font:16px/1.5 ui-monospace,SFMono-Regular,Consolas,monospace; background:var(--paper); color:var(--ink); }}
|
|
128
|
+
main {{ max-width:1080px; margin:auto; padding:36px 24px 64px; }}
|
|
129
|
+
h1 {{ margin:.2rem 0; font-size:clamp(1.8rem,5vw,3.4rem); line-height:1.05; }}
|
|
130
|
+
.eyebrow {{ color:var(--cool); font-weight:700; letter-spacing:.08em; text-transform:uppercase; }}
|
|
131
|
+
.result {{ border-left:8px solid var(--accent); padding:18px 22px; background:#fff; box-shadow:0 10px 30px #16303b18; }}
|
|
132
|
+
.grid {{ display:grid; grid-template-columns:repeat(auto-fit,minmax(260px,1fr)); gap:16px; margin:22px 0; }}
|
|
133
|
+
.card {{ border:1px solid #16202a2c; border-radius:8px; padding:16px; background:#ffffffb8; }}
|
|
134
|
+
table {{ width:100%; border-collapse:collapse; margin:12px 0 24px; }}
|
|
135
|
+
th,td {{ padding:9px; border-bottom:1px solid #16202a2c; text-align:left; vertical-align:top; }}
|
|
136
|
+
pre {{ overflow:auto; padding:16px; border-radius:8px; background:#10232b; color:#eaf5f2; }}
|
|
137
|
+
.limit {{ border:1px solid #ef5b35; padding:14px; background:#fff4ef; }}
|
|
138
|
+
@media (prefers-color-scheme:dark) {{ body {{ --paper:#0f1b20; --ink:#eaf5f2; }} .result,.card {{ background:#16272e; }} .limit {{ background:#36221e; }} }}
|
|
139
|
+
</style>
|
|
140
|
+
</head>
|
|
141
|
+
<body><main>
|
|
142
|
+
<p class="eyebrow">PhaseProbe evidence report · schema {html.escape(str(data.get("schema_version", "unknown")))}</p>
|
|
143
|
+
<section class="result"><h1>{status}</h1><p>Model: <strong>{model}</strong></p></section>
|
|
144
|
+
<section class="grid">
|
|
145
|
+
<article class="card"><h2>Baseline</h2><p>{html.escape(str(baseline.get("classification", "n/a")))}</p><p>Trace SHA-256: <code>{html.escape(str(baseline.get("trace_sha256", "n/a")))}</code></p></article>
|
|
146
|
+
<article class="card"><h2>Changed</h2><p>{html.escape(str(changed.get("classification", "n/a")))}</p><p>Trace SHA-256: <code>{html.escape(str(changed.get("trace_sha256", "n/a")))}</code></p></article>
|
|
147
|
+
</section>
|
|
148
|
+
<h2>Finding</h2><pre>{finding_json}</pre>
|
|
149
|
+
<h2>Search and refinement history</h2>
|
|
150
|
+
<table><thead><tr><th>Phase</th><th>Value / delta</th><th>Classification</th><th>Bracket</th></tr></thead><tbody>{"".join(history_rows)}</tbody></table>
|
|
151
|
+
<h2>Configuration</h2><pre>{config_json}</pre>
|
|
152
|
+
<h2>Replay and generated test</h2><p>The run directory contains a versioned <code>replay.json</code> fixture with declared <strong>{replay_mode}</strong> comparison. Artifact SHA-256 integrity is retained in both modes. The <code>generate-test</code> command validates and copies that fixture into a fixed pytest template.</p>
|
|
153
|
+
<h2>Scientific limitations</h2><p class="limit">{html.escape(limitations)}</p>
|
|
154
|
+
</main></body></html>
|
|
155
|
+
"""
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def regenerate_reports(run_directory: Path) -> tuple[Path, Path]:
|
|
159
|
+
"""Regenerate JSON and HTML evidence from a bounded run directory."""
|
|
160
|
+
|
|
161
|
+
run_path = run_directory / "run.json"
|
|
162
|
+
try:
|
|
163
|
+
parsed = json.loads(run_path.read_text(encoding="utf-8"))
|
|
164
|
+
except (OSError, json.JSONDecodeError) as exc:
|
|
165
|
+
raise ValueError(f"cannot load run evidence from {run_path}: {exc}") from exc
|
|
166
|
+
if not isinstance(parsed, dict):
|
|
167
|
+
raise ValueError(f"run evidence in {run_path} is not a JSON object")
|
|
168
|
+
data = cast(dict[str, object], parsed)
|
|
169
|
+
json_path = run_directory / "report.json"
|
|
170
|
+
html_path = run_directory / "report.html"
|
|
171
|
+
json_path.write_text(json_report(data), encoding="utf-8")
|
|
172
|
+
html_path.write_text(html_report(data), encoding="utf-8")
|
|
173
|
+
return json_path, html_path
|