runproof-engine 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,197 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import dataclass, field
4
+ from typing import Any
5
+
6
+ from .utils import safe_value
7
+
8
+
9
+ @dataclass
10
+ class Difference:
11
+ area: str
12
+ path: str
13
+ kind: str
14
+ before: Any
15
+ after: Any
16
+ explanation: str
17
+ confidence: str = "evidence"
18
+
19
+ def to_dict(self) -> dict[str, Any]:
20
+ return {
21
+ "area": self.area,
22
+ "path": self.path,
23
+ "kind": self.kind,
24
+ "before": safe_value(self.before),
25
+ "after": safe_value(self.after),
26
+ "explanation": self.explanation,
27
+ "confidence": self.confidence,
28
+ }
29
+
30
+
31
+ @dataclass
32
+ class RunDiff:
33
+ left_run_id: str
34
+ right_run_id: str
35
+ differences: list[Difference] = field(default_factory=list)
36
+
37
+ @property
38
+ def identical(self) -> bool:
39
+ return not self.differences
40
+
41
+ @property
42
+ def first_divergent_step(self) -> str | None:
43
+ for difference in self.differences:
44
+ if difference.area == "steps":
45
+ return difference.path.split(".")[0]
46
+ return None
47
+
48
+ def to_dict(self) -> dict[str, Any]:
49
+ return {
50
+ "left_run_id": self.left_run_id,
51
+ "right_run_id": self.right_run_id,
52
+ "identical": self.identical,
53
+ "first_divergent_step": self.first_divergent_step,
54
+ "differences": [difference.to_dict() for difference in self.differences],
55
+ }
56
+
57
+ def render(self) -> str:
58
+ if self.identical:
59
+ return f"Runs {self.left_run_id} and {self.right_run_id} are observably identical."
60
+ lines = [
61
+ f"Run diff: {self.left_run_id} -> {self.right_run_id}",
62
+ f"Differences: {len(self.differences)}",
63
+ ]
64
+ if self.first_divergent_step:
65
+ lines.append(f"First divergent step: {self.first_divergent_step}")
66
+ for difference in self.differences:
67
+ lines.append(
68
+ f"- [{difference.area}] {difference.path}: {difference.explanation} "
69
+ f"(confidence={difference.confidence})"
70
+ )
71
+ return "\n".join(lines)
72
+
73
+
74
+ def compare_manifests(left: dict[str, Any], right: dict[str, Any]) -> RunDiff:
75
+ left_run = left.get("run", {})
76
+ right_run = right.get("run", {})
77
+ result = RunDiff(
78
+ left_run_id=str(left_run.get("run_id", "unknown")),
79
+ right_run_id=str(right_run.get("run_id", "unknown")),
80
+ )
81
+ _compare_inputs(left.get("inputs", []), right.get("inputs", []), result)
82
+ _compare_steps(left.get("steps", []), right.get("steps", []), result)
83
+ _compare_outputs(left.get("outputs", []), right.get("outputs", []), result)
84
+ _compare_checks(left.get("checks", []), right.get("checks", []), result)
85
+ _compare_environment(left.get("environment"), right.get("environment"), result)
86
+ if left_run.get("status") != right_run.get("status"):
87
+ result.differences.append(Difference(
88
+ area="run",
89
+ path="status",
90
+ kind="changed",
91
+ before=left_run.get("status"),
92
+ after=right_run.get("status"),
93
+ explanation="run status changed",
94
+ ))
95
+ return result
96
+
97
+
98
+ def _index(records: list[dict[str, Any]], key: str = "name") -> dict[str, dict[str, Any]]:
99
+ return {str(item.get(key, index)): item for index, item in enumerate(records)}
100
+
101
+
102
+ def _compare_inputs(left: list[dict[str, Any]], right: list[dict[str, Any]], result: RunDiff) -> None:
103
+ left_map, right_map = _index(left), _index(right)
104
+ for name in sorted(set(left_map) | set(right_map)):
105
+ if name not in left_map:
106
+ result.differences.append(Difference("inputs", name, "added", None, right_map[name], "input was added"))
107
+ continue
108
+ if name not in right_map:
109
+ result.differences.append(Difference("inputs", name, "removed", left_map[name], None, "input was removed"))
110
+ continue
111
+ before, after = left_map[name], right_map[name]
112
+ if before.get("sha256") != after.get("sha256"):
113
+ result.differences.append(Difference(
114
+ "inputs", f"{name}.sha256", "changed", before.get("sha256"), after.get("sha256"),
115
+ "input content changed according to its SHA-256 fingerprint",
116
+ ))
117
+ for field_name in ("size_bytes", "rows", "columns", "schema"):
118
+ if before.get(field_name) != after.get(field_name) and (field_name in before or field_name in after):
119
+ result.differences.append(Difference(
120
+ "inputs", f"{name}.{field_name}", "changed", before.get(field_name), after.get(field_name),
121
+ f"input metadata field '{field_name}' changed",
122
+ ))
123
+
124
+
125
+ def _compare_steps(left: list[dict[str, Any]], right: list[dict[str, Any]], result: RunDiff) -> None:
126
+ left_map, right_map = _index(left), _index(right)
127
+ for name in sorted(set(left_map) | set(right_map)):
128
+ if name not in left_map:
129
+ result.differences.append(Difference("steps", name, "added", None, right_map[name], "step was added"))
130
+ continue
131
+ if name not in right_map:
132
+ result.differences.append(Difference("steps", name, "removed", left_map[name], None, "step was removed"))
133
+ continue
134
+ before, after = left_map[name], right_map[name]
135
+ if before.get("status") != after.get("status"):
136
+ result.differences.append(Difference("steps", f"{name}.status", "changed", before.get("status"), after.get("status"), "step status changed"))
137
+ before_output = (before.get("output") or {}).get("fingerprint")
138
+ after_output = (after.get("output") or {}).get("fingerprint")
139
+ if before_output != after_output:
140
+ result.differences.append(Difference(
141
+ "steps", f"{name}.output.fingerprint", "changed", before_output, after_output,
142
+ "step output changed; downstream artifacts may be affected",
143
+ ))
144
+ before_function = (before.get("function") or {}).get("source_sha256")
145
+ after_function = (after.get("function") or {}).get("source_sha256")
146
+ if before_function != after_function:
147
+ result.differences.append(Difference(
148
+ "steps", f"{name}.function.source_sha256", "changed", before_function, after_function,
149
+ "step source fingerprint changed",
150
+ ))
151
+
152
+
153
+ def _compare_outputs(left: list[dict[str, Any]], right: list[dict[str, Any]], result: RunDiff) -> None:
154
+ left_map, right_map = _index(left), _index(right)
155
+ for name in sorted(set(left_map) | set(right_map)):
156
+ if name not in left_map:
157
+ result.differences.append(Difference("outputs", name, "added", None, right_map[name], "output was added"))
158
+ elif name not in right_map:
159
+ result.differences.append(Difference("outputs", name, "removed", left_map[name], None, "output was removed"))
160
+ elif left_map[name].get("sha256") != right_map[name].get("sha256"):
161
+ result.differences.append(Difference(
162
+ "outputs", f"{name}.sha256", "changed", left_map[name].get("sha256"), right_map[name].get("sha256"),
163
+ "artifact content changed",
164
+ ))
165
+
166
+
167
+ def _compare_checks(left: list[dict[str, Any]], right: list[dict[str, Any]], result: RunDiff) -> None:
168
+ left_map, right_map = _index(left), _index(right)
169
+ for name in sorted(set(left_map) | set(right_map)):
170
+ if name not in left_map or name not in right_map:
171
+ result.differences.append(Difference("checks", name, "changed", left_map.get(name), right_map.get(name), "check set changed"))
172
+ elif left_map[name].get("passed") != right_map[name].get("passed"):
173
+ result.differences.append(Difference(
174
+ "checks", f"{name}.passed", "changed", left_map[name].get("passed"), right_map[name].get("passed"),
175
+ "check outcome changed",
176
+ ))
177
+
178
+
179
+ def _compare_environment(left: dict[str, Any] | None, right: dict[str, Any] | None, result: RunDiff) -> None:
180
+ if left is None or right is None:
181
+ if left != right:
182
+ result.differences.append(Difference("environment", "snapshot", "changed", bool(left), bool(right), "environment snapshot availability changed"))
183
+ return
184
+ for field_name in ("python_version", "platform", "machine"):
185
+ if left.get(field_name) != right.get(field_name):
186
+ result.differences.append(Difference(
187
+ "environment", field_name, "changed", left.get(field_name), right.get(field_name),
188
+ f"environment field '{field_name}' changed",
189
+ ))
190
+ left_packages = {item.get("name"): item.get("version") for item in left.get("packages", [])}
191
+ right_packages = {item.get("name"): item.get("version") for item in right.get("packages", [])}
192
+ for name in sorted(set(left_packages) | set(right_packages)):
193
+ if left_packages.get(name) != right_packages.get(name):
194
+ result.differences.append(Difference(
195
+ "environment", f"packages.{name}", "changed", left_packages.get(name), right_packages.get(name),
196
+ "installed package version changed",
197
+ ))
@@ -0,0 +1,29 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import dataclass, field
4
+ from typing import Any
5
+
6
+
7
+ class PolicyDenied(PermissionError):
8
+ """Raised when a requested action is not allowed by the run policy."""
9
+
10
+
11
+ @dataclass
12
+ class Policy:
13
+ allowed_actions: set[str] = field(default_factory=lambda: {"read", "write_artifact", "execute_python"})
14
+ approval_required: set[str] = field(default_factory=lambda: {"install_package", "network", "upload", "publish", "delete", "shell"})
15
+ denied_actions: set[str] = field(default_factory=set)
16
+
17
+ def authorize(self, action: str, *, approved: bool = False, target: str | None = None) -> dict[str, Any]:
18
+ action = str(action)
19
+ if action in self.denied_actions:
20
+ raise PolicyDenied(f"action denied by policy: {action}")
21
+ if action not in self.allowed_actions and action not in self.approval_required:
22
+ raise PolicyDenied(f"action is not declared in policy: {action}")
23
+ if action in self.approval_required and not approved:
24
+ raise PolicyDenied(f"approval required for action: {action}")
25
+ return {"action": action, "target": target, "approved": approved}
26
+
27
+
28
+ def safe_default_policy() -> Policy:
29
+ return Policy()
File without changes
@@ -0,0 +1,131 @@
1
+ from __future__ import annotations
2
+
3
+ import os
4
+ from collections.abc import Callable
5
+ from dataclasses import dataclass, field
6
+ from pathlib import Path
7
+ from typing import Any
8
+
9
+ from .diff import RunDiff, compare_manifests
10
+ from .utils import file_metadata, read_json, sha256_file
11
+
12
+
13
+ @dataclass
14
+ class ReplayReport:
15
+ status: str
16
+ run_id: str
17
+ mode: str
18
+ reasons: list[str] = field(default_factory=list)
19
+ compared_run_id: str | None = None
20
+
21
+ def to_dict(self) -> dict[str, Any]:
22
+ return {
23
+ "status": self.status,
24
+ "run_id": self.run_id,
25
+ "mode": self.mode,
26
+ "reasons": self.reasons,
27
+ "compared_run_id": self.compared_run_id,
28
+ }
29
+
30
+ def __str__(self) -> str:
31
+ suffix = f"; reasons={self.reasons}" if self.reasons else ""
32
+ return f"ReplayReport(status={self.status!r}, run_id={self.run_id!r}, mode={self.mode!r}{suffix})"
33
+
34
+
35
+ class LoadedRun:
36
+ def __init__(self, artifact_dir: str | os.PathLike[str]) -> None:
37
+ self.artifact_dir = Path(artifact_dir).expanduser().resolve()
38
+ self.manifest_path = self.artifact_dir / "manifest.json"
39
+ if not self.manifest_path.is_file():
40
+ raise FileNotFoundError(f"RunProof manifest not found: {self.manifest_path}")
41
+ self.manifest = read_json(self.manifest_path)
42
+
43
+ @property
44
+ def run_id(self) -> str:
45
+ return str(self.manifest.get("run", {}).get("run_id", self.artifact_dir.name))
46
+
47
+ @property
48
+ def name(self) -> str:
49
+ return str(self.manifest.get("run", {}).get("name", self.artifact_dir.parent.name))
50
+
51
+ @property
52
+ def status(self) -> str:
53
+ return str(self.manifest.get("run", {}).get("status", "unknown"))
54
+
55
+ def verify_integrity(self) -> ReplayReport:
56
+ reasons: list[str] = []
57
+ for record in self.manifest.get("inputs", []):
58
+ source = Path(record.get("path", ""))
59
+ if source.is_file():
60
+ if file_metadata(source).get("sha256") != record.get("sha256"):
61
+ reasons.append(f"input changed: {record.get('name', source.name)}")
62
+ continue
63
+ captured = record.get("captured_copy")
64
+ captured_path = (self.artifact_dir / captured).resolve() if captured else None
65
+ if captured_path and captured_path.is_file() and file_metadata(captured_path).get("sha256") == record.get("sha256"):
66
+ continue
67
+ reasons.append(f"input missing: {source}")
68
+ for record in self.manifest.get("outputs", []):
69
+ output = (self.artifact_dir / record.get("path", "")).resolve()
70
+ if not output.is_file():
71
+ reasons.append(f"output missing: {output}")
72
+ elif file_metadata(output).get("sha256") != record.get("sha256"):
73
+ reasons.append(f"artifact changed: {record.get('name', output.name)}")
74
+ integrity_file = self.artifact_dir / "integrity.json"
75
+ if integrity_file.is_file():
76
+ try:
77
+ integrity = read_json(integrity_file)
78
+ for record in integrity.get("files", []):
79
+ artifact = (self.artifact_dir / record["path"]).resolve()
80
+ if not artifact.is_file():
81
+ reasons.append(f"integrity file missing: {record['path']}")
82
+ elif sha256_file(artifact) != record.get("sha256"):
83
+ reasons.append(f"integrity mismatch: {record['path']}")
84
+ except (KeyError, TypeError, ValueError):
85
+ reasons.append("integrity manifest is invalid")
86
+ status = "verified" if not reasons else "non_reproducible"
87
+ return ReplayReport(status=status, run_id=self.run_id, mode="integrity", reasons=reasons)
88
+
89
+ def replay(
90
+ self,
91
+ *,
92
+ mode: str = "strict",
93
+ runner: Callable[[LoadedRun], str | os.PathLike[str] | LoadedRun] | None = None,
94
+ ) -> ReplayReport:
95
+ """Validate replay prerequisites and optionally compare a user-supplied re-execution.
96
+
97
+ RunProof does not guess how to reconstruct arbitrary Python state. A runner is
98
+ required to execute the original workflow again; without it, strict mode only
99
+ verifies captured evidence and reports that no new execution was performed.
100
+ """
101
+ if mode not in {"strict", "fresh"}:
102
+ raise ValueError("mode must be 'strict' or 'fresh'")
103
+ if runner is None:
104
+ integrity = self.verify_integrity()
105
+ if integrity.status != "verified":
106
+ integrity.mode = mode
107
+ return integrity
108
+ return ReplayReport(
109
+ status="replay_ready",
110
+ run_id=self.run_id,
111
+ mode=mode,
112
+ reasons=["integrity verified; pass runner=... to execute and compare the workflow"],
113
+ )
114
+ new_artifact = runner(self)
115
+ other = new_artifact if isinstance(new_artifact, LoadedRun) else load_run(new_artifact)
116
+ comparison = self.diff(other)
117
+ return ReplayReport(
118
+ status="verified" if comparison.identical else "non_reproducible",
119
+ run_id=self.run_id,
120
+ mode=mode,
121
+ reasons=[] if comparison.identical else [difference.explanation for difference in comparison.differences],
122
+ compared_run_id=other.run_id,
123
+ )
124
+
125
+ def diff(self, other: LoadedRun | str | os.PathLike[str]) -> RunDiff:
126
+ other_run = other if isinstance(other, LoadedRun) else load_run(other)
127
+ return compare_manifests(self.manifest, other_run.manifest)
128
+
129
+
130
+ def load_run(artifact_dir: str | os.PathLike[str]) -> LoadedRun:
131
+ return LoadedRun(artifact_dir)
@@ -0,0 +1,197 @@
1
+ from __future__ import annotations
2
+
3
+ import csv
4
+ import dataclasses
5
+ import hashlib
6
+ import inspect
7
+ import json
8
+ import os
9
+ import platform
10
+ import sys
11
+ from datetime import datetime, timezone
12
+ from importlib import metadata
13
+ from pathlib import Path
14
+ from typing import Any
15
+
16
+ SECRET_MARKERS = (
17
+ "secret",
18
+ "password",
19
+ "passwd",
20
+ "token",
21
+ "api_key",
22
+ "apikey",
23
+ "private_key",
24
+ "authorization",
25
+ "credential",
26
+ )
27
+
28
+
29
+ def utc_now() -> str:
30
+ return datetime.now(timezone.utc).replace(microsecond=0).isoformat().replace("+00:00", "Z")
31
+
32
+
33
+ def sha256_bytes(value: bytes) -> str:
34
+ return hashlib.sha256(value).hexdigest()
35
+
36
+
37
+ def sha256_file(path: str | os.PathLike[str], chunk_size: int = 1024 * 1024) -> str:
38
+ digest = hashlib.sha256()
39
+ with Path(path).open("rb") as handle:
40
+ while chunk := handle.read(chunk_size):
41
+ digest.update(chunk)
42
+ return digest.hexdigest()
43
+
44
+
45
+ def canonical_json(value: Any) -> str:
46
+ return json.dumps(value, sort_keys=True, ensure_ascii=False, separators=(",", ":"), default=str)
47
+
48
+
49
+ def fingerprint(value: Any) -> str:
50
+ return sha256_bytes(canonical_json(safe_value(value, max_items=200)).encode("utf-8"))
51
+
52
+
53
+ def _redacted_key(key: Any) -> bool:
54
+ normalized = str(key).lower().replace("-", "_")
55
+ return any(marker in normalized for marker in SECRET_MARKERS)
56
+
57
+
58
+ def safe_value(value: Any, *, max_items: int = 100, max_text: int = 500) -> Any:
59
+ """Return a bounded, JSON-safe representation without exposing likely secrets."""
60
+ if value is None or isinstance(value, (bool, int, float)):
61
+ return value
62
+ if isinstance(value, (str, Path)):
63
+ text = str(value)
64
+ return text if len(text) <= max_text else text[:max_text] + "…"
65
+ if dataclasses.is_dataclass(value):
66
+ return safe_value(dataclasses.asdict(value), max_items=max_items, max_text=max_text)
67
+ if isinstance(value, dict):
68
+ items = list(value.items())[:max_items]
69
+ return {
70
+ str(key): "[REDACTED]" if _redacted_key(key) else safe_value(item, max_items=max_items, max_text=max_text)
71
+ for key, item in items
72
+ }
73
+ if isinstance(value, (list, tuple, set)):
74
+ values = list(value)
75
+ bounded = [safe_value(item, max_items=max_items, max_text=max_text) for item in values[:max_items]]
76
+ if len(values) > max_items:
77
+ bounded.append(f"… {len(values) - max_items} more items")
78
+ return bounded
79
+ if hasattr(value, "to_dict") and callable(value.to_dict):
80
+ try:
81
+ return safe_value(value.to_dict(), max_items=max_items, max_text=max_text)
82
+ except (AttributeError, KeyError, TypeError, ValueError, RuntimeError):
83
+ pass
84
+ if hasattr(value, "__dict__"):
85
+ try:
86
+ return {
87
+ "type": f"{type(value).__module__}.{type(value).__qualname__}",
88
+ "attributes": safe_value(vars(value), max_items=max_items, max_text=max_text),
89
+ }
90
+ except (AttributeError, TypeError, ValueError, RuntimeError):
91
+ pass
92
+ return {
93
+ "type": f"{type(value).__module__}.{type(value).__qualname__}",
94
+ "repr": str(value)[:max_text],
95
+ }
96
+
97
+
98
+ def summarize(value: Any) -> dict[str, Any]:
99
+ """Describe a runtime value without serializing an unbounded payload."""
100
+ summary: dict[str, Any] = {
101
+ "type": f"{type(value).__module__}.{type(value).__qualname__}",
102
+ "fingerprint": fingerprint(value),
103
+ }
104
+ if value is None or isinstance(value, (bool, int, float, str)):
105
+ summary["value"] = safe_value(value)
106
+ elif isinstance(value, (list, tuple, set, dict)):
107
+ summary["length"] = len(value)
108
+ if isinstance(value, dict):
109
+ summary["keys"] = [str(key) for key in list(value.keys())[:100]]
110
+ else:
111
+ for attribute in ("shape", "columns", "dtypes", "size"):
112
+ if hasattr(value, attribute):
113
+ try:
114
+ summary[attribute] = safe_value(getattr(value, attribute))
115
+ except (AttributeError, KeyError, TypeError, ValueError, RuntimeError):
116
+ pass
117
+ return summary
118
+
119
+
120
+ def function_descriptor(function: Any) -> dict[str, Any]:
121
+ descriptor: dict[str, Any] = {
122
+ "module": getattr(function, "__module__", None),
123
+ "qualname": getattr(function, "__qualname__", repr(function)),
124
+ }
125
+ try:
126
+ source = inspect.getsource(function)
127
+ descriptor["source_sha256"] = sha256_bytes(source.encode("utf-8"))
128
+ except (OSError, TypeError):
129
+ descriptor["source_sha256"] = None
130
+ return descriptor
131
+
132
+
133
+ def environment_snapshot() -> dict[str, Any]:
134
+ packages: list[dict[str, str]] = []
135
+ try:
136
+ for distribution in metadata.distributions():
137
+ name = distribution.metadata.get("Name")
138
+ version = distribution.version
139
+ if name and version:
140
+ packages.append({"name": name.lower(), "version": version})
141
+ packages.sort(key=lambda item: item["name"])
142
+ except (ImportError, OSError, RuntimeError, ValueError):
143
+ packages = []
144
+ return {
145
+ "python_version": sys.version,
146
+ "python_executable": sys.executable,
147
+ "platform": platform.platform(),
148
+ "machine": platform.machine(),
149
+ "packages": packages,
150
+ }
151
+
152
+
153
+ def _tabular_metadata(file_path: Path) -> dict[str, Any]:
154
+ suffix = file_path.suffix.lower()
155
+ if suffix == ".csv":
156
+ try:
157
+ with file_path.open("r", encoding="utf-8-sig", newline="") as handle:
158
+ reader = csv.reader(handle)
159
+ columns = next(reader, [])
160
+ rows = sum(1 for _ in reader)
161
+ return {"rows": rows, "columns": columns, "schema": {column: "unknown" for column in columns}}
162
+ except (UnicodeDecodeError, csv.Error):
163
+ return {}
164
+ if suffix == ".jsonl":
165
+ try:
166
+ with file_path.open("r", encoding="utf-8") as handle:
167
+ rows = sum(1 for line in handle if line.strip())
168
+ return {"rows": rows}
169
+ except UnicodeDecodeError:
170
+ return {}
171
+ return {}
172
+
173
+
174
+ def file_metadata(path: str | os.PathLike[str], *, copy_path: str | None = None) -> dict[str, Any]:
175
+ file_path = Path(path).expanduser().resolve()
176
+ if not file_path.is_file():
177
+ raise FileNotFoundError(file_path)
178
+ metadata_record: dict[str, Any] = {
179
+ "path": str(file_path),
180
+ "name": file_path.name,
181
+ "size_bytes": file_path.stat().st_size,
182
+ "sha256": sha256_file(file_path),
183
+ }
184
+ metadata_record.update(_tabular_metadata(file_path))
185
+ if copy_path:
186
+ metadata_record["captured_copy"] = copy_path
187
+ return metadata_record
188
+
189
+
190
+ def write_json(path: str | os.PathLike[str], value: Any) -> None:
191
+ target = Path(path)
192
+ target.parent.mkdir(parents=True, exist_ok=True)
193
+ target.write_text(json.dumps(value, indent=2, ensure_ascii=False, default=str) + "\n", encoding="utf-8")
194
+
195
+
196
+ def read_json(path: str | os.PathLike[str]) -> Any:
197
+ return json.loads(Path(path).read_text(encoding="utf-8"))