runproof-engine 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- runproof_engine/__init__.py +33 -0
- runproof_engine/cli.py +69 -0
- runproof_engine/core.py +557 -0
- runproof_engine/diff.py +197 -0
- runproof_engine/policy.py +29 -0
- runproof_engine/py.typed +0 -0
- runproof_engine/replay.py +131 -0
- runproof_engine/utils.py +197 -0
- runproof_engine-0.1.0.dist-info/LICENSE +151 -0
- runproof_engine-0.1.0.dist-info/METADATA +97 -0
- runproof_engine-0.1.0.dist-info/RECORD +14 -0
- runproof_engine-0.1.0.dist-info/WHEEL +5 -0
- runproof_engine-0.1.0.dist-info/entry_points.txt +2 -0
- runproof_engine-0.1.0.dist-info/top_level.txt +1 -0
runproof_engine/diff.py
ADDED
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
from .utils import safe_value
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@dataclass
|
|
10
|
+
class Difference:
|
|
11
|
+
area: str
|
|
12
|
+
path: str
|
|
13
|
+
kind: str
|
|
14
|
+
before: Any
|
|
15
|
+
after: Any
|
|
16
|
+
explanation: str
|
|
17
|
+
confidence: str = "evidence"
|
|
18
|
+
|
|
19
|
+
def to_dict(self) -> dict[str, Any]:
|
|
20
|
+
return {
|
|
21
|
+
"area": self.area,
|
|
22
|
+
"path": self.path,
|
|
23
|
+
"kind": self.kind,
|
|
24
|
+
"before": safe_value(self.before),
|
|
25
|
+
"after": safe_value(self.after),
|
|
26
|
+
"explanation": self.explanation,
|
|
27
|
+
"confidence": self.confidence,
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass
|
|
32
|
+
class RunDiff:
|
|
33
|
+
left_run_id: str
|
|
34
|
+
right_run_id: str
|
|
35
|
+
differences: list[Difference] = field(default_factory=list)
|
|
36
|
+
|
|
37
|
+
@property
|
|
38
|
+
def identical(self) -> bool:
|
|
39
|
+
return not self.differences
|
|
40
|
+
|
|
41
|
+
@property
|
|
42
|
+
def first_divergent_step(self) -> str | None:
|
|
43
|
+
for difference in self.differences:
|
|
44
|
+
if difference.area == "steps":
|
|
45
|
+
return difference.path.split(".")[0]
|
|
46
|
+
return None
|
|
47
|
+
|
|
48
|
+
def to_dict(self) -> dict[str, Any]:
|
|
49
|
+
return {
|
|
50
|
+
"left_run_id": self.left_run_id,
|
|
51
|
+
"right_run_id": self.right_run_id,
|
|
52
|
+
"identical": self.identical,
|
|
53
|
+
"first_divergent_step": self.first_divergent_step,
|
|
54
|
+
"differences": [difference.to_dict() for difference in self.differences],
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
def render(self) -> str:
|
|
58
|
+
if self.identical:
|
|
59
|
+
return f"Runs {self.left_run_id} and {self.right_run_id} are observably identical."
|
|
60
|
+
lines = [
|
|
61
|
+
f"Run diff: {self.left_run_id} -> {self.right_run_id}",
|
|
62
|
+
f"Differences: {len(self.differences)}",
|
|
63
|
+
]
|
|
64
|
+
if self.first_divergent_step:
|
|
65
|
+
lines.append(f"First divergent step: {self.first_divergent_step}")
|
|
66
|
+
for difference in self.differences:
|
|
67
|
+
lines.append(
|
|
68
|
+
f"- [{difference.area}] {difference.path}: {difference.explanation} "
|
|
69
|
+
f"(confidence={difference.confidence})"
|
|
70
|
+
)
|
|
71
|
+
return "\n".join(lines)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def compare_manifests(left: dict[str, Any], right: dict[str, Any]) -> RunDiff:
|
|
75
|
+
left_run = left.get("run", {})
|
|
76
|
+
right_run = right.get("run", {})
|
|
77
|
+
result = RunDiff(
|
|
78
|
+
left_run_id=str(left_run.get("run_id", "unknown")),
|
|
79
|
+
right_run_id=str(right_run.get("run_id", "unknown")),
|
|
80
|
+
)
|
|
81
|
+
_compare_inputs(left.get("inputs", []), right.get("inputs", []), result)
|
|
82
|
+
_compare_steps(left.get("steps", []), right.get("steps", []), result)
|
|
83
|
+
_compare_outputs(left.get("outputs", []), right.get("outputs", []), result)
|
|
84
|
+
_compare_checks(left.get("checks", []), right.get("checks", []), result)
|
|
85
|
+
_compare_environment(left.get("environment"), right.get("environment"), result)
|
|
86
|
+
if left_run.get("status") != right_run.get("status"):
|
|
87
|
+
result.differences.append(Difference(
|
|
88
|
+
area="run",
|
|
89
|
+
path="status",
|
|
90
|
+
kind="changed",
|
|
91
|
+
before=left_run.get("status"),
|
|
92
|
+
after=right_run.get("status"),
|
|
93
|
+
explanation="run status changed",
|
|
94
|
+
))
|
|
95
|
+
return result
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _index(records: list[dict[str, Any]], key: str = "name") -> dict[str, dict[str, Any]]:
|
|
99
|
+
return {str(item.get(key, index)): item for index, item in enumerate(records)}
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _compare_inputs(left: list[dict[str, Any]], right: list[dict[str, Any]], result: RunDiff) -> None:
|
|
103
|
+
left_map, right_map = _index(left), _index(right)
|
|
104
|
+
for name in sorted(set(left_map) | set(right_map)):
|
|
105
|
+
if name not in left_map:
|
|
106
|
+
result.differences.append(Difference("inputs", name, "added", None, right_map[name], "input was added"))
|
|
107
|
+
continue
|
|
108
|
+
if name not in right_map:
|
|
109
|
+
result.differences.append(Difference("inputs", name, "removed", left_map[name], None, "input was removed"))
|
|
110
|
+
continue
|
|
111
|
+
before, after = left_map[name], right_map[name]
|
|
112
|
+
if before.get("sha256") != after.get("sha256"):
|
|
113
|
+
result.differences.append(Difference(
|
|
114
|
+
"inputs", f"{name}.sha256", "changed", before.get("sha256"), after.get("sha256"),
|
|
115
|
+
"input content changed according to its SHA-256 fingerprint",
|
|
116
|
+
))
|
|
117
|
+
for field_name in ("size_bytes", "rows", "columns", "schema"):
|
|
118
|
+
if before.get(field_name) != after.get(field_name) and (field_name in before or field_name in after):
|
|
119
|
+
result.differences.append(Difference(
|
|
120
|
+
"inputs", f"{name}.{field_name}", "changed", before.get(field_name), after.get(field_name),
|
|
121
|
+
f"input metadata field '{field_name}' changed",
|
|
122
|
+
))
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _compare_steps(left: list[dict[str, Any]], right: list[dict[str, Any]], result: RunDiff) -> None:
|
|
126
|
+
left_map, right_map = _index(left), _index(right)
|
|
127
|
+
for name in sorted(set(left_map) | set(right_map)):
|
|
128
|
+
if name not in left_map:
|
|
129
|
+
result.differences.append(Difference("steps", name, "added", None, right_map[name], "step was added"))
|
|
130
|
+
continue
|
|
131
|
+
if name not in right_map:
|
|
132
|
+
result.differences.append(Difference("steps", name, "removed", left_map[name], None, "step was removed"))
|
|
133
|
+
continue
|
|
134
|
+
before, after = left_map[name], right_map[name]
|
|
135
|
+
if before.get("status") != after.get("status"):
|
|
136
|
+
result.differences.append(Difference("steps", f"{name}.status", "changed", before.get("status"), after.get("status"), "step status changed"))
|
|
137
|
+
before_output = (before.get("output") or {}).get("fingerprint")
|
|
138
|
+
after_output = (after.get("output") or {}).get("fingerprint")
|
|
139
|
+
if before_output != after_output:
|
|
140
|
+
result.differences.append(Difference(
|
|
141
|
+
"steps", f"{name}.output.fingerprint", "changed", before_output, after_output,
|
|
142
|
+
"step output changed; downstream artifacts may be affected",
|
|
143
|
+
))
|
|
144
|
+
before_function = (before.get("function") or {}).get("source_sha256")
|
|
145
|
+
after_function = (after.get("function") or {}).get("source_sha256")
|
|
146
|
+
if before_function != after_function:
|
|
147
|
+
result.differences.append(Difference(
|
|
148
|
+
"steps", f"{name}.function.source_sha256", "changed", before_function, after_function,
|
|
149
|
+
"step source fingerprint changed",
|
|
150
|
+
))
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _compare_outputs(left: list[dict[str, Any]], right: list[dict[str, Any]], result: RunDiff) -> None:
|
|
154
|
+
left_map, right_map = _index(left), _index(right)
|
|
155
|
+
for name in sorted(set(left_map) | set(right_map)):
|
|
156
|
+
if name not in left_map:
|
|
157
|
+
result.differences.append(Difference("outputs", name, "added", None, right_map[name], "output was added"))
|
|
158
|
+
elif name not in right_map:
|
|
159
|
+
result.differences.append(Difference("outputs", name, "removed", left_map[name], None, "output was removed"))
|
|
160
|
+
elif left_map[name].get("sha256") != right_map[name].get("sha256"):
|
|
161
|
+
result.differences.append(Difference(
|
|
162
|
+
"outputs", f"{name}.sha256", "changed", left_map[name].get("sha256"), right_map[name].get("sha256"),
|
|
163
|
+
"artifact content changed",
|
|
164
|
+
))
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _compare_checks(left: list[dict[str, Any]], right: list[dict[str, Any]], result: RunDiff) -> None:
|
|
168
|
+
left_map, right_map = _index(left), _index(right)
|
|
169
|
+
for name in sorted(set(left_map) | set(right_map)):
|
|
170
|
+
if name not in left_map or name not in right_map:
|
|
171
|
+
result.differences.append(Difference("checks", name, "changed", left_map.get(name), right_map.get(name), "check set changed"))
|
|
172
|
+
elif left_map[name].get("passed") != right_map[name].get("passed"):
|
|
173
|
+
result.differences.append(Difference(
|
|
174
|
+
"checks", f"{name}.passed", "changed", left_map[name].get("passed"), right_map[name].get("passed"),
|
|
175
|
+
"check outcome changed",
|
|
176
|
+
))
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _compare_environment(left: dict[str, Any] | None, right: dict[str, Any] | None, result: RunDiff) -> None:
|
|
180
|
+
if left is None or right is None:
|
|
181
|
+
if left != right:
|
|
182
|
+
result.differences.append(Difference("environment", "snapshot", "changed", bool(left), bool(right), "environment snapshot availability changed"))
|
|
183
|
+
return
|
|
184
|
+
for field_name in ("python_version", "platform", "machine"):
|
|
185
|
+
if left.get(field_name) != right.get(field_name):
|
|
186
|
+
result.differences.append(Difference(
|
|
187
|
+
"environment", field_name, "changed", left.get(field_name), right.get(field_name),
|
|
188
|
+
f"environment field '{field_name}' changed",
|
|
189
|
+
))
|
|
190
|
+
left_packages = {item.get("name"): item.get("version") for item in left.get("packages", [])}
|
|
191
|
+
right_packages = {item.get("name"): item.get("version") for item in right.get("packages", [])}
|
|
192
|
+
for name in sorted(set(left_packages) | set(right_packages)):
|
|
193
|
+
if left_packages.get(name) != right_packages.get(name):
|
|
194
|
+
result.differences.append(Difference(
|
|
195
|
+
"environment", f"packages.{name}", "changed", left_packages.get(name), right_packages.get(name),
|
|
196
|
+
"installed package version changed",
|
|
197
|
+
))
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class PolicyDenied(PermissionError):
|
|
8
|
+
"""Raised when a requested action is not allowed by the run policy."""
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass
|
|
12
|
+
class Policy:
|
|
13
|
+
allowed_actions: set[str] = field(default_factory=lambda: {"read", "write_artifact", "execute_python"})
|
|
14
|
+
approval_required: set[str] = field(default_factory=lambda: {"install_package", "network", "upload", "publish", "delete", "shell"})
|
|
15
|
+
denied_actions: set[str] = field(default_factory=set)
|
|
16
|
+
|
|
17
|
+
def authorize(self, action: str, *, approved: bool = False, target: str | None = None) -> dict[str, Any]:
|
|
18
|
+
action = str(action)
|
|
19
|
+
if action in self.denied_actions:
|
|
20
|
+
raise PolicyDenied(f"action denied by policy: {action}")
|
|
21
|
+
if action not in self.allowed_actions and action not in self.approval_required:
|
|
22
|
+
raise PolicyDenied(f"action is not declared in policy: {action}")
|
|
23
|
+
if action in self.approval_required and not approved:
|
|
24
|
+
raise PolicyDenied(f"approval required for action: {action}")
|
|
25
|
+
return {"action": action, "target": target, "approved": approved}
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def safe_default_policy() -> Policy:
|
|
29
|
+
return Policy()
|
runproof_engine/py.typed
ADDED
|
File without changes
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
from collections.abc import Callable
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from .diff import RunDiff, compare_manifests
|
|
10
|
+
from .utils import file_metadata, read_json, sha256_file
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass
|
|
14
|
+
class ReplayReport:
|
|
15
|
+
status: str
|
|
16
|
+
run_id: str
|
|
17
|
+
mode: str
|
|
18
|
+
reasons: list[str] = field(default_factory=list)
|
|
19
|
+
compared_run_id: str | None = None
|
|
20
|
+
|
|
21
|
+
def to_dict(self) -> dict[str, Any]:
|
|
22
|
+
return {
|
|
23
|
+
"status": self.status,
|
|
24
|
+
"run_id": self.run_id,
|
|
25
|
+
"mode": self.mode,
|
|
26
|
+
"reasons": self.reasons,
|
|
27
|
+
"compared_run_id": self.compared_run_id,
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
def __str__(self) -> str:
|
|
31
|
+
suffix = f"; reasons={self.reasons}" if self.reasons else ""
|
|
32
|
+
return f"ReplayReport(status={self.status!r}, run_id={self.run_id!r}, mode={self.mode!r}{suffix})"
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class LoadedRun:
|
|
36
|
+
def __init__(self, artifact_dir: str | os.PathLike[str]) -> None:
|
|
37
|
+
self.artifact_dir = Path(artifact_dir).expanduser().resolve()
|
|
38
|
+
self.manifest_path = self.artifact_dir / "manifest.json"
|
|
39
|
+
if not self.manifest_path.is_file():
|
|
40
|
+
raise FileNotFoundError(f"RunProof manifest not found: {self.manifest_path}")
|
|
41
|
+
self.manifest = read_json(self.manifest_path)
|
|
42
|
+
|
|
43
|
+
@property
|
|
44
|
+
def run_id(self) -> str:
|
|
45
|
+
return str(self.manifest.get("run", {}).get("run_id", self.artifact_dir.name))
|
|
46
|
+
|
|
47
|
+
@property
|
|
48
|
+
def name(self) -> str:
|
|
49
|
+
return str(self.manifest.get("run", {}).get("name", self.artifact_dir.parent.name))
|
|
50
|
+
|
|
51
|
+
@property
|
|
52
|
+
def status(self) -> str:
|
|
53
|
+
return str(self.manifest.get("run", {}).get("status", "unknown"))
|
|
54
|
+
|
|
55
|
+
def verify_integrity(self) -> ReplayReport:
|
|
56
|
+
reasons: list[str] = []
|
|
57
|
+
for record in self.manifest.get("inputs", []):
|
|
58
|
+
source = Path(record.get("path", ""))
|
|
59
|
+
if source.is_file():
|
|
60
|
+
if file_metadata(source).get("sha256") != record.get("sha256"):
|
|
61
|
+
reasons.append(f"input changed: {record.get('name', source.name)}")
|
|
62
|
+
continue
|
|
63
|
+
captured = record.get("captured_copy")
|
|
64
|
+
captured_path = (self.artifact_dir / captured).resolve() if captured else None
|
|
65
|
+
if captured_path and captured_path.is_file() and file_metadata(captured_path).get("sha256") == record.get("sha256"):
|
|
66
|
+
continue
|
|
67
|
+
reasons.append(f"input missing: {source}")
|
|
68
|
+
for record in self.manifest.get("outputs", []):
|
|
69
|
+
output = (self.artifact_dir / record.get("path", "")).resolve()
|
|
70
|
+
if not output.is_file():
|
|
71
|
+
reasons.append(f"output missing: {output}")
|
|
72
|
+
elif file_metadata(output).get("sha256") != record.get("sha256"):
|
|
73
|
+
reasons.append(f"artifact changed: {record.get('name', output.name)}")
|
|
74
|
+
integrity_file = self.artifact_dir / "integrity.json"
|
|
75
|
+
if integrity_file.is_file():
|
|
76
|
+
try:
|
|
77
|
+
integrity = read_json(integrity_file)
|
|
78
|
+
for record in integrity.get("files", []):
|
|
79
|
+
artifact = (self.artifact_dir / record["path"]).resolve()
|
|
80
|
+
if not artifact.is_file():
|
|
81
|
+
reasons.append(f"integrity file missing: {record['path']}")
|
|
82
|
+
elif sha256_file(artifact) != record.get("sha256"):
|
|
83
|
+
reasons.append(f"integrity mismatch: {record['path']}")
|
|
84
|
+
except (KeyError, TypeError, ValueError):
|
|
85
|
+
reasons.append("integrity manifest is invalid")
|
|
86
|
+
status = "verified" if not reasons else "non_reproducible"
|
|
87
|
+
return ReplayReport(status=status, run_id=self.run_id, mode="integrity", reasons=reasons)
|
|
88
|
+
|
|
89
|
+
def replay(
|
|
90
|
+
self,
|
|
91
|
+
*,
|
|
92
|
+
mode: str = "strict",
|
|
93
|
+
runner: Callable[[LoadedRun], str | os.PathLike[str] | LoadedRun] | None = None,
|
|
94
|
+
) -> ReplayReport:
|
|
95
|
+
"""Validate replay prerequisites and optionally compare a user-supplied re-execution.
|
|
96
|
+
|
|
97
|
+
RunProof does not guess how to reconstruct arbitrary Python state. A runner is
|
|
98
|
+
required to execute the original workflow again; without it, strict mode only
|
|
99
|
+
verifies captured evidence and reports that no new execution was performed.
|
|
100
|
+
"""
|
|
101
|
+
if mode not in {"strict", "fresh"}:
|
|
102
|
+
raise ValueError("mode must be 'strict' or 'fresh'")
|
|
103
|
+
if runner is None:
|
|
104
|
+
integrity = self.verify_integrity()
|
|
105
|
+
if integrity.status != "verified":
|
|
106
|
+
integrity.mode = mode
|
|
107
|
+
return integrity
|
|
108
|
+
return ReplayReport(
|
|
109
|
+
status="replay_ready",
|
|
110
|
+
run_id=self.run_id,
|
|
111
|
+
mode=mode,
|
|
112
|
+
reasons=["integrity verified; pass runner=... to execute and compare the workflow"],
|
|
113
|
+
)
|
|
114
|
+
new_artifact = runner(self)
|
|
115
|
+
other = new_artifact if isinstance(new_artifact, LoadedRun) else load_run(new_artifact)
|
|
116
|
+
comparison = self.diff(other)
|
|
117
|
+
return ReplayReport(
|
|
118
|
+
status="verified" if comparison.identical else "non_reproducible",
|
|
119
|
+
run_id=self.run_id,
|
|
120
|
+
mode=mode,
|
|
121
|
+
reasons=[] if comparison.identical else [difference.explanation for difference in comparison.differences],
|
|
122
|
+
compared_run_id=other.run_id,
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
def diff(self, other: LoadedRun | str | os.PathLike[str]) -> RunDiff:
|
|
126
|
+
other_run = other if isinstance(other, LoadedRun) else load_run(other)
|
|
127
|
+
return compare_manifests(self.manifest, other_run.manifest)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def load_run(artifact_dir: str | os.PathLike[str]) -> LoadedRun:
|
|
131
|
+
return LoadedRun(artifact_dir)
|
runproof_engine/utils.py
ADDED
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import csv
|
|
4
|
+
import dataclasses
|
|
5
|
+
import hashlib
|
|
6
|
+
import inspect
|
|
7
|
+
import json
|
|
8
|
+
import os
|
|
9
|
+
import platform
|
|
10
|
+
import sys
|
|
11
|
+
from datetime import datetime, timezone
|
|
12
|
+
from importlib import metadata
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import Any
|
|
15
|
+
|
|
16
|
+
SECRET_MARKERS = (
|
|
17
|
+
"secret",
|
|
18
|
+
"password",
|
|
19
|
+
"passwd",
|
|
20
|
+
"token",
|
|
21
|
+
"api_key",
|
|
22
|
+
"apikey",
|
|
23
|
+
"private_key",
|
|
24
|
+
"authorization",
|
|
25
|
+
"credential",
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def utc_now() -> str:
|
|
30
|
+
return datetime.now(timezone.utc).replace(microsecond=0).isoformat().replace("+00:00", "Z")
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def sha256_bytes(value: bytes) -> str:
|
|
34
|
+
return hashlib.sha256(value).hexdigest()
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def sha256_file(path: str | os.PathLike[str], chunk_size: int = 1024 * 1024) -> str:
|
|
38
|
+
digest = hashlib.sha256()
|
|
39
|
+
with Path(path).open("rb") as handle:
|
|
40
|
+
while chunk := handle.read(chunk_size):
|
|
41
|
+
digest.update(chunk)
|
|
42
|
+
return digest.hexdigest()
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def canonical_json(value: Any) -> str:
|
|
46
|
+
return json.dumps(value, sort_keys=True, ensure_ascii=False, separators=(",", ":"), default=str)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def fingerprint(value: Any) -> str:
|
|
50
|
+
return sha256_bytes(canonical_json(safe_value(value, max_items=200)).encode("utf-8"))
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _redacted_key(key: Any) -> bool:
|
|
54
|
+
normalized = str(key).lower().replace("-", "_")
|
|
55
|
+
return any(marker in normalized for marker in SECRET_MARKERS)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def safe_value(value: Any, *, max_items: int = 100, max_text: int = 500) -> Any:
|
|
59
|
+
"""Return a bounded, JSON-safe representation without exposing likely secrets."""
|
|
60
|
+
if value is None or isinstance(value, (bool, int, float)):
|
|
61
|
+
return value
|
|
62
|
+
if isinstance(value, (str, Path)):
|
|
63
|
+
text = str(value)
|
|
64
|
+
return text if len(text) <= max_text else text[:max_text] + "…"
|
|
65
|
+
if dataclasses.is_dataclass(value):
|
|
66
|
+
return safe_value(dataclasses.asdict(value), max_items=max_items, max_text=max_text)
|
|
67
|
+
if isinstance(value, dict):
|
|
68
|
+
items = list(value.items())[:max_items]
|
|
69
|
+
return {
|
|
70
|
+
str(key): "[REDACTED]" if _redacted_key(key) else safe_value(item, max_items=max_items, max_text=max_text)
|
|
71
|
+
for key, item in items
|
|
72
|
+
}
|
|
73
|
+
if isinstance(value, (list, tuple, set)):
|
|
74
|
+
values = list(value)
|
|
75
|
+
bounded = [safe_value(item, max_items=max_items, max_text=max_text) for item in values[:max_items]]
|
|
76
|
+
if len(values) > max_items:
|
|
77
|
+
bounded.append(f"… {len(values) - max_items} more items")
|
|
78
|
+
return bounded
|
|
79
|
+
if hasattr(value, "to_dict") and callable(value.to_dict):
|
|
80
|
+
try:
|
|
81
|
+
return safe_value(value.to_dict(), max_items=max_items, max_text=max_text)
|
|
82
|
+
except (AttributeError, KeyError, TypeError, ValueError, RuntimeError):
|
|
83
|
+
pass
|
|
84
|
+
if hasattr(value, "__dict__"):
|
|
85
|
+
try:
|
|
86
|
+
return {
|
|
87
|
+
"type": f"{type(value).__module__}.{type(value).__qualname__}",
|
|
88
|
+
"attributes": safe_value(vars(value), max_items=max_items, max_text=max_text),
|
|
89
|
+
}
|
|
90
|
+
except (AttributeError, TypeError, ValueError, RuntimeError):
|
|
91
|
+
pass
|
|
92
|
+
return {
|
|
93
|
+
"type": f"{type(value).__module__}.{type(value).__qualname__}",
|
|
94
|
+
"repr": str(value)[:max_text],
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def summarize(value: Any) -> dict[str, Any]:
|
|
99
|
+
"""Describe a runtime value without serializing an unbounded payload."""
|
|
100
|
+
summary: dict[str, Any] = {
|
|
101
|
+
"type": f"{type(value).__module__}.{type(value).__qualname__}",
|
|
102
|
+
"fingerprint": fingerprint(value),
|
|
103
|
+
}
|
|
104
|
+
if value is None or isinstance(value, (bool, int, float, str)):
|
|
105
|
+
summary["value"] = safe_value(value)
|
|
106
|
+
elif isinstance(value, (list, tuple, set, dict)):
|
|
107
|
+
summary["length"] = len(value)
|
|
108
|
+
if isinstance(value, dict):
|
|
109
|
+
summary["keys"] = [str(key) for key in list(value.keys())[:100]]
|
|
110
|
+
else:
|
|
111
|
+
for attribute in ("shape", "columns", "dtypes", "size"):
|
|
112
|
+
if hasattr(value, attribute):
|
|
113
|
+
try:
|
|
114
|
+
summary[attribute] = safe_value(getattr(value, attribute))
|
|
115
|
+
except (AttributeError, KeyError, TypeError, ValueError, RuntimeError):
|
|
116
|
+
pass
|
|
117
|
+
return summary
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def function_descriptor(function: Any) -> dict[str, Any]:
|
|
121
|
+
descriptor: dict[str, Any] = {
|
|
122
|
+
"module": getattr(function, "__module__", None),
|
|
123
|
+
"qualname": getattr(function, "__qualname__", repr(function)),
|
|
124
|
+
}
|
|
125
|
+
try:
|
|
126
|
+
source = inspect.getsource(function)
|
|
127
|
+
descriptor["source_sha256"] = sha256_bytes(source.encode("utf-8"))
|
|
128
|
+
except (OSError, TypeError):
|
|
129
|
+
descriptor["source_sha256"] = None
|
|
130
|
+
return descriptor
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def environment_snapshot() -> dict[str, Any]:
|
|
134
|
+
packages: list[dict[str, str]] = []
|
|
135
|
+
try:
|
|
136
|
+
for distribution in metadata.distributions():
|
|
137
|
+
name = distribution.metadata.get("Name")
|
|
138
|
+
version = distribution.version
|
|
139
|
+
if name and version:
|
|
140
|
+
packages.append({"name": name.lower(), "version": version})
|
|
141
|
+
packages.sort(key=lambda item: item["name"])
|
|
142
|
+
except (ImportError, OSError, RuntimeError, ValueError):
|
|
143
|
+
packages = []
|
|
144
|
+
return {
|
|
145
|
+
"python_version": sys.version,
|
|
146
|
+
"python_executable": sys.executable,
|
|
147
|
+
"platform": platform.platform(),
|
|
148
|
+
"machine": platform.machine(),
|
|
149
|
+
"packages": packages,
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _tabular_metadata(file_path: Path) -> dict[str, Any]:
|
|
154
|
+
suffix = file_path.suffix.lower()
|
|
155
|
+
if suffix == ".csv":
|
|
156
|
+
try:
|
|
157
|
+
with file_path.open("r", encoding="utf-8-sig", newline="") as handle:
|
|
158
|
+
reader = csv.reader(handle)
|
|
159
|
+
columns = next(reader, [])
|
|
160
|
+
rows = sum(1 for _ in reader)
|
|
161
|
+
return {"rows": rows, "columns": columns, "schema": {column: "unknown" for column in columns}}
|
|
162
|
+
except (UnicodeDecodeError, csv.Error):
|
|
163
|
+
return {}
|
|
164
|
+
if suffix == ".jsonl":
|
|
165
|
+
try:
|
|
166
|
+
with file_path.open("r", encoding="utf-8") as handle:
|
|
167
|
+
rows = sum(1 for line in handle if line.strip())
|
|
168
|
+
return {"rows": rows}
|
|
169
|
+
except UnicodeDecodeError:
|
|
170
|
+
return {}
|
|
171
|
+
return {}
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def file_metadata(path: str | os.PathLike[str], *, copy_path: str | None = None) -> dict[str, Any]:
|
|
175
|
+
file_path = Path(path).expanduser().resolve()
|
|
176
|
+
if not file_path.is_file():
|
|
177
|
+
raise FileNotFoundError(file_path)
|
|
178
|
+
metadata_record: dict[str, Any] = {
|
|
179
|
+
"path": str(file_path),
|
|
180
|
+
"name": file_path.name,
|
|
181
|
+
"size_bytes": file_path.stat().st_size,
|
|
182
|
+
"sha256": sha256_file(file_path),
|
|
183
|
+
}
|
|
184
|
+
metadata_record.update(_tabular_metadata(file_path))
|
|
185
|
+
if copy_path:
|
|
186
|
+
metadata_record["captured_copy"] = copy_path
|
|
187
|
+
return metadata_record
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def write_json(path: str | os.PathLike[str], value: Any) -> None:
|
|
191
|
+
target = Path(path)
|
|
192
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
193
|
+
target.write_text(json.dumps(value, indent=2, ensure_ascii=False, default=str) + "\n", encoding="utf-8")
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def read_json(path: str | os.PathLike[str]) -> Any:
|
|
197
|
+
return json.loads(Path(path).read_text(encoding="utf-8"))
|