runproof-engine 0.1.2__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. runproof_engine-0.2.0/PKG-INFO +164 -0
  2. runproof_engine-0.2.0/README.md +121 -0
  3. {runproof-engine-0.1.2 → runproof_engine-0.2.0}/pyproject.toml +10 -6
  4. runproof_engine-0.2.0/src/runproof_engine/__init__.py +82 -0
  5. runproof_engine-0.2.0/src/runproof_engine/adapters.py +136 -0
  6. runproof_engine-0.2.0/src/runproof_engine/auto.py +334 -0
  7. runproof_engine-0.2.0/src/runproof_engine/cli.py +143 -0
  8. {runproof-engine-0.1.2 → runproof_engine-0.2.0}/src/runproof_engine/core.py +111 -4
  9. {runproof-engine-0.1.2 → runproof_engine-0.2.0}/src/runproof_engine/diff.py +72 -0
  10. runproof_engine-0.2.0/src/runproof_engine/environment.py +117 -0
  11. runproof_engine-0.2.0/src/runproof_engine/integrations.py +571 -0
  12. runproof_engine-0.2.0/src/runproof_engine/provenance.py +159 -0
  13. {runproof-engine-0.1.2 → runproof_engine-0.2.0}/src/runproof_engine/replay.py +27 -1
  14. runproof_engine-0.2.0/src/runproof_engine/tracing.py +135 -0
  15. runproof_engine-0.2.0/src/runproof_engine.egg-info/PKG-INFO +164 -0
  16. {runproof-engine-0.1.2 → runproof_engine-0.2.0}/src/runproof_engine.egg-info/SOURCES.txt +11 -1
  17. runproof_engine-0.2.0/src/runproof_engine.egg-info/requires.txt +31 -0
  18. runproof_engine-0.2.0/tests/test_auto.py +57 -0
  19. runproof_engine-0.2.0/tests/test_cli.py +65 -0
  20. {runproof-engine-0.1.2 → runproof_engine-0.2.0}/tests/test_core.py +47 -0
  21. runproof_engine-0.2.0/tests/test_fuzz.py +48 -0
  22. {runproof-engine-0.1.2 → runproof_engine-0.2.0}/tests/test_http.py +20 -1
  23. runproof_engine-0.2.0/tests/test_integrations.py +207 -0
  24. runproof_engine-0.2.0/tests/test_tracing.py +27 -0
  25. runproof-engine-0.1.2/PKG-INFO +0 -94
  26. runproof-engine-0.1.2/README.md +0 -73
  27. runproof-engine-0.1.2/src/runproof_engine/__init__.py +0 -33
  28. runproof-engine-0.1.2/src/runproof_engine/cli.py +0 -69
  29. runproof-engine-0.1.2/src/runproof_engine.egg-info/PKG-INFO +0 -94
  30. runproof-engine-0.1.2/src/runproof_engine.egg-info/requires.txt +0 -7
  31. runproof-engine-0.1.2/tests/test_cli.py +0 -22
  32. {runproof-engine-0.1.2 → runproof_engine-0.2.0}/LICENSE +0 -0
  33. {runproof-engine-0.1.2 → runproof_engine-0.2.0}/setup.cfg +0 -0
  34. {runproof-engine-0.1.2 → runproof_engine-0.2.0}/src/runproof_engine/policy.py +0 -0
  35. {runproof-engine-0.1.2 → runproof_engine-0.2.0}/src/runproof_engine/py.typed +0 -0
  36. {runproof-engine-0.1.2 → runproof_engine-0.2.0}/src/runproof_engine/utils.py +0 -0
  37. {runproof-engine-0.1.2 → runproof_engine-0.2.0}/src/runproof_engine.egg-info/dependency_links.txt +0 -0
  38. {runproof-engine-0.1.2 → runproof_engine-0.2.0}/src/runproof_engine.egg-info/entry_points.txt +0 -0
  39. {runproof-engine-0.1.2 → runproof_engine-0.2.0}/src/runproof_engine.egg-info/top_level.txt +0 -0
@@ -0,0 +1,164 @@
1
+ Metadata-Version: 2.4
2
+ Name: runproof-engine
3
+ Version: 0.2.0
4
+ Summary: Evidence, replay, and explainable diffs for real Python runs
5
+ Author: RunProof Contributors
6
+ License-Expression: Apache-2.0
7
+ Keywords: reproducibility,provenance,data-lineage,observability,python,replay,explainable-diff
8
+ Classifier: Development Status :: 3 - Alpha
9
+ Classifier: Intended Audience :: Developers
10
+ Classifier: Intended Audience :: Science/Research
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Programming Language :: Python :: 3 :: Only
13
+ Classifier: Topic :: Software Development :: Testing
14
+ Classifier: Topic :: Scientific/Engineering :: Information Analysis
15
+ Requires-Python: >=3.10
16
+ Description-Content-Type: text/markdown
17
+ License-File: LICENSE
18
+ Provides-Extra: data
19
+ Requires-Dist: pandas>=2.0; extra == "data"
20
+ Requires-Dist: polars>=1.0; extra == "data"
21
+ Provides-Extra: http
22
+ Requires-Dist: requests>=2.31; extra == "http"
23
+ Requires-Dist: httpx>=0.27; extra == "http"
24
+ Provides-Extra: database
25
+ Requires-Dist: SQLAlchemy>=2.0; extra == "database"
26
+ Requires-Dist: psycopg[binary]>=3.1; extra == "database"
27
+ Provides-Extra: jupyter
28
+ Requires-Dist: nbclient>=0.10; extra == "jupyter"
29
+ Requires-Dist: nbformat>=5.9; extra == "jupyter"
30
+ Provides-Extra: cloud
31
+ Requires-Dist: boto3>=1.34; extra == "cloud"
32
+ Provides-Extra: ml
33
+ Requires-Dist: joblib>=1.3; extra == "ml"
34
+ Provides-Extra: dev
35
+ Requires-Dist: pytest>=8; extra == "dev"
36
+ Requires-Dist: ruff>=0.6; extra == "dev"
37
+ Requires-Dist: build>=1.2; extra == "dev"
38
+ Requires-Dist: twine>=5; extra == "dev"
39
+ Requires-Dist: hypothesis>=6.100; extra == "dev"
40
+ Requires-Dist: bandit>=1.7; extra == "dev"
41
+ Requires-Dist: pip-audit>=2.7; extra == "dev"
42
+ Dynamic: license-file
43
+
44
+ # RunProof
45
+
46
+ RunProof is a Python library for recording, validating, replaying, and comparing real computational runs. It turns a Python execution into an inspectable artifact containing inputs, outputs, code references, environment metadata, checks, and an execution trace.
47
+
48
+ The distinctive feature is **Explainable Diff**: when two runs differ, RunProof compares input fingerprints, schemas, output summaries, step status, code fingerprints, and environment metadata, then reports evidence-backed causes instead of only saying that the runs are different.
49
+
50
+ ## What it is
51
+
52
+ RunProof is a local-first execution record and reproducibility layer for Python workflows. It is useful for data analysis, reports, ML experiments, research software, API workflows, and any process where the result must be explained later.
53
+
54
+ RunProof is not a reverse-engineering tool, a code generator, a replacement for Git, or a guarantee that a scientific conclusion is correct. It records and validates the execution that was declared to it.
55
+
56
+ ## Quick start
57
+
58
+ ```python
59
+ from runproof_engine import verified
60
+
61
+
62
+ def clean_rows(rows):
63
+ return [row for row in rows if row["amount"] >= 0]
64
+
65
+
66
+ def total(rows):
67
+ return sum(row["amount"] for row in rows)
68
+
69
+ with verified("sales_total", root="runs") as run:
70
+ rows = run.input("sales.json", name="sales")
71
+ cleaned = run.step("clean_rows", clean_rows, rows)
72
+ result = run.step("total", total, cleaned)
73
+ run.assert_true(result >= 0, "total must be non-negative")
74
+ run.output("total.json", {"total": result})
75
+
76
+ print(run.result.status)
77
+ print(run.result.artifact_dir)
78
+ ```
79
+
80
+ This creates a run directory with a manifest, input metadata, output metadata, trace events, checks, and an environment snapshot. The input is not silently replaced by generated data. File contents are fingerprinted, while copying full inputs is explicit and configurable.
81
+
82
+ ## Automatic capture
83
+
84
+ For whole-process observation, use `auto_run`. It installs a Python execution observer for the lifetime of the context. On Python 3.12 and newer it uses `sys.monitoring` when available; otherwise it falls back to `sys.settrace`. The observer records bounded call, return, and exception events without requiring every function to be wrapped manually.
85
+
86
+ ```python
87
+ from pathlib import Path
88
+ from runproof_engine import auto_run
89
+
90
+ with auto_run(
91
+ "automatic-report",
92
+ root="runs",
93
+ include_paths=[Path("src")],
94
+ ) as run:
95
+ report = build_report(real_input)
96
+ run.observe(report, name="report")
97
+ run.output("report.json", report)
98
+ ```
99
+
100
+ Automatic capture is an evidence layer, not a sandbox. It does not serialize every local variable, intercept every native extension, or make external services deterministic. Use adapters or explicit `run.input`, `run.step`, and `run.external_call` calls when a boundary needs stronger evidence.
101
+
102
+ ### Optional real-library adapters
103
+
104
+ When an optional dependency is installed, `auto_run` discovers and activates its adapter by default. The current adapters observe real operations through documented hooks or narrowly scoped wrappers:
105
+
106
+ | Integration | Captured evidence | Boundary behavior |
107
+ | --- | --- | --- |
108
+ | pandas | DataFrame reads/writes, summaries, file hashes | Local file outputs can be archived; in-memory and native execution remain bounded |
109
+ | Polars | DataFrame reads/writes, summaries, file hashes | Requires the optional `polars` package |
110
+ | requests | Prepared URL, redacted headers, status code | Response bodies are not consumed automatically; live network remains a boundary |
111
+ | HTTPX | Sync and async request/response hooks | Streaming bodies are not consumed automatically; live network remains a boundary |
112
+ | SQLite | SQL statements and database path | Database transaction state is a boundary; mutable files can become `non_reproducible` |
113
+ | SQLAlchemy and psycopg | Cursor statements and database dialect | Requires the optional database package and a live database policy |
114
+ | subprocess | Command summary, return code, stdout/stderr digests | Process environment and side effects remain a boundary |
115
+ | Boto3 | Service operation, redacted parameters, response summary | Cloud object/service state remains a boundary |
116
+ | Jupyter/nbclient | Notebook execution lifecycle | Kernel state and external effects remain a boundary |
117
+ | Torch and joblib | Model save/load file evidence | Device/native runtime state may remain a boundary |
118
+
119
+ The optional adapters never execute installation commands and never send captured data to RunProof services. They can be disabled or replaced by passing an explicit `adapters=` collection to `auto_run`.
120
+
121
+ ## Provenance, environment, and distributed tracing
122
+
123
+ Every completed run now contains `provenance.json`, a graph connecting the run to inputs, steps, values, outputs, checks, observations, and evidence boundaries. It is available through `load_run(path).provenance` and is designed to make later diffs explainable without pretending to prove causality.
124
+
125
+ The artifact also contains `environment/environment.lock.json`. It records the interpreter, platform, and installed package versions and can be compared with `compare_environment_lock(...)`. `reconstruction_plan(...)` returns reviewable pip commands but deliberately marks them as requiring approval; RunProof never installs packages silently. Package locks do not reproduce operating-system state, hardware, external services, or changing data.
126
+
127
+ For service-to-service workflows, use `run.span(...)` and propagate `run.traceparent()`. Spans are stored in `execution/spans.json` using a dependency-free W3C traceparent-compatible format. OpenTelemetry exporters can be added as adapters without making OpenTelemetry a runtime dependency.
128
+
129
+ ## Replay and comparison
130
+
131
+ ```python
132
+ from runproof_engine import load_run
133
+
134
+ previous = load_run("runs/sales_total/20260823-101500-abc123")
135
+ replayed = previous.replay(mode="strict")
136
+ print(replayed.status)
137
+
138
+ comparison = previous.diff(replayed)
139
+ print(comparison.to_dict())
140
+ print(comparison.render())
141
+ ```
142
+
143
+ Without a runner, `replay()` verifies the captured artifact and reports `replay_ready`; it does not silently re-execute arbitrary Python. To perform a real replay, provide a user-controlled runner that invokes the original workflow and returns the new artifact. The resulting run can then be compared with the previous run to identify evidence-backed changes.
144
+
145
+ ## Statuses
146
+
147
+ - `verified`: execution completed and all declared checks passed with no recorded evidence boundary.
148
+ - `verified_with_boundaries`: execution completed and the artifact is intact, but one or more external or non-deterministic effects remain outside the captured evidence.
149
+ - `verified_with_warnings`: execution completed but comparability or external-source evidence is limited.
150
+ - `failed`: execution or a required check failed.
151
+ - `blocked`: a declared policy prevented a sensitive action.
152
+ - `non_reproducible`: replay was attempted but did not match the captured run.
153
+
154
+ ## Privacy and real data
155
+
156
+ RunProof records metadata and hashes by default. Full input copying is opt-in. Secrets are redacted from environment snapshots and trace values. External adapters must provide privacy-safe request metadata instead of storing credentials or raw authorization headers.
157
+
158
+ ## Project status
159
+
160
+ The current repository implements the local core, automatic observation, provenance graph, environment lock, distributed trace context, and optional real-library integrations. It remains useful without a cloud account or a specific AI provider. Automatic observation still has explicit coverage boundaries; it is not a universal sandbox or a guarantee of perfect replay for arbitrary Python.
161
+
162
+ ## License
163
+
164
+ Apache-2.0. See `LICENSE`.
@@ -0,0 +1,121 @@
1
+ # RunProof
2
+
3
+ RunProof is a Python library for recording, validating, replaying, and comparing real computational runs. It turns a Python execution into an inspectable artifact containing inputs, outputs, code references, environment metadata, checks, and an execution trace.
4
+
5
+ The distinctive feature is **Explainable Diff**: when two runs differ, RunProof compares input fingerprints, schemas, output summaries, step status, code fingerprints, and environment metadata, then reports evidence-backed causes instead of only saying that the runs are different.
6
+
7
+ ## What it is
8
+
9
+ RunProof is a local-first execution record and reproducibility layer for Python workflows. It is useful for data analysis, reports, ML experiments, research software, API workflows, and any process where the result must be explained later.
10
+
11
+ RunProof is not a reverse-engineering tool, a code generator, a replacement for Git, or a guarantee that a scientific conclusion is correct. It records and validates the execution that was declared to it.
12
+
13
+ ## Quick start
14
+
15
+ ```python
16
+ from runproof_engine import verified
17
+
18
+
19
+ def clean_rows(rows):
20
+ return [row for row in rows if row["amount"] >= 0]
21
+
22
+
23
+ def total(rows):
24
+ return sum(row["amount"] for row in rows)
25
+
26
+ with verified("sales_total", root="runs") as run:
27
+ rows = run.input("sales.json", name="sales")
28
+ cleaned = run.step("clean_rows", clean_rows, rows)
29
+ result = run.step("total", total, cleaned)
30
+ run.assert_true(result >= 0, "total must be non-negative")
31
+ run.output("total.json", {"total": result})
32
+
33
+ print(run.result.status)
34
+ print(run.result.artifact_dir)
35
+ ```
36
+
37
+ This creates a run directory with a manifest, input metadata, output metadata, trace events, checks, and an environment snapshot. The input is not silently replaced by generated data. File contents are fingerprinted, while copying full inputs is explicit and configurable.
38
+
39
+ ## Automatic capture
40
+
41
+ For whole-process observation, use `auto_run`. It installs a Python execution observer for the lifetime of the context. On Python 3.12 and newer it uses `sys.monitoring` when available; otherwise it falls back to `sys.settrace`. The observer records bounded call, return, and exception events without requiring every function to be wrapped manually.
42
+
43
+ ```python
44
+ from pathlib import Path
45
+ from runproof_engine import auto_run
46
+
47
+ with auto_run(
48
+ "automatic-report",
49
+ root="runs",
50
+ include_paths=[Path("src")],
51
+ ) as run:
52
+ report = build_report(real_input)
53
+ run.observe(report, name="report")
54
+ run.output("report.json", report)
55
+ ```
56
+
57
+ Automatic capture is an evidence layer, not a sandbox. It does not serialize every local variable, intercept every native extension, or make external services deterministic. Use adapters or explicit `run.input`, `run.step`, and `run.external_call` calls when a boundary needs stronger evidence.
58
+
59
+ ### Optional real-library adapters
60
+
61
+ When an optional dependency is installed, `auto_run` discovers and activates its adapter by default. The current adapters observe real operations through documented hooks or narrowly scoped wrappers:
62
+
63
+ | Integration | Captured evidence | Boundary behavior |
64
+ | --- | --- | --- |
65
+ | pandas | DataFrame reads/writes, summaries, file hashes | Local file outputs can be archived; in-memory and native execution remain bounded |
66
+ | Polars | DataFrame reads/writes, summaries, file hashes | Requires the optional `polars` package |
67
+ | requests | Prepared URL, redacted headers, status code | Response bodies are not consumed automatically; live network remains a boundary |
68
+ | HTTPX | Sync and async request/response hooks | Streaming bodies are not consumed automatically; live network remains a boundary |
69
+ | SQLite | SQL statements and database path | Database transaction state is a boundary; mutable files can become `non_reproducible` |
70
+ | SQLAlchemy and psycopg | Cursor statements and database dialect | Requires the optional database package and a live database policy |
71
+ | subprocess | Command summary, return code, stdout/stderr digests | Process environment and side effects remain a boundary |
72
+ | Boto3 | Service operation, redacted parameters, response summary | Cloud object/service state remains a boundary |
73
+ | Jupyter/nbclient | Notebook execution lifecycle | Kernel state and external effects remain a boundary |
74
+ | Torch and joblib | Model save/load file evidence | Device/native runtime state may remain a boundary |
75
+
76
+ The optional adapters never execute installation commands and never send captured data to RunProof services. They can be disabled or replaced by passing an explicit `adapters=` collection to `auto_run`.
77
+
78
+ ## Provenance, environment, and distributed tracing
79
+
80
+ Every completed run now contains `provenance.json`, a graph connecting the run to inputs, steps, values, outputs, checks, observations, and evidence boundaries. It is available through `load_run(path).provenance` and is designed to make later diffs explainable without pretending to prove causality.
81
+
82
+ The artifact also contains `environment/environment.lock.json`. It records the interpreter, platform, and installed package versions and can be compared with `compare_environment_lock(...)`. `reconstruction_plan(...)` returns reviewable pip commands but deliberately marks them as requiring approval; RunProof never installs packages silently. Package locks do not reproduce operating-system state, hardware, external services, or changing data.
83
+
84
+ For service-to-service workflows, use `run.span(...)` and propagate `run.traceparent()`. Spans are stored in `execution/spans.json` using a dependency-free W3C traceparent-compatible format. OpenTelemetry exporters can be added as adapters without making OpenTelemetry a runtime dependency.
85
+
86
+ ## Replay and comparison
87
+
88
+ ```python
89
+ from runproof_engine import load_run
90
+
91
+ previous = load_run("runs/sales_total/20260823-101500-abc123")
92
+ replayed = previous.replay(mode="strict")
93
+ print(replayed.status)
94
+
95
+ comparison = previous.diff(replayed)
96
+ print(comparison.to_dict())
97
+ print(comparison.render())
98
+ ```
99
+
100
+ Without a runner, `replay()` verifies the captured artifact and reports `replay_ready`; it does not silently re-execute arbitrary Python. To perform a real replay, provide a user-controlled runner that invokes the original workflow and returns the new artifact. The resulting run can then be compared with the previous run to identify evidence-backed changes.
101
+
102
+ ## Statuses
103
+
104
+ - `verified`: execution completed and all declared checks passed with no recorded evidence boundary.
105
+ - `verified_with_boundaries`: execution completed and the artifact is intact, but one or more external or non-deterministic effects remain outside the captured evidence.
106
+ - `verified_with_warnings`: execution completed but comparability or external-source evidence is limited.
107
+ - `failed`: execution or a required check failed.
108
+ - `blocked`: a declared policy prevented a sensitive action.
109
+ - `non_reproducible`: replay was attempted but did not match the captured run.
110
+
111
+ ## Privacy and real data
112
+
113
+ RunProof records metadata and hashes by default. Full input copying is opt-in. Secrets are redacted from environment snapshots and trace values. External adapters must provide privacy-safe request metadata instead of storing credentials or raw authorization headers.
114
+
115
+ ## Project status
116
+
117
+ The current repository implements the local core, automatic observation, provenance graph, environment lock, distributed trace context, and optional real-library integrations. It remains useful without a cloud account or a specific AI provider. Automatic observation still has explicit coverage boundaries; it is not a universal sandbox or a guarantee of perfect replay for arbitrary Python.
118
+
119
+ ## License
120
+
121
+ Apache-2.0. See `LICENSE`.
@@ -1,14 +1,14 @@
1
1
  [build-system]
2
- requires = ["setuptools>=68"]
2
+ requires = ["setuptools>=83.0.0", "wheel>=0.46.2"]
3
3
  build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "runproof-engine"
7
- version = "0.1.2"
7
+ version = "0.2.0"
8
8
  description = "Evidence, replay, and explainable diffs for real Python runs"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
11
- license = { text = "Apache-2.0" }
11
+ license = "Apache-2.0"
12
12
  authors = [
13
13
  { name = "RunProof Contributors" }
14
14
  ]
@@ -17,7 +17,6 @@ classifiers = [
17
17
  "Development Status :: 3 - Alpha",
18
18
  "Intended Audience :: Developers",
19
19
  "Intended Audience :: Science/Research",
20
- "License :: OSI Approved :: Apache Software License",
21
20
  "Programming Language :: Python :: 3",
22
21
  "Programming Language :: Python :: 3 :: Only",
23
22
  "Topic :: Software Development :: Testing",
@@ -26,8 +25,13 @@ classifiers = [
26
25
  dependencies = []
27
26
 
28
27
  [project.optional-dependencies]
29
- data = ["pandas>=2.0"]
30
- dev = ["pytest>=8", "ruff>=0.6"]
28
+ data = ["pandas>=2.0", "polars>=1.0"]
29
+ http = ["requests>=2.31", "httpx>=0.27"]
30
+ database = ["SQLAlchemy>=2.0", "psycopg[binary]>=3.1"]
31
+ jupyter = ["nbclient>=0.10", "nbformat>=5.9"]
32
+ cloud = ["boto3>=1.34"]
33
+ ml = ["joblib>=1.3"]
34
+ dev = ["pytest>=8", "ruff>=0.6", "build>=1.2", "twine>=5", "hypothesis>=6.100", "bandit>=1.7", "pip-audit>=2.7"]
31
35
 
32
36
  [project.scripts]
33
37
  runproof = "runproof_engine.cli:main"
@@ -0,0 +1,82 @@
1
+ from .adapters import Adapter, AdapterInfo, UrllibAdapter, default_adapters
2
+ from .auto import AutoCapture, auto_run
3
+ from .core import (
4
+ CheckRecord,
5
+ ReplayUnavailable,
6
+ RunContext,
7
+ RunProofError,
8
+ RunResult,
9
+ StepRecord,
10
+ verified,
11
+ )
12
+ from .diff import Difference, RunDiff, compare_manifests
13
+ from .environment import compare_environment_lock, environment_lock, reconstruction_plan
14
+ from .integrations import (
15
+ Boto3Adapter,
16
+ HTTPXAdapter,
17
+ JoblibAdapter,
18
+ JupyterAdapter,
19
+ PandasAdapter,
20
+ PolarsAdapter,
21
+ PsycopgAdapter,
22
+ RequestsAdapter,
23
+ SQLAlchemyAdapter,
24
+ SQLiteAdapter,
25
+ SubprocessAdapter,
26
+ TorchAdapter,
27
+ available_adapters,
28
+ )
29
+ from .policy import Policy, PolicyDenied, safe_default_policy
30
+ from .provenance import ProvenanceEdge, ProvenanceGraph, ProvenanceNode
31
+ from .replay import LoadedRun, ReplayReport, load_run
32
+ from .tracing import Span, Tracer, format_traceparent, parse_traceparent
33
+
34
+ __all__ = [
35
+ "Adapter",
36
+ "AdapterInfo",
37
+ "AutoCapture",
38
+ "Boto3Adapter",
39
+ "CheckRecord",
40
+ "Difference",
41
+ "HTTPXAdapter",
42
+ "JoblibAdapter",
43
+ "JupyterAdapter",
44
+ "LoadedRun",
45
+ "PandasAdapter",
46
+ "PolarsAdapter",
47
+ "Policy",
48
+ "PolicyDenied",
49
+ "ProvenanceEdge",
50
+ "ProvenanceGraph",
51
+ "ProvenanceNode",
52
+ "PsycopgAdapter",
53
+ "ReplayReport",
54
+ "ReplayUnavailable",
55
+ "RequestsAdapter",
56
+ "RunContext",
57
+ "RunDiff",
58
+ "RunProofError",
59
+ "RunResult",
60
+ "SQLAlchemyAdapter",
61
+ "SQLiteAdapter",
62
+ "Span",
63
+ "StepRecord",
64
+ "SubprocessAdapter",
65
+ "TorchAdapter",
66
+ "Tracer",
67
+ "UrllibAdapter",
68
+ "auto_run",
69
+ "available_adapters",
70
+ "compare_environment_lock",
71
+ "compare_manifests",
72
+ "default_adapters",
73
+ "environment_lock",
74
+ "format_traceparent",
75
+ "load_run",
76
+ "parse_traceparent",
77
+ "reconstruction_plan",
78
+ "safe_default_policy",
79
+ "verified",
80
+ ]
81
+
82
+ __version__ = "0.2.0"
@@ -0,0 +1,136 @@
1
+ """Optional boundary adapters for automatic RunProof observation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ from dataclasses import dataclass
7
+ from typing import Any, Protocol
8
+ from urllib.parse import parse_qsl, urlencode, urlsplit, urlunsplit
9
+ from urllib.request import Request
10
+
11
+ from .core import RunContext
12
+
13
+
14
+ class Adapter(Protocol):
15
+ """Contract implemented by an automatic boundary adapter."""
16
+
17
+ name: str
18
+
19
+ def install(self, context: RunContext) -> Any:
20
+ """Install hooks and return a zero-argument uninstall callback."""
21
+
22
+
23
+ @dataclass(frozen=True)
24
+ class AdapterInfo:
25
+ name: str
26
+ version: str
27
+ capabilities: tuple[str, ...]
28
+
29
+
30
+ _SECRET_QUERY_MARKERS = (
31
+ "token",
32
+ "key",
33
+ "secret",
34
+ "password",
35
+ "credential",
36
+ "authorization",
37
+ )
38
+
39
+
40
+ def _safe_url(value: str) -> str:
41
+ """Keep URL structure while redacting likely secret query values."""
42
+ try:
43
+ parts = urlsplit(value)
44
+ query = []
45
+ for key, item in parse_qsl(parts.query, keep_blank_values=True):
46
+ safe_item = "[REDACTED]" if any(marker in key.lower() for marker in _SECRET_QUERY_MARKERS) else item
47
+ query.append((key, safe_item))
48
+ return urlunsplit((parts.scheme, parts.netloc, parts.path, urlencode(query), parts.fragment))
49
+ except ValueError:
50
+ return value[:500]
51
+
52
+
53
+ def _request_details(value: Any, data: Any) -> tuple[str, str, str | None]:
54
+ if isinstance(value, Request):
55
+ url = value.full_url
56
+ method = value.method or ("POST" if value.data is not None else "GET")
57
+ body = value.data
58
+ else:
59
+ url = str(value)
60
+ method = "POST" if data is not None else "GET"
61
+ body = data
62
+ body_bytes = body if isinstance(body, bytes) else str(body).encode("utf-8") if body is not None else None
63
+ digest = hashlib.sha256(body_bytes).hexdigest() if body_bytes else None
64
+ return method.upper(), _safe_url(url), digest
65
+
66
+
67
+ class UrllibAdapter:
68
+ """Observe calls through ``urllib.request.urlopen`` without consuming responses."""
69
+
70
+ name = "urllib"
71
+ info = AdapterInfo(
72
+ name="urllib",
73
+ version="1",
74
+ capabilities=("http-request-metadata", "response-status", "secret-query-redaction"),
75
+ )
76
+
77
+ def install(self, context: RunContext) -> Any:
78
+ import urllib.request as urllib_request
79
+
80
+ original = urllib_request.urlopen
81
+ adapter = self
82
+
83
+ def wrapped(*args: Any, **kwargs: Any) -> Any:
84
+ request_value = args[0] if args else kwargs.get("url")
85
+ data = args[1] if len(args) > 1 else kwargs.get("data")
86
+ method, url, body_sha256 = _request_details(request_value, data)
87
+ try:
88
+ response = original(*args, **kwargs)
89
+ except Exception as error:
90
+ context._event(
91
+ "http_request_failed",
92
+ {
93
+ "adapter": adapter.name,
94
+ "method": method,
95
+ "url": url,
96
+ "error_type": f"{type(error).__module__}.{type(error).__qualname__}",
97
+ "message": str(error),
98
+ },
99
+ )
100
+ raise
101
+ status_code = getattr(response, "status", None) or getattr(response, "code", None)
102
+ context._event(
103
+ "http_request_observed",
104
+ {
105
+ "adapter": adapter.name,
106
+ "method": method,
107
+ "url": url,
108
+ "body_sha256": body_sha256,
109
+ "status_code": status_code,
110
+ },
111
+ )
112
+ context.boundary(
113
+ "network_response",
114
+ target=url,
115
+ reason="automatic urllib adapter does not consume or archive the response body",
116
+ replay="requires_live_service",
117
+ )
118
+ return response
119
+
120
+ urllib_request.urlopen = wrapped
121
+ context._event("adapter_installed", {"name": self.name, "version": self.info.version})
122
+
123
+ def uninstall() -> None:
124
+ if urllib_request.urlopen is wrapped:
125
+ urllib_request.urlopen = original
126
+ context._event("adapter_uninstalled", {"name": self.name})
127
+
128
+ return uninstall
129
+
130
+
131
+ def default_adapters() -> tuple[Adapter, ...]:
132
+ """Return safe standard-library adapters enabled by automatic capture."""
133
+ return (UrllibAdapter(),)
134
+
135
+
136
+ __all__ = ["Adapter", "AdapterInfo", "UrllibAdapter", "default_adapters"]