replayguard 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. replayguard-0.1.0/.github/workflows/ci.yml +51 -0
  2. replayguard-0.1.0/.github/workflows/publish.yml +52 -0
  3. replayguard-0.1.0/.gitignore +14 -0
  4. replayguard-0.1.0/.pre-commit-hooks.yaml +11 -0
  5. replayguard-0.1.0/LICENSE +21 -0
  6. replayguard-0.1.0/PKG-INFO +297 -0
  7. replayguard-0.1.0/README.md +249 -0
  8. replayguard-0.1.0/VALIDATION.md +328 -0
  9. replayguard-0.1.0/action.yml +61 -0
  10. replayguard-0.1.0/experiments/README.md +86 -0
  11. replayguard-0.1.0/experiments/durable_invoke_probe.py +279 -0
  12. replayguard-0.1.0/experiments/durable_runtime_probe.py +200 -0
  13. replayguard-0.1.0/experiments/durable_suspend_resume_probe.py +376 -0
  14. replayguard-0.1.0/pyproject.toml +107 -0
  15. replayguard-0.1.0/scripts/verify.py +203 -0
  16. replayguard-0.1.0/src/replayguard/__init__.py +0 -0
  17. replayguard-0.1.0/src/replayguard/catalog.py +272 -0
  18. replayguard-0.1.0/src/replayguard/cli.py +254 -0
  19. replayguard-0.1.0/src/replayguard/dynamic/__init__.py +19 -0
  20. replayguard-0.1.0/src/replayguard/dynamic/diverge.py +246 -0
  21. replayguard-0.1.0/src/replayguard/dynamic/journal.py +95 -0
  22. replayguard-0.1.0/src/replayguard/dynamic/perturb.py +122 -0
  23. replayguard-0.1.0/src/replayguard/findings.py +68 -0
  24. replayguard-0.1.0/src/replayguard/frontends/__init__.py +0 -0
  25. replayguard-0.1.0/src/replayguard/frontends/java_frontend.py +791 -0
  26. replayguard-0.1.0/src/replayguard/frontends/python_frontend.py +754 -0
  27. replayguard-0.1.0/src/replayguard/frontends/rust_frontend.py +646 -0
  28. replayguard-0.1.0/src/replayguard/frontends/typescript_frontend.py +824 -0
  29. replayguard-0.1.0/src/replayguard/ir.py +195 -0
  30. replayguard-0.1.0/src/replayguard/report.py +160 -0
  31. replayguard-0.1.0/src/replayguard/rules.py +245 -0
  32. replayguard-0.1.0/tests/__init__.py +0 -0
  33. replayguard-0.1.0/tests/fixtures/java/BadHandler.java +95 -0
  34. replayguard-0.1.0/tests/fixtures/java/GoodHandler.java +67 -0
  35. replayguard-0.1.0/tests/fixtures/python/bad_handler.py +81 -0
  36. replayguard-0.1.0/tests/fixtures/python/good_handler.py +54 -0
  37. replayguard-0.1.0/tests/fixtures/rust/bad_handler.rs +56 -0
  38. replayguard-0.1.0/tests/fixtures/rust/good_handler.rs +47 -0
  39. replayguard-0.1.0/tests/fixtures/typescript/bad_handler.ts +69 -0
  40. replayguard-0.1.0/tests/fixtures/typescript/good_handler.ts +52 -0
  41. replayguard-0.1.0/tests/test_cli.py +198 -0
  42. replayguard-0.1.0/tests/test_cli_replay.py +128 -0
  43. replayguard-0.1.0/tests/test_core.py +231 -0
  44. replayguard-0.1.0/tests/test_dynamic.py +318 -0
  45. replayguard-0.1.0/tests/test_java_frontend.py +326 -0
  46. replayguard-0.1.0/tests/test_python_frontend.py +298 -0
  47. replayguard-0.1.0/tests/test_real_world_regressions.py +433 -0
  48. replayguard-0.1.0/tests/test_report.py +180 -0
  49. replayguard-0.1.0/tests/test_rule_matrix.py +229 -0
  50. replayguard-0.1.0/tests/test_rust_frontend.py +292 -0
  51. replayguard-0.1.0/tests/test_typescript_frontend.py +290 -0
@@ -0,0 +1,51 @@
1
+ name: ci
2
+
3
+ on:
4
+ push:
5
+ branches: [main]
6
+ pull_request:
7
+
8
+ jobs:
9
+ verify:
10
+ runs-on: ubuntu-latest
11
+ strategy:
12
+ fail-fast: false
13
+ matrix:
14
+ # The floor is what pyproject declares; the ceiling catches deprecations
15
+ # before they become someone else's bug report.
16
+ python-version: ["3.11", "3.13"]
17
+ steps:
18
+ - uses: actions/checkout@v4
19
+
20
+ - uses: actions/setup-python@v5
21
+ with:
22
+ python-version: ${{ matrix.python-version }}
23
+
24
+ - name: Install with every extra
25
+ run: |
26
+ python -m pip install --upgrade pip
27
+ python -m pip install -e ".[dev]" ruff
28
+
29
+ - name: Verify
30
+ # One command, five gates. Whatever passes locally passes here, which is
31
+ # the point of having a single entry point rather than a list of steps
32
+ # that drift apart from scripts/verify.py.
33
+ run: python scripts/verify.py
34
+
35
+ self-check:
36
+ # Dogfooding: the checker runs over its own fixtures and must exit clean on
37
+ # the good ones. A tool that cannot pass its own gate has no business
38
+ # annotating anyone else's pull request.
39
+ runs-on: ubuntu-latest
40
+ steps:
41
+ - uses: actions/checkout@v4
42
+ - uses: actions/setup-python@v5
43
+ with:
44
+ python-version: "3.13"
45
+ - run: python -m pip install -e ".[all]"
46
+ - name: Check the known-good fixtures
47
+ run: |
48
+ replayguard check tests/fixtures/python/good_handler.py
49
+ replayguard check tests/fixtures/typescript/good_handler.ts
50
+ replayguard check tests/fixtures/java/GoodHandler.java
51
+ replayguard check tests/fixtures/rust/good_handler.rs
@@ -0,0 +1,52 @@
1
+ name: publish
2
+
3
+ # Publishing runs on a published GitHub Release, not on a tag push, so
4
+ # creating a release is the single deliberate act that ships a version.
5
+ on:
6
+ release:
7
+ types: [published]
8
+
9
+ jobs:
10
+ build:
11
+ runs-on: ubuntu-latest
12
+ steps:
13
+ - uses: actions/checkout@v4
14
+
15
+ - uses: actions/setup-python@v5
16
+ with:
17
+ python-version: "3.13"
18
+
19
+ - name: Install
20
+ run: |
21
+ python -m pip install --upgrade pip
22
+ python -m pip install -e ".[dev]" ruff build
23
+
24
+ - name: Verify before shipping
25
+ # A release is built from whatever commit was tagged, which is not
26
+ # necessarily a commit CI has seen. Re-running the gate here is what
27
+ # stops a broken artifact reaching PyPI, where it cannot be replaced.
28
+ run: python scripts/verify.py
29
+
30
+ - name: Build sdist and wheel
31
+ run: python -m build
32
+
33
+ - uses: actions/upload-artifact@v4
34
+ with:
35
+ name: dist
36
+ path: dist/
37
+
38
+ publish:
39
+ needs: build
40
+ runs-on: ubuntu-latest
41
+ environment: pypi
42
+ permissions:
43
+ # Trusted publishing: PyPI verifies a short-lived OIDC token issued to
44
+ # this workflow, so no API token exists to leak, rotate, or commit.
45
+ id-token: write
46
+ steps:
47
+ - uses: actions/download-artifact@v4
48
+ with:
49
+ name: dist
50
+ path: dist/
51
+
52
+ - uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,14 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ .pytest_cache/
4
+ .mypy_cache/
5
+ *.egg-info/
6
+ dist/
7
+ build/
8
+ .venv/
9
+ venv/
10
+ *.sarif
11
+ .coverage
12
+ .coverage.*
13
+ coverage.xml
14
+ htmlcov/
@@ -0,0 +1,11 @@
1
+ - id: replayguard
2
+ name: replayguard (durable determinism)
3
+ description: >-
4
+ Check AWS Lambda durable functions for replay-determinism bugs before the
5
+ commit exists, rather than on a resume months later.
6
+ entry: replayguard check
7
+ language: python
8
+ # Only the four languages with a durable execution SDK. A repo is mostly not
9
+ # durable handlers, and running over everything would be wasted work.
10
+ types_or: [python, ts, java, rust]
11
+ pass_filenames: true
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Amrut Pagidipally
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,297 @@
1
+ Metadata-Version: 2.5
2
+ Name: replayguard
3
+ Version: 0.1.0
4
+ Summary: Determinism checker for AWS Lambda durable functions
5
+ Project-URL: Homepage, https://github.com/amrutp24/replayguard
6
+ Project-URL: Issues, https://github.com/amrutp24/replayguard/issues
7
+ Author-email: Amrut Pagidipally <amrut.pagidipally@gmail.com>
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: aws,determinism,durable-functions,lambda,linter,static-analysis
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Programming Language :: Python :: 3.11
14
+ Classifier: Programming Language :: Python :: 3.12
15
+ Classifier: Programming Language :: Python :: 3.13
16
+ Classifier: Topic :: Software Development :: Quality Assurance
17
+ Classifier: Topic :: Software Development :: Testing
18
+ Requires-Python: >=3.11
19
+ Provides-Extra: all
20
+ Requires-Dist: aws-durable-execution-sdk-python-testing>=1.2; extra == 'all'
21
+ Requires-Dist: aws-durable-execution-sdk-python>=1.7; extra == 'all'
22
+ Requires-Dist: tree-sitter-java>=0.23; extra == 'all'
23
+ Requires-Dist: tree-sitter-rust>=0.23; extra == 'all'
24
+ Requires-Dist: tree-sitter-typescript>=0.23; extra == 'all'
25
+ Requires-Dist: tree-sitter>=0.25; extra == 'all'
26
+ Provides-Extra: dev
27
+ Requires-Dist: aws-durable-execution-sdk-python-testing>=1.2; extra == 'dev'
28
+ Requires-Dist: aws-durable-execution-sdk-python>=1.7; extra == 'dev'
29
+ Requires-Dist: pytest-cov>=5; extra == 'dev'
30
+ Requires-Dist: pytest>=8; extra == 'dev'
31
+ Requires-Dist: tree-sitter-java>=0.23; extra == 'dev'
32
+ Requires-Dist: tree-sitter-rust>=0.23; extra == 'dev'
33
+ Requires-Dist: tree-sitter-typescript>=0.23; extra == 'dev'
34
+ Requires-Dist: tree-sitter>=0.25; extra == 'dev'
35
+ Provides-Extra: dynamic
36
+ Requires-Dist: aws-durable-execution-sdk-python-testing>=1.2; extra == 'dynamic'
37
+ Requires-Dist: aws-durable-execution-sdk-python>=1.7; extra == 'dynamic'
38
+ Provides-Extra: java
39
+ Requires-Dist: tree-sitter-java>=0.23; extra == 'java'
40
+ Requires-Dist: tree-sitter>=0.25; extra == 'java'
41
+ Provides-Extra: rust
42
+ Requires-Dist: tree-sitter-rust>=0.23; extra == 'rust'
43
+ Requires-Dist: tree-sitter>=0.25; extra == 'rust'
44
+ Provides-Extra: typescript
45
+ Requires-Dist: tree-sitter-typescript>=0.23; extra == 'typescript'
46
+ Requires-Dist: tree-sitter>=0.25; extra == 'typescript'
47
+ Description-Content-Type: text/markdown
48
+
49
+ # replayguard
50
+
51
+ A determinism checker for AWS Lambda durable functions.
52
+
53
+ Durable functions re-run your handler from the top on every resume. Completed
54
+ steps aren't re-executed; the SDK returns the checkpointed result and the
55
+ handler fast-forwards back to where it suspended. This only works if the
56
+ handler takes the same path every time, and AWS's docs are explicit about
57
+ whose job that is:
58
+
59
+ > Any code that is not inside a durable operation must be a pure function of the
60
+ > handler inputs and the results of completed operations.
61
+
62
+ No clocks, no randomness, no I/O, no writes to shared state outside a `step()`.
63
+ Nothing enforces this. Break the rule and nothing throws, your tests stay
64
+ green, and the bug surfaces whenever the workflow next resumes. An execution
65
+ can suspend for up to 366 days, so that can be months later, in production.
66
+ AWS's docs name double-charging as a possible consequence.
67
+
68
+ replayguard checks the rule two ways: statically, by analyzing handler source
69
+ in Python, TypeScript, Java, and Rust, and dynamically, by running a handler
70
+ twice under different clocks and diffing what it did.
71
+
72
+ ## Status
73
+
74
+ v0.1.0. Validated against 1,547 files of durable-function code written by
75
+ other people; [VALIDATION.md](VALIDATION.md) records what that established and
76
+ what it didn't.
77
+
78
+ Two known gaps. Calls aren't followed across files, so a handler that reaches
79
+ another module for its I/O passes clean. And three of the six rules have
80
+ working detectors but no confirmed real-world finding yet, because published
81
+ example code doesn't contain the mistakes they catch.
82
+
83
+ Reports from real codebases are the most useful thing anyone can contribute
84
+ right now, in either direction: a finding it caught, or a false positive it
85
+ shouldn't have raised.
86
+
87
+ ## Install
88
+
89
+ ```bash
90
+ pip install replayguard
91
+ ```
92
+
93
+ TypeScript, Java, and Rust need a parser:
94
+
95
+ ```bash
96
+ pip install 'replayguard[all]'
97
+ ```
98
+
99
+ ## Use
100
+
101
+ ```bash
102
+ replayguard check src/
103
+ replayguard check src/ --explain
104
+ replayguard check src/ --format sarif -o replayguard.sarif
105
+ ```
106
+
107
+ Exits non-zero when anything at or above `--fail-on` (default `error`) is
108
+ found, so it can gate a build without extra wiring.
109
+
110
+ ```
111
+ tests/fixtures/python/bad_handler.py
112
+ 25:17 error RG001 `time.time` runs outside a durable step
113
+ 37:14 error RG002 external I/O `requests.get` runs outside a durable step
114
+ 53:8 error RG003 step body writes to a captured variable `receipts`
115
+ 43:7 error RG004 branch condition depends on `datetime.datetime.now`
116
+ 68:4 error RG005 `step` name is built from `time.time`
117
+ 71:17 note RG900 could not resolve whether this code runs inside a step
118
+ ```
119
+
120
+ ## In CI
121
+
122
+ SARIF output means findings render inline on the pull request that introduced
123
+ them:
124
+
125
+ ```yaml
126
+ - uses: amrutp24/replayguard@v1
127
+ with:
128
+ path: src/
129
+ ```
130
+
131
+ The job needs `permissions: security-events: write` for the annotations. Set
132
+ `fail-on: never` to annotate without blocking the merge.
133
+
134
+ There is also a pre-commit hook:
135
+
136
+ ```yaml
137
+ - repo: https://github.com/amrutp24/replayguard
138
+ rev: v0.1.0
139
+ hooks:
140
+ - id: replayguard
141
+ ```
142
+
143
+ ## Rules
144
+
145
+ | ID | What it catches | Why it breaks replay |
146
+ |----|-----------------|----------------------|
147
+ | **RG001** | Clock, random, or identity source outside a step | Produces a different value on replay; everything derived from it diverges |
148
+ | **RG002** | Network or filesystem access outside a step | Diverges, and repeats the side effect on every replay |
149
+ | **RG003** | A step body writing to state it doesn't own | The write lands on the first run and is skipped on replay, so the outer state silently reverts |
150
+ | **RG004** | Control flow depending on a nondeterministic value | Replay can take the other branch, so the operation sequence no longer matches the journal |
151
+ | **RG005** | A step name built from an unstable source | Checkpoints match by name and order; a changed name can't be matched, so the step re-executes |
152
+ | **RG900** | Code whose region couldn't be resolved | Not a violation. A coverage gap, reported so a clean run means something |
153
+
154
+ `replayguard rules --explain` prints the rationale for each.
155
+
156
+ ## Language support
157
+
158
+ Python, TypeScript/JavaScript, and Java are the three runtimes with an
159
+ official AWS durable execution SDK. Rust has no official SDK; the frontend
160
+ targets [pgdad/durable-rust](https://github.com/pgdad/durable-rust), whose own
161
+ documentation says its determinism rules are documented rather than enforced.
162
+ Go and .NET have community proofs of concept only and are out of scope for
163
+ now.
164
+
165
+ | Runtime | Parser |
166
+ |---------|--------|
167
+ | Python | stdlib `ast` |
168
+ | TypeScript / JavaScript | tree-sitter |
169
+ | Java | tree-sitter |
170
+ | Rust | tree-sitter |
171
+
172
+ The SDKs differ in shape, not just syntax:
173
+
174
+ | | Handler | Step |
175
+ |---|---|---|
176
+ | Python | `@durable_execution` | `context.step(fn, name="x")` |
177
+ | JS/TS | `withDurableExecution(fn)` | `context.step("x", fn)` |
178
+ | Java | `extends DurableHandler<,>` | `ctx.step("x", Result.class, fn)` |
179
+ | Rust | param typed `*Context` | `ctx.step("x", \|\| async { .. })` |
180
+
181
+ The step body sits in a different argument position in each, and Java has a
182
+ two-argument overload besides, so bodies are located by kind rather than by
183
+ position. Everything lowers to one shared IR and the rules are written once,
184
+ with no knowledge of which language they're inspecting.
185
+
186
+ The semantic differences that matter are handled per-frontend, and RG003 is
187
+ where they show up:
188
+
189
+ - **Python**: a bare `x = 1` in a nested function creates a local binding, so
190
+ it can never be an outer write. Only mutation and `global`/`nonlocal` reach
191
+ out.
192
+ - **JavaScript**: the same assignment writes straight through to the enclosing
193
+ scope, so RG003 has more ways to fire.
194
+ - **Java**: captured locals must be effectively final, so reassigning one is a
195
+ compile error and that violation class can't exist. What remains is
196
+ collection mutation and field writes.
197
+ - **Rust**: the narrowest of the four. Step closures are `Send + 'static`, so
198
+ capturing a borrowed reference doesn't compile. What's left is interior
199
+ mutability through a shared handle, like an `Arc<Mutex<_>>` locked and
200
+ pushed to, or a `static mut`.
201
+
202
+ ## Design
203
+
204
+ ```
205
+ source ──▶ frontend ──▶ IR ──▶ rules ──▶ findings ──▶ reporter
206
+ (per-lang) (shared) (shared) text/json/sarif
207
+ ```
208
+
209
+ This is AST analysis, not pattern matching, because the questions need scope
210
+ resolution: RG003 has to know whether a mutated name belongs to the step body
211
+ or an enclosing scope, and RG004 has to know whether a branch condition
212
+ derives from a nondeterministic source. Neither can be answered by matching
213
+ source text.
214
+
215
+ False positives get particular attention, since a linter that fires on
216
+ correct code gets uninstalled. RG005 ignores computed step names unless they
217
+ interpolate something genuinely unstable (`` `item-${index}` `` is the pattern
218
+ AWS recommends), and RG900 reports unresolved regions instead of silently
219
+ passing them.
220
+
221
+ ## What it doesn't do
222
+
223
+ Calls are not followed across file boundaries. Within a file they are, and
224
+ findings name the route, but a handler that calls into another module for its
225
+ I/O will pass clean. This is the largest known blind spot.
226
+
227
+ Static analysis also can't see nondeterminism inside a third-party library,
228
+ data tainted several hops back, iteration order over an unordered collection,
229
+ or concurrent completion order. Some of that the dynamic half can catch.
230
+
231
+ ## Dynamic replay-divergence
232
+
233
+ The static rules reason about what code might do. The dynamic harness runs
234
+ the handler twice, once normally and once with the clock moved and entropy
235
+ reseeded, and diffs the operation journals:
236
+
237
+ ```bash
238
+ replayguard replay app.orders:handler --event '{"orderId": "A1"}'
239
+ ```
240
+
241
+ ```
242
+ replay-divergence: 1 divergence(s) found.
243
+
244
+ operation 0: operation name changed -- checkpoints match by name
245
+ control : step(op-1787442395)
246
+ perturbed : step(op-1787489626)
247
+ ```
248
+
249
+ Or as an assertion next to the handler, so a determinism regression fails the
250
+ build instead of surfacing on a resume months later:
251
+
252
+ ```python
253
+ from replayguard.dynamic import assert_deterministic
254
+
255
+ def test_handler_is_deterministic():
256
+ assert_deterministic(handler, {"orderId": "A1"})
257
+ ```
258
+
259
+ The harness needs no rule for the source of nondeterminism. A clock inside a
260
+ library, an iteration order, a value tainted many hops back: it measures the
261
+ effect, not the cause. On 51 AWS conformance handlers it produced zero false
262
+ alarms; on a handler whose step order comes from `random.sample`, which no
263
+ static rule covers, it diverges.
264
+
265
+ It can't prove determinism, only fail to disprove it, and the report says so.
266
+ Handlers that suspend on a callback can't be checked locally.
267
+
268
+ ## Validation
269
+
270
+ [VALIDATION.md](VALIDATION.md) records what has been tested against whose
271
+ code: which rules have confirmed real-world findings, which are still
272
+ unproven, the false positives that were found and fixed, and the bugs the
273
+ validation found in the tool itself. Read it before relying on a clean run.
274
+
275
+ ## Developing
276
+
277
+ ```bash
278
+ pip install -e ".[dev]"
279
+ python scripts/verify.py
280
+ ```
281
+
282
+ `verify.py` runs five gates: import, lint, tests with a coverage floor, the
283
+ CLI's exit codes and output formats, and a canary asserting the known-good
284
+ fixtures produce zero findings in every language. The canary matters most; a
285
+ false positive is a worse failure here than a missed bug.
286
+
287
+ ## Prior art
288
+
289
+ [Temporal's workflowcheck](https://github.com/temporalio/sdk-go) does this for
290
+ Temporal workflows, so the category is proven; it just didn't exist for AWS's
291
+ primitive. [durable-viz](https://github.com/gunnargrosch/durable-viz)
292
+ statically analyses durable handlers to draw flowcharts, but performs no
293
+ validation.
294
+
295
+ ## License
296
+
297
+ MIT