replayguard 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- replayguard-0.1.0/.github/workflows/ci.yml +51 -0
- replayguard-0.1.0/.github/workflows/publish.yml +52 -0
- replayguard-0.1.0/.gitignore +14 -0
- replayguard-0.1.0/.pre-commit-hooks.yaml +11 -0
- replayguard-0.1.0/LICENSE +21 -0
- replayguard-0.1.0/PKG-INFO +297 -0
- replayguard-0.1.0/README.md +249 -0
- replayguard-0.1.0/VALIDATION.md +328 -0
- replayguard-0.1.0/action.yml +61 -0
- replayguard-0.1.0/experiments/README.md +86 -0
- replayguard-0.1.0/experiments/durable_invoke_probe.py +279 -0
- replayguard-0.1.0/experiments/durable_runtime_probe.py +200 -0
- replayguard-0.1.0/experiments/durable_suspend_resume_probe.py +376 -0
- replayguard-0.1.0/pyproject.toml +107 -0
- replayguard-0.1.0/scripts/verify.py +203 -0
- replayguard-0.1.0/src/replayguard/__init__.py +0 -0
- replayguard-0.1.0/src/replayguard/catalog.py +272 -0
- replayguard-0.1.0/src/replayguard/cli.py +254 -0
- replayguard-0.1.0/src/replayguard/dynamic/__init__.py +19 -0
- replayguard-0.1.0/src/replayguard/dynamic/diverge.py +246 -0
- replayguard-0.1.0/src/replayguard/dynamic/journal.py +95 -0
- replayguard-0.1.0/src/replayguard/dynamic/perturb.py +122 -0
- replayguard-0.1.0/src/replayguard/findings.py +68 -0
- replayguard-0.1.0/src/replayguard/frontends/__init__.py +0 -0
- replayguard-0.1.0/src/replayguard/frontends/java_frontend.py +791 -0
- replayguard-0.1.0/src/replayguard/frontends/python_frontend.py +754 -0
- replayguard-0.1.0/src/replayguard/frontends/rust_frontend.py +646 -0
- replayguard-0.1.0/src/replayguard/frontends/typescript_frontend.py +824 -0
- replayguard-0.1.0/src/replayguard/ir.py +195 -0
- replayguard-0.1.0/src/replayguard/report.py +160 -0
- replayguard-0.1.0/src/replayguard/rules.py +245 -0
- replayguard-0.1.0/tests/__init__.py +0 -0
- replayguard-0.1.0/tests/fixtures/java/BadHandler.java +95 -0
- replayguard-0.1.0/tests/fixtures/java/GoodHandler.java +67 -0
- replayguard-0.1.0/tests/fixtures/python/bad_handler.py +81 -0
- replayguard-0.1.0/tests/fixtures/python/good_handler.py +54 -0
- replayguard-0.1.0/tests/fixtures/rust/bad_handler.rs +56 -0
- replayguard-0.1.0/tests/fixtures/rust/good_handler.rs +47 -0
- replayguard-0.1.0/tests/fixtures/typescript/bad_handler.ts +69 -0
- replayguard-0.1.0/tests/fixtures/typescript/good_handler.ts +52 -0
- replayguard-0.1.0/tests/test_cli.py +198 -0
- replayguard-0.1.0/tests/test_cli_replay.py +128 -0
- replayguard-0.1.0/tests/test_core.py +231 -0
- replayguard-0.1.0/tests/test_dynamic.py +318 -0
- replayguard-0.1.0/tests/test_java_frontend.py +326 -0
- replayguard-0.1.0/tests/test_python_frontend.py +298 -0
- replayguard-0.1.0/tests/test_real_world_regressions.py +433 -0
- replayguard-0.1.0/tests/test_report.py +180 -0
- replayguard-0.1.0/tests/test_rule_matrix.py +229 -0
- replayguard-0.1.0/tests/test_rust_frontend.py +292 -0
- replayguard-0.1.0/tests/test_typescript_frontend.py +290 -0
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
name: ci
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
verify:
|
|
10
|
+
runs-on: ubuntu-latest
|
|
11
|
+
strategy:
|
|
12
|
+
fail-fast: false
|
|
13
|
+
matrix:
|
|
14
|
+
# The floor is what pyproject declares; the ceiling catches deprecations
|
|
15
|
+
# before they become someone else's bug report.
|
|
16
|
+
python-version: ["3.11", "3.13"]
|
|
17
|
+
steps:
|
|
18
|
+
- uses: actions/checkout@v4
|
|
19
|
+
|
|
20
|
+
- uses: actions/setup-python@v5
|
|
21
|
+
with:
|
|
22
|
+
python-version: ${{ matrix.python-version }}
|
|
23
|
+
|
|
24
|
+
- name: Install with every extra
|
|
25
|
+
run: |
|
|
26
|
+
python -m pip install --upgrade pip
|
|
27
|
+
python -m pip install -e ".[dev]" ruff
|
|
28
|
+
|
|
29
|
+
- name: Verify
|
|
30
|
+
# One command, five gates. Whatever passes locally passes here, which is
|
|
31
|
+
# the point of having a single entry point rather than a list of steps
|
|
32
|
+
# that drift apart from scripts/verify.py.
|
|
33
|
+
run: python scripts/verify.py
|
|
34
|
+
|
|
35
|
+
self-check:
|
|
36
|
+
# Dogfooding: the checker runs over its own fixtures and must exit clean on
|
|
37
|
+
# the good ones. A tool that cannot pass its own gate has no business
|
|
38
|
+
# annotating anyone else's pull request.
|
|
39
|
+
runs-on: ubuntu-latest
|
|
40
|
+
steps:
|
|
41
|
+
- uses: actions/checkout@v4
|
|
42
|
+
- uses: actions/setup-python@v5
|
|
43
|
+
with:
|
|
44
|
+
python-version: "3.13"
|
|
45
|
+
- run: python -m pip install -e ".[all]"
|
|
46
|
+
- name: Check the known-good fixtures
|
|
47
|
+
run: |
|
|
48
|
+
replayguard check tests/fixtures/python/good_handler.py
|
|
49
|
+
replayguard check tests/fixtures/typescript/good_handler.ts
|
|
50
|
+
replayguard check tests/fixtures/java/GoodHandler.java
|
|
51
|
+
replayguard check tests/fixtures/rust/good_handler.rs
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
name: publish
|
|
2
|
+
|
|
3
|
+
# Publishing runs on a published GitHub Release, not on a tag push, so
|
|
4
|
+
# creating a release is the single deliberate act that ships a version.
|
|
5
|
+
on:
|
|
6
|
+
release:
|
|
7
|
+
types: [published]
|
|
8
|
+
|
|
9
|
+
jobs:
|
|
10
|
+
build:
|
|
11
|
+
runs-on: ubuntu-latest
|
|
12
|
+
steps:
|
|
13
|
+
- uses: actions/checkout@v4
|
|
14
|
+
|
|
15
|
+
- uses: actions/setup-python@v5
|
|
16
|
+
with:
|
|
17
|
+
python-version: "3.13"
|
|
18
|
+
|
|
19
|
+
- name: Install
|
|
20
|
+
run: |
|
|
21
|
+
python -m pip install --upgrade pip
|
|
22
|
+
python -m pip install -e ".[dev]" ruff build
|
|
23
|
+
|
|
24
|
+
- name: Verify before shipping
|
|
25
|
+
# A release is built from whatever commit was tagged, which is not
|
|
26
|
+
# necessarily a commit CI has seen. Re-running the gate here is what
|
|
27
|
+
# stops a broken artifact reaching PyPI, where it cannot be replaced.
|
|
28
|
+
run: python scripts/verify.py
|
|
29
|
+
|
|
30
|
+
- name: Build sdist and wheel
|
|
31
|
+
run: python -m build
|
|
32
|
+
|
|
33
|
+
- uses: actions/upload-artifact@v4
|
|
34
|
+
with:
|
|
35
|
+
name: dist
|
|
36
|
+
path: dist/
|
|
37
|
+
|
|
38
|
+
publish:
|
|
39
|
+
needs: build
|
|
40
|
+
runs-on: ubuntu-latest
|
|
41
|
+
environment: pypi
|
|
42
|
+
permissions:
|
|
43
|
+
# Trusted publishing: PyPI verifies a short-lived OIDC token issued to
|
|
44
|
+
# this workflow, so no API token exists to leak, rotate, or commit.
|
|
45
|
+
id-token: write
|
|
46
|
+
steps:
|
|
47
|
+
- uses: actions/download-artifact@v4
|
|
48
|
+
with:
|
|
49
|
+
name: dist
|
|
50
|
+
path: dist/
|
|
51
|
+
|
|
52
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
- id: replayguard
|
|
2
|
+
name: replayguard (durable determinism)
|
|
3
|
+
description: >-
|
|
4
|
+
Check AWS Lambda durable functions for replay-determinism bugs before the
|
|
5
|
+
commit exists, rather than on a resume months later.
|
|
6
|
+
entry: replayguard check
|
|
7
|
+
language: python
|
|
8
|
+
# Only the four languages with a durable execution SDK. A repo is mostly not
|
|
9
|
+
# durable handlers, and running over everything would be wasted work.
|
|
10
|
+
types_or: [python, ts, java, rust]
|
|
11
|
+
pass_filenames: true
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Amrut Pagidipally
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,297 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: replayguard
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Determinism checker for AWS Lambda durable functions
|
|
5
|
+
Project-URL: Homepage, https://github.com/amrutp24/replayguard
|
|
6
|
+
Project-URL: Issues, https://github.com/amrutp24/replayguard/issues
|
|
7
|
+
Author-email: Amrut Pagidipally <amrut.pagidipally@gmail.com>
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: aws,determinism,durable-functions,lambda,linter,static-analysis
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
16
|
+
Classifier: Topic :: Software Development :: Quality Assurance
|
|
17
|
+
Classifier: Topic :: Software Development :: Testing
|
|
18
|
+
Requires-Python: >=3.11
|
|
19
|
+
Provides-Extra: all
|
|
20
|
+
Requires-Dist: aws-durable-execution-sdk-python-testing>=1.2; extra == 'all'
|
|
21
|
+
Requires-Dist: aws-durable-execution-sdk-python>=1.7; extra == 'all'
|
|
22
|
+
Requires-Dist: tree-sitter-java>=0.23; extra == 'all'
|
|
23
|
+
Requires-Dist: tree-sitter-rust>=0.23; extra == 'all'
|
|
24
|
+
Requires-Dist: tree-sitter-typescript>=0.23; extra == 'all'
|
|
25
|
+
Requires-Dist: tree-sitter>=0.25; extra == 'all'
|
|
26
|
+
Provides-Extra: dev
|
|
27
|
+
Requires-Dist: aws-durable-execution-sdk-python-testing>=1.2; extra == 'dev'
|
|
28
|
+
Requires-Dist: aws-durable-execution-sdk-python>=1.7; extra == 'dev'
|
|
29
|
+
Requires-Dist: pytest-cov>=5; extra == 'dev'
|
|
30
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
31
|
+
Requires-Dist: tree-sitter-java>=0.23; extra == 'dev'
|
|
32
|
+
Requires-Dist: tree-sitter-rust>=0.23; extra == 'dev'
|
|
33
|
+
Requires-Dist: tree-sitter-typescript>=0.23; extra == 'dev'
|
|
34
|
+
Requires-Dist: tree-sitter>=0.25; extra == 'dev'
|
|
35
|
+
Provides-Extra: dynamic
|
|
36
|
+
Requires-Dist: aws-durable-execution-sdk-python-testing>=1.2; extra == 'dynamic'
|
|
37
|
+
Requires-Dist: aws-durable-execution-sdk-python>=1.7; extra == 'dynamic'
|
|
38
|
+
Provides-Extra: java
|
|
39
|
+
Requires-Dist: tree-sitter-java>=0.23; extra == 'java'
|
|
40
|
+
Requires-Dist: tree-sitter>=0.25; extra == 'java'
|
|
41
|
+
Provides-Extra: rust
|
|
42
|
+
Requires-Dist: tree-sitter-rust>=0.23; extra == 'rust'
|
|
43
|
+
Requires-Dist: tree-sitter>=0.25; extra == 'rust'
|
|
44
|
+
Provides-Extra: typescript
|
|
45
|
+
Requires-Dist: tree-sitter-typescript>=0.23; extra == 'typescript'
|
|
46
|
+
Requires-Dist: tree-sitter>=0.25; extra == 'typescript'
|
|
47
|
+
Description-Content-Type: text/markdown
|
|
48
|
+
|
|
49
|
+
# replayguard
|
|
50
|
+
|
|
51
|
+
A determinism checker for AWS Lambda durable functions.
|
|
52
|
+
|
|
53
|
+
Durable functions re-run your handler from the top on every resume. Completed
|
|
54
|
+
steps aren't re-executed; the SDK returns the checkpointed result and the
|
|
55
|
+
handler fast-forwards back to where it suspended. This only works if the
|
|
56
|
+
handler takes the same path every time, and AWS's docs are explicit about
|
|
57
|
+
whose job that is:
|
|
58
|
+
|
|
59
|
+
> Any code that is not inside a durable operation must be a pure function of the
|
|
60
|
+
> handler inputs and the results of completed operations.
|
|
61
|
+
|
|
62
|
+
No clocks, no randomness, no I/O, no writes to shared state outside a `step()`.
|
|
63
|
+
Nothing enforces this. Break the rule and nothing throws, your tests stay
|
|
64
|
+
green, and the bug surfaces whenever the workflow next resumes. An execution
|
|
65
|
+
can suspend for up to 366 days, so that can be months later, in production.
|
|
66
|
+
AWS's docs name double-charging as a possible consequence.
|
|
67
|
+
|
|
68
|
+
replayguard checks the rule two ways: statically, by analyzing handler source
|
|
69
|
+
in Python, TypeScript, Java, and Rust, and dynamically, by running a handler
|
|
70
|
+
twice under different clocks and diffing what it did.
|
|
71
|
+
|
|
72
|
+
## Status
|
|
73
|
+
|
|
74
|
+
v0.1.0. Validated against 1,547 files of durable-function code written by
|
|
75
|
+
other people; [VALIDATION.md](VALIDATION.md) records what that established and
|
|
76
|
+
what it didn't.
|
|
77
|
+
|
|
78
|
+
Two known gaps. Calls aren't followed across files, so a handler that reaches
|
|
79
|
+
another module for its I/O passes clean. And three of the six rules have
|
|
80
|
+
working detectors but no confirmed real-world finding yet, because published
|
|
81
|
+
example code doesn't contain the mistakes they catch.
|
|
82
|
+
|
|
83
|
+
Reports from real codebases are the most useful thing anyone can contribute
|
|
84
|
+
right now, in either direction: a finding it caught, or a false positive it
|
|
85
|
+
shouldn't have raised.
|
|
86
|
+
|
|
87
|
+
## Install
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
pip install replayguard
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
TypeScript, Java, and Rust need a parser:
|
|
94
|
+
|
|
95
|
+
```bash
|
|
96
|
+
pip install 'replayguard[all]'
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
## Use
|
|
100
|
+
|
|
101
|
+
```bash
|
|
102
|
+
replayguard check src/
|
|
103
|
+
replayguard check src/ --explain
|
|
104
|
+
replayguard check src/ --format sarif -o replayguard.sarif
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
Exits non-zero when anything at or above `--fail-on` (default `error`) is
|
|
108
|
+
found, so it can gate a build without extra wiring.
|
|
109
|
+
|
|
110
|
+
```
|
|
111
|
+
tests/fixtures/python/bad_handler.py
|
|
112
|
+
25:17 error RG001 `time.time` runs outside a durable step
|
|
113
|
+
37:14 error RG002 external I/O `requests.get` runs outside a durable step
|
|
114
|
+
53:8 error RG003 step body writes to a captured variable `receipts`
|
|
115
|
+
43:7 error RG004 branch condition depends on `datetime.datetime.now`
|
|
116
|
+
68:4 error RG005 `step` name is built from `time.time`
|
|
117
|
+
71:17 note RG900 could not resolve whether this code runs inside a step
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
## In CI
|
|
121
|
+
|
|
122
|
+
SARIF output means findings render inline on the pull request that introduced
|
|
123
|
+
them:
|
|
124
|
+
|
|
125
|
+
```yaml
|
|
126
|
+
- uses: amrutp24/replayguard@v1
|
|
127
|
+
with:
|
|
128
|
+
path: src/
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
The job needs `permissions: security-events: write` for the annotations. Set
|
|
132
|
+
`fail-on: never` to annotate without blocking the merge.
|
|
133
|
+
|
|
134
|
+
There is also a pre-commit hook:
|
|
135
|
+
|
|
136
|
+
```yaml
|
|
137
|
+
- repo: https://github.com/amrutp24/replayguard
|
|
138
|
+
rev: v0.1.0
|
|
139
|
+
hooks:
|
|
140
|
+
- id: replayguard
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
## Rules
|
|
144
|
+
|
|
145
|
+
| ID | What it catches | Why it breaks replay |
|
|
146
|
+
|----|-----------------|----------------------|
|
|
147
|
+
| **RG001** | Clock, random, or identity source outside a step | Produces a different value on replay; everything derived from it diverges |
|
|
148
|
+
| **RG002** | Network or filesystem access outside a step | Diverges, and repeats the side effect on every replay |
|
|
149
|
+
| **RG003** | A step body writing to state it doesn't own | The write lands on the first run and is skipped on replay, so the outer state silently reverts |
|
|
150
|
+
| **RG004** | Control flow depending on a nondeterministic value | Replay can take the other branch, so the operation sequence no longer matches the journal |
|
|
151
|
+
| **RG005** | A step name built from an unstable source | Checkpoints match by name and order; a changed name can't be matched, so the step re-executes |
|
|
152
|
+
| **RG900** | Code whose region couldn't be resolved | Not a violation. A coverage gap, reported so a clean run means something |
|
|
153
|
+
|
|
154
|
+
`replayguard rules --explain` prints the rationale for each.
|
|
155
|
+
|
|
156
|
+
## Language support
|
|
157
|
+
|
|
158
|
+
Python, TypeScript/JavaScript, and Java are the three runtimes with an
|
|
159
|
+
official AWS durable execution SDK. Rust has no official SDK; the frontend
|
|
160
|
+
targets [pgdad/durable-rust](https://github.com/pgdad/durable-rust), whose own
|
|
161
|
+
documentation says its determinism rules are documented rather than enforced.
|
|
162
|
+
Go and .NET have community proofs of concept only and are out of scope for
|
|
163
|
+
now.
|
|
164
|
+
|
|
165
|
+
| Runtime | Parser |
|
|
166
|
+
|---------|--------|
|
|
167
|
+
| Python | stdlib `ast` |
|
|
168
|
+
| TypeScript / JavaScript | tree-sitter |
|
|
169
|
+
| Java | tree-sitter |
|
|
170
|
+
| Rust | tree-sitter |
|
|
171
|
+
|
|
172
|
+
The SDKs differ in shape, not just syntax:
|
|
173
|
+
|
|
174
|
+
| | Handler | Step |
|
|
175
|
+
|---|---|---|
|
|
176
|
+
| Python | `@durable_execution` | `context.step(fn, name="x")` |
|
|
177
|
+
| JS/TS | `withDurableExecution(fn)` | `context.step("x", fn)` |
|
|
178
|
+
| Java | `extends DurableHandler<,>` | `ctx.step("x", Result.class, fn)` |
|
|
179
|
+
| Rust | param typed `*Context` | `ctx.step("x", \|\| async { .. })` |
|
|
180
|
+
|
|
181
|
+
The step body sits in a different argument position in each, and Java has a
|
|
182
|
+
two-argument overload besides, so bodies are located by kind rather than by
|
|
183
|
+
position. Everything lowers to one shared IR and the rules are written once,
|
|
184
|
+
with no knowledge of which language they're inspecting.
|
|
185
|
+
|
|
186
|
+
The semantic differences that matter are handled per-frontend, and RG003 is
|
|
187
|
+
where they show up:
|
|
188
|
+
|
|
189
|
+
- **Python**: a bare `x = 1` in a nested function creates a local binding, so
|
|
190
|
+
it can never be an outer write. Only mutation and `global`/`nonlocal` reach
|
|
191
|
+
out.
|
|
192
|
+
- **JavaScript**: the same assignment writes straight through to the enclosing
|
|
193
|
+
scope, so RG003 has more ways to fire.
|
|
194
|
+
- **Java**: captured locals must be effectively final, so reassigning one is a
|
|
195
|
+
compile error and that violation class can't exist. What remains is
|
|
196
|
+
collection mutation and field writes.
|
|
197
|
+
- **Rust**: the narrowest of the four. Step closures are `Send + 'static`, so
|
|
198
|
+
capturing a borrowed reference doesn't compile. What's left is interior
|
|
199
|
+
mutability through a shared handle, like an `Arc<Mutex<_>>` locked and
|
|
200
|
+
pushed to, or a `static mut`.
|
|
201
|
+
|
|
202
|
+
## Design
|
|
203
|
+
|
|
204
|
+
```
|
|
205
|
+
source ──▶ frontend ──▶ IR ──▶ rules ──▶ findings ──▶ reporter
|
|
206
|
+
(per-lang) (shared) (shared) text/json/sarif
|
|
207
|
+
```
|
|
208
|
+
|
|
209
|
+
This is AST analysis, not pattern matching, because the questions need scope
|
|
210
|
+
resolution: RG003 has to know whether a mutated name belongs to the step body
|
|
211
|
+
or an enclosing scope, and RG004 has to know whether a branch condition
|
|
212
|
+
derives from a nondeterministic source. Neither can be answered by matching
|
|
213
|
+
source text.
|
|
214
|
+
|
|
215
|
+
False positives get particular attention, since a linter that fires on
|
|
216
|
+
correct code gets uninstalled. RG005 ignores computed step names unless they
|
|
217
|
+
interpolate something genuinely unstable (`` `item-${index}` `` is the pattern
|
|
218
|
+
AWS recommends), and RG900 reports unresolved regions instead of silently
|
|
219
|
+
passing them.
|
|
220
|
+
|
|
221
|
+
## What it doesn't do
|
|
222
|
+
|
|
223
|
+
Calls are not followed across file boundaries. Within a file they are, and
|
|
224
|
+
findings name the route, but a handler that calls into another module for its
|
|
225
|
+
I/O will pass clean. This is the largest known blind spot.
|
|
226
|
+
|
|
227
|
+
Static analysis also can't see nondeterminism inside a third-party library,
|
|
228
|
+
data tainted several hops back, iteration order over an unordered collection,
|
|
229
|
+
or concurrent completion order. Some of that the dynamic half can catch.
|
|
230
|
+
|
|
231
|
+
## Dynamic replay-divergence
|
|
232
|
+
|
|
233
|
+
The static rules reason about what code might do. The dynamic harness runs
|
|
234
|
+
the handler twice, once normally and once with the clock moved and entropy
|
|
235
|
+
reseeded, and diffs the operation journals:
|
|
236
|
+
|
|
237
|
+
```bash
|
|
238
|
+
replayguard replay app.orders:handler --event '{"orderId": "A1"}'
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
```
|
|
242
|
+
replay-divergence: 1 divergence(s) found.
|
|
243
|
+
|
|
244
|
+
operation 0: operation name changed -- checkpoints match by name
|
|
245
|
+
control : step(op-1787442395)
|
|
246
|
+
perturbed : step(op-1787489626)
|
|
247
|
+
```
|
|
248
|
+
|
|
249
|
+
Or as an assertion next to the handler, so a determinism regression fails the
|
|
250
|
+
build instead of surfacing on a resume months later:
|
|
251
|
+
|
|
252
|
+
```python
|
|
253
|
+
from replayguard.dynamic import assert_deterministic
|
|
254
|
+
|
|
255
|
+
def test_handler_is_deterministic():
|
|
256
|
+
assert_deterministic(handler, {"orderId": "A1"})
|
|
257
|
+
```
|
|
258
|
+
|
|
259
|
+
The harness needs no rule for the source of nondeterminism. A clock inside a
|
|
260
|
+
library, an iteration order, a value tainted many hops back: it measures the
|
|
261
|
+
effect, not the cause. On 51 AWS conformance handlers it produced zero false
|
|
262
|
+
alarms; on a handler whose step order comes from `random.sample`, which no
|
|
263
|
+
static rule covers, it diverges.
|
|
264
|
+
|
|
265
|
+
It can't prove determinism, only fail to disprove it, and the report says so.
|
|
266
|
+
Handlers that suspend on a callback can't be checked locally.
|
|
267
|
+
|
|
268
|
+
## Validation
|
|
269
|
+
|
|
270
|
+
[VALIDATION.md](VALIDATION.md) records what has been tested against whose
|
|
271
|
+
code: which rules have confirmed real-world findings, which are still
|
|
272
|
+
unproven, the false positives that were found and fixed, and the bugs the
|
|
273
|
+
validation found in the tool itself. Read it before relying on a clean run.
|
|
274
|
+
|
|
275
|
+
## Developing
|
|
276
|
+
|
|
277
|
+
```bash
|
|
278
|
+
pip install -e ".[dev]"
|
|
279
|
+
python scripts/verify.py
|
|
280
|
+
```
|
|
281
|
+
|
|
282
|
+
`verify.py` runs five gates: import, lint, tests with a coverage floor, the
|
|
283
|
+
CLI's exit codes and output formats, and a canary asserting the known-good
|
|
284
|
+
fixtures produce zero findings in every language. The canary matters most; a
|
|
285
|
+
false positive is a worse failure here than a missed bug.
|
|
286
|
+
|
|
287
|
+
## Prior art
|
|
288
|
+
|
|
289
|
+
[Temporal's workflowcheck](https://github.com/temporalio/sdk-go) does this for
|
|
290
|
+
Temporal workflows, so the category is proven; it just didn't exist for AWS's
|
|
291
|
+
primitive. [durable-viz](https://github.com/gunnargrosch/durable-viz)
|
|
292
|
+
statically analyses durable handlers to draw flowcharts, but performs no
|
|
293
|
+
validation.
|
|
294
|
+
|
|
295
|
+
## License
|
|
296
|
+
|
|
297
|
+
MIT
|