prongs 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,48 @@
1
+ # Publish to PyPI when a GitHub Release is published, through the trusted
2
+ # publisher registered for this repository (workflow publishing.yml,
3
+ # environment pypi). Pushing a tag does nothing by itself: the click on
4
+ # "Publish release" is the gate.
5
+ #
6
+ # git tag v0.1.0 && git push origin v0.1.0
7
+ # gh release create v0.1.0 --draft --generate-notes # then publish it on GitHub
8
+ #
9
+ # The tag must match the version in src/prongs/__init__.py.
10
+
11
+ name: publishing
12
+
13
+ on:
14
+ release:
15
+ types: [published]
16
+
17
+ jobs:
18
+ build:
19
+ runs-on: ubuntu-latest
20
+ steps:
21
+ - uses: actions/checkout@v4
22
+ - uses: actions/setup-python@v5
23
+ with:
24
+ python-version: "3.12"
25
+ - name: Check the tag matches the package version
26
+ run: |
27
+ pkg=$(python -c "import re,pathlib; print(re.search(r'__version__ = \"(.+?)\"', pathlib.Path('src/prongs/__init__.py').read_text()).group(1))")
28
+ tag="${GITHUB_REF_NAME#v}"
29
+ [ "$pkg" = "$tag" ] || { echo "::error::tag v$tag but package version is $pkg"; exit 1; }
30
+ - run: pip install build
31
+ - run: python -m build
32
+ - uses: actions/upload-artifact@v4
33
+ with:
34
+ name: dist
35
+ path: dist/
36
+
37
+ publish:
38
+ needs: build
39
+ runs-on: ubuntu-latest
40
+ environment: pypi
41
+ permissions:
42
+ id-token: write
43
+ steps:
44
+ - uses: actions/download-artifact@v4
45
+ with:
46
+ name: dist
47
+ path: dist/
48
+ - uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,113 @@
1
+ # prongs's own CI, running on prongs.
2
+ #
3
+ # prongs calls pytest in-process and registers its plugin through an entry
4
+ # point, so the checkout under test is also the prongs doing the selecting.
5
+ # That is the point (this is the flow examples/github-actions.yml describes),
6
+ # but it means a pull request that breaks the selector could skip the tests
7
+ # that would catch it. So the merge gate is a plain full run; `prongs run`
8
+ # is the dogfood, and a test that fails in the full run but was not selected
9
+ # is a selection miss that fails the job on its own.
10
+ #
11
+ # push / nightly restore the newest map, fold the journals recent pull
12
+ # request jobs uploaded, run everything recorded, save the map
13
+ # pull request restore the newest map, `prongs run` the affected tests,
14
+ # then the full suite, upload this job's journals
15
+
16
+ name: tests
17
+
18
+ on:
19
+ push:
20
+ branches: [master]
21
+ pull_request:
22
+ schedule:
23
+ - cron: "17 3 * * *"
24
+
25
+ permissions:
26
+ contents: read
27
+ actions: read
28
+
29
+ jobs:
30
+ test:
31
+ runs-on: ubuntu-latest
32
+ strategy:
33
+ fail-fast: false
34
+ matrix:
35
+ python: ["3.12", "3.13"]
36
+ steps:
37
+ - uses: actions/checkout@v4
38
+ with:
39
+ fetch-depth: 0
40
+
41
+ - uses: actions/setup-python@v5
42
+ with:
43
+ python-version: ${{ matrix.python }}
44
+
45
+ - name: Install
46
+ run: pip install -e . pytest
47
+
48
+ - name: Restore the map
49
+ uses: actions/cache/restore@v4
50
+ with:
51
+ path: ${{ runner.temp }}/prongs-ci/map
52
+ key: prongs-map-py${{ matrix.python }}-${{ github.sha }}
53
+ restore-keys: prongs-map-py${{ matrix.python }}-
54
+
55
+ - name: Collect pull-request journals
56
+ if: github.event_name != 'pull_request'
57
+ env:
58
+ GH_TOKEN: ${{ github.token }}
59
+ run: |
60
+ mkdir -p "$RUNNER_TEMP/prongs-ci/journals"
61
+ for id in $(gh run list --workflow "${{ github.workflow }}" --event pull_request \
62
+ --status completed --limit 50 --json databaseId --jq '.[].databaseId'); do
63
+ gh run download "$id" --pattern "prongs-journals-py${{ matrix.python }}-*" \
64
+ --dir "$RUNNER_TEMP/prongs-ci/journals/$id" || true
65
+ done
66
+
67
+ - name: Restore
68
+ run: prongs ci restore "$RUNNER_TEMP/prongs-ci"
69
+
70
+ - name: Affected tests (prongs run)
71
+ id: affected
72
+ if: github.event_name == 'pull_request'
73
+ shell: bash
74
+ run: prongs run | tee "$RUNNER_TEMP/prongs-result.json"
75
+
76
+ - name: Full suite (the gate)
77
+ if: github.event_name == 'pull_request'
78
+ run: python -m pytest
79
+
80
+ - name: Selection miss check
81
+ if: github.event_name == 'pull_request' && always() && steps.affected.outcome != 'skipped'
82
+ run: |
83
+ # `prongs run` passed while the full suite failed: the selector
84
+ # skipped a test the change broke. Fail loudly, not just via pytest.
85
+ if [ "${{ steps.affected.outcome }}" = "success" ] && [ "${{ job.status }}" = "failure" ]; then
86
+ echo "::error::prongs run passed but the full suite failed: selection miss"
87
+ exit 1
88
+ fi
89
+
90
+ - name: Full suite, recorded
91
+ if: github.event_name != 'pull_request'
92
+ run: python -m pytest --prongs-cov
93
+
94
+ - name: Save this job's journals
95
+ if: always() && github.event_name == 'pull_request'
96
+ run: prongs ci save --no-map "$RUNNER_TEMP/prongs-out"
97
+
98
+ - uses: actions/upload-artifact@v4
99
+ if: always() && github.event_name == 'pull_request'
100
+ with:
101
+ name: prongs-journals-py${{ matrix.python }}-${{ github.run_id }}-${{ github.run_attempt }}
102
+ path: ${{ runner.temp }}/prongs-out
103
+ retention-days: 14
104
+
105
+ - name: Save the map
106
+ if: always() && github.event_name != 'pull_request'
107
+ run: prongs ci save --no-journals "$RUNNER_TEMP/prongs-ci/map"
108
+
109
+ - uses: actions/cache/save@v4
110
+ if: always() && github.event_name != 'pull_request'
111
+ with:
112
+ path: ${{ runner.temp }}/prongs-ci/map
113
+ key: prongs-map-py${{ matrix.python }}-${{ github.sha }}-${{ github.run_id }}-${{ github.run_attempt }}
@@ -0,0 +1,7 @@
1
+ testbeds/
2
+ results/
3
+ .venv/
4
+ __pycache__/
5
+ *.egg-info/
6
+ .prongs/
7
+ notes/
prongs-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Farhan Ali Raza
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
prongs-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,293 @@
1
+ Metadata-Version: 2.5
2
+ Name: prongs
3
+ Version: 0.1.0
4
+ Summary: Agent-native test runner for Python: runs only the tests a change can affect
5
+ Project-URL: Source, https://github.com/FarhanAliRaza/prongs
6
+ License-Expression: MIT
7
+ License-File: LICENSE
8
+ Classifier: Framework :: Pytest
9
+ Classifier: Programming Language :: Python :: 3
10
+ Classifier: Programming Language :: Python :: 3.12
11
+ Classifier: Programming Language :: Python :: 3.13
12
+ Classifier: Topic :: Software Development :: Testing
13
+ Requires-Python: >=3.12
14
+ Description-Content-Type: text/markdown
15
+
16
+ # prongs
17
+
18
+ prongs is a test runner for Python, built on pytest, for coding agents and developers iterating on large test suites.
19
+ It records which project functions each test executes.
20
+ After a change, it runs only the tests that change can affect.
21
+ Every skipped test comes with a receipt that says why it was safe to skip.
22
+
23
+ ## Why
24
+
25
+ - **A full suite per edit is too slow.** prongs diffs your change against a recorded map and runs the affected tests only.
26
+ - **Agents re-run everything because they cannot tell what is safe to skip.** Every `prongs run` prints one JSON document with a skip receipt: the rule, the evidence, and how old it is.
27
+ - **Selection tools are hard to trust.** Every run checks that collected tests equal selected plus run-all plus skipped. If the books do not balance, the run fails with exit code 3.
28
+ - **Trust needs measuring.** `prongs audit` runs the full suite in shadow mode and reports every status change that selection would have skipped.
29
+
30
+ ## Installation
31
+
32
+ ```console
33
+ pip install prongs
34
+ ```
35
+
36
+ prongs requires Python 3.12 or later, since it records with `sys.monitoring`.
37
+ It has no dependencies beyond pytest in your environment.
38
+ It registers itself as a pytest plugin through the `pytest11` entry point.
39
+ The recorder stays off unless you pass `--prongs-cov`, so plain pytest runs are unaffected.
40
+
41
+ ## Quick start
42
+
43
+ Run from the repository root, the directory you run pytest from.
44
+
45
+ ```console
46
+ python -m pytest --prongs-cov # one full run with the recorder on
47
+ prongs affected # what a change can affect, and why
48
+ prongs run # run those tests and report as JSON
49
+ ```
50
+
51
+ The first command writes a journal file under `.prongs/`.
52
+ `affected` and `run` fold it into the map before they select.
53
+ Add `.prongs/` to your `.gitignore`.
54
+
55
+ After editing one function, `prongs affected` prints (trimmed):
56
+
57
+ ```json
58
+ {
59
+ "mode": "select",
60
+ "n_total": 98,
61
+ "n_selected": 3,
62
+ "n_skipped": 95,
63
+ "changed_functions": [
64
+ "src/prongs/__main__.py::exception_line",
65
+ "src/prongs/config.py::settings"
66
+ ],
67
+ "run_all_reasons": [],
68
+ "selected_by_reason": {
69
+ "touches changed function src/prongs/config.py": 1,
70
+ "touches changed function src/prongs/__main__.py": 2
71
+ },
72
+ "tests": [
73
+ "tests/test_audit.py::test_a_control_run_separates_environment_drift_from_misses",
74
+ "tests/test_run.py::test_exception_line_skips_pytest_hints",
75
+ "tests/test_run.py::test_group_failures_keeps_one_traceback_per_root_cause"
76
+ ],
77
+ "history": {"window": 3, "recent_changes": 0, "flaky": 0, "flaky_window": 50},
78
+ "evidence": {"runs": 2, "last_full_run": {"id": 1, "scope": "full", "commit": "41e7c5c8..."}, "...": "..."},
79
+ "skip_receipt": {
80
+ "skipped": 95,
81
+ "rule": "a test is skipped only when, relative to the tree of the run that last observed it, no changed function or file is in its recorded dependency set; ...",
82
+ "map_age": {"commit": "41e7c5c8", "commits_behind_head": 1, "max_map_age": 50},
83
+ "unmapped_tests": 0,
84
+ "warnings": []
85
+ }
86
+ }
87
+ ```
88
+
89
+ With no map, `prongs run` runs the whole suite and records it, so it builds the map it lacked.
90
+
91
+ ## How it works
92
+
93
+ **Map.** The pytest plugin uses `sys.monitoring` to record the project functions each test enters.
94
+ Code that runs at import or collection time is recorded as its own set.
95
+ Stdlib and site-packages code is disabled on first sight and costs nothing after that.
96
+
97
+ **Journal and roll-up.** Every recorded session appends one journal file and never touches the map.
98
+ `prongs rollup` folds pending journal files into `.prongs/map.sqlite`.
99
+ `affected`, `run` and `audit` roll up first, unless you pass `--no-rollup`.
100
+ Partial runs (node ids, `-k`, `-m`, xdist workers) refresh exactly the tests they observed.
101
+ Concurrent sessions each write their own file.
102
+
103
+ **Provenance.** Each run stores HEAD and a content hash of every file that differed from HEAD, at start and finish.
104
+ The selector diffs against the tree the evidence actually saw, not the tree you assume.
105
+ A dirty file that still matches what the run saw is not a change.
106
+
107
+ **Selection.** Changed lines map to functions via AST spans, then to tests through the inverted map.
108
+ Each selected test carries a reason:
109
+
110
+ | Reason | When |
111
+ |---|---|
112
+ | `touches changed function` | the test entered a function whose lines changed |
113
+ | `imports/touches changed file` | the file was added, deleted, changed at module level, or changed mid-run, so it is taken whole |
114
+ | `changed function not in the map` | a new function in a known file: every test that touches the file runs |
115
+ | `test file changed` / `new test file` | the test module itself changed |
116
+ | `recent status change` | the outcome flipped within the last `history_window` rollups |
117
+ | `unmapped test (conservative)` | the map holds no dependency evidence for it |
118
+ | `evidence commit ... is not in this repository` | there is nothing to diff its evidence against |
119
+
120
+ Some changes run everything (`mode: "run_all"`).
121
+ These include a changed `conftest.py`, a changed tracked non-Python file, and module-level or import-time code in a production module.
122
+ A new file that did not exist when the map was recorded also runs everything.
123
+ As a safety net, a map more than `max_map_age` commits behind HEAD runs everything.
124
+
125
+ **Flaky tests.** A test that both passed and failed on one tree within `flaky_window` rollups is flaky.
126
+ Flaky tests still run.
127
+ A failing flaky test is re-run alone up to `flaky_retries` times.
128
+ A pass reports it under `flaky`; failing every time counts it as a failure.
129
+
130
+ **Books that balance.** Every `run` result carries a `conservation` block.
131
+ It checks `collected == selected + run_all + skipped`, and that every planned test produced a result.
132
+ Any mismatch makes the status `inconsistent` and the exit code 3.
133
+ A run that silently did less than asked is never reported as `passed`.
134
+
135
+ ## Commands
136
+
137
+ All commands print one JSON document on stdout, except `daemon`.
138
+
139
+ | Command | Purpose |
140
+ |---|---|
141
+ | `prongs affected` | Show what would run, and why |
142
+ | `prongs run` | Select, execute, and report results with receipts |
143
+ | `prongs audit` | Select, then run the full suite and report every skipped status change |
144
+ | `prongs rollup` | Fold pending journal files into the map |
145
+ | `prongs ci save DIR` | Write the map and this job's journal files as a CI artifact |
146
+ | `prongs ci restore PATH...` | Install CI artifacts into a fresh clone |
147
+ | `prongs daemon` | Start the warm daemon in the foreground |
148
+ | `prongs stop` | Stop the daemon |
149
+
150
+ `affected`, `run` and `audit` share these flags:
151
+
152
+ | Flag | Default | Meaning |
153
+ |---|---|---|
154
+ | `--base REV` | evidence commits | Diff against `REV` instead of the trees the evidence was observed on |
155
+ | `--rollup` / `--no-rollup` | on | Fold pending journal files into the map first |
156
+ | `--history-window N` | 3 | Select tests whose outcome flipped in the last N rollups |
157
+ | `--max-map-age N` | 50 | Run everything when the map is more than N commits behind HEAD |
158
+ | `--flaky-window N` | 50 | Flaky evidence older than N rollups expires |
159
+
160
+ `run` adds:
161
+
162
+ | Flag | Default | Meaning |
163
+ |---|---|---|
164
+ | `--record` / `--no-record` | on | Append this run to the journal |
165
+ | `--cov` / `--no-cov` | on | Record per-test coverage; `--no-cov` still appends outcomes and durations |
166
+ | `--flaky-retries N` | 2 | Re-run a failing flaky test alone up to N times; 0 makes a flaky failure a failure |
167
+
168
+ `audit` adds:
169
+
170
+ | Flag | Default | Meaning |
171
+ |---|---|---|
172
+ | `--pytest-args ARGS` | `$PRONGS_RUN_ARGS` | Extra pytest arguments for the full run |
173
+ | `--isolate` / `--no-isolate` | on | Re-run each miss alone to separate first-order misses from pollution |
174
+ | `--record` / `--no-record` | off | Append the full run to the real journal |
175
+ | `--control` / `--no-control` | off | Re-run each first-order miss on its baseline commit, checked out in place (clean tree only) |
176
+
177
+ `ci save` takes `--map` / `--no-map` and `--journals` / `--no-journals`, both on by default.
178
+ `ci restore` takes `--fetch` / `--no-fetch` (on) to deepen a shallow clone, and `--remote NAME` (default `origin`).
179
+
180
+ The pytest plugin adds:
181
+
182
+ | Option | Meaning |
183
+ |---|---|
184
+ | `--prongs-cov` | Record per-test coverage and outcomes as a journal file |
185
+ | `--prongs-journal DIR` | Journal directory for `--prongs-cov` |
186
+ | `--prongs-journal-key KEY` | Name of this run's journal file |
187
+
188
+ `run` uses the daemon when its socket exists, and otherwise runs pytest in-process.
189
+ If the warm image is stale, it falls back to a cold run and says so under `executor`.
190
+
191
+ ### Exit codes
192
+
193
+ | Code | `run` | `audit` | Other commands |
194
+ |---|---|---|---|
195
+ | 0 | passed, or nothing to run | no miss | done |
196
+ | 1 | tests failed | a first-order miss | |
197
+ | 2 | error: pytest exit 2-5, a collection error, a crash, an unknown revision | error, or no map to audit | error |
198
+ | 3 | inconsistent: the books do not balance | | |
199
+
200
+ ## Configuration
201
+
202
+ Settings are read from `[tool.prongs]` in `pyproject.toml`, then `PRONGS_<NAME>` environment variables, then command-line flags.
203
+
204
+ ```toml
205
+ [tool.prongs]
206
+ history_window = 3 # a test whose outcome flipped in the last N rollups is selected
207
+ max_map_age = 50 # a map more than N commits behind HEAD runs everything
208
+ flaky_window = 50 # flaky evidence older than N rollups expires (0: nothing is flaky)
209
+ flaky_retries = 2 # re-run a failing flaky test alone up to N times
210
+ ```
211
+
212
+ | Variable | Meaning |
213
+ |---|---|
214
+ | `PRONGS_HISTORY_WINDOW`, `PRONGS_MAX_MAP_AGE`, `PRONGS_FLAKY_WINDOW`, `PRONGS_FLAKY_RETRIES` | Override the settings above |
215
+ | `PRONGS_COV=1` | Turn the recorder on, like `--prongs-cov` |
216
+ | `PRONGS_DIR` | State directory, relative to the repository root (default `.prongs`); point worktrees at one directory to pool runs |
217
+ | `PRONGS_JOURNAL` | Journal directory only (default `<state dir>/journal`) |
218
+ | `PRONGS_JOURNAL_KEY` | Name of the recorded run's journal file |
219
+ | `PRONGS_RUN_ARGS` | Extra pytest arguments for every pytest run prongs starts: `run`, `audit`, flaky retries, and daemon children |
220
+ | `PRONGS_WARM_ARGS` | Extra pytest arguments for the daemon's warm-up collection |
221
+ | `PRONGS_WARM_DB_TEST` | A test node id the daemon runs once at warm-up, so forked children inherit the test database |
222
+
223
+ ## CI
224
+
225
+ A CI job starts from a fresh clone, so the map travels as an artifact.
226
+ Pushes to the default branch run the full suite with the recorder and save the map to a cache.
227
+ Pull requests restore that map, run `prongs run` as the gate, and upload their journal files.
228
+ The next push job folds those journals in as history, which feeds flaky detection.
229
+ `ci restore` deepens a shallow clone until the map's evidence commits are present.
230
+ With no map, `prongs run` runs everything and records it, so the gate never depends on the cache.
231
+ See [`examples/github-actions.yml`](examples/github-actions.yml) for the full, commented workflow.
232
+
233
+ ```yaml
234
+ - name: Restore
235
+ run: python -m prongs ci restore "$RUNNER_TEMP/prongs-ci"
236
+
237
+ - name: Tests (the affected ones)
238
+ if: github.event_name == 'pull_request'
239
+ shell: bash
240
+ run: python -m prongs run | tee "$RUNNER_TEMP/prongs-result.json"
241
+
242
+ - name: Tests (all of them, recorded)
243
+ if: github.event_name != 'pull_request'
244
+ run: python -m pytest --prongs-cov
245
+
246
+ - name: Save this job's journals
247
+ if: always() && github.event_name == 'pull_request'
248
+ run: python -m prongs ci save --no-map "$RUNNER_TEMP/prongs-out"
249
+
250
+ - name: Save the map
251
+ if: always() && github.event_name != 'pull_request'
252
+ run: python -m prongs ci save --no-journals "$RUNNER_TEMP/prongs-ci/map"
253
+ ```
254
+
255
+ ## Audit mode
256
+
257
+ `prongs audit` selects exactly as `prongs run` would, then runs the full suite with the recorder on.
258
+ The full run goes to a temporary journal, so an audit never feeds the map it audits.
259
+ It compares every test's status with the map's baseline.
260
+ A test whose status changed and was not selected is a candidate miss, reported with the receipt that skipped it.
261
+
262
+ Each candidate is re-run alone on the same tree:
263
+
264
+ - **First-order miss:** it still differs from its baseline alone. This is the selector's fault.
265
+ - **Second-order pollution:** it returns to its baseline alone. An already-selected failure caused it, not selection.
266
+
267
+ A known-flaky test that flips is reported under `flaky`, not as a miss.
268
+ With `--control`, a miss that also differs on its baseline commit is reported as `drift`.
269
+
270
+ ```console
271
+ prongs audit --base HEAD~1
272
+ ```
273
+
274
+ The audit exits 1 on any first-order miss.
275
+ Run it in CI alongside selection, and treat a first-order miss as the kill criterion for trusting selection on your suite.
276
+
277
+ ## Using it from an agent
278
+
279
+ `prongs run` is designed as the single command an agent calls after an edit.
280
+ It rolls up, selects, executes, and reports in one JSON document.
281
+ The output is built for a token budget: failures are grouped by exception line, tracebacks are truncated, passes are counted, not listed.
282
+ The skip receipt and `evidence` let the agent decide whether to trust the skip.
283
+ The exit code carries the verdict, so no output parsing is needed to gate on it.
284
+
285
+ ## Status
286
+
287
+ prongs is early (0.0.x).
288
+ The command-line interface, JSON output shape and map schema may change between releases.
289
+ The `bench/` directory holds the harnesses used to validate selection: a baseline timer, mutation testing and history replay around `audit`, and a local replay of the CI artifact flow.
290
+
291
+ ## License
292
+
293
+ MIT. See [LICENSE](LICENSE).