redherring 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. redherring-0.1.0/.gitignore +10 -0
  2. redherring-0.1.0/CHANGELOG.md +20 -0
  3. redherring-0.1.0/LICENSE +21 -0
  4. redherring-0.1.0/PKG-INFO +220 -0
  5. redherring-0.1.0/README.md +195 -0
  6. redherring-0.1.0/pyproject.toml +59 -0
  7. redherring-0.1.0/src/redherring/__init__.py +3 -0
  8. redherring-0.1.0/src/redherring/__main__.py +5 -0
  9. redherring-0.1.0/src/redherring/cache.py +106 -0
  10. redherring-0.1.0/src/redherring/cli.py +229 -0
  11. redherring-0.1.0/src/redherring/comment.py +39 -0
  12. redherring-0.1.0/src/redherring/github.py +320 -0
  13. redherring-0.1.0/src/redherring/infra.py +199 -0
  14. redherring-0.1.0/src/redherring/ledger.py +124 -0
  15. redherring-0.1.0/src/redherring/logtext.py +36 -0
  16. redherring-0.1.0/src/redherring/parsers.py +517 -0
  17. redherring-0.1.0/src/redherring/render.py +408 -0
  18. redherring-0.1.0/src/redherring/scan.py +455 -0
  19. redherring-0.1.0/src/redherring/why.py +292 -0
  20. redherring-0.1.0/tests/__init__.py +0 -0
  21. redherring-0.1.0/tests/fakegh.py +201 -0
  22. redherring-0.1.0/tests/fixtures/logs/cargo_tokio.log +18 -0
  23. redherring-0.1.0/tests/fixtures/logs/compose_airflow.log +13 -0
  24. redherring-0.1.0/tests/fixtures/logs/docker_apk_pulsar.log +10 -0
  25. redherring-0.1.0/tests/fixtures/logs/gate_langchain.log +5 -0
  26. redherring-0.1.0/tests/fixtures/logs/go_grafana.log +21 -0
  27. redherring-0.1.0/tests/fixtures/logs/go_timeout_prometheus.log +24 -0
  28. redherring-0.1.0/tests/fixtures/logs/jest_nextjs.log +46 -0
  29. redherring-0.1.0/tests/fixtures/logs/maven_keycloak.log +7 -0
  30. redherring-0.1.0/tests/fixtures/logs/maven_registry_keycloak.log +21 -0
  31. redherring-0.1.0/tests/fixtures/logs/nextest_uv.log +24 -0
  32. redherring-0.1.0/tests/fixtures/logs/pytest_airflow.log +16 -0
  33. redherring-0.1.0/tests/fixtures/logs/rspec_discourse.log +12 -0
  34. redherring-0.1.0/tests/fixtures/logs/ssl_pydantic.log +14 -0
  35. redherring-0.1.0/tests/fixtures/logs/tap_testem_discourse.log +10 -0
  36. redherring-0.1.0/tests/fixtures/logs/vitest_vite.log +24 -0
  37. redherring-0.1.0/tests/test_comment.py +52 -0
  38. redherring-0.1.0/tests/test_ledger.py +115 -0
  39. redherring-0.1.0/tests/test_parsers.py +677 -0
  40. redherring-0.1.0/tests/test_scan_why.py +316 -0
@@ -0,0 +1,10 @@
1
+ .venv/
2
+ __pycache__/
3
+ *.pyc
4
+ .pytest_cache/
5
+ .ruff_cache/
6
+ dist/
7
+ build/
8
+ *.egg-info/
9
+ *.sqlite
10
+ *.sqlite-*
@@ -0,0 +1,20 @@
1
+ # Changelog
2
+
3
+ ## 0.1.0 (2026-09-28)
4
+
5
+ First version.
6
+
7
+ - `redherring scan`: finds runs that went green only after a re-run of the same commit, reads the
8
+ failed attempts' logs, and reports flaky tests, flaky jobs by cause, runner time lost and time spent red.
9
+ - `redherring why`: explains one failed run. Each failure is a known flake, infrastructure, already
10
+ failing on the default branch, a policy check, or looks real. Exit codes for scripts and agents.
11
+ - Parsers for pytest, unittest, Jest, Vitest, Playwright, Mocha, node:test, Bun, TAP, go test,
12
+ gotestsum, cargo test, cargo-nextest, "failed tests:" lists, RSpec, Minitest, Maven Surefire, Gradle,
13
+ PHPUnit, dotnet test, CTest, XCTest and ExUnit.
14
+ - Infrastructure causes: network, package registry, rate limited, runner lost, runner environment,
15
+ job timeout, out of memory, disk full, service startup, test worker crash, GitHub service, AI provider.
16
+ - Checked by two independent verdict audits on 113 real failed jobs (see the project report).
17
+ - Text, Markdown and JSON output; a composite GitHub Action; a local cache of immutable history.
18
+ - `--ledger`: a JSON file of past evidence that outlives GitHub's run retention (from 1 Oct 2026 runs
19
+ are deleted after the log-retention period).
20
+ - `why --comment` / Action `comment: true`: one PR comment per workflow, edited in place.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 ixklo
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,220 @@
1
+ Metadata-Version: 2.5
2
+ Name: redherring
3
+ Version: 0.1.0
4
+ Summary: Find the CI failures that weren't your fault: flaky tests and infrastructure hiccups, mined from the GitHub Actions history you already have.
5
+ Project-URL: Homepage, https://github.com/ixklo/redherring
6
+ Project-URL: Issues, https://github.com/ixklo/redherring/issues
7
+ Author: ixklo
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: ci,coding-agents,devtools,flaky-tests,github-actions,testing
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Environment :: Console
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Classifier: Programming Language :: Python :: 3.13
19
+ Classifier: Topic :: Software Development :: Quality Assurance
20
+ Classifier: Topic :: Software Development :: Testing
21
+ Requires-Python: >=3.11
22
+ Requires-Dist: httpx>=0.27
23
+ Requires-Dist: rich>=13.7
24
+ Description-Content-Type: text/markdown
25
+
26
+ # redherring
27
+
28
+ **Find the CI failures that weren't your fault.**
29
+
30
+ redherring reads the GitHub Actions history your repo already has and finds every run that went red, then turned green when someone re-ran the *same commit*. Nothing in the code changed between those two attempts, so whatever failed first was not caused by the commit: a flaky test, a network blip, a package registry hiccup, a runner that vanished. redherring opens the failed logs, names the tests (over 20 test-runner output formats) or the infrastructure cause, and ranks them.
31
+
32
+ Then, when a build fails, `redherring why` tells you (or your coding agent) which failures are known red herrings and which ones look real.
33
+
34
+ No server, no signup, nothing to install in CI first. If your repo has Actions history, the first run already has answers.
35
+
36
+ ![redherring scan on astral-sh/uv: 2,129 runs in 14 days, 280 went red, 94 of those (34%) turned green on a plain re-run; a table of named flaky tests, several Windows-only; flaky jobs grouped by cause](docs/demo-scan.svg)
37
+
38
+ *Real output for a public repo, 28 September 2026. The "AI provider" cause is an AI code-review job failing with "Selected model is at capacity".*
39
+
40
+ > **How common is this?** We ran it on 122 popular repos (1.25 million workflow runs, two weeks). For the median repo about 2% of red runs are red herrings; for a quarter of them it's 7% or more (uv 34%, VS Code 23%, Next.js 15%), and 690 runner-hours went to failed jobs that a re-run fixed. **[Read the study →](docs/study/README.md)**
41
+
42
+ ## Install
43
+
44
+ redherring is a Python 3.11+ command-line tool. With [uv](https://docs.astral.sh/uv/):
45
+
46
+ ```sh
47
+ uvx --from git+https://github.com/ixklo/redherring redherring scan OWNER/REPO
48
+ ```
49
+
50
+ or `pipx install git+https://github.com/ixklo/redherring`.
51
+
52
+ It needs a GitHub token to read Actions logs. If you use the [GitHub CLI](https://cli.github.com/), it borrows `gh auth token` automatically; otherwise set `GH_TOKEN`. Any token works for public repositories; private ones need `actions: read`.
53
+
54
+ ## `redherring scan`: what flakes in this repo?
55
+
56
+ ```sh
57
+ redherring scan # the repo of the current checkout
58
+ redherring scan vitejs/vite --days 14
59
+ redherring scan owner/repo --workflow ci.yml --format md > FLAKY.md
60
+ ```
61
+
62
+ You get:
63
+
64
+ - **Flaky tests**: each failed and then passed on a re-run of the same commit, with how often, on how many commits, on which OS (plenty of tests only flake on Windows), and when last seen.
65
+ - **Flaky jobs with no test named**, grouped by cause: `network`, `package registry`, `rate limited`, `runner lost`, `runner environment`, `job timeout`, `out of memory`, `disk full`, `service startup`, `test worker crash`, `GitHub service`, `AI provider`, `many tests at once` (an environment problem, not ten flaky tests), or `unknown`.
66
+ - **What it cost**: runner time burned by the failed jobs, and how long commits sat red waiting for someone to press re-run.
67
+
68
+ `--format json` gives everything, including example job links, for your own dashboards.
69
+
70
+ ## `redherring why`: is this failure mine?
71
+
72
+ ```sh
73
+ redherring why https://github.com/owner/repo/actions/runs/123456789
74
+ redherring why 123456789 --repo owner/repo --format json
75
+ ```
76
+
77
+ ![redherring why on a failed uv PR run: one test is a known flake (failed then passed on re-runs twice in 14 days), the other looks real](docs/demo-why.svg)
78
+
79
+ For every failure in the run, one verdict:
80
+
81
+ | verdict | meaning |
82
+ |---|---|
83
+ | **RED HERRING** (known flaky) | This test failed and then passed on a re-run of the same commit in the last N days. |
84
+ | **RED HERRING** (infrastructure) | The log shows a network, registry, runner, or service problem, not a test failure. |
85
+ | **NOT THIS CHANGE** | The latest finished run on the default branch fails this same test. |
86
+ | **LOOKS REAL** | Never seen flaking, no infrastructure cause. Probably your change. |
87
+ | **UNCLEAR** / **POLICY CHECK** | Can't tell from history, or a PR policy check (labels, linked issue) that passes once the PR is updated. |
88
+
89
+ Exit codes make it scriptable: `0` all red herrings (or nothing failed), `1` something looks real, `3` unclear, `2` error.
90
+
91
+ ```sh
92
+ redherring why "$RUN" -q && gh run rerun "$RUN" --failed
93
+ ```
94
+
95
+ ### For coding agents
96
+
97
+ Agents lose a lot of time "fixing" failures that were never theirs, and sometimes weaken a perfectly good test to make a flake go green. Add this to your `AGENTS.md` or `CLAUDE.md`:
98
+
99
+ ```markdown
100
+ ## CI failures
101
+ Before changing code because CI failed, run `redherring why <run-url> --format json`.
102
+ - `recommendation: "rerun"`: every failure is a known flake, an infrastructure problem, or already
103
+ failing on main. Re-run the failed jobs (`gh run rerun <id> --failed`); don't edit code or tests for it.
104
+ - `recommendation: "investigate"`: fix only the findings with `"red_herring": false`.
105
+ - Never skip, loosen, or delete a test just because it is listed as flaky.
106
+ ```
107
+
108
+ Or install the ready-made skill: copy [`skills/redherring/`](skills/redherring/SKILL.md) into your agent's skills folder (for Claude Code, `.claude/skills/redherring/`). More setups are in [`examples/`](examples/).
109
+
110
+ ## GitHub Action
111
+
112
+ Explain every failed CI run, in its job summary and as a comment on the pull request:
113
+
114
+ ```yaml
115
+ # .github/workflows/redherring.yml
116
+ name: redherring
117
+ on:
118
+ workflow_run:
119
+ workflows: [CI] # the name of your test workflow
120
+ types: [completed]
121
+ permissions:
122
+ actions: read
123
+ contents: read
124
+ pull-requests: write # only needed for comment: true
125
+ jobs:
126
+ why:
127
+ if: github.event.workflow_run.conclusion == 'failure'
128
+ runs-on: ubuntu-latest
129
+ steps:
130
+ - uses: ixklo/redherring@v0.1.0
131
+ with:
132
+ comment: true
133
+ ```
134
+
135
+ The comment is posted once per PR and workflow, then edited in place on later failures, never duplicated.
136
+
137
+ Or post a weekly flaky-test report:
138
+
139
+ ```yaml
140
+ on:
141
+ schedule: [{ cron: "0 7 * * 1" }]
142
+ permissions:
143
+ actions: read
144
+ jobs:
145
+ scan:
146
+ runs-on: ubuntu-latest
147
+ steps:
148
+ - uses: ixklo/redherring@v0.1.0
149
+ with:
150
+ command: scan
151
+ days: "30"
152
+ ```
153
+
154
+ Inputs: `command` (`why` or `scan`), `run-id`, `days` (default 14), `workflow`, `ledger`, `comment`, `cache` (default on: keeps downloaded history between runs, because the Actions token allows about 1,000 API requests an hour), `fail-on-real`, `github-token`. Output: `recommendation`. From the command line, the same comment is `redherring why <run> --comment`.
155
+
156
+ ## Keep your flake history: the ledger
157
+
158
+ From 1 October 2026 GitHub deletes workflow runs once they pass your log-retention period (90 days by default). Past that, redherring has nothing to read. A ledger keeps the evidence:
159
+
160
+ ```sh
161
+ redherring scan --ledger .github/flake-ledger.json # merges new evidence into the file
162
+ redherring why "$RUN" --ledger .github/flake-ledger.json # uses it as extra history
163
+ ```
164
+
165
+ The file is small JSON (one entry per failed job: test names, cause, commit, OS, link). Keep it anywhere durable. You can commit it, or carry it between scheduled runs with `actions/cache`:
166
+
167
+ ```yaml
168
+ steps:
169
+ - uses: actions/cache@v6
170
+ with:
171
+ path: flake-ledger.json
172
+ key: redherring-ledger-${{ github.run_id }}
173
+ restore-keys: redherring-ledger-
174
+ - uses: ixklo/redherring@v0.1.0
175
+ with:
176
+ command: scan
177
+ ledger: flake-ledger.json
178
+ ```
179
+
180
+ Entries older than a year are dropped (`--keep-days`).
181
+
182
+ ## How it decides
183
+
184
+ redherring is deliberately strict about what it calls flaky. **Only one kind of evidence counts: the same commit failed and then passed.** That is what a re-run is. It does not guess from "failed on a PR, passed on main" or "this test fails a lot", because those are often real failures. A tool that cries wolf is worse than none.
185
+
186
+ Then, per failed job:
187
+
188
+ 1. **Tests.** The log is cleaned (timestamps, colour codes) and read by runner-specific parsers. Test ids keep only stable parts (file, suite, name; never line numbers or timings) so the same test lines up across commits and OSes.
189
+ 2. **Infrastructure.** If no test is named, the end of the log is matched against specific signatures (`ECONNRESET`, `Could not transfer artifact`, `The runner has received a shutdown signal`, `model is at capacity`, ...).
190
+ 3. **Not flakiness at all.** Roll-up jobs ("all required jobs passed") and PR policy checks are set aside and not counted.
191
+
192
+ Supported test output:
193
+
194
+ | ecosystem | runners |
195
+ |---|---|
196
+ | Python | pytest (incl. xdist, pytest-rerunfailures), unittest |
197
+ | JavaScript / TypeScript | Jest, Vitest (incl. workspaces), Playwright (incl. its own retries), Mocha, node:test (spec and TAP), Bun, TAP (QUnit/testem) |
198
+ | Go | `go test`, gotestsum, test timeouts |
199
+ | Rust | `cargo test`, cargo-nextest (incl. retries), `failed tests:` lists (Deno and other file-based runners) |
200
+ | Ruby | RSpec, Minitest |
201
+ | JVM | Maven Surefire, Gradle (incl. verbose test logging) |
202
+ | others | PHPUnit, .NET (`dotnet test`), CTest, XCTest, ExUnit |
203
+
204
+ Missing yours? A parser is one function and a test with a real log excerpt. See [CONTRIBUTING.md](CONTRIBUTING.md).
205
+
206
+ ## Limits, honestly
207
+
208
+ - **History has an expiry date.** Job logs are kept for 90 days by default, and [from 1 October 2026](https://github.blog/changelog/2026-08-27-actions-retention-will-cover-checks-workflow-runs-and-statuses/) GitHub deletes the workflow runs themselves once they pass the repo's log retention period (90 days by default, and at most 90 days for public repos). `--days` beyond that finds nothing; use a [ledger](#keep-your-flake-history-the-ledger) to keep what was found.
209
+ - **Only re-runs count.** Repos that never press re-run, or retry inside the job without reporting it, show fewer flakes than they have. Playwright, nextest and pytest-rerunfailures retries are recognised when they appear in a failed job's log.
210
+ - **Big repos cost API requests.** A busy monorepo can take a few thousand requests for 30 days. redherring waits out rate limits on its own, and caches everything immutable (attempt job lists, logs) so later runs only fetch what's new.
211
+ - **Some jobs stay "unknown".** In a study of 122 popular repos, nearly half of the red-herring jobs (46%) named neither a test nor a known infrastructure cause. Some logs genuinely say nothing ("exit 1"); others use output formats redherring doesn't parse yet. `why` treats those as "unclear", never as safe to re-run. Parser contributions fix this one format at a time.
212
+ - **GitHub Actions only**, for now.
213
+
214
+ ## Privacy
215
+
216
+ redherring runs on your machine (or in your own Actions runner) and talks only to the GitHub API. Downloaded logs are cached, compressed, in your user cache folder (`%LOCALAPPDATA%\redherring`, `~/Library/Caches/redherring`, or `~/.cache/redherring`; override with `REDHERRING_CACHE`). Nothing is sent anywhere else.
217
+
218
+ ## License
219
+
220
+ MIT
@@ -0,0 +1,195 @@
1
+ # redherring
2
+
3
+ **Find the CI failures that weren't your fault.**
4
+
5
+ redherring reads the GitHub Actions history your repo already has and finds every run that went red, then turned green when someone re-ran the *same commit*. Nothing in the code changed between those two attempts, so whatever failed first was not caused by the commit: a flaky test, a network blip, a package registry hiccup, a runner that vanished. redherring opens the failed logs, names the tests (over 20 test-runner output formats) or the infrastructure cause, and ranks them.
6
+
7
+ Then, when a build fails, `redherring why` tells you (or your coding agent) which failures are known red herrings and which ones look real.
8
+
9
+ No server, no signup, nothing to install in CI first. If your repo has Actions history, the first run already has answers.
10
+
11
+ ![redherring scan on astral-sh/uv: 2,129 runs in 14 days, 280 went red, 94 of those (34%) turned green on a plain re-run; a table of named flaky tests, several Windows-only; flaky jobs grouped by cause](docs/demo-scan.svg)
12
+
13
+ *Real output for a public repo, 28 September 2026. The "AI provider" cause is an AI code-review job failing with "Selected model is at capacity".*
14
+
15
+ > **How common is this?** We ran it on 122 popular repos (1.25 million workflow runs, two weeks). For the median repo about 2% of red runs are red herrings; for a quarter of them it's 7% or more (uv 34%, VS Code 23%, Next.js 15%), and 690 runner-hours went to failed jobs that a re-run fixed. **[Read the study →](docs/study/README.md)**
16
+
17
+ ## Install
18
+
19
+ redherring is a Python 3.11+ command-line tool. With [uv](https://docs.astral.sh/uv/):
20
+
21
+ ```sh
22
+ uvx --from git+https://github.com/ixklo/redherring redherring scan OWNER/REPO
23
+ ```
24
+
25
+ or `pipx install git+https://github.com/ixklo/redherring`.
26
+
27
+ It needs a GitHub token to read Actions logs. If you use the [GitHub CLI](https://cli.github.com/), it borrows `gh auth token` automatically; otherwise set `GH_TOKEN`. Any token works for public repositories; private ones need `actions: read`.
28
+
29
+ ## `redherring scan`: what flakes in this repo?
30
+
31
+ ```sh
32
+ redherring scan # the repo of the current checkout
33
+ redherring scan vitejs/vite --days 14
34
+ redherring scan owner/repo --workflow ci.yml --format md > FLAKY.md
35
+ ```
36
+
37
+ You get:
38
+
39
+ - **Flaky tests**: each failed and then passed on a re-run of the same commit, with how often, on how many commits, on which OS (plenty of tests only flake on Windows), and when last seen.
40
+ - **Flaky jobs with no test named**, grouped by cause: `network`, `package registry`, `rate limited`, `runner lost`, `runner environment`, `job timeout`, `out of memory`, `disk full`, `service startup`, `test worker crash`, `GitHub service`, `AI provider`, `many tests at once` (an environment problem, not ten flaky tests), or `unknown`.
41
+ - **What it cost**: runner time burned by the failed jobs, and how long commits sat red waiting for someone to press re-run.
42
+
43
+ `--format json` gives everything, including example job links, for your own dashboards.
44
+
45
+ ## `redherring why`: is this failure mine?
46
+
47
+ ```sh
48
+ redherring why https://github.com/owner/repo/actions/runs/123456789
49
+ redherring why 123456789 --repo owner/repo --format json
50
+ ```
51
+
52
+ ![redherring why on a failed uv PR run: one test is a known flake (failed then passed on re-runs twice in 14 days), the other looks real](docs/demo-why.svg)
53
+
54
+ For every failure in the run, one verdict:
55
+
56
+ | verdict | meaning |
57
+ |---|---|
58
+ | **RED HERRING** (known flaky) | This test failed and then passed on a re-run of the same commit in the last N days. |
59
+ | **RED HERRING** (infrastructure) | The log shows a network, registry, runner, or service problem, not a test failure. |
60
+ | **NOT THIS CHANGE** | The latest finished run on the default branch fails this same test. |
61
+ | **LOOKS REAL** | Never seen flaking, no infrastructure cause. Probably your change. |
62
+ | **UNCLEAR** / **POLICY CHECK** | Can't tell from history, or a PR policy check (labels, linked issue) that passes once the PR is updated. |
63
+
64
+ Exit codes make it scriptable: `0` all red herrings (or nothing failed), `1` something looks real, `3` unclear, `2` error.
65
+
66
+ ```sh
67
+ redherring why "$RUN" -q && gh run rerun "$RUN" --failed
68
+ ```
69
+
70
+ ### For coding agents
71
+
72
+ Agents lose a lot of time "fixing" failures that were never theirs, and sometimes weaken a perfectly good test to make a flake go green. Add this to your `AGENTS.md` or `CLAUDE.md`:
73
+
74
+ ```markdown
75
+ ## CI failures
76
+ Before changing code because CI failed, run `redherring why <run-url> --format json`.
77
+ - `recommendation: "rerun"`: every failure is a known flake, an infrastructure problem, or already
78
+ failing on main. Re-run the failed jobs (`gh run rerun <id> --failed`); don't edit code or tests for it.
79
+ - `recommendation: "investigate"`: fix only the findings with `"red_herring": false`.
80
+ - Never skip, loosen, or delete a test just because it is listed as flaky.
81
+ ```
82
+
83
+ Or install the ready-made skill: copy [`skills/redherring/`](skills/redherring/SKILL.md) into your agent's skills folder (for Claude Code, `.claude/skills/redherring/`). More setups are in [`examples/`](examples/).
84
+
85
+ ## GitHub Action
86
+
87
+ Explain every failed CI run, in its job summary and as a comment on the pull request:
88
+
89
+ ```yaml
90
+ # .github/workflows/redherring.yml
91
+ name: redherring
92
+ on:
93
+ workflow_run:
94
+ workflows: [CI] # the name of your test workflow
95
+ types: [completed]
96
+ permissions:
97
+ actions: read
98
+ contents: read
99
+ pull-requests: write # only needed for comment: true
100
+ jobs:
101
+ why:
102
+ if: github.event.workflow_run.conclusion == 'failure'
103
+ runs-on: ubuntu-latest
104
+ steps:
105
+ - uses: ixklo/redherring@v0.1.0
106
+ with:
107
+ comment: true
108
+ ```
109
+
110
+ The comment is posted once per PR and workflow, then edited in place on later failures, never duplicated.
111
+
112
+ Or post a weekly flaky-test report:
113
+
114
+ ```yaml
115
+ on:
116
+ schedule: [{ cron: "0 7 * * 1" }]
117
+ permissions:
118
+ actions: read
119
+ jobs:
120
+ scan:
121
+ runs-on: ubuntu-latest
122
+ steps:
123
+ - uses: ixklo/redherring@v0.1.0
124
+ with:
125
+ command: scan
126
+ days: "30"
127
+ ```
128
+
129
+ Inputs: `command` (`why` or `scan`), `run-id`, `days` (default 14), `workflow`, `ledger`, `comment`, `cache` (default on: keeps downloaded history between runs, because the Actions token allows about 1,000 API requests an hour), `fail-on-real`, `github-token`. Output: `recommendation`. From the command line, the same comment is `redherring why <run> --comment`.
130
+
131
+ ## Keep your flake history: the ledger
132
+
133
+ From 1 October 2026 GitHub deletes workflow runs once they pass your log-retention period (90 days by default). Past that, redherring has nothing to read. A ledger keeps the evidence:
134
+
135
+ ```sh
136
+ redherring scan --ledger .github/flake-ledger.json # merges new evidence into the file
137
+ redherring why "$RUN" --ledger .github/flake-ledger.json # uses it as extra history
138
+ ```
139
+
140
+ The file is small JSON (one entry per failed job: test names, cause, commit, OS, link). Keep it anywhere durable. You can commit it, or carry it between scheduled runs with `actions/cache`:
141
+
142
+ ```yaml
143
+ steps:
144
+ - uses: actions/cache@v6
145
+ with:
146
+ path: flake-ledger.json
147
+ key: redherring-ledger-${{ github.run_id }}
148
+ restore-keys: redherring-ledger-
149
+ - uses: ixklo/redherring@v0.1.0
150
+ with:
151
+ command: scan
152
+ ledger: flake-ledger.json
153
+ ```
154
+
155
+ Entries older than a year are dropped (`--keep-days`).
156
+
157
+ ## How it decides
158
+
159
+ redherring is deliberately strict about what it calls flaky. **Only one kind of evidence counts: the same commit failed and then passed.** That is what a re-run is. It does not guess from "failed on a PR, passed on main" or "this test fails a lot", because those are often real failures. A tool that cries wolf is worse than none.
160
+
161
+ Then, per failed job:
162
+
163
+ 1. **Tests.** The log is cleaned (timestamps, colour codes) and read by runner-specific parsers. Test ids keep only stable parts (file, suite, name; never line numbers or timings) so the same test lines up across commits and OSes.
164
+ 2. **Infrastructure.** If no test is named, the end of the log is matched against specific signatures (`ECONNRESET`, `Could not transfer artifact`, `The runner has received a shutdown signal`, `model is at capacity`, ...).
165
+ 3. **Not flakiness at all.** Roll-up jobs ("all required jobs passed") and PR policy checks are set aside and not counted.
166
+
167
+ Supported test output:
168
+
169
+ | ecosystem | runners |
170
+ |---|---|
171
+ | Python | pytest (incl. xdist, pytest-rerunfailures), unittest |
172
+ | JavaScript / TypeScript | Jest, Vitest (incl. workspaces), Playwright (incl. its own retries), Mocha, node:test (spec and TAP), Bun, TAP (QUnit/testem) |
173
+ | Go | `go test`, gotestsum, test timeouts |
174
+ | Rust | `cargo test`, cargo-nextest (incl. retries), `failed tests:` lists (Deno and other file-based runners) |
175
+ | Ruby | RSpec, Minitest |
176
+ | JVM | Maven Surefire, Gradle (incl. verbose test logging) |
177
+ | others | PHPUnit, .NET (`dotnet test`), CTest, XCTest, ExUnit |
178
+
179
+ Missing yours? A parser is one function and a test with a real log excerpt. See [CONTRIBUTING.md](CONTRIBUTING.md).
180
+
181
+ ## Limits, honestly
182
+
183
+ - **History has an expiry date.** Job logs are kept for 90 days by default, and [from 1 October 2026](https://github.blog/changelog/2026-08-27-actions-retention-will-cover-checks-workflow-runs-and-statuses/) GitHub deletes the workflow runs themselves once they pass the repo's log retention period (90 days by default, and at most 90 days for public repos). `--days` beyond that finds nothing; use a [ledger](#keep-your-flake-history-the-ledger) to keep what was found.
184
+ - **Only re-runs count.** Repos that never press re-run, or retry inside the job without reporting it, show fewer flakes than they have. Playwright, nextest and pytest-rerunfailures retries are recognised when they appear in a failed job's log.
185
+ - **Big repos cost API requests.** A busy monorepo can take a few thousand requests for 30 days. redherring waits out rate limits on its own, and caches everything immutable (attempt job lists, logs) so later runs only fetch what's new.
186
+ - **Some jobs stay "unknown".** In a study of 122 popular repos, nearly half of the red-herring jobs (46%) named neither a test nor a known infrastructure cause. Some logs genuinely say nothing ("exit 1"); others use output formats redherring doesn't parse yet. `why` treats those as "unclear", never as safe to re-run. Parser contributions fix this one format at a time.
187
+ - **GitHub Actions only**, for now.
188
+
189
+ ## Privacy
190
+
191
+ redherring runs on your machine (or in your own Actions runner) and talks only to the GitHub API. Downloaded logs are cached, compressed, in your user cache folder (`%LOCALAPPDATA%\redherring`, `~/Library/Caches/redherring`, or `~/.cache/redherring`; override with `REDHERRING_CACHE`). Nothing is sent anywhere else.
192
+
193
+ ## License
194
+
195
+ MIT
@@ -0,0 +1,59 @@
1
+ [project]
2
+ name = "redherring"
3
+ version = "0.1.0"
4
+ description = "Find the CI failures that weren't your fault: flaky tests and infrastructure hiccups, mined from the GitHub Actions history you already have."
5
+ readme = "README.md"
6
+ requires-python = ">=3.11"
7
+ license = "MIT"
8
+ license-files = ["LICENSE"]
9
+ authors = [{ name = "ixklo" }]
10
+ keywords = ["flaky-tests", "ci", "github-actions", "testing", "devtools", "coding-agents"]
11
+ classifiers = [
12
+ "Development Status :: 3 - Alpha",
13
+ "Environment :: Console",
14
+ "Intended Audience :: Developers",
15
+ "Operating System :: OS Independent",
16
+ "Programming Language :: Python :: 3",
17
+ "Programming Language :: Python :: 3.11",
18
+ "Programming Language :: Python :: 3.12",
19
+ "Programming Language :: Python :: 3.13",
20
+ "Topic :: Software Development :: Quality Assurance",
21
+ "Topic :: Software Development :: Testing",
22
+ ]
23
+ dependencies = [
24
+ "httpx>=0.27",
25
+ "rich>=13.7",
26
+ ]
27
+
28
+ [project.urls]
29
+ Homepage = "https://github.com/ixklo/redherring"
30
+ Issues = "https://github.com/ixklo/redherring/issues"
31
+
32
+ [project.scripts]
33
+ redherring = "redherring.cli:main"
34
+
35
+ [dependency-groups]
36
+ dev = [
37
+ "pytest>=8.0",
38
+ "ruff>=0.6",
39
+ ]
40
+
41
+ [build-system]
42
+ requires = ["hatchling>=1.26"]
43
+ build-backend = "hatchling.build"
44
+
45
+ [tool.hatch.build.targets.sdist]
46
+ include = ["/src", "/tests", "/README.md", "/LICENSE", "/CHANGELOG.md"]
47
+
48
+ [tool.pytest.ini_options]
49
+ testpaths = ["tests"]
50
+ addopts = "-q"
51
+
52
+ [tool.ruff]
53
+ line-length = 100
54
+ target-version = "py311"
55
+
56
+ [tool.ruff.lint]
57
+ select = ["E", "F", "I", "B", "UP", "SIM"]
58
+ # The formatter owns line length; long regexes and messages stay on one line.
59
+ ignore = ["E501"]
@@ -0,0 +1,3 @@
1
+ """redherring: find the CI failures that weren't your fault."""
2
+
3
+ __version__ = "0.1.0"
@@ -0,0 +1,5 @@
1
+ import sys
2
+
3
+ from .cli import main
4
+
5
+ sys.exit(main())
@@ -0,0 +1,106 @@
1
+ """A local cache of the parts of CI history that never change once a run is finished.
2
+
3
+ Job lists of completed attempts and job logs are immutable, so re-scans only pay for the
4
+ runs list. Logs are stored compressed so parsers can be improved without re-downloading.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ import os
11
+ import sqlite3
12
+ import sys
13
+ import threading
14
+ import zlib
15
+ from pathlib import Path
16
+
17
+ _SCHEMA = """
18
+ CREATE TABLE IF NOT EXISTS attempt_jobs (
19
+ repo TEXT NOT NULL, run_id INTEGER NOT NULL, attempt INTEGER NOT NULL,
20
+ jobs TEXT NOT NULL,
21
+ PRIMARY KEY (repo, run_id, attempt)
22
+ );
23
+ CREATE TABLE IF NOT EXISTS job_logs (
24
+ repo TEXT NOT NULL, job_id INTEGER NOT NULL,
25
+ status TEXT NOT NULL, -- ok | gone
26
+ log BLOB, -- zlib-compressed text when status = ok
27
+ PRIMARY KEY (repo, job_id)
28
+ );
29
+ """
30
+
31
+ # Logs above this (compressed) size are truncated to their tail: the failure is at the end.
32
+ _MAX_COMPRESSED = 4 * 1024 * 1024
33
+ _MAX_TEXT = 12 * 1024 * 1024
34
+
35
+
36
+ def default_path() -> Path:
37
+ if env := os.environ.get("REDHERRING_CACHE"):
38
+ return Path(env)
39
+ if sys.platform == "win32":
40
+ base = Path(os.environ.get("LOCALAPPDATA", Path.home() / "AppData" / "Local"))
41
+ elif sys.platform == "darwin":
42
+ base = Path.home() / "Library" / "Caches"
43
+ else:
44
+ base = Path(os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache"))
45
+ return base / "redherring" / "cache.sqlite"
46
+
47
+
48
+ class Cache:
49
+ def __init__(self, path: Path | str | None = None) -> None:
50
+ if path == ":memory:":
51
+ self._db = sqlite3.connect(":memory:", check_same_thread=False)
52
+ else:
53
+ p = Path(path) if path else default_path()
54
+ p.parent.mkdir(parents=True, exist_ok=True)
55
+ self._db = sqlite3.connect(p, check_same_thread=False, isolation_level=None)
56
+ self._db.execute("PRAGMA journal_mode=WAL")
57
+ self._db.executescript(_SCHEMA)
58
+ self._lock = threading.Lock()
59
+
60
+ def _exec(self, sql: str, args: tuple = ()) -> None:
61
+ with self._lock:
62
+ self._db.execute(sql, args)
63
+
64
+ def _one(self, sql: str, args: tuple = ()) -> tuple | None:
65
+ with self._lock:
66
+ return self._db.execute(sql, args).fetchone()
67
+
68
+ def close(self) -> None:
69
+ self._db.close()
70
+
71
+ def get_attempt_jobs(self, repo: str, run_id: int, attempt: int) -> list[dict] | None:
72
+ row = self._one(
73
+ "SELECT jobs FROM attempt_jobs WHERE repo=? AND run_id=? AND attempt=?",
74
+ (repo, run_id, attempt),
75
+ )
76
+ return json.loads(row[0]) if row else None
77
+
78
+ def put_attempt_jobs(self, repo: str, run_id: int, attempt: int, jobs: list[dict]) -> None:
79
+ self._exec(
80
+ "INSERT OR REPLACE INTO attempt_jobs VALUES (?, ?, ?, ?)",
81
+ (repo, run_id, attempt, json.dumps(jobs)),
82
+ )
83
+
84
+ def get_log(self, repo: str, job_id: int) -> tuple[str, str | None] | None:
85
+ """(status, text) if cached, else None. status is "ok" or "gone"."""
86
+ row = self._one(
87
+ "SELECT status, log FROM job_logs WHERE repo=? AND job_id=?", (repo, job_id)
88
+ )
89
+ if not row:
90
+ return None
91
+ status, blob = row
92
+ return status, zlib.decompress(blob).decode("utf-8") if blob else None
93
+
94
+ def put_log(self, repo: str, job_id: int, text: str | None) -> None:
95
+ if text is None:
96
+ self._exec(
97
+ "INSERT OR REPLACE INTO job_logs VALUES (?, ?, 'gone', NULL)", (repo, job_id)
98
+ )
99
+ return
100
+ if len(text) > _MAX_TEXT:
101
+ text = text[-_MAX_TEXT:]
102
+ blob = zlib.compress(text.encode("utf-8"), 6)
103
+ while len(blob) > _MAX_COMPRESSED and len(text) > 1024:
104
+ text = text[len(text) // 2 :]
105
+ blob = zlib.compress(text.encode("utf-8"), 6)
106
+ self._exec("INSERT OR REPLACE INTO job_logs VALUES (?, ?, 'ok', ?)", (repo, job_id, blob))