deadgate 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- deadgate/__init__.py +0 -0
- deadgate/checknames.py +180 -0
- deadgate/cli.py +124 -0
- deadgate/detectors.py +362 -0
- deadgate/predicates.py +141 -0
- deadgate/protection.py +238 -0
- deadgate/resolve.py +117 -0
- deadgate/sandbox.py +153 -0
- deadgate-0.1.0.dist-info/METADATA +158 -0
- deadgate-0.1.0.dist-info/RECORD +14 -0
- deadgate-0.1.0.dist-info/WHEEL +5 -0
- deadgate-0.1.0.dist-info/entry_points.txt +2 -0
- deadgate-0.1.0.dist-info/licenses/LICENSE +21 -0
- deadgate-0.1.0.dist-info/top_level.txt +1 -0
deadgate/__init__.py
ADDED
|
File without changes
|
deadgate/checknames.py
ADDED
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
"""Derive the check-run names a workflow job produces.
|
|
2
|
+
|
|
3
|
+
A required status check is identified by its CHECK-RUN NAME, which is not the job key in
|
|
4
|
+
the YAML. Getting this wrong is the way this tier lies: if derivation silently fails, no
|
|
5
|
+
required context matches any job, every finding resolves to "not required", and the tool
|
|
6
|
+
quietly downgrades real defects. So derivation reports its own confidence, and anything it
|
|
7
|
+
cannot work out is AMBIGUOUS rather than "no match".
|
|
8
|
+
|
|
9
|
+
The naming rules GitHub actually applies:
|
|
10
|
+
|
|
11
|
+
no `name:` the job key
|
|
12
|
+
no `name:` + a matrix `<job key> (<values, in matrix key order>)`
|
|
13
|
+
`name:` + a matrix that string, ALSO suffixed `(<values>)`
|
|
14
|
+
`name:` already naming matrix the string per leg, with expressions substituted, no suffix
|
|
15
|
+
`uses:` (reusable workflow) `<caller name> / <callee job name>`
|
|
16
|
+
|
|
17
|
+
The rule for a static `name:` with a matrix is append-the-leg-values, not use-it-once. Verified
|
|
18
|
+
against the authoritative surface: promptfoo's `main` branch requires the context
|
|
19
|
+
`Check Python (3.9)` while its workflow declares a static `name: Check Python` over a
|
|
20
|
+
`python-version` matrix, and its `Build on Node ${{ matrix.node }}` job, whose name already
|
|
21
|
+
references the matrix, is required as the unsuffixed `Build on Node 24.x`. Encoding the
|
|
22
|
+
opposite made every leg of a statically named matrix job match nothing.
|
|
23
|
+
|
|
24
|
+
Matrix legs and reusable callees are matched by PREFIX rather than enumerated, because the
|
|
25
|
+
prefix is decidable from the caller alone while the values often are not. That keeps a
|
|
26
|
+
`matrix: ${{ fromJson(needs.x.outputs.y) }}` job decidable instead of ambiguous.
|
|
27
|
+
"""
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
import itertools
|
|
31
|
+
import re
|
|
32
|
+
from dataclasses import dataclass
|
|
33
|
+
|
|
34
|
+
EXACT = "EXACT"
|
|
35
|
+
AMBIGUOUS = "AMBIGUOUS"
|
|
36
|
+
|
|
37
|
+
_EXPR = re.compile(r"\$\{\{(.+?)\}\}", re.S)
|
|
38
|
+
_MATRIX_REF = re.compile(r"^\s*matrix\.([A-Za-z_][\w-]*)\s*$")
|
|
39
|
+
MAX_COMBINATIONS = 256
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass(frozen=True)
|
|
43
|
+
class Derived:
|
|
44
|
+
"""Check-run names a job produces, plus how sure we are."""
|
|
45
|
+
names: frozenset[str]
|
|
46
|
+
prefixes: frozenset[str]
|
|
47
|
+
confidence: str
|
|
48
|
+
reason: str = ""
|
|
49
|
+
|
|
50
|
+
def matches(self, context: str) -> bool:
|
|
51
|
+
if self.confidence != EXACT:
|
|
52
|
+
raise ValueError("refusing to match on an AMBIGUOUS derivation")
|
|
53
|
+
return context in self.names or any(context.startswith(p) for p in self.prefixes)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _render(value) -> str:
|
|
57
|
+
"""A matrix value as GitHub interpolates it, which is not Python's str().
|
|
58
|
+
|
|
59
|
+
A YAML boolean renders `true`, not `True`, and null renders as the empty string. Getting
|
|
60
|
+
this wrong produces a name that matches nothing, and a name that matches nothing clears
|
|
61
|
+
the finding.
|
|
62
|
+
"""
|
|
63
|
+
if isinstance(value, bool):
|
|
64
|
+
return "true" if value else "false"
|
|
65
|
+
if value is None:
|
|
66
|
+
return ""
|
|
67
|
+
return str(value)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _literal_matrix(strategy) -> dict | None:
|
|
71
|
+
"""The matrix as literal lists, or None when any part of it is computed.
|
|
72
|
+
|
|
73
|
+
`include` and `exclude` add and remove legs by rules not worth guessing at: an
|
|
74
|
+
under-enumerated name set silently fails to match a context that IS required, which clears
|
|
75
|
+
the finding. Both are lists of mappings, so they make the matrix non-literal here.
|
|
76
|
+
"""
|
|
77
|
+
if not isinstance(strategy, dict):
|
|
78
|
+
return None
|
|
79
|
+
matrix = strategy.get("matrix")
|
|
80
|
+
if not isinstance(matrix, dict):
|
|
81
|
+
return None
|
|
82
|
+
axes = {}
|
|
83
|
+
for key, val in matrix.items():
|
|
84
|
+
# `include` and `exclude` land here too, and that is deliberate: both are lists of
|
|
85
|
+
# mappings, so the check below already makes the matrix non-literal and the caller
|
|
86
|
+
# falls through to AMBIGUOUS. An explicit guard for them was written first and was
|
|
87
|
+
# unreachable, which is the defect this tool is named after.
|
|
88
|
+
if not isinstance(val, list) or any(isinstance(v, (dict, list)) for v in val):
|
|
89
|
+
return None
|
|
90
|
+
if any("${{" in str(v) for v in val):
|
|
91
|
+
return None
|
|
92
|
+
axes[key] = [_render(v) for v in val]
|
|
93
|
+
return axes or None
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _substitute(template: str, axes: dict) -> frozenset[str] | None:
|
|
97
|
+
"""Expand a `name:` template over a literal matrix, or None if it cannot be expanded."""
|
|
98
|
+
refs = []
|
|
99
|
+
for raw in _EXPR.findall(template):
|
|
100
|
+
m = _MATRIX_REF.match(raw)
|
|
101
|
+
if not m or m.group(1) not in axes:
|
|
102
|
+
return None # an expression we cannot evaluate: not our business to guess
|
|
103
|
+
refs.append(m.group(1))
|
|
104
|
+
if not refs:
|
|
105
|
+
return frozenset({template})
|
|
106
|
+
used = {r: axes[r] for r in dict.fromkeys(refs)}
|
|
107
|
+
combos = list(itertools.product(*used.values()))
|
|
108
|
+
if len(combos) > MAX_COMBINATIONS:
|
|
109
|
+
return None
|
|
110
|
+
out = set()
|
|
111
|
+
for combo in combos:
|
|
112
|
+
values = dict(zip(used.keys(), combo))
|
|
113
|
+
# GitHub trims the rendered name, which matters when a leg's value is empty.
|
|
114
|
+
out.add(_EXPR.sub(lambda m: values[_MATRIX_REF.match(m.group(1)).group(1)],
|
|
115
|
+
template).strip())
|
|
116
|
+
# A job SKIPPED by its job-level if: emits one check run carrying the RAW template, with
|
|
117
|
+
# the expressions left literal and no matrix expansion. Skipped jobs are precisely this
|
|
118
|
+
# tool's target population, so the unexpanded form is a legitimate candidate name.
|
|
119
|
+
out.add(template.strip())
|
|
120
|
+
return frozenset(out)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def derive(job_key: str, job: dict, workflow_callable: bool = False) -> Derived:
|
|
124
|
+
"""The check-run names job `job_key` produces.
|
|
125
|
+
|
|
126
|
+
`workflow_callable` says this job's own workflow declares `on: workflow_call`. When another
|
|
127
|
+
workflow in the repository `uses:` it, GitHub names the check `<caller job> / <this job>`,
|
|
128
|
+
and which caller, if any, is not decidable from this file. Guessing the unprefixed name
|
|
129
|
+
clears a finding on a job that IS a required gate, so this is AMBIGUOUS instead.
|
|
130
|
+
"""
|
|
131
|
+
job = job if isinstance(job, dict) else {}
|
|
132
|
+
if workflow_callable:
|
|
133
|
+
return Derived(frozenset(), frozenset(), AMBIGUOUS,
|
|
134
|
+
"workflow is `uses:`-callable, so its checks may be named "
|
|
135
|
+
"`<caller> / <job>` and the caller is not knowable from this file")
|
|
136
|
+
template = job.get("name")
|
|
137
|
+
axes = _literal_matrix(job.get("strategy"))
|
|
138
|
+
has_matrix = isinstance(job.get("strategy"), dict) and job["strategy"].get("matrix") is not None
|
|
139
|
+
|
|
140
|
+
if template is None:
|
|
141
|
+
base = job_key
|
|
142
|
+
if "${{" in base:
|
|
143
|
+
return Derived(frozenset(), frozenset(), AMBIGUOUS, "job key contains an expression")
|
|
144
|
+
if job.get("uses"):
|
|
145
|
+
# A reusable call produces `caller / callee`, and a MATRIXED one produces
|
|
146
|
+
# `caller (values) / callee`. Checking uses: before the matrix emitted only the
|
|
147
|
+
# first prefix, so every matrixed reusable call matched nothing.
|
|
148
|
+
prefixes = {f"{base} / "}
|
|
149
|
+
if has_matrix:
|
|
150
|
+
prefixes.add(f"{base} (")
|
|
151
|
+
# A skipped reusable call emits the bare caller name with no callee suffix.
|
|
152
|
+
return Derived(frozenset({base}), frozenset(prefixes), EXACT,
|
|
153
|
+
"reusable workflow" + (" with matrix legs" if has_matrix else ""))
|
|
154
|
+
prefixes = frozenset({f"{base} ("}) if has_matrix else frozenset()
|
|
155
|
+
return Derived(frozenset({base}), prefixes, EXACT,
|
|
156
|
+
"job key" + (" with matrix legs" if has_matrix else ""))
|
|
157
|
+
|
|
158
|
+
template = str(template)
|
|
159
|
+
if "${{" not in template:
|
|
160
|
+
# A static name is SUFFIXED with the leg values when the job has a matrix. Encoding
|
|
161
|
+
# the opposite, that it is used once verbatim, made every leg match nothing.
|
|
162
|
+
prefixes = {f"{template} ("} if has_matrix else set()
|
|
163
|
+
if job.get("uses"):
|
|
164
|
+
prefixes.add(f"{template} / ")
|
|
165
|
+
return Derived(frozenset({template}), frozenset(prefixes), EXACT,
|
|
166
|
+
"reusable workflow" + (" with matrix legs" if has_matrix else ""))
|
|
167
|
+
return Derived(frozenset({template}), frozenset(prefixes), EXACT,
|
|
168
|
+
"explicit name" + (" with matrix legs" if has_matrix else ""))
|
|
169
|
+
|
|
170
|
+
if axes is None:
|
|
171
|
+
return Derived(frozenset(), frozenset(), AMBIGUOUS,
|
|
172
|
+
"name: has an expression and the matrix is not literal")
|
|
173
|
+
names = _substitute(template, axes)
|
|
174
|
+
if names is None:
|
|
175
|
+
return Derived(frozenset(), frozenset(), AMBIGUOUS,
|
|
176
|
+
"name: has an expression that does not resolve from the matrix")
|
|
177
|
+
if job.get("uses"):
|
|
178
|
+
return Derived(names, frozenset(f"{n} / " for n in names), EXACT,
|
|
179
|
+
"reusable workflow, name expanded over the matrix")
|
|
180
|
+
return Derived(names, frozenset(), EXACT, "name expanded over the matrix")
|
deadgate/cli.py
ADDED
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
"""deadgate: find CI checks that cannot fail."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
import pathlib
|
|
6
|
+
from dataclasses import replace
|
|
7
|
+
import sys
|
|
8
|
+
|
|
9
|
+
import yaml
|
|
10
|
+
|
|
11
|
+
from .detectors import scan_workflow
|
|
12
|
+
from .protection import fetch, gh_api, repo_meta, selftest
|
|
13
|
+
from .resolve import attribute, resolve, workflow_is_callable
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def workflows(root: pathlib.Path):
|
|
17
|
+
for pat in ("*.yml", "*.yaml"):
|
|
18
|
+
yield from sorted(root.rglob(f".github/workflows/{pat}"))
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def main(argv: list[str] | None = None) -> int:
|
|
22
|
+
ap = argparse.ArgumentParser(prog="deadgate", description=__doc__)
|
|
23
|
+
ap.add_argument("path", nargs="?", default=".", help="repository root")
|
|
24
|
+
ap.add_argument("--quiet", action="store_true", help="only print the summary")
|
|
25
|
+
ap.add_argument("--repo", metavar="OWNER/NAME",
|
|
26
|
+
help="resolve MEDIUM findings against what the branch actually requires. "
|
|
27
|
+
"Read-only GitHub API calls via `gh`. On a repository you do not "
|
|
28
|
+
"administer the required set is incomplete, so findings can only be "
|
|
29
|
+
"escalated, never cleared.")
|
|
30
|
+
ap.add_argument("--branch", help="branch to read protection from (default: the repo's default)")
|
|
31
|
+
ap.add_argument("--all", action="store_true",
|
|
32
|
+
help="include LOW findings (release and deploy pipelines, where a skip "
|
|
33
|
+
"is usually the intent). Hidden by default so the output stays actionable.")
|
|
34
|
+
args = ap.parse_args(argv)
|
|
35
|
+
|
|
36
|
+
root = pathlib.Path(args.path)
|
|
37
|
+
files = list(workflows(root))
|
|
38
|
+
if not files:
|
|
39
|
+
print(f"no workflow files under {root}/.github/workflows", file=sys.stderr)
|
|
40
|
+
return 0
|
|
41
|
+
|
|
42
|
+
prot = None
|
|
43
|
+
jobs_by_file: dict[str, dict] = {}
|
|
44
|
+
callable_files: set[str] = set()
|
|
45
|
+
if args.repo:
|
|
46
|
+
api = gh_api()
|
|
47
|
+
ok, why = selftest(api)
|
|
48
|
+
if not ok:
|
|
49
|
+
# Refuse rather than report every finding UNREADABLE, which would look like a
|
|
50
|
+
# result about the repository instead of a broken transport.
|
|
51
|
+
print(f"GitHub API transport self-test failed: {why}", file=sys.stderr)
|
|
52
|
+
return 2
|
|
53
|
+
default_branch, admin = repo_meta(args.repo, api)
|
|
54
|
+
branch = args.branch or default_branch
|
|
55
|
+
if not branch:
|
|
56
|
+
print(f"could not read {args.repo} from the GitHub API; is `gh` authenticated?",
|
|
57
|
+
file=sys.stderr)
|
|
58
|
+
return 2
|
|
59
|
+
prot = fetch(args.repo, branch, api, admin=admin)
|
|
60
|
+
print(f"branch {branch}: {prot.state}"
|
|
61
|
+
f"{f', {len(prot.required)} required check(s)' if prot.required else ''}"
|
|
62
|
+
f"{'' if prot.complete else ', required set INCOMPLETE (no admin)'}"
|
|
63
|
+
f"{f' [{prot.detail}]' if prot.detail else ''}\n")
|
|
64
|
+
|
|
65
|
+
findings = []
|
|
66
|
+
suppressed = 0
|
|
67
|
+
for f in files:
|
|
68
|
+
try:
|
|
69
|
+
doc = yaml.safe_load(f.read_text())
|
|
70
|
+
except yaml.YAMLError as exc:
|
|
71
|
+
# Refuse to report a clean result for a file we could not read. A parse
|
|
72
|
+
# failure counted as "no findings" is the exact defect this tool exists to find.
|
|
73
|
+
print(f"UNREADABLE {f}: {exc.__class__.__name__}", file=sys.stderr)
|
|
74
|
+
return 2
|
|
75
|
+
jobs_by_file[str(f)] = (doc or {}).get("jobs") or {}
|
|
76
|
+
if workflow_is_callable(doc):
|
|
77
|
+
callable_files.add(str(f))
|
|
78
|
+
for x in scan_workflow(doc):
|
|
79
|
+
why = ""
|
|
80
|
+
if prot is not None:
|
|
81
|
+
job = ((doc or {}).get("jobs") or {}).get(x.job)
|
|
82
|
+
r = resolve(x.severity, x.job, job if isinstance(job, dict) else {}, prot,
|
|
83
|
+
str(f) in callable_files)
|
|
84
|
+
why = f"{r.verdict}: {r.why}"
|
|
85
|
+
if r.moved:
|
|
86
|
+
why = f"{r.structural} -> {r.severity} {why}"
|
|
87
|
+
x = replace(x, severity=r.severity)
|
|
88
|
+
if x.severity == "LOW" and not args.all:
|
|
89
|
+
suppressed += 1
|
|
90
|
+
continue
|
|
91
|
+
findings.append((f, x, why))
|
|
92
|
+
|
|
93
|
+
if not args.quiet:
|
|
94
|
+
for f, x, why in findings:
|
|
95
|
+
rel = f.relative_to(root) if f.is_relative_to(root) else f
|
|
96
|
+
print(f"[{x.detector}/{x.severity}] {rel}::{x.job} {x.title}")
|
|
97
|
+
print(f" {x.detail}")
|
|
98
|
+
if why:
|
|
99
|
+
print(f" branch: {why}")
|
|
100
|
+
print(f" repro: {x.repro}\n")
|
|
101
|
+
|
|
102
|
+
tail = f", {suppressed} LOW hidden (use --all)" if suppressed else ""
|
|
103
|
+
print(f"{len(files)} workflow file(s), {len(findings)} finding(s){tail}")
|
|
104
|
+
|
|
105
|
+
if prot is not None and prot.required:
|
|
106
|
+
att = attribute(prot, jobs_by_file, frozenset(callable_files))
|
|
107
|
+
print(f"required checks: {len(att.attributed)}/{len(att.required)} attributed to a job "
|
|
108
|
+
f"in this repository")
|
|
109
|
+
if att.unattributed:
|
|
110
|
+
# Printed because a broken name derivation shows up here as a number instead of
|
|
111
|
+
# as silently cleared findings. Third-party checks land here legitimately.
|
|
112
|
+
print(f" unattributed: {', '.join(att.unattributed[:8])}"
|
|
113
|
+
f"{' ...' if len(att.unattributed) > 8 else ''}")
|
|
114
|
+
if att.undecidable_jobs:
|
|
115
|
+
print(f" jobs whose check name could not be derived: {len(att.undecidable_jobs)}")
|
|
116
|
+
if att.suspicious:
|
|
117
|
+
print(" WARNING: no required check matched any job here. Either every gate is "
|
|
118
|
+
"external, or the name derivation is broken. Do not read the severities "
|
|
119
|
+
"above as resolved.")
|
|
120
|
+
return 1 if findings else 0
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
if __name__ == "__main__":
|
|
124
|
+
raise SystemExit(main())
|
deadgate/detectors.py
ADDED
|
@@ -0,0 +1,362 @@
|
|
|
1
|
+
"""Structural detectors for CI checks that cannot fail.
|
|
2
|
+
|
|
3
|
+
Every detector reports a FINDING only when the defect is STRUCTURAL and decidable from
|
|
4
|
+
the workflow file alone. Anything needing branch-protection state or runtime history is
|
|
5
|
+
out of scope here and belongs to the API tier.
|
|
6
|
+
|
|
7
|
+
Design rule taken from gate_mutation_sweep.py: a detector must fire on the planted defect
|
|
8
|
+
AND stay quiet on a near miss. A detector that fires on everything gets the whole tool
|
|
9
|
+
switched off, which is worse than not shipping it.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import re
|
|
14
|
+
import os
|
|
15
|
+
from dataclasses import dataclass
|
|
16
|
+
|
|
17
|
+
# EXPERIMENT AFFORDANCE, not a product feature. With DEADGATE_NAIVE=1 the detectors run
|
|
18
|
+
# WITHOUT the suppressions and narrowings that tracing real repositories forced on them,
|
|
19
|
+
# and D4 is withheld because it did not exist then. It exists so the cost of each
|
|
20
|
+
# suppression can be measured against an identical corpus instead of two samples.
|
|
21
|
+
NAIVE = os.environ.get("DEADGATE_NAIVE") == "1"
|
|
22
|
+
|
|
23
|
+
FILTERS = {"grep", "jq", "head", "tail", "tee", "awk", "sed", "cut", "sort", "uniq", "wc", "tr"}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass(frozen=True)
|
|
27
|
+
class Finding:
|
|
28
|
+
detector: str
|
|
29
|
+
job: str
|
|
30
|
+
title: str
|
|
31
|
+
detail: str
|
|
32
|
+
repro: str
|
|
33
|
+
severity: str = "HIGH"
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _truthy_always(cond) -> bool:
|
|
37
|
+
"""Does this if: condition reduce to always()?"""
|
|
38
|
+
if cond is None:
|
|
39
|
+
return False
|
|
40
|
+
s = str(cond).strip()
|
|
41
|
+
s = re.sub(r"^\$\{\{\s*|\s*\}\}$", "", s).strip()
|
|
42
|
+
return s == "always()"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
# Conditions that are NOT skip-prone. A job guarded by one of these runs in normal
|
|
46
|
+
# operation, so treating it as "might skip" is a false alarm. Found by running against
|
|
47
|
+
# promptfoo main.yml, where package-acceptance carries `if: ${{ !cancelled() }}` and
|
|
48
|
+
# the corpus had nothing like it.
|
|
49
|
+
_NEVER_SKIPS = re.compile(r"^(always\(\)|!\s*cancelled\(\)|success\(\)\s*\|\|\s*failure\(\)|"
|
|
50
|
+
r"failure\(\)\s*\|\|\s*success\(\))$")
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _skip_prone(cond) -> bool:
|
|
54
|
+
"""A job-level if: that can realistically evaluate false and skip the job."""
|
|
55
|
+
if cond is None:
|
|
56
|
+
return False
|
|
57
|
+
s = re.sub(r"^\$\{\{\s*|\s*\}\}$", "", str(cond).strip()).strip()
|
|
58
|
+
return not _NEVER_SKIPS.match(s)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _steps_text(job: dict) -> str:
|
|
62
|
+
"""All text a job's steps can reference, so we can look for needs.* reads."""
|
|
63
|
+
out = []
|
|
64
|
+
for step in job.get("steps") or []:
|
|
65
|
+
if isinstance(step, dict):
|
|
66
|
+
out.append(str(step.get("run", "")))
|
|
67
|
+
out.append(str(step.get("if", "")))
|
|
68
|
+
env = step.get("env") or {}
|
|
69
|
+
if isinstance(env, dict):
|
|
70
|
+
out.extend(str(v) for v in env.values())
|
|
71
|
+
with_ = step.get("with") or {}
|
|
72
|
+
if isinstance(with_, dict):
|
|
73
|
+
out.extend(str(v) for v in with_.values())
|
|
74
|
+
out.append(str(job.get("env") or ""))
|
|
75
|
+
# The job-level if: is a result check too. A job guarded by
|
|
76
|
+
# `if: always() && needs.x.outputs.y` IS reading its upstream, and missing this
|
|
77
|
+
# was a LEAKY bug found by running against real workflows, not by the corpus.
|
|
78
|
+
# Behind NAIVE so the flag reproduces the pre-fix behaviour rather than quietly
|
|
79
|
+
# keeping the fix in both arms, which made an A/B report a difference of zero.
|
|
80
|
+
if not NAIVE:
|
|
81
|
+
out.append(str(job.get("if") or ""))
|
|
82
|
+
return "\n".join(out)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def d2_fanin_without_result_check(jobs: dict) -> list[Finding]:
|
|
86
|
+
"""A fan-in job that runs on always() and never reads needs.*.result.
|
|
87
|
+
|
|
88
|
+
It will be GREEN when the jobs it gates FAILED. If it is a required check, the
|
|
89
|
+
branch protection it provides is decorative.
|
|
90
|
+
"""
|
|
91
|
+
found = []
|
|
92
|
+
for name, job in jobs.items():
|
|
93
|
+
if not isinstance(job, dict):
|
|
94
|
+
continue
|
|
95
|
+
needs = job.get("needs")
|
|
96
|
+
if not needs or not _truthy_always(job.get("if")):
|
|
97
|
+
continue
|
|
98
|
+
text = _steps_text(job)
|
|
99
|
+
if re.search(r"needs\.[A-Za-z0-9_\-]+\.(result|outputs)", text):
|
|
100
|
+
continue
|
|
101
|
+
upstream = ", ".join(needs) if isinstance(needs, list) else str(needs)
|
|
102
|
+
found.append(Finding(
|
|
103
|
+
detector="D2",
|
|
104
|
+
job=name,
|
|
105
|
+
title="fan-in gate cannot fail",
|
|
106
|
+
detail=(f"job '{name}' needs [{upstream}] and runs on always(), but no step reads "
|
|
107
|
+
f"needs.*.result. It reports success even when [{upstream}] fail."),
|
|
108
|
+
repro=(f"Make any of [{upstream}] exit 1 and re-run. '{name}' still succeeds, "
|
|
109
|
+
f"and any branch protection requiring it still passes."),
|
|
110
|
+
))
|
|
111
|
+
return found
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def d1_skippable_upstream(jobs: dict, doc: dict | None = None) -> list[Finding]:
|
|
115
|
+
"""A conditional job that something depends on, where the dependent never checks the result.
|
|
116
|
+
|
|
117
|
+
GitHub documents that a skipped job reports Success and does not prevent a merge even
|
|
118
|
+
as a required check, and that a skipped conclusion is treated as success for dependent
|
|
119
|
+
checks. So skipping the real work silently satisfies the gate.
|
|
120
|
+
"""
|
|
121
|
+
found = []
|
|
122
|
+
conditional = {n for n, j in jobs.items()
|
|
123
|
+
if isinstance(j, dict) and _skip_prone(j.get("if"))}
|
|
124
|
+
# A dependent whose own if: reads the upstream's outputs is deliberately gated on it:
|
|
125
|
+
# the change-detection pattern, which is intentional, and whose real failure mode is
|
|
126
|
+
# D4's. That exclusion happens in _steps_text, which reads the job-level if:. An
|
|
127
|
+
# explicit second check here was DEAD CODE and is removed; the A/B that was supposed
|
|
128
|
+
# to prove it worked reported a difference of exactly zero and found it instead.
|
|
129
|
+
for name, job in jobs.items():
|
|
130
|
+
if not isinstance(job, dict):
|
|
131
|
+
continue
|
|
132
|
+
needs = job.get("needs")
|
|
133
|
+
if not needs:
|
|
134
|
+
continue
|
|
135
|
+
needs_list = needs if isinstance(needs, list) else [needs]
|
|
136
|
+
skippable = [n for n in needs_list if n in conditional]
|
|
137
|
+
if not skippable:
|
|
138
|
+
continue
|
|
139
|
+
text = _steps_text(job)
|
|
140
|
+
if re.search(r"needs\.[A-Za-z0-9_\-]+\.(result|outputs)", text):
|
|
141
|
+
continue
|
|
142
|
+
sev = _gate_severity(name, job, doc or {})
|
|
143
|
+
for up in skippable:
|
|
144
|
+
found.append(Finding(
|
|
145
|
+
detector="D1",
|
|
146
|
+
severity=sev,
|
|
147
|
+
job=name,
|
|
148
|
+
title="gate satisfied by a skipped job",
|
|
149
|
+
detail=(f"job '{name}' depends on '{up}', which is conditional. A skipped job "
|
|
150
|
+
f"reports Success, and '{name}' never reads needs.{up}.result."),
|
|
151
|
+
repro=(f"Open a PR where '{up}'s if: condition is false. '{up}' skips, "
|
|
152
|
+
f"'{name}' succeeds, and nothing ran."),
|
|
153
|
+
))
|
|
154
|
+
return found
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
_CANNOT_FAIL = re.compile(r"^\s*(echo|printf)\b")
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _substitution_bodies(line: str) -> list[str]:
|
|
161
|
+
"""Bodies of $( ... ), innermost first, so a pipe inside one is analysed on its own."""
|
|
162
|
+
out, stack = [], []
|
|
163
|
+
i = 0
|
|
164
|
+
while i < len(line) - 1:
|
|
165
|
+
if line[i] == "$" and line[i + 1] == "(":
|
|
166
|
+
stack.append(i + 2); i += 2; continue
|
|
167
|
+
if line[i] == ")" and stack:
|
|
168
|
+
out.append(line[stack.pop():i])
|
|
169
|
+
i += 1
|
|
170
|
+
return out
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def _upstream_can_fail(line: str) -> bool:
|
|
174
|
+
"""Can the command BEFORE the final pipe actually fail?
|
|
175
|
+
|
|
176
|
+
Two shapes are benign and together they were 10 of 20 sampled findings:
|
|
177
|
+
echo "literal" | cut the head cannot fail
|
|
178
|
+
echo "X=$(echo literal | cut)" >> file the pipe lives INSIDE a substitution whose
|
|
179
|
+
own head is echo, and splitting the whole
|
|
180
|
+
line on its last pipe misses that.
|
|
181
|
+
"""
|
|
182
|
+
for body in _substitution_bodies(line):
|
|
183
|
+
if "|" in body:
|
|
184
|
+
head = body.rsplit("|", 1)[0].strip()
|
|
185
|
+
if _CANNOT_FAIL.match(head) and "$(" not in head and "`" not in head:
|
|
186
|
+
return False
|
|
187
|
+
head = line.rsplit("|", 1)[0].strip()
|
|
188
|
+
if "$(" in head or "`" in head:
|
|
189
|
+
# the outer head holds a substitution we already judged benign above
|
|
190
|
+
return True
|
|
191
|
+
return not _CANNOT_FAIL.match(head)
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def _result_is_emptiness_checked(var: str, body: str) -> bool:
|
|
195
|
+
"""Did the author guard the empty case themselves, in the same step?
|
|
196
|
+
|
|
197
|
+
`X=$(find ... | head -1)` followed by `if [ -z "$X" ]; then exit 1` is a handled
|
|
198
|
+
case, not a defect. Reporting it anyway is how a tool gets uninstalled.
|
|
199
|
+
"""
|
|
200
|
+
if not var:
|
|
201
|
+
return False
|
|
202
|
+
return bool(re.search(rf"-[zn]\s+\"?\$\{{?{re.escape(var)}\}}?", body))
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def _d3_severity(step: dict, job_name: str, job: dict, doc: dict, line: str) -> str:
|
|
206
|
+
"""Does the masked exit status actually decide anything?
|
|
207
|
+
|
|
208
|
+
A pipeline whose result is exported, or which sits in a step that can fail the build,
|
|
209
|
+
masks a decision. A pipeline in a cleanup or logging step masks nothing anybody reads.
|
|
210
|
+
Tracing twenty D3 findings by hand, the ones that mattered all had a consumer and the
|
|
211
|
+
ones that did not were fire-and-forget.
|
|
212
|
+
"""
|
|
213
|
+
body = str(step.get("run") or "")
|
|
214
|
+
if not _on_pull_request(doc):
|
|
215
|
+
return "LOW"
|
|
216
|
+
exported = "GITHUB_OUTPUT" in body or "GITHUB_ENV" in body
|
|
217
|
+
decides = bool(re.search(r"\bexit\s+[1-9]|::error::", body))
|
|
218
|
+
assigned = re.match(r"\s*([A-Za-z_][A-Za-z0-9_]*)=", line)
|
|
219
|
+
consumed = bool(assigned and re.search(rf"\$\{{?{re.escape(assigned.group(1))}\b",
|
|
220
|
+
body.replace(line, "", 1)))
|
|
221
|
+
if decides or exported or consumed:
|
|
222
|
+
return "HIGH"
|
|
223
|
+
return "MEDIUM"
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def d3_pipe_masked_exit(jobs: dict, doc: dict | None = None) -> list[Finding]:
|
|
227
|
+
"""A run: step whose exit status is the LAST command in a pipe, with no pipefail.
|
|
228
|
+
|
|
229
|
+
The step's status reports the filter's success, not the real command's. An outage
|
|
230
|
+
reads as a pass.
|
|
231
|
+
"""
|
|
232
|
+
found = []
|
|
233
|
+
for name, job in jobs.items():
|
|
234
|
+
if not isinstance(job, dict):
|
|
235
|
+
continue
|
|
236
|
+
for step in job.get("steps") or []:
|
|
237
|
+
if not isinstance(step, dict):
|
|
238
|
+
continue
|
|
239
|
+
run = step.get("run")
|
|
240
|
+
if not run or "|" not in str(run):
|
|
241
|
+
continue
|
|
242
|
+
body = str(run)
|
|
243
|
+
if re.search(r"set\s+[-a-z]*o?\s*[-a-z]*pipefail|set\s+-o\s+pipefail", body):
|
|
244
|
+
continue
|
|
245
|
+
for line in body.splitlines():
|
|
246
|
+
line = line.strip()
|
|
247
|
+
if "|" not in line or line.startswith("#") or "||" in line:
|
|
248
|
+
continue
|
|
249
|
+
tail = line.rsplit("|", 1)[1].strip().split()
|
|
250
|
+
if not tail:
|
|
251
|
+
continue
|
|
252
|
+
if tail[0] in FILTERS and (NAIVE or _upstream_can_fail(line)):
|
|
253
|
+
assigned = (re.match(r"([A-Za-z_][A-Za-z0-9_]*)=", line) or [None, ""])[1]
|
|
254
|
+
if not NAIVE and _result_is_emptiness_checked(assigned, body):
|
|
255
|
+
continue
|
|
256
|
+
label = step.get("name") or line[:40]
|
|
257
|
+
found.append(Finding(
|
|
258
|
+
detector="D3",
|
|
259
|
+
severity=_d3_severity(step, name, job, doc or {}, line),
|
|
260
|
+
job=name,
|
|
261
|
+
title="exit status masked by a pipe",
|
|
262
|
+
detail=(f"step '{label}' in job '{name}' ends a pipeline with "
|
|
263
|
+
f"'{tail[0]}' and does not set pipefail. The step's status is "
|
|
264
|
+
f"{tail[0]}'s, so a failure upstream of the pipe passes."),
|
|
265
|
+
repro=(f"Make the command before '| {tail[0]}' fail. The step still "
|
|
266
|
+
f"succeeds. Add 'set -o pipefail' and it fails correctly."),
|
|
267
|
+
))
|
|
268
|
+
break
|
|
269
|
+
return found
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
# Severity for D4 is decided by TWO questions the workflow file can answer:
|
|
273
|
+
# does this workflow run on pull_request, so the job is a candidate required check, and
|
|
274
|
+
# is the gated job a VERIFICATION job, so skipping it means nothing was checked.
|
|
275
|
+
# Skipping a deploy because no release was cut is intended. Skipping the tests because the
|
|
276
|
+
# path filter crashed is not. Without the branch-protection API this is the honest ceiling,
|
|
277
|
+
# and the tier says which question it could not answer.
|
|
278
|
+
_CHECK_JOB = re.compile(
|
|
279
|
+
r"\b(test|tests|lint|check|checks|verify|typecheck|type-check|coverage|audit|"
|
|
280
|
+
r"security|e2e|unit|integration|spec|validate|ci)\b", re.I)
|
|
281
|
+
_SHIP_JOB = re.compile(
|
|
282
|
+
r"\b(deploy|publish|release|upload|notify|docs|announce|changelog|tag|sign|"
|
|
283
|
+
r"docker|image|artifact)\b", re.I)
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
def _on_pull_request(doc: dict) -> bool:
|
|
287
|
+
on = (doc or {}).get("on") or (doc or {}).get(True)
|
|
288
|
+
if isinstance(on, str):
|
|
289
|
+
return on == "pull_request"
|
|
290
|
+
if isinstance(on, list):
|
|
291
|
+
return "pull_request" in on
|
|
292
|
+
if isinstance(on, dict):
|
|
293
|
+
return "pull_request" in on or "pull_request_target" in on
|
|
294
|
+
return False
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def _gate_severity(job_name: str, job: dict, doc: dict) -> str:
|
|
298
|
+
name = f"{job_name} {job.get('name') or ''}"
|
|
299
|
+
if _SHIP_JOB.search(name) and not _CHECK_JOB.search(name):
|
|
300
|
+
return "LOW" # a release step, where skipping is usually the intent
|
|
301
|
+
if not _on_pull_request(doc):
|
|
302
|
+
return "LOW" # never runs on a PR, so it is not a merge gate
|
|
303
|
+
if _CHECK_JOB.search(name):
|
|
304
|
+
return "HIGH" # a PR verification job that can silently not run
|
|
305
|
+
return "MEDIUM"
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
def d4_outputs_gate_without_result_check(jobs: dict, doc: dict | None = None) -> list[Finding]:
|
|
309
|
+
"""A job gated on an upstream's OUTPUTS, with nothing checking that upstream SUCCEEDED.
|
|
310
|
+
|
|
311
|
+
The common change-detection shape: if: needs.detect.outputs.rust == 'true'
|
|
312
|
+
|
|
313
|
+
That is a deliberate optimisation and is NOT a defect by itself. The defect is what
|
|
314
|
+
happens when `detect` FAILS rather than decides: a failed job sets no outputs, the
|
|
315
|
+
comparison is false, the dependent job SKIPS, and a skipped job reports Success. One
|
|
316
|
+
broken detector silently disables the tests it gates, and the merge is green.
|
|
317
|
+
|
|
318
|
+
A job that also reads needs.<up>.result is doing it correctly and is not reported.
|
|
319
|
+
"""
|
|
320
|
+
found = []
|
|
321
|
+
for name, job in jobs.items():
|
|
322
|
+
if not isinstance(job, dict):
|
|
323
|
+
continue
|
|
324
|
+
cond = str(job.get("if") or "")
|
|
325
|
+
if not cond:
|
|
326
|
+
continue
|
|
327
|
+
ups = set(re.findall(r"needs\.([A-Za-z0-9_\-]+)\.outputs\.", cond))
|
|
328
|
+
if not ups:
|
|
329
|
+
continue
|
|
330
|
+
checked = set(re.findall(r"needs\.([A-Za-z0-9_\-]+)\.result", cond + _steps_text(job)))
|
|
331
|
+
sev = _gate_severity(name, job, doc or {})
|
|
332
|
+
for up in sorted(ups - checked):
|
|
333
|
+
found.append(Finding(
|
|
334
|
+
detector="D4",
|
|
335
|
+
severity=sev,
|
|
336
|
+
job=name,
|
|
337
|
+
title="tests silently disabled if the gating job fails",
|
|
338
|
+
detail=(f"job '{name}' runs only when '{up}' outputs say so, and nothing checks "
|
|
339
|
+
f"needs.{up}.result. If '{up}' FAILS, its outputs are unset, the "
|
|
340
|
+
f"condition is false, '{name}' skips, and a skipped job reports Success."),
|
|
341
|
+
repro=(f"Make '{up}' exit 1. '{name}' does not run, reports Success, and any "
|
|
342
|
+
f"branch protection requiring it passes with nothing tested."),
|
|
343
|
+
))
|
|
344
|
+
return found
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
DETECTORS = (d1_skippable_upstream, d2_fanin_without_result_check,
|
|
348
|
+
d3_pipe_masked_exit, d4_outputs_gate_without_result_check)
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def active_detectors():
|
|
352
|
+
return DETECTORS[:3] if NAIVE else DETECTORS
|
|
353
|
+
|
|
354
|
+
|
|
355
|
+
def scan_workflow(doc: dict) -> list[Finding]:
|
|
356
|
+
jobs = (doc or {}).get("jobs") or {}
|
|
357
|
+
if not isinstance(jobs, dict):
|
|
358
|
+
return []
|
|
359
|
+
out = []
|
|
360
|
+
for fn in active_detectors():
|
|
361
|
+
out.extend(fn(jobs, doc) if fn is not d2_fanin_without_result_check else fn(jobs))
|
|
362
|
+
return out
|