deploy-guard-engine 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- deploy_guard/__init__.py +10 -0
- deploy_guard/__main__.py +4 -0
- deploy_guard/analysis/__init__.py +39 -0
- deploy_guard/analysis/callgraph.py +149 -0
- deploy_guard/analysis/context.py +31 -0
- deploy_guard/analysis/findings.py +333 -0
- deploy_guard/analysis/nullability.py +461 -0
- deploy_guard/analysis/paths.py +219 -0
- deploy_guard/cli.py +190 -0
- deploy_guard/config.py +72 -0
- deploy_guard/engine.py +270 -0
- deploy_guard/explain/__init__.py +16 -0
- deploy_guard/explain/explainer.py +504 -0
- deploy_guard/explain/render.py +60 -0
- deploy_guard/explain/traceback_parse.py +89 -0
- deploy_guard/frontend/__init__.py +10 -0
- deploy_guard/frontend/python_cfg.py +332 -0
- deploy_guard/frontend/python_frontend.py +182 -0
- deploy_guard/generators/__init__.py +11 -0
- deploy_guard/generators/base.py +12 -0
- deploy_guard/generators/import_smoke.py +94 -0
- deploy_guard/ingest/__init__.py +5 -0
- deploy_guard/ingest/discover.py +166 -0
- deploy_guard/ir/__init__.py +24 -0
- deploy_guard/ir/cfg.py +114 -0
- deploy_guard/ir/model.py +128 -0
- deploy_guard/py.typed +0 -0
- deploy_guard/report/__init__.py +5 -0
- deploy_guard/report/render.py +262 -0
- deploy_guard/store.py +50 -0
- deploy_guard_engine-0.1.0.dist-info/METADATA +163 -0
- deploy_guard_engine-0.1.0.dist-info/RECORD +35 -0
- deploy_guard_engine-0.1.0.dist-info/WHEEL +4 -0
- deploy_guard_engine-0.1.0.dist-info/entry_points.txt +3 -0
- deploy_guard_engine-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,461 @@
|
|
|
1
|
+
"""Flow-sensitive nullability analysis.
|
|
2
|
+
|
|
3
|
+
Instead of enumerating every path (which blows up on branchy code), this runs
|
|
4
|
+
a classic forward data-flow fixpoint over the CFG: it propagates, for each
|
|
5
|
+
local variable, whether it is ``None`` / not ``None`` / maybe ``None`` /
|
|
6
|
+
unknown, and *merges* those facts at every branch join. That is linear-ish
|
|
7
|
+
in the size of the function and covers 100% of it regardless of how many
|
|
8
|
+
branches it has.
|
|
9
|
+
|
|
10
|
+
It then flags every attribute access or subscript on a value that can be
|
|
11
|
+
``None`` at that point - the "AttributeError/TypeError in production" class of
|
|
12
|
+
bug - and reconstructs an illustrative branch path to the offending line.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import ast
|
|
18
|
+
from collections import deque
|
|
19
|
+
from dataclasses import dataclass
|
|
20
|
+
from enum import Enum
|
|
21
|
+
|
|
22
|
+
from deploy_guard.ir.cfg import CFG, BasicBlock, EdgeKind
|
|
23
|
+
from deploy_guard.ir.model import FunctionDef
|
|
24
|
+
|
|
25
|
+
# Common stdlib calls whose result is Optional.
|
|
26
|
+
_OPTIONAL_RE_METHODS = {"match", "search", "fullmatch"}
|
|
27
|
+
_BUILTIN_NONNULL = {
|
|
28
|
+
"list", "dict", "set", "tuple", "str", "bytes", "bytearray", "int", "float",
|
|
29
|
+
"bool", "frozenset", "object", "complex",
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class NV(str, Enum):
|
|
34
|
+
BOTTOM = "bottom" # not yet assigned on any path reaching here
|
|
35
|
+
NULL = "null" # definitely None
|
|
36
|
+
NOTNULL = "notnull" # definitely not None
|
|
37
|
+
NULLABLE = "nullable" # known to be None on at least one path
|
|
38
|
+
TOP = "top" # unknown (e.g. result of an unresolved call)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def join(a: NV, b: NV) -> NV:
|
|
42
|
+
if a == b:
|
|
43
|
+
return a
|
|
44
|
+
if a == NV.BOTTOM:
|
|
45
|
+
return b
|
|
46
|
+
if b == NV.BOTTOM:
|
|
47
|
+
return a
|
|
48
|
+
if NV.TOP in (a, b):
|
|
49
|
+
return NV.TOP
|
|
50
|
+
# {NULL, NOTNULL, NULLABLE} in any mix -> NULLABLE
|
|
51
|
+
return NV.NULLABLE
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
@dataclass
|
|
55
|
+
class _State:
|
|
56
|
+
env: dict[str, NV]
|
|
57
|
+
why: dict[str, str]
|
|
58
|
+
|
|
59
|
+
def copy(self) -> _State:
|
|
60
|
+
return _State(dict(self.env), dict(self.why))
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _join_states(states: list[_State]) -> _State:
|
|
64
|
+
out = _State({}, {})
|
|
65
|
+
keys: set[str] = set()
|
|
66
|
+
for s in states:
|
|
67
|
+
keys |= s.env.keys()
|
|
68
|
+
for k in keys:
|
|
69
|
+
vals = [s.env.get(k, NV.BOTTOM) for s in states]
|
|
70
|
+
merged = vals[0]
|
|
71
|
+
for v in vals[1:]:
|
|
72
|
+
merged = join(merged, v)
|
|
73
|
+
out.env[k] = merged
|
|
74
|
+
if merged in (NV.NULL, NV.NULLABLE):
|
|
75
|
+
reasons = {s.why.get(k) for s in states if s.why.get(k)}
|
|
76
|
+
out.why[k] = (
|
|
77
|
+
next(iter(reasons)) if len(reasons) == 1
|
|
78
|
+
else "None on at least one branch reaching this point"
|
|
79
|
+
)
|
|
80
|
+
return out
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
class NullabilityAnalysis:
|
|
84
|
+
def __init__(self, fn: FunctionDef, cfg: CFG, ctx=None) -> None:
|
|
85
|
+
self.fn = fn
|
|
86
|
+
self.cfg = cfg
|
|
87
|
+
self.ctx = ctx # optional AnalysisContext for interprocedural nullability
|
|
88
|
+
self.file = str(fn.span.file)
|
|
89
|
+
self._deref_hits: list[tuple[int, int, str, NV, str]] = []
|
|
90
|
+
# (block_id, lineno, varname, state, reason)
|
|
91
|
+
|
|
92
|
+
# -- public -----------------------------------------------------------
|
|
93
|
+
|
|
94
|
+
def find_none_derefs(self) -> list[NullFinding]:
|
|
95
|
+
self._run()
|
|
96
|
+
out: list[NullFinding] = []
|
|
97
|
+
seen: set[tuple[str, int]] = set()
|
|
98
|
+
for block_id, lineno, var, state, reason in self._deref_hits:
|
|
99
|
+
key = (var, lineno)
|
|
100
|
+
if key in seen:
|
|
101
|
+
continue
|
|
102
|
+
seen.add(key)
|
|
103
|
+
out.append(
|
|
104
|
+
NullFinding(
|
|
105
|
+
var=var,
|
|
106
|
+
lineno=lineno,
|
|
107
|
+
definite=(state == NV.NULL),
|
|
108
|
+
reason=reason,
|
|
109
|
+
witness=_witness_conditions(self.cfg, block_id),
|
|
110
|
+
)
|
|
111
|
+
)
|
|
112
|
+
return out
|
|
113
|
+
|
|
114
|
+
# -- fixpoint -------------------------------------------------------
|
|
115
|
+
|
|
116
|
+
def _run(self) -> None:
|
|
117
|
+
cfg = self.cfg
|
|
118
|
+
in_state: dict[int, _State] = {}
|
|
119
|
+
out_state: dict[int, _State] = {}
|
|
120
|
+
entry_state = self._initial_state()
|
|
121
|
+
|
|
122
|
+
work: deque[int] = deque([cfg.entry])
|
|
123
|
+
in_state[cfg.entry] = entry_state
|
|
124
|
+
cap = max(50, len(cfg.blocks) * 8)
|
|
125
|
+
steps = 0
|
|
126
|
+
|
|
127
|
+
while work and steps < cap:
|
|
128
|
+
steps += 1
|
|
129
|
+
bid = work.popleft()
|
|
130
|
+
block = cfg.blocks[bid]
|
|
131
|
+
|
|
132
|
+
preds = [
|
|
133
|
+
(pid, e)
|
|
134
|
+
for pid in block.pred
|
|
135
|
+
for e in cfg.blocks[pid].succ
|
|
136
|
+
if e.target == bid and pid in out_state
|
|
137
|
+
]
|
|
138
|
+
if bid == cfg.entry:
|
|
139
|
+
cur_in = entry_state.copy()
|
|
140
|
+
elif preds:
|
|
141
|
+
cur_in = _join_states(
|
|
142
|
+
[self._refine_for_edge(out_state[pid], e) for pid, e in preds]
|
|
143
|
+
)
|
|
144
|
+
else:
|
|
145
|
+
cur_in = in_state.get(bid, _State({}, {})).copy()
|
|
146
|
+
|
|
147
|
+
in_state[bid] = cur_in
|
|
148
|
+
new_out = self._transfer_block(block, cur_in.copy(), record=True)
|
|
149
|
+
|
|
150
|
+
if bid not in out_state or new_out.env != out_state[bid].env:
|
|
151
|
+
out_state[bid] = new_out
|
|
152
|
+
for e in block.succ:
|
|
153
|
+
if e.target in cfg.blocks:
|
|
154
|
+
work.append(e.target)
|
|
155
|
+
|
|
156
|
+
# Final pass: re-record derefs against the converged in-states only
|
|
157
|
+
# (the loop above may have recorded against pre-fixpoint states).
|
|
158
|
+
self._deref_hits.clear()
|
|
159
|
+
for bid, block in cfg.blocks.items():
|
|
160
|
+
if bid in in_state:
|
|
161
|
+
self._transfer_block(block, in_state[bid].copy(), record=True)
|
|
162
|
+
|
|
163
|
+
# -- initial / transfer ------------------------------------------
|
|
164
|
+
|
|
165
|
+
def _initial_state(self) -> _State:
|
|
166
|
+
st = _State({}, {})
|
|
167
|
+
for p in self.fn.params:
|
|
168
|
+
if p.name in ("self", "cls"):
|
|
169
|
+
st.env[p.name] = NV.NOTNULL
|
|
170
|
+
elif p.annotation and _ann_is_optional(p.annotation):
|
|
171
|
+
st.env[p.name] = NV.NULLABLE
|
|
172
|
+
st.why[p.name] = f"parameter '{p.name}' is declared Optional"
|
|
173
|
+
elif p.default == "None":
|
|
174
|
+
st.env[p.name] = NV.NULLABLE
|
|
175
|
+
st.why[p.name] = f"parameter '{p.name}' defaults to None"
|
|
176
|
+
elif p.annotation:
|
|
177
|
+
st.env[p.name] = NV.NOTNULL
|
|
178
|
+
else:
|
|
179
|
+
st.env[p.name] = NV.TOP
|
|
180
|
+
return st
|
|
181
|
+
|
|
182
|
+
def _transfer_block(self, block: BasicBlock, state: _State, *, record: bool) -> _State:
|
|
183
|
+
for stmt in block.statements:
|
|
184
|
+
node = stmt.node
|
|
185
|
+
if record:
|
|
186
|
+
self._scan_derefs(node, state, block.id)
|
|
187
|
+
self._apply_stmt(node, state)
|
|
188
|
+
return state
|
|
189
|
+
|
|
190
|
+
def _apply_stmt(self, node: ast.AST, state: _State) -> None:
|
|
191
|
+
if isinstance(node, ast.Assign):
|
|
192
|
+
nv, why = self._eval(node.value, state)
|
|
193
|
+
for tgt in node.targets:
|
|
194
|
+
self._bind(tgt, nv, why, state)
|
|
195
|
+
elif isinstance(node, ast.AnnAssign) and node.value is not None:
|
|
196
|
+
nv, why = self._eval(node.value, state)
|
|
197
|
+
self._bind(node.target, nv, why, state)
|
|
198
|
+
elif isinstance(node, ast.AugAssign):
|
|
199
|
+
# x += ... : x was used, so treat as not-None afterwards.
|
|
200
|
+
self._bind(node.target, NV.NOTNULL, "", state)
|
|
201
|
+
elif isinstance(node, ast.Assert):
|
|
202
|
+
self._refine(node.test, True, state)
|
|
203
|
+
elif isinstance(node, ast.NamedExpr): # walrus at statement level (rare)
|
|
204
|
+
nv, why = self._eval(node.value, state)
|
|
205
|
+
self._bind(node.target, nv, why, state)
|
|
206
|
+
# imports, expr-statements, pass, return, raise: no local rebinding we track
|
|
207
|
+
|
|
208
|
+
def _bind(self, target: ast.AST, nv: NV, why: str, state: _State) -> None:
|
|
209
|
+
if isinstance(target, ast.Name):
|
|
210
|
+
state.env[target.id] = nv
|
|
211
|
+
if why and nv in (NV.NULL, NV.NULLABLE):
|
|
212
|
+
state.why[target.id] = why
|
|
213
|
+
else:
|
|
214
|
+
state.why.pop(target.id, None)
|
|
215
|
+
elif isinstance(target, (ast.Tuple, ast.List)):
|
|
216
|
+
for elt in target.elts:
|
|
217
|
+
inner = elt.value if isinstance(elt, ast.Starred) else elt
|
|
218
|
+
self._bind(inner, NV.TOP, "", state)
|
|
219
|
+
# Attribute / Subscript targets: not tracked (we only model locals)
|
|
220
|
+
|
|
221
|
+
# -- expression nullability ------------------------------------
|
|
222
|
+
|
|
223
|
+
def _eval(self, node: ast.AST | None, state: _State) -> tuple[NV, str]:
|
|
224
|
+
if node is None:
|
|
225
|
+
return NV.NULL, "explicit None"
|
|
226
|
+
if isinstance(node, ast.Constant):
|
|
227
|
+
return (NV.NULL, "explicit None") if node.value is None else (NV.NOTNULL, "")
|
|
228
|
+
if isinstance(node, ast.Name):
|
|
229
|
+
return state.env.get(node.id, NV.TOP), state.why.get(node.id, "")
|
|
230
|
+
if isinstance(node, ast.NamedExpr):
|
|
231
|
+
nv, why = self._eval(node.value, state)
|
|
232
|
+
self._bind(node.target, nv, why, state)
|
|
233
|
+
return nv, why
|
|
234
|
+
if isinstance(node, (ast.List, ast.Dict, ast.Set, ast.Tuple, ast.ListComp,
|
|
235
|
+
ast.DictComp, ast.SetComp, ast.GeneratorExp, ast.JoinedStr,
|
|
236
|
+
ast.Lambda, ast.FormattedValue)):
|
|
237
|
+
return NV.NOTNULL, ""
|
|
238
|
+
if isinstance(node, ast.IfExp):
|
|
239
|
+
a, wa = self._eval(node.body, state)
|
|
240
|
+
b, wb = self._eval(node.orelse, state)
|
|
241
|
+
merged = join(a, b)
|
|
242
|
+
if merged in (NV.NULL, NV.NULLABLE):
|
|
243
|
+
return NV.NULLABLE, "one arm of the conditional expression is None"
|
|
244
|
+
return merged, ""
|
|
245
|
+
if isinstance(node, ast.BoolOp):
|
|
246
|
+
# `a or b` / `a and b` - result is one of the operands.
|
|
247
|
+
vals = [self._eval(v, state) for v in node.values]
|
|
248
|
+
merged = vals[0][0]
|
|
249
|
+
for v, _ in vals[1:]:
|
|
250
|
+
merged = join(merged, v)
|
|
251
|
+
return (NV.NULLABLE, "value may come from a None operand") if merged in (
|
|
252
|
+
NV.NULL, NV.NULLABLE
|
|
253
|
+
) else (NV.TOP, "")
|
|
254
|
+
if isinstance(node, ast.Call):
|
|
255
|
+
return self._eval_call(node, state)
|
|
256
|
+
if isinstance(node, ast.Await):
|
|
257
|
+
return NV.TOP, ""
|
|
258
|
+
return NV.TOP, ""
|
|
259
|
+
|
|
260
|
+
def _eval_call(self, node: ast.Call, state: _State) -> tuple[NV, str]:
|
|
261
|
+
func = node.func
|
|
262
|
+
if isinstance(func, ast.Name):
|
|
263
|
+
if func.id in _BUILTIN_NONNULL or (func.id[:1].isupper()):
|
|
264
|
+
return NV.NOTNULL, ""
|
|
265
|
+
qn = self._callee_can_be_none(func.id)
|
|
266
|
+
if qn is not None:
|
|
267
|
+
return NV.NULLABLE, f"`{qn.rsplit('.', 1)[-1]}()` can return None"
|
|
268
|
+
return NV.TOP, ""
|
|
269
|
+
if isinstance(func, ast.Attribute):
|
|
270
|
+
# dict.get(k) / os.environ.get(k) with no or None default -> Optional
|
|
271
|
+
if func.attr == "get" and len(node.args) <= 1 and not node.keywords:
|
|
272
|
+
return NV.NULLABLE, "result of .get() with no default"
|
|
273
|
+
if func.attr == "get" and len(node.args) == 2:
|
|
274
|
+
nv, why = self._eval(node.args[1], state)
|
|
275
|
+
if nv == NV.NULL:
|
|
276
|
+
return NV.NULL, "the default passed to .get() is None"
|
|
277
|
+
return nv, why
|
|
278
|
+
if (
|
|
279
|
+
isinstance(func.value, ast.Name)
|
|
280
|
+
and func.value.id == "re"
|
|
281
|
+
and func.attr in _OPTIONAL_RE_METHODS
|
|
282
|
+
):
|
|
283
|
+
return NV.NULLABLE, f"result of re.{func.attr}() can be None"
|
|
284
|
+
# self.method() / module.func() resolving to a project function
|
|
285
|
+
dotted = (
|
|
286
|
+
f"{func.value.id}.{func.attr}"
|
|
287
|
+
if isinstance(func.value, ast.Name)
|
|
288
|
+
else func.attr
|
|
289
|
+
)
|
|
290
|
+
qn = self._callee_can_be_none(dotted)
|
|
291
|
+
if qn is not None:
|
|
292
|
+
return NV.NULLABLE, f"`{qn.rsplit('.', 1)[-1]}()` can return None"
|
|
293
|
+
return NV.TOP, ""
|
|
294
|
+
|
|
295
|
+
def _callee_can_be_none(self, name: str) -> str | None:
|
|
296
|
+
if self.ctx is None:
|
|
297
|
+
return None
|
|
298
|
+
try:
|
|
299
|
+
return self.ctx.callee_can_be_none(name)
|
|
300
|
+
except Exception: # pragma: no cover - never let this break analysis
|
|
301
|
+
return None
|
|
302
|
+
|
|
303
|
+
# -- guard refinement ------------------------------------------
|
|
304
|
+
|
|
305
|
+
def _refine_for_edge(self, state: _State, edge) -> _State:
|
|
306
|
+
if edge.guard is None or edge.kind not in (EdgeKind.TRUE, EdgeKind.FALSE):
|
|
307
|
+
return state
|
|
308
|
+
new = state.copy()
|
|
309
|
+
self._refine(edge.guard, not edge.negated, new)
|
|
310
|
+
return new
|
|
311
|
+
|
|
312
|
+
def _refine(self, test: ast.AST, positive: bool, state: _State) -> None:
|
|
313
|
+
if isinstance(test, ast.UnaryOp) and isinstance(test.op, ast.Not):
|
|
314
|
+
self._refine(test.operand, not positive, state)
|
|
315
|
+
return
|
|
316
|
+
if isinstance(test, ast.BoolOp) and isinstance(test.op, ast.And) and positive:
|
|
317
|
+
for v in test.values:
|
|
318
|
+
self._refine(v, True, state)
|
|
319
|
+
return
|
|
320
|
+
if isinstance(test, ast.BoolOp) and isinstance(test.op, ast.Or) and not positive:
|
|
321
|
+
for v in test.values:
|
|
322
|
+
self._refine(v, False, state)
|
|
323
|
+
return
|
|
324
|
+
if isinstance(test, ast.Compare) and len(test.ops) == 1:
|
|
325
|
+
left, op, right = test.left, test.ops[0], test.comparators[0]
|
|
326
|
+
name = _as_name(left) or _as_name(right)
|
|
327
|
+
other = right if _as_name(left) else left
|
|
328
|
+
is_none_cmp = isinstance(other, ast.Constant) and other.value is None
|
|
329
|
+
if name and is_none_cmp and isinstance(op, (ast.Is, ast.Eq)):
|
|
330
|
+
_set(state, name, NV.NULL if positive else NV.NOTNULL)
|
|
331
|
+
return
|
|
332
|
+
if name and is_none_cmp and isinstance(op, (ast.IsNot, ast.NotEq)):
|
|
333
|
+
_set(state, name, NV.NOTNULL if positive else NV.NULL)
|
|
334
|
+
return
|
|
335
|
+
if isinstance(test, ast.Call) and isinstance(test.func, ast.Name) \
|
|
336
|
+
and test.func.id == "isinstance" and test.args:
|
|
337
|
+
name = _as_name(test.args[0])
|
|
338
|
+
if name and positive:
|
|
339
|
+
_set(state, name, NV.NOTNULL)
|
|
340
|
+
return
|
|
341
|
+
name = _as_name(test)
|
|
342
|
+
if name and positive:
|
|
343
|
+
_set(state, name, NV.NOTNULL)
|
|
344
|
+
|
|
345
|
+
# -- deref scanning ------------------------------------------
|
|
346
|
+
|
|
347
|
+
def _scan_derefs(self, node: ast.AST, state: _State, block_id: int) -> None:
|
|
348
|
+
"""Report `x.attr` / `x[...]` where x can be None, honouring the
|
|
349
|
+
short-circuit narrowing of `and` / `or` and the arms of `a if c else b`
|
|
350
|
+
so that `x is None or x.attr` is *not* flagged."""
|
|
351
|
+
self._scan_expr(node, state, block_id)
|
|
352
|
+
|
|
353
|
+
def _scan_expr(self, node: ast.AST, state: _State, block_id: int) -> None:
|
|
354
|
+
if isinstance(node, ast.BoolOp):
|
|
355
|
+
# `a or b`: b runs only when a was falsy; `a and b`: b when a truthy.
|
|
356
|
+
narrowed = state.copy()
|
|
357
|
+
positive = isinstance(node.op, ast.And)
|
|
358
|
+
for operand in node.values:
|
|
359
|
+
self._scan_expr(operand, narrowed, block_id)
|
|
360
|
+
self._refine(operand, positive, narrowed)
|
|
361
|
+
return
|
|
362
|
+
if isinstance(node, ast.IfExp):
|
|
363
|
+
self._scan_expr(node.test, state, block_id)
|
|
364
|
+
t = state.copy()
|
|
365
|
+
self._refine(node.test, True, t)
|
|
366
|
+
self._scan_expr(node.body, t, block_id)
|
|
367
|
+
f = state.copy()
|
|
368
|
+
self._refine(node.test, False, f)
|
|
369
|
+
self._scan_expr(node.orelse, f, block_id)
|
|
370
|
+
return
|
|
371
|
+
|
|
372
|
+
target = None
|
|
373
|
+
if isinstance(node, ast.Attribute) or isinstance(node, ast.Subscript):
|
|
374
|
+
target = node.value
|
|
375
|
+
if isinstance(target, ast.Name):
|
|
376
|
+
nv = state.env.get(target.id, NV.TOP)
|
|
377
|
+
if nv in (NV.NULL, NV.NULLABLE):
|
|
378
|
+
self._deref_hits.append(
|
|
379
|
+
(
|
|
380
|
+
block_id,
|
|
381
|
+
getattr(node, "lineno", 0),
|
|
382
|
+
target.id,
|
|
383
|
+
nv,
|
|
384
|
+
state.why.get(target.id, "may be None here"),
|
|
385
|
+
)
|
|
386
|
+
)
|
|
387
|
+
# avoid a cascade of hits on the same var in the same statement
|
|
388
|
+
state.env[target.id] = NV.NOTNULL
|
|
389
|
+
|
|
390
|
+
for child in ast.iter_child_nodes(node):
|
|
391
|
+
if isinstance(child, ast.stmt) or isinstance(child, _SCOPE_NODES):
|
|
392
|
+
continue
|
|
393
|
+
self._scan_expr(child, state, block_id)
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
@dataclass
|
|
397
|
+
class NullFinding:
|
|
398
|
+
var: str
|
|
399
|
+
lineno: int
|
|
400
|
+
definite: bool
|
|
401
|
+
reason: str
|
|
402
|
+
witness: list[str]
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
# --------------------------------------------------------------------------
|
|
406
|
+
# helpers
|
|
407
|
+
# --------------------------------------------------------------------------
|
|
408
|
+
|
|
409
|
+
# Never descend into a nested statement body or a nested function/class/lambda
|
|
410
|
+
# scope while scanning one statement's own expressions.
|
|
411
|
+
_SCOPE_NODES = (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef, ast.Lambda)
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
def _as_name(node: ast.AST) -> str | None:
|
|
415
|
+
return node.id if isinstance(node, ast.Name) else None
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
def _set(state: _State, name: str, nv: NV) -> None:
|
|
419
|
+
state.env[name] = nv
|
|
420
|
+
if nv == NV.NOTNULL:
|
|
421
|
+
state.why.pop(name, None)
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
def _ann_is_optional(ann: str) -> bool:
|
|
425
|
+
text = ann.replace(" ", "")
|
|
426
|
+
return (
|
|
427
|
+
"Optional[" in text
|
|
428
|
+
or "|None" in text
|
|
429
|
+
or "None|" in text
|
|
430
|
+
or ("Union[" in text and "None" in text)
|
|
431
|
+
)
|
|
432
|
+
|
|
433
|
+
|
|
434
|
+
def _witness_conditions(cfg: CFG, target_block: int, limit: int = 12) -> list[str]:
|
|
435
|
+
"""A shortest branch path from entry to ``target_block``, as guard strings."""
|
|
436
|
+
from collections import deque as _dq
|
|
437
|
+
|
|
438
|
+
prev: dict[int, tuple[int, object]] = {}
|
|
439
|
+
q: _dq[int] = _dq([cfg.entry])
|
|
440
|
+
seen = {cfg.entry}
|
|
441
|
+
while q:
|
|
442
|
+
bid = q.popleft()
|
|
443
|
+
if bid == target_block:
|
|
444
|
+
break
|
|
445
|
+
for e in cfg.blocks[bid].succ:
|
|
446
|
+
if e.target in cfg.blocks and e.target not in seen:
|
|
447
|
+
seen.add(e.target)
|
|
448
|
+
prev[e.target] = (bid, e)
|
|
449
|
+
q.append(e.target)
|
|
450
|
+
|
|
451
|
+
if target_block not in prev and target_block != cfg.entry:
|
|
452
|
+
return []
|
|
453
|
+
chain: list[str] = []
|
|
454
|
+
cur = target_block
|
|
455
|
+
while cur in prev:
|
|
456
|
+
pid, e = prev[cur]
|
|
457
|
+
if getattr(e, "guard_src", None) and e.kind in (EdgeKind.TRUE, EdgeKind.FALSE):
|
|
458
|
+
cond = e.guard_src if not e.negated else f"not ({e.guard_src})"
|
|
459
|
+
chain.append(cond)
|
|
460
|
+
cur = pid
|
|
461
|
+
return list(reversed(chain))[:limit]
|
|
@@ -0,0 +1,219 @@
|
|
|
1
|
+
"""Enumerate execution paths through a CFG and fold them into a behavior spec.
|
|
2
|
+
|
|
3
|
+
An :class:`ExecPath` is one route from function entry to a terminal point,
|
|
4
|
+
carrying the branch conditions taken to get there and what the function does
|
|
5
|
+
at the end (returns a value, returns ``None`` implicitly, or raises).
|
|
6
|
+
|
|
7
|
+
Bounds keep this finite:
|
|
8
|
+
|
|
9
|
+
* ``unroll`` - a block may appear at most this many times on one path, so
|
|
10
|
+
loops are explored for 0 and 1 iterations by default.
|
|
11
|
+
* ``limit`` - a hard cap on the number of paths; hitting it sets
|
|
12
|
+
``PathSet.truncated``.
|
|
13
|
+
|
|
14
|
+
Exception edges are ignored in M1 (explicit ``raise`` still shows up via the
|
|
15
|
+
block terminator). Exception-flow sensitivity arrives with abstract
|
|
16
|
+
interpretation in M2.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import ast
|
|
22
|
+
from dataclasses import dataclass, field
|
|
23
|
+
|
|
24
|
+
from deploy_guard.ir.cfg import CFG, EdgeKind, Terminator
|
|
25
|
+
from deploy_guard.ir.model import FunctionDef
|
|
26
|
+
|
|
27
|
+
DEFAULT_LIMIT = 2000
|
|
28
|
+
DEFAULT_UNROLL = 2
|
|
29
|
+
|
|
30
|
+
_TERMINAL = {Terminator.RETURN, Terminator.RAISE, Terminator.IMPLICIT_RETURN}
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass
|
|
34
|
+
class ExecPath:
|
|
35
|
+
block_ids: list[int]
|
|
36
|
+
conditions: list[str]
|
|
37
|
+
outcome: str # returns | implicit-none | raises | truncated
|
|
38
|
+
value: str | None = None # return value / raised expr, as source text
|
|
39
|
+
lineno: int | None = None
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class PathSet:
|
|
44
|
+
qualname: str
|
|
45
|
+
paths: list[ExecPath] = field(default_factory=list)
|
|
46
|
+
truncated: bool = False # hit the hard path cap - spec is partial
|
|
47
|
+
loop_bounded: bool = False # a loop was explored only to unroll depth
|
|
48
|
+
notes: list[str] = field(default_factory=list)
|
|
49
|
+
|
|
50
|
+
@property
|
|
51
|
+
def outcomes(self) -> set[str]:
|
|
52
|
+
return {p.outcome for p in self.paths}
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _dedupe_consecutive(items: list[str]) -> list[str]:
|
|
56
|
+
out: list[str] = []
|
|
57
|
+
for it in items:
|
|
58
|
+
if not out or out[-1] != it:
|
|
59
|
+
out.append(it)
|
|
60
|
+
return out
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _terminal_outcome(term: Terminator) -> str:
|
|
64
|
+
if term == Terminator.RETURN:
|
|
65
|
+
return "returns"
|
|
66
|
+
if term == Terminator.RAISE:
|
|
67
|
+
return "raises"
|
|
68
|
+
return "implicit-none"
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _terminal_value(term: Terminator, node: ast.AST | None) -> str | None:
|
|
72
|
+
if term == Terminator.IMPLICIT_RETURN:
|
|
73
|
+
return "None"
|
|
74
|
+
if isinstance(node, ast.Return):
|
|
75
|
+
return ast.unparse(node.value) if node.value is not None else "None"
|
|
76
|
+
if isinstance(node, ast.Raise):
|
|
77
|
+
if node.exc is not None:
|
|
78
|
+
return ast.unparse(node.exc)
|
|
79
|
+
return "<re-raise>"
|
|
80
|
+
return None
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def enumerate_paths(
|
|
84
|
+
cfg: CFG,
|
|
85
|
+
*,
|
|
86
|
+
limit: int = DEFAULT_LIMIT,
|
|
87
|
+
unroll: int = DEFAULT_UNROLL,
|
|
88
|
+
) -> PathSet:
|
|
89
|
+
result = PathSet(qualname=cfg.func_qualname)
|
|
90
|
+
|
|
91
|
+
def record(blocks: list[int], conds: list[str], outcome: str,
|
|
92
|
+
value: str | None, lineno: int | None) -> None:
|
|
93
|
+
result.paths.append(
|
|
94
|
+
ExecPath(
|
|
95
|
+
block_ids=blocks,
|
|
96
|
+
conditions=_dedupe_consecutive(conds),
|
|
97
|
+
outcome=outcome,
|
|
98
|
+
value=value,
|
|
99
|
+
lineno=lineno,
|
|
100
|
+
)
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
def visit(bid: int, trail: list[int], counts: dict[int, int], conds: list[str]) -> None:
|
|
104
|
+
if len(result.paths) >= limit:
|
|
105
|
+
result.truncated = True
|
|
106
|
+
return
|
|
107
|
+
|
|
108
|
+
block = cfg.blocks[bid]
|
|
109
|
+
trail = [*trail, bid]
|
|
110
|
+
counts = {**counts, bid: counts.get(bid, 0) + 1}
|
|
111
|
+
|
|
112
|
+
if block.terminator in _TERMINAL:
|
|
113
|
+
record(
|
|
114
|
+
trail,
|
|
115
|
+
conds,
|
|
116
|
+
_terminal_outcome(block.terminator),
|
|
117
|
+
_terminal_value(block.terminator, block.term_node),
|
|
118
|
+
getattr(block.term_node, "lineno", block.first_line),
|
|
119
|
+
)
|
|
120
|
+
return
|
|
121
|
+
|
|
122
|
+
succ = [
|
|
123
|
+
e
|
|
124
|
+
for e in block.succ
|
|
125
|
+
if e.kind != EdgeKind.EXCEPTION and e.target in cfg.blocks
|
|
126
|
+
]
|
|
127
|
+
if not succ:
|
|
128
|
+
# A dead end with no terminator: e.g. break/continue whose loop
|
|
129
|
+
# target was outside a modelled region, or the synthetic exit.
|
|
130
|
+
outcome = "implicit-none" if bid == cfg.exit else "falls-through"
|
|
131
|
+
record(trail, conds, outcome, "None" if outcome == "implicit-none" else None,
|
|
132
|
+
block.first_line)
|
|
133
|
+
return
|
|
134
|
+
|
|
135
|
+
for edge in succ:
|
|
136
|
+
if counts.get(edge.target, 0) >= unroll:
|
|
137
|
+
# A loop back-edge explored to its unroll depth: stop following
|
|
138
|
+
# it, but the loop-exit edge still yields a complete path, so
|
|
139
|
+
# this is normal bounding, not truncation.
|
|
140
|
+
result.loop_bounded = True
|
|
141
|
+
continue
|
|
142
|
+
next_conds = conds
|
|
143
|
+
if edge.guard_src is not None and edge.kind in (
|
|
144
|
+
EdgeKind.TRUE, EdgeKind.FALSE,
|
|
145
|
+
):
|
|
146
|
+
cond = edge.guard_src if not edge.negated else f"not ({edge.guard_src})"
|
|
147
|
+
next_conds = [*conds, cond]
|
|
148
|
+
visit(edge.target, trail, counts, next_conds)
|
|
149
|
+
|
|
150
|
+
visit(cfg.entry, [], {}, [])
|
|
151
|
+
if result.truncated:
|
|
152
|
+
result.notes.append(
|
|
153
|
+
f"path enumeration hit the {limit}-path cap; behavior spec is partial"
|
|
154
|
+
)
|
|
155
|
+
elif result.loop_bounded:
|
|
156
|
+
result.notes.append(f"loops explored up to {unroll - 1} iteration(s)")
|
|
157
|
+
return result
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
# --------------------------------------------------------------------------
|
|
161
|
+
# Behavior spec: the human-facing view of the path set.
|
|
162
|
+
# --------------------------------------------------------------------------
|
|
163
|
+
|
|
164
|
+
@dataclass
|
|
165
|
+
class BehaviorEntry:
|
|
166
|
+
when: list[str]
|
|
167
|
+
outcome: str
|
|
168
|
+
value: str | None
|
|
169
|
+
lineno: int | None
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
@dataclass
|
|
173
|
+
class BehaviorSpec:
|
|
174
|
+
qualname: str
|
|
175
|
+
entries: list[BehaviorEntry] = field(default_factory=list)
|
|
176
|
+
truncated: bool = False
|
|
177
|
+
notes: list[str] = field(default_factory=list)
|
|
178
|
+
|
|
179
|
+
@property
|
|
180
|
+
def returns_value(self) -> bool:
|
|
181
|
+
return any(
|
|
182
|
+
e.outcome == "returns" and (e.value or "None") != "None" for e in self.entries
|
|
183
|
+
)
|
|
184
|
+
|
|
185
|
+
@property
|
|
186
|
+
def returns_none_implicitly(self) -> bool:
|
|
187
|
+
return any(
|
|
188
|
+
e.outcome == "implicit-none"
|
|
189
|
+
or (e.outcome == "returns" and (e.value or "None") == "None")
|
|
190
|
+
for e in self.entries
|
|
191
|
+
)
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def behavior_spec(
|
|
195
|
+
fn: FunctionDef,
|
|
196
|
+
cfg: CFG,
|
|
197
|
+
*,
|
|
198
|
+
limit: int = DEFAULT_LIMIT,
|
|
199
|
+
unroll: int = DEFAULT_UNROLL,
|
|
200
|
+
) -> BehaviorSpec:
|
|
201
|
+
pset = enumerate_paths(cfg, limit=limit, unroll=unroll)
|
|
202
|
+
spec = BehaviorSpec(
|
|
203
|
+
qualname=fn.qualname, truncated=pset.truncated, notes=list(pset.notes)
|
|
204
|
+
)
|
|
205
|
+
seen: set[tuple] = set()
|
|
206
|
+
for path in pset.paths:
|
|
207
|
+
key = (tuple(path.conditions), path.outcome, path.value)
|
|
208
|
+
if key in seen:
|
|
209
|
+
continue
|
|
210
|
+
seen.add(key)
|
|
211
|
+
spec.entries.append(
|
|
212
|
+
BehaviorEntry(
|
|
213
|
+
when=path.conditions,
|
|
214
|
+
outcome=path.outcome,
|
|
215
|
+
value=path.value,
|
|
216
|
+
lineno=path.lineno,
|
|
217
|
+
)
|
|
218
|
+
)
|
|
219
|
+
return spec
|