diffgenome 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. diffgenome/__init__.py +7 -0
  2. diffgenome/__main__.py +240 -0
  3. diffgenome/_collectors/go/dg/dg.go +623 -0
  4. diffgenome/_collectors/go/go.mod +3 -0
  5. diffgenome/_collectors/go/instrument/facts.go +346 -0
  6. diffgenome/_collectors/go/instrument/main.go +484 -0
  7. diffgenome/_collectors/node/instrument.js +289 -0
  8. diffgenome/_collectors/node/jest-setup.js +40 -0
  9. diffgenome/_collectors/node/package-lock.json +35 -0
  10. diffgenome/_collectors/node/package.json +11 -0
  11. diffgenome/_collectors/node/runtime.js +426 -0
  12. diffgenome/ambiguity.py +122 -0
  13. diffgenome/api.py +67 -0
  14. diffgenome/change.py +86 -0
  15. diffgenome/change_artifact.py +310 -0
  16. diffgenome/collect/__init__.py +2 -0
  17. diffgenome/collect/go_test.py +271 -0
  18. diffgenome/collect/node_jest.py +319 -0
  19. diffgenome/collect/py_monitoring.py +985 -0
  20. diffgenome/collect/py_runtime.py +116 -0
  21. diffgenome/collect/py_symbols.py +238 -0
  22. diffgenome/collect/pytest_plugin.py +130 -0
  23. diffgenome/compose.py +469 -0
  24. diffgenome/dependence.py +264 -0
  25. diffgenome/evaluate.py +669 -0
  26. diffgenome/frontends/__init__.py +0 -0
  27. diffgenome/frontends/python_ir.py +335 -0
  28. diffgenome/genome.py +1016 -0
  29. diffgenome/genome_pipeline.py +674 -0
  30. diffgenome/genome_prompt.py +33 -0
  31. diffgenome/genome_state.py +2118 -0
  32. diffgenome/graph.py +426 -0
  33. diffgenome/llm.py +189 -0
  34. diffgenome/model.py +364 -0
  35. diffgenome/mvp.py +398 -0
  36. diffgenome/probe.py +509 -0
  37. diffgenome/projection.py +308 -0
  38. diffgenome/py.typed +0 -0
  39. diffgenome/render.py +118 -0
  40. diffgenome/report.py +363 -0
  41. diffgenome/resolve.py +37 -0
  42. diffgenome/runtime.py +74 -0
  43. diffgenome/runtime_evidence.py +261 -0
  44. diffgenome/sandbox.py +166 -0
  45. diffgenome/serialize.py +96 -0
  46. diffgenome/sites.py +19 -0
  47. diffgenome/static_types.py +69 -0
  48. diffgenome/structure.py +462 -0
  49. diffgenome-0.1.0.dist-info/METADATA +139 -0
  50. diffgenome-0.1.0.dist-info/RECORD +53 -0
  51. diffgenome-0.1.0.dist-info/WHEEL +4 -0
  52. diffgenome-0.1.0.dist-info/entry_points.txt +2 -0
  53. diffgenome-0.1.0.dist-info/licenses/LICENSE +202 -0
@@ -0,0 +1,2118 @@
1
+ """Genome checking and prediction over the stronger substrate (experiment 09).
2
+
3
+ Adds to `diffgenome.genome`:
4
+ - `Mechanics`: deterministic intra-procedural facts (decision sites, local def-use, control
5
+ requirements, field stores) from `diffgenome.dependence`, keyed by site id;
6
+ - `Observations`: per execution, observed branch outcomes at sites and bucketed state on
7
+ entry and exit (observed state deltas);
8
+ - `establish_state()`: status for decisions anchored at sites, transitions, procedures, and
9
+ the new evidence kinds (branch / control / dataflow / delta / store). The checker remains
10
+ the only authority on status;
11
+ - path conditions: an execution's branch vector over genome sites, with the site's source
12
+ predicate attached;
13
+ - `predict_sequence()`: state-dependent prediction over a sequence of calls, where
14
+ transitions change state and later decisions read it. Anything unestablished that a
15
+ prediction needs stops it as indeterminate; nothing is skipped.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import hashlib
21
+ import itertools
22
+ import re
23
+ from dataclasses import dataclass, field
24
+ from typing import Any
25
+
26
+ from diffgenome.genome import (
27
+ HYPOTHESIS,
28
+ OBSERVED,
29
+ REJECTED,
30
+ STATIC,
31
+ SUPPORTED,
32
+ VERIFIED,
33
+ Decision,
34
+ EvidenceRef,
35
+ Genome,
36
+ Item,
37
+ Substrate,
38
+ Variable,
39
+ bindings_of,
40
+ derive,
41
+ eval_predicate,
42
+ )
43
+ from diffgenome.model import CallNode, Execution
44
+ from diffgenome.structure import (
45
+ Skeleton,
46
+ build_skeleton,
47
+ occurrences,
48
+ placement_problems,
49
+ repetitions,
50
+ resolve,
51
+ )
52
+
53
+ RANK = {REJECTED: -1, HYPOTHESIS: 0, SUPPORTED: 1, VERIFIED: 2, STATIC: 2, OBSERVED: 3}
54
+
55
+
56
+ def _name_match(claimed: str, symbol: str) -> bool:
57
+ """`ensure_backend` or `ModelManager.ensure_backend` matches
58
+ `py:...ModelManager.ensure_backend`."""
59
+ claimed = claimed.split(":", 1)[1] if claimed[:3] in ("go:", "py:", "js:") else claimed
60
+ tail = symbol.split(":", 1)[-1]
61
+ return tail == claimed or tail.endswith("." + claimed)
62
+
63
+
64
+ def _callee_match(claimed: str, callee: str) -> bool:
65
+ """A claimed entity against a call expression (`self._backend.unload`)."""
66
+ last = claimed.rsplit(".", 1)[-1]
67
+ return callee in (claimed, last) or callee.endswith("." + last)
68
+
69
+
70
+ # --------------------------------------------------------------------------- boundary facts
71
+
72
+
73
+ def _d(canonical: str) -> str:
74
+ return hashlib.sha256(canonical.encode()).hexdigest()[:16]
75
+
76
+
77
+ # Collectors digest values as sha256(canonical)[:16]; the canonical forms of "no value"
78
+ # (Python None, Go nil, JS null/undefined) and of booleans are known, so these facts are
79
+ # decidable from digests alone, without capturing any value.
80
+ NULL_DIGESTS = frozenset({_d("NoneType:None"), _d("nil"), _d("null"), _d("undefined")})
81
+ NULL_SHAPES = frozenset({"NoneType", "nil", "null", "undefined"})
82
+ BOOL_DIGESTS = {
83
+ _d("bool:True"): True, _d("bool:False"): False, # Python
84
+ _d("bool:true"): True, _d("bool:false"): False, # Go
85
+ _d("boolean:true"): True, _d("boolean:false"): False, # JS
86
+ } # fmt: skip
87
+
88
+
89
+ @dataclass(frozen=True)
90
+ class Identity:
91
+ """An opaque value identity: the collector's digest of a boundary value. It is never
92
+ decoded; predicates compare identities with == and != only."""
93
+
94
+ digest: str
95
+
96
+ def __repr__(self) -> str:
97
+ return f"Identity(#{self.digest[:6]})"
98
+
99
+
100
+ _GO_INT_KINDS = frozenset(
101
+ {"int", "int8", "int16", "int32", "int64", "uint", "uint8", "uint16", "uint32", "uint64"}
102
+ )
103
+
104
+
105
+ def literal_canonical(spec: dict[str, Any]) -> str | None:
106
+ """The canonical form a runtime collector digests for one allowed source literal, or
107
+ None. Allowed (design-value-identity.md, 3): bool, null, integers up to 64 bits, printable
108
+ strings up to 64 characters without quotes or backslashes. Go and Python only."""
109
+ lang, typ, value = spec.get("lang"), spec.get("type"), spec.get("value")
110
+ if isinstance(value, str) and (
111
+ len(value) > 64 or not value.isprintable() or '"' in value or "'" in value or "\\" in value
112
+ ):
113
+ return None
114
+ is_int = isinstance(value, int) and not isinstance(value, bool) and abs(value) < 2**63
115
+ if lang == "go":
116
+ if typ == "string" and isinstance(value, str):
117
+ return f'string:"{value}"'
118
+ if typ in _GO_INT_KINDS and is_int:
119
+ return f"{typ}:{value}"
120
+ if typ == "bool" and isinstance(value, bool):
121
+ return f"bool:{'true' if value else 'false'}"
122
+ if typ == "nil":
123
+ return "nil"
124
+ if lang == "python":
125
+ if typ == "str" and isinstance(value, str):
126
+ return f"str:{value!r}"
127
+ if typ == "int" and is_int:
128
+ return f"int:{value!r}"
129
+ if typ == "bool" and isinstance(value, bool):
130
+ return f"bool:{value!r}"
131
+ if typ in ("None", "NoneType"):
132
+ return "NoneType:None"
133
+ return None
134
+
135
+
136
+ def literal_in_source(spec: dict[str, Any], root: Any) -> bool:
137
+ """The literal is written on the cited source line (the checker never guesses values)."""
138
+ src = spec.get("source") or {}
139
+ try:
140
+ lines = (root / str(src.get("file", ""))).read_text(encoding="utf-8").splitlines()
141
+ text = lines[int(src.get("line", 0)) - 1]
142
+ except (OSError, ValueError, IndexError, TypeError):
143
+ return False
144
+ value = spec.get("value")
145
+ if isinstance(value, str):
146
+ return f'"{value}"' in text or f"'{value}'" in text
147
+ if isinstance(value, bool):
148
+ return re.search(r"\b(true|false|True|False)\b", text) is not None
149
+ if isinstance(value, int):
150
+ return re.search(rf"(?<![\w.]){value}(?![\w.])", text) is not None
151
+ return re.search(r"\b(nil|None|null)\b", text) is not None
152
+
153
+
154
+ def boundary_fact(node: Any, point: str, kind: str, literal: dict[str, Any] | None = None) -> Any:
155
+ """An identity-level fact at a call boundary, or None when not decidable.
156
+
157
+ point: "arg:<name>" | "result". kind:
158
+ is_set the value is not null/none/nil
159
+ changed_from:arg:<name> the value differs from that argument (digest inequality)
160
+ size the collection's size (argument shapes carry it)
161
+ bool a boolean, decoded from its digest"""
162
+ if not isinstance(node, CallNode):
163
+ return None
164
+ args = {a: (shape, dg) for a, shape, dg in node.args}
165
+ if point == "result":
166
+ shape, dg = None, node.result
167
+ elif point.startswith("arg:") and point[4:] in args:
168
+ shape, dg = args[point[4:]]
169
+ else:
170
+ return None
171
+ if kind == "is_set":
172
+ if shape is not None:
173
+ return shape.split("[", 1)[0] not in NULL_SHAPES
174
+ return (dg not in NULL_DIGESTS) if dg else None
175
+ if kind.startswith("changed_from:"):
176
+ src = kind.split(":", 1)[1]
177
+ if not src.startswith("arg:") or src[4:] not in args:
178
+ return None
179
+ other = args[src[4:]][1]
180
+ return (dg != other) if dg and other else None
181
+ if kind == "size":
182
+ m = re.fullmatch(r"[^\[]*\[(\d+)\]", shape or "")
183
+ return int(m.group(1)) if m else None
184
+ if kind == "bool":
185
+ return BOOL_DIGESTS.get(dg or "")
186
+ if kind == "identity":
187
+ return Identity(dg) if dg else None
188
+ if kind == "equals_literal":
189
+ canon = literal_canonical(literal or {})
190
+ return (dg == _d(canon)) if canon and dg else None
191
+ return None
192
+
193
+
194
+ def _exit_point(kind: str, point: str) -> bool:
195
+ """Result facts and change facts describe the call's exit; argument facts its entry."""
196
+ return point == "result" or kind.startswith("changed_from:")
197
+
198
+
199
+ def exit_matches(claim: str, observed: str) -> bool | None:
200
+ """A claimed exit (`returned`, `returned-error[:kind]`, `raised[:kind]`, `panic[:kind]`,
201
+ `cancelled`; `completed` = `returned`) against a call's observed outcome (model.Outcome).
202
+ A kind matches the observed identity exactly or as its dotted suffix. None if unknown."""
203
+ if not observed or observed == "unknown":
204
+ return None
205
+ claim = claim.replace("returned_error", "returned-error")
206
+ claim = "returned" if claim == "completed" else claim
207
+ c_cat, _, c_kind = claim.partition(":")
208
+ o_cat, _, o_kind = observed.partition(":")
209
+ if c_cat != o_cat:
210
+ return False
211
+ if not c_kind:
212
+ return True
213
+ # identities may carry the runtime prefix on either side (Experiment 10, Case B: a claim
214
+ # copied from an observed exit kept its `go:` and never matched the observation)
215
+ o_kind = o_kind.split(":", 1)[1] if o_kind[:3] in ("py:", "go:", "js:") else o_kind
216
+ c_kind = c_kind.split(":", 1)[1] if c_kind[:3] in ("py:", "go:", "js:") else c_kind
217
+ return o_kind == c_kind or o_kind.endswith("." + c_kind)
218
+
219
+
220
+ def _enclosing_call(ob: ExecObs, node_id: int, entity: str) -> Any:
221
+ n: Any = ob.execution.nodes[node_id] if node_id < len(ob.execution.nodes) else None
222
+ while n is not None:
223
+ if isinstance(n, CallNode) and _name_match(entity, n.symbol):
224
+ return n
225
+ n = ob.execution.nodes[n.parent] if n.parent is not None else None
226
+ return None
227
+
228
+
229
+ # --------------------------------------------------------------------------- mechanics
230
+
231
+
232
+ class Mechanics:
233
+ def __init__(self, functions: list[dict[str, Any]]) -> None:
234
+ self.functions = functions
235
+ self.sites: dict[str, dict[str, Any]] = {}
236
+ for f in functions:
237
+ for s in f["sites"]:
238
+ self.sites[s["site"]] = {**s, "symbol": f["symbol"], "file": f["file"]}
239
+
240
+ def function_of(self, site: str) -> dict[str, Any] | None:
241
+ sym = self.sites.get(site, {}).get("symbol")
242
+ return next((f for f in self.functions if f["symbol"] == sym), None)
243
+
244
+ def functions_named(self, entity: str) -> list[dict[str, Any]]:
245
+ return [f for f in self.functions if _name_match(entity, f["symbol"])]
246
+
247
+ def site_at(self, file: str, line: int, entity: str | None = None) -> list[str]:
248
+ """Sites whose condition spans `line` of `file`, preferring the entity's function."""
249
+ hits = [
250
+ s["site"]
251
+ for s in self.sites.values()
252
+ if s["file"] == file and s["span"] and s["span"][0] <= line <= s["span"][2]
253
+ ]
254
+ if entity:
255
+ own = [h for h in hits if _name_match(entity, self.sites[h]["symbol"])]
256
+ if own:
257
+ return own
258
+ return hits
259
+
260
+ def reachable_under(self, site: str, outcome: bool, callee: str) -> bool | None:
261
+ """In the site's function: can a call of `callee` run after the site takes
262
+ `outcome`? A call qualifies when it lies after the site or inside its `outcome`
263
+ branch, and none of its requirements conflicts with the site's own requirements
264
+ plus (site, outcome). None when the function has no call after the site to judge."""
265
+ fn = self.function_of(site)
266
+ rec = self.sites.get(site)
267
+ if fn is None or rec is None:
268
+ return None
269
+ ctx = {(s, o) for s, o in rec["requires"]} | {(site, outcome)}
270
+ end_line = rec["span"][2] if rec.get("span") else rec["line"]
271
+ candidates = []
272
+ for c in fn["calls"]:
273
+ if not _callee_match(callee, c["callee"]):
274
+ continue
275
+ req = {(s, o) for s, o in c["requires"]}
276
+ if (
277
+ c["line"] <= end_line
278
+ and (site, outcome) not in req
279
+ and (site, not outcome) not in req
280
+ ):
281
+ continue # before the site: not a consequence of it
282
+ candidates.append(req)
283
+ if not candidates:
284
+ return None
285
+ return any(
286
+ not any((s, not o) in ctx for s, o in req if o is not None) for req in candidates
287
+ )
288
+
289
+ def exits_under(self, site: str, outcome: bool) -> bool | None:
290
+ rec = self.sites.get(site)
291
+ if rec is None:
292
+ return None
293
+ return bool(rec["then_exits"] if outcome else rec["else_exits"])
294
+
295
+
296
+ # --------------------------------------------------------------------------- observations
297
+
298
+
299
+ @dataclass
300
+ class BranchEvent:
301
+ site: str
302
+ outcome: bool
303
+ node: int
304
+ seq: int
305
+ symbol: str
306
+
307
+
308
+ @dataclass
309
+ class ExecObs:
310
+ execution: Execution
311
+ branches: list[BranchEvent] = field(default_factory=list)
312
+
313
+ def children_of(self, node: int) -> dict[int, list[int]]:
314
+ kids: dict[int, list[int]] = {}
315
+ for n in self.execution.nodes:
316
+ if n.parent is not None:
317
+ kids.setdefault(n.parent, []).append(n.id)
318
+ return kids
319
+
320
+ def subtree_after(self, node: int, seq: int) -> list[str]:
321
+ """Symbols of calls under `node` created at or after `seq`, then of calls under the
322
+ node's caller created after `node` (the caller's continuation), in order."""
323
+ kids = self.children_of(node)
324
+ out: list[str] = []
325
+
326
+ def walk(n: int, floor: int) -> None:
327
+ for k in kids.get(n, []):
328
+ if k >= floor:
329
+ nd = self.execution.nodes[k]
330
+ out.append(
331
+ getattr(nd, "symbol", None) or getattr(nd, "claimed_target", None) or ""
332
+ )
333
+ walk(k, 0)
334
+
335
+ walk(node, seq)
336
+ parent = self.execution.nodes[node].parent
337
+ if parent is not None:
338
+ walk(parent, node + 1)
339
+ return out
340
+
341
+
342
+ class Observations:
343
+ def __init__(self, executions: list[Execution]) -> None:
344
+ self.by_test: dict[str, ExecObs] = {}
345
+ for ex in executions:
346
+ ob = ExecObs(ex)
347
+ for b in ex.branches:
348
+ sym = getattr(ex.nodes[b.node], "symbol", "") if b.node < len(ex.nodes) else ""
349
+ ob.branches.append(BranchEvent(b.site, b.outcome, b.node, b.seq, sym))
350
+ self.by_test[ex.stimulus_ref] = ob
351
+
352
+ def get(self, test: str) -> ExecObs | None:
353
+ if test in self.by_test:
354
+ return self.by_test[test]
355
+ hits = [
356
+ v for k, v in self.by_test.items() if k.endswith("::" + test) or k.endswith("/" + test)
357
+ ]
358
+ return hits[0] if len(hits) == 1 else None
359
+
360
+ def outcomes_at(self, site: str) -> dict[bool, list[str]]:
361
+ out: dict[bool, list[str]] = {True: [], False: []}
362
+ for test, ob in self.by_test.items():
363
+ for b in ob.branches:
364
+ if b.site == site and test not in out[b.outcome]:
365
+ out[b.outcome].append(test)
366
+ return out
367
+
368
+
369
+ # --------------------------------------------------------------------------- checking
370
+
371
+
372
+ class StateSubstrate(Substrate):
373
+ """Substrate with mechanics and observations; checks the new evidence kinds."""
374
+
375
+ def __init__(self, base: Substrate, mech: Mechanics, obs: Observations) -> None:
376
+ super().__init__(
377
+ list(base.paths.values()), base.observed_edges, base.source_root, base.decision_sites
378
+ )
379
+ self.mech = mech
380
+ self.obs = obs
381
+ # sites evaluated at runtime; a site may be observed without static facts (a file or
382
+ # nested function the front end did not lower): it exists, but is not anchored
383
+ self.runtime_sites = {b.site for ob in obs.by_test.values() for b in ob.branches}
384
+
385
+ def site_exists(self, site: str | None) -> bool:
386
+ return bool(site) and (site in self.mech.sites or site in self.runtime_sites)
387
+
388
+ def check(self, ref: EvidenceRef) -> None:
389
+ k = ref.kind
390
+ if k == "branch":
391
+ ob = self.obs.get(ref.test or "")
392
+ if ob is None:
393
+ ref.ok, ref.why = False, f"no execution {ref.test!r}"
394
+ elif not self.site_exists(ref.site):
395
+ # nothing knows this site: missing support, not an observed contradiction
396
+ ref.ok, ref.why, ref.unanchored = False, f"unknown site {ref.site}", True
397
+ else:
398
+ ref.ok = any(b.site == ref.site and b.outcome == ref.outcome for b in ob.branches)
399
+ ref.why = "" if ref.ok else f"{ref.site} never evaluated to {ref.outcome} there"
400
+ return
401
+ if k == "control":
402
+ r = self.mech.reachable_under(ref.site or "", bool(ref.outcome), ref.callee or "")
403
+ ref.ok = r is True and (
404
+ self.mech.reachable_under(ref.site or "", not ref.outcome, ref.callee or "")
405
+ is False
406
+ )
407
+ ref.why = (
408
+ "" if ref.ok else f"{ref.callee} is not controlled by {ref.site}={ref.outcome}"
409
+ )
410
+ return
411
+ if k == "dataflow":
412
+ rec = self.mech.sites.get(ref.site or "") or {}
413
+ ref.ok = bool(rec) and any(
414
+ (ref.origin or "") in o for op in rec.get("operands", []) for o in op["origins"]
415
+ )
416
+ ref.why = "" if ref.ok else f"no operand of {ref.site} originates at {ref.origin}"
417
+ return
418
+ if k == "boundary":
419
+ ob = self.obs.get(ref.test or "")
420
+ if ob is None:
421
+ ref.ok, ref.why = False, f"no execution {ref.test!r}"
422
+ return
423
+ ref.ok = any(
424
+ isinstance(n, CallNode)
425
+ and _name_match(ref.entity or "", n.symbol)
426
+ and boundary_fact(n, ref.point or "", ref.binding or "is_set") == ref.value
427
+ for n in ob.execution.nodes
428
+ )
429
+ ref.why = (
430
+ ""
431
+ if ref.ok
432
+ else f"no {ref.entity} call with {ref.point} {ref.binding} = {ref.value!r}"
433
+ )
434
+ return
435
+ if k == "outcome":
436
+ ob = self.obs.get(ref.test or "")
437
+ if ob is None:
438
+ ref.ok, ref.why = False, f"no execution {ref.test!r}"
439
+ return
440
+ ref.ok = any(
441
+ isinstance(n, CallNode)
442
+ and _name_match(ref.entity or "", n.symbol)
443
+ and exit_matches(ref.exit or "", n.outcome) is True
444
+ for n in ob.execution.nodes
445
+ )
446
+ ref.why = "" if ref.ok else f"no {ref.entity} call ending {ref.exit}"
447
+ return
448
+ if k == "store":
449
+ ref.ok = any(
450
+ any(st["path"] == ref.path for st in f["stores"])
451
+ for f in self.mech.functions_named(ref.entity or "")
452
+ )
453
+ ref.why = "" if ref.ok else f"{ref.entity} has no store of {ref.path}"
454
+ return
455
+ if k == "delta":
456
+ ob = self.obs.get(ref.test or "")
457
+ if ob is None:
458
+ ref.ok, ref.why = False, f"no execution {ref.test!r}"
459
+ return
460
+ ref.ok = False
461
+ for n in ob.execution.nodes:
462
+ if (
463
+ isinstance(n, CallNode)
464
+ and _name_match(ref.entity or "", n.symbol)
465
+ and n.state_after
466
+ ):
467
+ before = {a: b for a, b, _ in n.state}
468
+ after = {a: b for a, b, _ in n.state_after}
469
+ if (
470
+ before.get(ref.fact or "") == ref.before
471
+ and after.get(ref.fact or "") == ref.after
472
+ ):
473
+ ref.ok = True
474
+ break
475
+ ref.why = (
476
+ "" if ref.ok else f"no {ref.entity} call with {ref.fact}: {ref.before}→{ref.after}"
477
+ )
478
+ return
479
+ super().check(ref)
480
+
481
+
482
+ def _link_site(d: Decision, sub: StateSubstrate) -> tuple[str | None, str]:
483
+ if d.site:
484
+ if d.site in sub.mech.sites:
485
+ return d.site, "given"
486
+ if d.site in sub.runtime_sites:
487
+ return d.site, "given; observed at runtime, no static facts"
488
+ return None, f"unknown site {d.site}"
489
+ cited = {
490
+ s
491
+ for r in d.evidence
492
+ if r.kind == "source" and r.ok and r.file and r.line
493
+ for s in sub.mech.site_at(r.file, r.line)
494
+ }
495
+ own = {s for s in cited if _name_match(d.entity, sub.mech.sites[s]["symbol"])}
496
+ uniq = sorted(own or cited) # the decision's own function first
497
+ if len(uniq) == 1:
498
+ return uniq[0], "linked by the only cited `if` line"
499
+ return None, "no unique cited decision site" if not uniq else f"cites {len(uniq)} sites"
500
+
501
+
502
+ def _static_consistent(d: Decision, site: str, sub: StateSubstrate) -> list[str]:
503
+ """Claimed consequences vs the site's static control facts. Only calls that exist in the
504
+ site's function and lie after it are judged; 'stops' is not checked statically, since
505
+ whether a genome-level step follows is not a local property."""
506
+ problems: list[str] = []
507
+ if site not in sub.mech.sites:
508
+ return problems # no static facts to be consistent with
509
+ for b in (d.true_branch, d.false_branch):
510
+ for c in b.calls:
511
+ if _name_match(c, sub.mech.sites[site]["symbol"]):
512
+ continue # the deciding function itself
513
+ if sub.mech.reachable_under(site, b.when, c) is False:
514
+ problems.append(f"{c} unreachable when {site}={b.when}")
515
+ for a in b.absent:
516
+ if sub.mech.reachable_under(site, b.when, a) is True:
517
+ problems.append(f"{a} reachable when {site}={b.when}")
518
+ return problems
519
+
520
+
521
+ def _symbol(n: Any) -> str:
522
+ return getattr(n, "symbol", None) or getattr(n, "claimed_target", None) or ""
523
+
524
+
525
+ def episode_calls(ob: ExecObs, ev: BranchEvent, mech: Mechanics) -> list[str]:
526
+ """Calls that belong to the episode a branch evaluation decides, in order.
527
+
528
+ Node ids and branch `seq` share one clock, so every bound is a time cut:
529
+ - inside the deciding call: everything after the evaluation, up to the next evaluation
530
+ of the same site in that call (the next loop iteration) and, when the site lies in a
531
+ loop, up to the first direct call that the static facts place outside every loop
532
+ (control has left the loop);
533
+ - the caller's continuation only when the taken branch leaves the deciding function
534
+ (static exit facts), up to the next call of the same function or the next
535
+ evaluation of the site anywhere.
536
+ Deferred callbacks and later iterations are therefore outside the episode; claims about
537
+ them need `absent_scope: "run"`."""
538
+ nodes = ob.execution.nodes
539
+ kids = ob.children_of(ev.node) # the whole parent -> children map
540
+ later = [
541
+ b.seq for b in ob.branches if b.site == ev.site and b.node == ev.node and b.seq > ev.seq
542
+ ]
543
+ end: int | None = min(later) if later else None
544
+ rec = mech.sites.get(ev.site)
545
+ fn = mech.function_of(ev.site)
546
+ if rec is not None and fn is not None and any(r[0] == "?loop" for r in rec["requires"]):
547
+ for k in kids.get(ev.node, []):
548
+ if k < ev.seq or (end is not None and k >= end):
549
+ continue
550
+ tail = _symbol(nodes[k]).split(":", 1)[-1]
551
+ matches = [c for c in fn["calls"] if _callee_match(tail, c["callee"])]
552
+ if matches and all(not any(r[0] == "?loop" for r in c["requires"]) for c in matches):
553
+ end = k
554
+ break
555
+ out: list[str] = []
556
+
557
+ def walk(tree: dict[int, list[int]], n: int, floor: int, ceil: int | None) -> None:
558
+ for k in tree.get(n, []):
559
+ if k >= floor and (ceil is None or k < ceil):
560
+ out.append(_symbol(nodes[k]))
561
+ walk(tree, k, 0, ceil)
562
+
563
+ walk(kids, ev.node, ev.seq, end)
564
+ parent = nodes[ev.node].parent if ev.node < len(nodes) else None
565
+ if end is None and parent is not None and mech.exits_under(ev.site, ev.outcome):
566
+ me = _symbol(nodes[ev.node])
567
+ nxt = [b.seq for b in ob.branches if b.site == ev.site and b.seq > ev.seq]
568
+ stop: int | None = min(nxt) if nxt else None
569
+ for k in kids.get(parent, []):
570
+ if k <= ev.node:
571
+ continue
572
+ if (stop is not None and k >= stop) or _symbol(nodes[k]) == me:
573
+ break
574
+ out.append(_symbol(nodes[k]))
575
+ walk(kids, k, 0, stop)
576
+ return out
577
+
578
+
579
+ def _observed_contradictions(d: Decision, site: str, sub: StateSubstrate) -> list[str]:
580
+ bad = []
581
+ for test, ob in sub.obs.by_test.items():
582
+ for ev in ob.branches:
583
+ if ev.site != site:
584
+ continue
585
+ b = d.true_branch if ev.outcome else d.false_branch
586
+ if b.outcome and b.outcome.get("entity") and b.outcome.get("is"):
587
+ call = _enclosing_call(ob, ev.node, b.outcome["entity"])
588
+ ok = exit_matches(b.outcome["is"], call.outcome) if call is not None else None
589
+ if ok is False:
590
+ bad.append(
591
+ f"{test}: {b.outcome['entity']} ended {call.outcome}, "
592
+ f"not {b.outcome['is']}, after {site}={ev.outcome}"
593
+ )
594
+ if not b.absent:
595
+ continue
596
+ after = (
597
+ ob.subtree_after(ev.node, ev.seq)
598
+ if b.absent_scope == "run"
599
+ else episode_calls(ob, ev, sub.mech)
600
+ )
601
+ for a in b.absent:
602
+ if any(_name_match(a, s) for s in after if s):
603
+ bad.append(f"{test}: {a} ran after {site}={ev.outcome}")
604
+ return sorted(set(bad))
605
+
606
+
607
+ def _execution_constants(ob: ExecObs) -> dict[str, str]:
608
+ """Global facts holding a single bucket throughout the execution."""
609
+ seen: dict[str, set[str]] = {}
610
+ for m in ob.execution.nodes:
611
+ if isinstance(m, CallNode):
612
+ for a, b, _ in (*m.state, *m.state_after):
613
+ if a.startswith("global."):
614
+ seen.setdefault(a, set()).add(b)
615
+ return {a: next(iter(bs)) for a, bs in seen.items() if len(bs) == 1}
616
+
617
+
618
+ def _local_agreement(
619
+ g: Genome, d: Decision, site: str, sub: StateSubstrate, out_of_scope: set[str]
620
+ ) -> tuple[list[str], list[str]]:
621
+ """Assumes the state at the site equals the call's entry state except for fields the
622
+ function itself stores before the site. Interleaved tasks (another coroutine running
623
+ across an await) break that assumption, so executions outside the genome's sequential
624
+ scope are excluded."""
625
+ rec = sub.mech.sites.get(site)
626
+ if rec is None:
627
+ # without static facts it is unknown which fields the function writes before the
628
+ # site, so the entry state cannot stand in for the state at the site
629
+ return [], []
630
+ fn = sub.mech.function_of(site) or {"stores": []}
631
+ stored_before = {st["path"] for st in fn["stores"] if st["line"] < rec["line"]}
632
+ agree: list[str] = []
633
+ disagree: list[str] = []
634
+ for test, ob in sub.obs.by_test.items():
635
+ if test in out_of_scope or test.split("::")[-1] in out_of_scope:
636
+ continue
637
+ constants = _execution_constants(ob)
638
+ for ev in ob.branches:
639
+ if ev.site != site:
640
+ continue
641
+ node = ob.execution.nodes[ev.node]
642
+ if not isinstance(node, CallNode):
643
+ continue
644
+ facts = {**constants, **{a: b for a, b, _ in node.state if a not in stored_before}}
645
+ v = eval_predicate(d.predicate, derive(g, _bind(g, facts, node, "entry", ob)))
646
+ if v is None:
647
+ continue
648
+ name = test.split("/")[-1].split("::")[-1]
649
+ (agree if v == ev.outcome else disagree).append(
650
+ f"{name}: predicate {v} on observed state, site {ev.outcome}"
651
+ )
652
+ return agree, disagree
653
+
654
+
655
+ def _is_ancestor(ob: ExecObs, anc: int, node: int) -> bool:
656
+ p = ob.execution.nodes[node].parent
657
+ while p is not None:
658
+ if p == anc:
659
+ return True
660
+ p = ob.execution.nodes[p].parent
661
+ return False
662
+
663
+
664
+ def scoped_value(ob: ExecObs, b: dict[str, Any], before: int | None = None) -> tuple[Any, str]:
665
+ """The value of a binding across one execution: the boundary's value in every occurrence
666
+ of its entity (only those that ran before node `before`, when given: argument points of
667
+ occurrences started earlier, result points of occurrences finished earlier). Returns
668
+ (value, "") when all agree, (None, reason) when missing or ambiguous."""
669
+ at = b.get("at") or {}
670
+ entity, point = str(at.get("entity", "")), str(at.get("point", ""))
671
+ kind = str(b.get("kind", "identity"))
672
+ vals = []
673
+ for n in ob.execution.nodes:
674
+ if not (isinstance(n, CallNode) and _name_match(entity, n.symbol)):
675
+ continue
676
+ if before is not None:
677
+ if n.id >= before:
678
+ continue
679
+ if _exit_point(kind, point) and _is_ancestor(ob, n.id, before):
680
+ continue # still running: its result is not known yet
681
+ v = boundary_fact(n, point, kind, b.get("literal"))
682
+ if v is not None:
683
+ vals.append(v)
684
+ if not vals:
685
+ return None, "missing"
686
+ if any(v != vals[0] for v in vals[1:]):
687
+ return None, "ambiguous"
688
+ return vals[0], ""
689
+
690
+
691
+ def _bind(
692
+ g: Genome,
693
+ facts: dict[str, str],
694
+ node: Any = None,
695
+ phase: str = "entry",
696
+ ob: ExecObs | None = None,
697
+ ) -> dict[str, Any]:
698
+ """Observed buckets -> genome variable values, through each variable's binding(s). With
699
+ a call node, variables bound at that entity's boundary (`observed_as.at`) are added:
700
+ argument facts at "entry", result and change facts at "exit". With the execution too,
701
+ bindings of `scope: "execution"` at OTHER entities take the value those boundaries had
702
+ earlier in the execution, when it is unique (design-value-identity.md, 2)."""
703
+ out: dict[str, Any] = {}
704
+ for v in g.variables:
705
+ for b in bindings_of(v):
706
+ if v.name in out:
707
+ break
708
+ at = b.get("at")
709
+ if isinstance(at, dict):
710
+ kind = str(b.get("kind", "is_set"))
711
+ if kind == "equals_literal" and b.get("_literal_ok") is not True:
712
+ continue # unchecked or not in source: never used
713
+ point = str(at.get("point", ""))
714
+ if node is not None and _name_match(str(at.get("entity", "")), node.symbol):
715
+ if _exit_point(kind, point) != (phase == "exit"):
716
+ continue
717
+ val = boundary_fact(node, point, kind, b.get("literal"))
718
+ elif (
719
+ b.get("scope") == "execution"
720
+ and ob is not None
721
+ and node is not None
722
+ and phase == "entry"
723
+ ):
724
+ val, _ = scoped_value(ob, b, before=node.id)
725
+ else:
726
+ continue
727
+ if val is not None:
728
+ out[v.name] = val
729
+ continue
730
+ fact = b.get("fact")
731
+ if not fact or fact not in facts:
732
+ continue
733
+ bucket = facts[fact]
734
+ kind = b.get("kind", "is_set")
735
+ if kind == "is_set":
736
+ out[v.name] = bucket != "none"
737
+ elif kind == "bool" and bucket.startswith("bool:"):
738
+ out[v.name] = bucket == "bool:true"
739
+ elif kind == "sign" and bucket.startswith("num:"):
740
+ out[v.name] = {"num:zero": 0, "num:pos": 1, "num:neg": -1}[bucket]
741
+ return out
742
+
743
+
744
+ def identity_claim(v: Variable, obs: Observations) -> tuple[str, str, int, int]:
745
+ """Check a variable observed at >= 2 identity boundaries: every observation of it in one
746
+ execution is the same value. Returns (status or "", reason, agreeing executions, distinct
747
+ identities). Rejected on any execution where all sides are present and differ; verified
748
+ when equal in >= 2 executions carrying >= 2 distinct identities (a contrast, so that a
749
+ constant coincidence cannot verify it); supported when equal but vacuous; "" (unknown)
750
+ when no execution shows every side."""
751
+ sides = [b for b in bindings_of(v) if b.get("kind") == "identity"]
752
+ agree: list[Any] = []
753
+ bad: list[str] = []
754
+ for test, ob in obs.by_test.items():
755
+ vals = [scoped_value(ob, b)[0] for b in sides]
756
+ if any(x is None for x in vals):
757
+ continue
758
+ if all(x == vals[0] for x in vals[1:]):
759
+ agree.append(vals[0])
760
+ else:
761
+ bad.append(test.split("::")[-1])
762
+ distinct = len(set(agree))
763
+ if bad:
764
+ return (
765
+ REJECTED,
766
+ f"identity contradicted: sides differ in {', '.join(bad[:3])}",
767
+ len(agree),
768
+ distinct,
769
+ )
770
+ if len(agree) >= 2 and distinct >= 2:
771
+ return (
772
+ VERIFIED,
773
+ (
774
+ f"identity verified: equal in {len(agree)} execution(s) carrying {distinct} "
775
+ "distinct values"
776
+ ),
777
+ len(agree),
778
+ distinct,
779
+ )
780
+ if agree:
781
+ return (
782
+ SUPPORTED,
783
+ (
784
+ f"identity observed in {len(agree)} execution(s) with {distinct} distinct "
785
+ "value(s): no contrast, not verified"
786
+ ),
787
+ len(agree),
788
+ distinct,
789
+ )
790
+ return "", "identity unknown: no execution shows every side", 0, 0
791
+
792
+
793
+ def _abstract_equal(kind: str, predicted: Any, observed: Any) -> bool:
794
+ if kind == "sign":
795
+ sign = (
796
+ (predicted > 0) - (predicted < 0) if isinstance(predicted, int | float) else predicted
797
+ )
798
+ return bool(sign == observed)
799
+ return bool(predicted == observed)
800
+
801
+
802
+ def eval_expr(expr: str, env: dict[str, Any]) -> Any:
803
+ """Tiny expression language for transitions. Returns None when not decidable."""
804
+ e = expr.strip()
805
+ if e in ("true", "True"):
806
+ return True
807
+ if e in ("false", "False"):
808
+ return False
809
+ if e in ("none", "None", "null"):
810
+ return None
811
+ if re.fullmatch(r"-?\d+", e):
812
+ return int(e)
813
+ m = re.fullmatch(r"max\(\s*0\s*,\s*([\w.]+)\s*-\s*(\d+)\s*\)", e)
814
+ if m:
815
+ v = env.get(m.group(1))
816
+ return None if not isinstance(v, int) else max(0, v - int(m.group(2)))
817
+ m = re.fullmatch(r"([\w.]+)\s*([+-])\s*(\d+)", e)
818
+ if m:
819
+ v = env.get(m.group(1))
820
+ if not isinstance(v, int):
821
+ return None
822
+ return v + int(m.group(3)) if m.group(2) == "+" else v - int(m.group(3))
823
+ if e.startswith("!"):
824
+ v = env.get(e[1:].strip())
825
+ return None if not isinstance(v, bool) else not v
826
+ return env.get(e)
827
+
828
+
829
+ def agreement_from_facts(
830
+ g: Genome, test_facts: dict[str, dict[str, Any]]
831
+ ) -> dict[str, list[tuple[str, bool]]]:
832
+ """Per decision: (test, predicate value) for every test whose independently stated input
833
+ facts decide the decision's predicate. Used to check predicate semantics against the
834
+ outcome actually observed at the decision's site in that test."""
835
+ out: dict[str, list[tuple[str, bool]]] = {}
836
+ for d in g.decisions:
837
+ for test, facts in test_facts.items():
838
+ v = eval_predicate(d.predicate, derive(g, facts))
839
+ if v is not None:
840
+ out.setdefault(d.id, []).append((test, v))
841
+ return out
842
+
843
+
844
+ def genome_vocabulary(g: Genome) -> list[str]:
845
+ """Every entity a genome names: procedures, decisions, transitions and called entities."""
846
+ names = {p.entity for p in g.procedures} | {d.entity for d in g.decisions}
847
+ names |= {t.entity for t in g.transitions} | {r.entity for r in g.regions}
848
+ names |= {r.head for r in g.regions if r.head and not r.head.startswith("site:")}
849
+ step_lists = (
850
+ [p.steps for p in g.procedures]
851
+ + [r.steps for r in g.regions]
852
+ + [
853
+ [*b.steps, *(f"call:{c}" for c in b.calls)]
854
+ for d in g.decisions
855
+ for b in (d.true_branch, d.false_branch)
856
+ ]
857
+ )
858
+ for steps in step_lists:
859
+ names |= {st[5:] for st in steps if st.startswith("call:")}
860
+ return sorted(n for n in names if n)
861
+
862
+
863
+ def _reached(
864
+ steps: list[str],
865
+ decisions: dict[str, Decision],
866
+ vocab: list[str],
867
+ regions: dict[str, Any] | None = None,
868
+ ) -> tuple[list[str], list[tuple[str, str]]]:
869
+ """Calls and decision sites reached in a procedure's own body (its steps, its
870
+ decisions' branches and the bodies of regions it places), without entering other
871
+ entities' procedures."""
872
+ calls: list[str] = []
873
+ dsites: list[tuple[str, str]] = []
874
+ seen: set[str] = set()
875
+ stack = [list(steps)]
876
+ while stack:
877
+ for st in stack.pop():
878
+ kind, _, ref = st.partition(":")
879
+ if kind == "call":
880
+ calls.append(resolve(vocab, ref))
881
+ elif kind == "D" and ref in decisions and ref not in seen:
882
+ seen.add(ref)
883
+ d = decisions[ref]
884
+ if d.site:
885
+ dsites.append((d.id, d.site))
886
+ for br in (d.true_branch, d.false_branch):
887
+ stack.append(br.steps or [f"call:{c}" for c in br.calls])
888
+ elif kind == "R" and regions and ref in regions and f"R:{ref}" not in seen:
889
+ seen.add(f"R:{ref}")
890
+ stack.append(list(regions[ref].steps))
891
+ return calls, dsites
892
+
893
+
894
+ def procedure_structure_claims(g: Genome, sk: Skeleton) -> list[dict[str, Any]]:
895
+ """What each procedure claims about execution structure, judged against the observed
896
+ skeleton (diffgenome.structure). Absence of observation never rejects: a call or
897
+ decision on a path the tests never took is simply not established. Rejection needs
898
+ positive evidence.
899
+ - contains: `call:B` reached in A's own steps (and its decisions' branches) claims B
900
+ runs during A. Rejected if the nesting is inverted: every observed A ran
901
+ during B. Verified if every observed B ran during A, in >= 2 executions;
902
+ supported if some did; otherwise not established.
903
+ - evaluates: a decision reached in A's steps claims its site is evaluated during A.
904
+ Rejected if the site is evaluated in another function's own call and every
905
+ observed A ran during that function (the decider encloses A: its decisions
906
+ cannot be evaluated inside A). Verified if every evaluation was during A
907
+ (>= 2); supported if some; otherwise not established.
908
+ - order: consecutive steps x, y of A claim x before y. Verified when x precedes y in
909
+ every observed occurrence of A holding both (>= 2 executions), or inside
910
+ one repetition of a repeated region (>= 2 repetitions); rejected when the
911
+ opposite was observed; else supported. Counts are never claimed.
912
+ Recursion (a function running during itself) is outside these rules."""
913
+ vocab = sorted(set(genome_vocabulary(g)) | set(sk.seen))
914
+ decisions = {d.id: d for d in g.decisions}
915
+ region_items = {r.id: r for r in g.regions}
916
+ claims: list[dict[str, Any]] = []
917
+
918
+ def claim(p: Any, kind: str, what: str, status: str, why: str) -> None:
919
+ claims.append(
920
+ {"procedure": p.id, "entity": p.entity, "kind": kind, "claim": what,
921
+ "status": status, "why": why}
922
+ ) # fmt: skip
923
+
924
+ for p in g.procedures:
925
+ a = resolve(vocab, p.entity)
926
+ calls, dsites = _reached(p.steps, decisions, vocab, region_items)
927
+ for b in dict.fromkeys(calls):
928
+ n = sk.seen.get(b, 0)
929
+ inside = sk.inside_occ.get(b, {}).get(a, 0)
930
+ what = f"{b} during {a}"
931
+ if n == 0:
932
+ claim(p, "contains", what, "unobserved", f"{b} never observed")
933
+ elif b != a and sk.always_during(a, b):
934
+ claim(p, "contains", what, REJECTED,
935
+ f"inverted: every observed {a} ran during {b}") # fmt: skip
936
+ elif inside == 0:
937
+ claim(p, "contains", what, "not established",
938
+ f"{b} observed {n} time(s), never during {a}") # fmt: skip
939
+ elif inside == n and sk.ancestry.get(b, {}).get(a, 0) >= 2:
940
+ claim(p, "contains", what, VERIFIED, f"all {n} observed")
941
+ else:
942
+ claim(p, "contains", what, SUPPORTED, f"{inside} of {n} observed")
943
+ for did, site in dsites:
944
+ during = sk.site_during.get(site)
945
+ what = f"{did} ({site}) evaluated during {a}"
946
+ if not during:
947
+ claim(p, "evaluates", what, "unobserved", "site never evaluated")
948
+ continue
949
+ total = max(during.values())
950
+ owners = [o for o in sk.site_owner.get(site, {}) if o != a]
951
+ if a not in during and owners and all(sk.always_during(a, o) for o in owners):
952
+ claim(p, "evaluates", what, REJECTED,
953
+ f"inverted: evaluated in {', '.join(owners)}, and every observed "
954
+ f"{a} ran during it") # fmt: skip
955
+ elif a not in during:
956
+ claim(p, "evaluates", what, "not established",
957
+ f"evaluated {total} time(s), never during {a}") # fmt: skip
958
+ elif during[a] == total and total >= 2:
959
+ claim(p, "evaluates", what, VERIFIED, f"all {total} evaluations")
960
+ else:
961
+ claim(p, "evaluates", what, SUPPORTED, f"{during[a]} of {total} evaluations")
962
+ fams: list[str] = []
963
+ for st in p.steps:
964
+ kind, _, ref = st.partition(":")
965
+ if kind == "call":
966
+ fams.append(resolve(vocab, ref))
967
+ elif kind == "D" and ref in decisions and decisions[ref].site:
968
+ fams.append(f"site:{decisions[ref].site}")
969
+ elif kind == "R" and ref in region_items:
970
+ # a region is placed by its head
971
+ fams.append(_region_head(region_items[ref], vocab))
972
+ for x, y in itertools.pairwise(fams):
973
+ if x == y:
974
+ continue
975
+ what = f"{x} before {y} in {a}"
976
+ regions = [r for r in sk.regions.get(a, []) if r["head"] is not None]
977
+ if sk.strictly_before(a, x, y):
978
+ n_ex = sk.before[a][(x, y)][1]
979
+ claim(p, "order", what, VERIFIED if n_ex >= 2 else SUPPORTED,
980
+ f"all of x before all of y in {n_ex} execution(s)") # fmt: skip
981
+ elif sk.strictly_before(a, y, x):
982
+ claim(p, "order", what, REJECTED, "observed the other way round")
983
+ elif any((x, y) in r["within"] for r in regions):
984
+ n = max(r["within"].get((x, y), 0) for r in regions)
985
+ claim(p, "order", what, VERIFIED if n >= 2 else SUPPORTED,
986
+ f"inside one repetition, {n} repetition(s)") # fmt: skip
987
+ elif any((y, x) in r["within"] for r in regions):
988
+ claim(p, "order", what, REJECTED, "inside one repetition, observed reversed")
989
+ else:
990
+ claim(p, "order", what, SUPPORTED, "not established by observation")
991
+ claims += _region_claims(g, sk, vocab, decisions)
992
+ return claims
993
+
994
+
995
+ def _region_head(r: Any, vocab: list[str]) -> str:
996
+ return r.head if r.head.startswith("site:") else resolve(vocab, r.head)
997
+
998
+
999
+ def _region_claims(
1000
+ g: Genome, sk: Skeleton, vocab: list[str], decisions: dict[str, Decision]
1001
+ ) -> list[dict[str, Any]]:
1002
+ """A region claims (kind `region`) that its head starts every repetition of an observed
1003
+ repeated region of its enclosing entity; its body's consecutive steps are order claims
1004
+ inside ONE repetition; its calls and decisions run during the enclosing entity."""
1005
+ out: list[dict[str, Any]] = []
1006
+
1007
+ def claim(r: Any, kind: str, what: str, status: str, why: str) -> None:
1008
+ out.append(
1009
+ {"procedure": r.id, "entity": r.entity, "kind": kind, "claim": what,
1010
+ "status": status, "why": why}
1011
+ ) # fmt: skip
1012
+
1013
+ for r in g.regions:
1014
+ a = resolve(vocab, r.entity)
1015
+ head = _region_head(r, vocab)
1016
+ observed = [x for x in sk.regions.get(a, []) if x["head"] == head]
1017
+ what = f"region {r.id}: repetitions of {a} start with {head}"
1018
+ if observed:
1019
+ n_ex = sk.ancestry.get(head, {}).get(a, 0) if not head.startswith("site:") else 2
1020
+ claim(r, "region", what, VERIFIED if n_ex >= 2 else SUPPORTED,
1021
+ f"observed repeated region {observed[0]['members']}") # fmt: skip
1022
+ elif head in sk.families.get(a, {}) and sk.families[a][head][1] <= 1:
1023
+ claim(r, "region", what, REJECTED, f"{head} is observed in {a} but never repeated")
1024
+ else:
1025
+ claim(
1026
+ r,
1027
+ "region",
1028
+ what,
1029
+ "unobserved",
1030
+ f"no observed repeated region of {a} with that head",
1031
+ )
1032
+ within = observed[0]["within"] if observed else {}
1033
+ fams: list[str] = []
1034
+ for st in r.steps:
1035
+ kind, _, ref = st.partition(":")
1036
+ if kind == "call":
1037
+ fams.append(resolve(vocab, ref))
1038
+ elif kind == "D" and ref in decisions and decisions[ref].site:
1039
+ fams.append(f"site:{decisions[ref].site}")
1040
+ for x, y in itertools.pairwise(fams):
1041
+ if x == y:
1042
+ continue
1043
+ w = f"{x} before {y} in one repetition of {r.id}"
1044
+ if (x, y) in within:
1045
+ claim(r, "order", w, VERIFIED if within[(x, y)] >= 2 else SUPPORTED,
1046
+ f"{within[(x, y)]} repetition(s)") # fmt: skip
1047
+ elif (y, x) in within:
1048
+ claim(r, "order", w, REJECTED, "inside one repetition, observed reversed")
1049
+ else:
1050
+ claim(r, "order", w, SUPPORTED, "not established by observation")
1051
+ calls, dsites = _reached(r.steps, decisions, vocab)
1052
+ for b in dict.fromkeys(calls):
1053
+ if sk.seen.get(b, 0) and b != a and sk.always_during(a, b):
1054
+ claim(r, "contains", f"{b} during {a}", REJECTED,
1055
+ f"inverted: every observed {a} ran during {b}") # fmt: skip
1056
+ elif sk.inside_occ.get(b, {}).get(a, 0):
1057
+ claim(
1058
+ r,
1059
+ "contains",
1060
+ f"{b} during {a}",
1061
+ VERIFIED,
1062
+ "observed during the enclosing entity",
1063
+ )
1064
+ for did, site in dsites:
1065
+ during = sk.site_during.get(site) or {}
1066
+ owners = [o for o in sk.site_owner.get(site, {}) if o != a]
1067
+ if (
1068
+ during
1069
+ and a not in during
1070
+ and owners
1071
+ and all(sk.always_during(a, o) for o in owners)
1072
+ ):
1073
+ claim(r, "evaluates", f"{did} ({site}) evaluated during {a}", REJECTED,
1074
+ "inverted: the deciding function encloses the region's entity") # fmt: skip
1075
+ return out
1076
+
1077
+
1078
+ # evidence kinds that report what executed: a failure is an observed contradiction
1079
+ _OBSERVED_KINDS = ("execution", "branch", "delta", "boundary", "outcome")
1080
+
1081
+
1082
+ def establish_state(
1083
+ g: Genome,
1084
+ sub: StateSubstrate,
1085
+ agreement: dict[str, list[tuple[str, bool]]] | None = None,
1086
+ agreement_seq: dict[str, list[tuple[str, bool]]] | None = None,
1087
+ out_of_scope: set[str] | None = None,
1088
+ scenarios: dict[str, dict[str, Any]] | None = None,
1089
+ ) -> Genome:
1090
+ """Statuses over the stronger substrate. Same principle as `establish`: only evidence
1091
+ decides; decisions are verified through their site's observed outcomes and static
1092
+ control facts, transitions through observed state deltas."""
1093
+ items: list[Item] = [
1094
+ *g.variables, *g.data_dependencies, *g.decisions, *g.effects, *g.transitions,
1095
+ *g.procedures, *g.regions, *g.rules, *g.regimes,
1096
+ ] # fmt: skip
1097
+ for it in items:
1098
+ for ref in it.evidence:
1099
+ sub.check(ref)
1100
+ failed = [r for r in it.evidence if not r.ok]
1101
+ if not it.evidence:
1102
+ it.status, it.status_reason = HYPOTHESIS, "no evidence cited"
1103
+ elif [r for r in failed if r.kind in _OBSERVED_KINDS and not r.unanchored]:
1104
+ it.status = REJECTED
1105
+ it.status_reason = "contradicted: " + "; ".join(
1106
+ r.why for r in failed if r.kind in _OBSERVED_KINDS and not r.unanchored
1107
+ )
1108
+ elif failed:
1109
+ it.status = HYPOTHESIS
1110
+ it.status_reason = "evidence does not check out: " + "; ".join(r.why for r in failed)
1111
+ else:
1112
+ it.status = SUPPORTED
1113
+ it.status_reason = (
1114
+ "every cited reference checks out ("
1115
+ + ", ".join(sorted({r.kind for r in it.evidence}))
1116
+ + ")"
1117
+ )
1118
+ # literal bindings: usable only when the literal is allowed and written on its cited line
1119
+ for v in g.variables:
1120
+ for bd in bindings_of(v):
1121
+ if bd.get("kind") != "equals_literal":
1122
+ continue
1123
+ lit = bd.get("literal") or {}
1124
+ lit_ok = literal_canonical(lit) is not None and literal_in_source(lit, sub.source_root)
1125
+ bd["_literal_ok"] = lit_ok
1126
+ if not lit_ok and v.status != REJECTED:
1127
+ v.status = HYPOTHESIS
1128
+ v.status_reason = (
1129
+ f"literal {lit.get('value')!r} is not an allowed literal written on "
1130
+ f"{(lit.get('source') or {}).get('file')}:"
1131
+ f"{(lit.get('source') or {}).get('line')}"
1132
+ )
1133
+ # identity claims: a variable observed at >= 2 identity boundaries. The proposal names
1134
+ # the correspondence; the equality itself is observed (digests), never assumed.
1135
+ for v in g.variables:
1136
+ if sum(1 for x in bindings_of(v) if x.get("kind") == "identity") < 2:
1137
+ continue
1138
+ status, why, _, _ = identity_claim(v, sub.obs)
1139
+ if v.status == REJECTED or (
1140
+ v.status == HYPOTHESIS and not v.status_reason.startswith("no evidence cited")
1141
+ ):
1142
+ v.status_reason += f"; {why}" # its citations failed: identity does not rescue it
1143
+ elif status:
1144
+ v.status, v.status_reason = status, why
1145
+ else:
1146
+ v.status_reason += f"; {why}"
1147
+ # decisions: anchored at a site, both outcomes observed, statically consistent,
1148
+ # no observed contradiction, and not sharing the site with another decision
1149
+ anchors: dict[str, list[str]] = {}
1150
+ model_sited = {d.id for d in g.decisions if d.site}
1151
+ for d in g.decisions:
1152
+ if d.site and not sub.site_exists(d.site):
1153
+ d.status = HYPOTHESIS
1154
+ d.status_reason = f"names a decision site that does not exist: {d.site}"
1155
+ continue
1156
+ site, how = _link_site(d, sub)
1157
+ if site:
1158
+ d.site = d.site or site
1159
+ anchors.setdefault(site, []).append(d.id)
1160
+ d.status_reason += f"; site: {site or 'none'} ({how})"
1161
+ for d in g.decisions:
1162
+ if d.status != SUPPORTED:
1163
+ continue
1164
+ site = d.site if sub.site_exists(d.site) else None
1165
+ if site is None:
1166
+ d.status_reason += "; not verified: no decision site"
1167
+ continue
1168
+ anchored = site in sub.mech.sites
1169
+ if len(anchors.get(site, [])) > 1:
1170
+ others = ", ".join(x for x in anchors[site] if x != d.id)
1171
+ d.status_reason += (
1172
+ f"; not verified: site shared with {others} — the observable outcome decides "
1173
+ "only their disjunction"
1174
+ )
1175
+ continue
1176
+ seen = sub.obs.outcomes_at(site)
1177
+ problems = _static_consistent(d, site, sub)
1178
+ contradictions = _observed_contradictions(d, site, sub)
1179
+ # (1) LOCAL agreement: at every observed evaluation of the site, the predicate on the
1180
+ # call's observed entry state (plus execution-constant globals, minus fields the
1181
+ # function stores before the site) must give the observed outcome. Attributable to
1182
+ # this decision alone; a disagreement rejects a decision whose site the proposal named.
1183
+ local_agree, local_disagree = _local_agreement(g, d, site, sub, out_of_scope or set())
1184
+ # (2) scenario agreement (whole-genome replay, or stated test facts): informative but
1185
+ # not attributable to one decision; a disagreement only blocks verification.
1186
+ agree: list[str] = list(local_agree)
1187
+ disagree: list[str] = []
1188
+ for test, same in (agreement_seq or {}).get(d.id, []):
1189
+ (agree if same else disagree).append(
1190
+ f"{test.split('/')[-1].split('::')[-1]}: replayed outcome sequence "
1191
+ + ("matches" if same else "differs")
1192
+ )
1193
+ for test, value in (agreement or {}).get(d.id, []):
1194
+ ob = sub.obs.get(test)
1195
+ observed = {b.outcome for b in ob.branches if b.site == site} if ob else set()
1196
+ if not observed:
1197
+ continue
1198
+ (agree if observed == {value} else disagree).append(
1199
+ f"{test.split('/')[-1].split('::')[-1]}: predicate {value}, site {sorted(observed)}"
1200
+ )
1201
+ if local_disagree and d.id in model_sited:
1202
+ d.status = REJECTED
1203
+ d.status_reason = "contradicted at its own site by observed state: " + "; ".join(
1204
+ local_disagree[:3]
1205
+ )
1206
+ continue
1207
+ if local_disagree or disagree:
1208
+ d.contradicted = True
1209
+ d.status_reason += (
1210
+ "; not verified: predicate disagrees with observed outcomes: "
1211
+ + "; ".join((local_disagree + disagree)[:3])
1212
+ )
1213
+ continue
1214
+ if contradictions:
1215
+ d.status = REJECTED
1216
+ d.status_reason = "contradicted by observed branches: " + "; ".join(contradictions[:3])
1217
+ elif problems:
1218
+ d.status_reason += (
1219
+ "; not verified: inconsistent with static control facts: " + "; ".join(problems[:3])
1220
+ )
1221
+ elif not agree:
1222
+ d.status_reason += (
1223
+ "; not verified: neither the observed state at the site nor any stated input "
1224
+ "facts decide the predicate, so its meaning cannot be checked"
1225
+ )
1226
+ elif seen[True] and seen[False] and not anchored:
1227
+ d.status_reason += (
1228
+ f"; not verified: site {site} is observed both ways and the predicate agrees "
1229
+ f"{len(agree)} time(s), but it has no static facts (unanchored), so its "
1230
+ "consequences cannot be checked"
1231
+ )
1232
+ elif seen[True] and seen[False]:
1233
+ d.status = VERIFIED
1234
+ d.status_reason = (
1235
+ f"site {site} (`{sub.mech.sites[site]['pred']}`) observed true in "
1236
+ f"{len(seen[True])} and false in {len(seen[False])} execution(s); the predicate "
1237
+ f"agrees with the observed outcome {len(local_agree)} time(s) from observed "
1238
+ f"state and in {len(agree) - len(local_agree)} replayed/stated case(s), "
1239
+ "disagreeing in none; "
1240
+ "branch consequences consistent with static control facts; no contradiction"
1241
+ )
1242
+ else:
1243
+ missing = "true" if not seen[True] else "false"
1244
+ d.status_reason += f"; not verified: outcome {missing} never observed at {site}"
1245
+ # occurrence agreement (regions only): at each shown occurrence where a decision's site
1246
+ # was evaluated, its predicate on that occurrence's supplied facts must give the observed
1247
+ # outcome. A disagreement cannot be attributed to the predicate or to the supplied fact,
1248
+ # so the decision is CONTESTED: demoted to hypothesis (predictions that need it stop),
1249
+ # never rejected. Agreement everywhere counts as support for verification.
1250
+ if scenarios and g.regions:
1251
+ sk_occ = build_skeleton(
1252
+ [ob.execution for ob in sub.obs.by_test.values()], genome_vocabulary(g)
1253
+ )
1254
+ per: dict[str, list[tuple[str, int, bool]]] = {}
1255
+ for test, sc in scenarios.items():
1256
+ ob = sub.obs.get(test)
1257
+ if ob is None:
1258
+ continue
1259
+ for did, found in occurrence_agreement(g, sc, ob, sk_occ).items():
1260
+ per.setdefault(did, []).extend((test, i, same) for i, same in found)
1261
+ for d in g.decisions:
1262
+ rows = per.get(d.id, [])
1263
+ bad = [r for r in rows if not r[2]]
1264
+ if bad and d.status in (SUPPORTED, VERIFIED):
1265
+ d.status = HYPOTHESIS
1266
+ d.status_reason = (
1267
+ "contested: predicate disagrees with the observed outcome at "
1268
+ + ("; ".join(f"{t.split('::')[-1]}[{i}]" for t, i, _ in bad[:3]))
1269
+ + " given the supplied occurrence facts"
1270
+ )
1271
+ elif rows and d.status == SUPPORTED:
1272
+ d.status_reason += (
1273
+ f"; agrees with the observed outcome at {len(rows)} occurrence evaluation(s)"
1274
+ )
1275
+ # transitions: observed deltas at the entity's calls, through variable bindings. When
1276
+ # a procedure places the transition (T:id), it is checked where the genome says it
1277
+ # happens: the entity's procedure is run from the call's observed entry state, and the
1278
+ # transition is compared only if that run reaches it.
1279
+ kinds = {
1280
+ v.name: (bindings_of(v)[0].get("kind", "is_set") if bindings_of(v) else "is_set")
1281
+ for v in g.variables
1282
+ }
1283
+ step_lists = [p.steps for p in g.procedures] + [r.steps for r in g.regions]
1284
+ placed = {st[2:] for steps in step_lists for st in steps if st.startswith("T:")}
1285
+ placed |= {
1286
+ st[2:]
1287
+ for d in g.decisions
1288
+ for b in (d.true_branch, d.false_branch)
1289
+ for st in b.steps
1290
+ if st.startswith("T:")
1291
+ }
1292
+ for t in g.transitions:
1293
+ if t.status == REJECTED or not t.entity or not t.sets:
1294
+ continue
1295
+ consistent: list[str] = []
1296
+ contradicting: list[str] = []
1297
+ for test, ob in sub.obs.by_test.items():
1298
+ # global facts that hold one value throughout the execution are known at every
1299
+ # call in it (a setting read in a nested call is the same setting)
1300
+ seen_globals: dict[str, set[str]] = {}
1301
+ for m in ob.execution.nodes:
1302
+ if isinstance(m, CallNode):
1303
+ for a, b, _ in (*m.state, *m.state_after):
1304
+ if a.startswith("global."):
1305
+ seen_globals.setdefault(a, set()).add(b)
1306
+ constant = {a: next(iter(bs)) for a, bs in seen_globals.items() if len(bs) == 1}
1307
+ for n in ob.execution.nodes:
1308
+ if not (isinstance(n, CallNode) and _name_match(t.entity, n.symbol)):
1309
+ continue
1310
+ before = _bind(g, {**constant, **{a: b for a, b, _ in n.state}}, n, "entry", ob)
1311
+ after = _bind(g, {a: b for a, b, _ in n.state_after}, n, "exit")
1312
+ if not after:
1313
+ continue
1314
+ replayed: dict[str, Any] | None = None
1315
+ if t.id in placed:
1316
+ run = predict_sequence(
1317
+ g, {"state": dict(before), "calls": [{"entity": t.entity}]}, "hypothesis"
1318
+ )
1319
+ if not any(e[0] == "set" and e[3] == t.id for e in run.events):
1320
+ if run.indeterminate is None:
1321
+ continue # the genome says this call does not reach it
1322
+ # the genome cannot decide the path from the entry state alone
1323
+ # (decisions over unbound inputs): follow the path that executed
1324
+ events, _, problem = replay_observed(g, t.entity, ob, n.id, dict(before))
1325
+ sets = [e for e in events if e[0] == "set" and e[3] == t.id]
1326
+ if problem is not None or not sets:
1327
+ continue
1328
+ replayed = {e[1]: e[2] for e in sets}
1329
+ elif eval_predicate(t.when, before) is not True:
1330
+ continue
1331
+ for var, expr in t.sets.items():
1332
+ if var not in after:
1333
+ continue
1334
+ if replayed is not None:
1335
+ if var not in replayed:
1336
+ continue
1337
+ pred = replayed[var]
1338
+ else:
1339
+ pred = eval_expr(expr, before)
1340
+ if pred is None and expr.strip() not in ("none", "None", "null"):
1341
+ continue
1342
+ if _abstract_equal(kinds.get(var, "is_set"), pred, after[var]):
1343
+ consistent.append(test)
1344
+ else:
1345
+ contradicting.append(
1346
+ f"{test}: {var} observed {after[var]!r}, predicted {pred!r}"
1347
+ )
1348
+ if contradicting:
1349
+ t.status = REJECTED
1350
+ t.status_reason = "observed deltas contradict: " + "; ".join(contradicting[:3])
1351
+ elif consistent and t.status == SUPPORTED:
1352
+ t.status = VERIFIED
1353
+ t.status_reason = (
1354
+ f"observed deltas consistent in {len(set(consistent))} execution(s); "
1355
+ "none contradict"
1356
+ + (" (checked where the procedure applies it)" if t.id in placed else "")
1357
+ )
1358
+ elif not consistent:
1359
+ t.status_reason += "; not verified: no observed call of the entity with a bound delta"
1360
+ # procedure outcomes: where an observed call completes the procedure's path (replayed
1361
+ # along what executed, without a stopping branch), the named entity's exit must match
1362
+ for p in g.procedures:
1363
+ if not p.outcome or not p.outcome.get("is") or p.status == REJECTED:
1364
+ continue
1365
+ target = p.outcome.get("entity") or p.entity
1366
+ agreeing, disagreeing = 0, []
1367
+ for test, ob in sub.obs.by_test.items():
1368
+ for n in ob.execution.nodes:
1369
+ if not (isinstance(n, CallNode) and _name_match(p.entity, n.symbol)):
1370
+ continue
1371
+ env = _bind(g, {a: b for a, b, _ in n.state}, n, "entry")
1372
+ _, stopped, problem = replay_observed(g, p.entity, ob, n.id, env)
1373
+ if stopped or problem is not None:
1374
+ continue
1375
+ call = _enclosing_call(ob, n.id, target)
1376
+ ok = exit_matches(p.outcome["is"], call.outcome) if call is not None else None
1377
+ if ok is True:
1378
+ agreeing += 1
1379
+ elif ok is False:
1380
+ disagreeing.append(f"{test.split('::')[-1]}: {target} ended {call.outcome}")
1381
+ if disagreeing:
1382
+ p.status = REJECTED
1383
+ p.status_reason = f"observed exit contradicts outcome {p.outcome['is']}: " + "; ".join(
1384
+ disagreeing[:3]
1385
+ )
1386
+ elif agreeing:
1387
+ p.status_reason += f"; outcome {p.outcome['is']} observed in {agreeing} call(s)"
1388
+ # execution structure: where a procedure places calls and decisions, and their order,
1389
+ # judged against the skeleton of the observed executions
1390
+ sk = build_skeleton([ob.execution for ob in sub.obs.by_test.values()], genome_vocabulary(g))
1391
+ owners_by_id: dict[str, Item] = {q.id: q for q in (*g.procedures, *g.regions)}
1392
+ for c in procedure_structure_claims(g, sk):
1393
+ owner = owners_by_id[c["procedure"]]
1394
+ if c["status"] == REJECTED and owner.status != REJECTED:
1395
+ owner.status = REJECTED
1396
+ owner.status_reason = f"structure contradicted: {c['claim']}: {c['why']}"
1397
+ return g
1398
+
1399
+
1400
+ # --------------------------------------------------------------------------- path conditions
1401
+
1402
+
1403
+ def path_condition(ob: ExecObs, sites: set[str], mech: Mechanics) -> list[dict[str, Any]]:
1404
+ """The observed branch vector over `sites`, in order, with each site's predicate. An
1405
+ observed fact; a symbolic reading requires the predicate to be representable."""
1406
+ return [
1407
+ {
1408
+ "site": b.site,
1409
+ "outcome": b.outcome,
1410
+ "pred": mech.sites.get(b.site, {}).get("pred"),
1411
+ "in": b.symbol,
1412
+ }
1413
+ for b in ob.branches
1414
+ if b.site in sites
1415
+ ]
1416
+
1417
+
1418
+ def regimes(obs: Observations, sites: set[str]) -> dict[tuple[tuple[str, bool], ...], list[str]]:
1419
+ """Executions grouped by identical branch vectors over `sites`."""
1420
+ out: dict[tuple[tuple[str, bool], ...], list[str]] = {}
1421
+ for test, ob in obs.by_test.items():
1422
+ key = tuple((b.site, b.outcome) for b in ob.branches if b.site in sites)
1423
+ out.setdefault(key, []).append(test)
1424
+ return out
1425
+
1426
+
1427
+ # --------------------------------------------------------------------------- sequence prediction
1428
+
1429
+
1430
+ @dataclass
1431
+ class SeqPrediction:
1432
+ events: list[tuple[Any, ...]] = field(
1433
+ default_factory=list
1434
+ ) # ("call", e) | ("branch", site, v) | ("set", var, val)
1435
+ state: dict[str, Any] = field(default_factory=dict)
1436
+ indeterminate: str | None = None
1437
+ # the predicted occurrence tree: {"entity", "parent": index | None,
1438
+ # "events": [("call", entity) | ("branch", "site:<id>")]} (see diffgenome.structure)
1439
+ occurrences: list[dict[str, Any]] = field(default_factory=list)
1440
+
1441
+
1442
+ def predict_sequence(
1443
+ g: Genome, scenario: dict[str, Any], min_status: str = SUPPORTED, max_depth: int = 12
1444
+ ) -> SeqPrediction:
1445
+ """Predict a sequence of calls under an initial state. Each call runs the entity's
1446
+ procedure (or, without one, its decisions in order); transitions change the state that
1447
+ later decisions read. Unestablished items that are needed stop the prediction."""
1448
+ decisions = {d.id: d for d in g.decisions}
1449
+ transitions = {t.id: t for t in g.transitions}
1450
+ procedures = {p.entity: p for p in g.procedures}
1451
+ regions = {r.id: r for r in g.regions}
1452
+ # occurrence facts are local to one occurrence; state-bound variables persist
1453
+ state_vars = {v.name for v in g.variables if any(b.get("fact") for b in bindings_of(v))}
1454
+ ctx: list[dict[str, Any]] = [] # the scenario item whose occurrences a region reads
1455
+ pred = SeqPrediction(state=dict(scenario.get("state") or {}))
1456
+
1457
+ def usable(it: Item) -> bool:
1458
+ # eligibility: a supported item contradicted on a shown execution may not drive a
1459
+ # prediction; diagnostics that ask for hypothesis-level prediction still use it
1460
+ if it.contradicted and it.status != VERIFIED and min_status != HYPOTHESIS:
1461
+ return False
1462
+ return RANK.get(it.status, 0) >= RANK[min_status]
1463
+
1464
+ def entity_key(e: str) -> str | None:
1465
+ for k in procedures:
1466
+ if k == e or k.endswith("." + e) or e.endswith("." + k):
1467
+ return k
1468
+ return None
1469
+
1470
+ def new_occurrence(entity: str, parent: int | None) -> int:
1471
+ pred.occurrences.append({"entity": entity, "parent": parent, "events": []})
1472
+ if parent is not None:
1473
+ pred.occurrences[parent]["events"].append(("call", entity))
1474
+ return len(pred.occurrences) - 1
1475
+
1476
+ def run_steps(steps: list[str], env: dict[str, Any], depth: int, occ: int) -> bool:
1477
+ """Returns True if the enclosing entity stops."""
1478
+ for st in steps:
1479
+ if pred.indeterminate:
1480
+ return True
1481
+ kind, _, ref = st.partition(":")
1482
+ if kind == "call":
1483
+ pred.events.append(("call", ref))
1484
+ key = entity_key(ref)
1485
+ child = new_occurrence(key or ref, occ)
1486
+ if key is not None:
1487
+ run_entity(key, env, depth + 1, child)
1488
+ elif kind == "T":
1489
+ t = transitions.get(ref)
1490
+ if t is None or not usable(t):
1491
+ pred.indeterminate = (
1492
+ f"transition {ref} is {'missing' if t is None else t.status}"
1493
+ )
1494
+ return True
1495
+ w = eval_predicate(t.when, derive(g, env))
1496
+ if w is None:
1497
+ pred.indeterminate = f"{ref}: `{t.when}` needs facts it was not given"
1498
+ return True
1499
+ if w:
1500
+ for var, expr in t.sets.items():
1501
+ val = eval_expr(expr, env)
1502
+ if val is None and expr.strip() not in ("none", "None", "null"):
1503
+ pred.indeterminate = f"{ref}: cannot evaluate {var} := {expr}"
1504
+ return True
1505
+ env[var] = val
1506
+ pred.events.append(("set", var, val, ref))
1507
+ elif kind == "R":
1508
+ r = regions.get(ref)
1509
+ if r is None or not usable(r):
1510
+ pred.indeterminate = f"region {ref} is {'missing' if r is None else r.status}"
1511
+ return True
1512
+ items = (ctx[-1].get("occurrences") or {}).get(ref) if ctx else None
1513
+ if items is None:
1514
+ pred.indeterminate = f"occurrence facts not supplied for region {ref}"
1515
+ return True
1516
+ for k, item in enumerate(items):
1517
+ before = dict(env)
1518
+ local = item.get("facts") or {}
1519
+ env.update(local)
1520
+ pred.events.append(("occurrence", ref, k))
1521
+ ctx.append(item)
1522
+ stopped = run_steps(r.steps, env, depth, occ)
1523
+ ctx.pop()
1524
+ # the occurrence's supplied facts are local to it; what the body's
1525
+ # transitions set (e.g. "a match was found") is its effect and persists
1526
+ for var in local:
1527
+ if var in state_vars:
1528
+ continue
1529
+ if var in before:
1530
+ env[var] = before[var]
1531
+ else:
1532
+ env.pop(var, None)
1533
+ if stopped: # `stops` ends the enclosing entity; there is no `continue`
1534
+ return True
1535
+ elif kind == "D":
1536
+ d = decisions.get(ref)
1537
+ if d is None or not usable(d):
1538
+ why = (
1539
+ "missing"
1540
+ if d is None
1541
+ else (
1542
+ f"{d.status}, contradicted on a shown execution"
1543
+ if d.contradicted and d.status != VERIFIED
1544
+ else d.status
1545
+ )
1546
+ )
1547
+ pred.indeterminate = f"decision {ref} is {why}"
1548
+ return True
1549
+ v = eval_predicate(d.predicate, derive(g, env))
1550
+ if v is None:
1551
+ pred.indeterminate = f"{ref}: `{d.predicate}` needs facts it was not given"
1552
+ return True
1553
+ if d.site:
1554
+ pred.events.append(("branch", d.site, v))
1555
+ pred.occurrences[occ]["events"].append(("branch", f"site:{d.site}"))
1556
+ b = d.true_branch if v else d.false_branch
1557
+ if b.outcome and b.outcome.get("is"):
1558
+ pred.events.append(
1559
+ ("outcome", b.outcome.get("entity") or d.entity, b.outcome["is"])
1560
+ )
1561
+ inner = b.steps # `calls` are descriptive (design-value-identity.md, 4)
1562
+ if run_steps(inner, env, depth, occ) or b.stops:
1563
+ return True
1564
+ else:
1565
+ pred.indeterminate = f"unknown step {st!r}"
1566
+ return True
1567
+ return False
1568
+
1569
+ def run_entity(key: str, env: dict[str, Any], depth: int, occ: int) -> None:
1570
+ if depth > max_depth:
1571
+ pred.indeterminate = f"depth limit at {key}"
1572
+ return
1573
+ p = procedures[key]
1574
+ if not usable(p):
1575
+ pred.indeterminate = f"procedure {p.id} ({key}) is {p.status}"
1576
+ return
1577
+ stopped = run_steps(p.steps, env, depth, occ)
1578
+ if not stopped and not pred.indeterminate and p.outcome and p.outcome.get("is"):
1579
+ pred.events.append(("outcome", p.outcome.get("entity") or p.entity, p.outcome["is"]))
1580
+
1581
+ env = pred.state
1582
+ for call in scenario.get("calls") or []:
1583
+ env.update(call.get("facts") or {})
1584
+ e = call["entity"]
1585
+ pred.events.append(("call", e))
1586
+ key = entity_key(e)
1587
+ top = new_occurrence(key or e, None)
1588
+ ctx[:] = [call]
1589
+ if key is None:
1590
+ # an entity the genome says nothing about is a call with no modeled behavior;
1591
+ # one the genome has decisions for but no procedure cannot be predicted
1592
+ if any(
1593
+ d.entity == e or d.entity.endswith("." + e) or e.endswith("." + d.entity)
1594
+ for d in g.decisions
1595
+ ):
1596
+ pred.indeterminate = f"no procedure for {e}, which has decisions"
1597
+ break
1598
+ continue
1599
+ run_entity(key, env, 0, top)
1600
+ if pred.indeterminate:
1601
+ break
1602
+ return pred
1603
+
1604
+
1605
+ def replay_observed(
1606
+ g: Genome, entity: str, ob: ExecObs, node_id: int, env: dict[str, Any], depth: int = 0
1607
+ ) -> tuple[list[tuple[Any, ...]], bool, str | None]:
1608
+ """Run an entity's procedure for ONE observed call, guided by what executed.
1609
+
1610
+ Each decision takes the outcomes observed at its site in that call (in order). A
1611
+ decision whose predicate is atomic (`v` or `!v`) binds v accordingly. A `call:X` step
1612
+ descends into the next observed call of X beneath this one. Transitions apply when their
1613
+ `when` holds in the resulting environment. Nothing is guessed: a missing observation or
1614
+ an undecidable `when` ends the replay with a problem.
1615
+
1616
+ Returns (events, stopped, problem). Events: ("branch", site, v), ("set", var, value,
1617
+ transition id, node), ("outcome", entity, exit, node)."""
1618
+ procedures = {p.entity: p for p in g.procedures}
1619
+ decisions = {d.id: d for d in g.decisions}
1620
+ transitions = {t.id: t for t in g.transitions}
1621
+
1622
+ def key_of(e: str) -> str | None:
1623
+ return next(
1624
+ (k for k in procedures if k == e or k.endswith("." + e) or e.endswith("." + k)), None
1625
+ )
1626
+
1627
+ key = key_of(entity)
1628
+ if key is None:
1629
+ return [], False, f"no procedure for {entity}"
1630
+ nodes = ob.execution.nodes
1631
+ kids = ob.children_of(node_id)
1632
+ pending: dict[str, list[bool]] = {}
1633
+ for b in ob.branches:
1634
+ if b.node == node_id:
1635
+ pending.setdefault(b.site, []).append(b.outcome)
1636
+
1637
+ def subtree(n: int) -> list[int]:
1638
+ out: list[int] = []
1639
+ stack = list(reversed(kids.get(n, [])))
1640
+ while stack:
1641
+ k = stack.pop()
1642
+ out.append(k)
1643
+ stack.extend(reversed(kids.get(k, [])))
1644
+ return out
1645
+
1646
+ below = sorted(subtree(node_id))
1647
+ cursor = -1
1648
+ events: list[tuple[Any, ...]] = []
1649
+
1650
+ def run(steps: list[str]) -> tuple[bool, str | None]:
1651
+ nonlocal cursor
1652
+ for st in steps:
1653
+ kind, _, ref = st.partition(":")
1654
+ if kind == "call":
1655
+ nxt = next(
1656
+ (
1657
+ k
1658
+ for k in below
1659
+ if k > cursor
1660
+ and isinstance(nodes[k], CallNode)
1661
+ and _name_match(ref, _symbol(nodes[k]))
1662
+ ),
1663
+ None,
1664
+ )
1665
+ if nxt is None:
1666
+ return False, f"{ref} was not observed under this call"
1667
+ inner = subtree(nxt)
1668
+ cursor = max([nxt, *inner])
1669
+ if key_of(ref) is not None and depth < 12:
1670
+ ev, _, prob = replay_observed(g, ref, ob, nxt, env, depth + 1)
1671
+ events.extend(ev)
1672
+ if prob is not None:
1673
+ return False, prob
1674
+ elif kind == "D":
1675
+ d = decisions.get(ref)
1676
+ if d is None or not d.site:
1677
+ return False, f"decision {ref} is missing or has no site"
1678
+ queue = pending.get(d.site) or []
1679
+ if not queue:
1680
+ return False, f"{ref}: no observed evaluation of {d.site} in this call"
1681
+ v = queue.pop(0)
1682
+ events.append(("branch", d.site, v))
1683
+ m = re.fullmatch(r"\s*(!?)\s*([\w.]+)\s*", d.predicate)
1684
+ if m and m.group(2) not in ("true", "false"):
1685
+ env[m.group(2)] = (not v) if m.group(1) else v
1686
+ b = d.true_branch if v else d.false_branch
1687
+ if b.outcome and b.outcome.get("is"):
1688
+ events.append(
1689
+ ("outcome", b.outcome.get("entity") or d.entity, b.outcome["is"], node_id)
1690
+ )
1691
+ stopped, prob = run(b.steps)
1692
+ if prob is not None:
1693
+ return stopped, prob
1694
+ if stopped or b.stops:
1695
+ return True, None
1696
+ elif kind == "R":
1697
+ return False, f"region {ref}: repetitions are not replayed"
1698
+ elif kind == "T":
1699
+ t = transitions.get(ref)
1700
+ if t is None:
1701
+ return False, f"transition {ref} is missing"
1702
+ w = eval_predicate(t.when, derive(g, env))
1703
+ if w is None:
1704
+ return False, f"{ref}: `{t.when}` is undecidable on the observed path"
1705
+ if w:
1706
+ for var, expr in t.sets.items():
1707
+ val = eval_expr(expr, env)
1708
+ if val is None and expr.strip() not in ("none", "None", "null"):
1709
+ return False, f"{ref}: cannot evaluate {var} := {expr}"
1710
+ env[var] = val
1711
+ events.append(("set", var, val, ref, node_id))
1712
+ else:
1713
+ return False, f"unknown step {st!r}"
1714
+ return False, None
1715
+
1716
+ p = procedures[key]
1717
+ stopped, problem = run(p.steps)
1718
+ if not stopped and problem is None and p.outcome and p.outcome.get("is"):
1719
+ events.append(("outcome", p.outcome.get("entity") or p.entity, p.outcome["is"], node_id))
1720
+ return events, stopped, problem
1721
+
1722
+
1723
+ def _occurrence_values(
1724
+ g: Genome,
1725
+ body: list[str],
1726
+ rep: dict[str, Any],
1727
+ occs: dict[int, Any],
1728
+ ob: ExecObs,
1729
+ vocab: list[str],
1730
+ ) -> dict[str, Any]:
1731
+ """What one observed repetition shows about the genome's variables: boundary facts of
1732
+ the calls inside it (boundary-bound variables), and the variables that atomic decisions
1733
+ reached from the region body (`v` / `!v`) bind from their outcomes in it."""
1734
+ out: dict[str, Any] = {}
1735
+ nodes = ob.execution.nodes
1736
+ inside = [occs[i] for i in rep["occurrences"]]
1737
+ for v in g.variables:
1738
+ for b in bindings_of(v):
1739
+ at = b.get("at")
1740
+ if not isinstance(at, dict) or v.name in out:
1741
+ continue
1742
+ if b.get("kind") == "equals_literal" and b.get("_literal_ok") is not True:
1743
+ continue
1744
+ target = resolve(vocab, str(at.get("entity", "")))
1745
+ vals = [
1746
+ boundary_fact(
1747
+ nodes[o.id],
1748
+ str(at.get("point", "")),
1749
+ str(b.get("kind", "is_set")),
1750
+ b.get("literal"),
1751
+ )
1752
+ for o in inside
1753
+ if o.entity == target
1754
+ ]
1755
+ vals = [x for x in vals if x is not None]
1756
+ if vals and all(x == vals[0] for x in vals[1:]):
1757
+ out[v.name] = vals[0]
1758
+ procedures = {p.entity: p for p in g.procedures}
1759
+ decisions = {d.id: d for d in g.decisions}
1760
+ first: dict[str, bool] = {}
1761
+ for site, outcome in rep["branches"]:
1762
+ first.setdefault(site, outcome)
1763
+ seen: set[str] = set()
1764
+ stack = [list(body)]
1765
+ while stack:
1766
+ for st in stack.pop():
1767
+ kind, _, ref = st.partition(":")
1768
+ if kind == "call":
1769
+ keys = [k for k in procedures if k == ref or k.endswith("." + ref)]
1770
+ keys += [k for k in procedures if ref.endswith("." + k)]
1771
+ key = keys[0] if keys else None
1772
+ if key is not None and key not in seen:
1773
+ seen.add(key)
1774
+ stack.append(list(procedures[key].steps))
1775
+ elif kind == "D" and ref in decisions and ref not in seen:
1776
+ seen.add(ref)
1777
+ d = decisions[ref]
1778
+ m = re.fullmatch(r"\s*(!?)\s*([\w.]+)\s*", d.predicate)
1779
+ if d.site in first and m and m.group(2) not in ("true", "false"):
1780
+ out.setdefault(m.group(2), (not first[d.site]) if m.group(1) else first[d.site])
1781
+ for br in (d.true_branch, d.false_branch):
1782
+ stack.append(br.steps)
1783
+ return out
1784
+
1785
+
1786
+ def _subtree_values(g: Genome, ob: ExecObs, n: Any, names: set[str]) -> dict[str, Any]:
1787
+ """Values of boundary-bound variables at calls running inside call `n`, when unique."""
1788
+ out: dict[str, Any] = {}
1789
+ if not names:
1790
+ return out
1791
+ for v in g.variables:
1792
+ if v.name not in names:
1793
+ continue
1794
+ for bd in bindings_of(v):
1795
+ at = bd.get("at")
1796
+ if not isinstance(at, dict) or v.name in out:
1797
+ continue
1798
+ kind = str(bd.get("kind", "is_set"))
1799
+ if kind == "equals_literal" and bd.get("_literal_ok") is not True:
1800
+ continue
1801
+ vals = [
1802
+ boundary_fact(m, str(at.get("point", "")), kind, bd.get("literal"))
1803
+ for m in ob.execution.nodes
1804
+ if isinstance(m, CallNode)
1805
+ and m.id != n.id
1806
+ and _name_match(str(at.get("entity", "")), m.symbol)
1807
+ and _is_ancestor(ob, n.id, m.id)
1808
+ ]
1809
+ vals = [x for x in vals if x is not None]
1810
+ if vals and all(x == vals[0] for x in vals[1:]):
1811
+ out[v.name] = vals[0]
1812
+ return out
1813
+
1814
+
1815
+ def check_scenario_call_facts(
1816
+ g: Genome, scenario: dict[str, Any], ob: ExecObs
1817
+ ) -> list[dict[str, Any]]:
1818
+ """Facts a scenario states for one of its calls, for variables bound at that call's own
1819
+ boundary or within the execution (scope "execution"), compared with the aligned observed
1820
+ call (the k-th call of that entity). Identity variables carry labels, never decoded: the
1821
+ label pattern must equal the identity pattern (same label iff same identity)."""
1822
+ out: list[dict[str, Any]] = []
1823
+ pairs: list[tuple[str, Any, Any]] = []
1824
+ counts: dict[str, int] = {}
1825
+ kinds = {
1826
+ v.name: (bindings_of(v)[0].get("kind", "is_set") if bindings_of(v) else "is_set")
1827
+ for v in g.variables
1828
+ }
1829
+ for call in scenario.get("calls") or []:
1830
+ e = call["entity"]
1831
+ k = counts.get(e, 0)
1832
+ counts[e] = k + 1
1833
+ facts = call.get("facts") or {}
1834
+ if not facts:
1835
+ continue
1836
+ nodes = [
1837
+ n for n in ob.execution.nodes if isinstance(n, CallNode) and _name_match(e, n.symbol)
1838
+ ]
1839
+ if k >= len(nodes):
1840
+ continue
1841
+ n = nodes[k]
1842
+ seen = {
1843
+ **_bind(g, {a: b for a, b, _ in n.state}, n, "entry", ob),
1844
+ **_bind(g, {a: b for a, b, _ in n.state_after}, n, "exit", ob),
1845
+ }
1846
+ # a fact of this call may concern a boundary INSIDE it (a callee's argument or
1847
+ # result); it is checked there when that boundary's value is unique within the call
1848
+ inner = _subtree_values(g, ob, n, set(facts) - set(seen))
1849
+ seen = {**inner, **seen}
1850
+ for var, val in facts.items():
1851
+ if var not in seen:
1852
+ continue
1853
+ if isinstance(seen[var], Identity):
1854
+ pairs.append((var, val, seen[var]))
1855
+ continue
1856
+ same = _abstract_equal(kinds.get(var, "is_set"), val, seen[var])
1857
+ out.append(
1858
+ {"call": e, "fact": var, "supplied": val, "observed": seen[var],
1859
+ "status": "confirmed" if same else "contradicted"}
1860
+ ) # fmt: skip
1861
+ for i, (va, la, ia) in enumerate(pairs):
1862
+ for vb, lb, ib in pairs[i + 1 :]:
1863
+ consistent = (la == lb) == (ia == ib)
1864
+ out.append(
1865
+ {"fact": f"{va}={la!r} vs {vb}={lb!r}",
1866
+ "status": "confirmed" if consistent else "contradicted",
1867
+ "why": "" if consistent else (
1868
+ "labels equal but identities differ" if la == lb
1869
+ else "labels differ but identities are equal")}
1870
+ ) # fmt: skip
1871
+ return out
1872
+
1873
+
1874
+ def check_scenario_occurrences(
1875
+ g: Genome, scenario: dict[str, Any], ob: ExecObs, sk: Skeleton
1876
+ ) -> list[dict[str, Any]]:
1877
+ """Align a scenario's supplied region occurrences with the observed repetitions (the
1878
+ k-th supplied occurrence with the k-th observed repetition of that region, inside the
1879
+ matching observed call) and check each supplied fact. Statuses: confirmed, contradicted,
1880
+ unobservable. A supplied count that differs from the observed count is contradicted."""
1881
+ vocab = sorted(set(genome_vocabulary(g)) | set(sk.seen))
1882
+ occs = occurrences(ob.execution, vocab)
1883
+ regions = {r.id: r for r in g.regions}
1884
+ out: list[dict[str, Any]] = []
1885
+ seen_calls: dict[str, int] = {}
1886
+ for call in scenario.get("calls") or []:
1887
+ e = resolve(vocab, call["entity"])
1888
+ k = seen_calls.get(e, 0)
1889
+ seen_calls[e] = k + 1
1890
+ supplied = call.get("occurrences") or {}
1891
+ if not supplied:
1892
+ continue
1893
+ cands = sorted((o for o in occs.values() if o.entity == e), key=lambda o: o.id)
1894
+ if k >= len(cands):
1895
+ out.append(
1896
+ {"call": e, "status": "contradicted", "why": f"no observed call #{k} of {e}"}
1897
+ )
1898
+ continue
1899
+ for rid, items in supplied.items():
1900
+ r = regions.get(rid)
1901
+ if r is None:
1902
+ continue
1903
+ head = _region_head(r, vocab)
1904
+ observed = [
1905
+ x for x in sk.regions.get(resolve(vocab, r.entity), []) if x["head"] == head
1906
+ ]
1907
+ if not observed:
1908
+ out.append({"region": rid, "status": "unobservable", "why": "no observed region"})
1909
+ continue
1910
+ # the region lives in its own entity: the scenario call's occurrence itself, or
1911
+ # the one occurrence of that entity running during it
1912
+ r_entity = resolve(vocab, r.entity)
1913
+ call_occ = cands[k]
1914
+ homes = [
1915
+ o
1916
+ for o in occs.values()
1917
+ if o.entity == r_entity
1918
+ and (o.id == call_occ.id or call_occ.entity in o.ancestors)
1919
+ and call_occ.first <= o.id <= call_occ.last
1920
+ ]
1921
+ if len(homes) != 1:
1922
+ out.append(
1923
+ {"region": rid, "status": "unobservable",
1924
+ "why": f"{len(homes)} occurrence(s) of {r_entity} in this call"}
1925
+ ) # fmt: skip
1926
+ continue
1927
+ reps = repetitions(occs, homes[0].id, observed[0]["members"], head)
1928
+ if len(reps) != len(items):
1929
+ out.append(
1930
+ {"region": rid, "status": "contradicted",
1931
+ "why": f"{len(items)} occurrence(s) supplied, {len(reps)} observed"}
1932
+ ) # fmt: skip
1933
+ for i, (item, rep) in enumerate(zip(items, reps, strict=False)):
1934
+ seen_vals = _occurrence_values(g, r.steps, rep, occs, ob, vocab)
1935
+ for var, val in (item.get("facts") or {}).items():
1936
+ if var not in seen_vals:
1937
+ status = "unobservable"
1938
+ elif seen_vals[var] == val:
1939
+ status = "confirmed"
1940
+ else:
1941
+ status = "contradicted"
1942
+ out.append(
1943
+ {"region": rid, "occurrence": i, "fact": var, "supplied": val,
1944
+ "observed": seen_vals.get(var), "status": status}
1945
+ ) # fmt: skip
1946
+ return out
1947
+
1948
+
1949
+ def occurrence_agreement(
1950
+ g: Genome, scenario: dict[str, Any], ob: ExecObs, sk: Skeleton
1951
+ ) -> dict[str, list[tuple[int, bool]]]:
1952
+ """Per decision: at every evaluation of its site inside an aligned occurrence of a
1953
+ region, (occurrence ordinal, whether its predicate on that occurrence's supplied facts,
1954
+ plus the scenario call's facts, gives the observed outcome). Only meaningful where the
1955
+ checker can see the execution (shown tests)."""
1956
+ vocab = sorted(set(genome_vocabulary(g)) | set(sk.seen))
1957
+ occs = occurrences(ob.execution, vocab)
1958
+ regions = {r.id: r for r in g.regions}
1959
+ decisions = [d for d in g.decisions if d.site]
1960
+ out: dict[str, list[tuple[int, bool]]] = {}
1961
+ seen_calls: dict[str, int] = {}
1962
+ for call in scenario.get("calls") or []:
1963
+ e = resolve(vocab, call["entity"])
1964
+ k = seen_calls.get(e, 0)
1965
+ seen_calls[e] = k + 1
1966
+ cands = sorted((o for o in occs.values() if o.entity == e), key=lambda o: o.id)
1967
+ if k >= len(cands):
1968
+ continue
1969
+ call_occ = cands[k]
1970
+ for rid, items in (call.get("occurrences") or {}).items():
1971
+ r = regions.get(rid)
1972
+ if r is None:
1973
+ continue
1974
+ head = _region_head(r, vocab)
1975
+ observed = [
1976
+ x for x in sk.regions.get(resolve(vocab, r.entity), []) if x["head"] == head
1977
+ ]
1978
+ homes = [
1979
+ o for o in occs.values()
1980
+ if o.entity == resolve(vocab, r.entity)
1981
+ and (o.id == call_occ.id or call_occ.entity in o.ancestors)
1982
+ and call_occ.first <= o.id <= call_occ.last
1983
+ ] # fmt: skip
1984
+ if not observed or len(homes) != 1:
1985
+ continue
1986
+ reps = repetitions(occs, homes[0].id, observed[0]["members"], head)
1987
+ if len(reps) != len(items):
1988
+ continue
1989
+ for i, (item, rp) in enumerate(zip(items, reps, strict=True)):
1990
+ env = derive(g, {**(call.get("facts") or {}), **(item.get("facts") or {})})
1991
+ for d in decisions:
1992
+ outcomes = [v for site, v in rp["branches"] if site == d.site]
1993
+ if not outcomes:
1994
+ continue
1995
+ val = eval_predicate(d.predicate, env)
1996
+ if val is None:
1997
+ continue
1998
+ out.setdefault(d.id, []).extend((i, val == o) for o in outcomes)
1999
+ return out
2000
+
2001
+
2002
+ def change_sites(mech: Mechanics, obs: Observations, symbols: list[str]) -> set[str]:
2003
+ """Every decision site of the given functions and of functions nested in them: their
2004
+ static sites, plus sites observed at runtime inside calls of them. A genome is scored
2005
+ on all of these, not only on the sites it chooses to cover."""
2006
+ tails = [s.split(":", 1)[-1] for s in symbols]
2007
+
2008
+ def mine(sym: str) -> bool:
2009
+ t = sym.split(":", 1)[-1]
2010
+ return any(t == x or t.startswith(x + ".<locals>.") for x in tails)
2011
+
2012
+ out = {sid for sid, rec in mech.sites.items() if mine(rec["symbol"])}
2013
+ for ob in obs.by_test.values():
2014
+ out |= {b.site for b in ob.branches if mine(b.symbol)}
2015
+ return out
2016
+
2017
+
2018
+ def compare_sequence(
2019
+ g: Genome,
2020
+ pred: SeqPrediction,
2021
+ ob: ExecObs,
2022
+ sites: set[str],
2023
+ required_sites: set[str] | None = None,
2024
+ skeleton: Skeleton | None = None,
2025
+ scenario: dict[str, Any] | None = None,
2026
+ ) -> dict[str, Any]:
2027
+ """Predicted branch events vs the observed branch vector over the genome's sites (exact,
2028
+ in order), and the predicted final state vs the last observed exit state of each bound
2029
+ variable. With `required_sites` (e.g. `change_sites`), an observed evaluation of a
2030
+ required site that no decision covers makes the prediction incomplete: indeterminate,
2031
+ never a match."""
2032
+ uncovered = sorted(
2033
+ {b.site for b in ob.branches if required_sites and b.site in required_sites} - sites
2034
+ )
2035
+ observed_vec = [(b.site, b.outcome) for b in ob.branches if b.site in sites]
2036
+ predicted_vec = [(e[1], e[2]) for e in pred.events if e[0] == "branch"]
2037
+ last: dict[str, str] = {}
2038
+ for n in ob.execution.nodes:
2039
+ if isinstance(n, CallNode) and n.state_after:
2040
+ last.update({a: b for a, b, _ in n.state_after if a.startswith(("self.", "global."))})
2041
+ observed_state = _bind(g, last)
2042
+ kinds = {
2043
+ v.name: (bindings_of(v)[0].get("kind", "is_set") if bindings_of(v) else "is_set")
2044
+ for v in g.variables
2045
+ }
2046
+ state_diff = {
2047
+ k: (pred.state.get(k), v)
2048
+ for k, v in observed_state.items()
2049
+ if k in pred.state and not _abstract_equal(kinds.get(k, "is_set"), pred.state[k], v)
2050
+ }
2051
+ # predicted exits, per entity, against the exits of that entity's observed calls: every
2052
+ # observed call must match a predicted exit and every predicted exit an observed call
2053
+ predicted_exits: dict[str, list[str]] = {}
2054
+ for e in pred.events:
2055
+ if e[0] == "outcome":
2056
+ predicted_exits.setdefault(e[1], []).append(e[2])
2057
+ outcome_diff: dict[str, Any] = {}
2058
+ for ent, claims in predicted_exits.items():
2059
+ seen = [
2060
+ n.outcome
2061
+ for n in ob.execution.nodes
2062
+ if isinstance(n, CallNode) and _name_match(ent, n.symbol)
2063
+ ]
2064
+ unmatched_obs = [o for o in seen if not any(exit_matches(c, o) for c in claims)]
2065
+ unmatched_pred = [c for c in claims if not any(exit_matches(c, o) for o in seen)]
2066
+ if unmatched_obs or unmatched_pred:
2067
+ outcome_diff[ent] = {"predicted": sorted(set(claims)), "observed": sorted(set(seen))}
2068
+ # observed execution structure: a prediction that places a call or a decision where it
2069
+ # was never observed (hoisted out of a modeled caller, or evaluated in the wrong call)
2070
+ # predicts a structure the evidence does not have: indeterminate, never a guess
2071
+ placement: list[str] = []
2072
+ if skeleton is not None:
2073
+ vocab = sorted(skeleton.seen)
2074
+ tree = [{**o, "entity": resolve(vocab, o["entity"])} for o in pred.occurrences]
2075
+ placement = placement_problems(tree, skeleton)
2076
+ # supplied occurrence facts, where the checker can see the execution: a scenario that is
2077
+ # wrong about its own input cannot support a prediction
2078
+ occ_checks: list[dict[str, Any]] = []
2079
+ if skeleton is not None and scenario is not None:
2080
+ occ_checks = check_scenario_call_facts(g, scenario, ob)
2081
+ if g.regions:
2082
+ occ_checks += check_scenario_occurrences(g, scenario, ob, skeleton)
2083
+ contradicted = [c for c in occ_checks if c["status"] == "contradicted"]
2084
+ indeterminate = (
2085
+ pred.indeterminate
2086
+ or (f"no decision covers observed site(s) {', '.join(uncovered)}" if uncovered else None)
2087
+ or (f"placement: {'; '.join(placement[:3])}" if placement else None)
2088
+ or (
2089
+ "occurrence facts contradicted: "
2090
+ + "; ".join(
2091
+ f"{c.get('region')}[{c.get('occurrence', '?')}] {c.get('fact', '')} "
2092
+ f"{c.get('why', '')}".strip()
2093
+ for c in contradicted[:3]
2094
+ )
2095
+ if contradicted
2096
+ else None
2097
+ )
2098
+ )
2099
+ return {
2100
+ "execution": ob.execution.stimulus_ref,
2101
+ "indeterminate": indeterminate,
2102
+ "uncovered_sites": uncovered,
2103
+ "placement_problems": placement,
2104
+ "occurrence_checks": {
2105
+ st: sum(1 for c in occ_checks if c["status"] == st)
2106
+ for st in ("confirmed", "contradicted", "unobservable")
2107
+ },
2108
+ "predicted_branches": predicted_vec,
2109
+ "observed_branches": observed_vec,
2110
+ "branches_exact": indeterminate is None and predicted_vec == observed_vec,
2111
+ "state_checked": sorted(k for k in observed_state if k in pred.state),
2112
+ "state_diff": state_diff,
2113
+ "outcome_diff": outcome_diff,
2114
+ "match": indeterminate is None
2115
+ and predicted_vec == observed_vec
2116
+ and not state_diff
2117
+ and not outcome_diff,
2118
+ }