certo-math 0.11.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- certo/__init__.py +45 -0
- certo/asymptotics.py +233 -0
- certo/audit.py +299 -0
- certo/binding.py +137 -0
- certo/catalogue.py +136 -0
- certo/cdcl.py +276 -0
- certo/certificate.py +4427 -0
- certo/cli.py +2375 -0
- certo/cnf.py +231 -0
- certo/cover.py +129 -0
- certo/cycles.py +196 -0
- certo/dataspec.py +130 -0
- certo/doctor.py +341 -0
- certo/drup.py +245 -0
- certo/engines/__init__.py +1 -0
- certo/engines/algebra.py +1084 -0
- certo/engines/bb.py +256 -0
- certo/engines/bisect.py +128 -0
- certo/engines/bounds.py +116 -0
- certo/engines/cegis.py +180 -0
- certo/engines/compose.py +186 -0
- certo/engines/domain.py +207 -0
- certo/engines/farkas.py +175 -0
- certo/engines/graphsearch.py +308 -0
- certo/engines/induct.py +151 -0
- certo/engines/lp.py +341 -0
- certo/engines/mixed.py +244 -0
- certo/engines/order.py +106 -0
- certo/engines/sat.py +166 -0
- certo/engines/shrink.py +343 -0
- certo/engines/smt.py +433 -0
- certo/entry.py +132 -0
- certo/equitable.py +277 -0
- certo/exact.py +327 -0
- certo/existence.py +124 -0
- certo/family.py +171 -0
- certo/graphs.py +341 -0
- certo/growth.py +130 -0
- certo/i18n.py +69 -0
- certo/interchange.py +159 -0
- certo/lattice.py +447 -0
- certo/lean.py +114 -0
- certo/leancheck.py +238 -0
- certo/leanexport.py +359 -0
- certo/ledger.py +124 -0
- certo/limits.py +31 -0
- certo/linarith.py +288 -0
- certo/linsolve.py +229 -0
- certo/lint.py +707 -0
- certo/locales/en.json +1342 -0
- certo/locales/es.json +1342 -0
- certo/mcp_server.py +1794 -0
- certo/moment.py +116 -0
- certo/numbers.py +192 -0
- certo/numerics.py +295 -0
- certo/orbits.py +362 -0
- certo/orderinfer.py +241 -0
- certo/packing.py +289 -0
- certo/parametric.py +328 -0
- certo/paramsym.py +425 -0
- certo/peak.py +151 -0
- certo/polynomials.py +297 -0
- certo/propositional.py +174 -0
- certo/rangebound.py +228 -0
- certo/ratio.py +120 -0
- certo/reducers.py +135 -0
- certo/repro.py +178 -0
- certo/resultants.py +225 -0
- certo/routing.py +269 -0
- certo/simplex.py +168 -0
- certo/sos.py +236 -0
- certo/spec.py +1541 -0
- certo/status.py +89 -0
- certo/status_report.py +315 -0
- certo/structures.py +290 -0
- certo/symmetry.py +169 -0
- certo/toric.py +276 -0
- certo/tree.py +95 -0
- certo/z3util.py +82 -0
- certo_math-0.11.1.dist-info/METADATA +317 -0
- certo_math-0.11.1.dist-info/RECORD +85 -0
- certo_math-0.11.1.dist-info/WHEEL +5 -0
- certo_math-0.11.1.dist-info/entry_points.txt +3 -0
- certo_math-0.11.1.dist-info/licenses/LICENSE +21 -0
- certo_math-0.11.1.dist-info/top_level.txt +1 -0
certo/__init__.py
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""certo: a laboratory for supporting mathematical proofs.
|
|
2
|
+
|
|
3
|
+
Three cross-cutting rules, true of every command:
|
|
4
|
+
|
|
5
|
+
1. Every command returns a certificate, or says explicitly why not.
|
|
6
|
+
2. Six result states: unsat / sat / unknown_solver / timeout /
|
|
7
|
+
resource_exhausted / out_of_theory. Only the first two are conclusive;
|
|
8
|
+
the rest mean "no answer", each for a different reason.
|
|
9
|
+
3. Determinism by work budget (z3 rlimit, SAT conflict budget), not by
|
|
10
|
+
clock, so the same command gives the same answer on another machine.
|
|
11
|
+
That covers our engines, not a user-supplied predicate.
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from .certificate import Certificate, VerifyReport, verify
|
|
16
|
+
from .cnf import CNF, CNFSpec
|
|
17
|
+
from .graphs import Graph
|
|
18
|
+
from . import doctor, reducers
|
|
19
|
+
from .structures import SetFamily, family_from_masks, mask_to_set, set_to_mask
|
|
20
|
+
from .packing import PackingSpec, loads_from_dual
|
|
21
|
+
from .i18n import set_lang, t
|
|
22
|
+
from .limits import Limits
|
|
23
|
+
from .polynomials import Poly
|
|
24
|
+
from .spec import (BisectSpec, BoundSpec, DomainSpec, IdealSpec,
|
|
25
|
+
InductSpec, Lemma, LPSpec, NumberSpec, OrderSpec,
|
|
26
|
+
BindSpec, ConeSpec, CoverSpec, CycleSpec, EliminateSpec, EquitableQuotientSpec, LinearSystemSpec, MatrixSpec, ParametricSymmetrySpec, EntrySpec, FamilySpec, MomentSpec, ParametricSpec, SymmetrySpec, PeakSpec, RatioSpec,
|
|
27
|
+
SOSSpec,
|
|
28
|
+
MultiSpec, Outcome, ProofSpec, Spec, SweepSpec,
|
|
29
|
+
SynthSpec, load_spec)
|
|
30
|
+
from .status import Result, Status, Verdict
|
|
31
|
+
|
|
32
|
+
__version__ = "0.11.1"
|
|
33
|
+
|
|
34
|
+
__all__ = [
|
|
35
|
+
"Spec", "SynthSpec", "LPSpec", "SweepSpec", "CNF", "CNFSpec",
|
|
36
|
+
"BisectSpec", "BoundSpec", "DomainSpec", "IdealSpec", "InductSpec",
|
|
37
|
+
"NumberSpec", "OrderSpec", "SOSSpec", "EliminateSpec", "ParametricSpec", "PeakSpec", "FamilySpec", "RatioSpec", "MomentSpec", "EntrySpec", "SymmetrySpec", "CoverSpec", "ConeSpec", "CycleSpec", "BindSpec", "EquitableQuotientSpec", "MatrixSpec", "LinearSystemSpec", "ParametricSymmetrySpec", "Poly", "MultiSpec",
|
|
38
|
+
"ProofSpec", "Lemma",
|
|
39
|
+
"Outcome", "load_spec",
|
|
40
|
+
"Limits", "Status", "Verdict", "Result",
|
|
41
|
+
"Certificate", "VerifyReport", "verify",
|
|
42
|
+
"Graph", "PackingSpec", "loads_from_dual", "doctor", "reducers", "SetFamily",
|
|
43
|
+
"family_from_masks", "mask_to_set", "set_to_mask", "set_lang", "t",
|
|
44
|
+
"__version__",
|
|
45
|
+
]
|
certo/asymptotics.py
ADDED
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
"""Orders of magnitude: does this term decay, or is it Theta(1)?
|
|
2
|
+
|
|
3
|
+
A user described the bug this exists for better than I could have. They had a
|
|
4
|
+
symbolic constraint, substituted asymptotic magnitudes by hand -- `d` is about
|
|
5
|
+
`n^2`, `C` about `n`, `|W|` about `n^2` -- and asked whether a term decays in
|
|
6
|
+
`n`. Four bugs lived in that step. One of them,
|
|
7
|
+
|
|
8
|
+
5|k| W C^2 / (u^3 d^2 p^10)
|
|
9
|
+
|
|
10
|
+
was constant in `n`, and it was invisible to BOTH Lean and `prove`, for the
|
|
11
|
+
same reason: **it is not an infeasibility.** It is a feasibility that does not
|
|
12
|
+
improve with `n`, and a solver asked "is this satisfiable" will keep saying
|
|
13
|
+
yes, correctly, forever.
|
|
14
|
+
|
|
15
|
+
That is a different question from any the tool could ask. So:
|
|
16
|
+
|
|
17
|
+
order(expr, {"d": 2, "C": 1, "W": 2, "u": 0, "p": 0, "k": 0})
|
|
18
|
+
|
|
19
|
+
substitutes `n^a` for each symbol, collects the expression into powers of `n`
|
|
20
|
+
with EXACT rational coefficients, and reports the leading exponent. Negative
|
|
21
|
+
means it decays, positive means it grows, and zero -- the case that hurt --
|
|
22
|
+
means Theta(1): the term is not going away however large `n` gets.
|
|
23
|
+
|
|
24
|
+
Two things it is careful about, because both would make it lie.
|
|
25
|
+
|
|
26
|
+
Cancellation is real. Two terms with the same exponent whose coefficients sum
|
|
27
|
+
to zero do not contribute, and since the coefficients are `Fraction` that is
|
|
28
|
+
decided rather than estimated.
|
|
29
|
+
|
|
30
|
+
And `≍` hides a constant. This certifies the EXPONENT of `n`, not the
|
|
31
|
+
constant in front of it. A `Theta(1)` term with a coefficient of 1e-9 may be
|
|
32
|
+
perfectly fine in practice; what the certificate says is that it does not
|
|
33
|
+
shrink, and it says only that.
|
|
34
|
+
"""
|
|
35
|
+
from __future__ import annotations
|
|
36
|
+
|
|
37
|
+
from fractions import Fraction
|
|
38
|
+
|
|
39
|
+
import z3
|
|
40
|
+
|
|
41
|
+
from .i18n import t
|
|
42
|
+
|
|
43
|
+
#: A Laurent monomial is {symbol: exponent}, exponents may be negative -- which
|
|
44
|
+
#: is the whole point, since the interesting terms are quotients.
|
|
45
|
+
DECAYS, CONSTANT, GROWS = "decays", "constant", "grows"
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
class NotAsymptotic(ValueError):
|
|
49
|
+
"""Raised with the offending subterm, because "it failed" is useless."""
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _mono_mul(a: dict, b: dict) -> dict:
|
|
53
|
+
out = dict(a)
|
|
54
|
+
for k, v in b.items():
|
|
55
|
+
out[k] = out.get(k, 0) + v
|
|
56
|
+
if out[k] == 0:
|
|
57
|
+
del out[k]
|
|
58
|
+
return out
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _key(m: dict):
|
|
62
|
+
return tuple(sorted(m.items()))
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class Laurent:
|
|
66
|
+
"""A sum of Laurent monomials with exact rational coefficients."""
|
|
67
|
+
|
|
68
|
+
__slots__ = ("terms",)
|
|
69
|
+
|
|
70
|
+
def __init__(self, terms=None):
|
|
71
|
+
self.terms = {}
|
|
72
|
+
for m, c in (terms or {}).items():
|
|
73
|
+
c = Fraction(c)
|
|
74
|
+
if c:
|
|
75
|
+
self.terms[m] = c
|
|
76
|
+
|
|
77
|
+
@classmethod
|
|
78
|
+
def const(cls, c):
|
|
79
|
+
return cls({(): Fraction(c)})
|
|
80
|
+
|
|
81
|
+
@classmethod
|
|
82
|
+
def var(cls, name):
|
|
83
|
+
return cls({((name, 1),): Fraction(1)})
|
|
84
|
+
|
|
85
|
+
def __add__(self, other):
|
|
86
|
+
out = dict(self.terms)
|
|
87
|
+
for m, c in other.terms.items():
|
|
88
|
+
out[m] = out.get(m, Fraction(0)) + c
|
|
89
|
+
if not out[m]:
|
|
90
|
+
del out[m]
|
|
91
|
+
return Laurent(out)
|
|
92
|
+
|
|
93
|
+
def __neg__(self):
|
|
94
|
+
return Laurent({m: -c for m, c in self.terms.items()})
|
|
95
|
+
|
|
96
|
+
def __sub__(self, other):
|
|
97
|
+
return self + (-other)
|
|
98
|
+
|
|
99
|
+
def __mul__(self, other):
|
|
100
|
+
out = {}
|
|
101
|
+
for m1, c1 in self.terms.items():
|
|
102
|
+
for m2, c2 in other.terms.items():
|
|
103
|
+
m = _key(_mono_mul(dict(m1), dict(m2)))
|
|
104
|
+
out[m] = out.get(m, Fraction(0)) + c1 * c2
|
|
105
|
+
if not out[m]:
|
|
106
|
+
del out[m]
|
|
107
|
+
return Laurent(out)
|
|
108
|
+
|
|
109
|
+
def inverse(self):
|
|
110
|
+
"""Only a single monomial can be inverted, and that is the honest limit.
|
|
111
|
+
|
|
112
|
+
`1/(x + y)` is not a Laurent polynomial and its order depends on which
|
|
113
|
+
of x and y dominates -- a question this cannot answer and should not
|
|
114
|
+
pretend to. Divide by a product, not by a sum.
|
|
115
|
+
"""
|
|
116
|
+
if len(self.terms) != 1:
|
|
117
|
+
raise NotAsymptotic(t("asym.divide_sum", n=len(self.terms)))
|
|
118
|
+
(m, c), = self.terms.items()
|
|
119
|
+
return Laurent({_key({k: -v for k, v in dict(m).items()}):
|
|
120
|
+
Fraction(1) / c})
|
|
121
|
+
|
|
122
|
+
def power(self, n: int):
|
|
123
|
+
out = Laurent.const(1)
|
|
124
|
+
base = self if n >= 0 else self.inverse()
|
|
125
|
+
for _ in range(abs(n)):
|
|
126
|
+
out = out * base
|
|
127
|
+
return out
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def parse(expr) -> Laurent:
|
|
131
|
+
"""A z3 arithmetic term as a Laurent polynomial. Division included."""
|
|
132
|
+
if z3.is_int_value(expr):
|
|
133
|
+
return Laurent.const(expr.as_long())
|
|
134
|
+
if z3.is_rational_value(expr):
|
|
135
|
+
return Laurent.const(expr.as_fraction())
|
|
136
|
+
if z3.is_const(expr) and expr.decl().kind() == z3.Z3_OP_UNINTERPRETED:
|
|
137
|
+
return Laurent.var(str(expr))
|
|
138
|
+
if z3.is_add(expr):
|
|
139
|
+
out = Laurent()
|
|
140
|
+
for c in expr.children():
|
|
141
|
+
out = out + parse(c)
|
|
142
|
+
return out
|
|
143
|
+
if z3.is_sub(expr):
|
|
144
|
+
a, b = expr.children()
|
|
145
|
+
return parse(a) - parse(b)
|
|
146
|
+
if z3.is_mul(expr):
|
|
147
|
+
out = Laurent.const(1)
|
|
148
|
+
for c in expr.children():
|
|
149
|
+
out = out * parse(c)
|
|
150
|
+
return out
|
|
151
|
+
if z3.is_app_of(expr, z3.Z3_OP_UMINUS):
|
|
152
|
+
return -parse(expr.arg(0))
|
|
153
|
+
if z3.is_div(expr) or z3.is_idiv(expr):
|
|
154
|
+
a, b = expr.children()
|
|
155
|
+
return parse(a) * parse(b).inverse()
|
|
156
|
+
if z3.is_app_of(expr, z3.Z3_OP_POWER):
|
|
157
|
+
from .linarith import _literal_exponent
|
|
158
|
+
|
|
159
|
+
base, exp = expr.children()
|
|
160
|
+
n = _literal_exponent(exp)
|
|
161
|
+
if n is None:
|
|
162
|
+
# A negative literal exponent is fine here -- unlike in linarith,
|
|
163
|
+
# a quotient IS the shape we are here for.
|
|
164
|
+
n = _negative_exponent(exp)
|
|
165
|
+
if n is None:
|
|
166
|
+
raise NotAsymptotic(t("asym.bad_exponent", expr=str(expr)))
|
|
167
|
+
return parse(base).power(n)
|
|
168
|
+
raise NotAsymptotic(t("asym.not_arithmetic", expr=str(expr)))
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _negative_exponent(exp):
|
|
172
|
+
if z3.is_int_value(exp):
|
|
173
|
+
return exp.as_long()
|
|
174
|
+
if z3.is_rational_value(exp):
|
|
175
|
+
f = exp.as_fraction()
|
|
176
|
+
return f.numerator if f.denominator == 1 else None
|
|
177
|
+
return None
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
# ---------------------------------------------------------------------------
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def order(expr, orders: dict, var: str = "n"):
|
|
184
|
+
"""The exponent of `var`, and the terms that produce it.
|
|
185
|
+
|
|
186
|
+
`orders` maps each symbol to its exponent: `{"d": 2}` for `d` about `n^2`,
|
|
187
|
+
`{"u": 0}` for `u` about a constant, `{"r": Fraction(1,2)}` for a square
|
|
188
|
+
root. A symbol with no entry is an error rather than an assumption -- the
|
|
189
|
+
whole point is that the assignment is explicit.
|
|
190
|
+
"""
|
|
191
|
+
poly = parse(expr) if not isinstance(expr, Laurent) else expr
|
|
192
|
+
missing = sorted({s for m in poly.terms for s, _ in m} - set(orders))
|
|
193
|
+
if missing:
|
|
194
|
+
raise NotAsymptotic(t("asym.unassigned", names=", ".join(missing)))
|
|
195
|
+
|
|
196
|
+
by_exp: dict = {}
|
|
197
|
+
rows = []
|
|
198
|
+
for m, c in poly.terms.items():
|
|
199
|
+
e = sum((Fraction(orders[s]) * p for s, p in m), Fraction(0))
|
|
200
|
+
by_exp[e] = by_exp.get(e, Fraction(0)) + c
|
|
201
|
+
rows.append({"monomial": _render(m), "coefficient": str(c),
|
|
202
|
+
"exponent": str(e)})
|
|
203
|
+
|
|
204
|
+
# Cancellation is real, and with exact coefficients it is decided rather
|
|
205
|
+
# than estimated: a term whose coefficients sum to zero is not there.
|
|
206
|
+
surviving = {e: c for e, c in by_exp.items() if c}
|
|
207
|
+
degree = max(surviving) if surviving else None
|
|
208
|
+
return {
|
|
209
|
+
"terms": sorted(rows, key=lambda r: Fraction(r["exponent"]),
|
|
210
|
+
reverse=True),
|
|
211
|
+
"collected": sorted(({"exponent": str(e), "coefficient": str(c)}
|
|
212
|
+
for e, c in surviving.items()),
|
|
213
|
+
key=lambda r: Fraction(r["exponent"]),
|
|
214
|
+
reverse=True),
|
|
215
|
+
"degree": None if degree is None else str(degree),
|
|
216
|
+
"verdict": _verdict(degree),
|
|
217
|
+
"cancelled": len(by_exp) - len(surviving),
|
|
218
|
+
"var": var,
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _verdict(degree):
|
|
223
|
+
if degree is None:
|
|
224
|
+
return DECAYS # identically zero decays as fast as anything
|
|
225
|
+
if degree < 0:
|
|
226
|
+
return DECAYS
|
|
227
|
+
return CONSTANT if degree == 0 else GROWS
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def _render(m) -> str:
|
|
231
|
+
if not m:
|
|
232
|
+
return "1"
|
|
233
|
+
return " ".join("{}^{}".format(s, p) if p != 1 else s for s, p in m)
|
certo/audit.py
ADDED
|
@@ -0,0 +1,299 @@
|
|
|
1
|
+
"""Does each hypothesis earn its place? Drop it and go looking.
|
|
2
|
+
|
|
3
|
+
`core` answers the other half. It says WHICH hypotheses an unsat core needed
|
|
4
|
+
and drops the rest, which catches a theorem stated with slack. It cannot
|
|
5
|
+
catch the opposite mistake, and the opposite mistake is the expensive one:
|
|
6
|
+
|
|
7
|
+
a theorem stated TOO STRONGLY, formalised, and only then found to have
|
|
8
|
+
been about a smaller class than anybody wanted
|
|
9
|
+
|
|
10
|
+
So this drops each hypothesis in turn and hunts a counterexample to what
|
|
11
|
+
remains. Four answers per hypothesis, and they are genuinely different:
|
|
12
|
+
|
|
13
|
+
NEEDED, with a witness. Without it the claim is false, and here is the
|
|
14
|
+
assignment that breaks it. That witness is the useful artefact: it says not
|
|
15
|
+
only that the hypothesis matters but HOW, which is what tells you whether
|
|
16
|
+
you wrote the right one.
|
|
17
|
+
|
|
18
|
+
REDUNDANT. The claim still follows without it, so the theorem is weaker than
|
|
19
|
+
it looks and can be stated without it. `core` finds these too; they appear
|
|
20
|
+
here because a per-hypothesis report that silently omitted them would read
|
|
21
|
+
as "all needed".
|
|
22
|
+
|
|
23
|
+
DOMAIN. Dropping it does not make the claim false -- it makes the claim
|
|
24
|
+
MEANINGLESS, because it was holding up a well-definedness condition. See
|
|
25
|
+
below; this one exists because reporting it as either of the other two
|
|
26
|
+
would have been a lie in a different direction each time.
|
|
27
|
+
|
|
28
|
+
UNKNOWN. The solver did not settle it inside the budget. Never folded into
|
|
29
|
+
any of the other three -- "we did not find a counterexample" is not "there
|
|
30
|
+
is none", and that distinction is the whole discipline here.
|
|
31
|
+
|
|
32
|
+
DIVISION IS TOTAL IN SMT, AND THAT IS A TRAP. `n/0` is not an error in Z3; it
|
|
33
|
+
is some fixed but unspecified value, supplied by an internal function the
|
|
34
|
+
solver is free to interpret however it likes. So dropping `d != 0` and asking
|
|
35
|
+
for a counterexample gets you one immediately: `d = 0`, with `div0` chosen to
|
|
36
|
+
make the goal false. The hypothesis then reads `needed`, which is true by
|
|
37
|
+
accident and false in substance -- it is needed for the statement to MEAN
|
|
38
|
+
something, not for it to be true.
|
|
39
|
+
|
|
40
|
+
So the divisors are collected up front and every search is GUARDED by them.
|
|
41
|
+
A hypothesis whose drop leaves a counterexample only outside the domain is
|
|
42
|
+
reported as DOMAIN, naming the obligation it was carrying, rather than as
|
|
43
|
+
`needed` with a witness that divides by zero.
|
|
44
|
+
|
|
45
|
+
The guards are also what the witnesses are checked against on the way back,
|
|
46
|
+
and Z3's internal `div0`/`mod0` are excluded from a witness entirely: they
|
|
47
|
+
are the solver's bookkeeping, never part of the problem, and a witness that
|
|
48
|
+
carried them could not be re-applied.
|
|
49
|
+
|
|
50
|
+
WHAT THIS IS FOR is the step before formalisation. Formalising a theorem whose
|
|
51
|
+
hypotheses were never tested is how a month goes into proving something that
|
|
52
|
+
is true of a smaller class than the paper claims, and the test is cheap: one
|
|
53
|
+
satisfiability query per hypothesis.
|
|
54
|
+
|
|
55
|
+
WHAT IT IS NOT is a proof that the hypothesis set is minimal. Dropping them ONE
|
|
56
|
+
at a time says nothing about dropping two -- a pair can be jointly redundant
|
|
57
|
+
with neither redundant alone. Said in the certificate, every time.
|
|
58
|
+
"""
|
|
59
|
+
from __future__ import annotations
|
|
60
|
+
|
|
61
|
+
from .i18n import t as _t
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class NotAuditable(ValueError):
|
|
65
|
+
"""Raised with the reason, because a bare failure helps nobody."""
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
NEEDED, REDUNDANT, UNKNOWN = "needed", "redundant", "unknown"
|
|
69
|
+
#: A hypothesis holding up a well-definedness condition rather than a
|
|
70
|
+
#: mathematical one. Not `needed` -- the claim does not become false without
|
|
71
|
+
#: it -- and emphatically not `redundant`, because removing it does not give
|
|
72
|
+
#: a more general theorem, it gives a statement about `n/0`.
|
|
73
|
+
DOMAIN = "domain"
|
|
74
|
+
|
|
75
|
+
VERDICTS = (NEEDED, REDUNDANT, DOMAIN, UNKNOWN)
|
|
76
|
+
|
|
77
|
+
#: Z3 totalises division and modulo with these; they are the solver's
|
|
78
|
+
#: bookkeeping and never part of the problem, so they are not part of a
|
|
79
|
+
#: witness either.
|
|
80
|
+
INTERNAL = ("div0", "mod0", "rem0")
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def divisors(*expressions) -> list:
|
|
84
|
+
"""Every divisor appearing anywhere in these formulas.
|
|
85
|
+
|
|
86
|
+
A numeral divisor needs no obligation -- `n/3` is defined for every `n` --
|
|
87
|
+
so only the ones that could vanish are collected, deduplicated by their
|
|
88
|
+
printed form because two occurrences of the same expression are one
|
|
89
|
+
obligation.
|
|
90
|
+
"""
|
|
91
|
+
import z3
|
|
92
|
+
|
|
93
|
+
kinds = {z3.Z3_OP_DIV, z3.Z3_OP_IDIV, z3.Z3_OP_MOD, z3.Z3_OP_REM}
|
|
94
|
+
found, seen = [], set()
|
|
95
|
+
|
|
96
|
+
def walk(e):
|
|
97
|
+
if not z3.is_app(e):
|
|
98
|
+
return
|
|
99
|
+
if e.decl().kind() in kinds:
|
|
100
|
+
den = e.arg(1)
|
|
101
|
+
if not (z3.is_int_value(den) or z3.is_rational_value(den)):
|
|
102
|
+
key = den.sexpr()
|
|
103
|
+
if key not in seen:
|
|
104
|
+
seen.add(key)
|
|
105
|
+
found.append(den)
|
|
106
|
+
for i in range(e.num_args()):
|
|
107
|
+
walk(e.arg(i))
|
|
108
|
+
|
|
109
|
+
for expr in expressions:
|
|
110
|
+
if expr is not None:
|
|
111
|
+
walk(expr)
|
|
112
|
+
return found
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def obligations_for(spec) -> list:
|
|
116
|
+
"""The domain obligations of a spec, as `(text, formula)` pairs."""
|
|
117
|
+
import z3
|
|
118
|
+
|
|
119
|
+
out = []
|
|
120
|
+
for den in divisors(spec.goal, *[f for _n, f in spec.assumptions]):
|
|
121
|
+
out.append((den.sexpr(), den != 0))
|
|
122
|
+
return out
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _assignment(model) -> dict:
|
|
126
|
+
"""The witness, without the solver's own bookkeeping.
|
|
127
|
+
|
|
128
|
+
Only arity-zero declarations of a sort that can be written back down:
|
|
129
|
+
`div0` and `mod0` are functions Z3 invented to make division total, and a
|
|
130
|
+
witness carrying them cannot be re-applied by anybody, including us.
|
|
131
|
+
"""
|
|
132
|
+
known = {"Real", "Int", "Bool"}
|
|
133
|
+
return {str(d): [d.range().name(), str(model[d])]
|
|
134
|
+
for d in model.decls()
|
|
135
|
+
if d.arity() == 0 and d.range().name() in known
|
|
136
|
+
and str(d) not in INTERNAL}
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def audit(spec, limits=None) -> dict:
|
|
140
|
+
"""One satisfiability query per hypothesis: what breaks without it."""
|
|
141
|
+
import z3
|
|
142
|
+
|
|
143
|
+
from .limits import Limits
|
|
144
|
+
from .status import Status
|
|
145
|
+
|
|
146
|
+
if spec.goal is None:
|
|
147
|
+
raise NotAuditable(_t("audit.no_goal"))
|
|
148
|
+
if not spec.assumptions:
|
|
149
|
+
raise NotAuditable(_t("audit.no_hypotheses"))
|
|
150
|
+
|
|
151
|
+
lim = limits or Limits()
|
|
152
|
+
duties = obligations_for(spec)
|
|
153
|
+
rows = []
|
|
154
|
+
for dropped, _formula in spec.assumptions:
|
|
155
|
+
keep = [(n, f) for n, f in spec.assumptions if n != dropped]
|
|
156
|
+
s = z3.Solver()
|
|
157
|
+
lim.apply_to(s)
|
|
158
|
+
for _n, f in keep:
|
|
159
|
+
s.add(f)
|
|
160
|
+
# A counterexample to the claim, under everything EXCEPT this one --
|
|
161
|
+
# and INSIDE the domain, because `n/0` is a value Z3 makes up and a
|
|
162
|
+
# counterexample that uses it is about the solver, not the theorem.
|
|
163
|
+
s.add(z3.Not(spec.goal))
|
|
164
|
+
for _text, guard in duties:
|
|
165
|
+
s.add(guard)
|
|
166
|
+
got = s.check()
|
|
167
|
+
|
|
168
|
+
if got == z3.sat:
|
|
169
|
+
rows.append({"hypothesis": dropped, "verdict": NEEDED,
|
|
170
|
+
"witness": _assignment(s.model())})
|
|
171
|
+
elif got == z3.unsat:
|
|
172
|
+
lost = _lost_obligations(keep, duties, lim)
|
|
173
|
+
if lost:
|
|
174
|
+
rows.append({"hypothesis": dropped, "verdict": DOMAIN,
|
|
175
|
+
"witness": None, "obligations": lost})
|
|
176
|
+
else:
|
|
177
|
+
rows.append({"hypothesis": dropped, "verdict": REDUNDANT,
|
|
178
|
+
"witness": None})
|
|
179
|
+
else:
|
|
180
|
+
rows.append({"hypothesis": dropped, "verdict": UNKNOWN,
|
|
181
|
+
"witness": None,
|
|
182
|
+
"why": str(s.reason_unknown() or "")})
|
|
183
|
+
_ = Status
|
|
184
|
+
|
|
185
|
+
counts = {v: sum(1 for r in rows if r["verdict"] == v) for v in VERDICTS}
|
|
186
|
+
return {"rows": rows, "counts": counts,
|
|
187
|
+
"obligations": [text for text, _g in duties],
|
|
188
|
+
"ok": counts[UNKNOWN] == 0,
|
|
189
|
+
"redundant": [r["hypothesis"] for r in rows
|
|
190
|
+
if r["verdict"] == REDUNDANT],
|
|
191
|
+
"domain": [r["hypothesis"] for r in rows
|
|
192
|
+
if r["verdict"] == DOMAIN]}
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _lost_obligations(keep, duties, lim) -> list:
|
|
196
|
+
"""Which well-definedness conditions the remaining hypotheses no longer
|
|
197
|
+
force.
|
|
198
|
+
|
|
199
|
+
This is what separates DOMAIN from REDUNDANT, and it is a question about
|
|
200
|
+
the hypotheses rather than about the goal: if what is left still entails
|
|
201
|
+
every divisor being non-zero, the dropped one really was carrying nothing
|
|
202
|
+
and `redundant` is the honest answer.
|
|
203
|
+
"""
|
|
204
|
+
import z3
|
|
205
|
+
|
|
206
|
+
lost = []
|
|
207
|
+
for text, guard in duties:
|
|
208
|
+
s = z3.Solver()
|
|
209
|
+
lim.apply_to(s)
|
|
210
|
+
for _n, f in keep:
|
|
211
|
+
s.add(f)
|
|
212
|
+
s.add(z3.Not(guard))
|
|
213
|
+
if s.check() != z3.unsat: # the obligation is no longer forced
|
|
214
|
+
lost.append(text)
|
|
215
|
+
return lost
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def declared_obligations(payload) -> dict:
|
|
219
|
+
"""Re-derive the domain obligations from the formulas that travelled.
|
|
220
|
+
|
|
221
|
+
A certificate declaring FEWER divisors than its own formulas contain is a
|
|
222
|
+
certificate whose searches ran unguarded, and every `needed` verdict in it
|
|
223
|
+
could be a division by zero. So this is derived rather than read -- the
|
|
224
|
+
same reason a branch-and-bound node rebuilds its own linear program.
|
|
225
|
+
|
|
226
|
+
A certificate written before obligations existed declares none, and one
|
|
227
|
+
that lists MORE than it needs has only searched a smaller region, which is
|
|
228
|
+
sound and merely weaker -- so neither of those fails.
|
|
229
|
+
"""
|
|
230
|
+
import z3
|
|
231
|
+
|
|
232
|
+
formulas = [z3.And(*z3.parse_smt2_string(smt2))
|
|
233
|
+
for smt2 in (payload.get("hypotheses_smt2") or {}).values()]
|
|
234
|
+
goal = z3.And(*z3.parse_smt2_string(payload["goal_smt2"]))
|
|
235
|
+
found = [d.sexpr() for d in divisors(goal, *formulas)]
|
|
236
|
+
|
|
237
|
+
declared = payload.get("obligations")
|
|
238
|
+
if declared is None:
|
|
239
|
+
# Nothing was claimed, so nothing is contradicted. The rows are still
|
|
240
|
+
# checked against the guards by `recheck`, which is where an unguarded
|
|
241
|
+
# witness actually fails.
|
|
242
|
+
return {"ok": True, "found": found, "missing": [], "declared": []}
|
|
243
|
+
missing = [d for d in found if d not in set(declared)]
|
|
244
|
+
return {"ok": not missing, "found": found, "missing": missing,
|
|
245
|
+
"declared": list(declared)}
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def recheck(payload, limits=None) -> dict:
|
|
249
|
+
"""Re-run every witness against the formulas it claims to break.
|
|
250
|
+
|
|
251
|
+
The witness is the whole content of a NEEDED verdict, and checking one is
|
|
252
|
+
evaluation rather than search: substitute the assignment, and the kept
|
|
253
|
+
hypotheses must hold while the goal must not. A verdict with a witness
|
|
254
|
+
that does not do that is a verdict about nothing.
|
|
255
|
+
"""
|
|
256
|
+
import z3
|
|
257
|
+
|
|
258
|
+
from .limits import Limits
|
|
259
|
+
|
|
260
|
+
lim = limits or Limits()
|
|
261
|
+
formulas = {}
|
|
262
|
+
for name, smt2 in (payload.get("hypotheses_smt2") or {}).items():
|
|
263
|
+
formulas[name] = z3.And(*z3.parse_smt2_string(smt2))
|
|
264
|
+
goal = z3.And(*z3.parse_smt2_string(payload["goal_smt2"]))
|
|
265
|
+
duties = divisors(goal, *formulas.values())
|
|
266
|
+
|
|
267
|
+
bad, checked = [], 0
|
|
268
|
+
for row in payload["rows"]:
|
|
269
|
+
if row["verdict"] != NEEDED or not row.get("witness"):
|
|
270
|
+
continue
|
|
271
|
+
checked += 1
|
|
272
|
+
subs = []
|
|
273
|
+
for name, (sort, value) in row["witness"].items():
|
|
274
|
+
try:
|
|
275
|
+
var = {"Real": z3.Real, "Int": z3.Int,
|
|
276
|
+
"Bool": z3.Bool}[sort](name)
|
|
277
|
+
lit = {"Real": z3.RealVal, "Int": z3.IntVal,
|
|
278
|
+
"Bool": lambda v: z3.BoolVal(v == "True")}[sort](value)
|
|
279
|
+
except (KeyError, ValueError):
|
|
280
|
+
bad.append(row["hypothesis"])
|
|
281
|
+
break
|
|
282
|
+
subs.append((var, lit))
|
|
283
|
+
else:
|
|
284
|
+
kept = [f for n, f in formulas.items()
|
|
285
|
+
if n != row["hypothesis"]]
|
|
286
|
+
claim = z3.And(*kept) if kept else z3.BoolVal(True)
|
|
287
|
+
s = z3.Solver()
|
|
288
|
+
lim.apply_to(s)
|
|
289
|
+
# The witness must satisfy what was KEPT, stay inside the domain,
|
|
290
|
+
# and break the goal. Without the middle one a witness that
|
|
291
|
+
# divides by zero re-checks happily, which is how the defect that
|
|
292
|
+
# put this line here got past the first version.
|
|
293
|
+
inside = [z3.substitute(d, *subs) != 0 for d in duties]
|
|
294
|
+
s.add(z3.Not(z3.And(z3.substitute(claim, *subs),
|
|
295
|
+
*inside,
|
|
296
|
+
z3.Not(z3.substitute(goal, *subs)))))
|
|
297
|
+
if s.check() != z3.unsat:
|
|
298
|
+
bad.append(row["hypothesis"])
|
|
299
|
+
return {"checked": checked, "bad": sorted(set(bad))}
|