cdclkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
cdclkit/mus.py ADDED
@@ -0,0 +1,159 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # Copyright (c) 2026 Carlo Perassi. Licensed under the Apache License 2.0.
3
+ """Minimal unsatisfiable subsets: *why* is this formula unsatisfiable?
4
+
5
+ An UNSAT answer says a formula has no model. A **MUS** says which clauses are
6
+ to blame: an unsatisfiable subset of the clauses such that removing any one of
7
+ them makes it satisfiable. For a specification with 10,000 constraints and a
8
+ contradiction somewhere in it, that difference is the difference between "your
9
+ model is broken" and "these six lines conflict".
10
+
11
+ Formally, `M ⊆ F` is a MUS when `M` is unsatisfiable and every proper subset of
12
+ `M` is satisfiable. Minimal, not minimum: a formula can have many MUSes of
13
+ different sizes, and finding the smallest is a harder (Σ₂ᵖ-complete) problem.
14
+ Every algorithm here returns *a* MUS.
15
+
16
+ The machinery is selector variables. Clause `cᵢ` becomes `¬sᵢ ∨ cᵢ`, so
17
+ assuming `sᵢ` switches the clause on and assuming nothing leaves it inert. Then
18
+ "is this subset unsatisfiable?" is one incremental solve under assumptions, the
19
+ solver keeps everything it learned across all of them, and the unsatisfiable
20
+ core the solver already computes (`Solver.conflict`) gives a free head start:
21
+ it is an unsatisfiable subset, just not necessarily a minimal one.
22
+
23
+ Two algorithms:
24
+
25
+ **Deletion-based.** Try each candidate clause in turn; if the rest are still
26
+ unsatisfiable, drop it permanently. Exactly `|M|` solver calls after the
27
+ initial core, each one UNSAT-or-SAT, and the result is minimal by construction.
28
+ Simple and hard to get wrong.
29
+
30
+ **QuickXplain** (Junker 2004). Divide and conquer: split the candidate set,
31
+ recurse on halves, and only descend into a half that is actually needed. When
32
+ the MUS is small relative to the formula it takes O(|M| log(|F|/|M|)) calls
33
+ instead of O(|F|), which is a large win on the realistic case where a handful
34
+ of constraints out of thousands are guilty.
35
+ """
36
+
37
+ from __future__ import annotations
38
+
39
+ from typing import Sequence
40
+
41
+ from dratify.cnf import CNF
42
+ from dratify.lits import mk_lit, neg
43
+ from .solver import Solver
44
+
45
+ __all__ = ["MUSExtractor", "mus", "shrink_core"]
46
+
47
+
48
+ class MUSExtractor:
49
+ """Wraps a formula with selector variables for incremental subset testing."""
50
+
51
+ def __init__(self, formula: CNF) -> None:
52
+ self.formula = formula
53
+ self.solver = Solver(formula.nvars)
54
+ self.selectors: list[int] = []
55
+ for clause in formula.clauses:
56
+ s = mk_lit(self.solver.new_var())
57
+ self.selectors.append(s)
58
+ # (~s or clause): assuming s switches the clause on
59
+ self.solver.add_clause([neg(s)] + list(clause))
60
+ self.calls = 0
61
+
62
+ def is_unsat(self, subset: Sequence[int]) -> bool:
63
+ """True when the clauses indexed by ``subset`` are unsatisfiable."""
64
+ self.calls += 1
65
+ return not self.solver.solve([self.selectors[i] for i in subset])
66
+
67
+ def _all(self, subset: Sequence[int] | None) -> list[int]:
68
+ cands = list(range(len(self.selectors))) if subset is None else list(subset)
69
+ return cands if self.is_unsat(cands) else []
70
+
71
+ def core(self, subset: Sequence[int] | None = None) -> list[int]:
72
+ """An unsatisfiable core: the subset the solver actually used.
73
+
74
+ Not minimal, but usually far smaller than the input, and it costs
75
+ nothing extra -- the solver computes it during the failed solve.
76
+ """
77
+ candidates = list(range(len(self.selectors))) if subset is None else list(subset)
78
+ if not self.is_unsat(candidates):
79
+ return []
80
+ by_sel = {self.selectors[i]: i for i in candidates}
81
+ used = [by_sel[l] for l in self.solver.conflict if l in by_sel]
82
+ return sorted(used) if used else candidates
83
+
84
+ # -- algorithms ---------------------------------------------------------
85
+
86
+ def deletion(self, subset: Sequence[int] | None = None, use_core: bool = True) -> list[int]:
87
+ """Deletion-based MUS extraction.
88
+
89
+ ``use_core`` first shrinks the candidate set to the solver's own
90
+ unsatisfiable core, which is nearly always worth it: one solve replaces
91
+ many. Set it to False to measure the algorithms on equal footing.
92
+ """
93
+ candidates = self.core(subset) if use_core else self._all(subset)
94
+ if not candidates:
95
+ return []
96
+ keep: list[int] = []
97
+ remaining = list(candidates)
98
+ while remaining:
99
+ c = remaining.pop()
100
+ if self.is_unsat(keep + remaining):
101
+ continue # c is not needed
102
+ keep.append(c)
103
+ return sorted(keep)
104
+
105
+ def quickxplain(self, subset: Sequence[int] | None = None,
106
+ use_core: bool = True) -> list[int]:
107
+ """Junker's QuickXplain: divide and conquer over the candidate set."""
108
+ candidates = self.core(subset) if use_core else self._all(subset)
109
+ if not candidates:
110
+ return []
111
+ if self.is_unsat([]):
112
+ return []
113
+ return sorted(self._qx([], [], candidates))
114
+
115
+ def _qx(self, background: list[int], delta: list[int], candidates: list[int]) -> list[int]:
116
+ if delta and self.is_unsat(background):
117
+ return []
118
+ if len(candidates) == 1:
119
+ return list(candidates)
120
+ mid = len(candidates) // 2
121
+ first, second = candidates[:mid], candidates[mid:]
122
+ d1 = self._qx(background + first, first, second)
123
+ d2 = self._qx(background + d1, d1, first)
124
+ return d1 + d2
125
+
126
+ # -- verification -------------------------------------------------------
127
+
128
+ def verify(self, subset: Sequence[int]) -> tuple[bool, str]:
129
+ """Check the defining property: unsatisfiable, and minimally so.
130
+
131
+ Costs ``|subset| + 1`` solver calls. Cheap enough to run in anger, and
132
+ the test suite always does -- a "minimal" set nobody checked is just a
133
+ set.
134
+ """
135
+ if not self.is_unsat(subset):
136
+ return False, "the subset is satisfiable"
137
+ for i in subset:
138
+ rest = [j for j in subset if j != i]
139
+ if self.is_unsat(rest):
140
+ return False, f"clause {i} is redundant: the rest is still unsatisfiable"
141
+ return True, f"verified: unsatisfiable, and all {len(subset)} clauses are necessary"
142
+
143
+
144
+ def mus(formula: CNF, method: str = "deletion") -> list[int]:
145
+ """Return the indices of a minimal unsatisfiable subset of ``formula``.
146
+
147
+ Returns an empty list when the formula is satisfiable.
148
+ """
149
+ ex = MUSExtractor(formula)
150
+ if method == "deletion":
151
+ return ex.deletion()
152
+ if method == "quickxplain":
153
+ return ex.quickxplain()
154
+ raise ValueError(f"unknown MUS method {method!r}")
155
+
156
+
157
+ def shrink_core(formula: CNF) -> list[int]:
158
+ """The cheap option: the solver's own unsatisfiable core, unminimised."""
159
+ return MUSExtractor(formula).core()
cdclkit/native.py ADDED
@@ -0,0 +1,111 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # Copyright (c) 2026 Carlo Perassi. Licensed under the Apache License 2.0.
3
+ """Optional native engine: loader and capability probe.
4
+
5
+ cdclkit's Python core has no dependencies and never will. The native engine is a
6
+ strictly optional accelerator, and this module is the only place that knows
7
+ whether it exists.
8
+
9
+ The contract, in both directions:
10
+
11
+ * **If the compiled module is absent, nothing breaks.** `available()` returns
12
+ False, every caller falls back to Python, and the test suite still passes in
13
+ full under a plain `python3` with no Rust toolchain anywhere. That property
14
+ is tested, not assumed -- it is what lets the port proceed in small commits
15
+ without the repo ever being in a broken state.
16
+ * **If it is present, it is opt-in.** Nothing selects the native path on its
17
+ own. A caller asks for it explicitly (`--engine native`, or
18
+ `CDCLKIT_ENGINE=native`), because a silent switch between two implementations
19
+ is how differential bugs hide.
20
+
21
+ Building it::
22
+
23
+ python3 -m venv .venv
24
+ .venv/bin/pip install maturin
25
+ make native # or: cd native && ../.venv/bin/maturin develop --release
26
+
27
+ The build installs into `.venv`, so `.venv/bin/python` sees the native engine
28
+ and the system interpreter does not. That separation is deliberate: it keeps a
29
+ dependency-free path available at all times.
30
+ """
31
+
32
+ from __future__ import annotations
33
+
34
+ import os
35
+
36
+ __all__ = [
37
+ "available",
38
+ "require",
39
+ "module",
40
+ "version",
41
+ "build_hint",
42
+ "engine_requested",
43
+ ]
44
+
45
+ try: # pragma: no cover - the import result is the thing being reported
46
+ import cdclkit_native as _native
47
+ except ImportError: # pragma: no cover
48
+ _native = None
49
+ else: # pragma: no cover - depends on whether the module was built
50
+ # The Rust checker lives in the `dratify` crate, which this crate embeds,
51
+ # but `dratify` ships no Python bindings of its own yet. Hand ours over so
52
+ # that check_proof(engine="auto") uses it instead of falling back to the
53
+ # pure-Python checker, which is ~18x slower on large proofs.
54
+ #
55
+ # This is not the silent engine switch the rest of this module warns about.
56
+ # Two checkers must agree by construction -- if they ever disagree that is
57
+ # a bug worth surfacing, not a difference worth preserving. The solver is
58
+ # the thing that stays explicitly opt-in.
59
+ try:
60
+ import dratify as _dratify
61
+ _dratify.register_native(_native)
62
+ except (ImportError, AttributeError): # dratify < 0.1.1 has no seam
63
+ pass
64
+
65
+
66
+ BUILD_HINT = (
67
+ "the native engine is not built for this interpreter. Build it with:\n"
68
+ " python3 -m venv .venv && .venv/bin/pip install maturin\n"
69
+ " make native\n"
70
+ "then run with .venv/bin/python. The pure-Python engine needs none of this."
71
+ )
72
+
73
+
74
+ def available() -> bool:
75
+ """True when the compiled native module can be imported."""
76
+ return _native is not None
77
+
78
+
79
+ def module():
80
+ """The native module, or None. Prefer `require()` when you need it."""
81
+ return _native
82
+
83
+
84
+ def require():
85
+ """Return the native module, raising a useful error when it is missing."""
86
+ if _native is None:
87
+ raise RuntimeError(BUILD_HINT)
88
+ return _native
89
+
90
+
91
+ def version() -> str | None:
92
+ return getattr(_native, "__version__", None) if _native else None
93
+
94
+
95
+ def build_hint() -> str:
96
+ return BUILD_HINT
97
+
98
+
99
+ def engine_requested(default: str = "python") -> str:
100
+ """Which engine the environment asks for.
101
+
102
+ Reads `CDCLKIT_ENGINE`; an explicit request for an unavailable engine is an
103
+ error rather than a silent downgrade, because "I asked for native and got
104
+ Python timings" is a bad way to spend an afternoon.
105
+ """
106
+ want = os.environ.get("CDCLKIT_ENGINE", default).strip().lower()
107
+ if want not in ("python", "native"):
108
+ raise ValueError(f"unknown CDCLKIT_ENGINE {want!r} (expected python or native)")
109
+ if want == "native" and not available():
110
+ raise RuntimeError("CDCLKIT_ENGINE=native but " + BUILD_HINT)
111
+ return want
cdclkit/pipeline.py ADDED
@@ -0,0 +1,212 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # Copyright (c) 2026 Carlo Perassi. Licensed under the Apache License 2.0.
3
+ """Choosing when to preprocess, instead of always or never.
4
+
5
+ Preprocessing is worth 1.3-1.5x on structured instances and a dead loss on
6
+ instances that solve in milliseconds -- `bench/compare.py` measures both. The
7
+ crafted families in the benchmark set finish in under a millisecond, so any
8
+ preprocessing at all is a hundredfold overhead; the factoring and colouring
9
+ instances take seconds, and preprocessing pays for itself several times over.
10
+
11
+ Deciding by looking at the formula does not work well. Clause count is a poor
12
+ predictor: `queens(40)` has 18 611 clauses and solves in zero conflicts, while
13
+ `factor(16b)` has 8 051 and needs sixteen thousand. Size tells you how big the
14
+ formula is, not how hard it is.
15
+
16
+ So this asks the solver instead. Run with a small conflict budget; if the
17
+ instance falls over quickly, nothing was spent. If the budget is exhausted, the
18
+ instance is hard enough that preprocessing will repay its cost, so preprocess
19
+ and solve properly. The wasted work is bounded by the budget and the learnt
20
+ clauses from the probe are discarded -- a real cost, but a fixed and small one
21
+ against an unbounded win.
22
+
23
+ The policy is deliberately simple and its parameter is exposed, because the
24
+ right budget depends on the ratio between preprocessing cost and solve cost on
25
+ the machine in front of you.
26
+ """
27
+
28
+ from __future__ import annotations
29
+
30
+ import time
31
+ from typing import Sequence
32
+
33
+ from dratify.cnf import CNF
34
+ from .solver import Config, Solver
35
+
36
+ __all__ = ["solve_adaptive", "PipelineResult"]
37
+
38
+ #: Conflicts to spend probing before deciding to preprocess. At ~100k
39
+ #: conflicts/second on the native engine this is a ~10 ms probe, which is
40
+ #: cheaper than the fastest preprocessing run measured (~1 ms native, ~3 ms
41
+ #: Python) on anything but the smallest instances, and negligible against the
42
+ #: seconds-long solves where preprocessing matters.
43
+ DEFAULT_PROBE = 1000
44
+
45
+
46
+ class PipelineResult:
47
+ __slots__ = ("sat", "model", "conflicts", "seconds", "preprocessed",
48
+ "probe_conflicts", "prep_seconds", "clauses_before",
49
+ "clauses_after")
50
+
51
+ def __init__(self) -> None:
52
+ self.sat: bool | None = None
53
+ self.model: list[bool] | None = None
54
+ self.conflicts = 0
55
+ self.seconds = 0.0
56
+ self.preprocessed = False
57
+ self.probe_conflicts = 0
58
+ self.prep_seconds = 0.0
59
+ self.clauses_before = 0
60
+ self.clauses_after = 0
61
+
62
+ def report(self) -> str:
63
+ if not self.preprocessed:
64
+ return (f"c solved during the {self.probe_conflicts}-conflict probe; "
65
+ f"preprocessing skipped")
66
+ return (f"c probe exhausted at {self.probe_conflicts} conflicts -> "
67
+ f"preprocessed {self.clauses_before} -> {self.clauses_after} "
68
+ f"clauses in {self.prep_seconds*1000:.0f} ms")
69
+
70
+
71
+ def _native_available() -> bool:
72
+ from . import native
73
+
74
+ return native.available()
75
+
76
+
77
+ def _solve(f: CNF, cfg: Config, budget: int | None, engine: str,
78
+ seconds: float | None = None):
79
+ """Returns (status, model, conflicts). status None means budget exhausted."""
80
+ if engine == "native":
81
+ from . import native
82
+
83
+ n = native.require()
84
+ s = n.Solver(
85
+ f.nvars, restart=cfg.restart, ccmin=cfg.ccmin,
86
+ phase_saving=cfg.phase_saving, init_phase=cfg.init_phase,
87
+ target_phase=cfg.target_phase, target_reset=cfg.target_reset,
88
+ walk_flips=cfg.walk_flips, walk_interval=cfg.walk_interval,
89
+ walk_patience=cfg.walk_patience,
90
+ walk_min_conflicts=cfg.walk_min_conflicts,
91
+ var_decay=cfg.var_decay, var_decay_max=cfg.var_decay_max,
92
+ cla_decay=cfg.cla_decay, luby_base=float(cfg.luby_base),
93
+ first_reduce=cfg.first_reduce, reduce_inc=cfg.reduce_inc,
94
+ glue_keep=cfg.glue_keep, block_restart=cfg.block_restart,
95
+ )
96
+ for c in f.clauses:
97
+ if not s.add_clause(list(c)):
98
+ return False, None, s.conflicts
99
+ res = s.solve(budget, seconds)
100
+ if res is None:
101
+ return None, None, s.conflicts
102
+ return res, (list(s.model) if res else None), s.conflicts
103
+
104
+ s = Solver(f.nvars, config=cfg)
105
+ if not s.add_cnf(f):
106
+ return False, None, s.stats.conflicts
107
+ res = s.solve(max_conflicts=budget, deadline=(
108
+ None if seconds is None else time.perf_counter() + seconds))
109
+ if res is None:
110
+ return None, None, s.stats.conflicts
111
+ return res, (list(s.model) if res else None), s.stats.conflicts
112
+
113
+
114
+ def _preprocess(f: CNF, engine: str):
115
+ """Returns (reduced_formula, unsat, reconstruct_fn, seconds)."""
116
+ t0 = time.perf_counter()
117
+ if engine == "native" and _native_available():
118
+ from . import native
119
+
120
+ n = native.require()
121
+ p = n.Preprocessor(f.nvars)
122
+ for c in f.clauses:
123
+ p.add_clause(list(c))
124
+ p.run(3)
125
+ red = CNF(f.nvars)
126
+ for c in p.reduced():
127
+ red.add(c)
128
+ red.nvars = f.nvars
129
+ return red, p.unsat, p.reconstruct, time.perf_counter() - t0
130
+
131
+ from .preprocess import Preprocessor
132
+
133
+ p = Preprocessor(f)
134
+ red = p.run()
135
+ return red, p.unsat, p.reconstruct, time.perf_counter() - t0
136
+
137
+
138
+ def solve_adaptive(
139
+ f: CNF,
140
+ engine: str = "native",
141
+ probe: int = DEFAULT_PROBE,
142
+ config: Config | None = None,
143
+ always_preprocess: bool = False,
144
+ never_preprocess: bool = False,
145
+ jobs: int = 1,
146
+ seconds: float | None = None,
147
+ ) -> PipelineResult:
148
+ """Solve `f`, preprocessing only when a short probe says it is worth it.
149
+
150
+ `engine` selects "native" (falling back to Python when the module is
151
+ absent) or "python". `always_preprocess` / `never_preprocess` force the
152
+ decision, which is what the benchmark harness uses to measure the policy
153
+ against its own extremes.
154
+
155
+ `jobs > 1` runs the *post-probe* solve as a parallel portfolio. The probe
156
+ itself stays sequential and in-process on purpose: spawning five workers
157
+ costs ~60 ms, which is more than an easy instance takes to solve outright,
158
+ so paying it before knowing the instance is hard would throw away exactly
159
+ what the probe is for.
160
+ """
161
+ cfg = config or Config()
162
+ if engine == "native" and not _native_available():
163
+ engine = "python"
164
+
165
+ r = PipelineResult()
166
+ r.clauses_before = f.nclauses
167
+ t_start = time.perf_counter()
168
+
169
+ if not never_preprocess and not always_preprocess:
170
+ status, model, conflicts = _solve(f, cfg, probe, engine)
171
+ r.probe_conflicts = conflicts
172
+ if status is not None:
173
+ r.sat, r.model, r.conflicts = status, model, conflicts
174
+ r.seconds = time.perf_counter() - t_start
175
+ r.clauses_after = f.nclauses
176
+ return r
177
+
178
+ if never_preprocess:
179
+ status, model, conflicts = _solve(f, cfg, None, engine, seconds)
180
+ r.sat, r.model, r.conflicts = status, model, conflicts
181
+ r.seconds = time.perf_counter() - t_start
182
+ r.clauses_after = f.nclauses
183
+ return r
184
+
185
+ red, unsat, reconstruct, prep_s = _preprocess(f, engine)
186
+ r.preprocessed = True
187
+ r.prep_seconds = prep_s
188
+ r.clauses_after = red.nclauses
189
+
190
+ if unsat:
191
+ r.sat = False
192
+ r.seconds = time.perf_counter() - t_start
193
+ return r
194
+
195
+ if jobs > 1:
196
+ from .portfolio import solve_portfolio
197
+
198
+ # preprocess_workers=0 because this formula is *already* preprocessed;
199
+ # asking for preprocessing workers here would both redo the work and
200
+ # force the process-based path, paying ~60 ms of startup for nothing
201
+ pr = solve_portfolio(red, jobs=jobs, engine=engine, preprocess_workers=0)
202
+ status = pr.sat
203
+ model = pr.model
204
+ conflicts = pr.stats.get("conflicts", 0)
205
+ else:
206
+ status, model, conflicts = _solve(red, cfg, None, engine, seconds)
207
+ r.conflicts = conflicts + r.probe_conflicts
208
+ r.sat = status
209
+ if status:
210
+ r.model = list(reconstruct(model))
211
+ r.seconds = time.perf_counter() - t_start
212
+ return r