tiny-datalog 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
tiny_datalog/prolog.py ADDED
@@ -0,0 +1,335 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ prolog.py — a miniature top-down Horn-clause interpreter: the other side
4
+ of the Datalog boundary.
5
+
6
+ Datalog is Horn-clause logic with function symbols confiscated; that ban
7
+ is what makes bottom-up evaluation terminate. This module puts the
8
+ function symbols back — `s(N)`, `cons(H, T)` — and pays the price:
9
+ top-down SLD resolution with proper unification and no termination
10
+ guarantee. Two bounds cut the search off: a depth bound on each proof
11
+ branch and a budget on the total number of resolution steps. The depth
12
+ bound alone is not enough — with two clauses per goal a depth of 100
13
+ still allows about 2^100 branches — so the step budget is what keeps a
14
+ query from running for longer than anyone will wait. Either way the
15
+ interpreter tells you when it was cut off, i.e. when the answer set may
16
+ be incomplete.
17
+
18
+ Differences from real Prolog, on purpose:
19
+
20
+ * unification includes the occurs check (Prolog omits it for speed, and
21
+ `X = s(X)` quietly builds an infinite term);
22
+ * no cut, no arithmetic, no I/O — just SLD resolution over Horn clauses;
23
+ * `not` is negation as failure via a depth-bounded sub-proof.
24
+
25
+ Same syntax and parser as datalog.py, which *parses* compound terms but
26
+ rejects them at validation — run datalog.py on programs/peano.pl to see
27
+ the boundary stated as an error message.
28
+
29
+ CLI
30
+ ---
31
+ python3 prolog.py programs/peano.pl -q 'add(s(zero), s(s(zero)), X)'
32
+ python3 prolog.py programs/peano.pl -q 'nat(X)' --max-solutions 5
33
+ """
34
+
35
+ from __future__ import annotations
36
+
37
+ import argparse
38
+ import sys
39
+
40
+ from collections import defaultdict
41
+
42
+ from tiny_datalog.datalog import (
43
+ Atom, Const, DatalogError, Literal, parse, parse_goal, ParseError,
44
+ read_program, Rule, Struct, Var)
45
+
46
+
47
+ # ---------------------------------------------------------------------------
48
+ # Unification (with occurs check)
49
+ # ---------------------------------------------------------------------------
50
+
51
+ def _walk(term, subst):
52
+ while isinstance(term, Var) and term.name in subst:
53
+ term = subst[term.name]
54
+ return term
55
+
56
+
57
+ def _occurs(name, term, subst):
58
+ term = _walk(term, subst)
59
+ if isinstance(term, Var):
60
+ return term.name == name
61
+ if isinstance(term, Struct):
62
+ return any(_occurs(name, a, subst) for a in term.args)
63
+ return False
64
+
65
+
66
+ def unify(a, b, subst):
67
+ """Return an extended substitution unifying a and b, or None."""
68
+ a, b = _walk(a, subst), _walk(b, subst)
69
+ if isinstance(a, Var):
70
+ if isinstance(b, Var) and b.name == a.name:
71
+ return subst
72
+ if _occurs(a.name, b, subst):
73
+ return None
74
+ s = dict(subst)
75
+ s[a.name] = b
76
+ return s
77
+ if isinstance(b, Var):
78
+ return unify(b, a, subst)
79
+ if isinstance(a, Const) and isinstance(b, Const):
80
+ return subst if a.value == b.value else None
81
+ if isinstance(a, Struct) and isinstance(b, Struct):
82
+ if a.functor != b.functor or len(a.args) != len(b.args):
83
+ return None
84
+ for x, y in zip(a.args, b.args):
85
+ subst = unify(x, y, subst)
86
+ if subst is None:
87
+ return None
88
+ return subst
89
+ return None
90
+
91
+
92
+ def _unify_atoms(goal, head, subst):
93
+ if goal.pred != head.pred or len(goal.args) != len(head.args):
94
+ return None
95
+ for x, y in zip(goal.args, head.args):
96
+ subst = unify(x, y, subst)
97
+ if subst is None:
98
+ return None
99
+ return subst
100
+
101
+
102
+ def resolve(term, subst):
103
+ """Deep-substitute for printing an answer term."""
104
+ term = _walk(term, subst)
105
+ if isinstance(term, Struct):
106
+ return Struct(term.functor, tuple(resolve(a, subst) for a in term.args))
107
+ return term
108
+
109
+
110
+ def _is_ground(atom, subst):
111
+ """True iff every argument resolves to a variable-free term."""
112
+ def ground(t):
113
+ t = _walk(t, subst)
114
+ if isinstance(t, Var):
115
+ return False
116
+ if isinstance(t, Struct):
117
+ return all(ground(a) for a in t.args)
118
+ return True
119
+ return all(ground(a) for a in atom.args)
120
+
121
+
122
+ # ---------------------------------------------------------------------------
123
+ # SLD resolution
124
+ # ---------------------------------------------------------------------------
125
+
126
+ class PrologEngine:
127
+ """Bounded SLD resolution over Horn clauses. After a solve, the
128
+ `truncated` flag records whether the depth bound or the step budget
129
+ cut the search off (in which case "no more solutions" is not a proof
130
+ of absence)."""
131
+
132
+ def __init__(self, clauses):
133
+ self.clauses = list(clauses)
134
+ # index clauses by (pred, arity) so each resolution step only
135
+ # scans candidates that could possibly unify; source order is
136
+ # preserved within a bucket, so SLD semantics are unchanged
137
+ self.by_pred = defaultdict(list)
138
+ for c in self.clauses:
139
+ self.by_pred[(c.head.pred, len(c.head.args))].append(c)
140
+ self.counter = 0
141
+ self.truncated = False
142
+ self.steps, self.max_steps = 0, 100_000
143
+
144
+ def _rename(self, rule):
145
+ """Standardise a clause apart with fresh variable names. The
146
+ `#` makes them names the parser can never produce, so a renamed
147
+ clause variable can never collide with one the user wrote."""
148
+ self.counter += 1
149
+ n = self.counter
150
+
151
+ def rt(term):
152
+ if isinstance(term, Var):
153
+ return Var("%s#%d" % (term.name, n))
154
+ if isinstance(term, Struct):
155
+ return Struct(term.functor, tuple(rt(a) for a in term.args))
156
+ return term
157
+
158
+ def ra(atom):
159
+ return Atom(atom.pred, tuple(rt(a) for a in atom.args))
160
+
161
+ return Rule(ra(rule.head),
162
+ tuple(Literal(ra(l.atom), l.negated) for l in rule.body))
163
+
164
+ def solve(self, goals, subst=None, depth=100):
165
+ """Yield substitutions proving the goal list, left to right."""
166
+ if subst is None:
167
+ subst = {}
168
+ if not goals:
169
+ yield subst
170
+ return
171
+ if depth <= 0:
172
+ self.truncated = True
173
+ return
174
+ goal, rest = goals[0], goals[1:]
175
+ if goal.negated:
176
+ # Negation as failure, with two honesty guards.
177
+ # (1) The goal must be ground: `not p(X)` with X unbound would
178
+ # mean "no instance of p is provable at all", whose answer
179
+ # depends on literal order — the classic "floundering" trap.
180
+ if not _is_ground(goal.atom, subst):
181
+ raise DatalogError(
182
+ "negation as failure needs a ground goal, but %s has "
183
+ "unbound variables — reorder the body so positive "
184
+ "literals bind them first" % (goal.atom,))
185
+ # (2) Failure must be finite: if the sub-proof was cut off by
186
+ # a bound, "no proof found" means unproven, not disproven —
187
+ # so the negated goal must fail, not succeed. (If it *was*
188
+ # proved, the failure is genuine, whatever else was cut off.)
189
+ outer = self.truncated
190
+ self.truncated = False
191
+ proved = False
192
+ for _ in self.solve([Literal(goal.atom)], subst, depth - 1):
193
+ proved = True
194
+ break
195
+ truncated = self.truncated and not proved
196
+ self.truncated = outer or truncated
197
+ if proved or truncated:
198
+ return
199
+ yield from self.solve(rest, subst, depth - 1)
200
+ return
201
+ for clause in self.by_pred.get((goal.atom.pred,
202
+ len(goal.atom.args)), ()):
203
+ if self.steps >= self.max_steps:
204
+ self.truncated = True
205
+ return
206
+ self.steps += 1
207
+ renamed = self._rename(clause)
208
+ s2 = _unify_atoms(goal.atom, renamed.head, subst)
209
+ if s2 is not None:
210
+ yield from self.solve(list(renamed.body) + rest, s2,
211
+ depth - 1)
212
+
213
+ def query(self, atom, depth=100, max_solutions=None, max_steps=100_000):
214
+ """Solve a single goal; return (answers, incomplete) where answers
215
+ is a list of {var_name: term} dicts for the query's variables and
216
+ incomplete is True when the search was truncated — by the depth
217
+ bound, the step budget or the solution cap — so "no more answers"
218
+ is not a proof of absence."""
219
+ if max_solutions is not None and max_solutions < 1:
220
+ raise DatalogError("max_solutions must be at least 1 (got %s)"
221
+ % max_solutions)
222
+ self.truncated = False
223
+ self.steps, self.max_steps = 0, max_steps
224
+ qvars = []
225
+
226
+ def collect(term):
227
+ if (isinstance(term, Var) and not term.anonymous
228
+ and term not in qvars):
229
+ qvars.append(term)
230
+ elif isinstance(term, Struct):
231
+ for a in term.args:
232
+ collect(a)
233
+
234
+ for a in atom.args:
235
+ collect(a)
236
+ answers, seen = [], set()
237
+ capped = False
238
+ try:
239
+ for s in self.solve([Literal(atom)], depth=depth):
240
+ answer = {v.name: resolve(v, s) for v in qvars}
241
+ # dedupe on the terms themselves, not their printed
242
+ # form: two different terms can print identically
243
+ key = tuple(answer[v.name] for v in qvars)
244
+ if key in seen:
245
+ continue
246
+ seen.add(key)
247
+ answers.append(answer)
248
+ if max_solutions is not None and len(answers) >= max_solutions:
249
+ capped = True
250
+ break
251
+ except RecursionError:
252
+ raise DatalogError(
253
+ "the proof search exceeded Python's recursion capacity; "
254
+ "lower --depth (currently %d)" % depth)
255
+ return answers, self.truncated or capped
256
+
257
+
258
+ def load(text):
259
+ return PrologEngine(parse(text))
260
+
261
+
262
+ # ---------------------------------------------------------------------------
263
+ # CLI
264
+ # ---------------------------------------------------------------------------
265
+
266
+ def _positive_int(text):
267
+ n = int(text)
268
+ if n < 1:
269
+ raise argparse.ArgumentTypeError("must be at least 1, not %d" % n)
270
+ return n
271
+
272
+
273
+ def main(argv=None):
274
+ ap = argparse.ArgumentParser(
275
+ description="Top-down SLD resolution over Horn clauses with "
276
+ "function symbols — what Datalog deliberately isn't.")
277
+ ap.add_argument("file", help="Horn-clause program (.pl)")
278
+ ap.add_argument("-q", "--query", action="append", default=[],
279
+ metavar="GOAL", help="goal to prove (repeatable)")
280
+ ap.add_argument("--depth", type=int, default=100,
281
+ help="resolution depth bound (default 100)")
282
+ ap.add_argument("--max-solutions", type=_positive_int, default=10,
283
+ help="stop after this many answers (default 10)")
284
+ ap.add_argument("--max-steps", type=_positive_int, default=100_000,
285
+ help="resolution-step budget; the depth bound alone "
286
+ "can still allow exponentially many branches "
287
+ "(default 100000)")
288
+ args = ap.parse_args(argv)
289
+
290
+ try:
291
+ engine = load(read_program(args.file))
292
+ except DatalogError as exc:
293
+ print("error: %s" % exc, file=sys.stderr)
294
+ return 1
295
+ if not args.query:
296
+ print("loaded %d clauses; pass -q 'goal(...)' to prove something"
297
+ % len(engine.clauses))
298
+ return 0
299
+ for q in args.query:
300
+ try:
301
+ atom = parse_goal(q)
302
+ except ParseError as exc:
303
+ print("error: %s" % exc, file=sys.stderr)
304
+ return 1
305
+ print("?- %s" % atom)
306
+ try:
307
+ answers, incomplete = engine.query(
308
+ atom, depth=args.depth, max_solutions=args.max_solutions,
309
+ max_steps=args.max_steps)
310
+ except DatalogError as exc:
311
+ print("error: %s" % exc, file=sys.stderr)
312
+ return 1
313
+ if not answers:
314
+ bound = ("step budget %d" % args.max_steps
315
+ if engine.steps >= args.max_steps
316
+ else "depth %d" % args.depth)
317
+ print(" false." if not incomplete else
318
+ " false (search truncated at %s — unproven, "
319
+ "not disproven)." % bound)
320
+ continue
321
+ for answer in answers:
322
+ if answer:
323
+ print(" " + ", ".join("%s = %s" % (v, t)
324
+ for v, t in answer.items()))
325
+ else:
326
+ print(" true.")
327
+ note = " (more may exist: search truncated)" if incomplete else ""
328
+ print(" (%d solution%s%s)" % (len(answers),
329
+ "" if len(answers) == 1 else "s",
330
+ note))
331
+ return 0
332
+
333
+
334
+ if __name__ == "__main__":
335
+ sys.exit(main())
@@ -0,0 +1,182 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ semantics.py — ground semantics: stable models and the well-founded
4
+ model. (Lesson 5, which ends with a tour of this module.)
5
+
6
+ Stratifiability is a *syntactic* test, and rejection by the stratified
7
+ engine is not a semantic verdict: win(X) :- move(X, Y), not win(Y) is
8
+ unstratifiable yet has perfectly sensible stable models. This module
9
+ computes the two classical semantics that answer the real question —
10
+ "what, if anything, does this program mean?":
11
+
12
+ * **Stable models** (Gelfond–Lifschitz, 1988). A set of facts M is
13
+ stable if it *justifies itself*: assume M is exactly what's true,
14
+ simplify every `not` under that assumption (the "reduct"), and check
15
+ that the simplified — now negation-free — program derives exactly M
16
+ back. No unsupported beliefs, no dropped conclusions. A program with
17
+ no stable model (the café paradox, `p :- not p`) is genuinely
18
+ paradoxical.
19
+ * **The well-founded model** (Van Gelder–Ross–Schlipf, 1991). Three
20
+ valued and always defined: every fact comes out true, false, or
21
+ *undefined*. What can be settled is settled; what is genuinely
22
+ circular is named as such.
23
+
24
+ Both are computed over the program's grounding, and the stable-model
25
+ search is exhaustive over candidate sets — fine for teaching-sized
26
+ programs, and honest about it. Industrial ASP solvers (clingo) get the
27
+ same answers by conflict-driven search instead.
28
+ """
29
+
30
+ from __future__ import annotations
31
+
32
+ from collections import defaultdict
33
+
34
+ from tiny_datalog.datalog import (
35
+ Const, DatalogError, Program, _aggregate_of, _match, _sort_key, validate)
36
+
37
+
38
+ def _instantiate_atom(atom, subst):
39
+ return (atom.pred, tuple(a.value if isinstance(a, Const) else subst[a.name]
40
+ for a in atom.args))
41
+
42
+
43
+ def ground_program(clauses):
44
+ """Ground the program.
45
+
46
+ Returns (facts, ground_rules, candidates): facts is the set of ground
47
+ atoms (pred, args) from unit clauses; ground_rules is a list of
48
+ (head, pos_atoms, neg_atoms) triples over ground atoms; candidates is
49
+ the set of rule-derivable ground atoms when every negation is assumed
50
+ to succeed.
51
+
52
+ `candidates` is an upper bound on any stable model: the
53
+ Gelfond–Lifschitz operator is antimonotone (more assumed facts block
54
+ more rules), so every stable model M satisfies M = Γ(M) ⊆ Γ(∅) — and
55
+ Γ(∅) is exactly "derive with all negations granted". The exhaustive
56
+ search below therefore only needs subsets of this envelope.
57
+ """
58
+ validate(clauses)
59
+ for r in clauses:
60
+ if r.retract:
61
+ # `q~.` is an update, not a fact; let the base engine's
62
+ # Program reject it with its own explanation
63
+ Program([r])
64
+ if r.body and _aggregate_of(r.head):
65
+ raise DatalogError(
66
+ "stable-model and well-founded semantics for aggregates "
67
+ "are beyond this module (and still debated in the "
68
+ "literature): %s" % r)
69
+ facts = {(r.head.pred, tuple(a.value for a in r.head.args))
70
+ for r in clauses if not r.body}
71
+ rules = [r for r in clauses if r.body]
72
+
73
+ # Least model ignoring negation = the envelope of possibly-true atoms.
74
+ rels = defaultdict(set)
75
+ for pred, args in facts:
76
+ rels[pred].add(args)
77
+
78
+ def substitutions(rule):
79
+ # Join the positive body literals against the envelope; negated
80
+ # literals are skipped (treated as satisfied) at this stage.
81
+ substs = [{}]
82
+ for lit in rule.body:
83
+ if lit.negated:
84
+ continue
85
+ new = []
86
+ for s in substs:
87
+ for tup in rels.get(lit.atom.pred, ()):
88
+ m = _match(lit.atom.args, tup, s)
89
+ if m is not None:
90
+ new.append(m)
91
+ substs = new
92
+ return substs
93
+
94
+ changed = True
95
+ while changed:
96
+ changed = False
97
+ for rule in rules:
98
+ for s in substitutions(rule):
99
+ pred, args = _instantiate_atom(rule.head, s)
100
+ if args not in rels[pred]:
101
+ rels[pred].add(args)
102
+ changed = True
103
+
104
+ # Instantiate every rule over the envelope: each grounding whose
105
+ # positive body lies inside the envelope becomes one ground rule.
106
+ ground_rules = []
107
+ seen = set()
108
+ for rule in rules:
109
+ for s in substitutions(rule):
110
+ gr = (_instantiate_atom(rule.head, s),
111
+ tuple(_instantiate_atom(l.atom, s)
112
+ for l in rule.body if not l.negated),
113
+ tuple(_instantiate_atom(l.atom, s)
114
+ for l in rule.body if l.negated))
115
+ if gr not in seen:
116
+ seen.add(gr)
117
+ ground_rules.append(gr)
118
+ candidates = {gr[0] for gr in ground_rules} - facts
119
+ return facts, ground_rules, candidates
120
+
121
+
122
+ def _gamma(S, facts, ground_rules):
123
+ """The Gelfond–Lifschitz operator: least model of the reduct of the
124
+ ground program with respect to S.
125
+
126
+ The reduct deletes every rule with a negative literal `not a` where
127
+ a ∈ S, and strips the surviving rules' negative literals. What's
128
+ left is negation-free, so it has a unique least model — computed here
129
+ by plain forward chaining. M is a stable model iff Γ(M) == M.
130
+ """
131
+ derived = set(facts)
132
+ changed = True
133
+ while changed:
134
+ changed = False
135
+ for head, pos, neg in ground_rules:
136
+ if head in derived:
137
+ continue
138
+ if any(a in S for a in neg):
139
+ continue # rule deleted by the reduct
140
+ if all(a in derived for a in pos):
141
+ derived.add(head)
142
+ changed = True
143
+ return derived
144
+
145
+
146
+ def stable_models(clauses, limit_atoms=16, grounding=None):
147
+ """All stable models of the program, as sets of ground atoms (pred,
148
+ args), EDB facts included. Exhaustive search over subsets of the
149
+ candidate atoms; small programs only, and says so. Pass a
150
+ precomputed `grounding` (from ground_program) to avoid regrounding."""
151
+ facts, ground_rules, candidates = grounding or ground_program(clauses)
152
+ atoms = sorted(candidates, key=lambda a: (a[0], _sort_key(a[1])))
153
+ if len(atoms) > limit_atoms:
154
+ raise DatalogError(
155
+ "stable-model search is limited to %d candidate atoms; this "
156
+ "program grounds to %d" % (limit_atoms, len(atoms)))
157
+ models = []
158
+ for mask in range(1 << len(atoms)):
159
+ M = {atoms[i] for i in range(len(atoms)) if mask >> i & 1} | facts
160
+ if _gamma(M, facts, ground_rules) == M:
161
+ models.append(M)
162
+ return models
163
+
164
+
165
+ def well_founded(clauses, grounding=None):
166
+ """The well-founded (three-valued) model, via Van Gelder's alternating
167
+ fixpoint. Returns (true_atoms, undefined_atoms); everything else is
168
+ false.
169
+
170
+ The trick: Γ is antimonotone, so Γ∘Γ is monotone and has a least
171
+ fixpoint T (the surely-true atoms) reachable by plain iteration.
172
+ Γ(T) is then the upper envelope of the possibly-true; the gap
173
+ Γ(T) - T is precisely the undefined zone — for the café paradox,
174
+ exactly Bob's three atoms."""
175
+ facts, ground_rules, _candidates = grounding or ground_program(clauses)
176
+ T = set(facts)
177
+ while True:
178
+ upper = _gamma(T, facts, ground_rules)
179
+ T2 = _gamma(upper, facts, ground_rules)
180
+ if T2 == T:
181
+ return T, upper - T
182
+ T = T2