tiny-datalog 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tiny_datalog/__init__.py +42 -0
- tiny_datalog/containment.py +228 -0
- tiny_datalog/datalog.py +1455 -0
- tiny_datalog/incremental.py +424 -0
- tiny_datalog/magic.py +175 -0
- tiny_datalog/prolog.py +335 -0
- tiny_datalog/semantics.py +182 -0
- tiny_datalog/semiring.py +383 -0
- tiny_datalog/subsumption.py +475 -0
- tiny_datalog/tabling.py +220 -0
- tiny_datalog-0.1.0.dist-info/METADATA +360 -0
- tiny_datalog-0.1.0.dist-info/RECORD +16 -0
- tiny_datalog-0.1.0.dist-info/WHEEL +5 -0
- tiny_datalog-0.1.0.dist-info/entry_points.txt +8 -0
- tiny_datalog-0.1.0.dist-info/licenses/LICENSE +21 -0
- tiny_datalog-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,424 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
incremental.py — incremental maintenance of Datalog materialisations.
|
|
4
|
+
|
|
5
|
+
Evaluate once; then, as base facts arrive and depart, repair the derived
|
|
6
|
+
relations instead of recomputing them from scratch.
|
|
7
|
+
|
|
8
|
+
* Insertion is semi-naive evaluation *resumed*: the database is already a
|
|
9
|
+
fixpoint of the old facts, so every genuinely new derivation must use at
|
|
10
|
+
least one inserted fact — exactly the delta discipline the engine
|
|
11
|
+
already runs.
|
|
12
|
+
* Deletion ships in two strategies, because the field did:
|
|
13
|
+
- DRed (delete-and-rederive, Gupta–Mumick–Subrahmanian 1993): first
|
|
14
|
+
over-delete everything that has *any* derivation through a deleted
|
|
15
|
+
fact, then re-derive the survivors that still have support from what
|
|
16
|
+
remains.
|
|
17
|
+
- Backward/Forward (Motik–Nenov–Piro–Horrocks 2015, the algorithm in
|
|
18
|
+
RDFox): compute the same affected set, but instead of tearing it
|
|
19
|
+
down, check each fact by *backward chaining* for an alternative
|
|
20
|
+
derivation before touching it. Facts with independent support are
|
|
21
|
+
never disturbed. The search must be well-founded — a fact may not
|
|
22
|
+
support itself through a cycle — which is the same reason counting
|
|
23
|
+
derivations fails on recursion (lesson 8's diverging count semiring).
|
|
24
|
+
|
|
25
|
+
Both repair to exactly the recomputed state; they differ in where the
|
|
26
|
+
work goes. DRed pays teardown-plus-rebuild in proportion to derivation
|
|
27
|
+
*redundancy*; B/F pays proof search in proportion to how hard survival
|
|
28
|
+
is to confirm. Positive programs only — incrementalising negation is
|
|
29
|
+
exactly where the modern theory earns its keep.
|
|
30
|
+
|
|
31
|
+
Run `python3 incremental.py` for a demo: delete one edge from a graph
|
|
32
|
+
with an alternative route and watch DRed over-delete, then re-derive.
|
|
33
|
+
|
|
34
|
+
API
|
|
35
|
+
---
|
|
36
|
+
inc = IncrementalEngine(program_text)
|
|
37
|
+
inc.insert("edge(n2, n9).") -> {"inserted": 1, "derived": 2}
|
|
38
|
+
inc.delete("edge(n3, n4).") -> {"deleted": 1, "over_deleted": 7,
|
|
39
|
+
"rederived": 4, "net_removed": 3}
|
|
40
|
+
inc.delete(..., strategy="bf") # {"affected": 7, "confirmed": 4, ...}
|
|
41
|
+
inc.rels # always the same as recomputing fresh
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
from __future__ import annotations
|
|
45
|
+
|
|
46
|
+
import argparse
|
|
47
|
+
import os
|
|
48
|
+
import sys
|
|
49
|
+
import time
|
|
50
|
+
from collections import defaultdict
|
|
51
|
+
|
|
52
|
+
from tiny_datalog.datalog import (
|
|
53
|
+
DatalogError, Engine, format_fact, parse, Program, read_program, validate,
|
|
54
|
+
_aggregate_of, _match, _sort_key)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class IncrementalEngine:
|
|
58
|
+
def __init__(self, text):
|
|
59
|
+
clauses = parse(text)
|
|
60
|
+
for r in clauses:
|
|
61
|
+
if r.weight is not None:
|
|
62
|
+
raise DatalogError(
|
|
63
|
+
"weights are not supported by the incremental "
|
|
64
|
+
"engine: %s" % r)
|
|
65
|
+
if any(lit.negated for lit in r.body) or _aggregate_of(r.head):
|
|
66
|
+
raise DatalogError(
|
|
67
|
+
"incremental maintenance supports positive, "
|
|
68
|
+
"aggregate-free programs only (negation and "
|
|
69
|
+
"aggregation are non-monotone; maintaining them is "
|
|
70
|
+
"where the modern theory earns its keep): %s" % r)
|
|
71
|
+
self.program = Program(clauses)
|
|
72
|
+
self.engine = Engine(self.program)
|
|
73
|
+
self.engine.run()
|
|
74
|
+
self.rels = self.engine.rels
|
|
75
|
+
self.rules = self.program.rules
|
|
76
|
+
self.base = {(c.head.pred, tuple(a.value for a in c.head.args))
|
|
77
|
+
for c in clauses if not c.body}
|
|
78
|
+
|
|
79
|
+
# -- helpers ------------------------------------------------------------
|
|
80
|
+
|
|
81
|
+
def _facts_of(self, clauses):
|
|
82
|
+
"""Validate incoming fact clauses against the loaded program.
|
|
83
|
+
validate() (seeded with the program's arity map) enforces the
|
|
84
|
+
same invariants Program enforces at load time — ground,
|
|
85
|
+
function-free, arity-consistent — so the materialisation can
|
|
86
|
+
never be corrupted through this door."""
|
|
87
|
+
for c in clauses:
|
|
88
|
+
if c.body:
|
|
89
|
+
raise DatalogError("expected facts only, got a rule: %s" % c)
|
|
90
|
+
if c.weight is not None:
|
|
91
|
+
raise DatalogError(
|
|
92
|
+
"weights are not supported by the incremental "
|
|
93
|
+
"engine: %s" % c)
|
|
94
|
+
validate(clauses, self.program.arity)
|
|
95
|
+
return [(c.head.pred, tuple(a.value for a in c.head.args))
|
|
96
|
+
for c in clauses]
|
|
97
|
+
|
|
98
|
+
def _delta_fires(self, delta):
|
|
99
|
+
"""One round of delta-joins: yield every (head_pred, tuple) some
|
|
100
|
+
rule derives using at least one fact from `delta`. This is the
|
|
101
|
+
engine's own semi-naive discipline exposed as a generator —
|
|
102
|
+
insert() aims it at what's new, delete() aims it at what's
|
|
103
|
+
dying. Same mechanism, two directions."""
|
|
104
|
+
for rule in self.rules:
|
|
105
|
+
for i, lit in enumerate(rule.body):
|
|
106
|
+
if not delta.get(lit.atom.pred):
|
|
107
|
+
continue
|
|
108
|
+
for tup in self.engine._eval_rule(rule, delta_occ=i,
|
|
109
|
+
delta=delta):
|
|
110
|
+
yield rule.head.pred, tup
|
|
111
|
+
|
|
112
|
+
def _propagate(self, delta):
|
|
113
|
+
"""Semi-naive continuation: `delta` maps pred -> tuples already
|
|
114
|
+
added to rels. Returns the set of additionally derived facts."""
|
|
115
|
+
derived = set()
|
|
116
|
+
while delta:
|
|
117
|
+
new_delta = defaultdict(set)
|
|
118
|
+
for pred, tup in self._delta_fires(delta):
|
|
119
|
+
if tup not in self.rels[pred]:
|
|
120
|
+
new_delta[pred].add(tup)
|
|
121
|
+
for pred, tups in new_delta.items():
|
|
122
|
+
self.rels[pred] |= tups
|
|
123
|
+
derived |= {(pred, t) for t in tups}
|
|
124
|
+
delta = new_delta
|
|
125
|
+
return derived
|
|
126
|
+
|
|
127
|
+
def _deleter(self, strategy):
|
|
128
|
+
if strategy == "dred":
|
|
129
|
+
return self._delete_facts
|
|
130
|
+
if strategy == "bf":
|
|
131
|
+
return self._bf_delete_facts
|
|
132
|
+
raise DatalogError("unknown deletion strategy %r: use 'dred' or "
|
|
133
|
+
"'bf'" % (strategy,))
|
|
134
|
+
|
|
135
|
+
# -- the API ------------------------------------------------------------
|
|
136
|
+
|
|
137
|
+
def insert(self, facts_text):
|
|
138
|
+
"""Add base facts; repair derived relations by delta propagation."""
|
|
139
|
+
clauses = parse(facts_text)
|
|
140
|
+
if any(c.retract for c in clauses):
|
|
141
|
+
raise DatalogError(
|
|
142
|
+
"insert() received a retraction — use delete() or apply()")
|
|
143
|
+
return self._insert_facts(self._facts_of(clauses))
|
|
144
|
+
|
|
145
|
+
def delete(self, facts_text, strategy="dred"):
|
|
146
|
+
"""Remove base facts (a trailing `~` is allowed but optional
|
|
147
|
+
here); repair derived relations with DRed or Backward/Forward."""
|
|
148
|
+
deleter = self._deleter(strategy)
|
|
149
|
+
return deleter(self._facts_of(parse(facts_text)))
|
|
150
|
+
|
|
151
|
+
def apply(self, script, strategy="dred"):
|
|
152
|
+
"""Apply a mixed update script in one call: plain facts insert,
|
|
153
|
+
`fact~.` retracts. Deletions run first, then insertions;
|
|
154
|
+
returns the combined stats.
|
|
155
|
+
|
|
156
|
+
inc.apply("edge(n3, n4)~. edge(n2, n9).")
|
|
157
|
+
"""
|
|
158
|
+
# Validate the whole script before touching anything, so a bad
|
|
159
|
+
# fact anywhere in it leaves the materialisation unchanged.
|
|
160
|
+
clauses = parse(script)
|
|
161
|
+
deleter = self._deleter(strategy)
|
|
162
|
+
facts = self._facts_of(clauses)
|
|
163
|
+
deletes = [f for c, f in zip(clauses, facts) if c.retract]
|
|
164
|
+
inserts = [f for c, f in zip(clauses, facts) if not c.retract]
|
|
165
|
+
stats = {}
|
|
166
|
+
if deletes:
|
|
167
|
+
stats.update(deleter(deletes))
|
|
168
|
+
if inserts:
|
|
169
|
+
stats.update(self._insert_facts(inserts))
|
|
170
|
+
return stats
|
|
171
|
+
|
|
172
|
+
def _insert_facts(self, facts):
|
|
173
|
+
delta = defaultdict(set)
|
|
174
|
+
inserted = 0
|
|
175
|
+
for pred, tup in facts:
|
|
176
|
+
# a brand-new predicate fixes its arity here, so a later
|
|
177
|
+
# insert cannot store the same predicate at another arity
|
|
178
|
+
self.program.arity.setdefault(pred, len(tup))
|
|
179
|
+
self.base.add((pred, tup))
|
|
180
|
+
if tup not in self.rels[pred]:
|
|
181
|
+
self.rels[pred].add(tup)
|
|
182
|
+
delta[pred].add(tup)
|
|
183
|
+
inserted += 1
|
|
184
|
+
derived = self._propagate(delta)
|
|
185
|
+
return {"inserted": inserted, "derived": len(derived)}
|
|
186
|
+
|
|
187
|
+
def _delete_facts(self, facts):
|
|
188
|
+
for f in facts:
|
|
189
|
+
if f not in self.base:
|
|
190
|
+
raise DatalogError(
|
|
191
|
+
"can only delete base facts; %s is not one"
|
|
192
|
+
% format_fact(*f))
|
|
193
|
+
|
|
194
|
+
# Phase 1: over-delete. A fact is a candidate if any derivation
|
|
195
|
+
# of it passes through a deleted fact — computed with delta-joins
|
|
196
|
+
# against the *pre-deletion* database.
|
|
197
|
+
frontier = defaultdict(set)
|
|
198
|
+
candidates = set()
|
|
199
|
+
for pred, tup in facts:
|
|
200
|
+
self.base.discard((pred, tup))
|
|
201
|
+
if tup in self.rels[pred]:
|
|
202
|
+
frontier[pred].add(tup)
|
|
203
|
+
candidates.add((pred, tup))
|
|
204
|
+
while frontier:
|
|
205
|
+
nxt = defaultdict(set)
|
|
206
|
+
for pred, tup in self._delta_fires(frontier):
|
|
207
|
+
if tup in self.rels[pred] and (pred, tup) not in candidates:
|
|
208
|
+
candidates.add((pred, tup))
|
|
209
|
+
nxt[pred].add(tup)
|
|
210
|
+
frontier = nxt
|
|
211
|
+
|
|
212
|
+
for pred, tup in candidates:
|
|
213
|
+
self.rels[pred].discard(tup)
|
|
214
|
+
|
|
215
|
+
# Explicit base facts swept up as collateral survive.
|
|
216
|
+
seed = defaultdict(set)
|
|
217
|
+
for pred, tup in candidates:
|
|
218
|
+
if (pred, tup) in self.base:
|
|
219
|
+
self.rels[pred].add(tup)
|
|
220
|
+
seed[pred].add(tup)
|
|
221
|
+
|
|
222
|
+
# Phase 2: re-derive candidates that still have a derivation from
|
|
223
|
+
# the surviving facts, then propagate. Only rules that can head
|
|
224
|
+
# a candidate need re-evaluating.
|
|
225
|
+
cand_preds = {p for p, _t in candidates}
|
|
226
|
+
for rule in self.rules:
|
|
227
|
+
if rule.head.pred not in cand_preds:
|
|
228
|
+
continue
|
|
229
|
+
for tup in self.engine._eval_rule(rule):
|
|
230
|
+
f = (rule.head.pred, tup)
|
|
231
|
+
if f in candidates and tup not in self.rels[rule.head.pred]:
|
|
232
|
+
self.rels[rule.head.pred].add(tup)
|
|
233
|
+
seed[rule.head.pred].add(tup)
|
|
234
|
+
self._propagate(seed)
|
|
235
|
+
|
|
236
|
+
rederived = sum(1 for pred, tup in candidates
|
|
237
|
+
if tup in self.rels[pred])
|
|
238
|
+
# keep the "same as recomputing fresh" invariant exact: a fresh
|
|
239
|
+
# engine has no entry for a predicate with no facts at all
|
|
240
|
+
for pred in [p for p, ts in self.rels.items() if not ts]:
|
|
241
|
+
del self.rels[pred]
|
|
242
|
+
return {"deleted": len(facts),
|
|
243
|
+
"over_deleted": len(candidates),
|
|
244
|
+
"rederived": rederived,
|
|
245
|
+
"net_removed": len(candidates) - rederived}
|
|
246
|
+
|
|
247
|
+
def _bf_delete_facts(self, facts):
|
|
248
|
+
"""Backward/Forward: forward-propagate the affected set against
|
|
249
|
+
the intact database, then decide each affected fact by backward
|
|
250
|
+
proof search before removing anything. `blocked` carries the
|
|
251
|
+
current proof path, so support is well-founded by construction —
|
|
252
|
+
a fact cannot survive by deriving itself around a cycle."""
|
|
253
|
+
for f in facts:
|
|
254
|
+
if f not in self.base:
|
|
255
|
+
raise DatalogError("can only delete base facts; %s is "
|
|
256
|
+
"not one" % format_fact(*f))
|
|
257
|
+
self.base -= set(facts)
|
|
258
|
+
frontier, affected = defaultdict(set), set()
|
|
259
|
+
for pred, tup in facts:
|
|
260
|
+
if tup in self.rels.get(pred, ()):
|
|
261
|
+
frontier[pred].add(tup)
|
|
262
|
+
affected.add((pred, tup))
|
|
263
|
+
while frontier:
|
|
264
|
+
nxt = defaultdict(set)
|
|
265
|
+
for pred, tup in self._delta_fires(frontier):
|
|
266
|
+
if tup in self.rels[pred] and (pred, tup) not in affected:
|
|
267
|
+
affected.add((pred, tup))
|
|
268
|
+
nxt[pred].add(tup)
|
|
269
|
+
frontier = nxt
|
|
270
|
+
|
|
271
|
+
by_head = defaultdict(list)
|
|
272
|
+
for r in self.rules:
|
|
273
|
+
by_head[r.head.pred].append(r)
|
|
274
|
+
proven, checks = {}, [0]
|
|
275
|
+
|
|
276
|
+
def usable(f, blocked):
|
|
277
|
+
if f in blocked:
|
|
278
|
+
return False
|
|
279
|
+
if f not in affected or f in self.base:
|
|
280
|
+
return True
|
|
281
|
+
if f in proven:
|
|
282
|
+
return proven[f]
|
|
283
|
+
return prove(f, blocked | {f})
|
|
284
|
+
|
|
285
|
+
def prove(f, blocked):
|
|
286
|
+
# only successes are cached: failure under a nonempty proof
|
|
287
|
+
# path may succeed by another route; False is recorded only
|
|
288
|
+
# after a top-level search exhausts them all
|
|
289
|
+
checks[0] += 1
|
|
290
|
+
pred, tup = f
|
|
291
|
+
for rule in by_head[pred]:
|
|
292
|
+
subst = _match(rule.head.args, tup, {})
|
|
293
|
+
if subst is not None and solve(rule.body, 0, subst, blocked):
|
|
294
|
+
proven[f] = True
|
|
295
|
+
return True
|
|
296
|
+
return False
|
|
297
|
+
|
|
298
|
+
def solve(body, i, subst, blocked):
|
|
299
|
+
if i == len(body):
|
|
300
|
+
return True
|
|
301
|
+
atom = body[i].atom
|
|
302
|
+
for tup in sorted(self.rels.get(atom.pred, ()), key=_sort_key):
|
|
303
|
+
ext = _match(atom.args, tup, subst)
|
|
304
|
+
if (ext is not None and usable((atom.pred, tup), blocked)
|
|
305
|
+
and solve(body, i + 1, ext, blocked)):
|
|
306
|
+
return True
|
|
307
|
+
return False
|
|
308
|
+
|
|
309
|
+
removed = 0
|
|
310
|
+
order = sorted(affected, key=lambda f: (f[0], _sort_key(f[1])))
|
|
311
|
+
for f in order:
|
|
312
|
+
alive = (f in self.base or proven.get(f) is True
|
|
313
|
+
or (f not in proven and prove(f, {f})))
|
|
314
|
+
if alive:
|
|
315
|
+
continue
|
|
316
|
+
proven[f] = False
|
|
317
|
+
self.rels[f[0]].discard(f[1])
|
|
318
|
+
removed += 1
|
|
319
|
+
for pred in [p for p, ts in self.rels.items() if not ts]:
|
|
320
|
+
del self.rels[pred]
|
|
321
|
+
return {"deleted": len(facts), "affected": len(affected),
|
|
322
|
+
"confirmed": len(affected) - removed, "removed": removed,
|
|
323
|
+
"backward_checks": checks[0]}
|
|
324
|
+
|
|
325
|
+
def total_facts(self):
|
|
326
|
+
return sum(len(ts) for ts in self.rels.values())
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
# ---------------------------------------------------------------------------
|
|
330
|
+
# Demo
|
|
331
|
+
# ---------------------------------------------------------------------------
|
|
332
|
+
|
|
333
|
+
# The demo reads a teaching program from the repo root, one level above
|
|
334
|
+
# this package. An install ships the package alone, so the file is
|
|
335
|
+
# absent there — hence the explanation rather than a bare traceback.
|
|
336
|
+
_DEMO_FILE = os.path.join(
|
|
337
|
+
os.path.dirname(os.path.dirname(os.path.abspath(__file__))),
|
|
338
|
+
"programs", "dred-graph.dl")
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
def _demo():
|
|
342
|
+
try:
|
|
343
|
+
with open(_DEMO_FILE) as fh:
|
|
344
|
+
text = fh.read()
|
|
345
|
+
except FileNotFoundError:
|
|
346
|
+
raise SystemExit(
|
|
347
|
+
"the no-argument demo needs programs/dred-graph.dl, which "
|
|
348
|
+
"ships with the repository rather than with the installed "
|
|
349
|
+
"package — clone tiny-datalog and run it there, or pass a "
|
|
350
|
+
".dl file of your own.")
|
|
351
|
+
t0 = time.perf_counter()
|
|
352
|
+
inc = IncrementalEngine(text)
|
|
353
|
+
print("Initial: %d path facts, materialised in %.3fs"
|
|
354
|
+
% (len(inc.rels["path"]), time.perf_counter() - t0))
|
|
355
|
+
print()
|
|
356
|
+
print('delete("edge(n3, n4).") — n2\'s shortcut keeps most paths alive:')
|
|
357
|
+
stats = inc.delete("edge(n3, n4).")
|
|
358
|
+
print(" %r" % stats)
|
|
359
|
+
print(" over-deleted %d candidates, re-derived %d — only %d facts "
|
|
360
|
+
"actually died" % (stats["over_deleted"], stats["rederived"],
|
|
361
|
+
stats["net_removed"]))
|
|
362
|
+
print(" now: %d path facts" % len(inc.rels["path"]))
|
|
363
|
+
print()
|
|
364
|
+
print('insert("edge(n3, n4).") — put it back:')
|
|
365
|
+
stats = inc.insert("edge(n3, n4).")
|
|
366
|
+
print(" %r" % stats)
|
|
367
|
+
print(" now: %d path facts" % len(inc.rels["path"]))
|
|
368
|
+
# sanity: identical to computing from scratch
|
|
369
|
+
fresh = Engine(Program(parse(text)))
|
|
370
|
+
fresh.run()
|
|
371
|
+
assert inc.rels["path"] == fresh.rels["path"]
|
|
372
|
+
print()
|
|
373
|
+
print("Repaired state verified equal to a from-scratch recomputation.")
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
def main(argv=None):
|
|
377
|
+
ap = argparse.ArgumentParser(
|
|
378
|
+
description="Materialise a program, then repair it under updates "
|
|
379
|
+
"instead of recomputing. Update scripts mix inserts "
|
|
380
|
+
"(plain facts) and retractions (`fact~.`).")
|
|
381
|
+
ap.add_argument("file", nargs="?",
|
|
382
|
+
help="program to materialise; omit to run the "
|
|
383
|
+
"built-in DRed demo")
|
|
384
|
+
ap.add_argument("-u", "--update", action="append", default=[],
|
|
385
|
+
metavar="SCRIPT",
|
|
386
|
+
help="update script applied in order, e.g. "
|
|
387
|
+
"'edge(n3, n4)~. edge(n2, n9).' (repeatable)")
|
|
388
|
+
ap.add_argument("-p", "--print", dest="show", action="store_true",
|
|
389
|
+
help="print derived relations after the updates")
|
|
390
|
+
ap.add_argument("--strategy", choices=("dred", "bf"), default="dred",
|
|
391
|
+
help="deletion strategy: DRed (default) or "
|
|
392
|
+
"Backward/Forward")
|
|
393
|
+
args = ap.parse_args(argv)
|
|
394
|
+
|
|
395
|
+
if not args.file:
|
|
396
|
+
return _demo()
|
|
397
|
+
try:
|
|
398
|
+
inc = IncrementalEngine(read_program(args.file))
|
|
399
|
+
print("materialised: %d facts" % inc.total_facts())
|
|
400
|
+
for script in args.update:
|
|
401
|
+
t0 = time.perf_counter()
|
|
402
|
+
stats = inc.apply(script, args.strategy)
|
|
403
|
+
elapsed = time.perf_counter() - t0
|
|
404
|
+
print("%s\n -> %r in %.3fs" % (script.strip(), stats, elapsed))
|
|
405
|
+
if args.update:
|
|
406
|
+
t0 = time.perf_counter()
|
|
407
|
+
fresh = Engine(Program(parse(read_program(args.file))))
|
|
408
|
+
fresh.run()
|
|
409
|
+
print(" (a from-scratch rebuild of this program: %.3fs)"
|
|
410
|
+
% (time.perf_counter() - t0))
|
|
411
|
+
except DatalogError as exc:
|
|
412
|
+
print("error: %s" % exc, file=sys.stderr)
|
|
413
|
+
return 1
|
|
414
|
+
if args.show:
|
|
415
|
+
from tiny_datalog.datalog import _sort_key, format_fact
|
|
416
|
+
print()
|
|
417
|
+
for pred in sorted(inc.program.idb):
|
|
418
|
+
for tup in sorted(inc.rels.get(pred, ()), key=_sort_key):
|
|
419
|
+
print(format_fact(pred, tup))
|
|
420
|
+
return 0
|
|
421
|
+
|
|
422
|
+
|
|
423
|
+
if __name__ == "__main__":
|
|
424
|
+
sys.exit(main())
|
tiny_datalog/magic.py
ADDED
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
magic.py — the magic-sets transformation: goal-directed queries for a
|
|
4
|
+
bottom-up engine. (Lesson 7, which ends with a tour of this module.)
|
|
5
|
+
|
|
6
|
+
The problem: bottom-up evaluation computes *whole relations*, but a query
|
|
7
|
+
like path(n5, X) only needs facts reachable from n5. Top-down engines
|
|
8
|
+
(Prolog) get this focus for free by starting from the goal — at the cost
|
|
9
|
+
of possible non-termination.
|
|
10
|
+
|
|
11
|
+
The 1986 resolution: don't change the evaluator, change the *program*.
|
|
12
|
+
Given the query, rewrite the rules so that bottom-up evaluation of the
|
|
13
|
+
rewritten program derives only what a top-down engine would have visited.
|
|
14
|
+
Three ingredients:
|
|
15
|
+
|
|
16
|
+
* An **adornment** records which arguments of a predicate are bound at
|
|
17
|
+
call time: path(n5, X) calls path with pattern "bf" (bound, free).
|
|
18
|
+
Different call patterns get separately specialised copies of the rules,
|
|
19
|
+
named pred#bf, pred#fb, ...
|
|
20
|
+
* A **magic predicate** magic#pred#adorn holds the bound-argument tuples
|
|
21
|
+
actually demanded — "someone needs paths starting at n5". It is seeded
|
|
22
|
+
with the query's constants and grows exactly like Prolog's call stack
|
|
23
|
+
would: through each rule, bindings flow left to right (a "sideways
|
|
24
|
+
information passing strategy", here: the evaluator's own join order).
|
|
25
|
+
* Every specialised rule is **guarded** by its magic predicate, so it can
|
|
26
|
+
only fire for demanded bindings.
|
|
27
|
+
|
|
28
|
+
Negation gets the simple sound treatment: a negated subgoal is never
|
|
29
|
+
specialised — its predicate (and everything it depends on) is included
|
|
30
|
+
untransformed and computed in full. Since negative edges then point only
|
|
31
|
+
from the rewritten world into the original one, the rewriting is
|
|
32
|
+
stratified whenever the original program is.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
from __future__ import annotations
|
|
36
|
+
|
|
37
|
+
from collections import defaultdict
|
|
38
|
+
|
|
39
|
+
from tiny_datalog.datalog import (
|
|
40
|
+
Atom, Const, Engine, Literal, Program, Rule, Var, _aggregate_of,
|
|
41
|
+
check_query_atom, match_answers, validate)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _adorned_name(pred, adorn):
|
|
45
|
+
return "%s#%s" % (pred, adorn)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _magic_name(pred, adorn):
|
|
49
|
+
return "magic#%s#%s" % (pred, adorn)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def magic_transform(clauses, query):
|
|
53
|
+
"""Magic-sets rewriting of the program for `query` (an Atom, possibly
|
|
54
|
+
with variables). Returns (transformed_clauses, answer_pred): evaluate
|
|
55
|
+
the transformed program and read the query's answers from answer_pred.
|
|
56
|
+
"""
|
|
57
|
+
check_query_atom(query, validate(clauses))
|
|
58
|
+
idb = {c.head.pred for c in clauses if c.body}
|
|
59
|
+
# Aggregate-headed predicates are never specialised: an aggregate
|
|
60
|
+
# needs its whole group, so demand restriction would change answers.
|
|
61
|
+
# Like negated subgoals, they are computed in full.
|
|
62
|
+
agg_preds = {c.head.pred for c in clauses
|
|
63
|
+
if c.body and _aggregate_of(c.head)}
|
|
64
|
+
if query.pred not in idb or query.pred in agg_preds:
|
|
65
|
+
return list(clauses), query.pred # evaluate in full
|
|
66
|
+
|
|
67
|
+
defs = defaultdict(list) # IDB pred -> its clauses (rules AND facts)
|
|
68
|
+
edb_facts = []
|
|
69
|
+
for c in clauses:
|
|
70
|
+
if c.head.pred in idb:
|
|
71
|
+
defs[c.head.pred].append(c)
|
|
72
|
+
else:
|
|
73
|
+
edb_facts.append(c)
|
|
74
|
+
|
|
75
|
+
# The query's own adornment: constants are bound, variables free.
|
|
76
|
+
query_adorn = "".join("b" if isinstance(a, Const) else "f"
|
|
77
|
+
for a in query.args)
|
|
78
|
+
|
|
79
|
+
out = []
|
|
80
|
+
full_needed = set() # predicates under negation: evaluate in full
|
|
81
|
+
seen = set()
|
|
82
|
+
# Worklist of (predicate, adornment) call patterns still to specialise.
|
|
83
|
+
# Processing one pattern may discover new ones in rule bodies — exactly
|
|
84
|
+
# how a top-down engine discovers subgoals.
|
|
85
|
+
work = [(query.pred, query_adorn)]
|
|
86
|
+
while work:
|
|
87
|
+
pred, adorn = work.pop()
|
|
88
|
+
if (pred, adorn) in seen:
|
|
89
|
+
continue
|
|
90
|
+
seen.add((pred, adorn))
|
|
91
|
+
for clause in defs[pred]:
|
|
92
|
+
head = clause.head
|
|
93
|
+
# Variables bound on entry: those in the head's 'b' positions.
|
|
94
|
+
bound_vars = {head.args[i].name
|
|
95
|
+
for i, ch in enumerate(adorn)
|
|
96
|
+
if ch == "b" and isinstance(head.args[i], Var)}
|
|
97
|
+
magic_args = tuple(head.args[i]
|
|
98
|
+
for i, ch in enumerate(adorn) if ch == "b")
|
|
99
|
+
# `prefix` = the magic guard + body literals transformed so
|
|
100
|
+
# far. A snapshot of it, taken just before an IDB subgoal, is
|
|
101
|
+
# that subgoal's demand context: "if evaluation got this far,
|
|
102
|
+
# these bindings are being asked for."
|
|
103
|
+
prefix = [Literal(Atom(_magic_name(pred, adorn), magic_args))]
|
|
104
|
+
negatives = []
|
|
105
|
+
for lit in clause.body:
|
|
106
|
+
if lit.negated:
|
|
107
|
+
negatives.append(lit) # untouched; defined in full
|
|
108
|
+
if lit.atom.pred in idb:
|
|
109
|
+
full_needed.add(lit.atom.pred)
|
|
110
|
+
continue
|
|
111
|
+
if lit.atom.pred in agg_preds:
|
|
112
|
+
prefix.append(lit) # aggregate: keep original name
|
|
113
|
+
full_needed.add(lit.atom.pred)
|
|
114
|
+
elif lit.atom.pred in idb:
|
|
115
|
+
# Adorn the subgoal from what is bound *right now*.
|
|
116
|
+
sub = "".join(
|
|
117
|
+
"b" if isinstance(a, Const) or a.name in bound_vars
|
|
118
|
+
else "f"
|
|
119
|
+
for a in lit.atom.args)
|
|
120
|
+
sub_bound = tuple(a for a, ch in zip(lit.atom.args, sub)
|
|
121
|
+
if ch == "b")
|
|
122
|
+
# Magic rule: this subgoal's bindings are demanded
|
|
123
|
+
# whenever the prefix so far succeeds.
|
|
124
|
+
out.append(Rule(Atom(_magic_name(lit.atom.pred, sub),
|
|
125
|
+
sub_bound),
|
|
126
|
+
tuple(prefix)))
|
|
127
|
+
work.append((lit.atom.pred, sub))
|
|
128
|
+
prefix.append(
|
|
129
|
+
Literal(Atom(_adorned_name(lit.atom.pred, sub),
|
|
130
|
+
lit.atom.args)))
|
|
131
|
+
else:
|
|
132
|
+
prefix.append(lit) # EDB literal: kept as-is
|
|
133
|
+
# A positive literal, once joined, binds all its variables
|
|
134
|
+
# (aggregate heads included: their tuples are ordinary)
|
|
135
|
+
# for everything to its right — the evaluator's own order.
|
|
136
|
+
bound_vars |= {a.name for a in lit.atom.args
|
|
137
|
+
if isinstance(a, Var)}
|
|
138
|
+
# The specialised rule: original head under its adorned name,
|
|
139
|
+
# guarded by magic, negatives at the end (they only filter).
|
|
140
|
+
out.append(Rule(Atom(_adorned_name(pred, adorn), head.args),
|
|
141
|
+
tuple(prefix) + tuple(negatives)))
|
|
142
|
+
|
|
143
|
+
# Include negated subgoals' definitions untransformed, transitively.
|
|
144
|
+
stack = list(full_needed)
|
|
145
|
+
included = set()
|
|
146
|
+
while stack:
|
|
147
|
+
p = stack.pop()
|
|
148
|
+
if p in included:
|
|
149
|
+
continue
|
|
150
|
+
included.add(p)
|
|
151
|
+
for c in defs[p]:
|
|
152
|
+
out.append(c)
|
|
153
|
+
for lit in c.body:
|
|
154
|
+
if lit.atom.pred in idb and lit.atom.pred not in included:
|
|
155
|
+
stack.append(lit.atom.pred)
|
|
156
|
+
|
|
157
|
+
# Seed the query's magic predicate with the query constants, keep EDB.
|
|
158
|
+
seed_args = tuple(a for a in query.args if isinstance(a, Const))
|
|
159
|
+
out.append(Rule(Atom(_magic_name(query.pred, query_adorn), seed_args), ()))
|
|
160
|
+
out.extend(edb_facts)
|
|
161
|
+
|
|
162
|
+
# The same magic rule can be generated from several adornment passes;
|
|
163
|
+
# rules are hashable (frozen dataclasses), so dedupe preserving order.
|
|
164
|
+
return list(dict.fromkeys(out)), _adorned_name(query.pred, query_adorn)
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def magic_query(clauses, query):
|
|
168
|
+
"""Answer `query` via magic-sets rewriting + semi-naive evaluation.
|
|
169
|
+
Returns (engine, answers): the engine that ran the rewritten program,
|
|
170
|
+
and the set of ground tuples matching the query."""
|
|
171
|
+
transformed, answer_pred = magic_transform(clauses, query)
|
|
172
|
+
engine = Engine(Program(transformed))
|
|
173
|
+
engine.run()
|
|
174
|
+
answers = set(match_answers(query, engine.rels.get(answer_pred, ())))
|
|
175
|
+
return engine, answers
|