tiny-datalog 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tiny_datalog/__init__.py +42 -0
- tiny_datalog/containment.py +228 -0
- tiny_datalog/datalog.py +1455 -0
- tiny_datalog/incremental.py +424 -0
- tiny_datalog/magic.py +175 -0
- tiny_datalog/prolog.py +335 -0
- tiny_datalog/semantics.py +182 -0
- tiny_datalog/semiring.py +383 -0
- tiny_datalog/subsumption.py +475 -0
- tiny_datalog/tabling.py +220 -0
- tiny_datalog-0.1.0.dist-info/METADATA +360 -0
- tiny_datalog-0.1.0.dist-info/RECORD +16 -0
- tiny_datalog-0.1.0.dist-info/WHEEL +5 -0
- tiny_datalog-0.1.0.dist-info/entry_points.txt +8 -0
- tiny_datalog-0.1.0.dist-info/licenses/LICENSE +21 -0
- tiny_datalog-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,475 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
subsumption.py — KL-ONE-style concept subsumption, compiled to Datalog.
|
|
4
|
+
(Lesson 12, which ends with a tour of this module.)
|
|
5
|
+
|
|
6
|
+
KL-ONE (Brachman, late 1970s) organised knowledge as *concepts* with
|
|
7
|
+
structured definitions, and its party trick was the classifier: state
|
|
8
|
+
what a Father is, and the system *discovers* where Father belongs in the
|
|
9
|
+
concept hierarchy. The reasoning service underneath is **subsumption**:
|
|
10
|
+
C is subsumed by D (written C ⊑ D) iff every possible instance of C must
|
|
11
|
+
be an instance of D — a statement about definitions, not about any
|
|
12
|
+
particular database.
|
|
13
|
+
|
|
14
|
+
This module implements subsumption for the EL concept language —
|
|
15
|
+
conjunction and existential restriction — by the standard
|
|
16
|
+
completion-rule calculus. EL is the tractable core that the
|
|
17
|
+
SNOMED-scale reasoners (ELK, Snorocket) are built on; they implement
|
|
18
|
+
its extensions (EL++/ELH: top, bottom, role hierarchies, right
|
|
19
|
+
identities), which SNOMED CT actually requires and this module does
|
|
20
|
+
not. See "Where this stops", below. The twist that earns it a place in this
|
|
21
|
+
repository: after normalisation, the completion rules are *literally a
|
|
22
|
+
positive Datalog program*, and this module simply compiles the ontology
|
|
23
|
+
to facts + seven rules and hands them to the engine from datalog.py.
|
|
24
|
+
Classification is a fixpoint; goal-directed subsumption checks even work
|
|
25
|
+
under magic sets. (`--emit` prints the compiled Datalog so you can run
|
|
26
|
+
it yourself.)
|
|
27
|
+
|
|
28
|
+
Ontology syntax (parsed with the repository's own parser — concept
|
|
29
|
+
expressions are the compound terms Datalog itself forbids):
|
|
30
|
+
|
|
31
|
+
isa(man, person). % primitive: necessary only
|
|
32
|
+
define(parent, and(person, some(has_child, person))).
|
|
33
|
+
% defined: necessary AND
|
|
34
|
+
% sufficient — classifiable
|
|
35
|
+
role(has_child). % optional declaration
|
|
36
|
+
disjoint(cat, dog). % cat ⊓ dog ⊑ ⊥ (EL⊥)
|
|
37
|
+
|
|
38
|
+
Expressions: atomic names, and(...) with two or more conjuncts, and
|
|
39
|
+
some(role, expression). The predicates subs, link, concept, isa1, isa2,
|
|
40
|
+
isa_some, some_isa, the concept name bot (⊥) and concept names beginning
|
|
41
|
+
gen_ are reserved.
|
|
42
|
+
|
|
43
|
+
Normalisation introduces fresh names (gen_1, gen_2, ...) for nested
|
|
44
|
+
complex expressions — one inclusion per fresh name, direction chosen by
|
|
45
|
+
which side of ⊑ the expression sits on. This is a conservative
|
|
46
|
+
extension: subsumptions among the *named* concepts are unchanged.
|
|
47
|
+
|
|
48
|
+
Where this stops
|
|
49
|
+
----------------
|
|
50
|
+
EL⊥: EL plus disjointness (`disjoint/2` = A ⊓ B ⊑ ⊥), enough to detect
|
|
51
|
+
unsatisfiable definitions via two extra rules (⊥ is below everything;
|
|
52
|
+
∃r.⊥ is ⊥). No ⊤, no role hierarchies (`subrole/2` is rejected, not
|
|
53
|
+
ignored), no role chains, nominals, datatypes, or ABox: definitions
|
|
54
|
+
only, never individuals. The completion-rule *method* extends to all of that —
|
|
55
|
+
that is exactly how EL++ reasoners are built — but these seven rules
|
|
56
|
+
are complete only for what is listed above.
|
|
57
|
+
"""
|
|
58
|
+
|
|
59
|
+
from __future__ import annotations
|
|
60
|
+
|
|
61
|
+
import argparse
|
|
62
|
+
import sys
|
|
63
|
+
import time
|
|
64
|
+
|
|
65
|
+
from collections import defaultdict
|
|
66
|
+
|
|
67
|
+
from tiny_datalog.datalog import (
|
|
68
|
+
Atom, Const, DatalogError, Engine, Literal, parse, Program, read_program,
|
|
69
|
+
Rule, Struct, Var)
|
|
70
|
+
|
|
71
|
+
_RESERVED = {"subs", "link", "concept", "isa1", "isa2", "isa_some",
|
|
72
|
+
"some_isa", "bot"}
|
|
73
|
+
|
|
74
|
+
class Ontology:
|
|
75
|
+
"""An EL TBox, normalised on load into the four axiom forms:
|
|
76
|
+
|
|
77
|
+
isa1(A, B) A ⊑ B
|
|
78
|
+
isa2(A1, A2, B) A1 ⊓ A2 ⊑ B
|
|
79
|
+
isa_some(A, r, B) A ⊑ ∃r.B
|
|
80
|
+
some_isa(r, A, B) ∃r.A ⊑ B
|
|
81
|
+
"""
|
|
82
|
+
|
|
83
|
+
def __init__(self):
|
|
84
|
+
self.isa1 = []
|
|
85
|
+
self.isa2 = []
|
|
86
|
+
self.isa_some = []
|
|
87
|
+
self.some_isa = []
|
|
88
|
+
self.disjoint = [] # (A, B) pairs: A ⊓ B ⊑ ⊥
|
|
89
|
+
self.concepts = set() # every atomic name, fresh ones included
|
|
90
|
+
self.named = set() # concepts the user actually named
|
|
91
|
+
self.roles = set()
|
|
92
|
+
self.told = set() # (sub, super) pairs stated syntactically
|
|
93
|
+
self._memo = {} # normalised expression -> fresh name
|
|
94
|
+
self._fresh = 0
|
|
95
|
+
self._supers = None # classification cache
|
|
96
|
+
|
|
97
|
+
# -- loading ------------------------------------------------------------
|
|
98
|
+
|
|
99
|
+
@classmethod
|
|
100
|
+
def from_text(cls, text):
|
|
101
|
+
ont = cls()
|
|
102
|
+
for clause in parse(text):
|
|
103
|
+
if clause.body:
|
|
104
|
+
raise DatalogError(
|
|
105
|
+
"an ontology contains only facts, not rules: %s" % clause)
|
|
106
|
+
pred, args = clause.head.pred, clause.head.args
|
|
107
|
+
if pred == "isa" and len(args) == 2:
|
|
108
|
+
ont._axiom(args[0], args[1], told=True)
|
|
109
|
+
elif pred == "define" and len(args) == 2:
|
|
110
|
+
if not isinstance(args[0], Const):
|
|
111
|
+
raise DatalogError(
|
|
112
|
+
"define/2 needs an atomic concept name: %s" % clause)
|
|
113
|
+
# A definition is an equivalence: both inclusions.
|
|
114
|
+
ont._axiom(args[0], args[1], told=True)
|
|
115
|
+
ont._axiom(args[1], args[0])
|
|
116
|
+
elif pred == "role" and len(args) == 1:
|
|
117
|
+
ont.roles.add(ont._role_name(args[0]))
|
|
118
|
+
elif pred == "primitive" and len(args) == 1:
|
|
119
|
+
ont._concept_name(args[0])
|
|
120
|
+
elif pred == "disjoint" and len(args) == 2:
|
|
121
|
+
ont.disjoint.append((ont._concept_name(args[0]),
|
|
122
|
+
ont._concept_name(args[1])))
|
|
123
|
+
else:
|
|
124
|
+
raise DatalogError(
|
|
125
|
+
"unknown ontology statement %s/%d (expected isa/2, "
|
|
126
|
+
"define/2, role/1, primitive/1 or disjoint/2): %s"
|
|
127
|
+
% (pred, len(args), clause))
|
|
128
|
+
return ont
|
|
129
|
+
|
|
130
|
+
# -- names --------------------------------------------------------------
|
|
131
|
+
|
|
132
|
+
def _concept_name(self, term):
|
|
133
|
+
if not isinstance(term, Const) or not isinstance(term.value, str):
|
|
134
|
+
raise DatalogError("expected an atomic concept name, got %s" % (term,))
|
|
135
|
+
if term.value in _RESERVED or term.value.startswith("gen_"):
|
|
136
|
+
raise DatalogError("%r is reserved by the compilation%s" % (term.value,
|
|
137
|
+
"" if term.value in _RESERVED else # gen_N would merge with a fresh name
|
|
138
|
+
" (gen_ names are the fresh concepts of normalisation)"))
|
|
139
|
+
self.concepts.add(term.value)
|
|
140
|
+
self.named.add(term.value)
|
|
141
|
+
return term.value
|
|
142
|
+
|
|
143
|
+
def _role_name(self, term):
|
|
144
|
+
if not isinstance(term, Const) or not isinstance(term.value, str):
|
|
145
|
+
raise DatalogError("expected an atomic role name, got %s" % (term,))
|
|
146
|
+
self.roles.add(term.value)
|
|
147
|
+
return term.value
|
|
148
|
+
|
|
149
|
+
def _gen(self):
|
|
150
|
+
self._fresh += 1
|
|
151
|
+
name = "gen_%d" % self._fresh
|
|
152
|
+
self.concepts.add(name) # fresh names are concepts, but not named
|
|
153
|
+
return name
|
|
154
|
+
|
|
155
|
+
# -- normalisation ------------------------------------------------------
|
|
156
|
+
#
|
|
157
|
+
# The direction of each fresh-name inclusion follows the expression's
|
|
158
|
+
# side of ⊑. On the right (A ⊑ ...expr...) the fresh name goes BELOW
|
|
159
|
+
# the expression; on the left (...expr... ⊑ B) it goes ABOVE. Either
|
|
160
|
+
# way the extension is conservative — a model may always interpret the
|
|
161
|
+
# fresh name as exactly the expression.
|
|
162
|
+
|
|
163
|
+
@staticmethod
|
|
164
|
+
def _conjuncts(expr):
|
|
165
|
+
# flatten nested and(...)s into one list
|
|
166
|
+
if isinstance(expr, Struct) and expr.functor == "and":
|
|
167
|
+
out = []
|
|
168
|
+
for a in expr.args:
|
|
169
|
+
out.extend(Ontology._conjuncts(a))
|
|
170
|
+
return out
|
|
171
|
+
return [expr]
|
|
172
|
+
|
|
173
|
+
def _axiom(self, lhs, rhs, told=False):
|
|
174
|
+
"""Assert lhs ⊑ rhs for arbitrary expressions."""
|
|
175
|
+
if isinstance(lhs, Const):
|
|
176
|
+
self._sub_atom_expr(self._concept_name(lhs), rhs, told=told)
|
|
177
|
+
elif isinstance(rhs, Const):
|
|
178
|
+
self._sub_expr_atom(lhs, self._concept_name(rhs))
|
|
179
|
+
else:
|
|
180
|
+
mid = self._gen() # Ĉ ⊑ D̂ → Ĉ ⊑ A, A ⊑ D̂
|
|
181
|
+
self._sub_expr_atom(lhs, mid)
|
|
182
|
+
self._sub_atom_expr(mid, rhs)
|
|
183
|
+
|
|
184
|
+
def _sub_atom_expr(self, a, expr, told=False):
|
|
185
|
+
"""a ⊑ expr, a atomic."""
|
|
186
|
+
if isinstance(expr, Const):
|
|
187
|
+
b = self._concept_name(expr)
|
|
188
|
+
self.isa1.append((a, b))
|
|
189
|
+
if told:
|
|
190
|
+
self.told.add((a, b))
|
|
191
|
+
elif isinstance(expr, Struct) and expr.functor == "and":
|
|
192
|
+
for c in self._conjuncts(expr):
|
|
193
|
+
self._sub_atom_expr(a, c, told=told)
|
|
194
|
+
elif isinstance(expr, Struct) and expr.functor == "some" \
|
|
195
|
+
and len(expr.args) == 2:
|
|
196
|
+
r = self._role_name(expr.args[0])
|
|
197
|
+
self.isa_some.append((a, r, self._below(expr.args[1])))
|
|
198
|
+
else:
|
|
199
|
+
raise DatalogError("not an EL concept expression: %s" % (expr,))
|
|
200
|
+
|
|
201
|
+
def _sub_expr_atom(self, expr, b):
|
|
202
|
+
"""expr ⊑ b, b atomic."""
|
|
203
|
+
if isinstance(expr, Const):
|
|
204
|
+
self.isa1.append((self._concept_name(expr), b))
|
|
205
|
+
elif isinstance(expr, Struct) and expr.functor == "and":
|
|
206
|
+
atoms = [self._above(c) for c in self._conjuncts(expr)]
|
|
207
|
+
# binarise a1 ⊓ a2 ⊓ a3 ⊑ b through fresh intermediates
|
|
208
|
+
while len(atoms) > 2:
|
|
209
|
+
t = self._gen()
|
|
210
|
+
self.isa2.append((atoms[0], atoms[1], t))
|
|
211
|
+
atoms = [t] + atoms[2:]
|
|
212
|
+
if len(atoms) == 1:
|
|
213
|
+
self.isa1.append((atoms[0], b))
|
|
214
|
+
else:
|
|
215
|
+
self.isa2.append((atoms[0], atoms[1], b))
|
|
216
|
+
elif isinstance(expr, Struct) and expr.functor == "some" \
|
|
217
|
+
and len(expr.args) == 2:
|
|
218
|
+
r = self._role_name(expr.args[0])
|
|
219
|
+
self.some_isa.append((r, self._above(expr.args[1]), b))
|
|
220
|
+
else:
|
|
221
|
+
raise DatalogError("not an EL concept expression: %s" % (expr,))
|
|
222
|
+
|
|
223
|
+
def _below(self, expr):
|
|
224
|
+
"""Atomic name for expr in a right-hand position (name ⊑ expr)."""
|
|
225
|
+
if isinstance(expr, Const):
|
|
226
|
+
return self._concept_name(expr)
|
|
227
|
+
key = ("below", str(expr))
|
|
228
|
+
if key not in self._memo:
|
|
229
|
+
self._memo[key] = a = self._gen()
|
|
230
|
+
self._sub_atom_expr(a, expr)
|
|
231
|
+
return self._memo[key]
|
|
232
|
+
|
|
233
|
+
def _above(self, expr):
|
|
234
|
+
"""Atomic name for expr in a left-hand position (expr ⊑ name)."""
|
|
235
|
+
if isinstance(expr, Const):
|
|
236
|
+
return self._concept_name(expr)
|
|
237
|
+
key = ("above", str(expr))
|
|
238
|
+
if key not in self._memo:
|
|
239
|
+
self._memo[key] = a = self._gen()
|
|
240
|
+
self._sub_expr_atom(expr, a)
|
|
241
|
+
return self._memo[key]
|
|
242
|
+
|
|
243
|
+
# -- compilation to Datalog ---------------------------------------------
|
|
244
|
+
|
|
245
|
+
def datalog(self):
|
|
246
|
+
"""The completion calculus as a Datalog program (AST clauses).
|
|
247
|
+
|
|
248
|
+
Facts encode the normalised TBox; the rules are the EL⊥
|
|
249
|
+
completion rules (CR1–CR4 plus reflexivity). subs(C, D) in the
|
|
250
|
+
fixpoint means C ⊑ D."""
|
|
251
|
+
def atom(pred, *names):
|
|
252
|
+
return Atom(pred, tuple(Const(n) for n in names))
|
|
253
|
+
|
|
254
|
+
names = sorted(self.concepts) + (["bot"] if self.disjoint else [])
|
|
255
|
+
clauses = [Rule(atom("concept", c), ()) for c in names]
|
|
256
|
+
clauses += [Rule(atom("isa2", a, b, "bot"), ()) for a, b in self.disjoint]
|
|
257
|
+
clauses += [Rule(atom("isa1", a, b), ()) for a, b in self.isa1]
|
|
258
|
+
clauses += [Rule(atom("isa2", a1, a2, b), ())
|
|
259
|
+
for a1, a2, b in self.isa2]
|
|
260
|
+
clauses += [Rule(atom("isa_some", a, r, b), ())
|
|
261
|
+
for a, r, b in self.isa_some]
|
|
262
|
+
clauses += [Rule(atom("some_isa", r, a, b), ())
|
|
263
|
+
for r, a, b in self.some_isa]
|
|
264
|
+
|
|
265
|
+
C, D, D1, D2, Dp, E, R = (Var(n) for n in
|
|
266
|
+
("C", "D", "D1", "D2", "Dp", "E", "R"))
|
|
267
|
+
lit = lambda pred, *args: Literal(Atom(pred, tuple(args)))
|
|
268
|
+
clauses += [
|
|
269
|
+
# every concept subsumes itself
|
|
270
|
+
Rule(Atom("subs", (C, C)), (lit("concept", C),)),
|
|
271
|
+
# CR1: climb told inclusions
|
|
272
|
+
Rule(Atom("subs", (C, E)),
|
|
273
|
+
(lit("subs", C, D), lit("isa1", D, E))),
|
|
274
|
+
# CR2: two subsumers combine through a conjunction axiom
|
|
275
|
+
Rule(Atom("subs", (C, E)),
|
|
276
|
+
(lit("subs", C, D1), lit("subs", C, D2),
|
|
277
|
+
lit("isa2", D1, D2, E))),
|
|
278
|
+
# CR3: a subsumer with an existential creates a role link
|
|
279
|
+
Rule(Atom("link", (C, R, E)),
|
|
280
|
+
(lit("subs", C, D), lit("isa_some", D, R, E))),
|
|
281
|
+
# CR4: a role link whose target is subsumed appropriately
|
|
282
|
+
Rule(Atom("subs", (C, E)),
|
|
283
|
+
(lit("link", C, R, D), lit("subs", D, Dp),
|
|
284
|
+
lit("some_isa", R, Dp, E))),
|
|
285
|
+
# CR5/CR6 (EL⊥): ⊥ is below everything; ∃r.⊥ is ⊥
|
|
286
|
+
Rule(Atom("subs", (C, E)),
|
|
287
|
+
(lit("subs", C, Const("bot")), lit("concept", E))),
|
|
288
|
+
Rule(Atom("subs", (C, Const("bot"))),
|
|
289
|
+
(lit("link", C, R, D), lit("subs", D, Const("bot")))),
|
|
290
|
+
]
|
|
291
|
+
return clauses
|
|
292
|
+
|
|
293
|
+
def emit(self):
|
|
294
|
+
"""The compiled program as Datalog text — runnable by datalog.py."""
|
|
295
|
+
return "\n".join(str(c) for c in self.datalog()) + "\n"
|
|
296
|
+
|
|
297
|
+
# -- classification -----------------------------------------------------
|
|
298
|
+
|
|
299
|
+
def saturate(self):
|
|
300
|
+
"""CR1-CR6 as an indexed worklist saturation -- the same calculus
|
|
301
|
+
datalog() compiles, evaluated natively. Returns {concept: set of
|
|
302
|
+
subsumers, itself included}; "bot" in S(C) marks C unsatisfiable.
|
|
303
|
+
classify(fast=True) uses this; the tests hold it equal to the
|
|
304
|
+
compiled-Datalog path on every ontology they touch."""
|
|
305
|
+
concepts = set(self.concepts) | ({"bot"} if self.disjoint else set())
|
|
306
|
+
by_sub = defaultdict(list)
|
|
307
|
+
for a, b in self.isa1:
|
|
308
|
+
by_sub[a].append(b)
|
|
309
|
+
conj = defaultdict(list)
|
|
310
|
+
for a1, a2, b in (self.isa2
|
|
311
|
+
+ [(a, b, "bot") for a, b in self.disjoint]):
|
|
312
|
+
conj[a1].append((a2, b))
|
|
313
|
+
conj[a2].append((a1, b))
|
|
314
|
+
exist = defaultdict(list)
|
|
315
|
+
for a, r, b in self.isa_some:
|
|
316
|
+
exist[a].append((r, b))
|
|
317
|
+
some_isa = defaultdict(list)
|
|
318
|
+
for r, a, b in self.some_isa:
|
|
319
|
+
some_isa[(r, a)].append(b)
|
|
320
|
+
S = {c: {c} for c in concepts}
|
|
321
|
+
links = defaultdict(set)
|
|
322
|
+
incoming = defaultdict(set)
|
|
323
|
+
work = [(c, c) for c in concepts]
|
|
324
|
+
|
|
325
|
+
def add(c, d):
|
|
326
|
+
if d not in S[c]:
|
|
327
|
+
S[c].add(d)
|
|
328
|
+
work.append((c, d))
|
|
329
|
+
|
|
330
|
+
while work:
|
|
331
|
+
c, d = work.pop()
|
|
332
|
+
for e in by_sub.get(d, ()):
|
|
333
|
+
add(c, e)
|
|
334
|
+
for d2, e in conj.get(d, ()):
|
|
335
|
+
if d2 in S[c]:
|
|
336
|
+
add(c, e)
|
|
337
|
+
for r, e in exist.get(d, ()):
|
|
338
|
+
if (r, e) not in links[c]:
|
|
339
|
+
links[c].add((r, e))
|
|
340
|
+
incoming[e].add((c, r))
|
|
341
|
+
for dp in list(S[e]):
|
|
342
|
+
for f in some_isa.get((r, dp), ()):
|
|
343
|
+
add(c, f)
|
|
344
|
+
if "bot" in S[e]:
|
|
345
|
+
add(c, "bot")
|
|
346
|
+
for c2, r in incoming.get(c, ()):
|
|
347
|
+
for f in some_isa.get((r, d), ()):
|
|
348
|
+
add(c2, f)
|
|
349
|
+
if d == "bot":
|
|
350
|
+
add(c2, "bot")
|
|
351
|
+
if d == "bot":
|
|
352
|
+
for e in concepts:
|
|
353
|
+
add(c, e)
|
|
354
|
+
return S
|
|
355
|
+
|
|
356
|
+
def classify(self, fast=False):
|
|
357
|
+
"""{named concept: set of named strict subsumers}."""
|
|
358
|
+
if self._supers is None and fast:
|
|
359
|
+
S = self.saturate()
|
|
360
|
+
self._unsat = {c for c in self.named if "bot" in S.get(c, ())}
|
|
361
|
+
self._supers = {c: {d for d in S[c]
|
|
362
|
+
if d in self.named and d != c}
|
|
363
|
+
for c in self.named if c not in self._unsat}
|
|
364
|
+
for c in self._unsat:
|
|
365
|
+
self._supers[c] = set()
|
|
366
|
+
if self._supers is None:
|
|
367
|
+
engine = Engine(Program(self.datalog()))
|
|
368
|
+
engine.run()
|
|
369
|
+
self._supers = {c: set() for c in self.named}
|
|
370
|
+
self._unsat = {c for c, d in engine.rels["subs"]
|
|
371
|
+
if d == "bot" and c in self.named}
|
|
372
|
+
for sub, sup in engine.rels["subs"]:
|
|
373
|
+
if (sub in self.named and sup in self.named and sub != sup
|
|
374
|
+
and sub not in self._unsat): # ⊥ sits below all
|
|
375
|
+
self._supers[sub].add(sup)
|
|
376
|
+
return self._supers
|
|
377
|
+
|
|
378
|
+
def unsatisfiable(self):
|
|
379
|
+
"""Named concepts subsumed by ⊥ (only possible with disjoint/2)."""
|
|
380
|
+
self.classify()
|
|
381
|
+
return self._unsat
|
|
382
|
+
|
|
383
|
+
def direct_subsumers(self):
|
|
384
|
+
"""The transitive reduction of classify(): for each concept, its
|
|
385
|
+
immediate parents in the discovered hierarchy."""
|
|
386
|
+
supers = self.classify()
|
|
387
|
+
out = {}
|
|
388
|
+
for c, sup in supers.items():
|
|
389
|
+
out[c] = {d for d in sup
|
|
390
|
+
if not any(e != d and d in supers[e] and e not in supers[d]
|
|
391
|
+
for e in sup)}
|
|
392
|
+
return out
|
|
393
|
+
|
|
394
|
+
def equivalences(self):
|
|
395
|
+
supers = self.classify()
|
|
396
|
+
return sorted({tuple(sorted((c, d)))
|
|
397
|
+
for c, sup in supers.items()
|
|
398
|
+
for d in sup if c in supers[d]})
|
|
399
|
+
|
|
400
|
+
def load(text):
|
|
401
|
+
return Ontology.from_text(text)
|
|
402
|
+
|
|
403
|
+
# ---------------------------------------------------------------------------
|
|
404
|
+
# CLI
|
|
405
|
+
# ---------------------------------------------------------------------------
|
|
406
|
+
|
|
407
|
+
def main(argv=None):
|
|
408
|
+
ap = argparse.ArgumentParser(
|
|
409
|
+
description="Classify an EL ontology by compiling subsumption "
|
|
410
|
+
"to Datalog.")
|
|
411
|
+
ap.add_argument("file", help="ontology file (isa/2, define/2 facts)")
|
|
412
|
+
ap.add_argument("-q", "--query", action="append", default=[],
|
|
413
|
+
metavar="CONCEPT",
|
|
414
|
+
help="print all subsumers of one concept (repeatable)")
|
|
415
|
+
ap.add_argument("--emit", action="store_true",
|
|
416
|
+
help="print the compiled Datalog program and exit")
|
|
417
|
+
ap.add_argument("--fast", action="store_true",
|
|
418
|
+
help="classify by native saturation instead of the "
|
|
419
|
+
"compiled Datalog program (same result; the "
|
|
420
|
+
"tests hold the two equal)")
|
|
421
|
+
args = ap.parse_args(argv)
|
|
422
|
+
|
|
423
|
+
try:
|
|
424
|
+
ont = load(read_program(args.file))
|
|
425
|
+
if args.emit:
|
|
426
|
+
sys.stdout.write(ont.emit())
|
|
427
|
+
return 0
|
|
428
|
+
t0 = time.perf_counter()
|
|
429
|
+
supers = ont.classify(fast=args.fast)
|
|
430
|
+
elapsed = time.perf_counter() - t0
|
|
431
|
+
print("(classified in %.3fs via %s)" % (
|
|
432
|
+
elapsed, "native saturation" if args.fast
|
|
433
|
+
else "the compiled Datalog program"))
|
|
434
|
+
direct = ont.direct_subsumers()
|
|
435
|
+
except DatalogError as exc:
|
|
436
|
+
print("error: %s" % exc, file=sys.stderr)
|
|
437
|
+
return 1
|
|
438
|
+
|
|
439
|
+
if args.query:
|
|
440
|
+
for c in args.query:
|
|
441
|
+
if c not in supers:
|
|
442
|
+
print("error: unknown concept %r" % c, file=sys.stderr)
|
|
443
|
+
return 1
|
|
444
|
+
names = (["⊥ (unsatisfiable)"] if c in ont.unsatisfiable()
|
|
445
|
+
else sorted(supers[c]) or ["(none)"])
|
|
446
|
+
print("%s ⊑ %s" % (c, ", ".join(names)))
|
|
447
|
+
return 0
|
|
448
|
+
|
|
449
|
+
told = ont.told
|
|
450
|
+
inferred = 0
|
|
451
|
+
print("Classification (%d named concepts):" % len(supers))
|
|
452
|
+
for c in sorted(direct):
|
|
453
|
+
if c in ont.unsatisfiable():
|
|
454
|
+
print(" %-14s ⊑ ⊥ (unsatisfiable)" % c)
|
|
455
|
+
continue
|
|
456
|
+
if not direct[c]:
|
|
457
|
+
print(" %-14s (top of hierarchy)" % c)
|
|
458
|
+
continue
|
|
459
|
+
parts = []
|
|
460
|
+
for d in sorted(direct[c]):
|
|
461
|
+
if (c, d) in told:
|
|
462
|
+
parts.append(d)
|
|
463
|
+
else:
|
|
464
|
+
parts.append(d + "*")
|
|
465
|
+
inferred += 1
|
|
466
|
+
print(" %-14s ⊑ %s" % (c, ", ".join(parts)))
|
|
467
|
+
for c, d in ont.equivalences():
|
|
468
|
+
print(" %-14s ≡ %s" % (c, d))
|
|
469
|
+
if inferred:
|
|
470
|
+
print(" (* = inferred by the classifier, not stated)")
|
|
471
|
+
return 0
|
|
472
|
+
|
|
473
|
+
|
|
474
|
+
if __name__ == "__main__":
|
|
475
|
+
sys.exit(main())
|
tiny_datalog/tabling.py
ADDED
|
@@ -0,0 +1,220 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
tabling.py — tabled top-down evaluation: the third vertex of the
|
|
4
|
+
evaluation triangle. (Lesson 15, which ends with a tour of this module.)
|
|
5
|
+
|
|
6
|
+
The course has shown three ways to answer a Datalog query:
|
|
7
|
+
|
|
8
|
+
* bottom-up semi-naive (datalog.py) — compute everything, look up;
|
|
9
|
+
* magic sets (magic.py) — compile the query's demand into the program,
|
|
10
|
+
then run bottom-up;
|
|
11
|
+
* SLD resolution (prolog.py) — chase the goal top-down, tuple at a time,
|
|
12
|
+
repeating subgoals and looping on left recursion.
|
|
13
|
+
|
|
14
|
+
Tabling is top-down done right: every subgoal (a predicate with a
|
|
15
|
+
pattern of bound arguments — an adornment *with its values*) gets a
|
|
16
|
+
**table** of answers, filled once and shared by every occurrence. A
|
|
17
|
+
subgoal that reaches itself — left recursion, which sends Prolog into
|
|
18
|
+
the abyss — simply reads its own table and grows it to fixpoint.
|
|
19
|
+
|
|
20
|
+
The implementation is the iterative QSQR formulation, chosen because it
|
|
21
|
+
is honest and small: each round re-solves every tabled subgoal
|
|
22
|
+
top-down, with recursive calls reading current table contents; new
|
|
23
|
+
subgoals encountered get empty tables; repeat until no table grows.
|
|
24
|
+
Termination is Datalog's usual gift — finitely many call patterns,
|
|
25
|
+
finitely many answers.
|
|
26
|
+
|
|
27
|
+
The punchline to check for yourself: run a bound query here and under
|
|
28
|
+
`datalog.py --magic --trace`, and compare this module's *tables* with
|
|
29
|
+
the magic predicates' contents. They are the same sets — magic sets is
|
|
30
|
+
tabling performed at compile time, tabling is magic sets performed at
|
|
31
|
+
run time.
|
|
32
|
+
|
|
33
|
+
Positive programs only (tabling under negation is SLG resolution, which
|
|
34
|
+
computes the well-founded semantics — XSB's whole claim to fame — and
|
|
35
|
+
is beyond this teaching module).
|
|
36
|
+
|
|
37
|
+
CLI
|
|
38
|
+
---
|
|
39
|
+
python3 tabling.py programs/reachability.dl -q 'path(n5, X)'
|
|
40
|
+
python3 tabling.py programs/left-recursive.dl -q 'ancestor(abe, X)'
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
from __future__ import annotations
|
|
44
|
+
|
|
45
|
+
import argparse
|
|
46
|
+
import sys
|
|
47
|
+
from collections import defaultdict
|
|
48
|
+
|
|
49
|
+
from tiny_datalog.datalog import (
|
|
50
|
+
check_query_atom, Const, DatalogError, format_fact, parse, parse_goal,
|
|
51
|
+
read_program, validate, _aggregate_of, _match, _sort_key)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class TabledEngine:
|
|
55
|
+
"""Iterative QSQR: tables keyed by call pattern, filled to fixpoint.
|
|
56
|
+
|
|
57
|
+
After query(), `tables` maps (pred, pattern) — pattern has a constant
|
|
58
|
+
per bound argument and None per free one — to the set of full answer
|
|
59
|
+
tuples for that subgoal, and `rounds` records the outer iterations."""
|
|
60
|
+
|
|
61
|
+
def __init__(self, clauses):
|
|
62
|
+
self.arity = validate(clauses)
|
|
63
|
+
for c in clauses:
|
|
64
|
+
if c.retract:
|
|
65
|
+
raise DatalogError("retraction is incremental.py's job: %s" % c)
|
|
66
|
+
if c.body and _aggregate_of(c.head):
|
|
67
|
+
raise DatalogError(
|
|
68
|
+
"tabled aggregation needs completion detection (real "
|
|
69
|
+
"SLG); this module is positive-rules-only: %s" % c)
|
|
70
|
+
for lit in c.body:
|
|
71
|
+
if lit.negated:
|
|
72
|
+
raise DatalogError(
|
|
73
|
+
"tabling under negation is SLG resolution (the "
|
|
74
|
+
"well-founded semantics, XSB) — beyond this "
|
|
75
|
+
"module: %s" % c)
|
|
76
|
+
self.by_pred = defaultdict(list) # (pred, arity) -> clauses
|
|
77
|
+
for c in clauses:
|
|
78
|
+
self.by_pred[(c.head.pred, len(c.head.args))].append(c)
|
|
79
|
+
self.tables = {}
|
|
80
|
+
self.rounds = 0
|
|
81
|
+
|
|
82
|
+
# -- call patterns ------------------------------------------------------
|
|
83
|
+
|
|
84
|
+
@staticmethod
|
|
85
|
+
def _pattern(atom, subst):
|
|
86
|
+
"""The call pattern of an atom under a substitution: the bound
|
|
87
|
+
arguments' values, None where still free."""
|
|
88
|
+
out = []
|
|
89
|
+
for a in atom.args:
|
|
90
|
+
if isinstance(a, Const):
|
|
91
|
+
out.append(a.value)
|
|
92
|
+
elif a.name in subst:
|
|
93
|
+
out.append(subst[a.name])
|
|
94
|
+
else:
|
|
95
|
+
out.append(None)
|
|
96
|
+
return tuple(out)
|
|
97
|
+
|
|
98
|
+
def _table(self, pred, pattern):
|
|
99
|
+
key = (pred, pattern)
|
|
100
|
+
if key not in self.tables:
|
|
101
|
+
self.tables[key] = set() # discovered a new subgoal
|
|
102
|
+
self._grew = True # ...which the fixpoint must revisit
|
|
103
|
+
return self.tables[key]
|
|
104
|
+
|
|
105
|
+
# -- one round of top-down solving --------------------------------------
|
|
106
|
+
|
|
107
|
+
def _prove(self, body, subst):
|
|
108
|
+
"""Solve a rule body left to right; recursive calls read the
|
|
109
|
+
current tables (never the clauses directly — that is the whole
|
|
110
|
+
trick: no infinite descent, the fixpoint loop supplies growth)."""
|
|
111
|
+
if not body:
|
|
112
|
+
yield subst
|
|
113
|
+
return
|
|
114
|
+
lit, rest = body[0], body[1:]
|
|
115
|
+
table = self._table(lit.atom.pred, self._pattern(lit.atom, subst))
|
|
116
|
+
for ans in table:
|
|
117
|
+
s = _match(lit.atom.args, ans, subst)
|
|
118
|
+
if s is not None:
|
|
119
|
+
yield from self._prove(rest, s)
|
|
120
|
+
|
|
121
|
+
def _answers_for(self, key):
|
|
122
|
+
"""Re-derive a subgoal's answers from its clauses, one step of
|
|
123
|
+
head unification plus a tabled body proof."""
|
|
124
|
+
pred, pattern = key
|
|
125
|
+
for clause in self.by_pred.get((pred, len(pattern)), ()):
|
|
126
|
+
# unify the head with the call pattern (bound args only)
|
|
127
|
+
seed = {}
|
|
128
|
+
ok = True
|
|
129
|
+
for a, v in zip(clause.head.args, pattern):
|
|
130
|
+
if v is None:
|
|
131
|
+
continue
|
|
132
|
+
if isinstance(a, Const):
|
|
133
|
+
if a.value != v:
|
|
134
|
+
ok = False
|
|
135
|
+
break
|
|
136
|
+
elif a.name in seed and seed[a.name] != v:
|
|
137
|
+
ok = False
|
|
138
|
+
break
|
|
139
|
+
else:
|
|
140
|
+
seed[a.name] = v
|
|
141
|
+
if not ok:
|
|
142
|
+
continue
|
|
143
|
+
for s in self._prove(list(clause.body), seed):
|
|
144
|
+
yield tuple(a.value if isinstance(a, Const) else s[a.name]
|
|
145
|
+
for a in clause.head.args)
|
|
146
|
+
|
|
147
|
+
# -- the fixpoint --------------------------------------------------------
|
|
148
|
+
|
|
149
|
+
def query(self, atom):
|
|
150
|
+
"""All answers to `atom`, computed by tabled top-down evaluation.
|
|
151
|
+
Also populates self.tables / self.rounds for inspection."""
|
|
152
|
+
check_query_atom(atom, self.arity)
|
|
153
|
+
root = (atom.pred, self._pattern(atom, {}))
|
|
154
|
+
self.tables = {root: set()}
|
|
155
|
+
self.rounds = 0
|
|
156
|
+
changed = True
|
|
157
|
+
while changed:
|
|
158
|
+
changed = False
|
|
159
|
+
self._grew = False
|
|
160
|
+
self.rounds += 1
|
|
161
|
+
for key in list(self.tables):
|
|
162
|
+
table = self.tables[key]
|
|
163
|
+
for ans in list(self._answers_for(key)):
|
|
164
|
+
if ans not in table:
|
|
165
|
+
table.add(ans)
|
|
166
|
+
changed = True
|
|
167
|
+
changed = changed or self._grew
|
|
168
|
+
return {t for t in self.tables[root]
|
|
169
|
+
if _match(atom.args, t, {}) is not None}
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
# ---------------------------------------------------------------------------
|
|
173
|
+
# CLI
|
|
174
|
+
# ---------------------------------------------------------------------------
|
|
175
|
+
|
|
176
|
+
def main(argv=None):
|
|
177
|
+
ap = argparse.ArgumentParser(
|
|
178
|
+
description="Tabled top-down (QSQR) evaluation of positive "
|
|
179
|
+
"Datalog — handles left recursion SLD cannot.")
|
|
180
|
+
ap.add_argument("file", help="Datalog program (.dl), positive rules only")
|
|
181
|
+
ap.add_argument("-q", "--query", action="append", default=[],
|
|
182
|
+
metavar="ATOM", help="goal to solve (repeatable)")
|
|
183
|
+
ap.add_argument("-t", "--tables", action="store_true",
|
|
184
|
+
help="also print every table (compare these with the "
|
|
185
|
+
"magic predicates from datalog.py --magic!)")
|
|
186
|
+
args = ap.parse_args(argv)
|
|
187
|
+
|
|
188
|
+
try:
|
|
189
|
+
engine = TabledEngine(parse(read_program(args.file)))
|
|
190
|
+
except DatalogError as exc:
|
|
191
|
+
print("error: %s" % exc, file=sys.stderr)
|
|
192
|
+
return 1
|
|
193
|
+
for q in args.query:
|
|
194
|
+
try:
|
|
195
|
+
atom = parse_goal(q)
|
|
196
|
+
answers = engine.query(atom)
|
|
197
|
+
except DatalogError as exc:
|
|
198
|
+
print("error: %s" % exc, file=sys.stderr)
|
|
199
|
+
return 1
|
|
200
|
+
print("?- %s [tabled]" % atom)
|
|
201
|
+
for tup in sorted(answers, key=_sort_key):
|
|
202
|
+
print(" " + format_fact(atom.pred, tup))
|
|
203
|
+
print(" (%d answer%s; %d subgoal table%s, %d rounds)"
|
|
204
|
+
% (len(answers), "" if len(answers) == 1 else "s",
|
|
205
|
+
len(engine.tables), "" if len(engine.tables) == 1 else "s",
|
|
206
|
+
engine.rounds))
|
|
207
|
+
if args.tables:
|
|
208
|
+
for (pred, pattern), tbl in sorted(
|
|
209
|
+
engine.tables.items(),
|
|
210
|
+
key=lambda kv: (kv[0][0], str(kv[0][1]))):
|
|
211
|
+
shown = ", ".join("_" if v is None else str(v)
|
|
212
|
+
for v in pattern)
|
|
213
|
+
print(" table %s(%s): %d answer%s"
|
|
214
|
+
% (pred, shown, len(tbl),
|
|
215
|
+
"" if len(tbl) == 1 else "s"))
|
|
216
|
+
return 0
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
if __name__ == "__main__":
|
|
220
|
+
sys.exit(main())
|