tiny-datalog 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tiny_datalog/__init__.py +42 -0
- tiny_datalog/containment.py +228 -0
- tiny_datalog/datalog.py +1455 -0
- tiny_datalog/incremental.py +424 -0
- tiny_datalog/magic.py +175 -0
- tiny_datalog/prolog.py +335 -0
- tiny_datalog/semantics.py +182 -0
- tiny_datalog/semiring.py +383 -0
- tiny_datalog/subsumption.py +475 -0
- tiny_datalog/tabling.py +220 -0
- tiny_datalog-0.1.0.dist-info/METADATA +360 -0
- tiny_datalog-0.1.0.dist-info/RECORD +16 -0
- tiny_datalog-0.1.0.dist-info/WHEEL +5 -0
- tiny_datalog-0.1.0.dist-info/entry_points.txt +8 -0
- tiny_datalog-0.1.0.dist-info/licenses/LICENSE +21 -0
- tiny_datalog-0.1.0.dist-info/top_level.txt +1 -0
tiny_datalog/__init__.py
ADDED
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""tiny-datalog — a logic engine small enough to read in an afternoon.
|
|
2
|
+
|
|
3
|
+
The engine proper is `tiny_datalog.datalog`; the names below are its
|
|
4
|
+
public surface, re-exported here so that library use reads as
|
|
5
|
+
|
|
6
|
+
from tiny_datalog import Engine, Program, parse
|
|
7
|
+
|
|
8
|
+
Everything else in the package is a satellite built on top of it:
|
|
9
|
+
`magic` (goal-directed rewriting), `semantics` (stable and well-founded
|
|
10
|
+
models), `semiring` (provenance-weighted evaluation), `incremental`
|
|
11
|
+
(maintenance under updates), `tabling` (top-down with memoing),
|
|
12
|
+
`subsumption` (KL-ONE style classification), `containment` (query
|
|
13
|
+
containment), and `prolog` (a Prolog reader for the same syntax).
|
|
14
|
+
Import those by name: `from tiny_datalog import semiring`.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from tiny_datalog.datalog import (
|
|
18
|
+
# AST
|
|
19
|
+
Var, Const, Struct, Atom, Literal, Rule,
|
|
20
|
+
# errors
|
|
21
|
+
DatalogError, ParseError, SafetyError, StratificationError,
|
|
22
|
+
# front end
|
|
23
|
+
parse, validate, stratify, parse_goal, check_query_atom,
|
|
24
|
+
# evaluation
|
|
25
|
+
Program, Engine, run_program,
|
|
26
|
+
# answers and justification
|
|
27
|
+
match_answers, explain, whynot,
|
|
28
|
+
# formatting
|
|
29
|
+
format_atom, format_fact,
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
__version__ = "0.1.0"
|
|
33
|
+
|
|
34
|
+
__all__ = [
|
|
35
|
+
"Var", "Const", "Struct", "Atom", "Literal", "Rule",
|
|
36
|
+
"DatalogError", "ParseError", "SafetyError", "StratificationError",
|
|
37
|
+
"parse", "validate", "stratify", "parse_goal", "check_query_atom",
|
|
38
|
+
"Program", "Engine", "run_program",
|
|
39
|
+
"match_answers", "explain", "whynot",
|
|
40
|
+
"format_atom", "format_fact",
|
|
41
|
+
"__version__",
|
|
42
|
+
]
|
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
containment.py — query containment and minimisation for conjunctive
|
|
4
|
+
queries, by homomorphism. (Lesson 16.)
|
|
5
|
+
|
|
6
|
+
Two questions an optimiser asks that evaluation never does:
|
|
7
|
+
|
|
8
|
+
* **Containment.** Does Q1 return a subset of Q2's answers, on *every*
|
|
9
|
+
database? Not "on this data" — on all data.
|
|
10
|
+
* **Minimisation.** Does Q have redundant atoms — a smaller body giving
|
|
11
|
+
identical answers everywhere?
|
|
12
|
+
|
|
13
|
+
Chandra and Merlin (1977) answered both with one idea: for conjunctive
|
|
14
|
+
queries, **Q2 ⊇ Q1 iff there is a homomorphism from Q2's body into Q1's
|
|
15
|
+
body** that fixes the head variables. Containment, a statement about
|
|
16
|
+
infinitely many databases, becomes a finite search for a variable
|
|
17
|
+
mapping.
|
|
18
|
+
|
|
19
|
+
The engine already contains most of this. `datalog._match` maps a rule
|
|
20
|
+
body into the *database* — a set of ground atoms. A homomorphism here
|
|
21
|
+
maps a rule body into *another rule body*, whose variables act like
|
|
22
|
+
fresh constants ("freezing" the query into a canonical database). Same
|
|
23
|
+
search, one level up the abstraction; that observation is the whole
|
|
24
|
+
lesson, and it is why this module is short.
|
|
25
|
+
|
|
26
|
+
The price is complexity: homomorphism-finding is NP-complete (it is
|
|
27
|
+
graph colouring in disguise), so containment is too. Classic queries
|
|
28
|
+
are small, and the backtracking search below is the standard practical
|
|
29
|
+
answer.
|
|
30
|
+
|
|
31
|
+
CLI
|
|
32
|
+
---
|
|
33
|
+
python3 containment.py programs/minimise.dl
|
|
34
|
+
python3 containment.py --contains 'q(X) :- e(X, Y), e(Y, Z).' \\
|
|
35
|
+
'q(X) :- e(X, Y).'
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
from __future__ import annotations
|
|
39
|
+
|
|
40
|
+
import argparse
|
|
41
|
+
import sys
|
|
42
|
+
|
|
43
|
+
from tiny_datalog.datalog import (
|
|
44
|
+
_aggregate_of, Const, DatalogError, parse, read_program, validate, Var)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _extend(mapping, source_args, target_args):
|
|
48
|
+
"""Extend a variable mapping so source_args maps onto target_args."""
|
|
49
|
+
out = dict(mapping)
|
|
50
|
+
for s, t in zip(source_args, target_args):
|
|
51
|
+
if isinstance(s, Const):
|
|
52
|
+
if s != t:
|
|
53
|
+
return None
|
|
54
|
+
elif s.name in out:
|
|
55
|
+
if out[s.name] != t:
|
|
56
|
+
return None
|
|
57
|
+
else:
|
|
58
|
+
out[s.name] = t
|
|
59
|
+
return out
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def find_homomorphism(source, target, seed=None):
|
|
63
|
+
"""A mapping of source's variables to target's terms such that every
|
|
64
|
+
source atom lands on some target atom, or None.
|
|
65
|
+
|
|
66
|
+
Backtracking search: try every target atom for the first source
|
|
67
|
+
atom, recurse. This is exactly `_match`'s job with a target of
|
|
68
|
+
non-ground atoms instead of tuples."""
|
|
69
|
+
mapping = dict(seed or {})
|
|
70
|
+
|
|
71
|
+
def search(i, current):
|
|
72
|
+
if i == len(source):
|
|
73
|
+
return current
|
|
74
|
+
atom = source[i]
|
|
75
|
+
for candidate in target:
|
|
76
|
+
if (candidate.pred != atom.pred
|
|
77
|
+
or len(candidate.args) != len(atom.args)):
|
|
78
|
+
continue
|
|
79
|
+
extended = _extend(current, atom.args, candidate.args)
|
|
80
|
+
if extended is not None:
|
|
81
|
+
found = search(i + 1, extended)
|
|
82
|
+
if found is not None:
|
|
83
|
+
return found
|
|
84
|
+
return None
|
|
85
|
+
|
|
86
|
+
return search(0, mapping)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _bodies(rule):
|
|
90
|
+
for lit in rule.body:
|
|
91
|
+
if lit.negated:
|
|
92
|
+
raise DatalogError(
|
|
93
|
+
"containment by homomorphism is a conjunctive-query "
|
|
94
|
+
"result; negation needs different theory: %s" % rule)
|
|
95
|
+
if _aggregate_of(rule.head):
|
|
96
|
+
# sum/count see duplicates; Chandra–Merlin is about sets.
|
|
97
|
+
raise DatalogError(
|
|
98
|
+
"containment by homomorphism is a set-semantics result; "
|
|
99
|
+
"aggregation needs different theory: %s" % rule)
|
|
100
|
+
return [lit.atom for lit in rule.body]
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _head_seed(outer, inner):
|
|
104
|
+
"""Head variables correspond positionally and must be preserved."""
|
|
105
|
+
if len(outer.head.args) != len(inner.head.args):
|
|
106
|
+
return None
|
|
107
|
+
seed = {}
|
|
108
|
+
for a, b in zip(outer.head.args, inner.head.args):
|
|
109
|
+
if isinstance(a, Var):
|
|
110
|
+
if a.name in seed and seed[a.name] != b:
|
|
111
|
+
return None
|
|
112
|
+
seed[a.name] = b
|
|
113
|
+
elif a != b:
|
|
114
|
+
return None
|
|
115
|
+
return seed
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def contains(outer, inner):
|
|
119
|
+
"""True if `outer` ⊇ `inner`: on every database, every answer to
|
|
120
|
+
inner is an answer to outer. Decided by a homomorphism from outer's
|
|
121
|
+
body into inner's (Chandra–Merlin)."""
|
|
122
|
+
source, target = _bodies(outer), _bodies(inner)
|
|
123
|
+
seed = _head_seed(outer, inner)
|
|
124
|
+
if seed is None:
|
|
125
|
+
return False
|
|
126
|
+
return find_homomorphism(source, target, seed) is not None
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def equivalent(a, b):
|
|
130
|
+
return contains(a, b) and contains(b, a)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def minimise(rule):
|
|
134
|
+
"""The smallest equivalent body: repeatedly drop an atom whose loss
|
|
135
|
+
a homomorphism can repair. For conjunctive queries this greedy
|
|
136
|
+
procedure is safe — the minimal form is unique up to renaming, so
|
|
137
|
+
the order atoms are tried in cannot change the result."""
|
|
138
|
+
atoms = _bodies(rule)
|
|
139
|
+
head_vars = {a.name for a in rule.head.args if isinstance(a, Var)}
|
|
140
|
+
changed = True
|
|
141
|
+
while changed:
|
|
142
|
+
changed = False
|
|
143
|
+
for i in range(len(atoms)):
|
|
144
|
+
reduced = atoms[:i] + atoms[i + 1:]
|
|
145
|
+
if not reduced:
|
|
146
|
+
continue
|
|
147
|
+
seed = {v: Var(v) for v in head_vars}
|
|
148
|
+
if find_homomorphism(atoms, reduced, seed) is not None:
|
|
149
|
+
atoms = reduced
|
|
150
|
+
changed = True
|
|
151
|
+
break
|
|
152
|
+
return atoms
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _fmt(head, atoms):
|
|
156
|
+
return "%s :- %s." % (head, ", ".join(str(a) for a in atoms))
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _parse_query_rule(text):
|
|
160
|
+
clauses = parse(text if text.rstrip().endswith(".") else text + ".")
|
|
161
|
+
if len(clauses) != 1 or not clauses[0].body:
|
|
162
|
+
raise DatalogError("expected a single rule: %r" % text)
|
|
163
|
+
validate(clauses)
|
|
164
|
+
return clauses[0]
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
# ---------------------------------------------------------------------------
|
|
168
|
+
# CLI
|
|
169
|
+
# ---------------------------------------------------------------------------
|
|
170
|
+
|
|
171
|
+
def main(argv=None):
|
|
172
|
+
ap = argparse.ArgumentParser(
|
|
173
|
+
description="Conjunctive-query containment and minimisation by "
|
|
174
|
+
"homomorphism (Chandra–Merlin).")
|
|
175
|
+
ap.add_argument("file", nargs="?",
|
|
176
|
+
help="program whose rules should be minimised")
|
|
177
|
+
ap.add_argument("--contains", nargs=2, metavar=("OUTER", "INNER"),
|
|
178
|
+
help="test whether OUTER contains INNER")
|
|
179
|
+
args = ap.parse_args(argv)
|
|
180
|
+
|
|
181
|
+
try:
|
|
182
|
+
if args.contains:
|
|
183
|
+
outer = _parse_query_rule(args.contains[0])
|
|
184
|
+
inner = _parse_query_rule(args.contains[1])
|
|
185
|
+
fwd, bwd = contains(outer, inner), contains(inner, outer)
|
|
186
|
+
print("outer: %s" % outer)
|
|
187
|
+
print("inner: %s" % inner)
|
|
188
|
+
if fwd and bwd:
|
|
189
|
+
print("=> equivalent (each contains the other)")
|
|
190
|
+
elif fwd:
|
|
191
|
+
print("=> outer contains inner, on every database")
|
|
192
|
+
elif bwd:
|
|
193
|
+
print("=> inner contains outer, on every database")
|
|
194
|
+
else:
|
|
195
|
+
print("=> neither contains the other")
|
|
196
|
+
return 0
|
|
197
|
+
|
|
198
|
+
if not args.file:
|
|
199
|
+
ap.error("give a program to minimise, or use --contains")
|
|
200
|
+
clauses = parse(read_program(args.file))
|
|
201
|
+
validate(clauses)
|
|
202
|
+
for rule in clauses:
|
|
203
|
+
if not rule.body:
|
|
204
|
+
continue
|
|
205
|
+
if any(lit.negated for lit in rule.body):
|
|
206
|
+
print("%s\n (skipped: negation is outside the theory)"
|
|
207
|
+
% rule)
|
|
208
|
+
continue
|
|
209
|
+
if _aggregate_of(rule.head):
|
|
210
|
+
print("%s\n (skipped: aggregation is outside the theory)"
|
|
211
|
+
% rule)
|
|
212
|
+
continue
|
|
213
|
+
atoms = minimise(rule)
|
|
214
|
+
before = len(rule.body)
|
|
215
|
+
if len(atoms) == before:
|
|
216
|
+
print("%s\n already minimal (%d atom%s)"
|
|
217
|
+
% (rule, before, "" if before == 1 else "s"))
|
|
218
|
+
else:
|
|
219
|
+
print("%s\n minimises to %s (%d atoms -> %d)"
|
|
220
|
+
% (rule, _fmt(rule.head, atoms), before, len(atoms)))
|
|
221
|
+
except DatalogError as exc:
|
|
222
|
+
print("error: %s" % exc, file=sys.stderr)
|
|
223
|
+
return 1
|
|
224
|
+
return 0
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
if __name__ == "__main__":
|
|
228
|
+
sys.exit(main())
|