tiny-datalog 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,42 @@
1
+ """tiny-datalog — a logic engine small enough to read in an afternoon.
2
+
3
+ The engine proper is `tiny_datalog.datalog`; the names below are its
4
+ public surface, re-exported here so that library use reads as
5
+
6
+ from tiny_datalog import Engine, Program, parse
7
+
8
+ Everything else in the package is a satellite built on top of it:
9
+ `magic` (goal-directed rewriting), `semantics` (stable and well-founded
10
+ models), `semiring` (provenance-weighted evaluation), `incremental`
11
+ (maintenance under updates), `tabling` (top-down with memoing),
12
+ `subsumption` (KL-ONE style classification), `containment` (query
13
+ containment), and `prolog` (a Prolog reader for the same syntax).
14
+ Import those by name: `from tiny_datalog import semiring`.
15
+ """
16
+
17
+ from tiny_datalog.datalog import (
18
+ # AST
19
+ Var, Const, Struct, Atom, Literal, Rule,
20
+ # errors
21
+ DatalogError, ParseError, SafetyError, StratificationError,
22
+ # front end
23
+ parse, validate, stratify, parse_goal, check_query_atom,
24
+ # evaluation
25
+ Program, Engine, run_program,
26
+ # answers and justification
27
+ match_answers, explain, whynot,
28
+ # formatting
29
+ format_atom, format_fact,
30
+ )
31
+
32
+ __version__ = "0.1.0"
33
+
34
+ __all__ = [
35
+ "Var", "Const", "Struct", "Atom", "Literal", "Rule",
36
+ "DatalogError", "ParseError", "SafetyError", "StratificationError",
37
+ "parse", "validate", "stratify", "parse_goal", "check_query_atom",
38
+ "Program", "Engine", "run_program",
39
+ "match_answers", "explain", "whynot",
40
+ "format_atom", "format_fact",
41
+ "__version__",
42
+ ]
@@ -0,0 +1,228 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ containment.py — query containment and minimisation for conjunctive
4
+ queries, by homomorphism. (Lesson 16.)
5
+
6
+ Two questions an optimiser asks that evaluation never does:
7
+
8
+ * **Containment.** Does Q1 return a subset of Q2's answers, on *every*
9
+ database? Not "on this data" — on all data.
10
+ * **Minimisation.** Does Q have redundant atoms — a smaller body giving
11
+ identical answers everywhere?
12
+
13
+ Chandra and Merlin (1977) answered both with one idea: for conjunctive
14
+ queries, **Q2 ⊇ Q1 iff there is a homomorphism from Q2's body into Q1's
15
+ body** that fixes the head variables. Containment, a statement about
16
+ infinitely many databases, becomes a finite search for a variable
17
+ mapping.
18
+
19
+ The engine already contains most of this. `datalog._match` maps a rule
20
+ body into the *database* — a set of ground atoms. A homomorphism here
21
+ maps a rule body into *another rule body*, whose variables act like
22
+ fresh constants ("freezing" the query into a canonical database). Same
23
+ search, one level up the abstraction; that observation is the whole
24
+ lesson, and it is why this module is short.
25
+
26
+ The price is complexity: homomorphism-finding is NP-complete (it is
27
+ graph colouring in disguise), so containment is too. Classic queries
28
+ are small, and the backtracking search below is the standard practical
29
+ answer.
30
+
31
+ CLI
32
+ ---
33
+ python3 containment.py programs/minimise.dl
34
+ python3 containment.py --contains 'q(X) :- e(X, Y), e(Y, Z).' \\
35
+ 'q(X) :- e(X, Y).'
36
+ """
37
+
38
+ from __future__ import annotations
39
+
40
+ import argparse
41
+ import sys
42
+
43
+ from tiny_datalog.datalog import (
44
+ _aggregate_of, Const, DatalogError, parse, read_program, validate, Var)
45
+
46
+
47
+ def _extend(mapping, source_args, target_args):
48
+ """Extend a variable mapping so source_args maps onto target_args."""
49
+ out = dict(mapping)
50
+ for s, t in zip(source_args, target_args):
51
+ if isinstance(s, Const):
52
+ if s != t:
53
+ return None
54
+ elif s.name in out:
55
+ if out[s.name] != t:
56
+ return None
57
+ else:
58
+ out[s.name] = t
59
+ return out
60
+
61
+
62
+ def find_homomorphism(source, target, seed=None):
63
+ """A mapping of source's variables to target's terms such that every
64
+ source atom lands on some target atom, or None.
65
+
66
+ Backtracking search: try every target atom for the first source
67
+ atom, recurse. This is exactly `_match`'s job with a target of
68
+ non-ground atoms instead of tuples."""
69
+ mapping = dict(seed or {})
70
+
71
+ def search(i, current):
72
+ if i == len(source):
73
+ return current
74
+ atom = source[i]
75
+ for candidate in target:
76
+ if (candidate.pred != atom.pred
77
+ or len(candidate.args) != len(atom.args)):
78
+ continue
79
+ extended = _extend(current, atom.args, candidate.args)
80
+ if extended is not None:
81
+ found = search(i + 1, extended)
82
+ if found is not None:
83
+ return found
84
+ return None
85
+
86
+ return search(0, mapping)
87
+
88
+
89
+ def _bodies(rule):
90
+ for lit in rule.body:
91
+ if lit.negated:
92
+ raise DatalogError(
93
+ "containment by homomorphism is a conjunctive-query "
94
+ "result; negation needs different theory: %s" % rule)
95
+ if _aggregate_of(rule.head):
96
+ # sum/count see duplicates; Chandra–Merlin is about sets.
97
+ raise DatalogError(
98
+ "containment by homomorphism is a set-semantics result; "
99
+ "aggregation needs different theory: %s" % rule)
100
+ return [lit.atom for lit in rule.body]
101
+
102
+
103
+ def _head_seed(outer, inner):
104
+ """Head variables correspond positionally and must be preserved."""
105
+ if len(outer.head.args) != len(inner.head.args):
106
+ return None
107
+ seed = {}
108
+ for a, b in zip(outer.head.args, inner.head.args):
109
+ if isinstance(a, Var):
110
+ if a.name in seed and seed[a.name] != b:
111
+ return None
112
+ seed[a.name] = b
113
+ elif a != b:
114
+ return None
115
+ return seed
116
+
117
+
118
+ def contains(outer, inner):
119
+ """True if `outer` ⊇ `inner`: on every database, every answer to
120
+ inner is an answer to outer. Decided by a homomorphism from outer's
121
+ body into inner's (Chandra–Merlin)."""
122
+ source, target = _bodies(outer), _bodies(inner)
123
+ seed = _head_seed(outer, inner)
124
+ if seed is None:
125
+ return False
126
+ return find_homomorphism(source, target, seed) is not None
127
+
128
+
129
+ def equivalent(a, b):
130
+ return contains(a, b) and contains(b, a)
131
+
132
+
133
+ def minimise(rule):
134
+ """The smallest equivalent body: repeatedly drop an atom whose loss
135
+ a homomorphism can repair. For conjunctive queries this greedy
136
+ procedure is safe — the minimal form is unique up to renaming, so
137
+ the order atoms are tried in cannot change the result."""
138
+ atoms = _bodies(rule)
139
+ head_vars = {a.name for a in rule.head.args if isinstance(a, Var)}
140
+ changed = True
141
+ while changed:
142
+ changed = False
143
+ for i in range(len(atoms)):
144
+ reduced = atoms[:i] + atoms[i + 1:]
145
+ if not reduced:
146
+ continue
147
+ seed = {v: Var(v) for v in head_vars}
148
+ if find_homomorphism(atoms, reduced, seed) is not None:
149
+ atoms = reduced
150
+ changed = True
151
+ break
152
+ return atoms
153
+
154
+
155
+ def _fmt(head, atoms):
156
+ return "%s :- %s." % (head, ", ".join(str(a) for a in atoms))
157
+
158
+
159
+ def _parse_query_rule(text):
160
+ clauses = parse(text if text.rstrip().endswith(".") else text + ".")
161
+ if len(clauses) != 1 or not clauses[0].body:
162
+ raise DatalogError("expected a single rule: %r" % text)
163
+ validate(clauses)
164
+ return clauses[0]
165
+
166
+
167
+ # ---------------------------------------------------------------------------
168
+ # CLI
169
+ # ---------------------------------------------------------------------------
170
+
171
+ def main(argv=None):
172
+ ap = argparse.ArgumentParser(
173
+ description="Conjunctive-query containment and minimisation by "
174
+ "homomorphism (Chandra–Merlin).")
175
+ ap.add_argument("file", nargs="?",
176
+ help="program whose rules should be minimised")
177
+ ap.add_argument("--contains", nargs=2, metavar=("OUTER", "INNER"),
178
+ help="test whether OUTER contains INNER")
179
+ args = ap.parse_args(argv)
180
+
181
+ try:
182
+ if args.contains:
183
+ outer = _parse_query_rule(args.contains[0])
184
+ inner = _parse_query_rule(args.contains[1])
185
+ fwd, bwd = contains(outer, inner), contains(inner, outer)
186
+ print("outer: %s" % outer)
187
+ print("inner: %s" % inner)
188
+ if fwd and bwd:
189
+ print("=> equivalent (each contains the other)")
190
+ elif fwd:
191
+ print("=> outer contains inner, on every database")
192
+ elif bwd:
193
+ print("=> inner contains outer, on every database")
194
+ else:
195
+ print("=> neither contains the other")
196
+ return 0
197
+
198
+ if not args.file:
199
+ ap.error("give a program to minimise, or use --contains")
200
+ clauses = parse(read_program(args.file))
201
+ validate(clauses)
202
+ for rule in clauses:
203
+ if not rule.body:
204
+ continue
205
+ if any(lit.negated for lit in rule.body):
206
+ print("%s\n (skipped: negation is outside the theory)"
207
+ % rule)
208
+ continue
209
+ if _aggregate_of(rule.head):
210
+ print("%s\n (skipped: aggregation is outside the theory)"
211
+ % rule)
212
+ continue
213
+ atoms = minimise(rule)
214
+ before = len(rule.body)
215
+ if len(atoms) == before:
216
+ print("%s\n already minimal (%d atom%s)"
217
+ % (rule, before, "" if before == 1 else "s"))
218
+ else:
219
+ print("%s\n minimises to %s (%d atoms -> %d)"
220
+ % (rule, _fmt(rule.head, atoms), before, len(atoms)))
221
+ except DatalogError as exc:
222
+ print("error: %s" % exc, file=sys.stderr)
223
+ return 1
224
+ return 0
225
+
226
+
227
+ if __name__ == "__main__":
228
+ sys.exit(main())