tiny-datalog 0.1.0__tar.gz → 0.1.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tiny_datalog-0.1.0 → tiny_datalog-0.1.2}/PKG-INFO +47 -16
- tiny_datalog-0.1.0/tiny_datalog.egg-info/PKG-INFO → tiny_datalog-0.1.2/README.md +25 -27
- tiny_datalog-0.1.2/pyproject.toml +56 -0
- {tiny_datalog-0.1.0 → tiny_datalog-0.1.2}/tiny_datalog/__init__.py +1 -1
- tiny_datalog-0.1.0/README.md → tiny_datalog-0.1.2/tiny_datalog.egg-info/PKG-INFO +58 -15
- tiny_datalog-0.1.0/pyproject.toml +0 -29
- {tiny_datalog-0.1.0 → tiny_datalog-0.1.2}/LICENSE +0 -0
- {tiny_datalog-0.1.0 → tiny_datalog-0.1.2}/setup.cfg +0 -0
- {tiny_datalog-0.1.0 → tiny_datalog-0.1.2}/tiny_datalog/containment.py +0 -0
- {tiny_datalog-0.1.0 → tiny_datalog-0.1.2}/tiny_datalog/datalog.py +0 -0
- {tiny_datalog-0.1.0 → tiny_datalog-0.1.2}/tiny_datalog/incremental.py +0 -0
- {tiny_datalog-0.1.0 → tiny_datalog-0.1.2}/tiny_datalog/magic.py +0 -0
- {tiny_datalog-0.1.0 → tiny_datalog-0.1.2}/tiny_datalog/prolog.py +0 -0
- {tiny_datalog-0.1.0 → tiny_datalog-0.1.2}/tiny_datalog/semantics.py +0 -0
- {tiny_datalog-0.1.0 → tiny_datalog-0.1.2}/tiny_datalog/semiring.py +0 -0
- {tiny_datalog-0.1.0 → tiny_datalog-0.1.2}/tiny_datalog/subsumption.py +0 -0
- {tiny_datalog-0.1.0 → tiny_datalog-0.1.2}/tiny_datalog/tabling.py +0 -0
- {tiny_datalog-0.1.0 → tiny_datalog-0.1.2}/tiny_datalog.egg-info/SOURCES.txt +0 -0
- {tiny_datalog-0.1.0 → tiny_datalog-0.1.2}/tiny_datalog.egg-info/dependency_links.txt +0 -0
- {tiny_datalog-0.1.0 → tiny_datalog-0.1.2}/tiny_datalog.egg-info/entry_points.txt +0 -0
- {tiny_datalog-0.1.0 → tiny_datalog-0.1.2}/tiny_datalog.egg-info/top_level.txt +0 -0
|
@@ -1,10 +1,31 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: tiny-datalog
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.2
|
|
4
4
|
Summary: A Datalog engine small enough to read in an afternoon
|
|
5
5
|
Author: Andrew Goodchild
|
|
6
6
|
License-Expression: MIT
|
|
7
7
|
Project-URL: Homepage, https://github.com/andrewgoodchild/tiny-datalog
|
|
8
|
+
Project-URL: Source, https://github.com/andrewgoodchild/tiny-datalog
|
|
9
|
+
Project-URL: Issues, https://github.com/andrewgoodchild/tiny-datalog/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/andrewgoodchild/tiny-datalog/releases
|
|
11
|
+
Project-URL: Lessons, https://github.com/andrewgoodchild/tiny-datalog/tree/main/lessons
|
|
12
|
+
Keywords: datalog,logic programming,deductive database,semi-naive evaluation,magic sets,provenance,stable models,teaching
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Intended Audience :: Education
|
|
16
|
+
Classifier: Intended Audience :: Science/Research
|
|
17
|
+
Classifier: Operating System :: OS Independent
|
|
18
|
+
Classifier: Programming Language :: Python :: 3
|
|
19
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
24
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
25
|
+
Classifier: Topic :: Database
|
|
26
|
+
Classifier: Topic :: Education
|
|
27
|
+
Classifier: Topic :: Software Development :: Interpreters
|
|
28
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
8
29
|
Requires-Python: >=3.9
|
|
9
30
|
Description-Content-Type: text/markdown
|
|
10
31
|
License-File: LICENSE
|
|
@@ -13,6 +34,7 @@ Dynamic: license-file
|
|
|
13
34
|
# tiny-datalog
|
|
14
35
|
|
|
15
36
|
[](https://github.com/andrewgoodchild/tiny-datalog/actions/workflows/ci.yml)
|
|
37
|
+
[](https://pypi.org/project/tiny-datalog/)
|
|
16
38
|
|
|
17
39
|
**A logic engine small enough to read in an afternoon, and a course
|
|
18
40
|
that builds it up from nothing.**
|
|
@@ -104,7 +126,7 @@ Nothing to install:
|
|
|
104
126
|
|
|
105
127
|
```sh
|
|
106
128
|
git clone https://github.com/andrewgoodchild/tiny-datalog && cd tiny-datalog
|
|
107
|
-
python3 tests.py #
|
|
129
|
+
python3 tests.py # 235 tests, ~12s
|
|
108
130
|
```
|
|
109
131
|
|
|
110
132
|
## Why the language choice decides what you can ask later
|
|
@@ -126,7 +148,7 @@ particular decision came out the way it did, and somebody else asks
|
|
|
126
148
|
whether the rules are even coherent before trusting any answer at all.
|
|
127
149
|
|
|
128
150
|
(The sign-off case has its own demonstration:
|
|
129
|
-
[lesson 17](lessons/17-writing-rules.md) writes a lending policy badly
|
|
151
|
+
[lesson 17](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/17-writing-rules.md) writes a lending policy badly
|
|
130
152
|
twice, and `--explain` names which of two rules wrongly let a
|
|
131
153
|
suspended staff member borrow — three lines, no debugger.)
|
|
132
154
|
|
|
@@ -144,7 +166,7 @@ cannot answer them. Datalog can, because it gave things up:
|
|
|
144
166
|
(Containment and equivalence become undecidable once recursion is
|
|
145
167
|
involved — Shmueli, 1993, which is why `containment.py` handles
|
|
146
168
|
conjunctive queries and refuses the rest.
|
|
147
|
-
[Lesson 16](lessons/16-containment.md) covers the boundary.)
|
|
169
|
+
[Lesson 16](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/16-containment.md) covers the boundary.)
|
|
148
170
|
|
|
149
171
|
Being declarative, recursive and terminating is not a feature list. It
|
|
150
172
|
is the trade that makes rules analysable, and three things follow from
|
|
@@ -174,7 +196,7 @@ the better tool, and that covers most problems.
|
|
|
174
196
|
The worked example above is one row of a table. The full version — 20
|
|
175
197
|
questions, each with the command that answers it and the lesson that
|
|
176
198
|
builds the machinery — is in
|
|
177
|
-
[getting started](lessons/getting-started.md), beside the reading
|
|
199
|
+
[getting started](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/getting-started.md), beside the reading
|
|
178
200
|
paths.
|
|
179
201
|
|
|
180
202
|
## Claims you can check
|
|
@@ -201,7 +223,7 @@ python3 benchmarks/generate.py chain 150 > chain150.dl
|
|
|
201
223
|
The last two are the same rewriting on the same program. Magic sets
|
|
202
224
|
pays in proportion to how much the query's bindings prune; when demand
|
|
203
225
|
is the whole relation the guards are pure overhead.
|
|
204
|
-
[Lesson 7](lessons/07-magic-sets.md) works through why.
|
|
226
|
+
[Lesson 7](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/07-magic-sets.md) works through why.
|
|
205
227
|
|
|
206
228
|
Correctness is checked by a seeded differential fuzzer that generates
|
|
207
229
|
stratified programs and demands semi-naive, naive, magic-sets and
|
|
@@ -209,6 +231,13 @@ tabled evaluation all agree, and that incremental maintenance matches
|
|
|
209
231
|
recomputation under random updates. 400 programs per run;
|
|
210
232
|
`TINY_DATALOG_FUZZ=3000 python3 tests.py DifferentialFuzzTests` soaks.
|
|
211
233
|
|
|
234
|
+
Answers are also checked against engines this repository did not write.
|
|
235
|
+
[`conformance/`](https://github.com/andrewgoodchild/tiny-datalog/tree/main/conformance)
|
|
236
|
+
runs the [datalog-conformance](https://pypi.org/project/datalog-conformance/)
|
|
237
|
+
corpus, harvested from Soufflé, Nemo and Crepe, through all four
|
|
238
|
+
strategies: 89 cases pass. The 25 it skips need arithmetic or 10⁵-fact
|
|
239
|
+
joins, and both are omissions this course makes on purpose.
|
|
240
|
+
|
|
212
241
|
## Learning Datalog
|
|
213
242
|
|
|
214
243
|
`lessons/` is a complete course, no prior exposure assumed, every
|
|
@@ -221,29 +250,29 @@ a lesson on authoring rules that survive review. And it is built to be
|
|
|
221
250
|
inherited: `git clone`, no dependencies, no hosted anything, and every
|
|
222
251
|
quoted transcript re-verified by CI — the exercises cannot rot. (For where each
|
|
223
252
|
technique ships — CodeQL, RDFox, Feldera, SNOMED and the rest —
|
|
224
|
-
[lesson 0](lessons/00-what-is-datalog.md) ends with the deployments.)
|
|
225
|
-
[lessons/getting-started.md](lessons/getting-started.md) has the titles
|
|
253
|
+
[lesson 0](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/00-what-is-datalog.md) ends with the deployments.)
|
|
254
|
+
[lessons/getting-started.md](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/getting-started.md) has the titles
|
|
226
255
|
and the reading order, and
|
|
227
|
-
[lessons/glossary.md](lessons/glossary.md) defines every technical term
|
|
256
|
+
[lessons/glossary.md](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/glossary.md) defines every technical term
|
|
228
257
|
the course uses, and
|
|
229
|
-
[lessons/references.md](lessons/references.md) collects every work the
|
|
258
|
+
[lessons/references.md](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/references.md) collects every work the
|
|
230
259
|
lessons cite.
|
|
231
260
|
|
|
232
261
|
Three of them teach things that are hard to find taught well anywhere
|
|
233
262
|
else, and they are the reason the course exists rather than just the
|
|
234
263
|
engine:
|
|
235
264
|
|
|
236
|
-
- **[Lesson 8](lessons/08-semirings.md)** proves that why-provenance
|
|
265
|
+
- **[Lesson 8](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/08-semirings.md)** proves that why-provenance
|
|
237
266
|
cannot be specialised into derivation counts, with a program that
|
|
238
267
|
prints the disproof: two facts with identical provenance and different
|
|
239
268
|
counts. That settles "materialise provenance once, specialise later,"
|
|
240
269
|
which is a real design-review question with a real answer.
|
|
241
|
-
- **[Lesson 16](lessons/16-containment.md)** shows that the containment
|
|
270
|
+
- **[Lesson 16](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/16-containment.md)** shows that the containment
|
|
242
271
|
test you need for query minimisation is the search already sitting in
|
|
243
272
|
`datalog.py`: `_match` maps a rule body into a database,
|
|
244
273
|
`find_homomorphism` maps a rule body into another rule body. Same
|
|
245
274
|
backtracking, one level up.
|
|
246
|
-
- **[Lesson 4](lessons/04-closed-and-open-worlds.md)** contrasts the
|
|
275
|
+
- **[Lesson 4](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/04-closed-and-open-worlds.md)** contrasts the
|
|
247
276
|
two reasoners in this repository, which disagree about what absence
|
|
248
277
|
means, and leaves you with a habit: when you see `not`, ask whose
|
|
249
278
|
authority says this is absent.
|
|
@@ -265,7 +294,7 @@ quietly:
|
|
|
265
294
|
- **Arithmetic and comparisons.** A built-in isn't a relation you can
|
|
266
295
|
enumerate, so it must be *evaluated* the moment its operands bind —
|
|
267
296
|
which entangles correctness with join order and forces terms to
|
|
268
|
-
become trees. [Lesson 14](lessons/14-arithmetic.md) is the whole
|
|
297
|
+
become trees. [Lesson 14](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/14-arithmetic.md) is the whole
|
|
269
298
|
story, including what to do instead.
|
|
270
299
|
- **Indexes and join planning.** Every join is a nested loop so the
|
|
271
300
|
algorithms stay one-screen readable. It is also why the magic-sets
|
|
@@ -277,7 +306,7 @@ quietly:
|
|
|
277
306
|
- **A REPL (interactive prompt) and packaging.** `git clone` and run.
|
|
278
307
|
|
|
279
308
|
Aggregation used to be on this list;
|
|
280
|
-
[lesson 13](lessons/13-aggregation.md) is what promoting an omission
|
|
309
|
+
[lesson 13](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/13-aggregation.md) is what promoting an omission
|
|
281
310
|
into a feature looks like.
|
|
282
311
|
|
|
283
312
|
## Using it in your own project
|
|
@@ -329,8 +358,10 @@ programs/ teaching programs, numbered by the lesson that uses
|
|
|
329
358
|
lessons/ getting started, glossary, and lessons 0–18
|
|
330
359
|
exercises/ worked answers, verified by the test suite
|
|
331
360
|
cases/ golden test cases — add one without writing Python
|
|
361
|
+
conformance/ the external datalog-conformance corpus (Soufflé, Nemo,
|
|
362
|
+
Crepe), run against all four evaluation strategies
|
|
332
363
|
benchmarks/ scaled input generators (chain/tree/clique/grid)
|
|
333
|
-
tests.py
|
|
364
|
+
tests.py 235 tests: every shipped program and exercise answer is
|
|
334
365
|
executed, a conformance suite runs every query through
|
|
335
366
|
every applicable strategy, and a seeded fuzzer checks
|
|
336
367
|
the same property on random programs
|
|
@@ -1,18 +1,7 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: tiny-datalog
|
|
3
|
-
Version: 0.1.0
|
|
4
|
-
Summary: A Datalog engine small enough to read in an afternoon
|
|
5
|
-
Author: Andrew Goodchild
|
|
6
|
-
License-Expression: MIT
|
|
7
|
-
Project-URL: Homepage, https://github.com/andrewgoodchild/tiny-datalog
|
|
8
|
-
Requires-Python: >=3.9
|
|
9
|
-
Description-Content-Type: text/markdown
|
|
10
|
-
License-File: LICENSE
|
|
11
|
-
Dynamic: license-file
|
|
12
|
-
|
|
13
1
|
# tiny-datalog
|
|
14
2
|
|
|
15
3
|
[](https://github.com/andrewgoodchild/tiny-datalog/actions/workflows/ci.yml)
|
|
4
|
+
[](https://pypi.org/project/tiny-datalog/)
|
|
16
5
|
|
|
17
6
|
**A logic engine small enough to read in an afternoon, and a course
|
|
18
7
|
that builds it up from nothing.**
|
|
@@ -104,7 +93,7 @@ Nothing to install:
|
|
|
104
93
|
|
|
105
94
|
```sh
|
|
106
95
|
git clone https://github.com/andrewgoodchild/tiny-datalog && cd tiny-datalog
|
|
107
|
-
python3 tests.py #
|
|
96
|
+
python3 tests.py # 235 tests, ~12s
|
|
108
97
|
```
|
|
109
98
|
|
|
110
99
|
## Why the language choice decides what you can ask later
|
|
@@ -126,7 +115,7 @@ particular decision came out the way it did, and somebody else asks
|
|
|
126
115
|
whether the rules are even coherent before trusting any answer at all.
|
|
127
116
|
|
|
128
117
|
(The sign-off case has its own demonstration:
|
|
129
|
-
[lesson 17](lessons/17-writing-rules.md) writes a lending policy badly
|
|
118
|
+
[lesson 17](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/17-writing-rules.md) writes a lending policy badly
|
|
130
119
|
twice, and `--explain` names which of two rules wrongly let a
|
|
131
120
|
suspended staff member borrow — three lines, no debugger.)
|
|
132
121
|
|
|
@@ -144,7 +133,7 @@ cannot answer them. Datalog can, because it gave things up:
|
|
|
144
133
|
(Containment and equivalence become undecidable once recursion is
|
|
145
134
|
involved — Shmueli, 1993, which is why `containment.py` handles
|
|
146
135
|
conjunctive queries and refuses the rest.
|
|
147
|
-
[Lesson 16](lessons/16-containment.md) covers the boundary.)
|
|
136
|
+
[Lesson 16](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/16-containment.md) covers the boundary.)
|
|
148
137
|
|
|
149
138
|
Being declarative, recursive and terminating is not a feature list. It
|
|
150
139
|
is the trade that makes rules analysable, and three things follow from
|
|
@@ -174,7 +163,7 @@ the better tool, and that covers most problems.
|
|
|
174
163
|
The worked example above is one row of a table. The full version — 20
|
|
175
164
|
questions, each with the command that answers it and the lesson that
|
|
176
165
|
builds the machinery — is in
|
|
177
|
-
[getting started](lessons/getting-started.md), beside the reading
|
|
166
|
+
[getting started](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/getting-started.md), beside the reading
|
|
178
167
|
paths.
|
|
179
168
|
|
|
180
169
|
## Claims you can check
|
|
@@ -201,7 +190,7 @@ python3 benchmarks/generate.py chain 150 > chain150.dl
|
|
|
201
190
|
The last two are the same rewriting on the same program. Magic sets
|
|
202
191
|
pays in proportion to how much the query's bindings prune; when demand
|
|
203
192
|
is the whole relation the guards are pure overhead.
|
|
204
|
-
[Lesson 7](lessons/07-magic-sets.md) works through why.
|
|
193
|
+
[Lesson 7](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/07-magic-sets.md) works through why.
|
|
205
194
|
|
|
206
195
|
Correctness is checked by a seeded differential fuzzer that generates
|
|
207
196
|
stratified programs and demands semi-naive, naive, magic-sets and
|
|
@@ -209,6 +198,13 @@ tabled evaluation all agree, and that incremental maintenance matches
|
|
|
209
198
|
recomputation under random updates. 400 programs per run;
|
|
210
199
|
`TINY_DATALOG_FUZZ=3000 python3 tests.py DifferentialFuzzTests` soaks.
|
|
211
200
|
|
|
201
|
+
Answers are also checked against engines this repository did not write.
|
|
202
|
+
[`conformance/`](https://github.com/andrewgoodchild/tiny-datalog/tree/main/conformance)
|
|
203
|
+
runs the [datalog-conformance](https://pypi.org/project/datalog-conformance/)
|
|
204
|
+
corpus, harvested from Soufflé, Nemo and Crepe, through all four
|
|
205
|
+
strategies: 89 cases pass. The 25 it skips need arithmetic or 10⁵-fact
|
|
206
|
+
joins, and both are omissions this course makes on purpose.
|
|
207
|
+
|
|
212
208
|
## Learning Datalog
|
|
213
209
|
|
|
214
210
|
`lessons/` is a complete course, no prior exposure assumed, every
|
|
@@ -221,29 +217,29 @@ a lesson on authoring rules that survive review. And it is built to be
|
|
|
221
217
|
inherited: `git clone`, no dependencies, no hosted anything, and every
|
|
222
218
|
quoted transcript re-verified by CI — the exercises cannot rot. (For where each
|
|
223
219
|
technique ships — CodeQL, RDFox, Feldera, SNOMED and the rest —
|
|
224
|
-
[lesson 0](lessons/00-what-is-datalog.md) ends with the deployments.)
|
|
225
|
-
[lessons/getting-started.md](lessons/getting-started.md) has the titles
|
|
220
|
+
[lesson 0](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/00-what-is-datalog.md) ends with the deployments.)
|
|
221
|
+
[lessons/getting-started.md](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/getting-started.md) has the titles
|
|
226
222
|
and the reading order, and
|
|
227
|
-
[lessons/glossary.md](lessons/glossary.md) defines every technical term
|
|
223
|
+
[lessons/glossary.md](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/glossary.md) defines every technical term
|
|
228
224
|
the course uses, and
|
|
229
|
-
[lessons/references.md](lessons/references.md) collects every work the
|
|
225
|
+
[lessons/references.md](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/references.md) collects every work the
|
|
230
226
|
lessons cite.
|
|
231
227
|
|
|
232
228
|
Three of them teach things that are hard to find taught well anywhere
|
|
233
229
|
else, and they are the reason the course exists rather than just the
|
|
234
230
|
engine:
|
|
235
231
|
|
|
236
|
-
- **[Lesson 8](lessons/08-semirings.md)** proves that why-provenance
|
|
232
|
+
- **[Lesson 8](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/08-semirings.md)** proves that why-provenance
|
|
237
233
|
cannot be specialised into derivation counts, with a program that
|
|
238
234
|
prints the disproof: two facts with identical provenance and different
|
|
239
235
|
counts. That settles "materialise provenance once, specialise later,"
|
|
240
236
|
which is a real design-review question with a real answer.
|
|
241
|
-
- **[Lesson 16](lessons/16-containment.md)** shows that the containment
|
|
237
|
+
- **[Lesson 16](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/16-containment.md)** shows that the containment
|
|
242
238
|
test you need for query minimisation is the search already sitting in
|
|
243
239
|
`datalog.py`: `_match` maps a rule body into a database,
|
|
244
240
|
`find_homomorphism` maps a rule body into another rule body. Same
|
|
245
241
|
backtracking, one level up.
|
|
246
|
-
- **[Lesson 4](lessons/04-closed-and-open-worlds.md)** contrasts the
|
|
242
|
+
- **[Lesson 4](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/04-closed-and-open-worlds.md)** contrasts the
|
|
247
243
|
two reasoners in this repository, which disagree about what absence
|
|
248
244
|
means, and leaves you with a habit: when you see `not`, ask whose
|
|
249
245
|
authority says this is absent.
|
|
@@ -265,7 +261,7 @@ quietly:
|
|
|
265
261
|
- **Arithmetic and comparisons.** A built-in isn't a relation you can
|
|
266
262
|
enumerate, so it must be *evaluated* the moment its operands bind —
|
|
267
263
|
which entangles correctness with join order and forces terms to
|
|
268
|
-
become trees. [Lesson 14](lessons/14-arithmetic.md) is the whole
|
|
264
|
+
become trees. [Lesson 14](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/14-arithmetic.md) is the whole
|
|
269
265
|
story, including what to do instead.
|
|
270
266
|
- **Indexes and join planning.** Every join is a nested loop so the
|
|
271
267
|
algorithms stay one-screen readable. It is also why the magic-sets
|
|
@@ -277,7 +273,7 @@ quietly:
|
|
|
277
273
|
- **A REPL (interactive prompt) and packaging.** `git clone` and run.
|
|
278
274
|
|
|
279
275
|
Aggregation used to be on this list;
|
|
280
|
-
[lesson 13](lessons/13-aggregation.md) is what promoting an omission
|
|
276
|
+
[lesson 13](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/13-aggregation.md) is what promoting an omission
|
|
281
277
|
into a feature looks like.
|
|
282
278
|
|
|
283
279
|
## Using it in your own project
|
|
@@ -329,8 +325,10 @@ programs/ teaching programs, numbered by the lesson that uses
|
|
|
329
325
|
lessons/ getting started, glossary, and lessons 0–18
|
|
330
326
|
exercises/ worked answers, verified by the test suite
|
|
331
327
|
cases/ golden test cases — add one without writing Python
|
|
328
|
+
conformance/ the external datalog-conformance corpus (Soufflé, Nemo,
|
|
329
|
+
Crepe), run against all four evaluation strategies
|
|
332
330
|
benchmarks/ scaled input generators (chain/tree/clique/grid)
|
|
333
|
-
tests.py
|
|
331
|
+
tests.py 235 tests: every shipped program and exercise answer is
|
|
334
332
|
executed, a conformance suite runs every query through
|
|
335
333
|
every applicable strategy, and a seeded fuzzer checks
|
|
336
334
|
the same property on random programs
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "tiny-datalog"
|
|
7
|
+
version = "0.1.2"
|
|
8
|
+
description = "A Datalog engine small enough to read in an afternoon"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
license-files = ["LICENSE"]
|
|
12
|
+
requires-python = ">=3.9"
|
|
13
|
+
authors = [{name = "Andrew Goodchild"}]
|
|
14
|
+
dependencies = []
|
|
15
|
+
keywords = ["datalog", "logic programming", "deductive database",
|
|
16
|
+
"semi-naive evaluation", "magic sets", "provenance",
|
|
17
|
+
"stable models", "teaching"]
|
|
18
|
+
# No "License ::" classifier: `license` above is an SPDX expression, and
|
|
19
|
+
# setuptools refuses a build that declares the licence both ways.
|
|
20
|
+
classifiers = [
|
|
21
|
+
"Development Status :: 3 - Alpha",
|
|
22
|
+
"Intended Audience :: Developers",
|
|
23
|
+
"Intended Audience :: Education",
|
|
24
|
+
"Intended Audience :: Science/Research",
|
|
25
|
+
"Operating System :: OS Independent",
|
|
26
|
+
"Programming Language :: Python :: 3",
|
|
27
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
28
|
+
"Programming Language :: Python :: 3.9",
|
|
29
|
+
"Programming Language :: Python :: 3.10",
|
|
30
|
+
"Programming Language :: Python :: 3.11",
|
|
31
|
+
"Programming Language :: Python :: 3.12",
|
|
32
|
+
"Programming Language :: Python :: 3.13",
|
|
33
|
+
"Topic :: Database",
|
|
34
|
+
"Topic :: Education",
|
|
35
|
+
"Topic :: Software Development :: Interpreters",
|
|
36
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
37
|
+
]
|
|
38
|
+
|
|
39
|
+
[project.urls]
|
|
40
|
+
Homepage = "https://github.com/andrewgoodchild/tiny-datalog"
|
|
41
|
+
Source = "https://github.com/andrewgoodchild/tiny-datalog"
|
|
42
|
+
Issues = "https://github.com/andrewgoodchild/tiny-datalog/issues"
|
|
43
|
+
Changelog = "https://github.com/andrewgoodchild/tiny-datalog/releases"
|
|
44
|
+
Lessons = "https://github.com/andrewgoodchild/tiny-datalog/tree/main/lessons"
|
|
45
|
+
|
|
46
|
+
[project.scripts]
|
|
47
|
+
tiny-datalog = "tiny_datalog.datalog:main"
|
|
48
|
+
tiny-datalog-prolog = "tiny_datalog.prolog:main"
|
|
49
|
+
tiny-datalog-semiring = "tiny_datalog.semiring:main"
|
|
50
|
+
tiny-datalog-tabling = "tiny_datalog.tabling:main"
|
|
51
|
+
tiny-datalog-incremental = "tiny_datalog.incremental:main"
|
|
52
|
+
tiny-datalog-subsumption = "tiny_datalog.subsumption:main"
|
|
53
|
+
tiny-datalog-containment = "tiny_datalog.containment:main"
|
|
54
|
+
|
|
55
|
+
[tool.setuptools.packages.find]
|
|
56
|
+
include = ["tiny_datalog*"]
|
|
@@ -1,6 +1,40 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: tiny-datalog
|
|
3
|
+
Version: 0.1.2
|
|
4
|
+
Summary: A Datalog engine small enough to read in an afternoon
|
|
5
|
+
Author: Andrew Goodchild
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/andrewgoodchild/tiny-datalog
|
|
8
|
+
Project-URL: Source, https://github.com/andrewgoodchild/tiny-datalog
|
|
9
|
+
Project-URL: Issues, https://github.com/andrewgoodchild/tiny-datalog/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/andrewgoodchild/tiny-datalog/releases
|
|
11
|
+
Project-URL: Lessons, https://github.com/andrewgoodchild/tiny-datalog/tree/main/lessons
|
|
12
|
+
Keywords: datalog,logic programming,deductive database,semi-naive evaluation,magic sets,provenance,stable models,teaching
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Intended Audience :: Education
|
|
16
|
+
Classifier: Intended Audience :: Science/Research
|
|
17
|
+
Classifier: Operating System :: OS Independent
|
|
18
|
+
Classifier: Programming Language :: Python :: 3
|
|
19
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
24
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
25
|
+
Classifier: Topic :: Database
|
|
26
|
+
Classifier: Topic :: Education
|
|
27
|
+
Classifier: Topic :: Software Development :: Interpreters
|
|
28
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
29
|
+
Requires-Python: >=3.9
|
|
30
|
+
Description-Content-Type: text/markdown
|
|
31
|
+
License-File: LICENSE
|
|
32
|
+
Dynamic: license-file
|
|
33
|
+
|
|
1
34
|
# tiny-datalog
|
|
2
35
|
|
|
3
36
|
[](https://github.com/andrewgoodchild/tiny-datalog/actions/workflows/ci.yml)
|
|
37
|
+
[](https://pypi.org/project/tiny-datalog/)
|
|
4
38
|
|
|
5
39
|
**A logic engine small enough to read in an afternoon, and a course
|
|
6
40
|
that builds it up from nothing.**
|
|
@@ -92,7 +126,7 @@ Nothing to install:
|
|
|
92
126
|
|
|
93
127
|
```sh
|
|
94
128
|
git clone https://github.com/andrewgoodchild/tiny-datalog && cd tiny-datalog
|
|
95
|
-
python3 tests.py #
|
|
129
|
+
python3 tests.py # 235 tests, ~12s
|
|
96
130
|
```
|
|
97
131
|
|
|
98
132
|
## Why the language choice decides what you can ask later
|
|
@@ -114,7 +148,7 @@ particular decision came out the way it did, and somebody else asks
|
|
|
114
148
|
whether the rules are even coherent before trusting any answer at all.
|
|
115
149
|
|
|
116
150
|
(The sign-off case has its own demonstration:
|
|
117
|
-
[lesson 17](lessons/17-writing-rules.md) writes a lending policy badly
|
|
151
|
+
[lesson 17](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/17-writing-rules.md) writes a lending policy badly
|
|
118
152
|
twice, and `--explain` names which of two rules wrongly let a
|
|
119
153
|
suspended staff member borrow — three lines, no debugger.)
|
|
120
154
|
|
|
@@ -132,7 +166,7 @@ cannot answer them. Datalog can, because it gave things up:
|
|
|
132
166
|
(Containment and equivalence become undecidable once recursion is
|
|
133
167
|
involved — Shmueli, 1993, which is why `containment.py` handles
|
|
134
168
|
conjunctive queries and refuses the rest.
|
|
135
|
-
[Lesson 16](lessons/16-containment.md) covers the boundary.)
|
|
169
|
+
[Lesson 16](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/16-containment.md) covers the boundary.)
|
|
136
170
|
|
|
137
171
|
Being declarative, recursive and terminating is not a feature list. It
|
|
138
172
|
is the trade that makes rules analysable, and three things follow from
|
|
@@ -162,7 +196,7 @@ the better tool, and that covers most problems.
|
|
|
162
196
|
The worked example above is one row of a table. The full version — 20
|
|
163
197
|
questions, each with the command that answers it and the lesson that
|
|
164
198
|
builds the machinery — is in
|
|
165
|
-
[getting started](lessons/getting-started.md), beside the reading
|
|
199
|
+
[getting started](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/getting-started.md), beside the reading
|
|
166
200
|
paths.
|
|
167
201
|
|
|
168
202
|
## Claims you can check
|
|
@@ -189,7 +223,7 @@ python3 benchmarks/generate.py chain 150 > chain150.dl
|
|
|
189
223
|
The last two are the same rewriting on the same program. Magic sets
|
|
190
224
|
pays in proportion to how much the query's bindings prune; when demand
|
|
191
225
|
is the whole relation the guards are pure overhead.
|
|
192
|
-
[Lesson 7](lessons/07-magic-sets.md) works through why.
|
|
226
|
+
[Lesson 7](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/07-magic-sets.md) works through why.
|
|
193
227
|
|
|
194
228
|
Correctness is checked by a seeded differential fuzzer that generates
|
|
195
229
|
stratified programs and demands semi-naive, naive, magic-sets and
|
|
@@ -197,6 +231,13 @@ tabled evaluation all agree, and that incremental maintenance matches
|
|
|
197
231
|
recomputation under random updates. 400 programs per run;
|
|
198
232
|
`TINY_DATALOG_FUZZ=3000 python3 tests.py DifferentialFuzzTests` soaks.
|
|
199
233
|
|
|
234
|
+
Answers are also checked against engines this repository did not write.
|
|
235
|
+
[`conformance/`](https://github.com/andrewgoodchild/tiny-datalog/tree/main/conformance)
|
|
236
|
+
runs the [datalog-conformance](https://pypi.org/project/datalog-conformance/)
|
|
237
|
+
corpus, harvested from Soufflé, Nemo and Crepe, through all four
|
|
238
|
+
strategies: 89 cases pass. The 25 it skips need arithmetic or 10⁵-fact
|
|
239
|
+
joins, and both are omissions this course makes on purpose.
|
|
240
|
+
|
|
200
241
|
## Learning Datalog
|
|
201
242
|
|
|
202
243
|
`lessons/` is a complete course, no prior exposure assumed, every
|
|
@@ -209,29 +250,29 @@ a lesson on authoring rules that survive review. And it is built to be
|
|
|
209
250
|
inherited: `git clone`, no dependencies, no hosted anything, and every
|
|
210
251
|
quoted transcript re-verified by CI — the exercises cannot rot. (For where each
|
|
211
252
|
technique ships — CodeQL, RDFox, Feldera, SNOMED and the rest —
|
|
212
|
-
[lesson 0](lessons/00-what-is-datalog.md) ends with the deployments.)
|
|
213
|
-
[lessons/getting-started.md](lessons/getting-started.md) has the titles
|
|
253
|
+
[lesson 0](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/00-what-is-datalog.md) ends with the deployments.)
|
|
254
|
+
[lessons/getting-started.md](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/getting-started.md) has the titles
|
|
214
255
|
and the reading order, and
|
|
215
|
-
[lessons/glossary.md](lessons/glossary.md) defines every technical term
|
|
256
|
+
[lessons/glossary.md](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/glossary.md) defines every technical term
|
|
216
257
|
the course uses, and
|
|
217
|
-
[lessons/references.md](lessons/references.md) collects every work the
|
|
258
|
+
[lessons/references.md](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/references.md) collects every work the
|
|
218
259
|
lessons cite.
|
|
219
260
|
|
|
220
261
|
Three of them teach things that are hard to find taught well anywhere
|
|
221
262
|
else, and they are the reason the course exists rather than just the
|
|
222
263
|
engine:
|
|
223
264
|
|
|
224
|
-
- **[Lesson 8](lessons/08-semirings.md)** proves that why-provenance
|
|
265
|
+
- **[Lesson 8](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/08-semirings.md)** proves that why-provenance
|
|
225
266
|
cannot be specialised into derivation counts, with a program that
|
|
226
267
|
prints the disproof: two facts with identical provenance and different
|
|
227
268
|
counts. That settles "materialise provenance once, specialise later,"
|
|
228
269
|
which is a real design-review question with a real answer.
|
|
229
|
-
- **[Lesson 16](lessons/16-containment.md)** shows that the containment
|
|
270
|
+
- **[Lesson 16](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/16-containment.md)** shows that the containment
|
|
230
271
|
test you need for query minimisation is the search already sitting in
|
|
231
272
|
`datalog.py`: `_match` maps a rule body into a database,
|
|
232
273
|
`find_homomorphism` maps a rule body into another rule body. Same
|
|
233
274
|
backtracking, one level up.
|
|
234
|
-
- **[Lesson 4](lessons/04-closed-and-open-worlds.md)** contrasts the
|
|
275
|
+
- **[Lesson 4](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/04-closed-and-open-worlds.md)** contrasts the
|
|
235
276
|
two reasoners in this repository, which disagree about what absence
|
|
236
277
|
means, and leaves you with a habit: when you see `not`, ask whose
|
|
237
278
|
authority says this is absent.
|
|
@@ -253,7 +294,7 @@ quietly:
|
|
|
253
294
|
- **Arithmetic and comparisons.** A built-in isn't a relation you can
|
|
254
295
|
enumerate, so it must be *evaluated* the moment its operands bind —
|
|
255
296
|
which entangles correctness with join order and forces terms to
|
|
256
|
-
become trees. [Lesson 14](lessons/14-arithmetic.md) is the whole
|
|
297
|
+
become trees. [Lesson 14](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/14-arithmetic.md) is the whole
|
|
257
298
|
story, including what to do instead.
|
|
258
299
|
- **Indexes and join planning.** Every join is a nested loop so the
|
|
259
300
|
algorithms stay one-screen readable. It is also why the magic-sets
|
|
@@ -265,7 +306,7 @@ quietly:
|
|
|
265
306
|
- **A REPL (interactive prompt) and packaging.** `git clone` and run.
|
|
266
307
|
|
|
267
308
|
Aggregation used to be on this list;
|
|
268
|
-
[lesson 13](lessons/13-aggregation.md) is what promoting an omission
|
|
309
|
+
[lesson 13](https://github.com/andrewgoodchild/tiny-datalog/blob/main/lessons/13-aggregation.md) is what promoting an omission
|
|
269
310
|
into a feature looks like.
|
|
270
311
|
|
|
271
312
|
## Using it in your own project
|
|
@@ -317,8 +358,10 @@ programs/ teaching programs, numbered by the lesson that uses
|
|
|
317
358
|
lessons/ getting started, glossary, and lessons 0–18
|
|
318
359
|
exercises/ worked answers, verified by the test suite
|
|
319
360
|
cases/ golden test cases — add one without writing Python
|
|
361
|
+
conformance/ the external datalog-conformance corpus (Soufflé, Nemo,
|
|
362
|
+
Crepe), run against all four evaluation strategies
|
|
320
363
|
benchmarks/ scaled input generators (chain/tree/clique/grid)
|
|
321
|
-
tests.py
|
|
364
|
+
tests.py 235 tests: every shipped program and exercise answer is
|
|
322
365
|
executed, a conformance suite runs every query through
|
|
323
366
|
every applicable strategy, and a seeded fuzzer checks
|
|
324
367
|
the same property on random programs
|
|
@@ -1,29 +0,0 @@
|
|
|
1
|
-
[build-system]
|
|
2
|
-
requires = ["setuptools>=77"]
|
|
3
|
-
build-backend = "setuptools.build_meta"
|
|
4
|
-
|
|
5
|
-
[project]
|
|
6
|
-
name = "tiny-datalog"
|
|
7
|
-
version = "0.1.0"
|
|
8
|
-
description = "A Datalog engine small enough to read in an afternoon"
|
|
9
|
-
readme = "README.md"
|
|
10
|
-
license = "MIT"
|
|
11
|
-
license-files = ["LICENSE"]
|
|
12
|
-
requires-python = ">=3.9"
|
|
13
|
-
authors = [{name = "Andrew Goodchild"}]
|
|
14
|
-
dependencies = []
|
|
15
|
-
|
|
16
|
-
[project.urls]
|
|
17
|
-
Homepage = "https://github.com/andrewgoodchild/tiny-datalog"
|
|
18
|
-
|
|
19
|
-
[project.scripts]
|
|
20
|
-
tiny-datalog = "tiny_datalog.datalog:main"
|
|
21
|
-
tiny-datalog-prolog = "tiny_datalog.prolog:main"
|
|
22
|
-
tiny-datalog-semiring = "tiny_datalog.semiring:main"
|
|
23
|
-
tiny-datalog-tabling = "tiny_datalog.tabling:main"
|
|
24
|
-
tiny-datalog-incremental = "tiny_datalog.incremental:main"
|
|
25
|
-
tiny-datalog-subsumption = "tiny_datalog.subsumption:main"
|
|
26
|
-
tiny-datalog-containment = "tiny_datalog.containment:main"
|
|
27
|
-
|
|
28
|
-
[tool.setuptools.packages.find]
|
|
29
|
-
include = ["tiny_datalog*"]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|