moofile 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- moofile-0.1.0/LICENSE +21 -0
- moofile-0.1.0/PKG-INFO +450 -0
- moofile-0.1.0/README.md +116 -0
- moofile-0.1.0/docs/README.md +433 -0
- moofile-0.1.0/moofile/__init__.py +44 -0
- moofile-0.1.0/moofile/aggregation.py +114 -0
- moofile-0.1.0/moofile/collection.py +462 -0
- moofile-0.1.0/moofile/errors.py +17 -0
- moofile-0.1.0/moofile/index.py +103 -0
- moofile-0.1.0/moofile/operators.py +37 -0
- moofile-0.1.0/moofile/query.py +211 -0
- moofile-0.1.0/moofile/storage.py +108 -0
- moofile-0.1.0/moofile.egg-info/PKG-INFO +450 -0
- moofile-0.1.0/moofile.egg-info/SOURCES.txt +20 -0
- moofile-0.1.0/moofile.egg-info/dependency_links.txt +1 -0
- moofile-0.1.0/moofile.egg-info/requires.txt +9 -0
- moofile-0.1.0/moofile.egg-info/top_level.txt +1 -0
- moofile-0.1.0/pyproject.toml +23 -0
- moofile-0.1.0/setup.cfg +4 -0
- moofile-0.1.0/tests/test_aggregation.py +133 -0
- moofile-0.1.0/tests/test_collection.py +509 -0
- moofile-0.1.0/tests/test_storage.py +191 -0
moofile-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Pat Wendorf (dungeons@gmail.com)
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
moofile-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,450 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: moofile
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Lightweight embedded document store — SQLite ergonomics, MongoDB-style queries
|
|
5
|
+
License: MIT
|
|
6
|
+
Requires-Python: >=3.10
|
|
7
|
+
Description-Content-Type: text/markdown
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
Requires-Dist: pymongo>=4.0
|
|
10
|
+
Requires-Dist: sortedcontainers>=2.0
|
|
11
|
+
Provides-Extra: pandas
|
|
12
|
+
Requires-Dist: pandas>=1.0; extra == "pandas"
|
|
13
|
+
Provides-Extra: dev
|
|
14
|
+
Requires-Dist: pytest>=7; extra == "dev"
|
|
15
|
+
Requires-Dist: pandas>=1.0; extra == "dev"
|
|
16
|
+
Dynamic: license-file
|
|
17
|
+
|
|
18
|
+
# MooFile
|
|
19
|
+
|
|
20
|
+
> A lightweight, embedded, single-file document store with a developer-friendly query API.
|
|
21
|
+
> No server. No infrastructure. Just a file and a library.
|
|
22
|
+
|
|
23
|
+
```python
|
|
24
|
+
from moofile import Collection, count, mean
|
|
25
|
+
|
|
26
|
+
with Collection("mydata.bson", indexes=["email", "age"]) as db:
|
|
27
|
+
db.insert({"name": "Alice", "email": "alice@example.com", "age": 30})
|
|
28
|
+
|
|
29
|
+
result = (
|
|
30
|
+
db.find({"age": {"$gt": 25}})
|
|
31
|
+
.sort("age", descending=True)
|
|
32
|
+
.limit(10)
|
|
33
|
+
.to_list()
|
|
34
|
+
)
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
---
|
|
38
|
+
|
|
39
|
+
## Why MooFile?
|
|
40
|
+
|
|
41
|
+
| | SQLite | JSON file | MongoDB | **MooFile** |
|
|
42
|
+
|---|---|---|---|---|
|
|
43
|
+
| No server | ✓ | ✓ | ✗ | **✓** |
|
|
44
|
+
| Document-oriented | ✗ | ✓ | ✓ | **✓** |
|
|
45
|
+
| Indexes | ✓ | ✗ | ✓ | **✓** |
|
|
46
|
+
| Developer-friendly API | ✗ (SQL) | ✓ (raw Python) | ✓ | **✓** |
|
|
47
|
+
| Single-file portability | ✓ | ✓ | ✗ | **✓** |
|
|
48
|
+
|
|
49
|
+
MooFile is the right tool when you want MongoDB-style ergonomics without running a server: local tooling, embedded applications, tests, small datasets, single-process services.
|
|
50
|
+
|
|
51
|
+
**Target dataset size:** megabytes to single-digit gigabytes.
|
|
52
|
+
|
|
53
|
+
---
|
|
54
|
+
|
|
55
|
+
## Installation
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
pip install moofile
|
|
59
|
+
# or, with pandas support for .to_df():
|
|
60
|
+
pip install "moofile[pandas]"
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
**Dependencies:** `pymongo` (for BSON encoding) and `sortedcontainers`.
|
|
64
|
+
|
|
65
|
+
---
|
|
66
|
+
|
|
67
|
+
## Quick Start
|
|
68
|
+
|
|
69
|
+
```python
|
|
70
|
+
from moofile import Collection
|
|
71
|
+
|
|
72
|
+
# Open or create a collection. Indexes are declared here.
|
|
73
|
+
db = Collection("users.bson", indexes=["email", "status"])
|
|
74
|
+
|
|
75
|
+
# Insert
|
|
76
|
+
alice = db.insert({"name": "Alice", "email": "alice@example.com", "age": 30, "status": "active"})
|
|
77
|
+
print(alice["_id"]) # auto-generated 24-char hex string
|
|
78
|
+
|
|
79
|
+
db.insert_many([
|
|
80
|
+
{"name": "Bob", "email": "bob@example.com", "age": 22, "status": "trial"},
|
|
81
|
+
{"name": "Carol", "email": "carol@example.com", "age": 40, "status": "active"},
|
|
82
|
+
])
|
|
83
|
+
|
|
84
|
+
# Query
|
|
85
|
+
active = db.find({"status": "active"}).to_list()
|
|
86
|
+
young = db.find({"age": {"$lt": 30}}).sort("age").to_list()
|
|
87
|
+
one = db.find_one({"email": "alice@example.com"})
|
|
88
|
+
|
|
89
|
+
# Update
|
|
90
|
+
db.update_one({"email": "alice@example.com"}, set={"age": 31})
|
|
91
|
+
db.update_many({"status": "trial"}, set={"status": "expired"})
|
|
92
|
+
|
|
93
|
+
# Delete
|
|
94
|
+
db.delete_one({"email": "carol@example.com"})
|
|
95
|
+
db.delete_many({"status": "expired"})
|
|
96
|
+
|
|
97
|
+
# Always close when done (or use a context manager)
|
|
98
|
+
db.close()
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
---
|
|
102
|
+
|
|
103
|
+
## File Layout
|
|
104
|
+
|
|
105
|
+
A MooFile database is two files:
|
|
106
|
+
|
|
107
|
+
```
|
|
108
|
+
users.bson ← append-only document store, source of truth
|
|
109
|
+
users.bson.meta ← index configuration (JSON, human-readable)
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
The `.meta` file is a small JSON file:
|
|
113
|
+
|
|
114
|
+
```json
|
|
115
|
+
{
|
|
116
|
+
"version": 1,
|
|
117
|
+
"indexes": ["email", "status"],
|
|
118
|
+
"created_at": "2025-01-01T00:00:00+00:00"
|
|
119
|
+
}
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
Indexes are **never persisted** — they are rebuilt in memory on every open by scanning the BSON file. If the `.meta` file is lost, delete it and reopen; the data is always safe in the `.bson` file.
|
|
123
|
+
|
|
124
|
+
---
|
|
125
|
+
|
|
126
|
+
## API Reference
|
|
127
|
+
|
|
128
|
+
### Opening a Collection
|
|
129
|
+
|
|
130
|
+
```python
|
|
131
|
+
db = Collection(
|
|
132
|
+
path, # path to the .bson file (created if absent)
|
|
133
|
+
indexes=[], # list of top-level field names to index
|
|
134
|
+
readonly=False, # True to prevent all writes
|
|
135
|
+
schema=None, # optional hints, ignored in v1
|
|
136
|
+
)
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
Use as a context manager for automatic cleanup:
|
|
140
|
+
|
|
141
|
+
```python
|
|
142
|
+
with Collection("data.bson", indexes=["email"]) as db:
|
|
143
|
+
db.insert({"email": "bob@example.com"})
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
---
|
|
147
|
+
|
|
148
|
+
### Insert
|
|
149
|
+
|
|
150
|
+
```python
|
|
151
|
+
doc = db.insert({"name": "alice", "age": 30})
|
|
152
|
+
# → dict with _id populated
|
|
153
|
+
|
|
154
|
+
docs = db.insert_many([{...}, {...}])
|
|
155
|
+
# → list of dicts with _id populated
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
- If `_id` is absent, a random 24-char hex string is generated.
|
|
159
|
+
- Providing a custom `_id` of any hashable type is allowed.
|
|
160
|
+
- `DuplicateKeyError` is raised if `_id` already exists.
|
|
161
|
+
|
|
162
|
+
---
|
|
163
|
+
|
|
164
|
+
### Find
|
|
165
|
+
|
|
166
|
+
```python
|
|
167
|
+
# Return all matching documents as a list
|
|
168
|
+
db.find({"status": "active"}).to_list()
|
|
169
|
+
|
|
170
|
+
# Return first match or None
|
|
171
|
+
db.find_one({"email": "alice@example.com"})
|
|
172
|
+
|
|
173
|
+
# Count without materialising documents
|
|
174
|
+
db.count({"status": "active"})
|
|
175
|
+
|
|
176
|
+
# Existence check
|
|
177
|
+
db.exists({"email": "alice@example.com"})
|
|
178
|
+
```
|
|
179
|
+
|
|
180
|
+
`.find()` returns a lazy `Query` object. No work is done until a terminal method is called.
|
|
181
|
+
|
|
182
|
+
---
|
|
183
|
+
|
|
184
|
+
### Query Chains
|
|
185
|
+
|
|
186
|
+
```python
|
|
187
|
+
results = (
|
|
188
|
+
db.find({"status": "active"})
|
|
189
|
+
.sort("age", descending=True)
|
|
190
|
+
.skip(20)
|
|
191
|
+
.limit(10)
|
|
192
|
+
.to_list()
|
|
193
|
+
)
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
**Builder methods** (each returns a new `Query`):
|
|
197
|
+
|
|
198
|
+
| Method | Description |
|
|
199
|
+
|---|---|
|
|
200
|
+
| `.sort(field, descending=False)` | Sort by field |
|
|
201
|
+
| `.skip(n)` | Skip the first n results |
|
|
202
|
+
| `.limit(n)` | Return at most n results |
|
|
203
|
+
| `.group(field)` | Group results by field |
|
|
204
|
+
| `.agg(*funcs)` | Apply aggregation functions to each group |
|
|
205
|
+
|
|
206
|
+
**Terminal methods** (trigger execution):
|
|
207
|
+
|
|
208
|
+
| Method | Returns |
|
|
209
|
+
|---|---|
|
|
210
|
+
| `.to_list()` | `list[dict]` |
|
|
211
|
+
| `.first()` | `dict` or `None` |
|
|
212
|
+
| `.count()` | `int` |
|
|
213
|
+
| `.to_df()` | `pandas.DataFrame` (requires pandas) |
|
|
214
|
+
|
|
215
|
+
---
|
|
216
|
+
|
|
217
|
+
### Filter Operators
|
|
218
|
+
|
|
219
|
+
#### Comparison
|
|
220
|
+
|
|
221
|
+
```python
|
|
222
|
+
{"age": 30} # implicit $eq
|
|
223
|
+
{"age": {"$eq": 30}} # explicit $eq
|
|
224
|
+
{"age": {"$ne": 30}} # not equal
|
|
225
|
+
{"age": {"$gt": 25}} # greater than
|
|
226
|
+
{"age": {"$gte": 25}} # greater than or equal
|
|
227
|
+
{"age": {"$lt": 40}} # less than
|
|
228
|
+
{"age": {"$lte": 40}} # less than or equal
|
|
229
|
+
{"age": {"$gte": 25, "$lt": 40}} # range
|
|
230
|
+
{"status": {"$in": ["active", "trial"]}}
|
|
231
|
+
{"status": {"$nin": ["expired", "archived"]}}
|
|
232
|
+
```
|
|
233
|
+
|
|
234
|
+
#### Logical
|
|
235
|
+
|
|
236
|
+
```python
|
|
237
|
+
{"$and": [{"age": {"$gt": 25}}, {"status": "active"}]}
|
|
238
|
+
{"$or": [{"status": "active"}, {"status": "trial"}]}
|
|
239
|
+
{"$not": {"status": "archived"}}
|
|
240
|
+
```
|
|
241
|
+
|
|
242
|
+
#### Element
|
|
243
|
+
|
|
244
|
+
```python
|
|
245
|
+
{"email": {"$exists": True}} # field must be present
|
|
246
|
+
{"email": {"$exists": False}} # field must be absent
|
|
247
|
+
```
|
|
248
|
+
|
|
249
|
+
#### Array
|
|
250
|
+
|
|
251
|
+
```python
|
|
252
|
+
# At least one element of 'tags' equals "vip"
|
|
253
|
+
{"tags": {"$elemMatch": {"$eq": "vip"}}}
|
|
254
|
+
|
|
255
|
+
# At least one element of 'scores' is > 90
|
|
256
|
+
{"scores": {"$elemMatch": {"$gt": 90}}}
|
|
257
|
+
|
|
258
|
+
# At least one element of 'items' matches a sub-document filter
|
|
259
|
+
{"items": {"$elemMatch": {"product": "xyz", "qty": {"$gt": 5}}}}
|
|
260
|
+
```
|
|
261
|
+
|
|
262
|
+
---
|
|
263
|
+
|
|
264
|
+
### Update
|
|
265
|
+
|
|
266
|
+
```python
|
|
267
|
+
# Update first match — raises DocumentNotFoundError if no match
|
|
268
|
+
db.update_one(
|
|
269
|
+
where={"email": "alice@example.com"},
|
|
270
|
+
set={"age": 31}, # $set: set field values
|
|
271
|
+
unset=["temp_field"], # $unset: remove fields
|
|
272
|
+
inc={"login_count": 1}, # $inc: increment numeric fields
|
|
273
|
+
)
|
|
274
|
+
|
|
275
|
+
# Update all matches — returns count of updated documents
|
|
276
|
+
n = db.update_many(
|
|
277
|
+
where={"status": "trial"},
|
|
278
|
+
set={"status": "expired"},
|
|
279
|
+
)
|
|
280
|
+
|
|
281
|
+
# Replace entire document — preserves _id, raises DocumentNotFoundError if no match
|
|
282
|
+
db.replace_one({"_id": "abc123"}, {"name": "Alice", "age": 32})
|
|
283
|
+
```
|
|
284
|
+
|
|
285
|
+
---
|
|
286
|
+
|
|
287
|
+
### Delete
|
|
288
|
+
|
|
289
|
+
```python
|
|
290
|
+
# Delete first match — returns True if deleted, False if nothing matched
|
|
291
|
+
db.delete_one({"_id": "abc123"})
|
|
292
|
+
|
|
293
|
+
# Delete all matches — returns count
|
|
294
|
+
n = db.delete_many({"status": "archived"})
|
|
295
|
+
```
|
|
296
|
+
|
|
297
|
+
---
|
|
298
|
+
|
|
299
|
+
### Aggregation
|
|
300
|
+
|
|
301
|
+
Group documents and compute aggregate statistics:
|
|
302
|
+
|
|
303
|
+
```python
|
|
304
|
+
from moofile import count, sum, mean, min, max, collect, first, last
|
|
305
|
+
|
|
306
|
+
results = (
|
|
307
|
+
db.find({"status": "active"})
|
|
308
|
+
.group("city")
|
|
309
|
+
.agg(
|
|
310
|
+
count(),
|
|
311
|
+
mean("age"),
|
|
312
|
+
sum("revenue"),
|
|
313
|
+
min("created_at"),
|
|
314
|
+
max("created_at"),
|
|
315
|
+
)
|
|
316
|
+
.sort("count", descending=True)
|
|
317
|
+
.limit(10)
|
|
318
|
+
.to_list()
|
|
319
|
+
)
|
|
320
|
+
```
|
|
321
|
+
|
|
322
|
+
**Aggregation functions:**
|
|
323
|
+
|
|
324
|
+
| Function | Output field | Description |
|
|
325
|
+
|---|---|---|
|
|
326
|
+
| `count()` | `"count"` | Number of documents in group |
|
|
327
|
+
| `sum("field")` | `"sum_field"` | Sum of field values |
|
|
328
|
+
| `mean("field")` | `"mean_field"` | Arithmetic mean of field values |
|
|
329
|
+
| `min("field")` | `"min_field"` | Minimum field value |
|
|
330
|
+
| `max("field")` | `"max_field"` | Maximum field value |
|
|
331
|
+
| `collect("field")` | `"collect_field"` | List of all values |
|
|
332
|
+
| `first("field")` | `"first_field"` | First value encountered |
|
|
333
|
+
| `last("field")` | `"last_field"` | Last value encountered |
|
|
334
|
+
|
|
335
|
+
Documents where the aggregated field is absent are excluded from the computation (but still counted by `count()`).
|
|
336
|
+
|
|
337
|
+
---
|
|
338
|
+
|
|
339
|
+
### Utility
|
|
340
|
+
|
|
341
|
+
```python
|
|
342
|
+
# Database statistics
|
|
343
|
+
s = db.stats()
|
|
344
|
+
# → {
|
|
345
|
+
# "documents": 42150,
|
|
346
|
+
# "dead_records": 3201,
|
|
347
|
+
# "file_size_bytes": 8421000,
|
|
348
|
+
# "dead_ratio": 0.07,
|
|
349
|
+
# }
|
|
350
|
+
|
|
351
|
+
# Compact the file (remove dead records)
|
|
352
|
+
db.compact()
|
|
353
|
+
|
|
354
|
+
# Rebuild indexes from scratch (useful after manual file manipulation)
|
|
355
|
+
db.reindex()
|
|
356
|
+
|
|
357
|
+
# Explicit close
|
|
358
|
+
db.close()
|
|
359
|
+
```
|
|
360
|
+
|
|
361
|
+
**When to compact:** when `dead_ratio` exceeds ~0.30 (30 %). Compaction is always explicit — MooFile never compacts automatically.
|
|
362
|
+
|
|
363
|
+
---
|
|
364
|
+
|
|
365
|
+
### Error Handling
|
|
366
|
+
|
|
367
|
+
```python
|
|
368
|
+
from moofile import (
|
|
369
|
+
MooFileError, # base exception
|
|
370
|
+
DuplicateKeyError, # _id conflict on insert
|
|
371
|
+
DocumentNotFoundError, # update_one / replace_one with no match
|
|
372
|
+
ReadOnlyError, # write attempted on read-only collection
|
|
373
|
+
)
|
|
374
|
+
```
|
|
375
|
+
|
|
376
|
+
All MooFile exceptions are subclasses of `MooFileError`.
|
|
377
|
+
|
|
378
|
+
---
|
|
379
|
+
|
|
380
|
+
## Index Usage
|
|
381
|
+
|
|
382
|
+
MooFile uses an index automatically when a filter's top-level field is indexed:
|
|
383
|
+
|
|
384
|
+
```python
|
|
385
|
+
db = Collection("data.bson", indexes=["email", "age"])
|
|
386
|
+
|
|
387
|
+
# Uses the 'email' index — O(log n) lookup
|
|
388
|
+
db.find({"email": "alice@example.com"})
|
|
389
|
+
|
|
390
|
+
# Uses the 'age' index — O(log n) range scan
|
|
391
|
+
db.find({"age": {"$gt": 25}})
|
|
392
|
+
|
|
393
|
+
# Full scan — 'name' is not indexed
|
|
394
|
+
db.find({"name": "Alice"})
|
|
395
|
+
```
|
|
396
|
+
|
|
397
|
+
Index rules:
|
|
398
|
+
- Only **top-level** fields can be indexed (no nested paths in v1).
|
|
399
|
+
- `_id` is always available for fast lookup regardless of declared indexes.
|
|
400
|
+
- Indexes are rebuilt in memory on every open.
|
|
401
|
+
- Declaring additional indexes is cheap — add them to the `indexes=` parameter and reopen.
|
|
402
|
+
|
|
403
|
+
---
|
|
404
|
+
|
|
405
|
+
## How It Works
|
|
406
|
+
|
|
407
|
+
The `.bson` file is **append-only**. Every insert, update, and delete appends a new record — nothing is ever modified in place.
|
|
408
|
+
|
|
409
|
+
```
|
|
410
|
+
[4 bytes: payload length] [1 byte: record type] [BSON payload]
|
|
411
|
+
```
|
|
412
|
+
|
|
413
|
+
Record types:
|
|
414
|
+
- `0x01` live document
|
|
415
|
+
- `0x02` tombstone (delete marker)
|
|
416
|
+
- `0x03` replacement (update marker)
|
|
417
|
+
|
|
418
|
+
On open, MooFile scans the file once from start to finish. The last record for any given `_id` wins. In-memory indexes are built from the live document set.
|
|
419
|
+
|
|
420
|
+
If the file is truncated mid-write (crash during a write), MooFile detects and removes the incomplete trailing record on the next open. You lose at most the last in-flight write; all prior records are safe.
|
|
421
|
+
|
|
422
|
+
---
|
|
423
|
+
|
|
424
|
+
## Thread Safety
|
|
425
|
+
|
|
426
|
+
Single-threaded only. Concurrent reads are safe. Concurrent writes are not protected — serialise writes at the application layer if needed.
|
|
427
|
+
|
|
428
|
+
---
|
|
429
|
+
|
|
430
|
+
## Non-Goals (v1)
|
|
431
|
+
|
|
432
|
+
- No server or network interface
|
|
433
|
+
- No replication or clustering
|
|
434
|
+
- No multi-process concurrent writes
|
|
435
|
+
- No `$lookup` / joins
|
|
436
|
+
- No nested field indexes
|
|
437
|
+
- No async API
|
|
438
|
+
|
|
439
|
+
---
|
|
440
|
+
|
|
441
|
+
## Examples
|
|
442
|
+
|
|
443
|
+
See the [`examples/`](../examples/) directory:
|
|
444
|
+
|
|
445
|
+
| File | Description |
|
|
446
|
+
|---|---|
|
|
447
|
+
| `basic_crud.py` | Insert, find, update, delete — the complete CRUD tour |
|
|
448
|
+
| `contacts_app.py` | A realistic contacts manager with filtering and updates |
|
|
449
|
+
| `analytics.py` | Sales analytics with `group().agg()` pipeline |
|
|
450
|
+
| `event_log.py` | Structured event log with time-based purging and compaction |
|
moofile-0.1.0/README.md
ADDED
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
# MooFile
|
|
2
|
+
|
|
3
|
+

|
|
4
|
+
|
|
5
|
+
> A lightweight, embedded, single-file document store with a developer-friendly query API.
|
|
6
|
+
> No server. No infrastructure. Just a file and a library.
|
|
7
|
+
|
|
8
|
+
```python
|
|
9
|
+
from moofile import Collection, count, mean
|
|
10
|
+
|
|
11
|
+
with Collection("mydata.bson", indexes=["email", "age"]) as db:
|
|
12
|
+
db.insert({"name": "Alice", "email": "alice@example.com", "age": 30})
|
|
13
|
+
|
|
14
|
+
result = (
|
|
15
|
+
db.find({"age": {"$gt": 25}})
|
|
16
|
+
.sort("age", descending=True)
|
|
17
|
+
.limit(10)
|
|
18
|
+
.to_list()
|
|
19
|
+
)
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
---
|
|
23
|
+
|
|
24
|
+
## Why MooFile?
|
|
25
|
+
|
|
26
|
+
| | SQLite | JSON file | MongoDB | **MooFile** |
|
|
27
|
+
|---|---|---|---|---|
|
|
28
|
+
| No server | ✓ | ✓ | ✗ | **✓** |
|
|
29
|
+
| Document-oriented | ✗ | ✓ | ✓ | **✓** |
|
|
30
|
+
| Indexes | ✓ | ✗ | ✓ | **✓** |
|
|
31
|
+
| Developer-friendly API | ✗ (SQL) | ✓ (raw Python) | ✓ | **✓** |
|
|
32
|
+
| Single-file portability | ✓ | ✓ | ✗ | **✓** |
|
|
33
|
+
|
|
34
|
+
MooFile is the right tool when you want MongoDB-style ergonomics without running a server: local tooling, embedded applications, tests, small datasets, single-process services.
|
|
35
|
+
|
|
36
|
+
**Target dataset size:** megabytes to single-digit gigabytes.
|
|
37
|
+
|
|
38
|
+
---
|
|
39
|
+
|
|
40
|
+
## Installation
|
|
41
|
+
|
|
42
|
+
```bash
|
|
43
|
+
pip install moofile
|
|
44
|
+
# or, with pandas support for .to_df():
|
|
45
|
+
pip install "moofile[pandas]"
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
**Dependencies:** `pymongo` (for BSON encoding) and `sortedcontainers`.
|
|
49
|
+
|
|
50
|
+
---
|
|
51
|
+
|
|
52
|
+
## Quick Start
|
|
53
|
+
|
|
54
|
+
```python
|
|
55
|
+
from moofile import Collection
|
|
56
|
+
|
|
57
|
+
# Open or create a collection. Indexes are declared here.
|
|
58
|
+
db = Collection("users.bson", indexes=["email", "status"])
|
|
59
|
+
|
|
60
|
+
# Insert
|
|
61
|
+
alice = db.insert({"name": "Alice", "email": "alice@example.com", "age": 30, "status": "active"})
|
|
62
|
+
print(alice["_id"]) # auto-generated 24-char hex string
|
|
63
|
+
|
|
64
|
+
db.insert_many([
|
|
65
|
+
{"name": "Bob", "email": "bob@example.com", "age": 22, "status": "trial"},
|
|
66
|
+
{"name": "Carol", "email": "carol@example.com", "age": 40, "status": "active"},
|
|
67
|
+
])
|
|
68
|
+
|
|
69
|
+
# Query
|
|
70
|
+
active = db.find({"status": "active"}).to_list()
|
|
71
|
+
young = db.find({"age": {"$lt": 30}}).sort("age").to_list()
|
|
72
|
+
one = db.find_one({"email": "alice@example.com"})
|
|
73
|
+
|
|
74
|
+
# Update
|
|
75
|
+
db.update_one({"email": "alice@example.com"}, set={"age": 31})
|
|
76
|
+
db.update_many({"status": "trial"}, set={"status": "expired"})
|
|
77
|
+
|
|
78
|
+
# Delete
|
|
79
|
+
db.delete_one({"email": "carol@example.com"})
|
|
80
|
+
db.delete_many({"status": "expired"})
|
|
81
|
+
|
|
82
|
+
# Always close when done (or use a context manager)
|
|
83
|
+
db.close()
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
---
|
|
87
|
+
|
|
88
|
+
## Full Documentation
|
|
89
|
+
|
|
90
|
+
See [`docs/README.md`](docs/README.md) for the complete API reference, including:
|
|
91
|
+
|
|
92
|
+
- Filter operators (`$gt`, `$lt`, `$in`, `$and`, `$or`, `$elemMatch`, ...)
|
|
93
|
+
- Query chains (`.sort()`, `.skip()`, `.limit()`, `.group()`, `.agg()`)
|
|
94
|
+
- Aggregation functions (`count`, `sum`, `mean`, `min`, `max`, `collect`, `first`, `last`)
|
|
95
|
+
- Update operators (`set`, `unset`, `inc`)
|
|
96
|
+
- Index usage and performance notes
|
|
97
|
+
- File format internals
|
|
98
|
+
|
|
99
|
+
---
|
|
100
|
+
|
|
101
|
+
## Examples
|
|
102
|
+
|
|
103
|
+
See the [`examples/`](examples/) directory:
|
|
104
|
+
|
|
105
|
+
| File | Description |
|
|
106
|
+
|---|---|
|
|
107
|
+
| `basic_crud.py` | Insert, find, update, delete — the complete CRUD tour |
|
|
108
|
+
| `contacts_app.py` | A realistic contacts manager with filtering and updates |
|
|
109
|
+
| `analytics.py` | Sales analytics with `group().agg()` pipeline |
|
|
110
|
+
| `event_log.py` | Structured event log with time-based purging and compaction |
|
|
111
|
+
|
|
112
|
+
---
|
|
113
|
+
|
|
114
|
+
## License
|
|
115
|
+
|
|
116
|
+
MIT — see [LICENSE](LICENSE).
|