moofile 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
moofile/__init__.py ADDED
@@ -0,0 +1,44 @@
1
+ """
2
+ MooFile — lightweight embedded document store.
3
+
4
+ from moofile import Collection, count, mean, sum
5
+
6
+ with Collection("data.bson", indexes=["email", "age"]) as db:
7
+ db.insert({"name": "alice", "email": "alice@example.com", "age": 30})
8
+
9
+ results = (
10
+ db.find({"age": {"$gt": 25}})
11
+ .sort("age", descending=True)
12
+ .to_list()
13
+ )
14
+ """
15
+
16
+ from .aggregation import collect, count, first, last, max, mean, min, sum
17
+ from .collection import Collection
18
+ from .errors import (
19
+ DocumentNotFoundError,
20
+ DuplicateKeyError,
21
+ MooFileError,
22
+ ReadOnlyError,
23
+ )
24
+
25
+ __version__ = "0.1.0"
26
+
27
+ __all__ = [
28
+ # Core
29
+ "Collection",
30
+ # Exceptions
31
+ "MooFileError",
32
+ "DuplicateKeyError",
33
+ "DocumentNotFoundError",
34
+ "ReadOnlyError",
35
+ # Aggregation functions
36
+ "count",
37
+ "sum",
38
+ "mean",
39
+ "min",
40
+ "max",
41
+ "collect",
42
+ "first",
43
+ "last",
44
+ ]
moofile/aggregation.py ADDED
@@ -0,0 +1,114 @@
1
+ """
2
+ Aggregation functions for use with Query.group().agg(...).
3
+
4
+ Output field naming convention:
5
+ count() -> "count"
6
+ sum("revenue") -> "sum_revenue"
7
+ mean("age") -> "mean_age"
8
+ min("score") -> "min_score"
9
+ max("score") -> "max_score"
10
+ collect("tags") -> "collect_tags"
11
+ first("name") -> "first_name"
12
+ last("name") -> "last_name"
13
+ """
14
+
15
+ import builtins as _builtins
16
+
17
+
18
+ class AggFunc:
19
+ """Descriptor for a single aggregation operation."""
20
+
21
+ __slots__ = ("output_name", "field", "_func")
22
+
23
+ def __init__(self, output_name: str, field, func) -> None:
24
+ self.output_name = output_name
25
+ self.field = field
26
+ self._func = func
27
+
28
+ def compute(self, docs: list):
29
+ return self._func(docs)
30
+
31
+
32
+ def count() -> AggFunc:
33
+ """Count the number of documents in the group."""
34
+ return AggFunc("count", None, lambda docs: len(docs))
35
+
36
+
37
+ def sum(field: str) -> AggFunc:
38
+ """Sum of field values across the group."""
39
+ _f = field
40
+ return AggFunc(
41
+ f"sum_{_f}",
42
+ _f,
43
+ lambda docs: _builtins.sum(d[_f] for d in docs if _f in d),
44
+ )
45
+
46
+
47
+ def mean(field: str) -> AggFunc:
48
+ """Arithmetic mean of field values across the group."""
49
+ _f = field
50
+
51
+ def _mean(docs):
52
+ vals = [d[_f] for d in docs if _f in d]
53
+ return _builtins.sum(vals) / len(vals) if vals else None
54
+
55
+ return AggFunc(f"mean_{_f}", _f, _mean)
56
+
57
+
58
+ def min(field: str) -> AggFunc:
59
+ """Minimum field value in the group."""
60
+ _f = field
61
+
62
+ def _min(docs):
63
+ vals = [d[_f] for d in docs if _f in d]
64
+ return _builtins.min(vals) if vals else None
65
+
66
+ return AggFunc(f"min_{_f}", _f, _min)
67
+
68
+
69
+ def max(field: str) -> AggFunc:
70
+ """Maximum field value in the group."""
71
+ _f = field
72
+
73
+ def _max(docs):
74
+ vals = [d[_f] for d in docs if _f in d]
75
+ return _builtins.max(vals) if vals else None
76
+
77
+ return AggFunc(f"max_{_f}", _f, _max)
78
+
79
+
80
+ def collect(field: str) -> AggFunc:
81
+ """Collect all field values in the group as a list."""
82
+ _f = field
83
+ return AggFunc(
84
+ f"collect_{_f}",
85
+ _f,
86
+ lambda docs: [d[_f] for d in docs if _f in d],
87
+ )
88
+
89
+
90
+ def first(field: str) -> AggFunc:
91
+ """First value of field encountered in the group."""
92
+ _f = field
93
+
94
+ def _first(docs):
95
+ for d in docs:
96
+ if _f in d:
97
+ return d[_f]
98
+ return None
99
+
100
+ return AggFunc(f"first_{_f}", _f, _first)
101
+
102
+
103
+ def last(field: str) -> AggFunc:
104
+ """Last value of field encountered in the group."""
105
+ _f = field
106
+
107
+ def _last(docs):
108
+ result = None
109
+ for d in docs:
110
+ if _f in d:
111
+ result = d[_f]
112
+ return result
113
+
114
+ return AggFunc(f"last_{_f}", _f, _last)
moofile/collection.py ADDED
@@ -0,0 +1,462 @@
1
+ """Main Collection class — the primary public interface to MooFile."""
2
+
3
+ import binascii
4
+ import json
5
+ import os
6
+ from datetime import datetime, timezone
7
+
8
+ from .errors import DocumentNotFoundError, DuplicateKeyError, ReadOnlyError
9
+ from .index import IndexManager
10
+ from .query import Query, matches
11
+ from .storage import (
12
+ RECORD_LIVE,
13
+ RECORD_REPLACEMENT,
14
+ RECORD_TOMBSTONE,
15
+ StorageEngine,
16
+ compact,
17
+ scan_file,
18
+ )
19
+
20
+
21
+ def _generate_id() -> str:
22
+ """Generate a random 12-byte hex string for use as _id."""
23
+ return binascii.hexlify(os.urandom(12)).decode()
24
+
25
+
26
+ class Collection:
27
+ """
28
+ An embedded, single-file document store.
29
+
30
+ Usage::
31
+
32
+ db = Collection("mydata.bson", indexes=["email", "age"])
33
+ doc = db.insert({"name": "alice", "email": "alice@example.com"})
34
+ results = db.find({"age": {"$gt": 25}}).sort("age").to_list()
35
+
36
+ Use as a context manager for automatic close::
37
+
38
+ with Collection("mydata.bson") as db:
39
+ db.insert({"name": "bob"})
40
+ """
41
+
42
+ def __init__(
43
+ self,
44
+ path: str,
45
+ indexes=None,
46
+ readonly: bool = False,
47
+ schema=None,
48
+ ) -> None:
49
+ self._path = path
50
+ self._readonly = readonly
51
+ self._schema = schema # informational only in v1; not enforced
52
+ self._meta_path = path + ".meta"
53
+ self._total_records: int = 0
54
+ self._storage: StorageEngine | None = None
55
+
56
+ declared = list(indexes or [])
57
+
58
+ if not readonly:
59
+ # Create the data file if it does not exist
60
+ if not os.path.exists(path):
61
+ open(path, "wb").close()
62
+ self._save_meta(declared)
63
+
64
+ loaded_indexes = self._load_meta(declared)
65
+ self._storage = StorageEngine(path, readonly=readonly)
66
+ self._index_manager = IndexManager(loaded_indexes)
67
+ self._load_from_file()
68
+
69
+ # -----------------------------------------------------------------------
70
+ # Insert
71
+ # -----------------------------------------------------------------------
72
+
73
+ def insert(self, doc: dict) -> dict:
74
+ """
75
+ Insert a single document.
76
+
77
+ If _id is absent it is generated automatically.
78
+ Returns the document with _id populated.
79
+ Raises DuplicateKeyError if _id already exists.
80
+ """
81
+ self._require_write()
82
+ doc = dict(doc)
83
+ if "_id" not in doc:
84
+ doc["_id"] = _generate_id()
85
+ if self._index_manager.get(doc["_id"]) is not None:
86
+ raise DuplicateKeyError(f"Duplicate _id: {doc['_id']!r}")
87
+ self._storage.append(RECORD_LIVE, doc)
88
+ self._index_manager.add(doc)
89
+ self._total_records += 1
90
+ return doc
91
+
92
+ def insert_many(self, docs: list) -> list:
93
+ """Insert multiple documents. Returns a list of inserted documents."""
94
+ return [self.insert(doc) for doc in docs]
95
+
96
+ # -----------------------------------------------------------------------
97
+ # Update
98
+ # -----------------------------------------------------------------------
99
+
100
+ def update_one(
101
+ self,
102
+ where: dict,
103
+ set: dict = None,
104
+ unset: list = None,
105
+ inc: dict = None,
106
+ ) -> bool:
107
+ """
108
+ Update the first document matching *where*.
109
+
110
+ Operators:
111
+ set – dict of field→value to set
112
+ unset – list of field names to remove
113
+ inc – dict of field→delta to increment
114
+
115
+ Returns True if a document was updated.
116
+ Raises DocumentNotFoundError if no document matches.
117
+ """
118
+ self._require_write()
119
+ docs = self._get_docs(where)
120
+ if not docs:
121
+ raise DocumentNotFoundError(f"No document matches: {where!r}")
122
+ old_doc = docs[0]
123
+ new_doc = _apply_update(old_doc, set, unset, inc)
124
+ self._storage.append(RECORD_REPLACEMENT, new_doc)
125
+ self._index_manager.remove(old_doc["_id"])
126
+ self._index_manager.add(new_doc)
127
+ self._total_records += 1
128
+ return True
129
+
130
+ def update_many(
131
+ self,
132
+ where: dict,
133
+ set: dict = None,
134
+ unset: list = None,
135
+ inc: dict = None,
136
+ ) -> int:
137
+ """
138
+ Update all documents matching *where*.
139
+
140
+ Returns the count of updated documents.
141
+ """
142
+ self._require_write()
143
+ docs = self._get_docs(where)
144
+ count = 0
145
+ for old_doc in docs:
146
+ new_doc = _apply_update(old_doc, set, unset, inc)
147
+ self._storage.append(RECORD_REPLACEMENT, new_doc)
148
+ self._index_manager.remove(old_doc["_id"])
149
+ self._index_manager.add(new_doc)
150
+ self._total_records += 1
151
+ count += 1
152
+ return count
153
+
154
+ def replace_one(self, where: dict, new_doc: dict) -> bool:
155
+ """
156
+ Replace the entire document matching *where* with *new_doc*.
157
+
158
+ The original _id is preserved.
159
+ Raises DocumentNotFoundError if no document matches.
160
+ """
161
+ self._require_write()
162
+ docs = self._get_docs(where)
163
+ if not docs:
164
+ raise DocumentNotFoundError(f"No document matches: {where!r}")
165
+ old_doc = docs[0]
166
+ replacement = dict(new_doc)
167
+ replacement["_id"] = old_doc["_id"]
168
+ self._storage.append(RECORD_REPLACEMENT, replacement)
169
+ self._index_manager.remove(old_doc["_id"])
170
+ self._index_manager.add(replacement)
171
+ self._total_records += 1
172
+ return True
173
+
174
+ # -----------------------------------------------------------------------
175
+ # Delete
176
+ # -----------------------------------------------------------------------
177
+
178
+ def delete_one(self, where: dict) -> bool:
179
+ """
180
+ Delete the first document matching *where*.
181
+
182
+ Returns True if a document was deleted, False if nothing matched.
183
+ """
184
+ self._require_write()
185
+ docs = self._get_docs(where)
186
+ if not docs:
187
+ return False
188
+ doc = docs[0]
189
+ self._storage.append(RECORD_TOMBSTONE, {"_id": doc["_id"]})
190
+ self._index_manager.remove(doc["_id"])
191
+ self._total_records += 1
192
+ return True
193
+
194
+ def delete_many(self, where: dict) -> int:
195
+ """
196
+ Delete all documents matching *where*.
197
+
198
+ Returns the count of deleted documents.
199
+ """
200
+ self._require_write()
201
+ docs = self._get_docs(where)
202
+ count = 0
203
+ for doc in docs:
204
+ self._storage.append(RECORD_TOMBSTONE, {"_id": doc["_id"]})
205
+ self._index_manager.remove(doc["_id"])
206
+ self._total_records += 1
207
+ count += 1
208
+ return count
209
+
210
+ # -----------------------------------------------------------------------
211
+ # Query
212
+ # -----------------------------------------------------------------------
213
+
214
+ def find(self, filter_dict: dict = None) -> Query:
215
+ """Return a lazy Query object. No work is done until a terminal method is called."""
216
+ return Query(self, filter_dict or {})
217
+
218
+ def find_one(self, filter_dict: dict = None):
219
+ """Return the first matching document, or None."""
220
+ return self.find(filter_dict or {}).first()
221
+
222
+ def count(self, filter_dict: dict = None) -> int:
223
+ """Count documents matching *filter_dict* (all documents if omitted)."""
224
+ return self._count_docs(filter_dict or {})
225
+
226
+ def exists(self, filter_dict: dict) -> bool:
227
+ """Return True if at least one document matches *filter_dict*."""
228
+ return self.find_one(filter_dict) is not None
229
+
230
+ # -----------------------------------------------------------------------
231
+ # Utility
232
+ # -----------------------------------------------------------------------
233
+
234
+ def stats(self) -> dict:
235
+ """
236
+ Return database statistics::
237
+
238
+ {
239
+ "documents": 42150,
240
+ "dead_records": 3201,
241
+ "file_size_bytes": 8421000,
242
+ "dead_ratio": 0.07,
243
+ }
244
+ """
245
+ live = len(self._index_manager._documents)
246
+ dead = self._total_records - live
247
+ file_size = os.path.getsize(self._path) if os.path.exists(self._path) else 0
248
+ ratio = dead / self._total_records if self._total_records > 0 else 0.0
249
+ return {
250
+ "documents": live,
251
+ "dead_records": dead,
252
+ "file_size_bytes": file_size,
253
+ "dead_ratio": ratio,
254
+ }
255
+
256
+ def compact(self) -> None:
257
+ """
258
+ Rewrite the data file keeping only the latest live version of each document.
259
+
260
+ Safe to interrupt — if it fails the original file is untouched.
261
+ """
262
+ self._require_write()
263
+ live_docs = self._index_manager.all_docs()
264
+ self._storage.close()
265
+ try:
266
+ compact(self._path, live_docs)
267
+ finally:
268
+ self._storage.reopen()
269
+ # After compaction total_records == live document count
270
+ self._total_records = len(live_docs)
271
+
272
+ def reindex(self) -> None:
273
+ """Rebuild all in-memory indexes by re-scanning the data file."""
274
+ self._load_from_file()
275
+
276
+ def close(self) -> None:
277
+ """Close the collection and release the file handle."""
278
+ if self._storage is not None:
279
+ self._storage.close()
280
+ self._storage = None
281
+
282
+ # --- Context manager protocol ---
283
+
284
+ def __enter__(self) -> "Collection":
285
+ return self
286
+
287
+ def __exit__(self, *args) -> None:
288
+ self.close()
289
+
290
+ def __del__(self) -> None:
291
+ self.close()
292
+
293
+ # -----------------------------------------------------------------------
294
+ # Internal helpers (used by Query)
295
+ # -----------------------------------------------------------------------
296
+
297
+ def _get_docs(self, filter_dict: dict) -> list:
298
+ """Return all documents matching filter_dict, using indexes when possible."""
299
+ if not filter_dict:
300
+ return self._index_manager.all_docs()
301
+
302
+ candidates = self._try_index(filter_dict)
303
+ if candidates is not None:
304
+ return [doc for doc in candidates if matches(doc, filter_dict)]
305
+
306
+ # Full scan
307
+ return [
308
+ doc
309
+ for doc in self._index_manager.all_docs()
310
+ if matches(doc, filter_dict)
311
+ ]
312
+
313
+ def _count_docs(self, filter_dict: dict) -> int:
314
+ if not filter_dict:
315
+ return len(self._index_manager._documents)
316
+ return len(self._get_docs(filter_dict))
317
+
318
+ def _try_index(self, filter_dict: dict) -> list | None:
319
+ """
320
+ Attempt to use an index for the filter.
321
+
322
+ Returns a (possibly over-broad) candidate list, or None if no index
323
+ can be applied and a full scan is needed.
324
+ """
325
+ # Logical operators at the top level — can't use a single index
326
+ for key in filter_dict:
327
+ if key.startswith("$"):
328
+ return None
329
+
330
+ # Look for the first top-level field that is indexed
331
+ for field, condition in filter_dict.items():
332
+ if field not in self._index_manager._indexes:
333
+ continue
334
+
335
+ if not isinstance(condition, dict):
336
+ # Implicit $eq — exact lookup
337
+ return self._index_manager.get_by_field_exact(field, condition)
338
+
339
+ op_keys = set(condition.keys())
340
+
341
+ if "$eq" in op_keys:
342
+ return self._index_manager.get_by_field_exact(field, condition["$eq"])
343
+
344
+ range_ops = op_keys & {"$gt", "$gte", "$lt", "$lte"}
345
+ if range_ops and not (op_keys - range_ops):
346
+ # Only range operators present — use range scan
347
+ min_val = max_val = None
348
+ min_inc = max_inc = True
349
+ for op, val in condition.items():
350
+ if op == "$gt":
351
+ min_val, min_inc = val, False
352
+ elif op == "$gte":
353
+ min_val, min_inc = val, True
354
+ elif op == "$lt":
355
+ max_val, max_inc = val, False
356
+ elif op == "$lte":
357
+ max_val, max_inc = val, True
358
+ return self._index_manager.get_by_field_range(
359
+ field, min_val, max_val, min_inc, max_inc
360
+ )
361
+
362
+ return None # no usable index found
363
+
364
+ # -----------------------------------------------------------------------
365
+ # File management helpers
366
+ # -----------------------------------------------------------------------
367
+
368
+ def _load_from_file(self) -> None:
369
+ """Scan the BSON file and build in-memory indexes from scratch."""
370
+ self._index_manager.clear()
371
+ self._total_records = 0
372
+
373
+ if not os.path.exists(self._path):
374
+ return
375
+
376
+ records, truncate_to = scan_file(self._path)
377
+
378
+ # Truncate partial trailing write if needed
379
+ if truncate_to is not None and not self._readonly:
380
+ self._storage.close()
381
+ with open(self._path, "r+b") as f:
382
+ f.truncate(truncate_to)
383
+ self._storage.reopen()
384
+
385
+ self._total_records = len(records)
386
+
387
+ # Replay records; the last record for any _id wins
388
+ for _offset, record_type, doc in records:
389
+ _id = doc.get("_id")
390
+ if _id is None:
391
+ continue
392
+ if record_type in (RECORD_LIVE, RECORD_REPLACEMENT):
393
+ # Remove any previous version from the index
394
+ if self._index_manager.get(_id) is not None:
395
+ self._index_manager.remove(_id)
396
+ self._index_manager.add(doc)
397
+ elif record_type == RECORD_TOMBSTONE:
398
+ self._index_manager.remove(_id)
399
+
400
+ def _save_meta(self, indexes: list) -> None:
401
+ """Persist (or update) the .meta file with the given index list."""
402
+ existing: dict = {}
403
+ if os.path.exists(self._meta_path):
404
+ try:
405
+ with open(self._meta_path) as f:
406
+ existing = json.load(f)
407
+ except (json.JSONDecodeError, OSError):
408
+ pass
409
+
410
+ existing_indexes = existing.get("indexes", [])
411
+ # Merge, preserving order and removing duplicates
412
+ merged = list(dict.fromkeys(existing_indexes + indexes))
413
+
414
+ meta = {
415
+ "version": 1,
416
+ "indexes": merged,
417
+ "created_at": existing.get(
418
+ "created_at",
419
+ datetime.now(timezone.utc).isoformat(),
420
+ ),
421
+ }
422
+ with open(self._meta_path, "w") as f:
423
+ json.dump(meta, f, indent=2)
424
+
425
+ def _load_meta(self, declared: list) -> list:
426
+ """Load persisted indexes from .meta and merge with declared indexes."""
427
+ if os.path.exists(self._meta_path):
428
+ try:
429
+ with open(self._meta_path) as f:
430
+ meta = json.load(f)
431
+ persisted = meta.get("indexes", [])
432
+ return list(dict.fromkeys(persisted + declared))
433
+ except (json.JSONDecodeError, OSError):
434
+ pass
435
+ return declared
436
+
437
+ def _require_write(self) -> None:
438
+ if self._readonly:
439
+ raise ReadOnlyError("Collection is open in read-only mode")
440
+
441
+
442
+ # ---------------------------------------------------------------------------
443
+ # Module-level helper
444
+ # ---------------------------------------------------------------------------
445
+
446
+ def _apply_update(
447
+ doc: dict,
448
+ set_dict: dict | None,
449
+ unset_list: list | None,
450
+ inc_dict: dict | None,
451
+ ) -> dict:
452
+ """Apply $set / $unset / $inc operators to a copy of *doc*."""
453
+ new_doc = dict(doc)
454
+ if set_dict:
455
+ new_doc.update(set_dict)
456
+ if unset_list:
457
+ for field in unset_list:
458
+ new_doc.pop(field, None)
459
+ if inc_dict:
460
+ for field, delta in inc_dict.items():
461
+ new_doc[field] = new_doc.get(field, 0) + delta
462
+ return new_doc
moofile/errors.py ADDED
@@ -0,0 +1,17 @@
1
+ """MooFile exception hierarchy."""
2
+
3
+
4
+ class MooFileError(Exception):
5
+ """Base exception for all MooFile errors."""
6
+
7
+
8
+ class DuplicateKeyError(MooFileError):
9
+ """Raised when inserting a document with a duplicate _id."""
10
+
11
+
12
+ class DocumentNotFoundError(MooFileError):
13
+ """Raised when update_one or replace_one finds no matching document."""
14
+
15
+
16
+ class ReadOnlyError(MooFileError):
17
+ """Raised when attempting a write operation on a read-only collection."""