query-cascade 0.3.7__tar.gz → 0.3.9__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {query_cascade-0.3.7 → query_cascade-0.3.9}/PKG-INFO +103 -1
- {query_cascade-0.3.7 → query_cascade-0.3.9}/README.md +102 -0
- {query_cascade-0.3.7 → query_cascade-0.3.9}/pyproject.toml +1 -1
- {query_cascade-0.3.7 → query_cascade-0.3.9}/src/cascade/__init__.py +4 -0
- query_cascade-0.3.9/src/cascade/_ast_rewrite.py +464 -0
- {query_cascade-0.3.7 → query_cascade-0.3.9}/src/cascade/_canonical.py +22 -5
- query_cascade-0.3.9/src/cascade/_collections.py +774 -0
- {query_cascade-0.3.7 → query_cascade-0.3.9}/src/cascade/_disk_cache.py +85 -3
- {query_cascade-0.3.7 → query_cascade-0.3.9}/src/cascade/_evaluator.py +311 -92
- {query_cascade-0.3.7 → query_cascade-0.3.9}/src/cascade/_graph_export.py +58 -2
- query_cascade-0.3.9/src/cascade/_incremental.py +643 -0
- {query_cascade-0.3.7 → query_cascade-0.3.9}/src/cascade/_persistence.py +3 -1
- {query_cascade-0.3.7 → query_cascade-0.3.9}/src/cascade/_runtime.py +1 -0
- {query_cascade-0.3.7 → query_cascade-0.3.9}/src/cascade/_serde.py +23 -7
- {query_cascade-0.3.7 → query_cascade-0.3.9}/src/cascade/_state.py +1 -0
- {query_cascade-0.3.7 → query_cascade-0.3.9}/src/cascade/_store.py +81 -18
- {query_cascade-0.3.7 → query_cascade-0.3.9}/src/cascade/engine.py +111 -35
- {query_cascade-0.3.7 → query_cascade-0.3.9}/src/query_cascade.egg-info/PKG-INFO +103 -1
- {query_cascade-0.3.7 → query_cascade-0.3.9}/src/query_cascade.egg-info/SOURCES.txt +10 -1
- query_cascade-0.3.9/tests/test_ast_rewrite.py +365 -0
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_async.py +10 -7
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_async_coverage.py +26 -22
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_code_versioning.py +34 -32
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_disk_cache.py +38 -23
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_engine.py +27 -7
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_engine_compute_many.py +12 -4
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_engine_concurrency.py +28 -9
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_engine_persistence_and_tracing.py +44 -10
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_engine_private.py +18 -8
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_engine_private2.py +9 -7
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_engine_sweep.py +33 -18
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_error_caching_persistence.py +10 -9
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_evaluator_and_synthetic_coverage.py +3 -1
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_examples.py +3 -1
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_fixed_point.py +8 -6
- query_cascade-0.3.9/tests/test_incremental_collections.py +454 -0
- query_cascade-0.3.9/tests/test_incremental_graph.py +153 -0
- query_cascade-0.3.9/tests/test_incremental_persistence.py +232 -0
- query_cascade-0.3.9/tests/test_incremental_pipeline.py +813 -0
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_internal_invariants.py +4 -1
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_memoize_false.py +10 -8
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_repro_touch_memo_keyerror.py +3 -1
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_scale_behavior.py +44 -16
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_serde.py +10 -6
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_stateful_engine_invariants.py +22 -6
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_store_coverage.py +16 -11
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_sweep_unaccessed.py +9 -8
- query_cascade-0.3.9/tests/test_ttl.py +66 -0
- {query_cascade-0.3.7 → query_cascade-0.3.9}/setup.cfg +0 -0
- {query_cascade-0.3.7 → query_cascade-0.3.9}/src/cascade/_errors.py +0 -0
- {query_cascade-0.3.7 → query_cascade-0.3.9}/src/cascade/_scheduler.py +0 -0
- {query_cascade-0.3.7 → query_cascade-0.3.9}/src/cascade/_synthetic_graph.py +0 -0
- {query_cascade-0.3.7 → query_cascade-0.3.9}/src/query_cascade.egg-info/dependency_links.txt +0 -0
- {query_cascade-0.3.7 → query_cascade-0.3.9}/src/query_cascade.egg-info/requires.txt +0 -0
- {query_cascade-0.3.7 → query_cascade-0.3.9}/src/query_cascade.egg-info/top_level.txt +0 -0
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_dependency_parallelism.py +0 -0
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_error_caching.py +0 -0
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_performance.py +0 -0
- {query_cascade-0.3.7 → query_cascade-0.3.9}/tests/test_subgraph_stats_export.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: query-cascade
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.9
|
|
4
4
|
Summary: Minimal demand-driven query framework for incremental computation.
|
|
5
5
|
Requires-Python: >=3.12
|
|
6
6
|
Description-Content-Type: text/markdown
|
|
@@ -139,6 +139,16 @@ A few things to know:
|
|
|
139
139
|
|
|
140
140
|
## Advanced Features
|
|
141
141
|
|
|
142
|
+
### Time-To-Live Cache Invalidation
|
|
143
|
+
Cascade supports expiring cache entries automatically based on wall-clock time. By passing the `ttl` flag to `@engine.query`, you can guarantee that a node will recompute its value if it is accessed after the specified number of seconds has elapsed since its last evaluation.
|
|
144
|
+
|
|
145
|
+
```python
|
|
146
|
+
@engine.query(ttl=5.0)
|
|
147
|
+
def fetch_external_data():
|
|
148
|
+
return request.get("https://api.example.com/data")
|
|
149
|
+
```
|
|
150
|
+
When a TTL expires, Cascade evaluates the function again. If the newly returned value is identical to the previous one, downstream queries are **not** re-evaluated—early bail-out works exactly as it does for regular dependency changes.
|
|
151
|
+
|
|
142
152
|
### Asynchronous Execution
|
|
143
153
|
Cascade natively supports asynchronous queries via `async def`. This is extremely useful for I/O-bound workflows (such as making database or network calls). When an asynchronous query evaluates, it runs cooperatively on the active `asyncio` event loop.
|
|
144
154
|
|
|
@@ -228,6 +238,95 @@ Set `stats=True` in the `Engine` constructor to track execution timing.
|
|
|
228
238
|
|
|
229
239
|
---
|
|
230
240
|
|
|
241
|
+
## Incremental Collections & Native Map/Reduce
|
|
242
|
+
|
|
243
|
+
`CascadeList`, `CascadeSet`, and `CascadeDict` behave like their builtin counterparts, but every mutation appends a diff to an event log and bumps a hidden engine input. Queries written with ordinary comprehensions and reducers are rewritten at registration time into incremental pipelines that consume only new diffs. Appending one element to a list of a million reruns your mapping function once.
|
|
244
|
+
|
|
245
|
+
```python
|
|
246
|
+
from cascade import Engine, CascadeList
|
|
247
|
+
|
|
248
|
+
engine = Engine()
|
|
249
|
+
docs = CascadeList(engine, ["alpha", "beta", "gamma"], name="docs")
|
|
250
|
+
|
|
251
|
+
@engine.query
|
|
252
|
+
def long_doc_count():
|
|
253
|
+
# Standard Python. Rewritten into filter -> len over the diff stream.
|
|
254
|
+
return len([d for d in docs if len(d) > 4])
|
|
255
|
+
|
|
256
|
+
long_doc_count() # walks all elements once
|
|
257
|
+
docs.append("delta") # one diff
|
|
258
|
+
long_doc_count() # processes only "delta"
|
|
259
|
+
```
|
|
260
|
+
|
|
261
|
+
### The collections
|
|
262
|
+
|
|
263
|
+
* **`CascadeList`** emits `{"action": "insert" | "update" | "remove", "uid": ..., "value": ..., "index": ...}`. Elements carry hidden monotonic uids, so a diff identifies its element stably while positions shift. `sort()` and `reverse()` reorder in place and keep uids attached to their values.
|
|
264
|
+
* **`CascadeSet`** emits `{"action": "add" | "remove", "value": ...}`. The value is its own identity; duplicate adds and absent discards emit nothing.
|
|
265
|
+
* **`CascadeDict`** emits `{"action": "upsert" | "remove", "key": ..., "value": ...}`. `keys()`, `values()`, and `items()` return intercepted views, so a pipeline over `d.keys()` treats a value-only upsert as a no-op.
|
|
266
|
+
|
|
267
|
+
Every diff carries a monotonic `rev` tag. Each consuming pipeline keeps its own checkpoint, so two queries reading one collection advance independently and each requests only the diffs it has not seen.
|
|
268
|
+
|
|
269
|
+
Plain reads (`for x in xs`, `len(xs)`, `xs[0]`, `x in xs`, `.copy()`, dict `.get()`) inside any query record a dependency on the collection, so ordinary loops still invalidate correctly even where nothing is rewritten.
|
|
270
|
+
|
|
271
|
+
### What gets rewritten
|
|
272
|
+
|
|
273
|
+
Rewriting covers comprehensions (`[...]`, `{...}`, `{k: v ...}`, and generator arguments), `map`, `filter`, `reversed`, and these reducers: `sum`, `len`, `min`, `max`, `any`, `all`, `sorted`, `list`, `set`, `dict`, and string-literal `.join`. Adjacent map and filter stages are fused into a single per-item function before execution.
|
|
274
|
+
|
|
275
|
+
| Reducer | Ingest per diff | Finalize |
|
|
276
|
+
|---------|-----------------|----------|
|
|
277
|
+
| `sum`, `len`, `any`, `all` | O(1) | O(1) |
|
|
278
|
+
| `min`, `max` | O(log N) | O(1) |
|
|
279
|
+
| `sorted` | O(log N) | O(output) |
|
|
280
|
+
| `list`, `set`, `dict`, `.join` | O(1) | O(output) |
|
|
281
|
+
| `reversed` | O(1) positional flip | inherited |
|
|
282
|
+
|
|
283
|
+
Everything else runs exactly as written. The boundaries are strict and always fail toward standard execution:
|
|
284
|
+
|
|
285
|
+
* `enumerate()` and `zip()` sources are explicitly excluded. A prepend shifts every enumerate index and zip needs aligned streams, so neither can be made incremental safely; the code is left untouched and runs the normal O(N) way.
|
|
286
|
+
* Arbitrary `for` loops are never rewritten. They still invalidate through read tracking.
|
|
287
|
+
* Unsupported call shapes stay native: multi-generator comprehensions, `sum(..., start)`, `min`/`max` with `key` or multiple arguments, `dict(**kwargs)`, `.join` on anything except a string literal, `sorted` nested inside another pipeline, comprehensions inside `lambda` bodies.
|
|
288
|
+
* Functions whose source is unavailable (`exec`, REPL), `async` functions, lambdas, and functions that shadow a relevant builtin name are registered unmodified.
|
|
289
|
+
* If the source of a rewritten site turns out at runtime to be a plain list, set, dict, or iterator, the site falls back to plain Python semantics, including error behavior.
|
|
290
|
+
|
|
291
|
+
Rewritten functions compile against the original filename and line numbers and reuse the original closure cells, so tracebacks point at your code and late-bound closures behave normally. `handle.raw.__cascade_rewritten__` tells you whether a query was rewritten.
|
|
292
|
+
|
|
293
|
+
### Correctness guards
|
|
294
|
+
|
|
295
|
+
The runtime fingerprints each stage function's bytecode, closure values, and defaults. If a closed-over value changes, the pipeline state rebuilds rather than mixing results from two function versions; if a captured value cannot be fingerprinted, the pipeline rebuilds on every run, which is slower but always correct. A stage function that reads tracked engine state (an input, another query, another collection) marks its pipeline impure, which also forces a rebuild per run, because its per-item results can change without the source collection emitting a diff.
|
|
296
|
+
|
|
297
|
+
A few semantic edges are worth knowing. Incremental `sorted` breaks ties by insertion order into the structure, which can differ from CPython's positional stability after mid-list churn. Ordered outputs from a `CascadeSet` source follow insertion order. Incremental `sum` over strings raises `TypeError` with the generic operand message rather than `sum()`'s specific one. Incremental floating-point sums update by running addition and subtraction, so they can drift from a fresh recompute within normal float tolerance.
|
|
298
|
+
|
|
299
|
+
### Opting out
|
|
300
|
+
|
|
301
|
+
```python
|
|
302
|
+
engine = Engine(incremental=False) # global default off
|
|
303
|
+
|
|
304
|
+
@engine.query(incremental=False) # per-query off
|
|
305
|
+
def q(): ...
|
|
306
|
+
|
|
307
|
+
@engine.query(incremental=True) # per-query on, overrides the global
|
|
308
|
+
def r(): ...
|
|
309
|
+
```
|
|
310
|
+
|
|
311
|
+
Opted-out queries still invalidate through plain read tracking.
|
|
312
|
+
|
|
313
|
+
### Persistence
|
|
314
|
+
|
|
315
|
+
A **named** collection on an engine with `cache_dir` event-sources its log to disk. A later session with the same name restores contents, uid continuity, and revision head, and disk-cached query memos verify against that head, so a warm process serves results without recomputing anything.
|
|
316
|
+
|
|
317
|
+
```python
|
|
318
|
+
engine = Engine(cache_dir="./cache")
|
|
319
|
+
items = CascadeList(engine, name="items", compact_every=1024)
|
|
320
|
+
```
|
|
321
|
+
|
|
322
|
+
The on-disk log compacts into a snapshot automatically every `compact_every` revisions, or on demand with `collection.compact()`. Restoring a name as a different kind raises `PersistentCacheError`, as does mutating a named, disk-backed collection with a value the canonical serializer cannot encode; the failed mutation is rolled back completely. Unnamed collections never persist. `engine.clear_disk_cache()` also wipes collection logs. Collections assume a single writer per name at a time.
|
|
323
|
+
|
|
324
|
+
### Inspection
|
|
325
|
+
|
|
326
|
+
`engine.inspect_pipelines()` lists every observed pipeline with its source, logical stages, fused stage count, checkpoint revision, and consuming queries. `engine.inspect_graph()` merges collection, map/filter, and reduce nodes into the dependency graph, so `export_mermaid` and `export_dot` render the full flow from collection to consuming query.
|
|
327
|
+
|
|
328
|
+
---
|
|
329
|
+
|
|
231
330
|
## Visualization
|
|
232
331
|
Cascade provides renderers for the dependency graph.
|
|
233
332
|
|
|
@@ -249,6 +348,7 @@ mermaid_text = export_mermaid(graph)
|
|
|
249
348
|
2. **Thread Safety:** While Cascade supports parallel query execution, the `Engine` object itself should be modified (`.set()`, `@engine.query`) from a single thread or with external synchronization.
|
|
250
349
|
3. **Persistence Security:** `engine.load()` and the persistent disk cache resolve `@dataclass` and `NamedTuple` types via `importlib`. Only load databases or open cache directories from trusted sources.
|
|
251
350
|
4. **Python Version:** Optimization for parallel CPU-bound work requires **CPython 3.14+ free-threaded** builds with `PYTHON_GIL=0`.
|
|
351
|
+
5. **Collections & `engine.save()`:** State snapshots via `engine.save()`/`engine.load()` do not carry collection event logs; use a named collection with `cache_dir` for durable collection state.
|
|
252
352
|
|
|
253
353
|
---
|
|
254
354
|
|
|
@@ -263,7 +363,9 @@ pip install query-cascade
|
|
|
263
363
|
| Script | What it shows |
|
|
264
364
|
|--------|----------------|
|
|
265
365
|
| `compiler_pipeline.py` | `source → parse → symbols → typecheck`, warnings accumulator, cache-hit narration |
|
|
366
|
+
| `incremental_collections.py` | CascadeList/Set/Dict diffs, comprehension rewriting, O(1) ingest, pipeline inspection |
|
|
266
367
|
| `async_execution.py` | Asynchronous query evaluation and asyncio event loop integration for IO-bound work |
|
|
368
|
+
| `ttl_invalidation.py` | Expiring stale query caches automatically based on wall-clock time |
|
|
267
369
|
| `error_caching.py` | Basic exception caching to prevent repeated re-evaluation on failure |
|
|
268
370
|
| `error_caching_persistence.py` | Disk cache hydration of exceptions across process runs |
|
|
269
371
|
| `code_versioning.py` | Automatic cache invalidation when a function's bytecode logic changes |
|
|
@@ -120,6 +120,16 @@ A few things to know:
|
|
|
120
120
|
|
|
121
121
|
## Advanced Features
|
|
122
122
|
|
|
123
|
+
### Time-To-Live Cache Invalidation
|
|
124
|
+
Cascade supports expiring cache entries automatically based on wall-clock time. By passing the `ttl` flag to `@engine.query`, you can guarantee that a node will recompute its value if it is accessed after the specified number of seconds has elapsed since its last evaluation.
|
|
125
|
+
|
|
126
|
+
```python
|
|
127
|
+
@engine.query(ttl=5.0)
|
|
128
|
+
def fetch_external_data():
|
|
129
|
+
return request.get("https://api.example.com/data")
|
|
130
|
+
```
|
|
131
|
+
When a TTL expires, Cascade evaluates the function again. If the newly returned value is identical to the previous one, downstream queries are **not** re-evaluated—early bail-out works exactly as it does for regular dependency changes.
|
|
132
|
+
|
|
123
133
|
### Asynchronous Execution
|
|
124
134
|
Cascade natively supports asynchronous queries via `async def`. This is extremely useful for I/O-bound workflows (such as making database or network calls). When an asynchronous query evaluates, it runs cooperatively on the active `asyncio` event loop.
|
|
125
135
|
|
|
@@ -209,6 +219,95 @@ Set `stats=True` in the `Engine` constructor to track execution timing.
|
|
|
209
219
|
|
|
210
220
|
---
|
|
211
221
|
|
|
222
|
+
## Incremental Collections & Native Map/Reduce
|
|
223
|
+
|
|
224
|
+
`CascadeList`, `CascadeSet`, and `CascadeDict` behave like their builtin counterparts, but every mutation appends a diff to an event log and bumps a hidden engine input. Queries written with ordinary comprehensions and reducers are rewritten at registration time into incremental pipelines that consume only new diffs. Appending one element to a list of a million reruns your mapping function once.
|
|
225
|
+
|
|
226
|
+
```python
|
|
227
|
+
from cascade import Engine, CascadeList
|
|
228
|
+
|
|
229
|
+
engine = Engine()
|
|
230
|
+
docs = CascadeList(engine, ["alpha", "beta", "gamma"], name="docs")
|
|
231
|
+
|
|
232
|
+
@engine.query
|
|
233
|
+
def long_doc_count():
|
|
234
|
+
# Standard Python. Rewritten into filter -> len over the diff stream.
|
|
235
|
+
return len([d for d in docs if len(d) > 4])
|
|
236
|
+
|
|
237
|
+
long_doc_count() # walks all elements once
|
|
238
|
+
docs.append("delta") # one diff
|
|
239
|
+
long_doc_count() # processes only "delta"
|
|
240
|
+
```
|
|
241
|
+
|
|
242
|
+
### The collections
|
|
243
|
+
|
|
244
|
+
* **`CascadeList`** emits `{"action": "insert" | "update" | "remove", "uid": ..., "value": ..., "index": ...}`. Elements carry hidden monotonic uids, so a diff identifies its element stably while positions shift. `sort()` and `reverse()` reorder in place and keep uids attached to their values.
|
|
245
|
+
* **`CascadeSet`** emits `{"action": "add" | "remove", "value": ...}`. The value is its own identity; duplicate adds and absent discards emit nothing.
|
|
246
|
+
* **`CascadeDict`** emits `{"action": "upsert" | "remove", "key": ..., "value": ...}`. `keys()`, `values()`, and `items()` return intercepted views, so a pipeline over `d.keys()` treats a value-only upsert as a no-op.
|
|
247
|
+
|
|
248
|
+
Every diff carries a monotonic `rev` tag. Each consuming pipeline keeps its own checkpoint, so two queries reading one collection advance independently and each requests only the diffs it has not seen.
|
|
249
|
+
|
|
250
|
+
Plain reads (`for x in xs`, `len(xs)`, `xs[0]`, `x in xs`, `.copy()`, dict `.get()`) inside any query record a dependency on the collection, so ordinary loops still invalidate correctly even where nothing is rewritten.
|
|
251
|
+
|
|
252
|
+
### What gets rewritten
|
|
253
|
+
|
|
254
|
+
Rewriting covers comprehensions (`[...]`, `{...}`, `{k: v ...}`, and generator arguments), `map`, `filter`, `reversed`, and these reducers: `sum`, `len`, `min`, `max`, `any`, `all`, `sorted`, `list`, `set`, `dict`, and string-literal `.join`. Adjacent map and filter stages are fused into a single per-item function before execution.
|
|
255
|
+
|
|
256
|
+
| Reducer | Ingest per diff | Finalize |
|
|
257
|
+
|---------|-----------------|----------|
|
|
258
|
+
| `sum`, `len`, `any`, `all` | O(1) | O(1) |
|
|
259
|
+
| `min`, `max` | O(log N) | O(1) |
|
|
260
|
+
| `sorted` | O(log N) | O(output) |
|
|
261
|
+
| `list`, `set`, `dict`, `.join` | O(1) | O(output) |
|
|
262
|
+
| `reversed` | O(1) positional flip | inherited |
|
|
263
|
+
|
|
264
|
+
Everything else runs exactly as written. The boundaries are strict and always fail toward standard execution:
|
|
265
|
+
|
|
266
|
+
* `enumerate()` and `zip()` sources are explicitly excluded. A prepend shifts every enumerate index and zip needs aligned streams, so neither can be made incremental safely; the code is left untouched and runs the normal O(N) way.
|
|
267
|
+
* Arbitrary `for` loops are never rewritten. They still invalidate through read tracking.
|
|
268
|
+
* Unsupported call shapes stay native: multi-generator comprehensions, `sum(..., start)`, `min`/`max` with `key` or multiple arguments, `dict(**kwargs)`, `.join` on anything except a string literal, `sorted` nested inside another pipeline, comprehensions inside `lambda` bodies.
|
|
269
|
+
* Functions whose source is unavailable (`exec`, REPL), `async` functions, lambdas, and functions that shadow a relevant builtin name are registered unmodified.
|
|
270
|
+
* If the source of a rewritten site turns out at runtime to be a plain list, set, dict, or iterator, the site falls back to plain Python semantics, including error behavior.
|
|
271
|
+
|
|
272
|
+
Rewritten functions compile against the original filename and line numbers and reuse the original closure cells, so tracebacks point at your code and late-bound closures behave normally. `handle.raw.__cascade_rewritten__` tells you whether a query was rewritten.
|
|
273
|
+
|
|
274
|
+
### Correctness guards
|
|
275
|
+
|
|
276
|
+
The runtime fingerprints each stage function's bytecode, closure values, and defaults. If a closed-over value changes, the pipeline state rebuilds rather than mixing results from two function versions; if a captured value cannot be fingerprinted, the pipeline rebuilds on every run, which is slower but always correct. A stage function that reads tracked engine state (an input, another query, another collection) marks its pipeline impure, which also forces a rebuild per run, because its per-item results can change without the source collection emitting a diff.
|
|
277
|
+
|
|
278
|
+
A few semantic edges are worth knowing. Incremental `sorted` breaks ties by insertion order into the structure, which can differ from CPython's positional stability after mid-list churn. Ordered outputs from a `CascadeSet` source follow insertion order. Incremental `sum` over strings raises `TypeError` with the generic operand message rather than `sum()`'s specific one. Incremental floating-point sums update by running addition and subtraction, so they can drift from a fresh recompute within normal float tolerance.
|
|
279
|
+
|
|
280
|
+
### Opting out
|
|
281
|
+
|
|
282
|
+
```python
|
|
283
|
+
engine = Engine(incremental=False) # global default off
|
|
284
|
+
|
|
285
|
+
@engine.query(incremental=False) # per-query off
|
|
286
|
+
def q(): ...
|
|
287
|
+
|
|
288
|
+
@engine.query(incremental=True) # per-query on, overrides the global
|
|
289
|
+
def r(): ...
|
|
290
|
+
```
|
|
291
|
+
|
|
292
|
+
Opted-out queries still invalidate through plain read tracking.
|
|
293
|
+
|
|
294
|
+
### Persistence
|
|
295
|
+
|
|
296
|
+
A **named** collection on an engine with `cache_dir` event-sources its log to disk. A later session with the same name restores contents, uid continuity, and revision head, and disk-cached query memos verify against that head, so a warm process serves results without recomputing anything.
|
|
297
|
+
|
|
298
|
+
```python
|
|
299
|
+
engine = Engine(cache_dir="./cache")
|
|
300
|
+
items = CascadeList(engine, name="items", compact_every=1024)
|
|
301
|
+
```
|
|
302
|
+
|
|
303
|
+
The on-disk log compacts into a snapshot automatically every `compact_every` revisions, or on demand with `collection.compact()`. Restoring a name as a different kind raises `PersistentCacheError`, as does mutating a named, disk-backed collection with a value the canonical serializer cannot encode; the failed mutation is rolled back completely. Unnamed collections never persist. `engine.clear_disk_cache()` also wipes collection logs. Collections assume a single writer per name at a time.
|
|
304
|
+
|
|
305
|
+
### Inspection
|
|
306
|
+
|
|
307
|
+
`engine.inspect_pipelines()` lists every observed pipeline with its source, logical stages, fused stage count, checkpoint revision, and consuming queries. `engine.inspect_graph()` merges collection, map/filter, and reduce nodes into the dependency graph, so `export_mermaid` and `export_dot` render the full flow from collection to consuming query.
|
|
308
|
+
|
|
309
|
+
---
|
|
310
|
+
|
|
212
311
|
## Visualization
|
|
213
312
|
Cascade provides renderers for the dependency graph.
|
|
214
313
|
|
|
@@ -230,6 +329,7 @@ mermaid_text = export_mermaid(graph)
|
|
|
230
329
|
2. **Thread Safety:** While Cascade supports parallel query execution, the `Engine` object itself should be modified (`.set()`, `@engine.query`) from a single thread or with external synchronization.
|
|
231
330
|
3. **Persistence Security:** `engine.load()` and the persistent disk cache resolve `@dataclass` and `NamedTuple` types via `importlib`. Only load databases or open cache directories from trusted sources.
|
|
232
331
|
4. **Python Version:** Optimization for parallel CPU-bound work requires **CPython 3.14+ free-threaded** builds with `PYTHON_GIL=0`.
|
|
332
|
+
5. **Collections & `engine.save()`:** State snapshots via `engine.save()`/`engine.load()` do not carry collection event logs; use a named collection with `cache_dir` for durable collection state.
|
|
233
333
|
|
|
234
334
|
---
|
|
235
335
|
|
|
@@ -244,7 +344,9 @@ pip install query-cascade
|
|
|
244
344
|
| Script | What it shows |
|
|
245
345
|
|--------|----------------|
|
|
246
346
|
| `compiler_pipeline.py` | `source → parse → symbols → typecheck`, warnings accumulator, cache-hit narration |
|
|
347
|
+
| `incremental_collections.py` | CascadeList/Set/Dict diffs, comprehension rewriting, O(1) ingest, pipeline inspection |
|
|
247
348
|
| `async_execution.py` | Asynchronous query evaluation and asyncio event loop integration for IO-bound work |
|
|
349
|
+
| `ttl_invalidation.py` | Expiring stale query caches automatically based on wall-clock time |
|
|
248
350
|
| `error_caching.py` | Basic exception caching to prevent repeated re-evaluation on failure |
|
|
249
351
|
| `error_caching_persistence.py` | Disk cache hydration of exceptions across process runs |
|
|
250
352
|
| `code_versioning.py` | Automatic cache invalidation when a function's bytecode logic changes |
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
from ._collections import CascadeDict, CascadeList, CascadeSet
|
|
1
2
|
from ._graph_export import export_dot, export_mermaid
|
|
2
3
|
from ._state import QueryKey
|
|
3
4
|
from .engine import (
|
|
@@ -14,6 +15,9 @@ from .engine import (
|
|
|
14
15
|
__all__ = [
|
|
15
16
|
"Accumulator",
|
|
16
17
|
"CancellationError",
|
|
18
|
+
"CascadeDict",
|
|
19
|
+
"CascadeList",
|
|
20
|
+
"CascadeSet",
|
|
17
21
|
"CycleError",
|
|
18
22
|
"Engine",
|
|
19
23
|
"PersistentCacheError",
|