client-query-cache 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. client_query_cache-0.3.0/PKG-INFO +59 -0
  2. client_query_cache-0.3.0/README.md +45 -0
  3. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/pyproject.toml +19 -3
  4. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/pyproject.toml.orig +13 -3
  5. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/collation.py +12 -0
  6. client_query_cache-0.3.0/src/client_query_cache/_core/count_reads.py +32 -0
  7. client_query_cache-0.3.0/src/client_query_cache/_core/cursor_capture.py +93 -0
  8. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/entries.py +2 -0
  9. client_query_cache-0.3.0/src/client_query_cache/_core/find_reads.py +69 -0
  10. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/manager.py +96 -7
  11. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/namespace.py +9 -1
  12. client_query_cache-0.3.0/src/client_query_cache/_core/query_filters.py +27 -0
  13. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/read_classification.py +9 -1
  14. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/read_validation.py +3 -7
  15. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/asynchronous/collection.py +68 -170
  16. client_query_cache-0.3.0/src/client_query_cache/asynchronous/cursors.py +315 -0
  17. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/synchronous/collection.py +74 -184
  18. client_query_cache-0.3.0/src/client_query_cache/synchronous/cursors.py +312 -0
  19. client_query_cache-0.2.0/PKG-INFO +0 -141
  20. client_query_cache-0.2.0/README.md +0 -127
  21. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/__init__.py +0 -0
  22. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/__init__.py +0 -0
  23. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/canonical.py +0 -0
  24. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/codec.py +0 -0
  25. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/collection_metadata.py +0 -0
  26. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/errors.py +0 -0
  27. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/find_one_reads.py +0 -0
  28. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/identity_reads.py +0 -0
  29. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/keys.py +0 -0
  30. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/lifecycle.py +0 -0
  31. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/locking.py +0 -0
  32. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/lru.py +0 -0
  33. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/order_sensitive_keys.py +0 -0
  34. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/projection.py +0 -0
  35. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/snapshots.py +0 -0
  36. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/stream_cost.py +0 -0
  37. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/stream_events.py +0 -0
  38. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/stream_health.py +0 -0
  39. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/stream_options.py +0 -0
  40. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/traversal.py +0 -0
  41. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/_core/unique_keys.py +0 -0
  42. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/asynchronous/__init__.py +0 -0
  43. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/asynchronous/database.py +0 -0
  44. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/asynchronous/manager.py +0 -0
  45. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/asynchronous/streams.py +0 -0
  46. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/otel.py +0 -0
  47. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/py.typed +0 -0
  48. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/synchronous/__init__.py +0 -0
  49. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/synchronous/database.py +0 -0
  50. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/synchronous/manager.py +0 -0
  51. {client_query_cache-0.2.0 → client_query_cache-0.3.0}/src/client_query_cache/synchronous/streams.py +0 -0
@@ -0,0 +1,59 @@
1
+ Metadata-Version: 2.3
2
+ Name: client-query-cache
3
+ Version: 0.3.0
4
+ Summary: Client-side caching for PyMongo, kept coherent using MongoDB change streams.
5
+ Author: Alessio Locatelli
6
+ Author-email: Alessio Locatelli <<software.development@secure.mailbox.org>>
7
+ Requires-Dist: pymongo>=4.18.1
8
+ Requires-Dist: opentelemetry-api>=1.45.0 ; extra == 'otel'
9
+ Requires-Python: >=3.14.6
10
+ Project-URL: Repository, https://github.com/alessio-locatelli/client-query-cache
11
+ Project-URL: Documentation, https://alessio-locatelli.github.io/client-query-cache/
12
+ Provides-Extra: otel
13
+ Description-Content-Type: text/markdown
14
+
15
+ # Coherent client-side caching for PyMongo
16
+
17
+ [![Python 3.14.6+](https://img.shields.io/badge/python-3.14.6%2B-blue.svg)](https://www.python.org/downloads/)
18
+
19
+ `client-query-cache` uses MongoDB change streams to keep cached reads coherent. It supports synchronous and asyncio applications without requiring a separate cache server.
20
+
21
+ **Built for production:** Designed for high-traffic, read-heavy applications, with cache hits served from memory to reduce latency and database load. Quality is backed by a 100% test-coverage requirement, real MongoDB integration and concurrency stress tests, and automated performance and memory regression checks.
22
+
23
+ ![Illustrative read latency on a logarithmic scale: direct local MongoDB read, 120 microseconds; direct Atlas M0 read, 79.4 milliseconds; cached read in either deployment, about 61 microseconds.](docs/user/assets/benchmark-latency-light.svg)
24
+
25
+ Results vary by workload and deployment. [Measurements and methodology](https://alessio-locatelli.github.io/client-query-cache/benchmarks/).
26
+
27
+ Caching requires [MongoDB 8.0+ on a replica set or sharded cluster](https://alessio-locatelli.github.io/client-query-cache/getting-started/installation/#requirements). Reads on standalone servers and older MongoDB versions run through PyMongo without caching.
28
+
29
+ Invalidation is asynchronous: a cached read can return a preceding value until the write's change-stream event is processed. Use PyMongo directly when a read must immediately observe a preceding write.
30
+
31
+ ## Quick start
32
+
33
+ ```bash
34
+ uv add client-query-cache
35
+ ```
36
+
37
+ Or use `pip install client-query-cache`.
38
+
39
+ ```python
40
+ from pymongo import MongoClient
41
+
42
+ from client_query_cache import CacheManager
43
+
44
+ with (
45
+ MongoClient("mongodb://localhost:27017") as client,
46
+ CacheManager(client) as cache,
47
+ ):
48
+ users = cache.cached(client["my_database"]["users"])
49
+ users.find_one({"_id": "alice"}) # Reads from MongoDB.
50
+ users.find_one({"_id": "alice"}) # Repeated reads can use the cache.
51
+ ```
52
+
53
+ ## References
54
+
55
+ [Getting started](https://alessio-locatelli.github.io/client-query-cache/getting-started/synchronous/) · [Benchmarks](https://alessio-locatelli.github.io/client-query-cache/benchmarks/) · [Examples](https://alessio-locatelli.github.io/client-query-cache/examples/)
56
+
57
+ ---
58
+
59
+ > MongoDB is a registered trademark of MongoDB, Inc. This project is independent and is not affiliated with or endorsed by MongoDB, Inc.
@@ -0,0 +1,45 @@
1
+ # Coherent client-side caching for PyMongo
2
+
3
+ [![Python 3.14.6+](https://img.shields.io/badge/python-3.14.6%2B-blue.svg)](https://www.python.org/downloads/)
4
+
5
+ `client-query-cache` uses MongoDB change streams to keep cached reads coherent. It supports synchronous and asyncio applications without requiring a separate cache server.
6
+
7
+ **Built for production:** Designed for high-traffic, read-heavy applications, with cache hits served from memory to reduce latency and database load. Quality is backed by a 100% test-coverage requirement, real MongoDB integration and concurrency stress tests, and automated performance and memory regression checks.
8
+
9
+ ![Illustrative read latency on a logarithmic scale: direct local MongoDB read, 120 microseconds; direct Atlas M0 read, 79.4 milliseconds; cached read in either deployment, about 61 microseconds.](docs/user/assets/benchmark-latency-light.svg)
10
+
11
+ Results vary by workload and deployment. [Measurements and methodology](https://alessio-locatelli.github.io/client-query-cache/benchmarks/).
12
+
13
+ Caching requires [MongoDB 8.0+ on a replica set or sharded cluster](https://alessio-locatelli.github.io/client-query-cache/getting-started/installation/#requirements). Reads on standalone servers and older MongoDB versions run through PyMongo without caching.
14
+
15
+ Invalidation is asynchronous: a cached read can return a preceding value until the write's change-stream event is processed. Use PyMongo directly when a read must immediately observe a preceding write.
16
+
17
+ ## Quick start
18
+
19
+ ```bash
20
+ uv add client-query-cache
21
+ ```
22
+
23
+ Or use `pip install client-query-cache`.
24
+
25
+ ```python
26
+ from pymongo import MongoClient
27
+
28
+ from client_query_cache import CacheManager
29
+
30
+ with (
31
+ MongoClient("mongodb://localhost:27017") as client,
32
+ CacheManager(client) as cache,
33
+ ):
34
+ users = cache.cached(client["my_database"]["users"])
35
+ users.find_one({"_id": "alice"}) # Reads from MongoDB.
36
+ users.find_one({"_id": "alice"}) # Repeated reads can use the cache.
37
+ ```
38
+
39
+ ## References
40
+
41
+ [Getting started](https://alessio-locatelli.github.io/client-query-cache/getting-started/synchronous/) · [Benchmarks](https://alessio-locatelli.github.io/client-query-cache/benchmarks/) · [Examples](https://alessio-locatelli.github.io/client-query-cache/examples/)
42
+
43
+ ---
44
+
45
+ > MongoDB is a registered trademark of MongoDB, Inc. This project is independent and is not affiliated with or endorsed by MongoDB, Inc.
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "client-query-cache"
3
- version = "0.2.0"
3
+ version = "0.3.0"
4
4
  description = "Client-side caching for PyMongo, kept coherent using MongoDB change streams."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.14.6"
@@ -27,7 +27,6 @@ dev = [
27
27
  "types-pygments>=2.21.0.20260819",
28
28
  "types-pysocks>=1.7.1.20260518",
29
29
  "types-simplejson>=4.1.0.20260724",
30
- "coverage>=7.16.1",
31
30
  "faker>=40.38.0",
32
31
  "hypothesis>=6.168.0",
33
32
  "pytest>=9.1.1",
@@ -45,16 +44,30 @@ dev = [
45
44
  "types-psutil>=7.2.2.20260906",
46
45
  "dnspython>=2.8.0",
47
46
  "python-snappy>=0.7.3",
47
+ "pytest-xdist[psutil]>=3.8.0",
48
+ "pytest-cov>=7.1.0",
48
49
  ]
49
- docs = ["zensical>=0.0.67"]
50
+ docs = [
51
+ "mike",
52
+ "tomli-w>=1.2.0",
53
+ "zensical>=0.0.68",
54
+ ]
55
+ memory = ["pytest-memray>=1.11.0"]
50
56
 
51
57
  [build-system]
52
58
  requires = ["uv_build>=0.12.18,<0.13.0"]
53
59
  build-backend = "uv_build"
54
60
 
61
+ [tool.uv]
62
+ environments = ["sys_platform == 'linux'"]
63
+
55
64
  [tool.uv.build-backend]
56
65
  module-root = "src"
57
66
 
67
+ [tool.uv.sources.mike]
68
+ git = "https://github.com/squidfunk/mike.git"
69
+ rev = "2d4ad799442f4592db8ad53b179bfb33db8c69ac"
70
+
58
71
  [tool.slotscheck]
59
72
  strict-imports = true
60
73
  require-superclass = true
@@ -90,3 +103,6 @@ level = "aggressive"
90
103
 
91
104
  [tool.ruff-extra-rules.redundant-dict-get]
92
105
  level = "aggressive"
106
+
107
+ [tool.ruff-extra-rules.defined-far-from-use]
108
+ level = "aggressive"
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "client-query-cache"
3
- version = "0.2.0"
3
+ version = "0.3.0"
4
4
  description = "Client-side caching for PyMongo, kept coherent using MongoDB change streams."
5
5
  authors = [
6
6
  { name = "Alessio Locatelli", email = "<software.development@secure.mailbox.org>" },
@@ -26,7 +26,6 @@ dev = [
26
26
  "types-pygments>=2.21.0.20260819",
27
27
  "types-pysocks>=1.7.1.20260518",
28
28
  "types-simplejson>=4.1.0.20260724",
29
- "coverage>=7.16.1",
30
29
  "faker>=40.38.0",
31
30
  "hypothesis>=6.168.0",
32
31
  "pytest>=9.1.1",
@@ -44,8 +43,11 @@ dev = [
44
43
  "types-psutil>=7.2.2.20260906",
45
44
  "dnspython>=2.8.0",
46
45
  "python-snappy>=0.7.3",
46
+ "pytest-xdist[psutil]>=3.8.0",
47
+ "pytest-cov>=7.1.0",
47
48
  ]
48
- docs = ["zensical>=0.0.67"]
49
+ docs = ["mike", "tomli-w>=1.2.0", "zensical>=0.0.68"]
50
+ memory = ["pytest-memray>=1.11.0"]
49
51
 
50
52
  [build-system]
51
53
  requires = ["uv_build>=0.12.18,<0.13.0"]
@@ -54,6 +56,12 @@ build-backend = "uv_build"
54
56
  [tool.uv.build-backend]
55
57
  module-root = "src"
56
58
 
59
+ [tool.uv]
60
+ environments = ["sys_platform == 'linux'"]
61
+
62
+ [tool.uv.sources]
63
+ mike = { git = "https://github.com/squidfunk/mike.git", rev = "2d4ad799442f4592db8ad53b179bfb33db8c69ac" } # pragma: allowlist secret
64
+
57
65
  [tool.slotscheck]
58
66
  strict-imports = true
59
67
  require-superclass = true
@@ -80,3 +88,5 @@ level = "aggressive"
80
88
  level = "aggressive"
81
89
  [tool.ruff-extra-rules.redundant-dict-get]
82
90
  level = "aggressive"
91
+ [tool.ruff-extra-rules.defined-far-from-use]
92
+ level = "aggressive"
@@ -2,6 +2,8 @@ from __future__ import annotations
2
2
 
3
3
  from typing import TYPE_CHECKING, Any
4
4
 
5
+ from pymongo.collation import Collation
6
+
5
7
  if TYPE_CHECKING:
6
8
  from collections.abc import Mapping
7
9
 
@@ -20,3 +22,13 @@ def normalize_collation(
20
22
  if locale == _SIMPLE_LOCALE:
21
23
  return None
22
24
  return collation
25
+
26
+
27
+ def collation_document(
28
+ collation: Collation | Mapping[str, Any] | None,
29
+ ) -> Mapping[str, Any] | None:
30
+ if collation is None:
31
+ return None
32
+ if isinstance(collation, Collation):
33
+ return collation.document
34
+ return dict(collation)
@@ -0,0 +1,32 @@
1
+ from __future__ import annotations
2
+
3
+ from typing import TYPE_CHECKING, TypedDict
4
+
5
+ from pymongo.collation import Collation
6
+
7
+ from client_query_cache._core.collation import collation_document
8
+
9
+ if TYPE_CHECKING:
10
+ from collections.abc import Mapping
11
+
12
+ _OMITTED = object()
13
+
14
+
15
+ class CountReadOptions(TypedDict):
16
+ cache_options: tuple[object, object, object, object]
17
+ extra_options: dict[str, object] # Can be empty for cacheable reads.
18
+
19
+
20
+ def count_read_options(kwargs: Mapping[str, object]) -> CountReadOptions:
21
+ extra_options = dict(kwargs)
22
+ skip = extra_options.pop("skip", 0)
23
+ limit = extra_options.pop("limit", _OMITTED)
24
+ collation = extra_options.pop("collation", None)
25
+ hint = extra_options.pop("hint", _OMITTED)
26
+ if isinstance(collation, Collation):
27
+ collation = collation_document(collation)
28
+ cache_options = (skip, limit, collation, hint)
29
+ return {
30
+ "cache_options": cache_options,
31
+ "extra_options": extra_options,
32
+ }
@@ -0,0 +1,93 @@
1
+ from __future__ import annotations
2
+
3
+ from typing import TYPE_CHECKING, Any
4
+
5
+ from bson.raw_bson import DEFAULT_RAW_BSON_OPTIONS, RawBSONDocument
6
+
7
+ from client_query_cache._core.codec import encode_value
8
+ from client_query_cache._core.errors import CacheClosedError
9
+
10
+ if TYPE_CHECKING:
11
+ from collections.abc import Mapping
12
+
13
+ from bson.codec_options import CodecOptions
14
+
15
+ from client_query_cache._core.entries import AdmissionOutcome
16
+ from client_query_cache._core.find_reads import FindSource
17
+ from client_query_cache._core.manager import CacheCore, NamespaceCapture
18
+
19
+
20
+ class CursorCapture:
21
+ """Own immutable document snapshots until consumption completes."""
22
+
23
+ __slots__ = (
24
+ "_active",
25
+ "_capture",
26
+ "_codec_options",
27
+ "_core",
28
+ "_discriminator",
29
+ "_documents",
30
+ "_find_source",
31
+ "_max_entry_bytes",
32
+ "retained_bytes",
33
+ )
34
+
35
+ def __init__(
36
+ self,
37
+ core: CacheCore,
38
+ capture: NamespaceCapture,
39
+ discriminator: object,
40
+ codec_options: CodecOptions[Mapping[str, Any]],
41
+ *,
42
+ find_source: FindSource | None = None,
43
+ ) -> None:
44
+ self._core = core
45
+ self._capture = capture
46
+ self._discriminator = discriminator
47
+ self._codec_options = codec_options
48
+ self._find_source = find_source
49
+ self._max_entry_bytes = core.snapshot().max_entry_bytes
50
+ self._documents: list[RawBSONDocument] = [] # An empty result is cacheable.
51
+ self.retained_bytes = 0 # Empty and abandoned captures retain no payload.
52
+ self._active = True
53
+
54
+ def append(self, document: Mapping[str, Any]) -> None:
55
+ if not self._active:
56
+ return
57
+ try:
58
+ encoded = encode_value(document, self._codec_options)
59
+ except Exception: # noqa: BLE001 - Application BSON encoders can raise arbitrary errors.
60
+ self.abandon()
61
+ return
62
+ if self.retained_bytes + len(encoded) > self._max_entry_bytes:
63
+ self._core.record_oversized_bypass()
64
+ self.abandon()
65
+ return
66
+ # Access the envelope without replaying application codec transformations.
67
+ envelope = RawBSONDocument(encoded, DEFAULT_RAW_BSON_OPTIONS)
68
+ self._documents.append(envelope["v"])
69
+ self.retained_bytes += len(encoded)
70
+
71
+ def abandon(self) -> None:
72
+ self._active = False
73
+ self._documents.clear()
74
+ self.retained_bytes = 0
75
+
76
+ def finish(self) -> AdmissionOutcome | None:
77
+ if not self._active:
78
+ return None
79
+ self._active = False
80
+ try:
81
+ return self._core.admit_namespace(
82
+ self._capture,
83
+ self._discriminator,
84
+ self._documents,
85
+ codec_options=self._codec_options,
86
+ find_source=self._find_source,
87
+ )
88
+ except CacheClosedError:
89
+ # The cursor owns its native resources independently of the manager.
90
+ return None
91
+ finally:
92
+ self._documents.clear()
93
+ self.retained_bytes = 0
@@ -6,6 +6,7 @@ from typing import TYPE_CHECKING, Any
6
6
 
7
7
  if TYPE_CHECKING:
8
8
  from client_query_cache._core.canonical import Canonical
9
+ from client_query_cache._core.find_reads import FindSource
9
10
  from client_query_cache._core.keys import NamespaceId
10
11
 
11
12
 
@@ -24,6 +25,7 @@ class CacheEntry:
24
25
  value: bytes
25
26
  namespace: NamespaceId
26
27
  identity: Canonical | None
28
+ find_source: FindSource | None = None
27
29
 
28
30
 
29
31
  @dataclass(frozen=True, slots=True)
@@ -0,0 +1,69 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import dataclass
4
+ from typing import TYPE_CHECKING
5
+
6
+ from client_query_cache._core.canonical import canonicalize
7
+ from client_query_cache._core.order_sensitive_keys import (
8
+ order_sensitive_discriminator_key,
9
+ )
10
+ from client_query_cache._core.query_filters import find_filter_key
11
+
12
+ if TYPE_CHECKING:
13
+ from client_query_cache._core.canonical import Canonical
14
+
15
+
16
+ @dataclass(frozen=True, slots=True)
17
+ class FindSource:
18
+ family: int # A compact hash, possibly zero; collisions require key equality.
19
+ limit: int # Zero denotes unlimited; negative and boolean limits are excluded.
20
+
21
+
22
+ @dataclass(frozen=True, slots=True)
23
+ class FindReadShape:
24
+ family: object
25
+ limit: int # Zero and negative limits retain exact-only lookup.
26
+
27
+ @property
28
+ def discriminator(self) -> object:
29
+ return (self.family, self.limit)
30
+
31
+ @property
32
+ def source(self) -> FindSource | None:
33
+ if isinstance(self.limit, bool) or self.limit < 0:
34
+ return None
35
+ return FindSource(hash(canonicalize(self.family)), self.limit)
36
+
37
+
38
+ def find_discriminator(
39
+ family: Canonical,
40
+ limit: int, # Zero and negative limits retain exact identity.
41
+ ) -> Canonical:
42
+ # Reuse the already canonical family without copying its query tree.
43
+ return canonicalize((family, limit))
44
+
45
+
46
+ def find_read_shape(
47
+ filter_document: object,
48
+ projection: object,
49
+ ordering: object,
50
+ skip: int, # Zero means no skipped documents.
51
+ limit: int, # Zero and negative limits use native semantics.
52
+ *,
53
+ collation: object,
54
+ codec: object,
55
+ ) -> FindReadShape:
56
+ return FindReadShape(
57
+ family=order_sensitive_discriminator_key(
58
+ (
59
+ "find",
60
+ find_filter_key(filter_document),
61
+ projection,
62
+ ordering,
63
+ skip,
64
+ collation,
65
+ codec,
66
+ )
67
+ ),
68
+ limit=limit,
69
+ )
@@ -4,7 +4,7 @@ import logging
4
4
  import threading
5
5
  from contextlib import contextmanager
6
6
  from dataclasses import dataclass, field
7
- from typing import TYPE_CHECKING
7
+ from typing import TYPE_CHECKING, cast
8
8
 
9
9
  from bson.errors import BSONError
10
10
 
@@ -16,6 +16,7 @@ from client_query_cache._core.errors import (
16
16
  CacheConfigurationError,
17
17
  UnsupportedCacheRequestError,
18
18
  )
19
+ from client_query_cache._core.find_reads import find_discriminator
19
20
  from client_query_cache._core.keys import (
20
21
  IdentityCacheKey,
21
22
  NamespaceCacheKey,
@@ -45,6 +46,7 @@ if TYPE_CHECKING:
45
46
  from bson.codec_options import CodecOptions
46
47
 
47
48
  from client_query_cache._core.canonical import Canonical
49
+ from client_query_cache._core.find_reads import FindReadShape, FindSource
48
50
  from client_query_cache._core.keys import AliasKey, CacheKey, NamespaceId
49
51
 
50
52
  DEFAULT_SHARED_BUDGET_BYTES = 64 * 1024 * 1024
@@ -107,6 +109,16 @@ def _maybe_prune_identity_locked(
107
109
 
108
110
  def _discard_entry_locked(state: NamespaceState, entry: CacheEntry) -> None:
109
111
  was_indexed = state.entry_index.pop(entry, _NOT_INDEXED) is not _NOT_INDEXED
112
+ if entry.find_source is not None:
113
+ family = entry.find_source.family
114
+ try:
115
+ bucket = state.find_families[family]
116
+ except KeyError:
117
+ pass # Invalidation can already have cleared the family.
118
+ else:
119
+ bucket.pop(entry, None)
120
+ if not bucket:
121
+ del state.find_families[family]
110
122
  if was_indexed and entry.identity is not None:
111
123
  identity_state = state.identities[entry.identity]
112
124
  identity_state.cached_ref_count -= 1
@@ -304,6 +316,9 @@ class _CacheCoreLifecycle(_CacheCoreBase):
304
316
  def record_bypass(self, reason: BypassReason = BypassReason.UNSPECIFIED) -> None:
305
317
  self._statistics.record_bypass(reason)
306
318
 
319
+ def record_oversized_bypass(self) -> None:
320
+ self._statistics.record_oversized_bypass()
321
+
307
322
  def snapshot(self) -> CacheSnapshot:
308
323
  used_bytes, entry_count = self._lru.snapshot_usage()
309
324
  hits, misses, evictions, bypasses, oversized_bypasses, bypass_reasons = (
@@ -361,6 +376,7 @@ class _CacheCoreNamespaceLifecycle(_CacheCoreBase):
361
376
  state = self._namespace(namespace)
362
377
  with self._namespace_section(state):
363
378
  state.generation += 1
379
+ state.find_families.clear()
364
380
  if not is_canonicalizable(identity):
365
381
  return
366
382
  try:
@@ -389,6 +405,7 @@ class _CacheCoreNamespaceLifecycle(_CacheCoreBase):
389
405
  state.aliases = {}
390
406
  reclaimed = list(state.entry_index.items())
391
407
  state.entry_index.clear()
408
+ state.find_families.clear()
392
409
  for entry, _key in reclaimed:
393
410
  if entry.identity is not None:
394
411
  state.identities[entry.identity].cached_ref_count -= 1
@@ -472,13 +489,13 @@ class _CacheCoreIdentityAdmission(_CacheCoreBase):
472
489
  ) -> AdmissionOutcome:
473
490
  self._ensure_active()
474
491
  try:
475
- canonical_shape = canonicalize(read_shape)
476
492
  try:
477
493
  encoded = encode_value(value, codec_options)
478
494
  except BSONError:
479
495
  return AdmissionOutcome.DECLINED_UNENCODABLE
480
496
  except OverflowError:
481
497
  return AdmissionOutcome.DECLINED_UNENCODABLE
498
+ canonical_shape = canonicalize(read_shape)
482
499
  weight = len(encoded)
483
500
  with self._admission_section(
484
501
  capture.namespace, capture.availability_generation, weight
@@ -595,15 +612,16 @@ class _CacheCoreNamespaceAdmission(_CacheCoreBase):
595
612
  value: object,
596
613
  *,
597
614
  codec_options: CodecOptions[Any] | None = None,
615
+ find_source: FindSource | None = None,
598
616
  ) -> AdmissionOutcome:
599
617
  self._ensure_active()
600
- canonical_discriminator = canonicalize(discriminator)
601
618
  try:
602
619
  encoded = encode_value(value, codec_options)
603
620
  except BSONError:
604
621
  return AdmissionOutcome.DECLINED_UNENCODABLE
605
622
  except OverflowError:
606
623
  return AdmissionOutcome.DECLINED_UNENCODABLE
624
+ canonical_discriminator = canonicalize(discriminator)
607
625
  weight = len(encoded)
608
626
  with self._admission_section(
609
627
  capture.namespace, capture.availability_generation, weight
@@ -621,9 +639,15 @@ class _CacheCoreNamespaceAdmission(_CacheCoreBase):
621
639
  value=encoded,
622
640
  namespace=capture.namespace,
623
641
  identity=None,
642
+ find_source=find_source,
624
643
  )
625
644
  admitted, displaced, evicted = self._lru.conditional_put(key, entry)
626
645
  self._process_evicted(evicted)
646
+
647
+ def publish_source() -> None:
648
+ if find_source is not None:
649
+ state.find_families.setdefault(find_source.family, {})[entry] = key
650
+
627
651
  return self._finalize_put(
628
652
  state,
629
653
  key,
@@ -631,7 +655,7 @@ class _CacheCoreNamespaceAdmission(_CacheCoreBase):
631
655
  admitted=admitted,
632
656
  displaced=displaced,
633
657
  is_still_valid=lambda: state.generation == capture.generation,
634
- on_admit=lambda: None,
658
+ on_admit=publish_source,
635
659
  )
636
660
 
637
661
 
@@ -671,19 +695,19 @@ class _CacheCoreUniqueKeyAdmission(_CacheCoreBase):
671
695
  codec_options: CodecOptions[Any] | None = None,
672
696
  ) -> AdmissionOutcome:
673
697
  self._ensure_active()
674
- namespace = namespace_capture.namespace
675
- canonical_discriminator = canonicalize(discriminator)
676
698
  canonical_identity = canonicalize(order_sensitive_key(identity))
677
699
  if canonical_identity is None:
678
700
  message = "identity must not be None"
679
701
  raise UnsupportedCacheRequestError(message)
680
- canonical_shape = canonicalize(read_shape)
681
702
  try:
682
703
  encoded = encode_value(value, codec_options)
683
704
  except BSONError:
684
705
  return AdmissionOutcome.DECLINED_UNENCODABLE
685
706
  except OverflowError:
686
707
  return AdmissionOutcome.DECLINED_UNENCODABLE
708
+ namespace = namespace_capture.namespace
709
+ canonical_discriminator = canonicalize(discriminator)
710
+ canonical_shape = canonicalize(read_shape)
687
711
  weight = len(encoded)
688
712
  with self._admission_section(
689
713
  namespace, namespace_capture.availability_generation, weight
@@ -779,9 +803,74 @@ class _CacheCoreUniqueKeyAdmission(_CacheCoreBase):
779
803
  return namespace_outcome
780
804
 
781
805
 
806
+ def _find_candidate_order(
807
+ candidate: tuple[CacheEntry, NamespaceCacheKey],
808
+ ) -> tuple[bool, int]:
809
+ descriptor = candidate[0].find_source
810
+ assert descriptor is not None
811
+ # Zero denotes unlimited and sorts after every covering positive limit.
812
+ return descriptor.limit == 0, descriptor.limit
813
+
814
+
782
815
  class _CacheCoreLookup(_CacheCoreBase):
783
816
  __slots__ = ()
784
817
 
818
+ def _probe_namespace_entry(
819
+ self,
820
+ state: NamespaceState,
821
+ key: NamespaceCacheKey,
822
+ expected: CacheEntry | None = None,
823
+ ) -> CacheEntry | None:
824
+ entry = self._lru.peek(key)
825
+ if entry is None or (expected is not None and entry is not expected):
826
+ return None
827
+ with self._namespace_section(state):
828
+ valid = (state.generation,) == entry.generation_key
829
+ return entry if valid else None
830
+
831
+ def lookup_find(
832
+ self,
833
+ namespace: NamespaceId,
834
+ shape: FindReadShape,
835
+ *,
836
+ codec_options: CodecOptions[Any] | None = None,
837
+ ) -> LookupResult:
838
+ self._ensure_active()
839
+ if not self._is_database_available(namespace.database):
840
+ self._statistics.record_bypass(BypassReason.STREAM_UNAVAILABLE)
841
+ return LookupResult(hit=False)
842
+ state = self._namespace(namespace)
843
+ family = canonicalize(shape.family)
844
+ key = NamespaceCacheKey(namespace, find_discriminator(family, shape.limit))
845
+ entry = self._probe_namespace_entry(state, key)
846
+ if entry is None and not isinstance(shape.limit, bool) and shape.limit > 0:
847
+ with self._namespace_section(state):
848
+ try:
849
+ candidates = tuple(state.find_families[hash(family)].items())
850
+ except KeyError:
851
+ candidates = ()
852
+ for token, source_key in sorted(candidates, key=_find_candidate_order):
853
+ descriptor = token.find_source
854
+ assert descriptor is not None
855
+ limit = descriptor.limit
856
+ if limit != 0 and limit < shape.limit:
857
+ continue
858
+ if source_key.discriminator != find_discriminator(family, limit):
859
+ continue
860
+ candidate = self._probe_namespace_entry(state, source_key, token)
861
+ if candidate is not None:
862
+ key, entry = source_key, candidate
863
+ break
864
+ if entry is None:
865
+ self._statistics.record_miss()
866
+ return LookupResult(hit=False)
867
+ self._lru.touch(key)
868
+ self._statistics.record_hit()
869
+ documents = cast("list[object]", decode_value(entry.value, codec_options))
870
+ if not isinstance(shape.limit, bool) and shape.limit > 0:
871
+ documents = documents[: shape.limit]
872
+ return LookupResult(hit=True, value=documents)
873
+
785
874
  def lookup_identity(
786
875
  self,
787
876
  namespace: NamespaceId,
@@ -7,7 +7,12 @@ from typing import TYPE_CHECKING
7
7
  if TYPE_CHECKING:
8
8
  from client_query_cache._core.canonical import Canonical
9
9
  from client_query_cache._core.entries import CacheEntry
10
- from client_query_cache._core.keys import AliasKey, CacheKey, NamespaceId
10
+ from client_query_cache._core.keys import (
11
+ AliasKey,
12
+ CacheKey,
13
+ NamespaceCacheKey,
14
+ NamespaceId,
15
+ )
11
16
 
12
17
 
13
18
  @dataclass(slots=True)
@@ -32,4 +37,7 @@ class NamespaceState:
32
37
  identities: dict[Canonical, IdentityState] = field(default_factory=dict)
33
38
  aliases: dict[AliasKey, Canonical] = field(default_factory=dict)
34
39
  entry_index: dict[CacheEntry, CacheKey] = field(default_factory=dict)
40
+ find_families: dict[int, dict[CacheEntry, NamespaceCacheKey]] = field(
41
+ default_factory=dict # Empty without compatible resident sources.
42
+ )
35
43
  identity_generation_watermark: int = 0