client-query-cache 0.1.0__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/PKG-INFO +42 -15
  2. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/README.md +40 -14
  3. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/pyproject.toml +4 -1
  4. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/pyproject.toml.orig +4 -1
  5. client_query_cache-0.2.0/src/client_query_cache/__init__.py +37 -0
  6. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/_core/collection_metadata.py +21 -5
  7. client_query_cache-0.2.0/src/client_query_cache/_core/find_one_reads.py +108 -0
  8. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/_core/locking.py +4 -4
  9. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/_core/manager.py +29 -14
  10. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/_core/projection.py +3 -1
  11. client_query_cache-0.2.0/src/client_query_cache/_core/read_classification.py +77 -0
  12. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/_core/read_validation.py +1 -1
  13. client_query_cache-0.2.0/src/client_query_cache/_core/snapshots.py +112 -0
  14. client_query_cache-0.2.0/src/client_query_cache/_core/stream_health.py +116 -0
  15. client_query_cache-0.2.0/src/client_query_cache/_core/traversal.py +20 -0
  16. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/_core/unique_keys.py +14 -1
  17. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/asynchronous/__init__.py +16 -0
  18. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/asynchronous/collection.py +224 -115
  19. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/asynchronous/database.py +10 -22
  20. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/asynchronous/manager.py +64 -6
  21. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/asynchronous/streams.py +44 -9
  22. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/otel.py +30 -10
  23. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/synchronous/__init__.py +16 -0
  24. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/synchronous/collection.py +224 -105
  25. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/synchronous/database.py +10 -16
  26. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/synchronous/manager.py +61 -6
  27. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/synchronous/streams.py +36 -7
  28. client_query_cache-0.1.0/src/client_query_cache/__init__.py +0 -15
  29. client_query_cache-0.1.0/src/client_query_cache/_core/snapshots.py +0 -67
  30. client_query_cache-0.1.0/src/client_query_cache/_core/stream_health.py +0 -47
  31. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/_core/__init__.py +0 -0
  32. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/_core/canonical.py +0 -0
  33. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/_core/codec.py +0 -0
  34. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/_core/collation.py +0 -0
  35. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/_core/entries.py +0 -0
  36. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/_core/errors.py +0 -0
  37. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/_core/identity_reads.py +0 -0
  38. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/_core/keys.py +0 -0
  39. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/_core/lifecycle.py +0 -0
  40. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/_core/lru.py +0 -0
  41. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/_core/namespace.py +0 -0
  42. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/_core/order_sensitive_keys.py +0 -0
  43. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/_core/stream_cost.py +0 -0
  44. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/_core/stream_events.py +0 -0
  45. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/_core/stream_options.py +0 -0
  46. {client_query_cache-0.1.0 → client_query_cache-0.2.0}/src/client_query_cache/py.typed +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.3
2
2
  Name: client-query-cache
3
- Version: 0.1.0
3
+ Version: 0.2.0
4
4
  Summary: Client-side caching for PyMongo, kept coherent using MongoDB change streams.
5
5
  Author: Alessio Locatelli
6
6
  Author-email: Alessio Locatelli <<software.development@secure.mailbox.org>>
@@ -8,6 +8,7 @@ Requires-Dist: pymongo>=4.18.1
8
8
  Requires-Dist: opentelemetry-api>=1.45.0 ; extra == 'otel'
9
9
  Requires-Python: >=3.14.6
10
10
  Project-URL: Repository, https://github.com/alessio-locatelli/client-query-cache
11
+ Project-URL: Documentation, https://alessio-locatelli.github.io/client-query-cache/
11
12
  Provides-Extra: otel
12
13
  Description-Content-Type: text/markdown
13
14
 
@@ -41,7 +42,7 @@ Or with pip: `pip install client-query-cache`.
41
42
 
42
43
  ## Usage
43
44
 
44
- `CacheManager` wraps a `pymongo.MongoClient` (or `pymongo.AsyncMongoClient`) that you construct and own. Its database and collection facades cache a narrow set of PyMongo's own read methods — `find_one`, `find`, `aggregate`, `count_documents`, `estimated_document_count`, and `distinct` — and keep cached results coherent as the underlying data changes. Every other operation, including all writes, is called directly on the facade the same way you'd call it on the wrapped PyMongo object:
45
+ `CacheManager` wraps a `pymongo.MongoClient` (or `pymongo.AsyncMongoClient`) that you construct and own. Keep using your PyMongo collection for writes and everything else, and ask the manager for a cached view of that collection for six reads — `find_one`, `find`, `aggregate`, `count_documents`, `estimated_document_count`, and `distinct`. The cache keeps those results coherent as the underlying data changes:
45
46
 
46
47
  ```python
47
48
  from pymongo import MongoClient
@@ -50,28 +51,45 @@ from client_query_cache import CacheManager
50
51
 
51
52
  with (
52
53
  MongoClient("mongodb://localhost:27017") as client,
53
- CacheManager(client) as manager,
54
+ CacheManager(client) as cache_manager,
54
55
  ):
55
- collection = manager["my_database"]["my_collection"]
56
+ collection = client["my_database"]["my_collection"]
57
+ cached_collection = cache_manager.cached(collection)
58
+
56
59
  collection.insert_one({"_id": "example", "value": 42})
57
60
 
58
- collection.find_one({"_id": "example"}) # cache miss: reads from MongoDB
59
- collection.find_one({"_id": "example"}) # cache hit: served from the cache
61
+ # Cache miss: reads from MongoDB.
62
+ cached_collection.find_one({"_id": "example"})
63
+ # Cache hit: served from the cache.
64
+ cached_collection.find_one({"_id": "example"})
60
65
 
61
- # Bridge stats like this into OpenTelemetry:
66
+ # Prints 1. Bridge stats like this into OpenTelemetry:
62
67
  # docs/architecture.md#opentelemetry-metrics
63
- print(manager.cache_core.snapshot().hits) # 1
68
+ print(cache_manager.snapshot().hits)
64
69
 
65
70
  collection.create_index("email", unique=True)
66
71
  collection.insert_one({"_id": "user-1", "email": "a@example.com"})
67
- collection.find_one({"email": "a@example.com"}) # also cached, like an `_id` lookup
72
+ # Also cached, like an `_id` lookup.
73
+ cached_collection.find_one({"email": "a@example.com"})
74
+ ```
75
+
76
+ `collection` is PyMongo's own object, so your editor and type checker see PyMongo's real methods and signatures. `cached_collection` has only the six cached reads; `find` and `aggregate` on it return a list instead of a cursor. Use `cached_collection.raw` (the same `collection`) whenever you need PyMongo's own cursor behavior.
77
+
78
+ `find_one` caches deterministic single-document queries, including compound filters, match-all reads and missing results. Use `sort` to choose the first matching document and `collation` to control string matching:
79
+
80
+ ```python
81
+ cached_collection.find_one(
82
+ {"status": "active"},
83
+ sort=[("updated_at", -1)],
84
+ collation={"locale": "en", "strength": 2},
85
+ )
68
86
  ```
69
87
 
70
- `find_one` caches a lookup by `_id` and by any other field the database enforces as unique, discovered automatically from the collection's own indexes — there's nothing to declare. Only a plain unique index qualifies: a partial, sparse, or hashed unique index, or a read whose collation doesn't match the index's collation, falls back to an uncached read instead.
88
+ Exact `_id` lookups and qualifying unique indexes allow cached reads to survive writes to other documents. Other queries are refreshed after any write to the collection. Updates become visible after the manager processes their change-stream events; use the PyMongo collection for reads that must immediately observe a preceding write.
71
89
 
72
90
  `CacheManager` starts a background change-stream task the first time a read touches a database, so close it (or use it as a context manager, as above) alongside the client — closing only the client leaves that background task running against a closed connection.
73
91
 
74
- The same facades are available for `pymongo.AsyncMongoClient` under `client_query_cache.asynchronous`, with the same methods as coroutines:
92
+ The same API is available for `pymongo.AsyncMongoClient` under `client_query_cache.asynchronous`, with the cached reads as coroutines:
75
93
 
76
94
  ```python
77
95
  import asyncio
@@ -84,13 +102,19 @@ from client_query_cache.asynchronous import CacheManager
84
102
  async def main() -> None:
85
103
  async with (
86
104
  AsyncMongoClient("mongodb://localhost:27017") as client,
87
- CacheManager(client) as manager,
105
+ CacheManager(client) as cache_manager,
88
106
  ):
89
- collection = manager["my_database"]["my_collection"]
107
+ collection = client["my_database"]["my_collection"]
108
+ cached_collection = cache_manager.cached(collection)
109
+
90
110
  await collection.insert_one({"_id": "example", "value": 42})
91
111
 
92
- await collection.find_one({"_id": "example"}) # cache miss
93
- await collection.find_one({"_id": "example"}) # cache hit
112
+ # Cache miss.
113
+ await cached_collection.find_one({"_id": "example"})
114
+ # Cache hit.
115
+ await cached_collection.find_one({"_id": "example"})
116
+ # A list, not a cursor.
117
+ await cached_collection.find({"value": 42})
94
118
 
95
119
 
96
120
  asyncio.run(main())
@@ -105,8 +129,11 @@ A read bypasses the cache instead of using it whenever caching it safely isn't p
105
129
 
106
130
  ## Documentation
107
131
 
132
+ Browse the [documentation site](https://alessio-locatelli.github.io/client-query-cache/) for searchable guides to the current `main` branch.
133
+
108
134
  - [API reference](docs/api-reference.md) — the complete public surface: construction, configuration, limits, ownership, and raw fallback.
109
135
  - [Architecture and operations](docs/architecture.md) — system requirements, capacity planning, retry/error handling, observability, security, and recovery behavior.
136
+ - [Examples](examples/README.md) — runnable programs that add the cache to real libraries that store data in MongoDB, such as requests-cache.
110
137
  - [Stream cost benchmarks](docs/stream-cost-benchmarks.md) — whether caching fits your workload, and the controlled benchmark reports backing that guidance.
111
138
 
112
139
  ---
@@ -28,7 +28,7 @@ Or with pip: `pip install client-query-cache`.
28
28
 
29
29
  ## Usage
30
30
 
31
- `CacheManager` wraps a `pymongo.MongoClient` (or `pymongo.AsyncMongoClient`) that you construct and own. Its database and collection facades cache a narrow set of PyMongo's own read methods — `find_one`, `find`, `aggregate`, `count_documents`, `estimated_document_count`, and `distinct` — and keep cached results coherent as the underlying data changes. Every other operation, including all writes, is called directly on the facade the same way you'd call it on the wrapped PyMongo object:
31
+ `CacheManager` wraps a `pymongo.MongoClient` (or `pymongo.AsyncMongoClient`) that you construct and own. Keep using your PyMongo collection for writes and everything else, and ask the manager for a cached view of that collection for six reads — `find_one`, `find`, `aggregate`, `count_documents`, `estimated_document_count`, and `distinct`. The cache keeps those results coherent as the underlying data changes:
32
32
 
33
33
  ```python
34
34
  from pymongo import MongoClient
@@ -37,28 +37,45 @@ from client_query_cache import CacheManager
37
37
 
38
38
  with (
39
39
  MongoClient("mongodb://localhost:27017") as client,
40
- CacheManager(client) as manager,
40
+ CacheManager(client) as cache_manager,
41
41
  ):
42
- collection = manager["my_database"]["my_collection"]
42
+ collection = client["my_database"]["my_collection"]
43
+ cached_collection = cache_manager.cached(collection)
44
+
43
45
  collection.insert_one({"_id": "example", "value": 42})
44
46
 
45
- collection.find_one({"_id": "example"}) # cache miss: reads from MongoDB
46
- collection.find_one({"_id": "example"}) # cache hit: served from the cache
47
+ # Cache miss: reads from MongoDB.
48
+ cached_collection.find_one({"_id": "example"})
49
+ # Cache hit: served from the cache.
50
+ cached_collection.find_one({"_id": "example"})
47
51
 
48
- # Bridge stats like this into OpenTelemetry:
52
+ # Prints 1. Bridge stats like this into OpenTelemetry:
49
53
  # docs/architecture.md#opentelemetry-metrics
50
- print(manager.cache_core.snapshot().hits) # 1
54
+ print(cache_manager.snapshot().hits)
51
55
 
52
56
  collection.create_index("email", unique=True)
53
57
  collection.insert_one({"_id": "user-1", "email": "a@example.com"})
54
- collection.find_one({"email": "a@example.com"}) # also cached, like an `_id` lookup
58
+ # Also cached, like an `_id` lookup.
59
+ cached_collection.find_one({"email": "a@example.com"})
60
+ ```
61
+
62
+ `collection` is PyMongo's own object, so your editor and type checker see PyMongo's real methods and signatures. `cached_collection` has only the six cached reads; `find` and `aggregate` on it return a list instead of a cursor. Use `cached_collection.raw` (the same `collection`) whenever you need PyMongo's own cursor behavior.
63
+
64
+ `find_one` caches deterministic single-document queries, including compound filters, match-all reads and missing results. Use `sort` to choose the first matching document and `collation` to control string matching:
65
+
66
+ ```python
67
+ cached_collection.find_one(
68
+ {"status": "active"},
69
+ sort=[("updated_at", -1)],
70
+ collation={"locale": "en", "strength": 2},
71
+ )
55
72
  ```
56
73
 
57
- `find_one` caches a lookup by `_id` and by any other field the database enforces as unique, discovered automatically from the collection's own indexes — there's nothing to declare. Only a plain unique index qualifies: a partial, sparse, or hashed unique index, or a read whose collation doesn't match the index's collation, falls back to an uncached read instead.
74
+ Exact `_id` lookups and qualifying unique indexes allow cached reads to survive writes to other documents. Other queries are refreshed after any write to the collection. Updates become visible after the manager processes their change-stream events; use the PyMongo collection for reads that must immediately observe a preceding write.
58
75
 
59
76
  `CacheManager` starts a background change-stream task the first time a read touches a database, so close it (or use it as a context manager, as above) alongside the client — closing only the client leaves that background task running against a closed connection.
60
77
 
61
- The same facades are available for `pymongo.AsyncMongoClient` under `client_query_cache.asynchronous`, with the same methods as coroutines:
78
+ The same API is available for `pymongo.AsyncMongoClient` under `client_query_cache.asynchronous`, with the cached reads as coroutines:
62
79
 
63
80
  ```python
64
81
  import asyncio
@@ -71,13 +88,19 @@ from client_query_cache.asynchronous import CacheManager
71
88
  async def main() -> None:
72
89
  async with (
73
90
  AsyncMongoClient("mongodb://localhost:27017") as client,
74
- CacheManager(client) as manager,
91
+ CacheManager(client) as cache_manager,
75
92
  ):
76
- collection = manager["my_database"]["my_collection"]
93
+ collection = client["my_database"]["my_collection"]
94
+ cached_collection = cache_manager.cached(collection)
95
+
77
96
  await collection.insert_one({"_id": "example", "value": 42})
78
97
 
79
- await collection.find_one({"_id": "example"}) # cache miss
80
- await collection.find_one({"_id": "example"}) # cache hit
98
+ # Cache miss.
99
+ await cached_collection.find_one({"_id": "example"})
100
+ # Cache hit.
101
+ await cached_collection.find_one({"_id": "example"})
102
+ # A list, not a cursor.
103
+ await cached_collection.find({"value": 42})
81
104
 
82
105
 
83
106
  asyncio.run(main())
@@ -92,8 +115,11 @@ A read bypasses the cache instead of using it whenever caching it safely isn't p
92
115
 
93
116
  ## Documentation
94
117
 
118
+ Browse the [documentation site](https://alessio-locatelli.github.io/client-query-cache/) for searchable guides to the current `main` branch.
119
+
95
120
  - [API reference](docs/api-reference.md) — the complete public surface: construction, configuration, limits, ownership, and raw fallback.
96
121
  - [Architecture and operations](docs/architecture.md) — system requirements, capacity planning, retry/error handling, observability, security, and recovery behavior.
122
+ - [Examples](examples/README.md) — runnable programs that add the cache to real libraries that store data in MongoDB, such as requests-cache.
97
123
  - [Stream cost benchmarks](docs/stream-cost-benchmarks.md) — whether caching fits your workload, and the controlled benchmark reports backing that guidance.
98
124
 
99
125
  ---
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "client-query-cache"
3
- version = "0.1.0"
3
+ version = "0.2.0"
4
4
  description = "Client-side caching for PyMongo, kept coherent using MongoDB change streams."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.14.6"
@@ -12,6 +12,7 @@ email = "<software.development@secure.mailbox.org>"
12
12
 
13
13
  [project.urls]
14
14
  Repository = "https://github.com/alessio-locatelli/client-query-cache"
15
+ Documentation = "https://alessio-locatelli.github.io/client-query-cache/"
15
16
 
16
17
  [project.optional-dependencies]
17
18
  otel = ["opentelemetry-api>=1.45.0"]
@@ -45,6 +46,7 @@ dev = [
45
46
  "dnspython>=2.8.0",
46
47
  "python-snappy>=0.7.3",
47
48
  ]
49
+ docs = ["zensical>=0.0.67"]
48
50
 
49
51
  [build-system]
50
52
  requires = ["uv_build>=0.12.18,<0.13.0"]
@@ -60,6 +62,7 @@ require-subclass = true
60
62
  exclude-modules = """
61
63
  (
62
64
  ^specifications
65
+ | ^examples
63
66
  )
64
67
  """
65
68
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "client-query-cache"
3
- version = "0.1.0"
3
+ version = "0.2.0"
4
4
  description = "Client-side caching for PyMongo, kept coherent using MongoDB change streams."
5
5
  authors = [
6
6
  { name = "Alessio Locatelli", email = "<software.development@secure.mailbox.org>" },
@@ -11,6 +11,7 @@ dependencies = ["pymongo>=4.18.1"]
11
11
 
12
12
  [project.urls]
13
13
  Repository = "https://github.com/alessio-locatelli/client-query-cache"
14
+ Documentation = "https://alessio-locatelli.github.io/client-query-cache/"
14
15
 
15
16
  [project.optional-dependencies]
16
17
  otel = ["opentelemetry-api>=1.45.0"]
@@ -44,6 +45,7 @@ dev = [
44
45
  "dnspython>=2.8.0",
45
46
  "python-snappy>=0.7.3",
46
47
  ]
48
+ docs = ["zensical>=0.0.67"]
47
49
 
48
50
  [build-system]
49
51
  requires = ["uv_build>=0.12.18,<0.13.0"]
@@ -59,6 +61,7 @@ require-subclass = true
59
61
  exclude-modules = '''
60
62
  (
61
63
  ^specifications
64
+ | ^examples
62
65
  )
63
66
  '''
64
67
 
@@ -0,0 +1,37 @@
1
+ from ._core.errors import (
2
+ CacheClosedError,
3
+ CacheConfigurationError,
4
+ CacheError,
5
+ UnsupportedCacheRequestError,
6
+ )
7
+ from .synchronous import (
8
+ BypassReason,
9
+ BypassReasonCount,
10
+ CacheCore,
11
+ CacheCoreConfig,
12
+ CachedCollection,
13
+ CachedDatabase,
14
+ CacheManager,
15
+ CacheSnapshot,
16
+ StreamCostSnapshot,
17
+ StreamHealthSnapshot,
18
+ StreamHealthStatus,
19
+ )
20
+
21
+ __all__ = [
22
+ "BypassReason",
23
+ "BypassReasonCount",
24
+ "CacheClosedError",
25
+ "CacheConfigurationError",
26
+ "CacheCore",
27
+ "CacheCoreConfig",
28
+ "CacheError",
29
+ "CacheManager",
30
+ "CacheSnapshot",
31
+ "CachedCollection",
32
+ "CachedDatabase",
33
+ "StreamCostSnapshot",
34
+ "StreamHealthSnapshot",
35
+ "StreamHealthStatus",
36
+ "UnsupportedCacheRequestError",
37
+ ]
@@ -5,6 +5,7 @@ from dataclasses import dataclass
5
5
  from typing import TYPE_CHECKING, Any
6
6
 
7
7
  from client_query_cache._core.collation import normalize_collation
8
+ from client_query_cache._core.snapshots import BypassReason
8
9
 
9
10
  if TYPE_CHECKING:
10
11
  from collections.abc import Mapping
@@ -12,10 +13,17 @@ if TYPE_CHECKING:
12
13
  from client_query_cache._core.keys import NamespaceId
13
14
 
14
15
 
16
+ _COLLECTION_REASONS = {
17
+ "collection": None,
18
+ "view": BypassReason.VIEW_COLLECTION,
19
+ "timeseries": BypassReason.TIME_SERIES_COLLECTION,
20
+ }
21
+
22
+
15
23
  @dataclass(frozen=True, slots=True)
16
24
  class CollectionMetadata:
17
25
  checked_epoch: int
18
- is_cacheable: bool
26
+ bypass_reason: BypassReason | None
19
27
  default_collation: Mapping[str, Any] | None
20
28
 
21
29
 
@@ -40,16 +48,24 @@ class CollectionMetadataCache:
40
48
 
41
49
  @dataclass(frozen=True, slots=True)
42
50
  class CollectionProbeResult:
43
- is_cacheable: bool
51
+ bypass_reason: BypassReason | None
44
52
  default_collation: Mapping[str, Any] | None
45
53
 
54
+ @property
55
+ def is_cacheable(self) -> bool:
56
+ return self.bypass_reason is None
57
+
46
58
 
47
59
  def interpret_list_collections_entry(
48
60
  entry: Mapping[str, Any] | None,
49
- ) -> CollectionProbeResult | None:
61
+ ) -> CollectionProbeResult:
50
62
  if entry is None:
51
- return None
63
+ return CollectionProbeResult(BypassReason.MISSING_COLLECTION, None)
52
64
  collection_type = entry["type"]
65
+ try:
66
+ bypass_reason = _COLLECTION_REASONS[collection_type]
67
+ except KeyError:
68
+ bypass_reason = BypassReason.METADATA_UNAVAILABLE
53
69
  try:
54
70
  options = entry["options"]
55
71
  except KeyError:
@@ -60,6 +76,6 @@ def interpret_list_collections_entry(
60
76
  collation = None
61
77
  default_collation = normalize_collation(collation)
62
78
  return CollectionProbeResult(
63
- is_cacheable=collection_type == "collection",
79
+ bypass_reason=bypass_reason,
64
80
  default_collation=default_collation,
65
81
  )
@@ -0,0 +1,108 @@
1
+ from __future__ import annotations
2
+
3
+ from collections.abc import Mapping, Sequence
4
+ from typing import Any
5
+
6
+ from pymongo.collation import Collation
7
+
8
+ from client_query_cache._core.collation import normalize_collation
9
+ from client_query_cache._core.order_sensitive_keys import (
10
+ order_sensitive_discriminator_key,
11
+ )
12
+
13
+ type CollationInput = Collation | Mapping[str, Any]
14
+
15
+ _COLLATION_FIELDS = frozenset(
16
+ {
17
+ "locale",
18
+ "caseLevel",
19
+ "caseFirst",
20
+ "strength",
21
+ "numericOrdering",
22
+ "alternate",
23
+ "maxVariable",
24
+ "normalization",
25
+ "backwards",
26
+ }
27
+ )
28
+
29
+
30
+ def normalize_find_one_filter(filter_query: object) -> Mapping[str, Any]:
31
+ if filter_query is None:
32
+ return {}
33
+ if isinstance(filter_query, Mapping):
34
+ return filter_query
35
+ return {"_id": filter_query}
36
+
37
+
38
+ def find_one_options_cacheable(
39
+ sort: Sequence[tuple[str, int]] | None,
40
+ collation: CollationInput | None,
41
+ ) -> bool:
42
+ if sort is not None and (
43
+ not isinstance(sort, (list, tuple))
44
+ or any(
45
+ not isinstance(pair, (list, tuple))
46
+ or len(pair) != 2
47
+ or not isinstance(pair[0], str)
48
+ or type(pair[1]) is not int
49
+ or pair[1] not in {-1, 1}
50
+ for pair in sort
51
+ )
52
+ ):
53
+ return False
54
+ if collation is None:
55
+ return True
56
+ document = collation.document if isinstance(collation, Collation) else collation
57
+ if not isinstance(document, dict) or not document.keys() <= _COLLATION_FIELDS:
58
+ return False
59
+ try:
60
+ validated = Collation(**document).document
61
+ except TypeError:
62
+ return False
63
+ except ValueError:
64
+ return False
65
+ if not validated["locale"]:
66
+ return False
67
+ if validated["locale"] == "simple":
68
+ return len(validated) == 1
69
+ for field, choices in (
70
+ ("strength", (1, 2, 3, 4, 5)),
71
+ ("caseFirst", ("upper", "lower", "off")),
72
+ ("alternate", ("non-ignorable", "shifted")),
73
+ ("maxVariable", ("punct", "space")),
74
+ ):
75
+ if field in validated and validated[field] not in choices:
76
+ return False
77
+ return True
78
+
79
+
80
+ def effective_find_one_collation(
81
+ collation: CollationInput | None,
82
+ default_collation: Mapping[str, Any] | None,
83
+ ) -> Mapping[str, Any] | None:
84
+ if collation is None:
85
+ return default_collation
86
+ document = collation.document if isinstance(collation, Collation) else collation
87
+ return normalize_collation(document)
88
+
89
+
90
+ def find_one_read_shape(
91
+ projection: Mapping[str, Any] | Sequence[str] | None,
92
+ sort: Sequence[tuple[str, int]] | None,
93
+ effective_collation: Mapping[str, Any] | None,
94
+ codec_identity: object,
95
+ ) -> object:
96
+ return order_sensitive_discriminator_key(
97
+ ("find_one", projection, sort, effective_collation, codec_identity)
98
+ )
99
+
100
+
101
+ def generic_find_one_discriminator(
102
+ filter_query: object,
103
+ read_shape: object,
104
+ index_generation: int, # Can be zero.
105
+ ) -> object:
106
+ return order_sensitive_discriminator_key(
107
+ ("find_one_generic", filter_query, read_shape, index_generation)
108
+ )
@@ -5,7 +5,7 @@ from contextlib import contextmanager
5
5
  from typing import TYPE_CHECKING
6
6
 
7
7
  if TYPE_CHECKING:
8
- from collections.abc import Iterator
8
+ from collections.abc import Generator
9
9
 
10
10
 
11
11
  class LockOrderViolationError(RuntimeError):
@@ -19,17 +19,17 @@ class LockOrderGuard:
19
19
  self._state = threading.local()
20
20
 
21
21
  @contextmanager
22
- def namespace_section(self) -> Iterator[None]:
22
+ def namespace_section(self) -> Generator[None]:
23
23
  with self._section(entering="in_namespace", forbidden="in_lru"):
24
24
  yield
25
25
 
26
26
  @contextmanager
27
- def lru_section(self) -> Iterator[None]:
27
+ def lru_section(self) -> Generator[None]:
28
28
  with self._section(entering="in_lru", forbidden="in_namespace"):
29
29
  yield
30
30
 
31
31
  @contextmanager
32
- def _section(self, *, entering: str, forbidden: str) -> Iterator[None]:
32
+ def _section(self, *, entering: str, forbidden: str) -> Generator[None]:
33
33
  if getattr(self._state, forbidden, False):
34
34
  message = f"cannot enter {entering!r} section while holding {forbidden!r}"
35
35
  raise LockOrderViolationError(message)
@@ -26,7 +26,11 @@ from client_query_cache._core.locking import LockOrderGuard
26
26
  from client_query_cache._core.lru import WeightedLru
27
27
  from client_query_cache._core.namespace import IdentityState, NamespaceState
28
28
  from client_query_cache._core.order_sensitive_keys import order_sensitive_key
29
- from client_query_cache._core.snapshots import CacheSnapshot, CacheStatistics
29
+ from client_query_cache._core.snapshots import (
30
+ BypassReason,
31
+ CacheSnapshot,
32
+ CacheStatistics,
33
+ )
30
34
  from client_query_cache._core.stream_cost import (
31
35
  DEFAULT_LAG_CAPTURE_WINDOW_CONFIG,
32
36
  LagCaptureWindowConfig,
@@ -35,7 +39,7 @@ from client_query_cache._core.stream_cost import (
35
39
  )
36
40
 
37
41
  if TYPE_CHECKING:
38
- from collections.abc import Callable, Iterator
42
+ from collections.abc import Callable, Generator
39
43
  from typing import Any
40
44
 
41
45
  from bson.codec_options import CodecOptions
@@ -183,7 +187,7 @@ class _CacheCoreBase:
183
187
  namespace: NamespaceId,
184
188
  availability_generation: int,
185
189
  weight: int,
186
- ) -> Iterator[AdmissionOutcome | None]:
190
+ ) -> Generator[AdmissionOutcome | None]:
187
191
  with self._availability_lock:
188
192
  try:
189
193
  available, current_generation = self._database_availability[
@@ -192,7 +196,11 @@ class _CacheCoreBase:
192
196
  except KeyError:
193
197
  available, current_generation = _DEFAULT_DATABASE_AVAILABILITY
194
198
  if not available or availability_generation != current_generation:
195
- self._statistics.record_bypass()
199
+ self._statistics.record_bypass(
200
+ BypassReason.STREAM_UNAVAILABLE
201
+ if not available
202
+ else BypassReason.ADMISSION_INVALIDATED
203
+ )
196
204
  yield AdmissionOutcome.DECLINED_UNAVAILABLE
197
205
  return
198
206
  if self._lru.is_oversize(weight):
@@ -220,7 +228,7 @@ class _CacheCoreBase:
220
228
  return state
221
229
 
222
230
  @contextmanager
223
- def _namespace_section(self, state: NamespaceState) -> Iterator[None]:
231
+ def _namespace_section(self, state: NamespaceState) -> Generator[None]:
224
232
  with self._guard.namespace_section(), state.lock:
225
233
  yield
226
234
 
@@ -293,12 +301,12 @@ class _CacheCoreLifecycle(_CacheCoreBase):
293
301
  self._database_namespaces.clear()
294
302
  logger.info("cache manager closed")
295
303
 
296
- def record_bypass(self) -> None:
297
- self._statistics.record_bypass()
304
+ def record_bypass(self, reason: BypassReason = BypassReason.UNSPECIFIED) -> None:
305
+ self._statistics.record_bypass(reason)
298
306
 
299
307
  def snapshot(self) -> CacheSnapshot:
300
308
  used_bytes, entry_count = self._lru.snapshot_usage()
301
- hits, misses, evictions, bypasses, oversized_bypasses = (
309
+ hits, misses, evictions, bypasses, oversized_bypasses, bypass_reasons = (
302
310
  self._statistics.snapshot()
303
311
  )
304
312
  return CacheSnapshot(
@@ -312,6 +320,7 @@ class _CacheCoreLifecycle(_CacheCoreBase):
312
320
  evictions=evictions,
313
321
  bypasses=bypasses,
314
322
  oversized_bypasses=oversized_bypasses,
323
+ bypass_reasons=bypass_reasons,
315
324
  )
316
325
 
317
326
 
@@ -466,7 +475,9 @@ class _CacheCoreIdentityAdmission(_CacheCoreBase):
466
475
  canonical_shape = canonicalize(read_shape)
467
476
  try:
468
477
  encoded = encode_value(value, codec_options)
469
- except BSONError, OverflowError:
478
+ except BSONError:
479
+ return AdmissionOutcome.DECLINED_UNENCODABLE
480
+ except OverflowError:
470
481
  return AdmissionOutcome.DECLINED_UNENCODABLE
471
482
  weight = len(encoded)
472
483
  with self._admission_section(
@@ -589,7 +600,9 @@ class _CacheCoreNamespaceAdmission(_CacheCoreBase):
589
600
  canonical_discriminator = canonicalize(discriminator)
590
601
  try:
591
602
  encoded = encode_value(value, codec_options)
592
- except BSONError, OverflowError:
603
+ except BSONError:
604
+ return AdmissionOutcome.DECLINED_UNENCODABLE
605
+ except OverflowError:
593
606
  return AdmissionOutcome.DECLINED_UNENCODABLE
594
607
  weight = len(encoded)
595
608
  with self._admission_section(
@@ -667,7 +680,9 @@ class _CacheCoreUniqueKeyAdmission(_CacheCoreBase):
667
680
  canonical_shape = canonicalize(read_shape)
668
681
  try:
669
682
  encoded = encode_value(value, codec_options)
670
- except BSONError, OverflowError:
683
+ except BSONError:
684
+ return AdmissionOutcome.DECLINED_UNENCODABLE
685
+ except OverflowError:
671
686
  return AdmissionOutcome.DECLINED_UNENCODABLE
672
687
  weight = len(encoded)
673
688
  with self._admission_section(
@@ -777,7 +792,7 @@ class _CacheCoreLookup(_CacheCoreBase):
777
792
  ) -> LookupResult:
778
793
  self._ensure_active()
779
794
  if not self._is_database_available(namespace.database):
780
- self._statistics.record_bypass()
795
+ self._statistics.record_bypass(BypassReason.STREAM_UNAVAILABLE)
781
796
  return LookupResult(hit=False)
782
797
  canonical_identity = canonicalize(order_sensitive_key(identity))
783
798
  key = IdentityCacheKey(namespace, canonical_identity, canonicalize(read_shape))
@@ -807,7 +822,7 @@ class _CacheCoreLookup(_CacheCoreBase):
807
822
  ) -> LookupResult:
808
823
  self._ensure_active()
809
824
  if not self._is_database_available(namespace.database):
810
- self._statistics.record_bypass()
825
+ self._statistics.record_bypass(BypassReason.STREAM_UNAVAILABLE)
811
826
  return LookupResult(hit=False)
812
827
  canonical_discriminator = canonicalize(discriminator)
813
828
  key = NamespaceCacheKey(namespace, canonical_discriminator)
@@ -834,7 +849,7 @@ class _CacheCoreLookup(_CacheCoreBase):
834
849
  ) -> Canonical | None:
835
850
  self._ensure_active()
836
851
  if not self._is_database_available(namespace.database):
837
- self._statistics.record_bypass()
852
+ self._statistics.record_bypass(BypassReason.STREAM_UNAVAILABLE)
838
853
  return None
839
854
  alias_key = canonical_alias_key(definition, value, collation)
840
855
  state = self._namespace(namespace)
@@ -29,7 +29,9 @@ def without_id(
29
29
  bson.encode(stripped, codec_options=codec_options),
30
30
  codec_options=codec_options,
31
31
  )
32
- except BSONError, OverflowError:
32
+ except BSONError:
33
+ return stripped
34
+ except OverflowError:
33
35
  return stripped
34
36
 
35
37