embedflow 0.7.0__tar.gz → 0.8.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (177) hide show
  1. {embedflow-0.7.0 → embedflow-0.8.0}/CHANGELOG.md +12 -0
  2. {embedflow-0.7.0 → embedflow-0.8.0}/CITATION.cff +2 -2
  3. {embedflow-0.7.0 → embedflow-0.8.0}/PKG-INFO +13 -2
  4. {embedflow-0.7.0 → embedflow-0.8.0}/README.md +12 -1
  5. {embedflow-0.7.0 → embedflow-0.8.0}/README_PYPI.md +12 -1
  6. {embedflow-0.7.0 → embedflow-0.8.0}/docs/cli.md +10 -1
  7. {embedflow-0.7.0 → embedflow-0.8.0}/docs/configuration.md +12 -0
  8. {embedflow-0.7.0 → embedflow-0.8.0}/docs/limitations.md +1 -1
  9. embedflow-0.8.0/docs/prewarming.md +84 -0
  10. {embedflow-0.7.0 → embedflow-0.8.0}/docs/releasing.md +6 -6
  11. embedflow-0.8.0/embedflow/__init__.py +59 -0
  12. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/cache/base.py +3 -3
  13. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/cache/persistent_cache.py +133 -18
  14. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/cli.py +381 -8
  15. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/config.py +81 -0
  16. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/migration/materializer.py +59 -4
  17. embedflow-0.8.0/embedflow/prewarm/__init__.py +23 -0
  18. embedflow-0.8.0/embedflow/prewarm/models.py +181 -0
  19. embedflow-0.8.0/embedflow/prewarm/planner.py +479 -0
  20. embedflow-0.8.0/embedflow/prewarm/runner.py +450 -0
  21. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/serving/engine.py +72 -7
  22. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/shadow/models.py +5 -0
  23. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/shadow/runner.py +28 -0
  24. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/shadow/telemetry.py +306 -9
  25. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow.egg-info/PKG-INFO +13 -2
  26. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow.egg-info/SOURCES.txt +9 -0
  27. embedflow-0.8.0/examples/prewarm/README.md +16 -0
  28. embedflow-0.8.0/examples/prewarm/demo.py +74 -0
  29. embedflow-0.8.0/examples/prewarm/embedflow.yaml.example +24 -0
  30. embedflow-0.8.0/examples/prewarm/run_demo.sh +12 -0
  31. {embedflow-0.7.0 → embedflow-0.8.0}/pyproject.toml +1 -1
  32. {embedflow-0.7.0 → embedflow-0.8.0}/scripts/release_gate.py +2 -2
  33. embedflow-0.7.0/embedflow/__init__.py +0 -31
  34. {embedflow-0.7.0 → embedflow-0.8.0}/CONTRIBUTING.md +0 -0
  35. {embedflow-0.7.0 → embedflow-0.8.0}/LICENSE +0 -0
  36. {embedflow-0.7.0 → embedflow-0.8.0}/MANIFEST.in +0 -0
  37. {embedflow-0.7.0 → embedflow-0.8.0}/SECURITY.md +0 -0
  38. {embedflow-0.7.0 → embedflow-0.8.0}/docs/api.md +0 -0
  39. {embedflow-0.7.0 → embedflow-0.8.0}/docs/assets/README.md +0 -0
  40. {embedflow-0.7.0 → embedflow-0.8.0}/docs/assets/candidate-gap-example.svg +0 -0
  41. {embedflow-0.7.0 → embedflow-0.8.0}/docs/assets/dashboard-screenshot.md +0 -0
  42. {embedflow-0.7.0 → embedflow-0.8.0}/docs/assets/terminal-demo.txt +0 -0
  43. {embedflow-0.7.0 → embedflow-0.8.0}/docs/concepts.md +0 -0
  44. {embedflow-0.7.0 → embedflow-0.8.0}/docs/contributing-benchmarks.md +0 -0
  45. {embedflow-0.7.0 → embedflow-0.8.0}/docs/economics.md +0 -0
  46. {embedflow-0.7.0 → embedflow-0.8.0}/docs/installation.md +0 -0
  47. {embedflow-0.7.0 → embedflow-0.8.0}/docs/integrations/faiss.md +0 -0
  48. {embedflow-0.7.0 → embedflow-0.8.0}/docs/integrations/milvus.md +0 -0
  49. {embedflow-0.7.0 → embedflow-0.8.0}/docs/integrations/pgvector.md +0 -0
  50. {embedflow-0.7.0 → embedflow-0.8.0}/docs/integrations/pinecone.md +0 -0
  51. {embedflow-0.7.0 → embedflow-0.8.0}/docs/integrations/qdrant.md +0 -0
  52. {embedflow-0.7.0 → embedflow-0.8.0}/docs/integrations/weaviate.md +0 -0
  53. {embedflow-0.7.0 → embedflow-0.8.0}/docs/methodology.md +0 -0
  54. {embedflow-0.7.0 → embedflow-0.8.0}/docs/planner.md +0 -0
  55. {embedflow-0.7.0 → embedflow-0.8.0}/docs/quickstart.md +0 -0
  56. {embedflow-0.7.0 → embedflow-0.8.0}/docs/registry.md +0 -0
  57. {embedflow-0.7.0 → embedflow-0.8.0}/docs/shadow-mode.md +0 -0
  58. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/__main__.py +0 -0
  59. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/analysis.py +0 -0
  60. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/api.py +0 -0
  61. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/cache/__init__.py +0 -0
  62. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/compatibility/__init__.py +0 -0
  63. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/compatibility/candidate_gap.py +0 -0
  64. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/compatibility/containment.py +0 -0
  65. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/compatibility/evaluate.py +0 -0
  66. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/compatibility/metrics.py +0 -0
  67. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/compatibility/migration_depth.py +0 -0
  68. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/compatibility/probe.py +0 -0
  69. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/compatibility/report.py +0 -0
  70. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/compatibility/t2.py +0 -0
  71. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/data/__init__.py +0 -0
  72. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/data/registry/__init__.py +0 -0
  73. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/data/registry/benchmark_profiles.jsonl +0 -0
  74. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/data/registry/checksums.sha256 +0 -0
  75. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/data/registry/migrations.jsonl +0 -0
  76. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/data/registry/registry_manifest.json +0 -0
  77. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/data/registry/research_summaries.json +0 -0
  78. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/data/registry/schema_version.json +0 -0
  79. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/frozen/T2_V1_FROZEN_SPEC.md +0 -0
  80. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/frozen/T2_V1_FROZEN_SPEC.sha256 +0 -0
  81. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/indexes/__init__.py +0 -0
  82. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/indexes/base.py +0 -0
  83. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/indexes/faiss_backend.py +0 -0
  84. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/indexes/milvus_backend.py +0 -0
  85. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/indexes/pgvector_backend.py +0 -0
  86. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/indexes/pinecone_backend.py +0 -0
  87. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/indexes/qdrant_backend.py +0 -0
  88. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/indexes/weaviate_backend.py +0 -0
  89. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/metrics/__init__.py +0 -0
  90. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/metrics/latency.py +0 -0
  91. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/migration/__init__.py +0 -0
  92. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/migration/compatibility.py +0 -0
  93. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/migration/facade.py +0 -0
  94. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/migration/planner.py +0 -0
  95. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/migration/state.py +0 -0
  96. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/models/__init__.py +0 -0
  97. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/models/base.py +0 -0
  98. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/models/huggingface.py +0 -0
  99. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/planner/__init__.py +0 -0
  100. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/planner/economics.py +0 -0
  101. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/planner/models.py +0 -0
  102. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/planner/planner.py +0 -0
  103. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/planner/rendering.py +0 -0
  104. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/registry/__init__.py +0 -0
  105. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/registry/loader.py +0 -0
  106. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/registry/matcher.py +0 -0
  107. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/registry/schema.py +0 -0
  108. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/runtime.py +0 -0
  109. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/serving/__init__.py +0 -0
  110. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/serving/api.py +0 -0
  111. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/serving/factory.py +0 -0
  112. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/serving/schemas.py +0 -0
  113. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/shadow/__init__.py +0 -0
  114. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/shadow/report.py +0 -0
  115. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow.egg-info/dependency_links.txt +0 -0
  116. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow.egg-info/entry_points.txt +0 -0
  117. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow.egg-info/requires.txt +0 -0
  118. {embedflow-0.7.0 → embedflow-0.8.0}/embedflow.egg-info/top_level.txt +0 -0
  119. {embedflow-0.7.0 → embedflow-0.8.0}/examples/faiss/README.md +0 -0
  120. {embedflow-0.7.0 → embedflow-0.8.0}/examples/faiss/documents.jsonl +0 -0
  121. {embedflow-0.7.0 → embedflow-0.8.0}/examples/faiss/embedflow.yaml +0 -0
  122. {embedflow-0.7.0 → embedflow-0.8.0}/examples/faiss/queries.jsonl +0 -0
  123. {embedflow-0.7.0 → embedflow-0.8.0}/examples/milvus/README.md +0 -0
  124. {embedflow-0.7.0 → embedflow-0.8.0}/examples/milvus/compose.yaml +0 -0
  125. {embedflow-0.7.0 → embedflow-0.8.0}/examples/milvus/embedflow.yaml.example +0 -0
  126. {embedflow-0.7.0 → embedflow-0.8.0}/examples/milvus/run_demo.sh +0 -0
  127. {embedflow-0.7.0 → embedflow-0.8.0}/examples/pgvector/README.md +0 -0
  128. {embedflow-0.7.0 → embedflow-0.8.0}/examples/pgvector/build_index.py +0 -0
  129. {embedflow-0.7.0 → embedflow-0.8.0}/examples/pgvector/compose.yaml +0 -0
  130. {embedflow-0.7.0 → embedflow-0.8.0}/examples/pgvector/embedflow.yaml +0 -0
  131. {embedflow-0.7.0 → embedflow-0.8.0}/examples/pgvector/init.sql +0 -0
  132. {embedflow-0.7.0 → embedflow-0.8.0}/examples/pgvector/queries.jsonl +0 -0
  133. {embedflow-0.7.0 → embedflow-0.8.0}/examples/pgvector/run_demo.sh +0 -0
  134. {embedflow-0.7.0 → embedflow-0.8.0}/examples/pinecone/README.md +0 -0
  135. {embedflow-0.7.0 → embedflow-0.8.0}/examples/pinecone/embedflow.yaml.example +0 -0
  136. {embedflow-0.7.0 → embedflow-0.8.0}/examples/pinecone/run_smoke.sh +0 -0
  137. {embedflow-0.7.0 → embedflow-0.8.0}/examples/planner/README.md +0 -0
  138. {embedflow-0.7.0 → embedflow-0.8.0}/examples/planner/run_demo.sh +0 -0
  139. {embedflow-0.7.0 → embedflow-0.8.0}/examples/qdrant/README.md +0 -0
  140. {embedflow-0.7.0 → embedflow-0.8.0}/examples/qdrant/build_index.py +0 -0
  141. {embedflow-0.7.0 → embedflow-0.8.0}/examples/qdrant/documents.jsonl +0 -0
  142. {embedflow-0.7.0 → embedflow-0.8.0}/examples/qdrant/embedflow.yaml +0 -0
  143. {embedflow-0.7.0 → embedflow-0.8.0}/examples/qdrant/queries.jsonl +0 -0
  144. {embedflow-0.7.0 → embedflow-0.8.0}/examples/research_analysis/README.md +0 -0
  145. {embedflow-0.7.0 → embedflow-0.8.0}/examples/research_analysis/documents.jsonl +0 -0
  146. {embedflow-0.7.0 → embedflow-0.8.0}/examples/research_analysis/embedflow.yaml +0 -0
  147. {embedflow-0.7.0 → embedflow-0.8.0}/examples/research_analysis/qrels.json +0 -0
  148. {embedflow-0.7.0 → embedflow-0.8.0}/examples/research_analysis/queries.jsonl +0 -0
  149. {embedflow-0.7.0 → embedflow-0.8.0}/examples/shadow/README.md +0 -0
  150. {embedflow-0.7.0 → embedflow-0.8.0}/examples/shadow/embedflow.yaml.example +0 -0
  151. {embedflow-0.7.0 → embedflow-0.8.0}/examples/shadow/run_demo.py +0 -0
  152. {embedflow-0.7.0 → embedflow-0.8.0}/examples/shadow/run_demo.sh +0 -0
  153. {embedflow-0.7.0 → embedflow-0.8.0}/examples/weaviate/README.md +0 -0
  154. {embedflow-0.7.0 → embedflow-0.8.0}/examples/weaviate/compose.yaml +0 -0
  155. {embedflow-0.7.0 → embedflow-0.8.0}/examples/weaviate/embedflow.yaml.example +0 -0
  156. {embedflow-0.7.0 → embedflow-0.8.0}/examples/weaviate/run_demo.sh +0 -0
  157. {embedflow-0.7.0 → embedflow-0.8.0}/frozen/T2_V1_FROZEN_SPEC.md +0 -0
  158. {embedflow-0.7.0 → embedflow-0.8.0}/frozen/T2_V1_FROZEN_SPEC.sha256 +0 -0
  159. {embedflow-0.7.0 → embedflow-0.8.0}/requirements-dev.txt +0 -0
  160. {embedflow-0.7.0 → embedflow-0.8.0}/requirements.txt +0 -0
  161. {embedflow-0.7.0 → embedflow-0.8.0}/scripts/milvus_fixture.py +0 -0
  162. {embedflow-0.7.0 → embedflow-0.8.0}/scripts/pinecone_smoke.py +0 -0
  163. {embedflow-0.7.0 → embedflow-0.8.0}/scripts/real_qdrant_smoke.py +0 -0
  164. {embedflow-0.7.0 → embedflow-0.8.0}/scripts/run_demo.sh +0 -0
  165. {embedflow-0.7.0 → embedflow-0.8.0}/scripts/run_tests.sh +0 -0
  166. {embedflow-0.7.0 → embedflow-0.8.0}/scripts/validate_milvus.py +0 -0
  167. {embedflow-0.7.0 → embedflow-0.8.0}/scripts/validate_pgvector_10k.py +0 -0
  168. {embedflow-0.7.0 → embedflow-0.8.0}/scripts/validate_weaviate.py +0 -0
  169. {embedflow-0.7.0 → embedflow-0.8.0}/scripts/weaviate_fixture.py +0 -0
  170. {embedflow-0.7.0 → embedflow-0.8.0}/scripts/weaviate_smoke.py +0 -0
  171. {embedflow-0.7.0 → embedflow-0.8.0}/setup.cfg +0 -0
  172. {embedflow-0.7.0 → embedflow-0.8.0}/src/__init__.py +0 -0
  173. {embedflow-0.7.0 → embedflow-0.8.0}/src/embed.py +0 -0
  174. {embedflow-0.7.0 → embedflow-0.8.0}/src/probe_features.py +0 -0
  175. {embedflow-0.7.0 → embedflow-0.8.0}/src/storage.py +0 -0
  176. {embedflow-0.7.0 → embedflow-0.8.0}/src/t2_v1.py +0 -0
  177. {embedflow-0.7.0 → embedflow-0.8.0}/src/utils.py +0 -0
@@ -1,5 +1,17 @@
1
1
  # Changelog
2
2
 
3
+ ## v0.8.0 — Traffic-aware target-vector prewarming
4
+
5
+ - Added bounded `prewarm plan`, `prewarm run`, and `prewarm status` workflows
6
+ that select uncached target documents by observed Shadow Mode candidate
7
+ occurrences.
8
+ - Added privacy-safe aggregate document popularity telemetry, content/model
9
+ fingerprint checks, deterministic tie-breaking, hard document/storage/time
10
+ budgets, atomic plan/run artifacts, and resumable materialization through the
11
+ existing target cache and worker.
12
+ - Observed candidate-occurrence coverage remains an operational warming signal;
13
+ it is not retrieval quality or recall evidence.
14
+
3
15
  ## v0.7.0 — Source-authoritative Shadow Mode
4
16
 
5
17
  - Added bounded, deterministic, source-authoritative Shadow Mode for observing
@@ -2,8 +2,8 @@ cff-version: 1.2.0
2
2
  title: "EmbedFlow: Upgrading Legacy Embeddings Without Full Upfront Re-Embedding"
3
3
  message: "If EmbedFlow contributes to your work, please cite this software release."
4
4
  type: software
5
- version: 0.7.0
6
- date-released: 2026-09-13
5
+ version: 0.8.0
6
+ date-released: 2026-09-15
7
7
  repository-code: "https://github.com/arnsri33/embedflow"
8
8
  url: "https://github.com/arnsri33/embedflow"
9
9
  license: AGPL-3.0-only
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: embedflow
3
- Version: 0.7.0
3
+ Version: 0.8.0
4
4
  Summary: Progressive embedding-model migration over existing vector indexes.
5
5
  Author: Arnav Srivastav
6
6
  License-Expression: AGPL-3.0-only
@@ -224,6 +224,17 @@ Reports contain operational and ranking-disagreement diagnostics—not qrel-base
224
224
  quality guarantees—and recommendations never route canary traffic. See the
225
225
  [Shadow Mode guide](https://github.com/arnsri33/embedflow/blob/main/docs/shadow-mode.md).
226
226
 
227
+ After collecting Shadow traffic, prioritize uncached target vectors by observed
228
+ candidate popularity:
229
+
230
+ ```bash
231
+ embedflow prewarm plan --config ./embedflow.yaml --since 24h --max-docs 50000 --output prewarm-plan.json
232
+ embedflow prewarm run --config ./embedflow.yaml --plan prewarm-plan.json
233
+ ```
234
+
235
+ Observed candidate-occurrence coverage is an operational warming signal, not
236
+ retrieval quality or recall. See the [prewarming guide](https://github.com/arnsri33/embedflow/blob/main/docs/prewarming.md).
237
+
227
238
  ## Research
228
239
 
229
240
  For candidate depth `K`, EmbedFlow measures:
@@ -278,7 +289,7 @@ cover the remaining commands and endpoints.
278
289
 
279
290
  ## Status
280
291
 
281
- EmbedFlow v0.7.0 is a pre-1.0 release for research and early real-world
292
+ EmbedFlow v0.8.0 is a pre-1.0 release for research and early real-world
282
293
  testing. T2-v1 is an empirical finite-tail diagnostic, partial rankings can
283
294
  differ from fully warm target reranking, and ANN fidelity needs a reference
284
295
  comparison to audit.
@@ -174,6 +174,17 @@ coverage, latency, and ranking-disagreement diagnostics; it does not claim
174
174
  retrieval-quality preservation without qrels and never routes canary traffic.
175
175
  See [`docs/shadow-mode.md`](https://github.com/arnsri33/embedflow/blob/main/docs/shadow-mode.md).
176
176
 
177
+ After collecting Shadow traffic, prioritize uncached target vectors by observed
178
+ candidate popularity:
179
+
180
+ ```bash
181
+ embedflow prewarm plan --config ./embedflow.yaml --since 24h --max-docs 50000 --output prewarm-plan.json
182
+ embedflow prewarm run --config ./embedflow.yaml --plan prewarm-plan.json
183
+ ```
184
+
185
+ Observed candidate-occurrence coverage is an operational warming signal, not
186
+ retrieval quality or recall. See [`docs/prewarming.md`](https://github.com/arnsri33/embedflow/blob/main/docs/prewarming.md).
187
+
177
188
  Search responses expose `COLD`, `PARTIAL`, or `WARM`, cache hits and misses,
178
189
  synchronous work, queued work, and stage timings. Once the candidate vectors
179
190
  are warm, target scoring over that candidate set is deterministic.
@@ -288,7 +299,7 @@ OpenAPI documentation; see
288
299
 
289
300
  ## Status
290
301
 
291
- EmbedFlow v0.7.0 is a pre-1.0 release for research and early real-world
302
+ EmbedFlow v0.8.0 is a pre-1.0 release for research and early real-world
292
303
  testing.
293
304
 
294
305
  - T2-v1 reports an empirical finite-tail diagnostic.
@@ -158,6 +158,17 @@ Reports contain operational and ranking-disagreement diagnostics—not qrel-base
158
158
  quality guarantees—and recommendations never route canary traffic. See the
159
159
  [Shadow Mode guide](https://github.com/arnsri33/embedflow/blob/main/docs/shadow-mode.md).
160
160
 
161
+ After collecting Shadow traffic, prioritize uncached target vectors by observed
162
+ candidate popularity:
163
+
164
+ ```bash
165
+ embedflow prewarm plan --config ./embedflow.yaml --since 24h --max-docs 50000 --output prewarm-plan.json
166
+ embedflow prewarm run --config ./embedflow.yaml --plan prewarm-plan.json
167
+ ```
168
+
169
+ Observed candidate-occurrence coverage is an operational warming signal, not
170
+ retrieval quality or recall. See the [prewarming guide](https://github.com/arnsri33/embedflow/blob/main/docs/prewarming.md).
171
+
161
172
  ## Research
162
173
 
163
174
  For candidate depth `K`, EmbedFlow measures:
@@ -212,7 +223,7 @@ cover the remaining commands and endpoints.
212
223
 
213
224
  ## Status
214
225
 
215
- EmbedFlow v0.7.0 is a pre-1.0 release for research and early real-world
226
+ EmbedFlow v0.8.0 is a pre-1.0 release for research and early real-world
216
227
  testing. T2-v1 is an empirical finite-tail diagnostic, partial rankings can
217
228
  differ from fully warm target reranking, and ANN fidelity needs a reference
218
229
  comparison to audit.
@@ -11,6 +11,9 @@ embedflow analyze --config ./embedflow.yaml --output-dir ./analysis
11
11
  embedflow evaluate --config ./experiment.yaml --output-dir ./results
12
12
  embedflow plan --config ./embedflow.yaml --queries ./probe_queries.jsonl
13
13
  embedflow shadow report --config ./embedflow.yaml --since 24h
14
+ embedflow prewarm plan --config ./embedflow.yaml --since 24h --max-docs 50000 --output prewarm-plan.json
15
+ embedflow prewarm run --config ./embedflow.yaml --plan prewarm-plan.json
16
+ embedflow prewarm status --config ./embedflow.yaml
14
17
  ```
15
18
 
16
19
  `analyze` is the no-target-index workflow. It uses probe queries and frozen
@@ -41,7 +44,13 @@ embedflow export-target --config ./embedflow.yaml --output-index ./target.index
41
44
  `status` reports cache coverage, hit/miss counters, queue depth, and
42
45
  materialization throughput. `audit-index` checks the source index against an
43
46
  exact/reference configuration where supported. `prewarm` schedules target
44
- document work; it does not change source-index results.
47
+ document work; it does not change source-index results. `prewarm plan` ranks
48
+ uncached IDs by privacy-safe Shadow candidate occurrence counts and is always
49
+ bounded by `--max-docs` (or the configured `prewarm.max_docs`). `prewarm run`
50
+ refuses mismatched/tampered fingerprints, skips vectors that became warm, and
51
+ uses the existing durable materializer. Observed candidate-occurrence coverage
52
+ is not retrieval quality or recall. Use `--docs-per-second` and
53
+ `--gpu-hourly-cost` only for explicitly labeled modeled economics.
45
54
 
46
55
  Use `embedflow serve --mode shadow` to opt into source-authoritative Shadow
47
56
  Mode for one process. `embedflow shadow report` reads the bounded telemetry
@@ -77,6 +77,18 @@ planner:
77
77
  corpus_name: null
78
78
  corpus_fingerprint: null
79
79
 
80
+ # Optional bounded traffic-aware target-cache prewarming defaults. CLI flags
81
+ # override these values for one invocation; there is no implicit full-corpus
82
+ # prewarm operation.
83
+ prewarm:
84
+ max_docs: 1000
85
+ strategy: traffic_hotset
86
+ target_observed_coverage: null
87
+ max_storage_gb: null
88
+ max_runtime_seconds: null
89
+ batch_size: null
90
+ max_retries: null
91
+
80
92
  # Source-authoritative observation mode. ``runtime.mode: migration`` (the
81
93
  # default) leaves Shadow Mode inactive. ``mode: shadow`` is an explicit opt-in.
82
94
  runtime:
@@ -1,6 +1,6 @@
1
1
  # Limitations and release scope
2
2
 
3
- EmbedFlow v0.7.0 is a pre-1.0 release for research and early real-world
3
+ EmbedFlow v0.8.0 is a pre-1.0 release for research and early real-world
4
4
  testing. The serving path is designed to make migration experiments concrete;
5
5
  production rollout still requires application-specific validation.
6
6
 
@@ -0,0 +1,84 @@
1
+ # Traffic-aware prewarming
2
+
3
+ Traffic-aware prewarming uses privacy-safe Shadow Mode aggregates to select the
4
+ uncached target documents that occurred most often in the source candidate
5
+ pool. It is a bounded operational warming tool in the existing migration
6
+ workflow:
7
+
8
+ ```text
9
+ PLAN -> SHADOW -> PREWARM -> CANARY -> MIGRATE
10
+ ```
11
+
12
+ It never writes the source index. `prewarm plan` only reads Shadow telemetry
13
+ and the target cache. `prewarm run` delegates document lookup, target encoding,
14
+ and cache writes to the existing materialization worker.
15
+
16
+ ## Plan and run
17
+
18
+ ```bash
19
+ embedflow prewarm plan \
20
+ --config embedflow.yaml --since 24h --max-docs 50000 \
21
+ --output prewarm-plan.json
22
+ embedflow prewarm run --config embedflow.yaml --plan prewarm-plan.json
23
+ embedflow prewarm status --config embedflow.yaml
24
+ ```
25
+
26
+ Plans are schema-versioned (`schema_version: 1`), deterministically ranked by
27
+ `candidate_occurrences DESC, document_id ASC`, and fingerprinted with the
28
+ source/target contracts, source-index identity, candidate K, and Shadow
29
+ configuration. Execution refuses a stale or tampered plan. A plan is always
30
+ bounded by `max_docs`; there is no implicit full-corpus operation.
31
+
32
+ Optional configuration defaults are available under `prewarm`:
33
+
34
+ ```yaml
35
+ prewarm:
36
+ max_docs: 1000
37
+ strategy: traffic_hotset
38
+ batch_size: 32
39
+ max_retries: 3
40
+ ```
41
+
42
+ `--target-observed-coverage`, `--max-storage-gb`, and `--max-runtime` (when a
43
+ real measured/user-supplied `--docs-per-second` is supplied) further reduce
44
+ the hard selection cap. `--gpu-hourly-cost` (or the existing planner/economics
45
+ configuration value) adds a user-supplied modeled cost. Runtime, cost, and
46
+ storage values are explicitly marked as modeled/unknown; no cloud price or
47
+ throughput is guessed.
48
+
49
+ ## Coverage and privacy
50
+
51
+ The plan reports **observed candidate-occurrence coverage**:
52
+
53
+ ```text
54
+ occurrences with a valid target vector / all source candidate occurrences
55
+ ```
56
+
57
+ This is not retrieval coverage, recall, nDCG, or a guarantee of migration
58
+ quality. It only describes the selected Shadow window and candidate K. Ties are
59
+ stable, and current target-cache entries count as warm only when their target
60
+ model and (when document text is available) content fingerprints match. Legacy
61
+ cache rows created before content bindings were introduced remain usable but
62
+ are treated as content-unknown.
63
+
64
+ Shadow aggregation stores document IDs and counters, not raw query text,
65
+ document text, vectors, or credentials. Treat the generated plan/ID list as
66
+ application data because selected document IDs are needed to execute it.
67
+
68
+ ## Resume and idempotence
69
+
70
+ The runner stores a plan-specific queue/state beside the target cache and uses
71
+ the target cache as the execution source of truth. Re-running a completed plan
72
+ skips already-warm vectors. Interrupts leave completed vectors valid and a
73
+ subsequent run can continue the remaining IDs. Documents that disappear or
74
+ change are resolved again at run time; stale content is not silently reused.
75
+
76
+ Two concurrent runners use the existing durable queue/cache deduplication. The
77
+ queue is at-least-once under process crashes, so operators should inspect
78
+ `prewarm status` and the run report for failures.
79
+
80
+ The source FAISS/Qdrant/pgvector/Pinecone/Milvus/Weaviate index is never
81
+ created, altered, rebuilt, upserted, or deleted by this feature. The API for
82
+ prewarming is currently Python/core plus CLI; dashboard execution controls are
83
+ intentionally not added to avoid turning an advisory plan into autonomous
84
+ traffic routing.
@@ -18,8 +18,8 @@ python -m twine check dist/*
18
18
  Inspect both archives before uploading:
19
19
 
20
20
  ```bash
21
- unzip -l dist/embedflow-0.7.0-py3-none-any.whl
22
- tar -tzf dist/embedflow-0.7.0.tar.gz
21
+ unzip -l dist/embedflow-0.8.0-py3-none-any.whl
22
+ tar -tzf dist/embedflow-0.8.0.tar.gz
23
23
  sha256sum dist/*
24
24
  ```
25
25
 
@@ -32,7 +32,7 @@ Test the wheel outside the source tree:
32
32
  ```bash
33
33
  python -m venv /tmp/embedflow-wheel-test
34
34
  /tmp/embedflow-wheel-test/bin/python -m pip install --upgrade pip
35
- /tmp/embedflow-wheel-test/bin/python -m pip install dist/embedflow-0.7.0-py3-none-any.whl
35
+ /tmp/embedflow-wheel-test/bin/python -m pip install dist/embedflow-0.8.0-py3-none-any.whl
36
36
  cd /tmp
37
37
  /tmp/embedflow-wheel-test/bin/python -c "import embedflow; print(embedflow.__version__)"
38
38
  /tmp/embedflow-wheel-test/bin/embedflow --help
@@ -65,7 +65,7 @@ python -m venv /tmp/embedflow-testpypi
65
65
  /tmp/embedflow-testpypi/bin/python -m pip install \
66
66
  --index-url https://test.pypi.org/simple/ \
67
67
  --extra-index-url https://pypi.org/simple/ \
68
- embedflow==0.7.0
68
+ embedflow==0.8.0
69
69
  cd /tmp
70
70
  /tmp/embedflow-testpypi/bin/python -c "import embedflow; print(embedflow.__version__)"
71
71
  /tmp/embedflow-testpypi/bin/embedflow --help
@@ -79,7 +79,7 @@ Test optional integrations in a second clean environment:
79
79
  /tmp/embedflow-testpypi/bin/python -m pip install \
80
80
  --index-url https://test.pypi.org/simple/ \
81
81
  --extra-index-url https://pypi.org/simple/ \
82
- "embedflow[faiss,dashboard,pinecone,milvus,weaviate]==0.7.0"
82
+ "embedflow[faiss,dashboard,pinecone,milvus,weaviate]==0.8.0"
83
83
  ```
84
84
 
85
85
  If the same filename already exists on TestPyPI, use a pre-release for the
@@ -115,7 +115,7 @@ above keeps the test step explicit.
115
115
  3. Run the final release gate and review the generated report.
116
116
  4. Configure the PyPI pending publisher and protected `pypi` environment.
117
117
  5. Create a Git tag and GitHub Release for the exact package version, for
118
- example `v0.7.0`.
118
+ example `v0.8.0`.
119
119
  6. Approve the `pypi` environment when the release workflow is ready.
120
120
  7. Verify the files and metadata on PyPI.
121
121
  8. Install from production PyPI in a directory outside this checkout.
@@ -0,0 +1,59 @@
1
+ """EmbedFlow: progressive embedding-model migration for existing indexes."""
2
+
3
+ __version__ = "0.8.0"
4
+
5
+ from .config import (
6
+ EmbedFlowConfig,
7
+ PlannerConfig,
8
+ PrewarmConfig,
9
+ RuntimeConfig,
10
+ ShadowConfig,
11
+ ShadowTelemetryConfig,
12
+ load_config,
13
+ )
14
+ from .prewarm import PrewarmPlan, PrewarmPlanner, PrewarmRunner, TrafficHotsetPlanner, load_prewarm_plan
15
+
16
+
17
+ def plan(*args, **kwargs):
18
+ """Generate an advisory migration plan without changing source traffic."""
19
+ from .planner import plan_migration
20
+ return plan_migration(*args, **kwargs)
21
+
22
+
23
+ def migrate(*args, **kwargs):
24
+ """Start progressive migration over an existing FAISS, Qdrant, pgvector, Pinecone, Milvus, or Weaviate index.
25
+
26
+ Imported lazily to keep the lightweight configuration package free of
27
+ model-serving dependencies at import time. See ``embedflow.migration``
28
+ for the ``MigrationSession`` type.
29
+ """
30
+ from .migration.facade import migrate as _migrate
31
+ return _migrate(*args, **kwargs)
32
+
33
+
34
+ def analyze_migration(*args, **kwargs):
35
+ """Run the leakage-safe no-target-index analysis programmatically."""
36
+ from .analysis import analyze_migration as _analyze_migration
37
+ return _analyze_migration(*args, **kwargs)
38
+
39
+
40
+ def prewarm_plan(telemetry, cache, **kwargs):
41
+ """Build a bounded traffic-hotset plan through the reusable API.
42
+
43
+ Planner-constructor options and ``plan`` options may be supplied together;
44
+ recognized plan options are routed to :meth:`PrewarmPlanner.plan`.
45
+ """
46
+ plan_keys = {"since_seconds", "start", "end", "max_docs", "target_observed_coverage",
47
+ "max_storage_gb", "max_runtime_seconds", "docs_per_second", "gpu_hourly_cost", "strategy"}
48
+ plan_kwargs = {key: kwargs.pop(key) for key in tuple(kwargs) if key in plan_keys}
49
+ return PrewarmPlanner(telemetry, cache, **kwargs).plan(**plan_kwargs)
50
+
51
+
52
+ def prewarm_run(plan, *args, **kwargs):
53
+ """Execute a validated prewarm plan through the existing materializer."""
54
+ run_keys = {"max_runtime_seconds", "progress"}
55
+ run_kwargs = {key: kwargs.pop(key) for key in tuple(kwargs) if key in run_keys}
56
+ return PrewarmRunner(*args, **kwargs).run(plan, **run_kwargs)
57
+
58
+
59
+ __all__ = ["EmbedFlowConfig", "PlannerConfig", "PrewarmConfig", "RuntimeConfig", "ShadowConfig", "ShadowTelemetryConfig", "load_config", "migrate", "plan", "analyze_migration", "prewarm_plan", "prewarm_run", "PrewarmPlan", "PrewarmPlanner", "TrafficHotsetPlanner", "PrewarmRunner", "load_prewarm_plan", "__version__"]
@@ -8,15 +8,15 @@ import numpy as np
8
8
 
9
9
  class TargetVectorCache(ABC):
10
10
  @abstractmethod
11
- def get(self, document_ids: list[str]) -> dict[str, np.ndarray]:
11
+ def get(self, document_ids: list[str], content_fingerprints: Any = None) -> dict[str, np.ndarray]:
12
12
  raise NotImplementedError
13
13
 
14
14
  @abstractmethod
15
- def put(self, document_ids: list[str], vectors: np.ndarray) -> None:
15
+ def put(self, document_ids: list[str], vectors: np.ndarray, content_fingerprints: Any = None) -> None:
16
16
  raise NotImplementedError
17
17
 
18
18
  @abstractmethod
19
- def contains(self, document_ids: list[str]) -> set[str]:
19
+ def contains(self, document_ids: list[str], content_fingerprints: Any = None) -> set[str]:
20
20
  raise NotImplementedError
21
21
 
22
22
  @abstractmethod
@@ -4,6 +4,7 @@ import hashlib
4
4
  import sqlite3
5
5
  import threading
6
6
  import time
7
+ from collections.abc import Mapping
7
8
  from pathlib import Path
8
9
  from typing import Any
9
10
 
@@ -25,7 +26,7 @@ class SQLiteVectorCache(TargetVectorCache):
25
26
  _LOOKUP_BATCH_SIZE = 900
26
27
 
27
28
  def __init__(self, path: str | Path, model_fingerprint: str, dimension: int,
28
- dtype: str = "float32"):
29
+ dtype: str = "float32", *, read_only: bool = False):
29
30
  p = Path(path)
30
31
  if p.suffix in {".sqlite", ".sqlite3", ".db"}:
31
32
  self.db_path = p
@@ -33,7 +34,13 @@ class SQLiteVectorCache(TargetVectorCache):
33
34
  else:
34
35
  self.root = p
35
36
  self.db_path = p / "cache.sqlite3"
36
- self.root.mkdir(parents=True, exist_ok=True)
37
+ self.read_only = bool(read_only)
38
+ # A planning/status inspection must not create a cache directory or
39
+ # migrate its schema. Normal serving retains the historical
40
+ # create-or-open behavior; read-only callers use SQLite's URI mode and
41
+ # receive an empty view when the cache has not been initialized yet.
42
+ if not self.read_only:
43
+ self.root.mkdir(parents=True, exist_ok=True)
37
44
  self.model_fingerprint = str(model_fingerprint).strip()
38
45
  if not self.model_fingerprint:
39
46
  raise ValueError("model_fingerprint must be non-empty")
@@ -53,7 +60,20 @@ class SQLiteVectorCache(TargetVectorCache):
53
60
  raise ValueError("cache dtype must be a numeric vector dtype")
54
61
  self._lock = threading.RLock()
55
62
  self._db = None
63
+ self._has_content_fingerprint = True
64
+ self._has_target_vectors = True
56
65
  try:
66
+ if self.read_only:
67
+ if self.db_path.exists():
68
+ self._db = sqlite3.connect(f"file:{self.db_path}?mode=ro", uri=True,
69
+ check_same_thread=False, timeout=30)
70
+ self._db.execute("PRAGMA busy_timeout=30000")
71
+ columns = {str(row[1]) for row in self._db.execute(
72
+ "PRAGMA table_info(target_vectors)").fetchall()}
73
+ self._has_target_vectors = bool(columns)
74
+ self._has_content_fingerprint = "content_fingerprint" in columns
75
+ self._closed = False
76
+ return
57
77
  self._db = sqlite3.connect(str(self.db_path), check_same_thread=False, timeout=30)
58
78
  self._db.execute("PRAGMA busy_timeout=30000")
59
79
  self._db.execute("PRAGMA journal_mode=WAL")
@@ -67,8 +87,18 @@ class SQLiteVectorCache(TargetVectorCache):
67
87
  checksum TEXT NOT NULL,
68
88
  created_at REAL NOT NULL,
69
89
  accessed_at REAL NOT NULL,
90
+ content_fingerprint TEXT,
70
91
  PRIMARY KEY(document_id, model_fingerprint)
71
92
  )""")
93
+ # Caches created by pre-0.8 releases do not have a content
94
+ # fingerprint. Keep those vectors readable for existing serving
95
+ # callers, while allowing prewarming to require a current-content
96
+ # match when one is available.
97
+ columns = {str(row[1]) for row in self._db.execute("PRAGMA table_info(target_vectors)").fetchall()}
98
+ if "content_fingerprint" not in columns:
99
+ self._db.execute("ALTER TABLE target_vectors ADD COLUMN content_fingerprint TEXT")
100
+ self._has_content_fingerprint = True
101
+ self._has_target_vectors = True
72
102
  self._db.execute("CREATE INDEX IF NOT EXISTS idx_target_vectors_model ON target_vectors(model_fingerprint)")
73
103
  self._db.execute("""CREATE TABLE IF NOT EXISTS cache_counters (
74
104
  model_fingerprint TEXT PRIMARY KEY,
@@ -92,7 +122,7 @@ class SQLiteVectorCache(TargetVectorCache):
92
122
  self._closed = False
93
123
 
94
124
  def _ensure_open(self) -> None:
95
- if self._closed or self._db is None:
125
+ if self._closed or (self._db is None and not self.read_only):
96
126
  raise RuntimeError("target vector cache is closed")
97
127
 
98
128
  def _decode(self, row: tuple[Any, ...]) -> np.ndarray:
@@ -106,20 +136,53 @@ class SQLiteVectorCache(TargetVectorCache):
106
136
  raise CacheCorruptionError(f"invalid cached vector for document {document_id}")
107
137
  return value.astype("float32", copy=False)
108
138
 
109
- def get(self, document_ids: list[str]) -> dict[str, np.ndarray]:
139
+ @staticmethod
140
+ def content_fingerprint(text: Any) -> str:
141
+ """Return the stable fingerprint used to bind a vector to content."""
142
+ return hashlib.sha256(str(text).encode("utf-8")).hexdigest()
143
+
144
+ @staticmethod
145
+ def _normalize_content_fingerprints(document_ids: list[str], values: Any) -> dict[str, str] | None:
146
+ if values is None:
147
+ return None
148
+ if isinstance(values, Mapping):
149
+ return {str(key): str(value) for key, value in values.items()}
150
+ try:
151
+ sequence = list(values)
152
+ except TypeError as exc:
153
+ raise ValueError("content_fingerprints must be a mapping or sequence") from exc
154
+ if len(sequence) != len(document_ids):
155
+ raise ValueError("content_fingerprints must match document_ids length")
156
+ return {str(document_id): str(value) for document_id, value in zip(document_ids, sequence)}
157
+
158
+ def get(self, document_ids: list[str], content_fingerprints: Any = None) -> dict[str, np.ndarray]:
159
+ if self.read_only:
160
+ # Preserve the familiar lookup contract for inspection callers
161
+ # without updating hit/miss/access timestamps in a read-only
162
+ # SQLite connection.
163
+ return self.peek(document_ids, content_fingerprints=content_fingerprints)
110
164
  ids = [str(x) for x in document_ids]
111
165
  if not ids: return {}
166
+ expected_content = self._normalize_content_fingerprints(ids, content_fingerprints)
112
167
  with self._lock:
113
168
  self._ensure_open()
114
169
  rows: list[tuple[Any, ...]] = []
115
170
  for start in range(0, len(ids), self._LOOKUP_BATCH_SIZE):
116
171
  chunk = ids[start:start + self._LOOKUP_BATCH_SIZE]
117
172
  placeholders = ",".join("?" for _ in chunk)
173
+ columns = ",content_fingerprint" if self._has_content_fingerprint else ""
118
174
  rows.extend(self._db.execute(
119
- f"SELECT document_id,model_fingerprint,dimension,dtype,vector,checksum,created_at,accessed_at "
175
+ f"SELECT document_id,model_fingerprint,dimension,dtype,vector,checksum,created_at,accessed_at{columns} "
120
176
  f"FROM target_vectors WHERE model_fingerprint=? AND document_id IN ({placeholders})",
121
177
  [self.model_fingerprint, *chunk]).fetchall())
122
178
  now = time.time()
179
+ if expected_content is not None:
180
+ # Rows from pre-0.8 caches have no content binding. Preserve
181
+ # their historical serving behavior while requiring an exact
182
+ # match whenever a binding is present (new writes and
183
+ # materialized vectors).
184
+ if self._has_content_fingerprint:
185
+ rows = [row for row in rows if row[8] is None or str(row[8]) == expected_content.get(str(row[0]))]
123
186
  values = {str(row[0]): self._decode(row) for row in rows}
124
187
  if rows:
125
188
  self._db.executemany("UPDATE target_vectors SET accessed_at=? WHERE document_id=? AND model_fingerprint=?",
@@ -141,7 +204,7 @@ class SQLiteVectorCache(TargetVectorCache):
141
204
  self._db.commit()
142
205
  return values
143
206
 
144
- def peek(self, document_ids: list[str]) -> dict[str, np.ndarray]:
207
+ def peek(self, document_ids: list[str], content_fingerprints: Any = None) -> dict[str, np.ndarray]:
145
208
  """Read cached vectors without changing hit/miss telemetry.
146
209
 
147
210
  Shadow analysis with ``materialize=false`` must not make ordinary
@@ -152,19 +215,27 @@ class SQLiteVectorCache(TargetVectorCache):
152
215
  ids = [str(x) for x in document_ids]
153
216
  if not ids:
154
217
  return {}
218
+ expected_content = self._normalize_content_fingerprints(ids, content_fingerprints)
155
219
  with self._lock:
156
220
  self._ensure_open()
221
+ if self._db is None or not self._has_target_vectors:
222
+ return {}
157
223
  rows: list[tuple[Any, ...]] = []
158
224
  for start in range(0, len(ids), self._LOOKUP_BATCH_SIZE):
159
225
  chunk = ids[start:start + self._LOOKUP_BATCH_SIZE]
160
226
  placeholders = ",".join("?" for _ in chunk)
227
+ columns = ",content_fingerprint" if self._has_content_fingerprint else ""
161
228
  rows.extend(self._db.execute(
162
- f"SELECT document_id,model_fingerprint,dimension,dtype,vector,checksum,created_at,accessed_at "
229
+ f"SELECT document_id,model_fingerprint,dimension,dtype,vector,checksum,created_at,accessed_at{columns} "
163
230
  f"FROM target_vectors WHERE model_fingerprint=? AND document_id IN ({placeholders})",
164
231
  [self.model_fingerprint, *chunk]).fetchall())
232
+ if expected_content is not None and self._has_content_fingerprint:
233
+ rows = [row for row in rows if row[8] is None or str(row[8]) == expected_content.get(str(row[0]))]
165
234
  return {str(row[0]): self._decode(row) for row in rows}
166
235
 
167
- def put(self, document_ids: list[str], vectors: np.ndarray) -> None:
236
+ def put(self, document_ids: list[str], vectors: np.ndarray, content_fingerprints: Any = None) -> None:
237
+ if self.read_only:
238
+ raise RuntimeError("target vector cache is read-only")
168
239
  ids = [str(x) for x in document_ids]
169
240
  values = np.asarray(vectors, dtype=self.dtype)
170
241
  if values.ndim != 2 or values.shape != (len(ids), self.dimension):
@@ -173,27 +244,70 @@ class SQLiteVectorCache(TargetVectorCache):
173
244
  raise ValueError("duplicate IDs in cache write")
174
245
  if not np.isfinite(values).all():
175
246
  raise ValueError("cannot cache non-finite vectors")
247
+ content = self._normalize_content_fingerprints(ids, content_fingerprints)
176
248
  now = time.time(); records = []
177
249
  for document_id, value in zip(ids, values):
178
250
  blob = np.ascontiguousarray(value).tobytes()
179
251
  records.append((document_id, self.model_fingerprint, self.dimension, self.dtype.name, sqlite3.Binary(blob),
180
- hashlib.sha256(blob).hexdigest(), now, now))
252
+ hashlib.sha256(blob).hexdigest(), now, now,
253
+ content.get(document_id) if content is not None else None))
181
254
  with self._lock:
182
255
  self._ensure_open()
183
- self._db.executemany("""INSERT INTO target_vectors
184
- (document_id,model_fingerprint,dimension,dtype,vector,checksum,created_at,accessed_at)
185
- VALUES (?,?,?,?,?,?,?,?)
186
- ON CONFLICT(document_id,model_fingerprint) DO UPDATE SET
187
- dimension=excluded.dimension,dtype=excluded.dtype,vector=excluded.vector,
188
- checksum=excluded.checksum,accessed_at=excluded.accessed_at""", records)
256
+ if content is None:
257
+ # Older callers do not know document content. Updating the
258
+ # vector must not erase a binding written by the new
259
+ # materializer, otherwise a later content change could be
260
+ # mistaken for a valid cache hit. New rows remain unbound and
261
+ # therefore retain historical compatibility.
262
+ self._db.executemany("""INSERT INTO target_vectors
263
+ (document_id,model_fingerprint,dimension,dtype,vector,checksum,created_at,accessed_at,content_fingerprint)
264
+ VALUES (?,?,?,?,?,?,?,?,NULL)
265
+ ON CONFLICT(document_id,model_fingerprint) DO UPDATE SET
266
+ dimension=excluded.dimension,dtype=excluded.dtype,vector=excluded.vector,
267
+ checksum=excluded.checksum,accessed_at=excluded.accessed_at""",
268
+ [record[:8] for record in records])
269
+ else:
270
+ self._db.executemany("""INSERT INTO target_vectors
271
+ (document_id,model_fingerprint,dimension,dtype,vector,checksum,created_at,accessed_at,content_fingerprint)
272
+ VALUES (?,?,?,?,?,?,?,?,?)
273
+ ON CONFLICT(document_id,model_fingerprint) DO UPDATE SET
274
+ dimension=excluded.dimension,dtype=excluded.dtype,vector=excluded.vector,
275
+ checksum=excluded.checksum,accessed_at=excluded.accessed_at,
276
+ content_fingerprint=excluded.content_fingerprint""", records)
189
277
  self._db.commit()
190
278
 
191
- def contains(self, document_ids: list[str]) -> set[str]:
192
- return set(self.get(document_ids))
279
+ def contains(self, document_ids: list[str], content_fingerprints: Any = None) -> set[str]:
280
+ if self.read_only:
281
+ return set(self.peek(document_ids, content_fingerprints=content_fingerprints))
282
+ return set(self.get(document_ids, content_fingerprints=content_fingerprints))
283
+
284
+ def content_fingerprints(self, document_ids: list[str]) -> dict[str, str | None]:
285
+ """Inspect content bindings without changing cache hit/miss counters."""
286
+ ids = [str(x) for x in document_ids]
287
+ if not ids:
288
+ return {}
289
+ with self._lock:
290
+ self._ensure_open()
291
+ if self._db is None or not self._has_target_vectors or not self._has_content_fingerprint:
292
+ return {}
293
+ rows: list[tuple[Any, ...]] = []
294
+ for start in range(0, len(ids), self._LOOKUP_BATCH_SIZE):
295
+ chunk = ids[start:start + self._LOOKUP_BATCH_SIZE]
296
+ placeholders = ",".join("?" for _ in chunk)
297
+ rows.extend(self._db.execute(
298
+ f"SELECT document_id,content_fingerprint FROM target_vectors "
299
+ f"WHERE model_fingerprint=? AND document_id IN ({placeholders})",
300
+ [self.model_fingerprint, *chunk]).fetchall())
301
+ return {str(row[0]): (None if row[1] is None else str(row[1])) for row in rows}
193
302
 
194
303
  def stats(self) -> dict[str, Any]:
195
304
  with self._lock:
196
305
  self._ensure_open()
306
+ if self._db is None or not self._has_target_vectors:
307
+ return {"cached_target_vectors": 0, "all_model_vectors": 0,
308
+ "model_fingerprint": self.model_fingerprint, "dimension": self.dimension,
309
+ "cache_path": str(self.db_path), "total_cache_hits": 0,
310
+ "total_cache_misses": 0, "recent_hit_rate": 0.0}
197
311
  count = self._db.execute("SELECT COUNT(*) FROM target_vectors WHERE model_fingerprint=?", (self.model_fingerprint,)).fetchone()[0]
198
312
  total = self._db.execute("SELECT COUNT(*) FROM target_vectors").fetchone()[0]
199
313
  counters = self._db.execute("SELECT hits,misses FROM cache_counters WHERE model_fingerprint=?",
@@ -213,7 +327,8 @@ class SQLiteVectorCache(TargetVectorCache):
213
327
  def close(self) -> None:
214
328
  with self._lock:
215
329
  if not self._closed:
216
- self._db.close()
330
+ if self._db is not None:
331
+ self._db.close()
217
332
  self._db = None
218
333
  self._closed = True
219
334