embedflow 0.6.0__tar.gz → 0.8.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (179) hide show
  1. {embedflow-0.6.0 → embedflow-0.8.0}/CHANGELOG.md +24 -0
  2. {embedflow-0.6.0 → embedflow-0.8.0}/CITATION.cff +2 -2
  3. {embedflow-0.6.0 → embedflow-0.8.0}/PKG-INFO +33 -2
  4. {embedflow-0.6.0 → embedflow-0.8.0}/README.md +31 -1
  5. {embedflow-0.6.0 → embedflow-0.8.0}/README_PYPI.md +32 -1
  6. {embedflow-0.6.0 → embedflow-0.8.0}/docs/api.md +8 -0
  7. {embedflow-0.6.0 → embedflow-0.8.0}/docs/cli.md +17 -1
  8. {embedflow-0.6.0 → embedflow-0.8.0}/docs/configuration.md +59 -0
  9. {embedflow-0.6.0 → embedflow-0.8.0}/docs/limitations.md +1 -1
  10. embedflow-0.8.0/docs/prewarming.md +84 -0
  11. {embedflow-0.6.0 → embedflow-0.8.0}/docs/releasing.md +8 -8
  12. embedflow-0.8.0/docs/shadow-mode.md +104 -0
  13. embedflow-0.8.0/embedflow/__init__.py +59 -0
  14. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/cache/base.py +3 -3
  15. embedflow-0.8.0/embedflow/cache/persistent_cache.py +336 -0
  16. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/cli.py +516 -13
  17. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/config.py +250 -0
  18. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/migration/facade.py +50 -4
  19. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/migration/materializer.py +124 -6
  20. embedflow-0.8.0/embedflow/prewarm/__init__.py +23 -0
  21. embedflow-0.8.0/embedflow/prewarm/models.py +181 -0
  22. embedflow-0.8.0/embedflow/prewarm/planner.py +479 -0
  23. embedflow-0.8.0/embedflow/prewarm/runner.py +450 -0
  24. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/runtime.py +60 -2
  25. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/serving/api.py +26 -3
  26. embedflow-0.8.0/embedflow/serving/engine.py +695 -0
  27. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/serving/schemas.py +1 -0
  28. embedflow-0.8.0/embedflow/shadow/__init__.py +9 -0
  29. embedflow-0.8.0/embedflow/shadow/models.py +65 -0
  30. embedflow-0.8.0/embedflow/shadow/report.py +48 -0
  31. embedflow-0.8.0/embedflow/shadow/runner.py +356 -0
  32. embedflow-0.8.0/embedflow/shadow/telemetry.py +1070 -0
  33. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow.egg-info/PKG-INFO +33 -2
  34. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow.egg-info/SOURCES.txt +19 -0
  35. embedflow-0.8.0/examples/prewarm/README.md +16 -0
  36. embedflow-0.8.0/examples/prewarm/demo.py +74 -0
  37. embedflow-0.8.0/examples/prewarm/embedflow.yaml.example +24 -0
  38. embedflow-0.8.0/examples/prewarm/run_demo.sh +12 -0
  39. embedflow-0.8.0/examples/shadow/README.md +13 -0
  40. embedflow-0.8.0/examples/shadow/embedflow.yaml.example +25 -0
  41. embedflow-0.8.0/examples/shadow/run_demo.py +95 -0
  42. embedflow-0.8.0/examples/shadow/run_demo.sh +10 -0
  43. {embedflow-0.6.0 → embedflow-0.8.0}/pyproject.toml +1 -1
  44. {embedflow-0.6.0 → embedflow-0.8.0}/scripts/release_gate.py +16 -4
  45. embedflow-0.6.0/embedflow/__init__.py +0 -31
  46. embedflow-0.6.0/embedflow/cache/persistent_cache.py +0 -198
  47. embedflow-0.6.0/embedflow/serving/engine.py +0 -228
  48. {embedflow-0.6.0 → embedflow-0.8.0}/CONTRIBUTING.md +0 -0
  49. {embedflow-0.6.0 → embedflow-0.8.0}/LICENSE +0 -0
  50. {embedflow-0.6.0 → embedflow-0.8.0}/MANIFEST.in +0 -0
  51. {embedflow-0.6.0 → embedflow-0.8.0}/SECURITY.md +0 -0
  52. {embedflow-0.6.0 → embedflow-0.8.0}/docs/assets/README.md +0 -0
  53. {embedflow-0.6.0 → embedflow-0.8.0}/docs/assets/candidate-gap-example.svg +0 -0
  54. {embedflow-0.6.0 → embedflow-0.8.0}/docs/assets/dashboard-screenshot.md +0 -0
  55. {embedflow-0.6.0 → embedflow-0.8.0}/docs/assets/terminal-demo.txt +0 -0
  56. {embedflow-0.6.0 → embedflow-0.8.0}/docs/concepts.md +0 -0
  57. {embedflow-0.6.0 → embedflow-0.8.0}/docs/contributing-benchmarks.md +0 -0
  58. {embedflow-0.6.0 → embedflow-0.8.0}/docs/economics.md +0 -0
  59. {embedflow-0.6.0 → embedflow-0.8.0}/docs/installation.md +0 -0
  60. {embedflow-0.6.0 → embedflow-0.8.0}/docs/integrations/faiss.md +0 -0
  61. {embedflow-0.6.0 → embedflow-0.8.0}/docs/integrations/milvus.md +0 -0
  62. {embedflow-0.6.0 → embedflow-0.8.0}/docs/integrations/pgvector.md +0 -0
  63. {embedflow-0.6.0 → embedflow-0.8.0}/docs/integrations/pinecone.md +0 -0
  64. {embedflow-0.6.0 → embedflow-0.8.0}/docs/integrations/qdrant.md +0 -0
  65. {embedflow-0.6.0 → embedflow-0.8.0}/docs/integrations/weaviate.md +0 -0
  66. {embedflow-0.6.0 → embedflow-0.8.0}/docs/methodology.md +0 -0
  67. {embedflow-0.6.0 → embedflow-0.8.0}/docs/planner.md +0 -0
  68. {embedflow-0.6.0 → embedflow-0.8.0}/docs/quickstart.md +0 -0
  69. {embedflow-0.6.0 → embedflow-0.8.0}/docs/registry.md +0 -0
  70. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/__main__.py +0 -0
  71. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/analysis.py +0 -0
  72. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/api.py +0 -0
  73. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/cache/__init__.py +0 -0
  74. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/compatibility/__init__.py +0 -0
  75. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/compatibility/candidate_gap.py +0 -0
  76. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/compatibility/containment.py +0 -0
  77. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/compatibility/evaluate.py +0 -0
  78. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/compatibility/metrics.py +0 -0
  79. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/compatibility/migration_depth.py +0 -0
  80. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/compatibility/probe.py +0 -0
  81. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/compatibility/report.py +0 -0
  82. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/compatibility/t2.py +0 -0
  83. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/data/__init__.py +0 -0
  84. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/data/registry/__init__.py +0 -0
  85. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/data/registry/benchmark_profiles.jsonl +0 -0
  86. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/data/registry/checksums.sha256 +0 -0
  87. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/data/registry/migrations.jsonl +0 -0
  88. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/data/registry/registry_manifest.json +0 -0
  89. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/data/registry/research_summaries.json +0 -0
  90. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/data/registry/schema_version.json +0 -0
  91. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/frozen/T2_V1_FROZEN_SPEC.md +0 -0
  92. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/frozen/T2_V1_FROZEN_SPEC.sha256 +0 -0
  93. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/indexes/__init__.py +0 -0
  94. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/indexes/base.py +0 -0
  95. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/indexes/faiss_backend.py +0 -0
  96. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/indexes/milvus_backend.py +0 -0
  97. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/indexes/pgvector_backend.py +0 -0
  98. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/indexes/pinecone_backend.py +0 -0
  99. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/indexes/qdrant_backend.py +0 -0
  100. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/indexes/weaviate_backend.py +0 -0
  101. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/metrics/__init__.py +0 -0
  102. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/metrics/latency.py +0 -0
  103. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/migration/__init__.py +0 -0
  104. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/migration/compatibility.py +0 -0
  105. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/migration/planner.py +0 -0
  106. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/migration/state.py +0 -0
  107. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/models/__init__.py +0 -0
  108. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/models/base.py +0 -0
  109. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/models/huggingface.py +0 -0
  110. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/planner/__init__.py +0 -0
  111. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/planner/economics.py +0 -0
  112. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/planner/models.py +0 -0
  113. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/planner/planner.py +0 -0
  114. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/planner/rendering.py +0 -0
  115. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/registry/__init__.py +0 -0
  116. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/registry/loader.py +0 -0
  117. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/registry/matcher.py +0 -0
  118. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/registry/schema.py +0 -0
  119. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/serving/__init__.py +0 -0
  120. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/serving/factory.py +0 -0
  121. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow.egg-info/dependency_links.txt +0 -0
  122. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow.egg-info/entry_points.txt +0 -0
  123. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow.egg-info/requires.txt +0 -0
  124. {embedflow-0.6.0 → embedflow-0.8.0}/embedflow.egg-info/top_level.txt +0 -0
  125. {embedflow-0.6.0 → embedflow-0.8.0}/examples/faiss/README.md +0 -0
  126. {embedflow-0.6.0 → embedflow-0.8.0}/examples/faiss/documents.jsonl +0 -0
  127. {embedflow-0.6.0 → embedflow-0.8.0}/examples/faiss/embedflow.yaml +0 -0
  128. {embedflow-0.6.0 → embedflow-0.8.0}/examples/faiss/queries.jsonl +0 -0
  129. {embedflow-0.6.0 → embedflow-0.8.0}/examples/milvus/README.md +0 -0
  130. {embedflow-0.6.0 → embedflow-0.8.0}/examples/milvus/compose.yaml +0 -0
  131. {embedflow-0.6.0 → embedflow-0.8.0}/examples/milvus/embedflow.yaml.example +0 -0
  132. {embedflow-0.6.0 → embedflow-0.8.0}/examples/milvus/run_demo.sh +0 -0
  133. {embedflow-0.6.0 → embedflow-0.8.0}/examples/pgvector/README.md +0 -0
  134. {embedflow-0.6.0 → embedflow-0.8.0}/examples/pgvector/build_index.py +0 -0
  135. {embedflow-0.6.0 → embedflow-0.8.0}/examples/pgvector/compose.yaml +0 -0
  136. {embedflow-0.6.0 → embedflow-0.8.0}/examples/pgvector/embedflow.yaml +0 -0
  137. {embedflow-0.6.0 → embedflow-0.8.0}/examples/pgvector/init.sql +0 -0
  138. {embedflow-0.6.0 → embedflow-0.8.0}/examples/pgvector/queries.jsonl +0 -0
  139. {embedflow-0.6.0 → embedflow-0.8.0}/examples/pgvector/run_demo.sh +0 -0
  140. {embedflow-0.6.0 → embedflow-0.8.0}/examples/pinecone/README.md +0 -0
  141. {embedflow-0.6.0 → embedflow-0.8.0}/examples/pinecone/embedflow.yaml.example +0 -0
  142. {embedflow-0.6.0 → embedflow-0.8.0}/examples/pinecone/run_smoke.sh +0 -0
  143. {embedflow-0.6.0 → embedflow-0.8.0}/examples/planner/README.md +0 -0
  144. {embedflow-0.6.0 → embedflow-0.8.0}/examples/planner/run_demo.sh +0 -0
  145. {embedflow-0.6.0 → embedflow-0.8.0}/examples/qdrant/README.md +0 -0
  146. {embedflow-0.6.0 → embedflow-0.8.0}/examples/qdrant/build_index.py +0 -0
  147. {embedflow-0.6.0 → embedflow-0.8.0}/examples/qdrant/documents.jsonl +0 -0
  148. {embedflow-0.6.0 → embedflow-0.8.0}/examples/qdrant/embedflow.yaml +0 -0
  149. {embedflow-0.6.0 → embedflow-0.8.0}/examples/qdrant/queries.jsonl +0 -0
  150. {embedflow-0.6.0 → embedflow-0.8.0}/examples/research_analysis/README.md +0 -0
  151. {embedflow-0.6.0 → embedflow-0.8.0}/examples/research_analysis/documents.jsonl +0 -0
  152. {embedflow-0.6.0 → embedflow-0.8.0}/examples/research_analysis/embedflow.yaml +0 -0
  153. {embedflow-0.6.0 → embedflow-0.8.0}/examples/research_analysis/qrels.json +0 -0
  154. {embedflow-0.6.0 → embedflow-0.8.0}/examples/research_analysis/queries.jsonl +0 -0
  155. {embedflow-0.6.0 → embedflow-0.8.0}/examples/weaviate/README.md +0 -0
  156. {embedflow-0.6.0 → embedflow-0.8.0}/examples/weaviate/compose.yaml +0 -0
  157. {embedflow-0.6.0 → embedflow-0.8.0}/examples/weaviate/embedflow.yaml.example +0 -0
  158. {embedflow-0.6.0 → embedflow-0.8.0}/examples/weaviate/run_demo.sh +0 -0
  159. {embedflow-0.6.0 → embedflow-0.8.0}/frozen/T2_V1_FROZEN_SPEC.md +0 -0
  160. {embedflow-0.6.0 → embedflow-0.8.0}/frozen/T2_V1_FROZEN_SPEC.sha256 +0 -0
  161. {embedflow-0.6.0 → embedflow-0.8.0}/requirements-dev.txt +0 -0
  162. {embedflow-0.6.0 → embedflow-0.8.0}/requirements.txt +0 -0
  163. {embedflow-0.6.0 → embedflow-0.8.0}/scripts/milvus_fixture.py +0 -0
  164. {embedflow-0.6.0 → embedflow-0.8.0}/scripts/pinecone_smoke.py +0 -0
  165. {embedflow-0.6.0 → embedflow-0.8.0}/scripts/real_qdrant_smoke.py +0 -0
  166. {embedflow-0.6.0 → embedflow-0.8.0}/scripts/run_demo.sh +0 -0
  167. {embedflow-0.6.0 → embedflow-0.8.0}/scripts/run_tests.sh +0 -0
  168. {embedflow-0.6.0 → embedflow-0.8.0}/scripts/validate_milvus.py +0 -0
  169. {embedflow-0.6.0 → embedflow-0.8.0}/scripts/validate_pgvector_10k.py +0 -0
  170. {embedflow-0.6.0 → embedflow-0.8.0}/scripts/validate_weaviate.py +0 -0
  171. {embedflow-0.6.0 → embedflow-0.8.0}/scripts/weaviate_fixture.py +0 -0
  172. {embedflow-0.6.0 → embedflow-0.8.0}/scripts/weaviate_smoke.py +0 -0
  173. {embedflow-0.6.0 → embedflow-0.8.0}/setup.cfg +0 -0
  174. {embedflow-0.6.0 → embedflow-0.8.0}/src/__init__.py +0 -0
  175. {embedflow-0.6.0 → embedflow-0.8.0}/src/embed.py +0 -0
  176. {embedflow-0.6.0 → embedflow-0.8.0}/src/probe_features.py +0 -0
  177. {embedflow-0.6.0 → embedflow-0.8.0}/src/storage.py +0 -0
  178. {embedflow-0.6.0 → embedflow-0.8.0}/src/t2_v1.py +0 -0
  179. {embedflow-0.6.0 → embedflow-0.8.0}/src/utils.py +0 -0
@@ -1,5 +1,29 @@
1
1
  # Changelog
2
2
 
3
+ ## v0.8.0 — Traffic-aware target-vector prewarming
4
+
5
+ - Added bounded `prewarm plan`, `prewarm run`, and `prewarm status` workflows
6
+ that select uncached target documents by observed Shadow Mode candidate
7
+ occurrences.
8
+ - Added privacy-safe aggregate document popularity telemetry, content/model
9
+ fingerprint checks, deterministic tie-breaking, hard document/storage/time
10
+ budgets, atomic plan/run artifacts, and resumable materialization through the
11
+ existing target cache and worker.
12
+ - Observed candidate-occurrence coverage remains an operational warming signal;
13
+ it is not retrieval quality or recall evidence.
14
+
15
+ ## v0.7.0 — Source-authoritative Shadow Mode
16
+
17
+ - Added bounded, deterministic, source-authoritative Shadow Mode for observing
18
+ target reranking on sampled traffic without delaying or changing source
19
+ responses.
20
+ - Added failure/timeout isolation, queue backpressure, optional asynchronous
21
+ target-cache materialization, target-coverage and ranking diagnostics, and
22
+ privacy-conscious SQLite telemetry.
23
+ - Added `embedflow shadow report`, API/status integration, documentation, and a
24
+ deterministic offline demonstration. Shadow reports remain operational
25
+ diagnostics and do not claim qrel-based retrieval quality.
26
+
3
27
  ## v0.6.0 — Migration planner
4
28
 
5
29
  - Added an advisory `embedflow plan` command and Python API that combine
@@ -2,8 +2,8 @@ cff-version: 1.2.0
2
2
  title: "EmbedFlow: Upgrading Legacy Embeddings Without Full Upfront Re-Embedding"
3
3
  message: "If EmbedFlow contributes to your work, please cite this software release."
4
4
  type: software
5
- version: 0.6.0
6
- date-released: 2026-09-13
5
+ version: 0.8.0
6
+ date-released: 2026-09-15
7
7
  repository-code: "https://github.com/arnsri33/embedflow"
8
8
  url: "https://github.com/arnsri33/embedflow"
9
9
  license: AGPL-3.0-only
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: embedflow
3
- Version: 0.6.0
3
+ Version: 0.8.0
4
4
  Summary: Progressive embedding-model migration over existing vector indexes.
5
5
  Author: Arnav Srivastav
6
6
  License-Expression: AGPL-3.0-only
@@ -204,6 +204,37 @@ behavior, candidate K, cache/economics projections, and staged rollout
204
204
  guidance. It never routes traffic or mutates the source index; `SAFE` is not a
205
205
  qrels-based retrieval-quality guarantee. See the [planner guide](https://github.com/arnsri33/embedflow/blob/main/docs/planner.md).
206
206
 
207
+ ### Source-authoritative Shadow Mode
208
+
209
+ Run the target migration path beside sampled traffic while always returning the
210
+ source result:
211
+
212
+ ```yaml
213
+ runtime: {mode: shadow}
214
+ shadow: {enabled: true, sample_rate: 0.10, candidate_k: 100, materialize: true}
215
+ ```
216
+
217
+ ```bash
218
+ embedflow serve --config ./embedflow.yaml
219
+ embedflow shadow report --config ./embedflow.yaml --since 24h --format json
220
+ ```
221
+
222
+ Shadow work is asynchronous, bounded, privacy-conscious, and failure isolated.
223
+ Reports contain operational and ranking-disagreement diagnostics—not qrel-based
224
+ quality guarantees—and recommendations never route canary traffic. See the
225
+ [Shadow Mode guide](https://github.com/arnsri33/embedflow/blob/main/docs/shadow-mode.md).
226
+
227
+ After collecting Shadow traffic, prioritize uncached target vectors by observed
228
+ candidate popularity:
229
+
230
+ ```bash
231
+ embedflow prewarm plan --config ./embedflow.yaml --since 24h --max-docs 50000 --output prewarm-plan.json
232
+ embedflow prewarm run --config ./embedflow.yaml --plan prewarm-plan.json
233
+ ```
234
+
235
+ Observed candidate-occurrence coverage is an operational warming signal, not
236
+ retrieval quality or recall. See the [prewarming guide](https://github.com/arnsri33/embedflow/blob/main/docs/prewarming.md).
237
+
207
238
  ## Research
208
239
 
209
240
  For candidate depth `K`, EmbedFlow measures:
@@ -258,7 +289,7 @@ cover the remaining commands and endpoints.
258
289
 
259
290
  ## Status
260
291
 
261
- EmbedFlow v0.6.0 is a pre-1.0 release for research and early real-world
292
+ EmbedFlow v0.8.0 is a pre-1.0 release for research and early real-world
262
293
  testing. T2-v1 is an empirical finite-tail diagnostic, partial rankings can
263
294
  differ from fully warm target reranking, and ANN fidelity needs a reference
264
295
  comparison to audit.
@@ -156,6 +156,35 @@ embedflow plan --config ./embedflow.yaml --queries ./probe_queries.jsonl --forma
156
156
  `SAFE` is an empirical finite-tail signal, not a retrieval-quality guarantee.
157
157
  See [`docs/planner.md`](https://github.com/arnsri33/embedflow/blob/main/docs/planner.md).
158
158
 
159
+ Observe the reviewed migration path on real traffic without changing the
160
+ source result:
161
+
162
+ ```yaml
163
+ runtime: {mode: shadow}
164
+ shadow: {enabled: true, sample_rate: 0.10, candidate_k: 100, materialize: true}
165
+ ```
166
+
167
+ ```bash
168
+ embedflow serve --config ./embedflow.yaml
169
+ embedflow shadow report --config ./embedflow.yaml --since 24h
170
+ ```
171
+
172
+ Shadow Mode is source-authoritative, bounded, and advisory. It records cache,
173
+ coverage, latency, and ranking-disagreement diagnostics; it does not claim
174
+ retrieval-quality preservation without qrels and never routes canary traffic.
175
+ See [`docs/shadow-mode.md`](https://github.com/arnsri33/embedflow/blob/main/docs/shadow-mode.md).
176
+
177
+ After collecting Shadow traffic, prioritize uncached target vectors by observed
178
+ candidate popularity:
179
+
180
+ ```bash
181
+ embedflow prewarm plan --config ./embedflow.yaml --since 24h --max-docs 50000 --output prewarm-plan.json
182
+ embedflow prewarm run --config ./embedflow.yaml --plan prewarm-plan.json
183
+ ```
184
+
185
+ Observed candidate-occurrence coverage is an operational warming signal, not
186
+ retrieval quality or recall. See [`docs/prewarming.md`](https://github.com/arnsri33/embedflow/blob/main/docs/prewarming.md).
187
+
159
188
  Search responses expose `COLD`, `PARTIAL`, or `WARM`, cache hits and misses,
160
189
  synchronous work, queued work, and stage timings. Once the candidate vectors
161
190
  are warm, target scoring over that candidate set is deterministic.
@@ -263,13 +292,14 @@ OpenAPI documentation; see
263
292
  - [CLI reference](https://github.com/arnsri33/embedflow/blob/main/docs/cli.md)
264
293
  - [API](https://github.com/arnsri33/embedflow/blob/main/docs/api.md)
265
294
  - [Economics](https://github.com/arnsri33/embedflow/blob/main/docs/economics.md)
295
+ - [Shadow Mode](https://github.com/arnsri33/embedflow/blob/main/docs/shadow-mode.md)
266
296
  - [Limitations](https://github.com/arnsri33/embedflow/blob/main/docs/limitations.md)
267
297
  - [Contributing](https://github.com/arnsri33/embedflow/blob/main/CONTRIBUTING.md)
268
298
  - [Security](https://github.com/arnsri33/embedflow/blob/main/SECURITY.md)
269
299
 
270
300
  ## Status
271
301
 
272
- EmbedFlow v0.6.0 is a pre-1.0 release for research and early real-world
302
+ EmbedFlow v0.8.0 is a pre-1.0 release for research and early real-world
273
303
  testing.
274
304
 
275
305
  - T2-v1 reports an empirical finite-tail diagnostic.
@@ -138,6 +138,37 @@ behavior, candidate K, cache/economics projections, and staged rollout
138
138
  guidance. It never routes traffic or mutates the source index; `SAFE` is not a
139
139
  qrels-based retrieval-quality guarantee. See the [planner guide](https://github.com/arnsri33/embedflow/blob/main/docs/planner.md).
140
140
 
141
+ ### Source-authoritative Shadow Mode
142
+
143
+ Run the target migration path beside sampled traffic while always returning the
144
+ source result:
145
+
146
+ ```yaml
147
+ runtime: {mode: shadow}
148
+ shadow: {enabled: true, sample_rate: 0.10, candidate_k: 100, materialize: true}
149
+ ```
150
+
151
+ ```bash
152
+ embedflow serve --config ./embedflow.yaml
153
+ embedflow shadow report --config ./embedflow.yaml --since 24h --format json
154
+ ```
155
+
156
+ Shadow work is asynchronous, bounded, privacy-conscious, and failure isolated.
157
+ Reports contain operational and ranking-disagreement diagnostics—not qrel-based
158
+ quality guarantees—and recommendations never route canary traffic. See the
159
+ [Shadow Mode guide](https://github.com/arnsri33/embedflow/blob/main/docs/shadow-mode.md).
160
+
161
+ After collecting Shadow traffic, prioritize uncached target vectors by observed
162
+ candidate popularity:
163
+
164
+ ```bash
165
+ embedflow prewarm plan --config ./embedflow.yaml --since 24h --max-docs 50000 --output prewarm-plan.json
166
+ embedflow prewarm run --config ./embedflow.yaml --plan prewarm-plan.json
167
+ ```
168
+
169
+ Observed candidate-occurrence coverage is an operational warming signal, not
170
+ retrieval quality or recall. See the [prewarming guide](https://github.com/arnsri33/embedflow/blob/main/docs/prewarming.md).
171
+
141
172
  ## Research
142
173
 
143
174
  For candidate depth `K`, EmbedFlow measures:
@@ -192,7 +223,7 @@ cover the remaining commands and endpoints.
192
223
 
193
224
  ## Status
194
225
 
195
- EmbedFlow v0.6.0 is a pre-1.0 release for research and early real-world
226
+ EmbedFlow v0.8.0 is a pre-1.0 release for research and early real-world
196
227
  testing. T2-v1 is an empirical finite-tail diagnostic, partial rankings can
197
228
  differ from fully warm target reranking, and ANN fidelity needs a reference
198
229
  comparison to audit.
@@ -21,6 +21,7 @@ Interactive OpenAPI documentation is available at
21
21
  | GET | `/metrics` | Aggregated latency and queue metrics |
22
22
  | GET | `/plan` | Current migration plan |
23
23
  | GET | `/economics` | Configured economics estimate |
24
+ | GET | `/shadow/report` | Bounded Shadow Mode diagnostics |
24
25
 
25
26
  ## Search
26
27
 
@@ -48,6 +49,13 @@ The response includes the result list and migration fields such as:
48
49
  request. A partial response scores the available target vectors; it can differ
49
50
  from the fully warm ranking.
50
51
 
52
+ When `runtime.mode=shadow`, `/search` returns the source-only result and marks
53
+ `migration.source_authoritative=true`; target work is scheduled in the
54
+ background. `/shadow/report`, `/status`, and `/metrics` expose aggregate shadow
55
+ observations without raw query/document text or vectors. Shadow failures and
56
+ timeouts are isolated from the response. Ranking overlap is diagnostic and is
57
+ not a qrels-based quality claim.
58
+
51
59
  The advisory migration planner is exposed through the Python API and the
52
60
  `embedflow plan` CLI. It is intentionally not a synchronous FastAPI endpoint:
53
61
  probe analysis may load models and perform bounded candidate work, so operators
@@ -10,6 +10,10 @@ embedflow init --config ./embedflow.yaml
10
10
  embedflow analyze --config ./embedflow.yaml --output-dir ./analysis
11
11
  embedflow evaluate --config ./experiment.yaml --output-dir ./results
12
12
  embedflow plan --config ./embedflow.yaml --queries ./probe_queries.jsonl
13
+ embedflow shadow report --config ./embedflow.yaml --since 24h
14
+ embedflow prewarm plan --config ./embedflow.yaml --since 24h --max-docs 50000 --output prewarm-plan.json
15
+ embedflow prewarm run --config ./embedflow.yaml --plan prewarm-plan.json
16
+ embedflow prewarm status --config ./embedflow.yaml
13
17
  ```
14
18
 
15
19
  `analyze` is the no-target-index workflow. It uses probe queries and frozen
@@ -40,7 +44,19 @@ embedflow export-target --config ./embedflow.yaml --output-index ./target.index
40
44
  `status` reports cache coverage, hit/miss counters, queue depth, and
41
45
  materialization throughput. `audit-index` checks the source index against an
42
46
  exact/reference configuration where supported. `prewarm` schedules target
43
- document work; it does not change source-index results.
47
+ document work; it does not change source-index results. `prewarm plan` ranks
48
+ uncached IDs by privacy-safe Shadow candidate occurrence counts and is always
49
+ bounded by `--max-docs` (or the configured `prewarm.max_docs`). `prewarm run`
50
+ refuses mismatched/tampered fingerprints, skips vectors that became warm, and
51
+ uses the existing durable materializer. Observed candidate-occurrence coverage
52
+ is not retrieval quality or recall. Use `--docs-per-second` and
53
+ `--gpu-hourly-cost` only for explicitly labeled modeled economics.
54
+
55
+ Use `embedflow serve --mode shadow` to opt into source-authoritative Shadow
56
+ Mode for one process. `embedflow shadow report` reads the bounded telemetry
57
+ store and supports `--format text|json|yaml`, `--since`, `--output`, and
58
+ `--quiet`. A valid `DEFER`/`EXPAND_K` report is still an analytical success;
59
+ only invalid configuration or initialization returns a non-zero exit code.
44
60
 
45
61
  ## Registry and profiles
46
62
 
@@ -77,9 +77,50 @@ planner:
77
77
  corpus_name: null
78
78
  corpus_fingerprint: null
79
79
 
80
+ # Optional bounded traffic-aware target-cache prewarming defaults. CLI flags
81
+ # override these values for one invocation; there is no implicit full-corpus
82
+ # prewarm operation.
83
+ prewarm:
84
+ max_docs: 1000
85
+ strategy: traffic_hotset
86
+ target_observed_coverage: null
87
+ max_storage_gb: null
88
+ max_runtime_seconds: null
89
+ batch_size: null
90
+ max_retries: null
91
+
92
+ # Source-authoritative observation mode. ``runtime.mode: migration`` (the
93
+ # default) leaves Shadow Mode inactive. ``mode: shadow`` is an explicit opt-in.
94
+ runtime:
95
+ mode: migration # migration, normal, source, or shadow
96
+
97
+ shadow:
98
+ enabled: true
99
+ sample_rate: 0.10
100
+ sample_seed: 42
101
+ candidate_k: 100
102
+ materialize: true
103
+ max_inflight: 32
104
+ queue_capacity: 1000
105
+ timeout_ms: 10000
106
+ shutdown_grace_ms: 1000
107
+ telemetry:
108
+ enabled: true
109
+ path: ./.embedflow/shadow.sqlite3
110
+ retain_query_records: false
111
+ retain_query_text: false
112
+ max_records: 10000
113
+ retention_days: null
114
+ report_k: 10
115
+ min_target_coverage_for_ranking: 1.0
116
+
80
117
  state_path: ./embedflow_state.json
81
118
  ```
82
119
 
120
+ Shadow Mode always returns the source-authoritative result before target work
121
+ finishes. See [`shadow-mode.md`](shadow-mode.md) for queue, timeout, privacy,
122
+ materialization, and report semantics.
123
+
83
124
  ## Model contracts
84
125
 
85
126
  Model configuration can include `revision`, `dimension`, `max_length`,
@@ -164,6 +205,24 @@ EMBEDFLOW_PLANNER_LATENCY_BUDGET_MS
164
205
  EMBEDFLOW_PLANNER_ACCESS_TRACE
165
206
  EMBEDFLOW_PLANNER_CORPUS_NAME
166
207
  EMBEDFLOW_PLANNER_CORPUS_FINGERPRINT
208
+ EMBEDFLOW_RUNTIME_MODE
209
+ EMBEDFLOW_SHADOW_ENABLED
210
+ EMBEDFLOW_SHADOW_SAMPLE_RATE
211
+ EMBEDFLOW_SHADOW_SAMPLE_SEED
212
+ EMBEDFLOW_SHADOW_CANDIDATE_K
213
+ EMBEDFLOW_SHADOW_MATERIALIZE
214
+ EMBEDFLOW_SHADOW_MAX_INFLIGHT
215
+ EMBEDFLOW_SHADOW_QUEUE_CAPACITY
216
+ EMBEDFLOW_SHADOW_TIMEOUT_MS
217
+ EMBEDFLOW_SHADOW_SHUTDOWN_GRACE_MS
218
+ EMBEDFLOW_SHADOW_TELEMETRY_ENABLED
219
+ EMBEDFLOW_SHADOW_TELEMETRY_PATH
220
+ EMBEDFLOW_SHADOW_RETAIN_QUERY_RECORDS
221
+ EMBEDFLOW_SHADOW_RETAIN_QUERY_TEXT
222
+ EMBEDFLOW_SHADOW_MAX_RECORDS
223
+ EMBEDFLOW_SHADOW_REPORT_K
224
+ EMBEDFLOW_SHADOW_RETENTION_DAYS
225
+ EMBEDFLOW_SHADOW_MIN_TARGET_COVERAGE
167
226
  ```
168
227
 
169
228
  ## Input files
@@ -1,6 +1,6 @@
1
1
  # Limitations and release scope
2
2
 
3
- EmbedFlow v0.6.0 is a pre-1.0 release for research and early real-world
3
+ EmbedFlow v0.8.0 is a pre-1.0 release for research and early real-world
4
4
  testing. The serving path is designed to make migration experiments concrete;
5
5
  production rollout still requires application-specific validation.
6
6
 
@@ -0,0 +1,84 @@
1
+ # Traffic-aware prewarming
2
+
3
+ Traffic-aware prewarming uses privacy-safe Shadow Mode aggregates to select the
4
+ uncached target documents that occurred most often in the source candidate
5
+ pool. It is a bounded operational warming tool in the existing migration
6
+ workflow:
7
+
8
+ ```text
9
+ PLAN -> SHADOW -> PREWARM -> CANARY -> MIGRATE
10
+ ```
11
+
12
+ It never writes the source index. `prewarm plan` only reads Shadow telemetry
13
+ and the target cache. `prewarm run` delegates document lookup, target encoding,
14
+ and cache writes to the existing materialization worker.
15
+
16
+ ## Plan and run
17
+
18
+ ```bash
19
+ embedflow prewarm plan \
20
+ --config embedflow.yaml --since 24h --max-docs 50000 \
21
+ --output prewarm-plan.json
22
+ embedflow prewarm run --config embedflow.yaml --plan prewarm-plan.json
23
+ embedflow prewarm status --config embedflow.yaml
24
+ ```
25
+
26
+ Plans are schema-versioned (`schema_version: 1`), deterministically ranked by
27
+ `candidate_occurrences DESC, document_id ASC`, and fingerprinted with the
28
+ source/target contracts, source-index identity, candidate K, and Shadow
29
+ configuration. Execution refuses a stale or tampered plan. A plan is always
30
+ bounded by `max_docs`; there is no implicit full-corpus operation.
31
+
32
+ Optional configuration defaults are available under `prewarm`:
33
+
34
+ ```yaml
35
+ prewarm:
36
+ max_docs: 1000
37
+ strategy: traffic_hotset
38
+ batch_size: 32
39
+ max_retries: 3
40
+ ```
41
+
42
+ `--target-observed-coverage`, `--max-storage-gb`, and `--max-runtime` (when a
43
+ real measured/user-supplied `--docs-per-second` is supplied) further reduce
44
+ the hard selection cap. `--gpu-hourly-cost` (or the existing planner/economics
45
+ configuration value) adds a user-supplied modeled cost. Runtime, cost, and
46
+ storage values are explicitly marked as modeled/unknown; no cloud price or
47
+ throughput is guessed.
48
+
49
+ ## Coverage and privacy
50
+
51
+ The plan reports **observed candidate-occurrence coverage**:
52
+
53
+ ```text
54
+ occurrences with a valid target vector / all source candidate occurrences
55
+ ```
56
+
57
+ This is not retrieval coverage, recall, nDCG, or a guarantee of migration
58
+ quality. It only describes the selected Shadow window and candidate K. Ties are
59
+ stable, and current target-cache entries count as warm only when their target
60
+ model and (when document text is available) content fingerprints match. Legacy
61
+ cache rows created before content bindings were introduced remain usable but
62
+ are treated as content-unknown.
63
+
64
+ Shadow aggregation stores document IDs and counters, not raw query text,
65
+ document text, vectors, or credentials. Treat the generated plan/ID list as
66
+ application data because selected document IDs are needed to execute it.
67
+
68
+ ## Resume and idempotence
69
+
70
+ The runner stores a plan-specific queue/state beside the target cache and uses
71
+ the target cache as the execution source of truth. Re-running a completed plan
72
+ skips already-warm vectors. Interrupts leave completed vectors valid and a
73
+ subsequent run can continue the remaining IDs. Documents that disappear or
74
+ change are resolved again at run time; stale content is not silently reused.
75
+
76
+ Two concurrent runners use the existing durable queue/cache deduplication. The
77
+ queue is at-least-once under process crashes, so operators should inspect
78
+ `prewarm status` and the run report for failures.
79
+
80
+ The source FAISS/Qdrant/pgvector/Pinecone/Milvus/Weaviate index is never
81
+ created, altered, rebuilt, upserted, or deleted by this feature. The API for
82
+ prewarming is currently Python/core plus CLI; dashboard execution controls are
83
+ intentionally not added to avoid turning an advisory plan into autonomous
84
+ traffic routing.
@@ -18,8 +18,8 @@ python -m twine check dist/*
18
18
  Inspect both archives before uploading:
19
19
 
20
20
  ```bash
21
- unzip -l dist/embedflow-0.6.0-py3-none-any.whl
22
- tar -tzf dist/embedflow-0.6.0.tar.gz
21
+ unzip -l dist/embedflow-0.8.0-py3-none-any.whl
22
+ tar -tzf dist/embedflow-0.8.0.tar.gz
23
23
  sha256sum dist/*
24
24
  ```
25
25
 
@@ -32,7 +32,7 @@ Test the wheel outside the source tree:
32
32
  ```bash
33
33
  python -m venv /tmp/embedflow-wheel-test
34
34
  /tmp/embedflow-wheel-test/bin/python -m pip install --upgrade pip
35
- /tmp/embedflow-wheel-test/bin/python -m pip install dist/embedflow-0.6.0-py3-none-any.whl
35
+ /tmp/embedflow-wheel-test/bin/python -m pip install dist/embedflow-0.8.0-py3-none-any.whl
36
36
  cd /tmp
37
37
  /tmp/embedflow-wheel-test/bin/python -c "import embedflow; print(embedflow.__version__)"
38
38
  /tmp/embedflow-wheel-test/bin/embedflow --help
@@ -65,7 +65,7 @@ python -m venv /tmp/embedflow-testpypi
65
65
  /tmp/embedflow-testpypi/bin/python -m pip install \
66
66
  --index-url https://test.pypi.org/simple/ \
67
67
  --extra-index-url https://pypi.org/simple/ \
68
- embedflow==0.6.0
68
+ embedflow==0.8.0
69
69
  cd /tmp
70
70
  /tmp/embedflow-testpypi/bin/python -c "import embedflow; print(embedflow.__version__)"
71
71
  /tmp/embedflow-testpypi/bin/embedflow --help
@@ -79,11 +79,11 @@ Test optional integrations in a second clean environment:
79
79
  /tmp/embedflow-testpypi/bin/python -m pip install \
80
80
  --index-url https://test.pypi.org/simple/ \
81
81
  --extra-index-url https://pypi.org/simple/ \
82
- "embedflow[faiss,dashboard,pinecone,milvus,weaviate]==0.6.0"
82
+ "embedflow[faiss,dashboard,pinecone,milvus,weaviate]==0.8.0"
83
83
  ```
84
84
 
85
- If the same filename already exists on TestPyPI, use a pre-release such as
86
- `0.6.0rc1` for the TestPyPI-only trial. Keep production `0.6.0` unchanged.
85
+ If the same filename already exists on TestPyPI, use a pre-release for the
86
+ TestPyPI-only trial. Never overwrite the production version.
87
87
 
88
88
  ## Trusted Publishing configuration
89
89
 
@@ -115,7 +115,7 @@ above keeps the test step explicit.
115
115
  3. Run the final release gate and review the generated report.
116
116
  4. Configure the PyPI pending publisher and protected `pypi` environment.
117
117
  5. Create a Git tag and GitHub Release for the exact package version, for
118
- example `v0.6.0`.
118
+ example `v0.8.0`.
119
119
  6. Approve the `pypi` environment when the release workflow is ready.
120
120
  7. Verify the files and metadata on PyPI.
121
121
  8. Install from production PyPI in a directory outside this checkout.
@@ -0,0 +1,104 @@
1
+ # Shadow Mode
2
+
3
+ Shadow Mode runs the configured target migration path beside real traffic while
4
+ the existing source result remains authoritative. It is an observation and
5
+ cache-warming tool, not a traffic router:
6
+
7
+ ```yaml
8
+ runtime:
9
+ mode: shadow
10
+
11
+ shadow:
12
+ enabled: true
13
+ sample_rate: 0.10
14
+ sample_seed: 42
15
+ candidate_k: 100
16
+ materialize: true
17
+ max_inflight: 32
18
+ queue_capacity: 1000
19
+ timeout_ms: 10000
20
+ telemetry:
21
+ enabled: true
22
+ path: ./.embedflow/shadow.sqlite3
23
+ retain_query_records: false
24
+ retain_query_text: false
25
+ ```
26
+
27
+ Install the optional backend/model extras required by the selected
28
+ configuration, then start the ordinary service:
29
+
30
+ ```bash
31
+ pip install "embedflow[dashboard]"
32
+ embedflow serve --config embedflow.yaml
33
+ # or override the file for one process:
34
+ embedflow serve --config embedflow.yaml --mode shadow
35
+ ```
36
+
37
+ The request path performs source encoding, source ANN retrieval, and document
38
+ resolution exactly as source-only serving. It returns that result without
39
+ waiting for target encoding, cache misses, reranking, telemetry, or the
40
+ materialization worker. Shadow jobs use a bounded queue and worker pool. A full
41
+ queue drops only shadow work; a target/model/cache/telemetry failure or timeout
42
+ is recorded and cannot change the primary response.
43
+
44
+ If the target model cannot be loaded during startup, source/shadow serving
45
+ still opens with target work marked unavailable; sampled requests record the
46
+ isolated target-encoding failure. Normal migration mode remains fail-fast for
47
+ the same startup error.
48
+
49
+ `candidate_k` is explicit. The planner may suggest a value, but Shadow Mode
50
+ does not silently select one. `materialize: false` reads already cached target
51
+ vectors without changing cache or queue state. With `materialize: true`, cold
52
+ candidate IDs are deduplicated in the existing persistent materialization queue
53
+ and warmed asynchronously; synchronous target misses remain zero in Shadow
54
+ Mode. A comparison with incomplete target vectors is reported as `partial`,
55
+ with target coverage shown separately.
56
+
57
+ ## Reports
58
+
59
+ Reports use the privacy-preserving SQLite telemetry store (by default beside
60
+ the configured cache):
61
+
62
+ ```bash
63
+ embedflow shadow report --config embedflow.yaml --since 24h
64
+ embedflow shadow report --config embedflow.yaml --since 24h --format json --output shadow-report.json
65
+ ```
66
+
67
+ Use `/shadow/report` or the `shadow` section of `/status` and `/metrics` in the
68
+ FastAPI service. JSON/YAML stdout contains only the requested artifact; progress
69
+ and errors belong on stderr. `--since` accepts seconds or `s`, `m`, `h`, `d`,
70
+ and `w` suffixes.
71
+
72
+ Reports contain sampled/completed/partial/failed/timed-out/dropped counts,
73
+ cache and materialization accounting, source and shadow latency distributions,
74
+ target coverage, top-1 agreement, and top-k overlap. Shadow latency is not
75
+ user-facing latency because it is off the primary critical path. Ranking
76
+ disagreement and overlap are diagnostics, not recall, nDCG, or quality loss;
77
+ without qrels/native-target evaluation no quality guarantee is made.
78
+
79
+ T2-v1 remains the frozen window-level diagnostic. EmbedFlow does not label an
80
+ individual production query `T2 SAFE`. A report recommendation is operational
81
+ guidance only:
82
+
83
+ - `CONTINUE_SHADOW` means evidence or coverage is still limited.
84
+ - `EXPAND_K` means the accumulated T2 window requests a larger candidate pool.
85
+ - `INVESTIGATE` means failures, timeouts, or uncertain T2 behavior need review.
86
+ - `READY_FOR_CANARY_EVALUATION` means an operator may consider a separately
87
+ designed canary; it never routes traffic automatically.
88
+
89
+ Raw query text, candidate text, source vectors, target vectors, and credentials
90
+ are not persisted in telemetry. Query IDs may be retained only when explicitly
91
+ enabled, and retention is bounded by `max_records`. The telemetry database is
92
+ segmented by source/target contract and candidate-K fingerprint so unrelated
93
+ migrations are not mixed.
94
+
95
+ On shutdown EmbedFlow stops accepting new shadow work, drops queued jobs when
96
+ necessary, and waits only the configured bounded grace period. A corrupted or
97
+ unwritable telemetry file degrades observability; it does not take source
98
+ serving down. Source indexes are read-only during Shadow Mode. The planner and
99
+ Shadow Mode are separate: generate a plan first, then copy its recommended K
100
+ explicitly into a reviewed Shadow configuration.
101
+
102
+ Shadow Mode is advisory and experimental operational instrumentation. It does
103
+ not provide autonomous rollout, rollback, qrel evaluation, or ANN-fidelity
104
+ proof.
@@ -0,0 +1,59 @@
1
+ """EmbedFlow: progressive embedding-model migration for existing indexes."""
2
+
3
+ __version__ = "0.8.0"
4
+
5
+ from .config import (
6
+ EmbedFlowConfig,
7
+ PlannerConfig,
8
+ PrewarmConfig,
9
+ RuntimeConfig,
10
+ ShadowConfig,
11
+ ShadowTelemetryConfig,
12
+ load_config,
13
+ )
14
+ from .prewarm import PrewarmPlan, PrewarmPlanner, PrewarmRunner, TrafficHotsetPlanner, load_prewarm_plan
15
+
16
+
17
+ def plan(*args, **kwargs):
18
+ """Generate an advisory migration plan without changing source traffic."""
19
+ from .planner import plan_migration
20
+ return plan_migration(*args, **kwargs)
21
+
22
+
23
+ def migrate(*args, **kwargs):
24
+ """Start progressive migration over an existing FAISS, Qdrant, pgvector, Pinecone, Milvus, or Weaviate index.
25
+
26
+ Imported lazily to keep the lightweight configuration package free of
27
+ model-serving dependencies at import time. See ``embedflow.migration``
28
+ for the ``MigrationSession`` type.
29
+ """
30
+ from .migration.facade import migrate as _migrate
31
+ return _migrate(*args, **kwargs)
32
+
33
+
34
+ def analyze_migration(*args, **kwargs):
35
+ """Run the leakage-safe no-target-index analysis programmatically."""
36
+ from .analysis import analyze_migration as _analyze_migration
37
+ return _analyze_migration(*args, **kwargs)
38
+
39
+
40
+ def prewarm_plan(telemetry, cache, **kwargs):
41
+ """Build a bounded traffic-hotset plan through the reusable API.
42
+
43
+ Planner-constructor options and ``plan`` options may be supplied together;
44
+ recognized plan options are routed to :meth:`PrewarmPlanner.plan`.
45
+ """
46
+ plan_keys = {"since_seconds", "start", "end", "max_docs", "target_observed_coverage",
47
+ "max_storage_gb", "max_runtime_seconds", "docs_per_second", "gpu_hourly_cost", "strategy"}
48
+ plan_kwargs = {key: kwargs.pop(key) for key in tuple(kwargs) if key in plan_keys}
49
+ return PrewarmPlanner(telemetry, cache, **kwargs).plan(**plan_kwargs)
50
+
51
+
52
+ def prewarm_run(plan, *args, **kwargs):
53
+ """Execute a validated prewarm plan through the existing materializer."""
54
+ run_keys = {"max_runtime_seconds", "progress"}
55
+ run_kwargs = {key: kwargs.pop(key) for key in tuple(kwargs) if key in run_keys}
56
+ return PrewarmRunner(*args, **kwargs).run(plan, **run_kwargs)
57
+
58
+
59
+ __all__ = ["EmbedFlowConfig", "PlannerConfig", "PrewarmConfig", "RuntimeConfig", "ShadowConfig", "ShadowTelemetryConfig", "load_config", "migrate", "plan", "analyze_migration", "prewarm_plan", "prewarm_run", "PrewarmPlan", "PrewarmPlanner", "TrafficHotsetPlanner", "PrewarmRunner", "load_prewarm_plan", "__version__"]
@@ -8,15 +8,15 @@ import numpy as np
8
8
 
9
9
  class TargetVectorCache(ABC):
10
10
  @abstractmethod
11
- def get(self, document_ids: list[str]) -> dict[str, np.ndarray]:
11
+ def get(self, document_ids: list[str], content_fingerprints: Any = None) -> dict[str, np.ndarray]:
12
12
  raise NotImplementedError
13
13
 
14
14
  @abstractmethod
15
- def put(self, document_ids: list[str], vectors: np.ndarray) -> None:
15
+ def put(self, document_ids: list[str], vectors: np.ndarray, content_fingerprints: Any = None) -> None:
16
16
  raise NotImplementedError
17
17
 
18
18
  @abstractmethod
19
- def contains(self, document_ids: list[str]) -> set[str]:
19
+ def contains(self, document_ids: list[str], content_fingerprints: Any = None) -> set[str]:
20
20
  raise NotImplementedError
21
21
 
22
22
  @abstractmethod