embedflow 0.6.0__tar.gz → 0.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {embedflow-0.6.0 → embedflow-0.8.0}/CHANGELOG.md +24 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/CITATION.cff +2 -2
- {embedflow-0.6.0 → embedflow-0.8.0}/PKG-INFO +33 -2
- {embedflow-0.6.0 → embedflow-0.8.0}/README.md +31 -1
- {embedflow-0.6.0 → embedflow-0.8.0}/README_PYPI.md +32 -1
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/api.md +8 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/cli.md +17 -1
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/configuration.md +59 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/limitations.md +1 -1
- embedflow-0.8.0/docs/prewarming.md +84 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/releasing.md +8 -8
- embedflow-0.8.0/docs/shadow-mode.md +104 -0
- embedflow-0.8.0/embedflow/__init__.py +59 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/cache/base.py +3 -3
- embedflow-0.8.0/embedflow/cache/persistent_cache.py +336 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/cli.py +516 -13
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/config.py +250 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/migration/facade.py +50 -4
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/migration/materializer.py +124 -6
- embedflow-0.8.0/embedflow/prewarm/__init__.py +23 -0
- embedflow-0.8.0/embedflow/prewarm/models.py +181 -0
- embedflow-0.8.0/embedflow/prewarm/planner.py +479 -0
- embedflow-0.8.0/embedflow/prewarm/runner.py +450 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/runtime.py +60 -2
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/serving/api.py +26 -3
- embedflow-0.8.0/embedflow/serving/engine.py +695 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/serving/schemas.py +1 -0
- embedflow-0.8.0/embedflow/shadow/__init__.py +9 -0
- embedflow-0.8.0/embedflow/shadow/models.py +65 -0
- embedflow-0.8.0/embedflow/shadow/report.py +48 -0
- embedflow-0.8.0/embedflow/shadow/runner.py +356 -0
- embedflow-0.8.0/embedflow/shadow/telemetry.py +1070 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow.egg-info/PKG-INFO +33 -2
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow.egg-info/SOURCES.txt +19 -0
- embedflow-0.8.0/examples/prewarm/README.md +16 -0
- embedflow-0.8.0/examples/prewarm/demo.py +74 -0
- embedflow-0.8.0/examples/prewarm/embedflow.yaml.example +24 -0
- embedflow-0.8.0/examples/prewarm/run_demo.sh +12 -0
- embedflow-0.8.0/examples/shadow/README.md +13 -0
- embedflow-0.8.0/examples/shadow/embedflow.yaml.example +25 -0
- embedflow-0.8.0/examples/shadow/run_demo.py +95 -0
- embedflow-0.8.0/examples/shadow/run_demo.sh +10 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/pyproject.toml +1 -1
- {embedflow-0.6.0 → embedflow-0.8.0}/scripts/release_gate.py +16 -4
- embedflow-0.6.0/embedflow/__init__.py +0 -31
- embedflow-0.6.0/embedflow/cache/persistent_cache.py +0 -198
- embedflow-0.6.0/embedflow/serving/engine.py +0 -228
- {embedflow-0.6.0 → embedflow-0.8.0}/CONTRIBUTING.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/LICENSE +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/MANIFEST.in +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/SECURITY.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/assets/README.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/assets/candidate-gap-example.svg +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/assets/dashboard-screenshot.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/assets/terminal-demo.txt +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/concepts.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/contributing-benchmarks.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/economics.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/installation.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/integrations/faiss.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/integrations/milvus.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/integrations/pgvector.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/integrations/pinecone.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/integrations/qdrant.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/integrations/weaviate.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/methodology.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/planner.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/quickstart.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/docs/registry.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/__main__.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/analysis.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/api.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/cache/__init__.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/compatibility/__init__.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/compatibility/candidate_gap.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/compatibility/containment.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/compatibility/evaluate.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/compatibility/metrics.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/compatibility/migration_depth.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/compatibility/probe.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/compatibility/report.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/compatibility/t2.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/data/__init__.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/data/registry/__init__.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/data/registry/benchmark_profiles.jsonl +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/data/registry/checksums.sha256 +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/data/registry/migrations.jsonl +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/data/registry/registry_manifest.json +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/data/registry/research_summaries.json +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/data/registry/schema_version.json +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/frozen/T2_V1_FROZEN_SPEC.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/frozen/T2_V1_FROZEN_SPEC.sha256 +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/indexes/__init__.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/indexes/base.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/indexes/faiss_backend.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/indexes/milvus_backend.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/indexes/pgvector_backend.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/indexes/pinecone_backend.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/indexes/qdrant_backend.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/indexes/weaviate_backend.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/metrics/__init__.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/metrics/latency.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/migration/__init__.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/migration/compatibility.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/migration/planner.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/migration/state.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/models/__init__.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/models/base.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/models/huggingface.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/planner/__init__.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/planner/economics.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/planner/models.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/planner/planner.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/planner/rendering.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/registry/__init__.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/registry/loader.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/registry/matcher.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/registry/schema.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/serving/__init__.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow/serving/factory.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow.egg-info/dependency_links.txt +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow.egg-info/entry_points.txt +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow.egg-info/requires.txt +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/embedflow.egg-info/top_level.txt +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/faiss/README.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/faiss/documents.jsonl +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/faiss/embedflow.yaml +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/faiss/queries.jsonl +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/milvus/README.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/milvus/compose.yaml +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/milvus/embedflow.yaml.example +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/milvus/run_demo.sh +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/pgvector/README.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/pgvector/build_index.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/pgvector/compose.yaml +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/pgvector/embedflow.yaml +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/pgvector/init.sql +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/pgvector/queries.jsonl +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/pgvector/run_demo.sh +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/pinecone/README.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/pinecone/embedflow.yaml.example +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/pinecone/run_smoke.sh +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/planner/README.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/planner/run_demo.sh +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/qdrant/README.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/qdrant/build_index.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/qdrant/documents.jsonl +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/qdrant/embedflow.yaml +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/qdrant/queries.jsonl +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/research_analysis/README.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/research_analysis/documents.jsonl +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/research_analysis/embedflow.yaml +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/research_analysis/qrels.json +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/research_analysis/queries.jsonl +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/weaviate/README.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/weaviate/compose.yaml +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/weaviate/embedflow.yaml.example +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/examples/weaviate/run_demo.sh +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/frozen/T2_V1_FROZEN_SPEC.md +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/frozen/T2_V1_FROZEN_SPEC.sha256 +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/requirements-dev.txt +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/requirements.txt +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/scripts/milvus_fixture.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/scripts/pinecone_smoke.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/scripts/real_qdrant_smoke.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/scripts/run_demo.sh +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/scripts/run_tests.sh +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/scripts/validate_milvus.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/scripts/validate_pgvector_10k.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/scripts/validate_weaviate.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/scripts/weaviate_fixture.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/scripts/weaviate_smoke.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/setup.cfg +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/src/__init__.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/src/embed.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/src/probe_features.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/src/storage.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/src/t2_v1.py +0 -0
- {embedflow-0.6.0 → embedflow-0.8.0}/src/utils.py +0 -0
|
@@ -1,5 +1,29 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## v0.8.0 — Traffic-aware target-vector prewarming
|
|
4
|
+
|
|
5
|
+
- Added bounded `prewarm plan`, `prewarm run`, and `prewarm status` workflows
|
|
6
|
+
that select uncached target documents by observed Shadow Mode candidate
|
|
7
|
+
occurrences.
|
|
8
|
+
- Added privacy-safe aggregate document popularity telemetry, content/model
|
|
9
|
+
fingerprint checks, deterministic tie-breaking, hard document/storage/time
|
|
10
|
+
budgets, atomic plan/run artifacts, and resumable materialization through the
|
|
11
|
+
existing target cache and worker.
|
|
12
|
+
- Observed candidate-occurrence coverage remains an operational warming signal;
|
|
13
|
+
it is not retrieval quality or recall evidence.
|
|
14
|
+
|
|
15
|
+
## v0.7.0 — Source-authoritative Shadow Mode
|
|
16
|
+
|
|
17
|
+
- Added bounded, deterministic, source-authoritative Shadow Mode for observing
|
|
18
|
+
target reranking on sampled traffic without delaying or changing source
|
|
19
|
+
responses.
|
|
20
|
+
- Added failure/timeout isolation, queue backpressure, optional asynchronous
|
|
21
|
+
target-cache materialization, target-coverage and ranking diagnostics, and
|
|
22
|
+
privacy-conscious SQLite telemetry.
|
|
23
|
+
- Added `embedflow shadow report`, API/status integration, documentation, and a
|
|
24
|
+
deterministic offline demonstration. Shadow reports remain operational
|
|
25
|
+
diagnostics and do not claim qrel-based retrieval quality.
|
|
26
|
+
|
|
3
27
|
## v0.6.0 — Migration planner
|
|
4
28
|
|
|
5
29
|
- Added an advisory `embedflow plan` command and Python API that combine
|
|
@@ -2,8 +2,8 @@ cff-version: 1.2.0
|
|
|
2
2
|
title: "EmbedFlow: Upgrading Legacy Embeddings Without Full Upfront Re-Embedding"
|
|
3
3
|
message: "If EmbedFlow contributes to your work, please cite this software release."
|
|
4
4
|
type: software
|
|
5
|
-
version: 0.
|
|
6
|
-
date-released: 2026-09-
|
|
5
|
+
version: 0.8.0
|
|
6
|
+
date-released: 2026-09-15
|
|
7
7
|
repository-code: "https://github.com/arnsri33/embedflow"
|
|
8
8
|
url: "https://github.com/arnsri33/embedflow"
|
|
9
9
|
license: AGPL-3.0-only
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: embedflow
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.8.0
|
|
4
4
|
Summary: Progressive embedding-model migration over existing vector indexes.
|
|
5
5
|
Author: Arnav Srivastav
|
|
6
6
|
License-Expression: AGPL-3.0-only
|
|
@@ -204,6 +204,37 @@ behavior, candidate K, cache/economics projections, and staged rollout
|
|
|
204
204
|
guidance. It never routes traffic or mutates the source index; `SAFE` is not a
|
|
205
205
|
qrels-based retrieval-quality guarantee. See the [planner guide](https://github.com/arnsri33/embedflow/blob/main/docs/planner.md).
|
|
206
206
|
|
|
207
|
+
### Source-authoritative Shadow Mode
|
|
208
|
+
|
|
209
|
+
Run the target migration path beside sampled traffic while always returning the
|
|
210
|
+
source result:
|
|
211
|
+
|
|
212
|
+
```yaml
|
|
213
|
+
runtime: {mode: shadow}
|
|
214
|
+
shadow: {enabled: true, sample_rate: 0.10, candidate_k: 100, materialize: true}
|
|
215
|
+
```
|
|
216
|
+
|
|
217
|
+
```bash
|
|
218
|
+
embedflow serve --config ./embedflow.yaml
|
|
219
|
+
embedflow shadow report --config ./embedflow.yaml --since 24h --format json
|
|
220
|
+
```
|
|
221
|
+
|
|
222
|
+
Shadow work is asynchronous, bounded, privacy-conscious, and failure isolated.
|
|
223
|
+
Reports contain operational and ranking-disagreement diagnostics—not qrel-based
|
|
224
|
+
quality guarantees—and recommendations never route canary traffic. See the
|
|
225
|
+
[Shadow Mode guide](https://github.com/arnsri33/embedflow/blob/main/docs/shadow-mode.md).
|
|
226
|
+
|
|
227
|
+
After collecting Shadow traffic, prioritize uncached target vectors by observed
|
|
228
|
+
candidate popularity:
|
|
229
|
+
|
|
230
|
+
```bash
|
|
231
|
+
embedflow prewarm plan --config ./embedflow.yaml --since 24h --max-docs 50000 --output prewarm-plan.json
|
|
232
|
+
embedflow prewarm run --config ./embedflow.yaml --plan prewarm-plan.json
|
|
233
|
+
```
|
|
234
|
+
|
|
235
|
+
Observed candidate-occurrence coverage is an operational warming signal, not
|
|
236
|
+
retrieval quality or recall. See the [prewarming guide](https://github.com/arnsri33/embedflow/blob/main/docs/prewarming.md).
|
|
237
|
+
|
|
207
238
|
## Research
|
|
208
239
|
|
|
209
240
|
For candidate depth `K`, EmbedFlow measures:
|
|
@@ -258,7 +289,7 @@ cover the remaining commands and endpoints.
|
|
|
258
289
|
|
|
259
290
|
## Status
|
|
260
291
|
|
|
261
|
-
EmbedFlow v0.
|
|
292
|
+
EmbedFlow v0.8.0 is a pre-1.0 release for research and early real-world
|
|
262
293
|
testing. T2-v1 is an empirical finite-tail diagnostic, partial rankings can
|
|
263
294
|
differ from fully warm target reranking, and ANN fidelity needs a reference
|
|
264
295
|
comparison to audit.
|
|
@@ -156,6 +156,35 @@ embedflow plan --config ./embedflow.yaml --queries ./probe_queries.jsonl --forma
|
|
|
156
156
|
`SAFE` is an empirical finite-tail signal, not a retrieval-quality guarantee.
|
|
157
157
|
See [`docs/planner.md`](https://github.com/arnsri33/embedflow/blob/main/docs/planner.md).
|
|
158
158
|
|
|
159
|
+
Observe the reviewed migration path on real traffic without changing the
|
|
160
|
+
source result:
|
|
161
|
+
|
|
162
|
+
```yaml
|
|
163
|
+
runtime: {mode: shadow}
|
|
164
|
+
shadow: {enabled: true, sample_rate: 0.10, candidate_k: 100, materialize: true}
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
```bash
|
|
168
|
+
embedflow serve --config ./embedflow.yaml
|
|
169
|
+
embedflow shadow report --config ./embedflow.yaml --since 24h
|
|
170
|
+
```
|
|
171
|
+
|
|
172
|
+
Shadow Mode is source-authoritative, bounded, and advisory. It records cache,
|
|
173
|
+
coverage, latency, and ranking-disagreement diagnostics; it does not claim
|
|
174
|
+
retrieval-quality preservation without qrels and never routes canary traffic.
|
|
175
|
+
See [`docs/shadow-mode.md`](https://github.com/arnsri33/embedflow/blob/main/docs/shadow-mode.md).
|
|
176
|
+
|
|
177
|
+
After collecting Shadow traffic, prioritize uncached target vectors by observed
|
|
178
|
+
candidate popularity:
|
|
179
|
+
|
|
180
|
+
```bash
|
|
181
|
+
embedflow prewarm plan --config ./embedflow.yaml --since 24h --max-docs 50000 --output prewarm-plan.json
|
|
182
|
+
embedflow prewarm run --config ./embedflow.yaml --plan prewarm-plan.json
|
|
183
|
+
```
|
|
184
|
+
|
|
185
|
+
Observed candidate-occurrence coverage is an operational warming signal, not
|
|
186
|
+
retrieval quality or recall. See [`docs/prewarming.md`](https://github.com/arnsri33/embedflow/blob/main/docs/prewarming.md).
|
|
187
|
+
|
|
159
188
|
Search responses expose `COLD`, `PARTIAL`, or `WARM`, cache hits and misses,
|
|
160
189
|
synchronous work, queued work, and stage timings. Once the candidate vectors
|
|
161
190
|
are warm, target scoring over that candidate set is deterministic.
|
|
@@ -263,13 +292,14 @@ OpenAPI documentation; see
|
|
|
263
292
|
- [CLI reference](https://github.com/arnsri33/embedflow/blob/main/docs/cli.md)
|
|
264
293
|
- [API](https://github.com/arnsri33/embedflow/blob/main/docs/api.md)
|
|
265
294
|
- [Economics](https://github.com/arnsri33/embedflow/blob/main/docs/economics.md)
|
|
295
|
+
- [Shadow Mode](https://github.com/arnsri33/embedflow/blob/main/docs/shadow-mode.md)
|
|
266
296
|
- [Limitations](https://github.com/arnsri33/embedflow/blob/main/docs/limitations.md)
|
|
267
297
|
- [Contributing](https://github.com/arnsri33/embedflow/blob/main/CONTRIBUTING.md)
|
|
268
298
|
- [Security](https://github.com/arnsri33/embedflow/blob/main/SECURITY.md)
|
|
269
299
|
|
|
270
300
|
## Status
|
|
271
301
|
|
|
272
|
-
EmbedFlow v0.
|
|
302
|
+
EmbedFlow v0.8.0 is a pre-1.0 release for research and early real-world
|
|
273
303
|
testing.
|
|
274
304
|
|
|
275
305
|
- T2-v1 reports an empirical finite-tail diagnostic.
|
|
@@ -138,6 +138,37 @@ behavior, candidate K, cache/economics projections, and staged rollout
|
|
|
138
138
|
guidance. It never routes traffic or mutates the source index; `SAFE` is not a
|
|
139
139
|
qrels-based retrieval-quality guarantee. See the [planner guide](https://github.com/arnsri33/embedflow/blob/main/docs/planner.md).
|
|
140
140
|
|
|
141
|
+
### Source-authoritative Shadow Mode
|
|
142
|
+
|
|
143
|
+
Run the target migration path beside sampled traffic while always returning the
|
|
144
|
+
source result:
|
|
145
|
+
|
|
146
|
+
```yaml
|
|
147
|
+
runtime: {mode: shadow}
|
|
148
|
+
shadow: {enabled: true, sample_rate: 0.10, candidate_k: 100, materialize: true}
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
```bash
|
|
152
|
+
embedflow serve --config ./embedflow.yaml
|
|
153
|
+
embedflow shadow report --config ./embedflow.yaml --since 24h --format json
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
Shadow work is asynchronous, bounded, privacy-conscious, and failure isolated.
|
|
157
|
+
Reports contain operational and ranking-disagreement diagnostics—not qrel-based
|
|
158
|
+
quality guarantees—and recommendations never route canary traffic. See the
|
|
159
|
+
[Shadow Mode guide](https://github.com/arnsri33/embedflow/blob/main/docs/shadow-mode.md).
|
|
160
|
+
|
|
161
|
+
After collecting Shadow traffic, prioritize uncached target vectors by observed
|
|
162
|
+
candidate popularity:
|
|
163
|
+
|
|
164
|
+
```bash
|
|
165
|
+
embedflow prewarm plan --config ./embedflow.yaml --since 24h --max-docs 50000 --output prewarm-plan.json
|
|
166
|
+
embedflow prewarm run --config ./embedflow.yaml --plan prewarm-plan.json
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
Observed candidate-occurrence coverage is an operational warming signal, not
|
|
170
|
+
retrieval quality or recall. See the [prewarming guide](https://github.com/arnsri33/embedflow/blob/main/docs/prewarming.md).
|
|
171
|
+
|
|
141
172
|
## Research
|
|
142
173
|
|
|
143
174
|
For candidate depth `K`, EmbedFlow measures:
|
|
@@ -192,7 +223,7 @@ cover the remaining commands and endpoints.
|
|
|
192
223
|
|
|
193
224
|
## Status
|
|
194
225
|
|
|
195
|
-
EmbedFlow v0.
|
|
226
|
+
EmbedFlow v0.8.0 is a pre-1.0 release for research and early real-world
|
|
196
227
|
testing. T2-v1 is an empirical finite-tail diagnostic, partial rankings can
|
|
197
228
|
differ from fully warm target reranking, and ANN fidelity needs a reference
|
|
198
229
|
comparison to audit.
|
|
@@ -21,6 +21,7 @@ Interactive OpenAPI documentation is available at
|
|
|
21
21
|
| GET | `/metrics` | Aggregated latency and queue metrics |
|
|
22
22
|
| GET | `/plan` | Current migration plan |
|
|
23
23
|
| GET | `/economics` | Configured economics estimate |
|
|
24
|
+
| GET | `/shadow/report` | Bounded Shadow Mode diagnostics |
|
|
24
25
|
|
|
25
26
|
## Search
|
|
26
27
|
|
|
@@ -48,6 +49,13 @@ The response includes the result list and migration fields such as:
|
|
|
48
49
|
request. A partial response scores the available target vectors; it can differ
|
|
49
50
|
from the fully warm ranking.
|
|
50
51
|
|
|
52
|
+
When `runtime.mode=shadow`, `/search` returns the source-only result and marks
|
|
53
|
+
`migration.source_authoritative=true`; target work is scheduled in the
|
|
54
|
+
background. `/shadow/report`, `/status`, and `/metrics` expose aggregate shadow
|
|
55
|
+
observations without raw query/document text or vectors. Shadow failures and
|
|
56
|
+
timeouts are isolated from the response. Ranking overlap is diagnostic and is
|
|
57
|
+
not a qrels-based quality claim.
|
|
58
|
+
|
|
51
59
|
The advisory migration planner is exposed through the Python API and the
|
|
52
60
|
`embedflow plan` CLI. It is intentionally not a synchronous FastAPI endpoint:
|
|
53
61
|
probe analysis may load models and perform bounded candidate work, so operators
|
|
@@ -10,6 +10,10 @@ embedflow init --config ./embedflow.yaml
|
|
|
10
10
|
embedflow analyze --config ./embedflow.yaml --output-dir ./analysis
|
|
11
11
|
embedflow evaluate --config ./experiment.yaml --output-dir ./results
|
|
12
12
|
embedflow plan --config ./embedflow.yaml --queries ./probe_queries.jsonl
|
|
13
|
+
embedflow shadow report --config ./embedflow.yaml --since 24h
|
|
14
|
+
embedflow prewarm plan --config ./embedflow.yaml --since 24h --max-docs 50000 --output prewarm-plan.json
|
|
15
|
+
embedflow prewarm run --config ./embedflow.yaml --plan prewarm-plan.json
|
|
16
|
+
embedflow prewarm status --config ./embedflow.yaml
|
|
13
17
|
```
|
|
14
18
|
|
|
15
19
|
`analyze` is the no-target-index workflow. It uses probe queries and frozen
|
|
@@ -40,7 +44,19 @@ embedflow export-target --config ./embedflow.yaml --output-index ./target.index
|
|
|
40
44
|
`status` reports cache coverage, hit/miss counters, queue depth, and
|
|
41
45
|
materialization throughput. `audit-index` checks the source index against an
|
|
42
46
|
exact/reference configuration where supported. `prewarm` schedules target
|
|
43
|
-
document work; it does not change source-index results.
|
|
47
|
+
document work; it does not change source-index results. `prewarm plan` ranks
|
|
48
|
+
uncached IDs by privacy-safe Shadow candidate occurrence counts and is always
|
|
49
|
+
bounded by `--max-docs` (or the configured `prewarm.max_docs`). `prewarm run`
|
|
50
|
+
refuses mismatched/tampered fingerprints, skips vectors that became warm, and
|
|
51
|
+
uses the existing durable materializer. Observed candidate-occurrence coverage
|
|
52
|
+
is not retrieval quality or recall. Use `--docs-per-second` and
|
|
53
|
+
`--gpu-hourly-cost` only for explicitly labeled modeled economics.
|
|
54
|
+
|
|
55
|
+
Use `embedflow serve --mode shadow` to opt into source-authoritative Shadow
|
|
56
|
+
Mode for one process. `embedflow shadow report` reads the bounded telemetry
|
|
57
|
+
store and supports `--format text|json|yaml`, `--since`, `--output`, and
|
|
58
|
+
`--quiet`. A valid `DEFER`/`EXPAND_K` report is still an analytical success;
|
|
59
|
+
only invalid configuration or initialization returns a non-zero exit code.
|
|
44
60
|
|
|
45
61
|
## Registry and profiles
|
|
46
62
|
|
|
@@ -77,9 +77,50 @@ planner:
|
|
|
77
77
|
corpus_name: null
|
|
78
78
|
corpus_fingerprint: null
|
|
79
79
|
|
|
80
|
+
# Optional bounded traffic-aware target-cache prewarming defaults. CLI flags
|
|
81
|
+
# override these values for one invocation; there is no implicit full-corpus
|
|
82
|
+
# prewarm operation.
|
|
83
|
+
prewarm:
|
|
84
|
+
max_docs: 1000
|
|
85
|
+
strategy: traffic_hotset
|
|
86
|
+
target_observed_coverage: null
|
|
87
|
+
max_storage_gb: null
|
|
88
|
+
max_runtime_seconds: null
|
|
89
|
+
batch_size: null
|
|
90
|
+
max_retries: null
|
|
91
|
+
|
|
92
|
+
# Source-authoritative observation mode. ``runtime.mode: migration`` (the
|
|
93
|
+
# default) leaves Shadow Mode inactive. ``mode: shadow`` is an explicit opt-in.
|
|
94
|
+
runtime:
|
|
95
|
+
mode: migration # migration, normal, source, or shadow
|
|
96
|
+
|
|
97
|
+
shadow:
|
|
98
|
+
enabled: true
|
|
99
|
+
sample_rate: 0.10
|
|
100
|
+
sample_seed: 42
|
|
101
|
+
candidate_k: 100
|
|
102
|
+
materialize: true
|
|
103
|
+
max_inflight: 32
|
|
104
|
+
queue_capacity: 1000
|
|
105
|
+
timeout_ms: 10000
|
|
106
|
+
shutdown_grace_ms: 1000
|
|
107
|
+
telemetry:
|
|
108
|
+
enabled: true
|
|
109
|
+
path: ./.embedflow/shadow.sqlite3
|
|
110
|
+
retain_query_records: false
|
|
111
|
+
retain_query_text: false
|
|
112
|
+
max_records: 10000
|
|
113
|
+
retention_days: null
|
|
114
|
+
report_k: 10
|
|
115
|
+
min_target_coverage_for_ranking: 1.0
|
|
116
|
+
|
|
80
117
|
state_path: ./embedflow_state.json
|
|
81
118
|
```
|
|
82
119
|
|
|
120
|
+
Shadow Mode always returns the source-authoritative result before target work
|
|
121
|
+
finishes. See [`shadow-mode.md`](shadow-mode.md) for queue, timeout, privacy,
|
|
122
|
+
materialization, and report semantics.
|
|
123
|
+
|
|
83
124
|
## Model contracts
|
|
84
125
|
|
|
85
126
|
Model configuration can include `revision`, `dimension`, `max_length`,
|
|
@@ -164,6 +205,24 @@ EMBEDFLOW_PLANNER_LATENCY_BUDGET_MS
|
|
|
164
205
|
EMBEDFLOW_PLANNER_ACCESS_TRACE
|
|
165
206
|
EMBEDFLOW_PLANNER_CORPUS_NAME
|
|
166
207
|
EMBEDFLOW_PLANNER_CORPUS_FINGERPRINT
|
|
208
|
+
EMBEDFLOW_RUNTIME_MODE
|
|
209
|
+
EMBEDFLOW_SHADOW_ENABLED
|
|
210
|
+
EMBEDFLOW_SHADOW_SAMPLE_RATE
|
|
211
|
+
EMBEDFLOW_SHADOW_SAMPLE_SEED
|
|
212
|
+
EMBEDFLOW_SHADOW_CANDIDATE_K
|
|
213
|
+
EMBEDFLOW_SHADOW_MATERIALIZE
|
|
214
|
+
EMBEDFLOW_SHADOW_MAX_INFLIGHT
|
|
215
|
+
EMBEDFLOW_SHADOW_QUEUE_CAPACITY
|
|
216
|
+
EMBEDFLOW_SHADOW_TIMEOUT_MS
|
|
217
|
+
EMBEDFLOW_SHADOW_SHUTDOWN_GRACE_MS
|
|
218
|
+
EMBEDFLOW_SHADOW_TELEMETRY_ENABLED
|
|
219
|
+
EMBEDFLOW_SHADOW_TELEMETRY_PATH
|
|
220
|
+
EMBEDFLOW_SHADOW_RETAIN_QUERY_RECORDS
|
|
221
|
+
EMBEDFLOW_SHADOW_RETAIN_QUERY_TEXT
|
|
222
|
+
EMBEDFLOW_SHADOW_MAX_RECORDS
|
|
223
|
+
EMBEDFLOW_SHADOW_REPORT_K
|
|
224
|
+
EMBEDFLOW_SHADOW_RETENTION_DAYS
|
|
225
|
+
EMBEDFLOW_SHADOW_MIN_TARGET_COVERAGE
|
|
167
226
|
```
|
|
168
227
|
|
|
169
228
|
## Input files
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Limitations and release scope
|
|
2
2
|
|
|
3
|
-
EmbedFlow v0.
|
|
3
|
+
EmbedFlow v0.8.0 is a pre-1.0 release for research and early real-world
|
|
4
4
|
testing. The serving path is designed to make migration experiments concrete;
|
|
5
5
|
production rollout still requires application-specific validation.
|
|
6
6
|
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
# Traffic-aware prewarming
|
|
2
|
+
|
|
3
|
+
Traffic-aware prewarming uses privacy-safe Shadow Mode aggregates to select the
|
|
4
|
+
uncached target documents that occurred most often in the source candidate
|
|
5
|
+
pool. It is a bounded operational warming tool in the existing migration
|
|
6
|
+
workflow:
|
|
7
|
+
|
|
8
|
+
```text
|
|
9
|
+
PLAN -> SHADOW -> PREWARM -> CANARY -> MIGRATE
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
It never writes the source index. `prewarm plan` only reads Shadow telemetry
|
|
13
|
+
and the target cache. `prewarm run` delegates document lookup, target encoding,
|
|
14
|
+
and cache writes to the existing materialization worker.
|
|
15
|
+
|
|
16
|
+
## Plan and run
|
|
17
|
+
|
|
18
|
+
```bash
|
|
19
|
+
embedflow prewarm plan \
|
|
20
|
+
--config embedflow.yaml --since 24h --max-docs 50000 \
|
|
21
|
+
--output prewarm-plan.json
|
|
22
|
+
embedflow prewarm run --config embedflow.yaml --plan prewarm-plan.json
|
|
23
|
+
embedflow prewarm status --config embedflow.yaml
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
Plans are schema-versioned (`schema_version: 1`), deterministically ranked by
|
|
27
|
+
`candidate_occurrences DESC, document_id ASC`, and fingerprinted with the
|
|
28
|
+
source/target contracts, source-index identity, candidate K, and Shadow
|
|
29
|
+
configuration. Execution refuses a stale or tampered plan. A plan is always
|
|
30
|
+
bounded by `max_docs`; there is no implicit full-corpus operation.
|
|
31
|
+
|
|
32
|
+
Optional configuration defaults are available under `prewarm`:
|
|
33
|
+
|
|
34
|
+
```yaml
|
|
35
|
+
prewarm:
|
|
36
|
+
max_docs: 1000
|
|
37
|
+
strategy: traffic_hotset
|
|
38
|
+
batch_size: 32
|
|
39
|
+
max_retries: 3
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
`--target-observed-coverage`, `--max-storage-gb`, and `--max-runtime` (when a
|
|
43
|
+
real measured/user-supplied `--docs-per-second` is supplied) further reduce
|
|
44
|
+
the hard selection cap. `--gpu-hourly-cost` (or the existing planner/economics
|
|
45
|
+
configuration value) adds a user-supplied modeled cost. Runtime, cost, and
|
|
46
|
+
storage values are explicitly marked as modeled/unknown; no cloud price or
|
|
47
|
+
throughput is guessed.
|
|
48
|
+
|
|
49
|
+
## Coverage and privacy
|
|
50
|
+
|
|
51
|
+
The plan reports **observed candidate-occurrence coverage**:
|
|
52
|
+
|
|
53
|
+
```text
|
|
54
|
+
occurrences with a valid target vector / all source candidate occurrences
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
This is not retrieval coverage, recall, nDCG, or a guarantee of migration
|
|
58
|
+
quality. It only describes the selected Shadow window and candidate K. Ties are
|
|
59
|
+
stable, and current target-cache entries count as warm only when their target
|
|
60
|
+
model and (when document text is available) content fingerprints match. Legacy
|
|
61
|
+
cache rows created before content bindings were introduced remain usable but
|
|
62
|
+
are treated as content-unknown.
|
|
63
|
+
|
|
64
|
+
Shadow aggregation stores document IDs and counters, not raw query text,
|
|
65
|
+
document text, vectors, or credentials. Treat the generated plan/ID list as
|
|
66
|
+
application data because selected document IDs are needed to execute it.
|
|
67
|
+
|
|
68
|
+
## Resume and idempotence
|
|
69
|
+
|
|
70
|
+
The runner stores a plan-specific queue/state beside the target cache and uses
|
|
71
|
+
the target cache as the execution source of truth. Re-running a completed plan
|
|
72
|
+
skips already-warm vectors. Interrupts leave completed vectors valid and a
|
|
73
|
+
subsequent run can continue the remaining IDs. Documents that disappear or
|
|
74
|
+
change are resolved again at run time; stale content is not silently reused.
|
|
75
|
+
|
|
76
|
+
Two concurrent runners use the existing durable queue/cache deduplication. The
|
|
77
|
+
queue is at-least-once under process crashes, so operators should inspect
|
|
78
|
+
`prewarm status` and the run report for failures.
|
|
79
|
+
|
|
80
|
+
The source FAISS/Qdrant/pgvector/Pinecone/Milvus/Weaviate index is never
|
|
81
|
+
created, altered, rebuilt, upserted, or deleted by this feature. The API for
|
|
82
|
+
prewarming is currently Python/core plus CLI; dashboard execution controls are
|
|
83
|
+
intentionally not added to avoid turning an advisory plan into autonomous
|
|
84
|
+
traffic routing.
|
|
@@ -18,8 +18,8 @@ python -m twine check dist/*
|
|
|
18
18
|
Inspect both archives before uploading:
|
|
19
19
|
|
|
20
20
|
```bash
|
|
21
|
-
unzip -l dist/embedflow-0.
|
|
22
|
-
tar -tzf dist/embedflow-0.
|
|
21
|
+
unzip -l dist/embedflow-0.8.0-py3-none-any.whl
|
|
22
|
+
tar -tzf dist/embedflow-0.8.0.tar.gz
|
|
23
23
|
sha256sum dist/*
|
|
24
24
|
```
|
|
25
25
|
|
|
@@ -32,7 +32,7 @@ Test the wheel outside the source tree:
|
|
|
32
32
|
```bash
|
|
33
33
|
python -m venv /tmp/embedflow-wheel-test
|
|
34
34
|
/tmp/embedflow-wheel-test/bin/python -m pip install --upgrade pip
|
|
35
|
-
/tmp/embedflow-wheel-test/bin/python -m pip install dist/embedflow-0.
|
|
35
|
+
/tmp/embedflow-wheel-test/bin/python -m pip install dist/embedflow-0.8.0-py3-none-any.whl
|
|
36
36
|
cd /tmp
|
|
37
37
|
/tmp/embedflow-wheel-test/bin/python -c "import embedflow; print(embedflow.__version__)"
|
|
38
38
|
/tmp/embedflow-wheel-test/bin/embedflow --help
|
|
@@ -65,7 +65,7 @@ python -m venv /tmp/embedflow-testpypi
|
|
|
65
65
|
/tmp/embedflow-testpypi/bin/python -m pip install \
|
|
66
66
|
--index-url https://test.pypi.org/simple/ \
|
|
67
67
|
--extra-index-url https://pypi.org/simple/ \
|
|
68
|
-
embedflow==0.
|
|
68
|
+
embedflow==0.8.0
|
|
69
69
|
cd /tmp
|
|
70
70
|
/tmp/embedflow-testpypi/bin/python -c "import embedflow; print(embedflow.__version__)"
|
|
71
71
|
/tmp/embedflow-testpypi/bin/embedflow --help
|
|
@@ -79,11 +79,11 @@ Test optional integrations in a second clean environment:
|
|
|
79
79
|
/tmp/embedflow-testpypi/bin/python -m pip install \
|
|
80
80
|
--index-url https://test.pypi.org/simple/ \
|
|
81
81
|
--extra-index-url https://pypi.org/simple/ \
|
|
82
|
-
"embedflow[faiss,dashboard,pinecone,milvus,weaviate]==0.
|
|
82
|
+
"embedflow[faiss,dashboard,pinecone,milvus,weaviate]==0.8.0"
|
|
83
83
|
```
|
|
84
84
|
|
|
85
|
-
If the same filename already exists on TestPyPI, use a pre-release
|
|
86
|
-
|
|
85
|
+
If the same filename already exists on TestPyPI, use a pre-release for the
|
|
86
|
+
TestPyPI-only trial. Never overwrite the production version.
|
|
87
87
|
|
|
88
88
|
## Trusted Publishing configuration
|
|
89
89
|
|
|
@@ -115,7 +115,7 @@ above keeps the test step explicit.
|
|
|
115
115
|
3. Run the final release gate and review the generated report.
|
|
116
116
|
4. Configure the PyPI pending publisher and protected `pypi` environment.
|
|
117
117
|
5. Create a Git tag and GitHub Release for the exact package version, for
|
|
118
|
-
example `v0.
|
|
118
|
+
example `v0.8.0`.
|
|
119
119
|
6. Approve the `pypi` environment when the release workflow is ready.
|
|
120
120
|
7. Verify the files and metadata on PyPI.
|
|
121
121
|
8. Install from production PyPI in a directory outside this checkout.
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
# Shadow Mode
|
|
2
|
+
|
|
3
|
+
Shadow Mode runs the configured target migration path beside real traffic while
|
|
4
|
+
the existing source result remains authoritative. It is an observation and
|
|
5
|
+
cache-warming tool, not a traffic router:
|
|
6
|
+
|
|
7
|
+
```yaml
|
|
8
|
+
runtime:
|
|
9
|
+
mode: shadow
|
|
10
|
+
|
|
11
|
+
shadow:
|
|
12
|
+
enabled: true
|
|
13
|
+
sample_rate: 0.10
|
|
14
|
+
sample_seed: 42
|
|
15
|
+
candidate_k: 100
|
|
16
|
+
materialize: true
|
|
17
|
+
max_inflight: 32
|
|
18
|
+
queue_capacity: 1000
|
|
19
|
+
timeout_ms: 10000
|
|
20
|
+
telemetry:
|
|
21
|
+
enabled: true
|
|
22
|
+
path: ./.embedflow/shadow.sqlite3
|
|
23
|
+
retain_query_records: false
|
|
24
|
+
retain_query_text: false
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
Install the optional backend/model extras required by the selected
|
|
28
|
+
configuration, then start the ordinary service:
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
pip install "embedflow[dashboard]"
|
|
32
|
+
embedflow serve --config embedflow.yaml
|
|
33
|
+
# or override the file for one process:
|
|
34
|
+
embedflow serve --config embedflow.yaml --mode shadow
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
The request path performs source encoding, source ANN retrieval, and document
|
|
38
|
+
resolution exactly as source-only serving. It returns that result without
|
|
39
|
+
waiting for target encoding, cache misses, reranking, telemetry, or the
|
|
40
|
+
materialization worker. Shadow jobs use a bounded queue and worker pool. A full
|
|
41
|
+
queue drops only shadow work; a target/model/cache/telemetry failure or timeout
|
|
42
|
+
is recorded and cannot change the primary response.
|
|
43
|
+
|
|
44
|
+
If the target model cannot be loaded during startup, source/shadow serving
|
|
45
|
+
still opens with target work marked unavailable; sampled requests record the
|
|
46
|
+
isolated target-encoding failure. Normal migration mode remains fail-fast for
|
|
47
|
+
the same startup error.
|
|
48
|
+
|
|
49
|
+
`candidate_k` is explicit. The planner may suggest a value, but Shadow Mode
|
|
50
|
+
does not silently select one. `materialize: false` reads already cached target
|
|
51
|
+
vectors without changing cache or queue state. With `materialize: true`, cold
|
|
52
|
+
candidate IDs are deduplicated in the existing persistent materialization queue
|
|
53
|
+
and warmed asynchronously; synchronous target misses remain zero in Shadow
|
|
54
|
+
Mode. A comparison with incomplete target vectors is reported as `partial`,
|
|
55
|
+
with target coverage shown separately.
|
|
56
|
+
|
|
57
|
+
## Reports
|
|
58
|
+
|
|
59
|
+
Reports use the privacy-preserving SQLite telemetry store (by default beside
|
|
60
|
+
the configured cache):
|
|
61
|
+
|
|
62
|
+
```bash
|
|
63
|
+
embedflow shadow report --config embedflow.yaml --since 24h
|
|
64
|
+
embedflow shadow report --config embedflow.yaml --since 24h --format json --output shadow-report.json
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
Use `/shadow/report` or the `shadow` section of `/status` and `/metrics` in the
|
|
68
|
+
FastAPI service. JSON/YAML stdout contains only the requested artifact; progress
|
|
69
|
+
and errors belong on stderr. `--since` accepts seconds or `s`, `m`, `h`, `d`,
|
|
70
|
+
and `w` suffixes.
|
|
71
|
+
|
|
72
|
+
Reports contain sampled/completed/partial/failed/timed-out/dropped counts,
|
|
73
|
+
cache and materialization accounting, source and shadow latency distributions,
|
|
74
|
+
target coverage, top-1 agreement, and top-k overlap. Shadow latency is not
|
|
75
|
+
user-facing latency because it is off the primary critical path. Ranking
|
|
76
|
+
disagreement and overlap are diagnostics, not recall, nDCG, or quality loss;
|
|
77
|
+
without qrels/native-target evaluation no quality guarantee is made.
|
|
78
|
+
|
|
79
|
+
T2-v1 remains the frozen window-level diagnostic. EmbedFlow does not label an
|
|
80
|
+
individual production query `T2 SAFE`. A report recommendation is operational
|
|
81
|
+
guidance only:
|
|
82
|
+
|
|
83
|
+
- `CONTINUE_SHADOW` means evidence or coverage is still limited.
|
|
84
|
+
- `EXPAND_K` means the accumulated T2 window requests a larger candidate pool.
|
|
85
|
+
- `INVESTIGATE` means failures, timeouts, or uncertain T2 behavior need review.
|
|
86
|
+
- `READY_FOR_CANARY_EVALUATION` means an operator may consider a separately
|
|
87
|
+
designed canary; it never routes traffic automatically.
|
|
88
|
+
|
|
89
|
+
Raw query text, candidate text, source vectors, target vectors, and credentials
|
|
90
|
+
are not persisted in telemetry. Query IDs may be retained only when explicitly
|
|
91
|
+
enabled, and retention is bounded by `max_records`. The telemetry database is
|
|
92
|
+
segmented by source/target contract and candidate-K fingerprint so unrelated
|
|
93
|
+
migrations are not mixed.
|
|
94
|
+
|
|
95
|
+
On shutdown EmbedFlow stops accepting new shadow work, drops queued jobs when
|
|
96
|
+
necessary, and waits only the configured bounded grace period. A corrupted or
|
|
97
|
+
unwritable telemetry file degrades observability; it does not take source
|
|
98
|
+
serving down. Source indexes are read-only during Shadow Mode. The planner and
|
|
99
|
+
Shadow Mode are separate: generate a plan first, then copy its recommended K
|
|
100
|
+
explicitly into a reviewed Shadow configuration.
|
|
101
|
+
|
|
102
|
+
Shadow Mode is advisory and experimental operational instrumentation. It does
|
|
103
|
+
not provide autonomous rollout, rollback, qrel evaluation, or ANN-fidelity
|
|
104
|
+
proof.
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
"""EmbedFlow: progressive embedding-model migration for existing indexes."""
|
|
2
|
+
|
|
3
|
+
__version__ = "0.8.0"
|
|
4
|
+
|
|
5
|
+
from .config import (
|
|
6
|
+
EmbedFlowConfig,
|
|
7
|
+
PlannerConfig,
|
|
8
|
+
PrewarmConfig,
|
|
9
|
+
RuntimeConfig,
|
|
10
|
+
ShadowConfig,
|
|
11
|
+
ShadowTelemetryConfig,
|
|
12
|
+
load_config,
|
|
13
|
+
)
|
|
14
|
+
from .prewarm import PrewarmPlan, PrewarmPlanner, PrewarmRunner, TrafficHotsetPlanner, load_prewarm_plan
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def plan(*args, **kwargs):
|
|
18
|
+
"""Generate an advisory migration plan without changing source traffic."""
|
|
19
|
+
from .planner import plan_migration
|
|
20
|
+
return plan_migration(*args, **kwargs)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def migrate(*args, **kwargs):
|
|
24
|
+
"""Start progressive migration over an existing FAISS, Qdrant, pgvector, Pinecone, Milvus, or Weaviate index.
|
|
25
|
+
|
|
26
|
+
Imported lazily to keep the lightweight configuration package free of
|
|
27
|
+
model-serving dependencies at import time. See ``embedflow.migration``
|
|
28
|
+
for the ``MigrationSession`` type.
|
|
29
|
+
"""
|
|
30
|
+
from .migration.facade import migrate as _migrate
|
|
31
|
+
return _migrate(*args, **kwargs)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def analyze_migration(*args, **kwargs):
|
|
35
|
+
"""Run the leakage-safe no-target-index analysis programmatically."""
|
|
36
|
+
from .analysis import analyze_migration as _analyze_migration
|
|
37
|
+
return _analyze_migration(*args, **kwargs)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def prewarm_plan(telemetry, cache, **kwargs):
|
|
41
|
+
"""Build a bounded traffic-hotset plan through the reusable API.
|
|
42
|
+
|
|
43
|
+
Planner-constructor options and ``plan`` options may be supplied together;
|
|
44
|
+
recognized plan options are routed to :meth:`PrewarmPlanner.plan`.
|
|
45
|
+
"""
|
|
46
|
+
plan_keys = {"since_seconds", "start", "end", "max_docs", "target_observed_coverage",
|
|
47
|
+
"max_storage_gb", "max_runtime_seconds", "docs_per_second", "gpu_hourly_cost", "strategy"}
|
|
48
|
+
plan_kwargs = {key: kwargs.pop(key) for key in tuple(kwargs) if key in plan_keys}
|
|
49
|
+
return PrewarmPlanner(telemetry, cache, **kwargs).plan(**plan_kwargs)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def prewarm_run(plan, *args, **kwargs):
|
|
53
|
+
"""Execute a validated prewarm plan through the existing materializer."""
|
|
54
|
+
run_keys = {"max_runtime_seconds", "progress"}
|
|
55
|
+
run_kwargs = {key: kwargs.pop(key) for key in tuple(kwargs) if key in run_keys}
|
|
56
|
+
return PrewarmRunner(*args, **kwargs).run(plan, **run_kwargs)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
__all__ = ["EmbedFlowConfig", "PlannerConfig", "PrewarmConfig", "RuntimeConfig", "ShadowConfig", "ShadowTelemetryConfig", "load_config", "migrate", "plan", "analyze_migration", "prewarm_plan", "prewarm_run", "PrewarmPlan", "PrewarmPlanner", "TrafficHotsetPlanner", "PrewarmRunner", "load_prewarm_plan", "__version__"]
|
|
@@ -8,15 +8,15 @@ import numpy as np
|
|
|
8
8
|
|
|
9
9
|
class TargetVectorCache(ABC):
|
|
10
10
|
@abstractmethod
|
|
11
|
-
def get(self, document_ids: list[str]) -> dict[str, np.ndarray]:
|
|
11
|
+
def get(self, document_ids: list[str], content_fingerprints: Any = None) -> dict[str, np.ndarray]:
|
|
12
12
|
raise NotImplementedError
|
|
13
13
|
|
|
14
14
|
@abstractmethod
|
|
15
|
-
def put(self, document_ids: list[str], vectors: np.ndarray) -> None:
|
|
15
|
+
def put(self, document_ids: list[str], vectors: np.ndarray, content_fingerprints: Any = None) -> None:
|
|
16
16
|
raise NotImplementedError
|
|
17
17
|
|
|
18
18
|
@abstractmethod
|
|
19
|
-
def contains(self, document_ids: list[str]) -> set[str]:
|
|
19
|
+
def contains(self, document_ids: list[str], content_fingerprints: Any = None) -> set[str]:
|
|
20
20
|
raise NotImplementedError
|
|
21
21
|
|
|
22
22
|
@abstractmethod
|