embedflow 0.7.0__tar.gz → 0.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {embedflow-0.7.0 → embedflow-0.8.0}/CHANGELOG.md +12 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/CITATION.cff +2 -2
- {embedflow-0.7.0 → embedflow-0.8.0}/PKG-INFO +13 -2
- {embedflow-0.7.0 → embedflow-0.8.0}/README.md +12 -1
- {embedflow-0.7.0 → embedflow-0.8.0}/README_PYPI.md +12 -1
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/cli.md +10 -1
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/configuration.md +12 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/limitations.md +1 -1
- embedflow-0.8.0/docs/prewarming.md +84 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/releasing.md +6 -6
- embedflow-0.8.0/embedflow/__init__.py +59 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/cache/base.py +3 -3
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/cache/persistent_cache.py +133 -18
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/cli.py +381 -8
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/config.py +81 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/migration/materializer.py +59 -4
- embedflow-0.8.0/embedflow/prewarm/__init__.py +23 -0
- embedflow-0.8.0/embedflow/prewarm/models.py +181 -0
- embedflow-0.8.0/embedflow/prewarm/planner.py +479 -0
- embedflow-0.8.0/embedflow/prewarm/runner.py +450 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/serving/engine.py +72 -7
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/shadow/models.py +5 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/shadow/runner.py +28 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/shadow/telemetry.py +306 -9
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow.egg-info/PKG-INFO +13 -2
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow.egg-info/SOURCES.txt +9 -0
- embedflow-0.8.0/examples/prewarm/README.md +16 -0
- embedflow-0.8.0/examples/prewarm/demo.py +74 -0
- embedflow-0.8.0/examples/prewarm/embedflow.yaml.example +24 -0
- embedflow-0.8.0/examples/prewarm/run_demo.sh +12 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/pyproject.toml +1 -1
- {embedflow-0.7.0 → embedflow-0.8.0}/scripts/release_gate.py +2 -2
- embedflow-0.7.0/embedflow/__init__.py +0 -31
- {embedflow-0.7.0 → embedflow-0.8.0}/CONTRIBUTING.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/LICENSE +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/MANIFEST.in +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/SECURITY.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/api.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/assets/README.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/assets/candidate-gap-example.svg +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/assets/dashboard-screenshot.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/assets/terminal-demo.txt +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/concepts.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/contributing-benchmarks.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/economics.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/installation.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/integrations/faiss.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/integrations/milvus.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/integrations/pgvector.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/integrations/pinecone.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/integrations/qdrant.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/integrations/weaviate.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/methodology.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/planner.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/quickstart.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/registry.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/docs/shadow-mode.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/__main__.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/analysis.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/api.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/cache/__init__.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/compatibility/__init__.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/compatibility/candidate_gap.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/compatibility/containment.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/compatibility/evaluate.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/compatibility/metrics.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/compatibility/migration_depth.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/compatibility/probe.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/compatibility/report.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/compatibility/t2.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/data/__init__.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/data/registry/__init__.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/data/registry/benchmark_profiles.jsonl +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/data/registry/checksums.sha256 +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/data/registry/migrations.jsonl +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/data/registry/registry_manifest.json +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/data/registry/research_summaries.json +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/data/registry/schema_version.json +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/frozen/T2_V1_FROZEN_SPEC.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/frozen/T2_V1_FROZEN_SPEC.sha256 +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/indexes/__init__.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/indexes/base.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/indexes/faiss_backend.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/indexes/milvus_backend.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/indexes/pgvector_backend.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/indexes/pinecone_backend.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/indexes/qdrant_backend.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/indexes/weaviate_backend.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/metrics/__init__.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/metrics/latency.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/migration/__init__.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/migration/compatibility.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/migration/facade.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/migration/planner.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/migration/state.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/models/__init__.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/models/base.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/models/huggingface.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/planner/__init__.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/planner/economics.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/planner/models.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/planner/planner.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/planner/rendering.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/registry/__init__.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/registry/loader.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/registry/matcher.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/registry/schema.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/runtime.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/serving/__init__.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/serving/api.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/serving/factory.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/serving/schemas.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/shadow/__init__.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow/shadow/report.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow.egg-info/dependency_links.txt +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow.egg-info/entry_points.txt +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow.egg-info/requires.txt +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/embedflow.egg-info/top_level.txt +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/faiss/README.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/faiss/documents.jsonl +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/faiss/embedflow.yaml +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/faiss/queries.jsonl +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/milvus/README.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/milvus/compose.yaml +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/milvus/embedflow.yaml.example +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/milvus/run_demo.sh +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/pgvector/README.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/pgvector/build_index.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/pgvector/compose.yaml +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/pgvector/embedflow.yaml +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/pgvector/init.sql +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/pgvector/queries.jsonl +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/pgvector/run_demo.sh +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/pinecone/README.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/pinecone/embedflow.yaml.example +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/pinecone/run_smoke.sh +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/planner/README.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/planner/run_demo.sh +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/qdrant/README.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/qdrant/build_index.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/qdrant/documents.jsonl +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/qdrant/embedflow.yaml +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/qdrant/queries.jsonl +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/research_analysis/README.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/research_analysis/documents.jsonl +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/research_analysis/embedflow.yaml +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/research_analysis/qrels.json +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/research_analysis/queries.jsonl +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/shadow/README.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/shadow/embedflow.yaml.example +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/shadow/run_demo.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/shadow/run_demo.sh +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/weaviate/README.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/weaviate/compose.yaml +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/weaviate/embedflow.yaml.example +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/examples/weaviate/run_demo.sh +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/frozen/T2_V1_FROZEN_SPEC.md +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/frozen/T2_V1_FROZEN_SPEC.sha256 +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/requirements-dev.txt +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/requirements.txt +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/scripts/milvus_fixture.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/scripts/pinecone_smoke.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/scripts/real_qdrant_smoke.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/scripts/run_demo.sh +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/scripts/run_tests.sh +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/scripts/validate_milvus.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/scripts/validate_pgvector_10k.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/scripts/validate_weaviate.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/scripts/weaviate_fixture.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/scripts/weaviate_smoke.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/setup.cfg +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/src/__init__.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/src/embed.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/src/probe_features.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/src/storage.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/src/t2_v1.py +0 -0
- {embedflow-0.7.0 → embedflow-0.8.0}/src/utils.py +0 -0
|
@@ -1,5 +1,17 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## v0.8.0 — Traffic-aware target-vector prewarming
|
|
4
|
+
|
|
5
|
+
- Added bounded `prewarm plan`, `prewarm run`, and `prewarm status` workflows
|
|
6
|
+
that select uncached target documents by observed Shadow Mode candidate
|
|
7
|
+
occurrences.
|
|
8
|
+
- Added privacy-safe aggregate document popularity telemetry, content/model
|
|
9
|
+
fingerprint checks, deterministic tie-breaking, hard document/storage/time
|
|
10
|
+
budgets, atomic plan/run artifacts, and resumable materialization through the
|
|
11
|
+
existing target cache and worker.
|
|
12
|
+
- Observed candidate-occurrence coverage remains an operational warming signal;
|
|
13
|
+
it is not retrieval quality or recall evidence.
|
|
14
|
+
|
|
3
15
|
## v0.7.0 — Source-authoritative Shadow Mode
|
|
4
16
|
|
|
5
17
|
- Added bounded, deterministic, source-authoritative Shadow Mode for observing
|
|
@@ -2,8 +2,8 @@ cff-version: 1.2.0
|
|
|
2
2
|
title: "EmbedFlow: Upgrading Legacy Embeddings Without Full Upfront Re-Embedding"
|
|
3
3
|
message: "If EmbedFlow contributes to your work, please cite this software release."
|
|
4
4
|
type: software
|
|
5
|
-
version: 0.
|
|
6
|
-
date-released: 2026-09-
|
|
5
|
+
version: 0.8.0
|
|
6
|
+
date-released: 2026-09-15
|
|
7
7
|
repository-code: "https://github.com/arnsri33/embedflow"
|
|
8
8
|
url: "https://github.com/arnsri33/embedflow"
|
|
9
9
|
license: AGPL-3.0-only
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: embedflow
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.8.0
|
|
4
4
|
Summary: Progressive embedding-model migration over existing vector indexes.
|
|
5
5
|
Author: Arnav Srivastav
|
|
6
6
|
License-Expression: AGPL-3.0-only
|
|
@@ -224,6 +224,17 @@ Reports contain operational and ranking-disagreement diagnostics—not qrel-base
|
|
|
224
224
|
quality guarantees—and recommendations never route canary traffic. See the
|
|
225
225
|
[Shadow Mode guide](https://github.com/arnsri33/embedflow/blob/main/docs/shadow-mode.md).
|
|
226
226
|
|
|
227
|
+
After collecting Shadow traffic, prioritize uncached target vectors by observed
|
|
228
|
+
candidate popularity:
|
|
229
|
+
|
|
230
|
+
```bash
|
|
231
|
+
embedflow prewarm plan --config ./embedflow.yaml --since 24h --max-docs 50000 --output prewarm-plan.json
|
|
232
|
+
embedflow prewarm run --config ./embedflow.yaml --plan prewarm-plan.json
|
|
233
|
+
```
|
|
234
|
+
|
|
235
|
+
Observed candidate-occurrence coverage is an operational warming signal, not
|
|
236
|
+
retrieval quality or recall. See the [prewarming guide](https://github.com/arnsri33/embedflow/blob/main/docs/prewarming.md).
|
|
237
|
+
|
|
227
238
|
## Research
|
|
228
239
|
|
|
229
240
|
For candidate depth `K`, EmbedFlow measures:
|
|
@@ -278,7 +289,7 @@ cover the remaining commands and endpoints.
|
|
|
278
289
|
|
|
279
290
|
## Status
|
|
280
291
|
|
|
281
|
-
EmbedFlow v0.
|
|
292
|
+
EmbedFlow v0.8.0 is a pre-1.0 release for research and early real-world
|
|
282
293
|
testing. T2-v1 is an empirical finite-tail diagnostic, partial rankings can
|
|
283
294
|
differ from fully warm target reranking, and ANN fidelity needs a reference
|
|
284
295
|
comparison to audit.
|
|
@@ -174,6 +174,17 @@ coverage, latency, and ranking-disagreement diagnostics; it does not claim
|
|
|
174
174
|
retrieval-quality preservation without qrels and never routes canary traffic.
|
|
175
175
|
See [`docs/shadow-mode.md`](https://github.com/arnsri33/embedflow/blob/main/docs/shadow-mode.md).
|
|
176
176
|
|
|
177
|
+
After collecting Shadow traffic, prioritize uncached target vectors by observed
|
|
178
|
+
candidate popularity:
|
|
179
|
+
|
|
180
|
+
```bash
|
|
181
|
+
embedflow prewarm plan --config ./embedflow.yaml --since 24h --max-docs 50000 --output prewarm-plan.json
|
|
182
|
+
embedflow prewarm run --config ./embedflow.yaml --plan prewarm-plan.json
|
|
183
|
+
```
|
|
184
|
+
|
|
185
|
+
Observed candidate-occurrence coverage is an operational warming signal, not
|
|
186
|
+
retrieval quality or recall. See [`docs/prewarming.md`](https://github.com/arnsri33/embedflow/blob/main/docs/prewarming.md).
|
|
187
|
+
|
|
177
188
|
Search responses expose `COLD`, `PARTIAL`, or `WARM`, cache hits and misses,
|
|
178
189
|
synchronous work, queued work, and stage timings. Once the candidate vectors
|
|
179
190
|
are warm, target scoring over that candidate set is deterministic.
|
|
@@ -288,7 +299,7 @@ OpenAPI documentation; see
|
|
|
288
299
|
|
|
289
300
|
## Status
|
|
290
301
|
|
|
291
|
-
EmbedFlow v0.
|
|
302
|
+
EmbedFlow v0.8.0 is a pre-1.0 release for research and early real-world
|
|
292
303
|
testing.
|
|
293
304
|
|
|
294
305
|
- T2-v1 reports an empirical finite-tail diagnostic.
|
|
@@ -158,6 +158,17 @@ Reports contain operational and ranking-disagreement diagnostics—not qrel-base
|
|
|
158
158
|
quality guarantees—and recommendations never route canary traffic. See the
|
|
159
159
|
[Shadow Mode guide](https://github.com/arnsri33/embedflow/blob/main/docs/shadow-mode.md).
|
|
160
160
|
|
|
161
|
+
After collecting Shadow traffic, prioritize uncached target vectors by observed
|
|
162
|
+
candidate popularity:
|
|
163
|
+
|
|
164
|
+
```bash
|
|
165
|
+
embedflow prewarm plan --config ./embedflow.yaml --since 24h --max-docs 50000 --output prewarm-plan.json
|
|
166
|
+
embedflow prewarm run --config ./embedflow.yaml --plan prewarm-plan.json
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
Observed candidate-occurrence coverage is an operational warming signal, not
|
|
170
|
+
retrieval quality or recall. See the [prewarming guide](https://github.com/arnsri33/embedflow/blob/main/docs/prewarming.md).
|
|
171
|
+
|
|
161
172
|
## Research
|
|
162
173
|
|
|
163
174
|
For candidate depth `K`, EmbedFlow measures:
|
|
@@ -212,7 +223,7 @@ cover the remaining commands and endpoints.
|
|
|
212
223
|
|
|
213
224
|
## Status
|
|
214
225
|
|
|
215
|
-
EmbedFlow v0.
|
|
226
|
+
EmbedFlow v0.8.0 is a pre-1.0 release for research and early real-world
|
|
216
227
|
testing. T2-v1 is an empirical finite-tail diagnostic, partial rankings can
|
|
217
228
|
differ from fully warm target reranking, and ANN fidelity needs a reference
|
|
218
229
|
comparison to audit.
|
|
@@ -11,6 +11,9 @@ embedflow analyze --config ./embedflow.yaml --output-dir ./analysis
|
|
|
11
11
|
embedflow evaluate --config ./experiment.yaml --output-dir ./results
|
|
12
12
|
embedflow plan --config ./embedflow.yaml --queries ./probe_queries.jsonl
|
|
13
13
|
embedflow shadow report --config ./embedflow.yaml --since 24h
|
|
14
|
+
embedflow prewarm plan --config ./embedflow.yaml --since 24h --max-docs 50000 --output prewarm-plan.json
|
|
15
|
+
embedflow prewarm run --config ./embedflow.yaml --plan prewarm-plan.json
|
|
16
|
+
embedflow prewarm status --config ./embedflow.yaml
|
|
14
17
|
```
|
|
15
18
|
|
|
16
19
|
`analyze` is the no-target-index workflow. It uses probe queries and frozen
|
|
@@ -41,7 +44,13 @@ embedflow export-target --config ./embedflow.yaml --output-index ./target.index
|
|
|
41
44
|
`status` reports cache coverage, hit/miss counters, queue depth, and
|
|
42
45
|
materialization throughput. `audit-index` checks the source index against an
|
|
43
46
|
exact/reference configuration where supported. `prewarm` schedules target
|
|
44
|
-
document work; it does not change source-index results.
|
|
47
|
+
document work; it does not change source-index results. `prewarm plan` ranks
|
|
48
|
+
uncached IDs by privacy-safe Shadow candidate occurrence counts and is always
|
|
49
|
+
bounded by `--max-docs` (or the configured `prewarm.max_docs`). `prewarm run`
|
|
50
|
+
refuses mismatched/tampered fingerprints, skips vectors that became warm, and
|
|
51
|
+
uses the existing durable materializer. Observed candidate-occurrence coverage
|
|
52
|
+
is not retrieval quality or recall. Use `--docs-per-second` and
|
|
53
|
+
`--gpu-hourly-cost` only for explicitly labeled modeled economics.
|
|
45
54
|
|
|
46
55
|
Use `embedflow serve --mode shadow` to opt into source-authoritative Shadow
|
|
47
56
|
Mode for one process. `embedflow shadow report` reads the bounded telemetry
|
|
@@ -77,6 +77,18 @@ planner:
|
|
|
77
77
|
corpus_name: null
|
|
78
78
|
corpus_fingerprint: null
|
|
79
79
|
|
|
80
|
+
# Optional bounded traffic-aware target-cache prewarming defaults. CLI flags
|
|
81
|
+
# override these values for one invocation; there is no implicit full-corpus
|
|
82
|
+
# prewarm operation.
|
|
83
|
+
prewarm:
|
|
84
|
+
max_docs: 1000
|
|
85
|
+
strategy: traffic_hotset
|
|
86
|
+
target_observed_coverage: null
|
|
87
|
+
max_storage_gb: null
|
|
88
|
+
max_runtime_seconds: null
|
|
89
|
+
batch_size: null
|
|
90
|
+
max_retries: null
|
|
91
|
+
|
|
80
92
|
# Source-authoritative observation mode. ``runtime.mode: migration`` (the
|
|
81
93
|
# default) leaves Shadow Mode inactive. ``mode: shadow`` is an explicit opt-in.
|
|
82
94
|
runtime:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Limitations and release scope
|
|
2
2
|
|
|
3
|
-
EmbedFlow v0.
|
|
3
|
+
EmbedFlow v0.8.0 is a pre-1.0 release for research and early real-world
|
|
4
4
|
testing. The serving path is designed to make migration experiments concrete;
|
|
5
5
|
production rollout still requires application-specific validation.
|
|
6
6
|
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
# Traffic-aware prewarming
|
|
2
|
+
|
|
3
|
+
Traffic-aware prewarming uses privacy-safe Shadow Mode aggregates to select the
|
|
4
|
+
uncached target documents that occurred most often in the source candidate
|
|
5
|
+
pool. It is a bounded operational warming tool in the existing migration
|
|
6
|
+
workflow:
|
|
7
|
+
|
|
8
|
+
```text
|
|
9
|
+
PLAN -> SHADOW -> PREWARM -> CANARY -> MIGRATE
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
It never writes the source index. `prewarm plan` only reads Shadow telemetry
|
|
13
|
+
and the target cache. `prewarm run` delegates document lookup, target encoding,
|
|
14
|
+
and cache writes to the existing materialization worker.
|
|
15
|
+
|
|
16
|
+
## Plan and run
|
|
17
|
+
|
|
18
|
+
```bash
|
|
19
|
+
embedflow prewarm plan \
|
|
20
|
+
--config embedflow.yaml --since 24h --max-docs 50000 \
|
|
21
|
+
--output prewarm-plan.json
|
|
22
|
+
embedflow prewarm run --config embedflow.yaml --plan prewarm-plan.json
|
|
23
|
+
embedflow prewarm status --config embedflow.yaml
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
Plans are schema-versioned (`schema_version: 1`), deterministically ranked by
|
|
27
|
+
`candidate_occurrences DESC, document_id ASC`, and fingerprinted with the
|
|
28
|
+
source/target contracts, source-index identity, candidate K, and Shadow
|
|
29
|
+
configuration. Execution refuses a stale or tampered plan. A plan is always
|
|
30
|
+
bounded by `max_docs`; there is no implicit full-corpus operation.
|
|
31
|
+
|
|
32
|
+
Optional configuration defaults are available under `prewarm`:
|
|
33
|
+
|
|
34
|
+
```yaml
|
|
35
|
+
prewarm:
|
|
36
|
+
max_docs: 1000
|
|
37
|
+
strategy: traffic_hotset
|
|
38
|
+
batch_size: 32
|
|
39
|
+
max_retries: 3
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
`--target-observed-coverage`, `--max-storage-gb`, and `--max-runtime` (when a
|
|
43
|
+
real measured/user-supplied `--docs-per-second` is supplied) further reduce
|
|
44
|
+
the hard selection cap. `--gpu-hourly-cost` (or the existing planner/economics
|
|
45
|
+
configuration value) adds a user-supplied modeled cost. Runtime, cost, and
|
|
46
|
+
storage values are explicitly marked as modeled/unknown; no cloud price or
|
|
47
|
+
throughput is guessed.
|
|
48
|
+
|
|
49
|
+
## Coverage and privacy
|
|
50
|
+
|
|
51
|
+
The plan reports **observed candidate-occurrence coverage**:
|
|
52
|
+
|
|
53
|
+
```text
|
|
54
|
+
occurrences with a valid target vector / all source candidate occurrences
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
This is not retrieval coverage, recall, nDCG, or a guarantee of migration
|
|
58
|
+
quality. It only describes the selected Shadow window and candidate K. Ties are
|
|
59
|
+
stable, and current target-cache entries count as warm only when their target
|
|
60
|
+
model and (when document text is available) content fingerprints match. Legacy
|
|
61
|
+
cache rows created before content bindings were introduced remain usable but
|
|
62
|
+
are treated as content-unknown.
|
|
63
|
+
|
|
64
|
+
Shadow aggregation stores document IDs and counters, not raw query text,
|
|
65
|
+
document text, vectors, or credentials. Treat the generated plan/ID list as
|
|
66
|
+
application data because selected document IDs are needed to execute it.
|
|
67
|
+
|
|
68
|
+
## Resume and idempotence
|
|
69
|
+
|
|
70
|
+
The runner stores a plan-specific queue/state beside the target cache and uses
|
|
71
|
+
the target cache as the execution source of truth. Re-running a completed plan
|
|
72
|
+
skips already-warm vectors. Interrupts leave completed vectors valid and a
|
|
73
|
+
subsequent run can continue the remaining IDs. Documents that disappear or
|
|
74
|
+
change are resolved again at run time; stale content is not silently reused.
|
|
75
|
+
|
|
76
|
+
Two concurrent runners use the existing durable queue/cache deduplication. The
|
|
77
|
+
queue is at-least-once under process crashes, so operators should inspect
|
|
78
|
+
`prewarm status` and the run report for failures.
|
|
79
|
+
|
|
80
|
+
The source FAISS/Qdrant/pgvector/Pinecone/Milvus/Weaviate index is never
|
|
81
|
+
created, altered, rebuilt, upserted, or deleted by this feature. The API for
|
|
82
|
+
prewarming is currently Python/core plus CLI; dashboard execution controls are
|
|
83
|
+
intentionally not added to avoid turning an advisory plan into autonomous
|
|
84
|
+
traffic routing.
|
|
@@ -18,8 +18,8 @@ python -m twine check dist/*
|
|
|
18
18
|
Inspect both archives before uploading:
|
|
19
19
|
|
|
20
20
|
```bash
|
|
21
|
-
unzip -l dist/embedflow-0.
|
|
22
|
-
tar -tzf dist/embedflow-0.
|
|
21
|
+
unzip -l dist/embedflow-0.8.0-py3-none-any.whl
|
|
22
|
+
tar -tzf dist/embedflow-0.8.0.tar.gz
|
|
23
23
|
sha256sum dist/*
|
|
24
24
|
```
|
|
25
25
|
|
|
@@ -32,7 +32,7 @@ Test the wheel outside the source tree:
|
|
|
32
32
|
```bash
|
|
33
33
|
python -m venv /tmp/embedflow-wheel-test
|
|
34
34
|
/tmp/embedflow-wheel-test/bin/python -m pip install --upgrade pip
|
|
35
|
-
/tmp/embedflow-wheel-test/bin/python -m pip install dist/embedflow-0.
|
|
35
|
+
/tmp/embedflow-wheel-test/bin/python -m pip install dist/embedflow-0.8.0-py3-none-any.whl
|
|
36
36
|
cd /tmp
|
|
37
37
|
/tmp/embedflow-wheel-test/bin/python -c "import embedflow; print(embedflow.__version__)"
|
|
38
38
|
/tmp/embedflow-wheel-test/bin/embedflow --help
|
|
@@ -65,7 +65,7 @@ python -m venv /tmp/embedflow-testpypi
|
|
|
65
65
|
/tmp/embedflow-testpypi/bin/python -m pip install \
|
|
66
66
|
--index-url https://test.pypi.org/simple/ \
|
|
67
67
|
--extra-index-url https://pypi.org/simple/ \
|
|
68
|
-
embedflow==0.
|
|
68
|
+
embedflow==0.8.0
|
|
69
69
|
cd /tmp
|
|
70
70
|
/tmp/embedflow-testpypi/bin/python -c "import embedflow; print(embedflow.__version__)"
|
|
71
71
|
/tmp/embedflow-testpypi/bin/embedflow --help
|
|
@@ -79,7 +79,7 @@ Test optional integrations in a second clean environment:
|
|
|
79
79
|
/tmp/embedflow-testpypi/bin/python -m pip install \
|
|
80
80
|
--index-url https://test.pypi.org/simple/ \
|
|
81
81
|
--extra-index-url https://pypi.org/simple/ \
|
|
82
|
-
"embedflow[faiss,dashboard,pinecone,milvus,weaviate]==0.
|
|
82
|
+
"embedflow[faiss,dashboard,pinecone,milvus,weaviate]==0.8.0"
|
|
83
83
|
```
|
|
84
84
|
|
|
85
85
|
If the same filename already exists on TestPyPI, use a pre-release for the
|
|
@@ -115,7 +115,7 @@ above keeps the test step explicit.
|
|
|
115
115
|
3. Run the final release gate and review the generated report.
|
|
116
116
|
4. Configure the PyPI pending publisher and protected `pypi` environment.
|
|
117
117
|
5. Create a Git tag and GitHub Release for the exact package version, for
|
|
118
|
-
example `v0.
|
|
118
|
+
example `v0.8.0`.
|
|
119
119
|
6. Approve the `pypi` environment when the release workflow is ready.
|
|
120
120
|
7. Verify the files and metadata on PyPI.
|
|
121
121
|
8. Install from production PyPI in a directory outside this checkout.
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
"""EmbedFlow: progressive embedding-model migration for existing indexes."""
|
|
2
|
+
|
|
3
|
+
__version__ = "0.8.0"
|
|
4
|
+
|
|
5
|
+
from .config import (
|
|
6
|
+
EmbedFlowConfig,
|
|
7
|
+
PlannerConfig,
|
|
8
|
+
PrewarmConfig,
|
|
9
|
+
RuntimeConfig,
|
|
10
|
+
ShadowConfig,
|
|
11
|
+
ShadowTelemetryConfig,
|
|
12
|
+
load_config,
|
|
13
|
+
)
|
|
14
|
+
from .prewarm import PrewarmPlan, PrewarmPlanner, PrewarmRunner, TrafficHotsetPlanner, load_prewarm_plan
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def plan(*args, **kwargs):
|
|
18
|
+
"""Generate an advisory migration plan without changing source traffic."""
|
|
19
|
+
from .planner import plan_migration
|
|
20
|
+
return plan_migration(*args, **kwargs)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def migrate(*args, **kwargs):
|
|
24
|
+
"""Start progressive migration over an existing FAISS, Qdrant, pgvector, Pinecone, Milvus, or Weaviate index.
|
|
25
|
+
|
|
26
|
+
Imported lazily to keep the lightweight configuration package free of
|
|
27
|
+
model-serving dependencies at import time. See ``embedflow.migration``
|
|
28
|
+
for the ``MigrationSession`` type.
|
|
29
|
+
"""
|
|
30
|
+
from .migration.facade import migrate as _migrate
|
|
31
|
+
return _migrate(*args, **kwargs)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def analyze_migration(*args, **kwargs):
|
|
35
|
+
"""Run the leakage-safe no-target-index analysis programmatically."""
|
|
36
|
+
from .analysis import analyze_migration as _analyze_migration
|
|
37
|
+
return _analyze_migration(*args, **kwargs)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def prewarm_plan(telemetry, cache, **kwargs):
|
|
41
|
+
"""Build a bounded traffic-hotset plan through the reusable API.
|
|
42
|
+
|
|
43
|
+
Planner-constructor options and ``plan`` options may be supplied together;
|
|
44
|
+
recognized plan options are routed to :meth:`PrewarmPlanner.plan`.
|
|
45
|
+
"""
|
|
46
|
+
plan_keys = {"since_seconds", "start", "end", "max_docs", "target_observed_coverage",
|
|
47
|
+
"max_storage_gb", "max_runtime_seconds", "docs_per_second", "gpu_hourly_cost", "strategy"}
|
|
48
|
+
plan_kwargs = {key: kwargs.pop(key) for key in tuple(kwargs) if key in plan_keys}
|
|
49
|
+
return PrewarmPlanner(telemetry, cache, **kwargs).plan(**plan_kwargs)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def prewarm_run(plan, *args, **kwargs):
|
|
53
|
+
"""Execute a validated prewarm plan through the existing materializer."""
|
|
54
|
+
run_keys = {"max_runtime_seconds", "progress"}
|
|
55
|
+
run_kwargs = {key: kwargs.pop(key) for key in tuple(kwargs) if key in run_keys}
|
|
56
|
+
return PrewarmRunner(*args, **kwargs).run(plan, **run_kwargs)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
__all__ = ["EmbedFlowConfig", "PlannerConfig", "PrewarmConfig", "RuntimeConfig", "ShadowConfig", "ShadowTelemetryConfig", "load_config", "migrate", "plan", "analyze_migration", "prewarm_plan", "prewarm_run", "PrewarmPlan", "PrewarmPlanner", "TrafficHotsetPlanner", "PrewarmRunner", "load_prewarm_plan", "__version__"]
|
|
@@ -8,15 +8,15 @@ import numpy as np
|
|
|
8
8
|
|
|
9
9
|
class TargetVectorCache(ABC):
|
|
10
10
|
@abstractmethod
|
|
11
|
-
def get(self, document_ids: list[str]) -> dict[str, np.ndarray]:
|
|
11
|
+
def get(self, document_ids: list[str], content_fingerprints: Any = None) -> dict[str, np.ndarray]:
|
|
12
12
|
raise NotImplementedError
|
|
13
13
|
|
|
14
14
|
@abstractmethod
|
|
15
|
-
def put(self, document_ids: list[str], vectors: np.ndarray) -> None:
|
|
15
|
+
def put(self, document_ids: list[str], vectors: np.ndarray, content_fingerprints: Any = None) -> None:
|
|
16
16
|
raise NotImplementedError
|
|
17
17
|
|
|
18
18
|
@abstractmethod
|
|
19
|
-
def contains(self, document_ids: list[str]) -> set[str]:
|
|
19
|
+
def contains(self, document_ids: list[str], content_fingerprints: Any = None) -> set[str]:
|
|
20
20
|
raise NotImplementedError
|
|
21
21
|
|
|
22
22
|
@abstractmethod
|
|
@@ -4,6 +4,7 @@ import hashlib
|
|
|
4
4
|
import sqlite3
|
|
5
5
|
import threading
|
|
6
6
|
import time
|
|
7
|
+
from collections.abc import Mapping
|
|
7
8
|
from pathlib import Path
|
|
8
9
|
from typing import Any
|
|
9
10
|
|
|
@@ -25,7 +26,7 @@ class SQLiteVectorCache(TargetVectorCache):
|
|
|
25
26
|
_LOOKUP_BATCH_SIZE = 900
|
|
26
27
|
|
|
27
28
|
def __init__(self, path: str | Path, model_fingerprint: str, dimension: int,
|
|
28
|
-
dtype: str = "float32"):
|
|
29
|
+
dtype: str = "float32", *, read_only: bool = False):
|
|
29
30
|
p = Path(path)
|
|
30
31
|
if p.suffix in {".sqlite", ".sqlite3", ".db"}:
|
|
31
32
|
self.db_path = p
|
|
@@ -33,7 +34,13 @@ class SQLiteVectorCache(TargetVectorCache):
|
|
|
33
34
|
else:
|
|
34
35
|
self.root = p
|
|
35
36
|
self.db_path = p / "cache.sqlite3"
|
|
36
|
-
self.
|
|
37
|
+
self.read_only = bool(read_only)
|
|
38
|
+
# A planning/status inspection must not create a cache directory or
|
|
39
|
+
# migrate its schema. Normal serving retains the historical
|
|
40
|
+
# create-or-open behavior; read-only callers use SQLite's URI mode and
|
|
41
|
+
# receive an empty view when the cache has not been initialized yet.
|
|
42
|
+
if not self.read_only:
|
|
43
|
+
self.root.mkdir(parents=True, exist_ok=True)
|
|
37
44
|
self.model_fingerprint = str(model_fingerprint).strip()
|
|
38
45
|
if not self.model_fingerprint:
|
|
39
46
|
raise ValueError("model_fingerprint must be non-empty")
|
|
@@ -53,7 +60,20 @@ class SQLiteVectorCache(TargetVectorCache):
|
|
|
53
60
|
raise ValueError("cache dtype must be a numeric vector dtype")
|
|
54
61
|
self._lock = threading.RLock()
|
|
55
62
|
self._db = None
|
|
63
|
+
self._has_content_fingerprint = True
|
|
64
|
+
self._has_target_vectors = True
|
|
56
65
|
try:
|
|
66
|
+
if self.read_only:
|
|
67
|
+
if self.db_path.exists():
|
|
68
|
+
self._db = sqlite3.connect(f"file:{self.db_path}?mode=ro", uri=True,
|
|
69
|
+
check_same_thread=False, timeout=30)
|
|
70
|
+
self._db.execute("PRAGMA busy_timeout=30000")
|
|
71
|
+
columns = {str(row[1]) for row in self._db.execute(
|
|
72
|
+
"PRAGMA table_info(target_vectors)").fetchall()}
|
|
73
|
+
self._has_target_vectors = bool(columns)
|
|
74
|
+
self._has_content_fingerprint = "content_fingerprint" in columns
|
|
75
|
+
self._closed = False
|
|
76
|
+
return
|
|
57
77
|
self._db = sqlite3.connect(str(self.db_path), check_same_thread=False, timeout=30)
|
|
58
78
|
self._db.execute("PRAGMA busy_timeout=30000")
|
|
59
79
|
self._db.execute("PRAGMA journal_mode=WAL")
|
|
@@ -67,8 +87,18 @@ class SQLiteVectorCache(TargetVectorCache):
|
|
|
67
87
|
checksum TEXT NOT NULL,
|
|
68
88
|
created_at REAL NOT NULL,
|
|
69
89
|
accessed_at REAL NOT NULL,
|
|
90
|
+
content_fingerprint TEXT,
|
|
70
91
|
PRIMARY KEY(document_id, model_fingerprint)
|
|
71
92
|
)""")
|
|
93
|
+
# Caches created by pre-0.8 releases do not have a content
|
|
94
|
+
# fingerprint. Keep those vectors readable for existing serving
|
|
95
|
+
# callers, while allowing prewarming to require a current-content
|
|
96
|
+
# match when one is available.
|
|
97
|
+
columns = {str(row[1]) for row in self._db.execute("PRAGMA table_info(target_vectors)").fetchall()}
|
|
98
|
+
if "content_fingerprint" not in columns:
|
|
99
|
+
self._db.execute("ALTER TABLE target_vectors ADD COLUMN content_fingerprint TEXT")
|
|
100
|
+
self._has_content_fingerprint = True
|
|
101
|
+
self._has_target_vectors = True
|
|
72
102
|
self._db.execute("CREATE INDEX IF NOT EXISTS idx_target_vectors_model ON target_vectors(model_fingerprint)")
|
|
73
103
|
self._db.execute("""CREATE TABLE IF NOT EXISTS cache_counters (
|
|
74
104
|
model_fingerprint TEXT PRIMARY KEY,
|
|
@@ -92,7 +122,7 @@ class SQLiteVectorCache(TargetVectorCache):
|
|
|
92
122
|
self._closed = False
|
|
93
123
|
|
|
94
124
|
def _ensure_open(self) -> None:
|
|
95
|
-
if self._closed or self._db is None:
|
|
125
|
+
if self._closed or (self._db is None and not self.read_only):
|
|
96
126
|
raise RuntimeError("target vector cache is closed")
|
|
97
127
|
|
|
98
128
|
def _decode(self, row: tuple[Any, ...]) -> np.ndarray:
|
|
@@ -106,20 +136,53 @@ class SQLiteVectorCache(TargetVectorCache):
|
|
|
106
136
|
raise CacheCorruptionError(f"invalid cached vector for document {document_id}")
|
|
107
137
|
return value.astype("float32", copy=False)
|
|
108
138
|
|
|
109
|
-
|
|
139
|
+
@staticmethod
|
|
140
|
+
def content_fingerprint(text: Any) -> str:
|
|
141
|
+
"""Return the stable fingerprint used to bind a vector to content."""
|
|
142
|
+
return hashlib.sha256(str(text).encode("utf-8")).hexdigest()
|
|
143
|
+
|
|
144
|
+
@staticmethod
|
|
145
|
+
def _normalize_content_fingerprints(document_ids: list[str], values: Any) -> dict[str, str] | None:
|
|
146
|
+
if values is None:
|
|
147
|
+
return None
|
|
148
|
+
if isinstance(values, Mapping):
|
|
149
|
+
return {str(key): str(value) for key, value in values.items()}
|
|
150
|
+
try:
|
|
151
|
+
sequence = list(values)
|
|
152
|
+
except TypeError as exc:
|
|
153
|
+
raise ValueError("content_fingerprints must be a mapping or sequence") from exc
|
|
154
|
+
if len(sequence) != len(document_ids):
|
|
155
|
+
raise ValueError("content_fingerprints must match document_ids length")
|
|
156
|
+
return {str(document_id): str(value) for document_id, value in zip(document_ids, sequence)}
|
|
157
|
+
|
|
158
|
+
def get(self, document_ids: list[str], content_fingerprints: Any = None) -> dict[str, np.ndarray]:
|
|
159
|
+
if self.read_only:
|
|
160
|
+
# Preserve the familiar lookup contract for inspection callers
|
|
161
|
+
# without updating hit/miss/access timestamps in a read-only
|
|
162
|
+
# SQLite connection.
|
|
163
|
+
return self.peek(document_ids, content_fingerprints=content_fingerprints)
|
|
110
164
|
ids = [str(x) for x in document_ids]
|
|
111
165
|
if not ids: return {}
|
|
166
|
+
expected_content = self._normalize_content_fingerprints(ids, content_fingerprints)
|
|
112
167
|
with self._lock:
|
|
113
168
|
self._ensure_open()
|
|
114
169
|
rows: list[tuple[Any, ...]] = []
|
|
115
170
|
for start in range(0, len(ids), self._LOOKUP_BATCH_SIZE):
|
|
116
171
|
chunk = ids[start:start + self._LOOKUP_BATCH_SIZE]
|
|
117
172
|
placeholders = ",".join("?" for _ in chunk)
|
|
173
|
+
columns = ",content_fingerprint" if self._has_content_fingerprint else ""
|
|
118
174
|
rows.extend(self._db.execute(
|
|
119
|
-
f"SELECT document_id,model_fingerprint,dimension,dtype,vector,checksum,created_at,accessed_at "
|
|
175
|
+
f"SELECT document_id,model_fingerprint,dimension,dtype,vector,checksum,created_at,accessed_at{columns} "
|
|
120
176
|
f"FROM target_vectors WHERE model_fingerprint=? AND document_id IN ({placeholders})",
|
|
121
177
|
[self.model_fingerprint, *chunk]).fetchall())
|
|
122
178
|
now = time.time()
|
|
179
|
+
if expected_content is not None:
|
|
180
|
+
# Rows from pre-0.8 caches have no content binding. Preserve
|
|
181
|
+
# their historical serving behavior while requiring an exact
|
|
182
|
+
# match whenever a binding is present (new writes and
|
|
183
|
+
# materialized vectors).
|
|
184
|
+
if self._has_content_fingerprint:
|
|
185
|
+
rows = [row for row in rows if row[8] is None or str(row[8]) == expected_content.get(str(row[0]))]
|
|
123
186
|
values = {str(row[0]): self._decode(row) for row in rows}
|
|
124
187
|
if rows:
|
|
125
188
|
self._db.executemany("UPDATE target_vectors SET accessed_at=? WHERE document_id=? AND model_fingerprint=?",
|
|
@@ -141,7 +204,7 @@ class SQLiteVectorCache(TargetVectorCache):
|
|
|
141
204
|
self._db.commit()
|
|
142
205
|
return values
|
|
143
206
|
|
|
144
|
-
def peek(self, document_ids: list[str]) -> dict[str, np.ndarray]:
|
|
207
|
+
def peek(self, document_ids: list[str], content_fingerprints: Any = None) -> dict[str, np.ndarray]:
|
|
145
208
|
"""Read cached vectors without changing hit/miss telemetry.
|
|
146
209
|
|
|
147
210
|
Shadow analysis with ``materialize=false`` must not make ordinary
|
|
@@ -152,19 +215,27 @@ class SQLiteVectorCache(TargetVectorCache):
|
|
|
152
215
|
ids = [str(x) for x in document_ids]
|
|
153
216
|
if not ids:
|
|
154
217
|
return {}
|
|
218
|
+
expected_content = self._normalize_content_fingerprints(ids, content_fingerprints)
|
|
155
219
|
with self._lock:
|
|
156
220
|
self._ensure_open()
|
|
221
|
+
if self._db is None or not self._has_target_vectors:
|
|
222
|
+
return {}
|
|
157
223
|
rows: list[tuple[Any, ...]] = []
|
|
158
224
|
for start in range(0, len(ids), self._LOOKUP_BATCH_SIZE):
|
|
159
225
|
chunk = ids[start:start + self._LOOKUP_BATCH_SIZE]
|
|
160
226
|
placeholders = ",".join("?" for _ in chunk)
|
|
227
|
+
columns = ",content_fingerprint" if self._has_content_fingerprint else ""
|
|
161
228
|
rows.extend(self._db.execute(
|
|
162
|
-
f"SELECT document_id,model_fingerprint,dimension,dtype,vector,checksum,created_at,accessed_at "
|
|
229
|
+
f"SELECT document_id,model_fingerprint,dimension,dtype,vector,checksum,created_at,accessed_at{columns} "
|
|
163
230
|
f"FROM target_vectors WHERE model_fingerprint=? AND document_id IN ({placeholders})",
|
|
164
231
|
[self.model_fingerprint, *chunk]).fetchall())
|
|
232
|
+
if expected_content is not None and self._has_content_fingerprint:
|
|
233
|
+
rows = [row for row in rows if row[8] is None or str(row[8]) == expected_content.get(str(row[0]))]
|
|
165
234
|
return {str(row[0]): self._decode(row) for row in rows}
|
|
166
235
|
|
|
167
|
-
def put(self, document_ids: list[str], vectors: np.ndarray) -> None:
|
|
236
|
+
def put(self, document_ids: list[str], vectors: np.ndarray, content_fingerprints: Any = None) -> None:
|
|
237
|
+
if self.read_only:
|
|
238
|
+
raise RuntimeError("target vector cache is read-only")
|
|
168
239
|
ids = [str(x) for x in document_ids]
|
|
169
240
|
values = np.asarray(vectors, dtype=self.dtype)
|
|
170
241
|
if values.ndim != 2 or values.shape != (len(ids), self.dimension):
|
|
@@ -173,27 +244,70 @@ class SQLiteVectorCache(TargetVectorCache):
|
|
|
173
244
|
raise ValueError("duplicate IDs in cache write")
|
|
174
245
|
if not np.isfinite(values).all():
|
|
175
246
|
raise ValueError("cannot cache non-finite vectors")
|
|
247
|
+
content = self._normalize_content_fingerprints(ids, content_fingerprints)
|
|
176
248
|
now = time.time(); records = []
|
|
177
249
|
for document_id, value in zip(ids, values):
|
|
178
250
|
blob = np.ascontiguousarray(value).tobytes()
|
|
179
251
|
records.append((document_id, self.model_fingerprint, self.dimension, self.dtype.name, sqlite3.Binary(blob),
|
|
180
|
-
hashlib.sha256(blob).hexdigest(), now, now
|
|
252
|
+
hashlib.sha256(blob).hexdigest(), now, now,
|
|
253
|
+
content.get(document_id) if content is not None else None))
|
|
181
254
|
with self._lock:
|
|
182
255
|
self._ensure_open()
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
256
|
+
if content is None:
|
|
257
|
+
# Older callers do not know document content. Updating the
|
|
258
|
+
# vector must not erase a binding written by the new
|
|
259
|
+
# materializer, otherwise a later content change could be
|
|
260
|
+
# mistaken for a valid cache hit. New rows remain unbound and
|
|
261
|
+
# therefore retain historical compatibility.
|
|
262
|
+
self._db.executemany("""INSERT INTO target_vectors
|
|
263
|
+
(document_id,model_fingerprint,dimension,dtype,vector,checksum,created_at,accessed_at,content_fingerprint)
|
|
264
|
+
VALUES (?,?,?,?,?,?,?,?,NULL)
|
|
265
|
+
ON CONFLICT(document_id,model_fingerprint) DO UPDATE SET
|
|
266
|
+
dimension=excluded.dimension,dtype=excluded.dtype,vector=excluded.vector,
|
|
267
|
+
checksum=excluded.checksum,accessed_at=excluded.accessed_at""",
|
|
268
|
+
[record[:8] for record in records])
|
|
269
|
+
else:
|
|
270
|
+
self._db.executemany("""INSERT INTO target_vectors
|
|
271
|
+
(document_id,model_fingerprint,dimension,dtype,vector,checksum,created_at,accessed_at,content_fingerprint)
|
|
272
|
+
VALUES (?,?,?,?,?,?,?,?,?)
|
|
273
|
+
ON CONFLICT(document_id,model_fingerprint) DO UPDATE SET
|
|
274
|
+
dimension=excluded.dimension,dtype=excluded.dtype,vector=excluded.vector,
|
|
275
|
+
checksum=excluded.checksum,accessed_at=excluded.accessed_at,
|
|
276
|
+
content_fingerprint=excluded.content_fingerprint""", records)
|
|
189
277
|
self._db.commit()
|
|
190
278
|
|
|
191
|
-
def contains(self, document_ids: list[str]) -> set[str]:
|
|
192
|
-
|
|
279
|
+
def contains(self, document_ids: list[str], content_fingerprints: Any = None) -> set[str]:
|
|
280
|
+
if self.read_only:
|
|
281
|
+
return set(self.peek(document_ids, content_fingerprints=content_fingerprints))
|
|
282
|
+
return set(self.get(document_ids, content_fingerprints=content_fingerprints))
|
|
283
|
+
|
|
284
|
+
def content_fingerprints(self, document_ids: list[str]) -> dict[str, str | None]:
|
|
285
|
+
"""Inspect content bindings without changing cache hit/miss counters."""
|
|
286
|
+
ids = [str(x) for x in document_ids]
|
|
287
|
+
if not ids:
|
|
288
|
+
return {}
|
|
289
|
+
with self._lock:
|
|
290
|
+
self._ensure_open()
|
|
291
|
+
if self._db is None or not self._has_target_vectors or not self._has_content_fingerprint:
|
|
292
|
+
return {}
|
|
293
|
+
rows: list[tuple[Any, ...]] = []
|
|
294
|
+
for start in range(0, len(ids), self._LOOKUP_BATCH_SIZE):
|
|
295
|
+
chunk = ids[start:start + self._LOOKUP_BATCH_SIZE]
|
|
296
|
+
placeholders = ",".join("?" for _ in chunk)
|
|
297
|
+
rows.extend(self._db.execute(
|
|
298
|
+
f"SELECT document_id,content_fingerprint FROM target_vectors "
|
|
299
|
+
f"WHERE model_fingerprint=? AND document_id IN ({placeholders})",
|
|
300
|
+
[self.model_fingerprint, *chunk]).fetchall())
|
|
301
|
+
return {str(row[0]): (None if row[1] is None else str(row[1])) for row in rows}
|
|
193
302
|
|
|
194
303
|
def stats(self) -> dict[str, Any]:
|
|
195
304
|
with self._lock:
|
|
196
305
|
self._ensure_open()
|
|
306
|
+
if self._db is None or not self._has_target_vectors:
|
|
307
|
+
return {"cached_target_vectors": 0, "all_model_vectors": 0,
|
|
308
|
+
"model_fingerprint": self.model_fingerprint, "dimension": self.dimension,
|
|
309
|
+
"cache_path": str(self.db_path), "total_cache_hits": 0,
|
|
310
|
+
"total_cache_misses": 0, "recent_hit_rate": 0.0}
|
|
197
311
|
count = self._db.execute("SELECT COUNT(*) FROM target_vectors WHERE model_fingerprint=?", (self.model_fingerprint,)).fetchone()[0]
|
|
198
312
|
total = self._db.execute("SELECT COUNT(*) FROM target_vectors").fetchone()[0]
|
|
199
313
|
counters = self._db.execute("SELECT hits,misses FROM cache_counters WHERE model_fingerprint=?",
|
|
@@ -213,7 +327,8 @@ class SQLiteVectorCache(TargetVectorCache):
|
|
|
213
327
|
def close(self) -> None:
|
|
214
328
|
with self._lock:
|
|
215
329
|
if not self._closed:
|
|
216
|
-
self._db
|
|
330
|
+
if self._db is not None:
|
|
331
|
+
self._db.close()
|
|
217
332
|
self._db = None
|
|
218
333
|
self._closed = True
|
|
219
334
|
|