embedflow 0.5.0__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {embedflow-0.5.0 → embedflow-0.6.0}/CHANGELOG.md +9 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/CITATION.cff +1 -1
- {embedflow-0.5.0 → embedflow-0.6.0}/PKG-INFO +18 -2
- {embedflow-0.5.0 → embedflow-0.6.0}/README.md +16 -1
- {embedflow-0.5.0 → embedflow-0.6.0}/README_PYPI.md +17 -1
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/api.md +5 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/cli.md +10 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/configuration.md +34 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/limitations.md +1 -1
- embedflow-0.6.0/docs/planner.md +111 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/releasing.md +7 -7
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/__init__.py +9 -3
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/cli.py +99 -1
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/config.py +128 -0
- embedflow-0.6.0/embedflow/planner/__init__.py +22 -0
- embedflow-0.6.0/embedflow/planner/economics.py +153 -0
- embedflow-0.6.0/embedflow/planner/models.py +97 -0
- embedflow-0.6.0/embedflow/planner/planner.py +1390 -0
- embedflow-0.6.0/embedflow/planner/rendering.py +93 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow.egg-info/PKG-INFO +18 -2
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow.egg-info/SOURCES.txt +8 -0
- embedflow-0.6.0/examples/planner/README.md +12 -0
- embedflow-0.6.0/examples/planner/run_demo.sh +14 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/pyproject.toml +1 -1
- {embedflow-0.5.0 → embedflow-0.6.0}/scripts/release_gate.py +3 -2
- {embedflow-0.5.0 → embedflow-0.6.0}/CONTRIBUTING.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/LICENSE +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/MANIFEST.in +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/SECURITY.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/assets/README.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/assets/candidate-gap-example.svg +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/assets/dashboard-screenshot.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/assets/terminal-demo.txt +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/concepts.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/contributing-benchmarks.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/economics.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/installation.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/integrations/faiss.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/integrations/milvus.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/integrations/pgvector.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/integrations/pinecone.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/integrations/qdrant.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/integrations/weaviate.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/methodology.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/quickstart.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/docs/registry.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/__main__.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/analysis.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/api.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/cache/__init__.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/cache/base.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/cache/persistent_cache.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/compatibility/__init__.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/compatibility/candidate_gap.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/compatibility/containment.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/compatibility/evaluate.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/compatibility/metrics.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/compatibility/migration_depth.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/compatibility/probe.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/compatibility/report.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/compatibility/t2.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/data/__init__.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/data/registry/__init__.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/data/registry/benchmark_profiles.jsonl +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/data/registry/checksums.sha256 +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/data/registry/migrations.jsonl +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/data/registry/registry_manifest.json +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/data/registry/research_summaries.json +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/data/registry/schema_version.json +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/frozen/T2_V1_FROZEN_SPEC.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/frozen/T2_V1_FROZEN_SPEC.sha256 +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/indexes/__init__.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/indexes/base.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/indexes/faiss_backend.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/indexes/milvus_backend.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/indexes/pgvector_backend.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/indexes/pinecone_backend.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/indexes/qdrant_backend.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/indexes/weaviate_backend.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/metrics/__init__.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/metrics/latency.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/migration/__init__.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/migration/compatibility.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/migration/facade.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/migration/materializer.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/migration/planner.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/migration/state.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/models/__init__.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/models/base.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/models/huggingface.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/registry/__init__.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/registry/loader.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/registry/matcher.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/registry/schema.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/runtime.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/serving/__init__.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/serving/api.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/serving/engine.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/serving/factory.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow/serving/schemas.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow.egg-info/dependency_links.txt +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow.egg-info/entry_points.txt +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow.egg-info/requires.txt +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/embedflow.egg-info/top_level.txt +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/faiss/README.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/faiss/documents.jsonl +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/faiss/embedflow.yaml +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/faiss/queries.jsonl +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/milvus/README.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/milvus/compose.yaml +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/milvus/embedflow.yaml.example +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/milvus/run_demo.sh +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/pgvector/README.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/pgvector/build_index.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/pgvector/compose.yaml +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/pgvector/embedflow.yaml +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/pgvector/init.sql +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/pgvector/queries.jsonl +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/pgvector/run_demo.sh +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/pinecone/README.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/pinecone/embedflow.yaml.example +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/pinecone/run_smoke.sh +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/qdrant/README.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/qdrant/build_index.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/qdrant/documents.jsonl +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/qdrant/embedflow.yaml +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/qdrant/queries.jsonl +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/research_analysis/README.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/research_analysis/documents.jsonl +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/research_analysis/embedflow.yaml +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/research_analysis/qrels.json +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/research_analysis/queries.jsonl +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/weaviate/README.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/weaviate/compose.yaml +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/weaviate/embedflow.yaml.example +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/examples/weaviate/run_demo.sh +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/frozen/T2_V1_FROZEN_SPEC.md +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/frozen/T2_V1_FROZEN_SPEC.sha256 +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/requirements-dev.txt +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/requirements.txt +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/scripts/milvus_fixture.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/scripts/pinecone_smoke.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/scripts/real_qdrant_smoke.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/scripts/run_demo.sh +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/scripts/run_tests.sh +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/scripts/validate_milvus.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/scripts/validate_pgvector_10k.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/scripts/validate_weaviate.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/scripts/weaviate_fixture.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/scripts/weaviate_smoke.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/setup.cfg +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/src/__init__.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/src/embed.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/src/probe_features.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/src/storage.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/src/t2_v1.py +0 -0
- {embedflow-0.5.0 → embedflow-0.6.0}/src/utils.py +0 -0
|
@@ -1,5 +1,14 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## v0.6.0 — Migration planner
|
|
4
|
+
|
|
5
|
+
- Added an advisory `embedflow plan` command and Python API that combine
|
|
6
|
+
source-index preflight, registry evidence, representative probes, frozen
|
|
7
|
+
T2-v1 diagnostics, candidate-depth selection, cache planning, economics, and
|
|
8
|
+
staged rollout guidance without routing traffic or mutating the source.
|
|
9
|
+
- Added structured JSON/YAML plan artifacts with explicit warnings and
|
|
10
|
+
measured/user-supplied/modeled/unknown provenance.
|
|
11
|
+
|
|
3
12
|
## v0.5.0 — Weaviate backend
|
|
4
13
|
|
|
5
14
|
- Added a read-only Weaviate v4 backend for existing externally-vectorized
|
|
@@ -2,7 +2,7 @@ cff-version: 1.2.0
|
|
|
2
2
|
title: "EmbedFlow: Upgrading Legacy Embeddings Without Full Upfront Re-Embedding"
|
|
3
3
|
message: "If EmbedFlow contributes to your work, please cite this software release."
|
|
4
4
|
type: software
|
|
5
|
-
version: 0.
|
|
5
|
+
version: 0.6.0
|
|
6
6
|
date-released: 2026-09-13
|
|
7
7
|
repository-code: "https://github.com/arnsri33/embedflow"
|
|
8
8
|
url: "https://github.com/arnsri33/embedflow"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: embedflow
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.6.0
|
|
4
4
|
Summary: Progressive embedding-model migration over existing vector indexes.
|
|
5
5
|
Author: Arnav Srivastav
|
|
6
6
|
License-Expression: AGPL-3.0-only
|
|
@@ -189,6 +189,21 @@ Selected core records (nDCG@10, `G(50)`):
|
|
|
189
189
|
See the [registry documentation](https://github.com/arnsri33/embedflow/blob/main/docs/registry.md)
|
|
190
190
|
for matching levels, contract fingerprints, and provenance.
|
|
191
191
|
|
|
192
|
+
## Advisory migration plans
|
|
193
|
+
|
|
194
|
+
Use representative query probes to produce a bounded migration recommendation
|
|
195
|
+
before serving:
|
|
196
|
+
|
|
197
|
+
```bash
|
|
198
|
+
embedflow plan --config ./embedflow.yaml --queries ./probe_queries.jsonl
|
|
199
|
+
embedflow plan --config ./embedflow.yaml --queries ./probe_queries.jsonl --format json --output migration-plan.json
|
|
200
|
+
```
|
|
201
|
+
|
|
202
|
+
The planner reports preflight, registry evidence, frozen T2-v1 finite-tail
|
|
203
|
+
behavior, candidate K, cache/economics projections, and staged rollout
|
|
204
|
+
guidance. It never routes traffic or mutates the source index; `SAFE` is not a
|
|
205
|
+
qrels-based retrieval-quality guarantee. See the [planner guide](https://github.com/arnsri33/embedflow/blob/main/docs/planner.md).
|
|
206
|
+
|
|
192
207
|
## Research
|
|
193
208
|
|
|
194
209
|
For candidate depth `K`, EmbedFlow measures:
|
|
@@ -229,6 +244,7 @@ See the [Weaviate guide](https://github.com/arnsri33/embedflow/blob/main/docs/in
|
|
|
229
244
|
```bash
|
|
230
245
|
embedflow --help
|
|
231
246
|
embedflow analyze --help
|
|
247
|
+
embedflow plan --help
|
|
232
248
|
embedflow serve --config ./embedflow.yaml
|
|
233
249
|
embedflow status --config ./embedflow.yaml
|
|
234
250
|
embedflow registry list
|
|
@@ -242,7 +258,7 @@ cover the remaining commands and endpoints.
|
|
|
242
258
|
|
|
243
259
|
## Status
|
|
244
260
|
|
|
245
|
-
EmbedFlow v0.
|
|
261
|
+
EmbedFlow v0.6.0 is a pre-1.0 release for research and early real-world
|
|
246
262
|
testing. T2-v1 is an empirical finite-tail diagnostic, partial rankings can
|
|
247
263
|
differ from fully warm target reranking, and ANN fidelity needs a reference
|
|
248
264
|
comparison to audit.
|
|
@@ -142,6 +142,20 @@ session = embedflow.migrate(
|
|
|
142
142
|
results = session.search("what causes auroras?", top_k=10)
|
|
143
143
|
```
|
|
144
144
|
|
|
145
|
+
## Plan a migration
|
|
146
|
+
|
|
147
|
+
Build a conservative, evidence-aware recommendation before serving. The
|
|
148
|
+
planner reuses backend preflight, registry matching, and frozen T2-v1; it is
|
|
149
|
+
advisory and never routes traffic or mutates the source index.
|
|
150
|
+
|
|
151
|
+
```bash
|
|
152
|
+
embedflow plan --config ./embedflow.yaml --queries ./probe_queries.jsonl
|
|
153
|
+
embedflow plan --config ./embedflow.yaml --queries ./probe_queries.jsonl --format json --output migration-plan.json
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
`SAFE` is an empirical finite-tail signal, not a retrieval-quality guarantee.
|
|
157
|
+
See [`docs/planner.md`](https://github.com/arnsri33/embedflow/blob/main/docs/planner.md).
|
|
158
|
+
|
|
145
159
|
Search responses expose `COLD`, `PARTIAL`, or `WARM`, cache hits and misses,
|
|
146
160
|
synchronous work, queued work, and stage timings. Once the candidate vectors
|
|
147
161
|
are warm, target scoring over that candidate set is deterministic.
|
|
@@ -227,6 +241,7 @@ Backend-specific setup and examples:
|
|
|
227
241
|
```bash
|
|
228
242
|
embedflow --help
|
|
229
243
|
embedflow analyze --help
|
|
244
|
+
embedflow plan --help
|
|
230
245
|
embedflow serve --config ./embedflow.yaml
|
|
231
246
|
embedflow status --config ./embedflow.yaml
|
|
232
247
|
embedflow registry list
|
|
@@ -254,7 +269,7 @@ OpenAPI documentation; see
|
|
|
254
269
|
|
|
255
270
|
## Status
|
|
256
271
|
|
|
257
|
-
EmbedFlow v0.
|
|
272
|
+
EmbedFlow v0.6.0 is a pre-1.0 release for research and early real-world
|
|
258
273
|
testing.
|
|
259
274
|
|
|
260
275
|
- T2-v1 reports an empirical finite-tail diagnostic.
|
|
@@ -123,6 +123,21 @@ Selected core records (nDCG@10, `G(50)`):
|
|
|
123
123
|
See the [registry documentation](https://github.com/arnsri33/embedflow/blob/main/docs/registry.md)
|
|
124
124
|
for matching levels, contract fingerprints, and provenance.
|
|
125
125
|
|
|
126
|
+
## Advisory migration plans
|
|
127
|
+
|
|
128
|
+
Use representative query probes to produce a bounded migration recommendation
|
|
129
|
+
before serving:
|
|
130
|
+
|
|
131
|
+
```bash
|
|
132
|
+
embedflow plan --config ./embedflow.yaml --queries ./probe_queries.jsonl
|
|
133
|
+
embedflow plan --config ./embedflow.yaml --queries ./probe_queries.jsonl --format json --output migration-plan.json
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
The planner reports preflight, registry evidence, frozen T2-v1 finite-tail
|
|
137
|
+
behavior, candidate K, cache/economics projections, and staged rollout
|
|
138
|
+
guidance. It never routes traffic or mutates the source index; `SAFE` is not a
|
|
139
|
+
qrels-based retrieval-quality guarantee. See the [planner guide](https://github.com/arnsri33/embedflow/blob/main/docs/planner.md).
|
|
140
|
+
|
|
126
141
|
## Research
|
|
127
142
|
|
|
128
143
|
For candidate depth `K`, EmbedFlow measures:
|
|
@@ -163,6 +178,7 @@ See the [Weaviate guide](https://github.com/arnsri33/embedflow/blob/main/docs/in
|
|
|
163
178
|
```bash
|
|
164
179
|
embedflow --help
|
|
165
180
|
embedflow analyze --help
|
|
181
|
+
embedflow plan --help
|
|
166
182
|
embedflow serve --config ./embedflow.yaml
|
|
167
183
|
embedflow status --config ./embedflow.yaml
|
|
168
184
|
embedflow registry list
|
|
@@ -176,7 +192,7 @@ cover the remaining commands and endpoints.
|
|
|
176
192
|
|
|
177
193
|
## Status
|
|
178
194
|
|
|
179
|
-
EmbedFlow v0.
|
|
195
|
+
EmbedFlow v0.6.0 is a pre-1.0 release for research and early real-world
|
|
180
196
|
testing. T2-v1 is an empirical finite-tail diagnostic, partial rankings can
|
|
181
197
|
differ from fully warm target reranking, and ANN fidelity needs a reference
|
|
182
198
|
comparison to audit.
|
|
@@ -48,6 +48,11 @@ The response includes the result list and migration fields such as:
|
|
|
48
48
|
request. A partial response scores the available target vectors; it can differ
|
|
49
49
|
from the fully warm ranking.
|
|
50
50
|
|
|
51
|
+
The advisory migration planner is exposed through the Python API and the
|
|
52
|
+
`embedflow plan` CLI. It is intentionally not a synchronous FastAPI endpoint:
|
|
53
|
+
probe analysis may load models and perform bounded candidate work, so operators
|
|
54
|
+
should generate a plan artifact out of band and publish/read it as needed.
|
|
55
|
+
|
|
51
56
|
`/status` includes safe backend metadata. For pgvector this names the
|
|
52
57
|
schema/table and vector contract; for Pinecone it names the host/index,
|
|
53
58
|
namespace, dimension, metric, and safe vector counts. Neither backend returns
|
|
@@ -9,6 +9,7 @@ list. The commands below are the main public entry points.
|
|
|
9
9
|
embedflow init --config ./embedflow.yaml
|
|
10
10
|
embedflow analyze --config ./embedflow.yaml --output-dir ./analysis
|
|
11
11
|
embedflow evaluate --config ./experiment.yaml --output-dir ./results
|
|
12
|
+
embedflow plan --config ./embedflow.yaml --queries ./probe_queries.jsonl
|
|
12
13
|
```
|
|
13
14
|
|
|
14
15
|
`analyze` is the no-target-index workflow. It uses probe queries and frozen
|
|
@@ -16,6 +17,15 @@ T2-v1 logic. `evaluate` is the labelled workflow; it computes source quality,
|
|
|
16
17
|
native target quality, target-within-source-candidates quality, candidate gap,
|
|
17
18
|
containment, and migration depth when the required inputs are available.
|
|
18
19
|
|
|
20
|
+
`plan` is the advisory migration planner. It runs source preflight and registry
|
|
21
|
+
matching, optionally samples a JSONL probe set, and reports a candidate-depth
|
|
22
|
+
recommendation, cache/economics projections, warnings, and staged rollout
|
|
23
|
+
guidance. Use `--format json` (or `yaml`) for a structured artifact. A plan
|
|
24
|
+
with no probes reports `T2-v1: NOT_RUN`; `SAFE` is a finite-tail diagnostic,
|
|
25
|
+
not a qrels-based retrieval-quality guarantee. A `DEFER`/`EXPAND_PROBE` result
|
|
26
|
+
is still a successful analytical command and exits zero; invalid configuration
|
|
27
|
+
or failed preflight exits nonzero.
|
|
28
|
+
|
|
19
29
|
## Serving and operations
|
|
20
30
|
|
|
21
31
|
```bash
|
|
@@ -58,6 +58,25 @@ economics:
|
|
|
58
58
|
gpu_price_per_hour: null
|
|
59
59
|
target_docs_per_second: null
|
|
60
60
|
|
|
61
|
+
# Optional advisory planner settings. CLI flags override these values.
|
|
62
|
+
planner:
|
|
63
|
+
max_probes: 250
|
|
64
|
+
seed: 42
|
|
65
|
+
k_grid: [20, 50, 100, 200, 500]
|
|
66
|
+
max_candidates: null
|
|
67
|
+
max_target_encodes: null
|
|
68
|
+
max_sync_misses: null
|
|
69
|
+
background_batch_size: null
|
|
70
|
+
gpu_hourly_cost: null
|
|
71
|
+
target_docs_per_second: null
|
|
72
|
+
queries_per_second: null
|
|
73
|
+
daily_queries: null
|
|
74
|
+
cache_hit_rate: null
|
|
75
|
+
latency_budget_ms: null
|
|
76
|
+
access_trace: null
|
|
77
|
+
corpus_name: null
|
|
78
|
+
corpus_fingerprint: null
|
|
79
|
+
|
|
61
80
|
state_path: ./embedflow_state.json
|
|
62
81
|
```
|
|
63
82
|
|
|
@@ -130,6 +149,21 @@ EMBEDFLOW_CANDIDATE_DEPTH
|
|
|
130
149
|
EMBEDFLOW_MAX_SYNC_MISSES
|
|
131
150
|
EMBEDFLOW_BACKGROUND_BATCH_SIZE
|
|
132
151
|
EMBEDFLOW_PROBE_KMAX
|
|
152
|
+
EMBEDFLOW_PLANNER_MAX_PROBES
|
|
153
|
+
EMBEDFLOW_PLANNER_SEED
|
|
154
|
+
EMBEDFLOW_PLANNER_MAX_CANDIDATES
|
|
155
|
+
EMBEDFLOW_PLANNER_MAX_TARGET_ENCODINGS
|
|
156
|
+
EMBEDFLOW_PLANNER_MAX_SYNC_MISSES
|
|
157
|
+
EMBEDFLOW_PLANNER_BACKGROUND_BATCH_SIZE
|
|
158
|
+
EMBEDFLOW_PLANNER_GPU_HOURLY_COST
|
|
159
|
+
EMBEDFLOW_PLANNER_TARGET_DOCS_PER_SECOND
|
|
160
|
+
EMBEDFLOW_PLANNER_QPS
|
|
161
|
+
EMBEDFLOW_PLANNER_DAILY_QUERIES
|
|
162
|
+
EMBEDFLOW_PLANNER_CACHE_HIT_RATE
|
|
163
|
+
EMBEDFLOW_PLANNER_LATENCY_BUDGET_MS
|
|
164
|
+
EMBEDFLOW_PLANNER_ACCESS_TRACE
|
|
165
|
+
EMBEDFLOW_PLANNER_CORPUS_NAME
|
|
166
|
+
EMBEDFLOW_PLANNER_CORPUS_FINGERPRINT
|
|
133
167
|
```
|
|
134
168
|
|
|
135
169
|
## Input files
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Limitations and release scope
|
|
2
2
|
|
|
3
|
-
EmbedFlow v0.
|
|
3
|
+
EmbedFlow v0.6.0 is a pre-1.0 release for research and early real-world
|
|
4
4
|
testing. The serving path is designed to make migration experiments concrete;
|
|
5
5
|
production rollout still requires application-specific validation.
|
|
6
6
|
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
# Migration planner
|
|
2
|
+
|
|
3
|
+
`embedflow plan` is a bounded, advisory analysis for moving from an existing
|
|
4
|
+
source vector index to a target embedding model. It composes the existing
|
|
5
|
+
backend audit, model contracts, retained registry evidence, probe retrieval,
|
|
6
|
+
frozen T2-v1 diagnostic, cache behavior, and economics helpers. It never
|
|
7
|
+
routes traffic and does not create, update, or delete source-index data.
|
|
8
|
+
|
|
9
|
+
## Quickstart
|
|
10
|
+
|
|
11
|
+
Install EmbedFlow with the optional dependency for the configured backend, then
|
|
12
|
+
provide representative query probes as JSONL:
|
|
13
|
+
|
|
14
|
+
```json
|
|
15
|
+
{"id":"q1","query":"how do I reset my password?"}
|
|
16
|
+
{"id":"q2","query":"what is the vacation policy?"}
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
Run a terminal plan or save a machine-readable artifact:
|
|
20
|
+
|
|
21
|
+
```bash
|
|
22
|
+
embedflow plan --config embedflow.yaml --queries probes.jsonl
|
|
23
|
+
embedflow plan --config embedflow.yaml --queries probes.jsonl \
|
|
24
|
+
--k-grid 20,50,100,200,500 --format json --output migration-plan.json
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
For a deterministic, network-free walkthrough, run `examples/planner/run_demo.sh`.
|
|
28
|
+
The demo uses the same planner API and tiny hash encoders; it is not a quality
|
|
29
|
+
benchmark.
|
|
30
|
+
|
|
31
|
+
## What the result means
|
|
32
|
+
|
|
33
|
+
The result is a versioned (`schema_version: 1`) object with source/target
|
|
34
|
+
contracts, preflight checks, evidence, candidate-depth diagnostics, cache and
|
|
35
|
+
rollout guidance, economics, and structured warnings. The Python API is:
|
|
36
|
+
|
|
37
|
+
```python
|
|
38
|
+
from embedflow.planner import MigrationPlanner
|
|
39
|
+
|
|
40
|
+
result = MigrationPlanner("embedflow.yaml").plan("probes.jsonl")
|
|
41
|
+
artifact = result.to_dict()
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
Recommendation states are deliberately small:
|
|
45
|
+
|
|
46
|
+
- `PROCEED`: probe evidence and preflight are strong enough for an operator-controlled shadow/canary sequence.
|
|
47
|
+
- `PROCEED_WITH_CAUTION`: probes are acceptable, but evidence coverage or contract/ANN provenance is limited.
|
|
48
|
+
- `EXPAND_PROBE`: T2-v1 asks for a deeper pool or the evidence is too small.
|
|
49
|
+
- `DEFER`: finite-tail behavior is `UNSAFE_OR_UNCERTAIN`; do not canary based on this plan.
|
|
50
|
+
- `BLOCKED`: structural/backend preflight failed; no candidate K is certified.
|
|
51
|
+
|
|
52
|
+
These are recommendations, not automatic deployment actions. The suggested
|
|
53
|
+
rollout is Validate → Shadow → small canary → expanded canary → target-primary
|
|
54
|
+
path, with the source remaining authoritative until the operator approves each
|
|
55
|
+
step. Stop conditions include source/target errors, p95 latency, queue depth,
|
|
56
|
+
cache hit rate, and any native/qrels evaluation regressions.
|
|
57
|
+
|
|
58
|
+
## Evidence and T2-v1
|
|
59
|
+
|
|
60
|
+
Registry matching preserves the existing hierarchy: `EXACT REGISTRY MATCH`,
|
|
61
|
+
`PRIOR EVIDENCE AVAILABLE` (same contracts, different/unknown corpus),
|
|
62
|
+
`RELATED EVIDENCE ONLY`, and `NO REGISTRY MATCH`. Prior or related evidence can
|
|
63
|
+
seed context but cannot override current probe results or certify a new corpus.
|
|
64
|
+
|
|
65
|
+
The planner calls the existing frozen T2-v1 implementation unchanged. T2-v1 is
|
|
66
|
+
a no-qrels finite-tail diagnostic of candidate behavior. `SAFE` does **not**
|
|
67
|
+
prove a candidate gap, nDCG, recall, or zero quality loss. Without actual qrels
|
|
68
|
+
or native-target evaluation, `candidate_gap` is `null`/`UNKNOWN`. ANN fidelity
|
|
69
|
+
is also `UNKNOWN` unless an exact/reference comparison is supplied by the
|
|
70
|
+
backend audit workflow.
|
|
71
|
+
|
|
72
|
+
No probes is a valid preflight/economics mode, but T2 is `NOT_RUN`, confidence
|
|
73
|
+
is `INSUFFICIENT`, and no K is presented as safe. Probe JSONL rows may use
|
|
74
|
+
`query` (or the existing `text` alias) and an optional `id`; empty, malformed,
|
|
75
|
+
or duplicate IDs are rejected. `--max-probes` and `--seed` provide deterministic
|
|
76
|
+
bounded sampling. Candidate documents are deduplicated by canonical ID and
|
|
77
|
+
encoded in bounded batches; `--max-target-encodes` can impose a hard cap.
|
|
78
|
+
|
|
79
|
+
## Cache, performance, and economics
|
|
80
|
+
|
|
81
|
+
Planning uses an ephemeral target-cache/state directory when it opens a runtime,
|
|
82
|
+
so the normal serving cache is not polluted. It recommends conservative sync
|
|
83
|
+
miss and background batch settings and defaults to traffic-driven progressive
|
|
84
|
+
warming when no access trace is supplied. An access trace can be supplied with
|
|
85
|
+
`--access-trace` as rows such as `{"document_id":"abc","count":192}` to
|
|
86
|
+
model hot-document coverage.
|
|
87
|
+
|
|
88
|
+
Performance values retain provenance (`measured`, `user_supplied`, `modeled`,
|
|
89
|
+
`registry`, or `unknown`). `--profile` performs only a small local encode/search
|
|
90
|
+
sample with one unmeasured warmup pass excluded; it is diagnostic and is not a
|
|
91
|
+
formal benchmark. Economics accepts `--target-docs-per-second` and
|
|
92
|
+
`--gpu-hourly-cost`; absent inputs remain `UNKNOWN`. Raw vector storage is
|
|
93
|
+
`documents × target dimension × dtype bytes` and excludes ANN overhead,
|
|
94
|
+
metadata, replicas, backups, and database overhead. Progressive percentages
|
|
95
|
+
are materialization scenarios, not claims that a percentage is sufficient.
|
|
96
|
+
|
|
97
|
+
## Configuration and privacy
|
|
98
|
+
|
|
99
|
+
An optional `planner:` section mirrors the CLI options (`max_probes`, `seed`,
|
|
100
|
+
`k_grid`, work caps, cache hints, throughput/cost inputs, and corpus identity).
|
|
101
|
+
CLI values take precedence. `EMBEDFLOW_PLANNER_*` environment overrides follow
|
|
102
|
+
the normal config mechanism. Probe text is never copied into plan artifacts or
|
|
103
|
+
telemetry by default; only counts, IDs used for diagnostics, and aggregates are
|
|
104
|
+
reported. Backend credentials are handled by the existing backend adapters and
|
|
105
|
+
are redacted from errors and structured output.
|
|
106
|
+
|
|
107
|
+
The planner does not provide qrel evaluation, traffic routing, rollback
|
|
108
|
+
automation, distributed scheduling, ANN tuning, integrated-vectorizer contract
|
|
109
|
+
verification, or production cost guarantees. Use `embedflow evaluate` with
|
|
110
|
+
qrels/native target rankings when empirical retrieval-quality claims are
|
|
111
|
+
required.
|
|
@@ -18,8 +18,8 @@ python -m twine check dist/*
|
|
|
18
18
|
Inspect both archives before uploading:
|
|
19
19
|
|
|
20
20
|
```bash
|
|
21
|
-
unzip -l dist/embedflow-0.
|
|
22
|
-
tar -tzf dist/embedflow-0.
|
|
21
|
+
unzip -l dist/embedflow-0.6.0-py3-none-any.whl
|
|
22
|
+
tar -tzf dist/embedflow-0.6.0.tar.gz
|
|
23
23
|
sha256sum dist/*
|
|
24
24
|
```
|
|
25
25
|
|
|
@@ -32,7 +32,7 @@ Test the wheel outside the source tree:
|
|
|
32
32
|
```bash
|
|
33
33
|
python -m venv /tmp/embedflow-wheel-test
|
|
34
34
|
/tmp/embedflow-wheel-test/bin/python -m pip install --upgrade pip
|
|
35
|
-
/tmp/embedflow-wheel-test/bin/python -m pip install dist/embedflow-0.
|
|
35
|
+
/tmp/embedflow-wheel-test/bin/python -m pip install dist/embedflow-0.6.0-py3-none-any.whl
|
|
36
36
|
cd /tmp
|
|
37
37
|
/tmp/embedflow-wheel-test/bin/python -c "import embedflow; print(embedflow.__version__)"
|
|
38
38
|
/tmp/embedflow-wheel-test/bin/embedflow --help
|
|
@@ -65,7 +65,7 @@ python -m venv /tmp/embedflow-testpypi
|
|
|
65
65
|
/tmp/embedflow-testpypi/bin/python -m pip install \
|
|
66
66
|
--index-url https://test.pypi.org/simple/ \
|
|
67
67
|
--extra-index-url https://pypi.org/simple/ \
|
|
68
|
-
embedflow==0.
|
|
68
|
+
embedflow==0.6.0
|
|
69
69
|
cd /tmp
|
|
70
70
|
/tmp/embedflow-testpypi/bin/python -c "import embedflow; print(embedflow.__version__)"
|
|
71
71
|
/tmp/embedflow-testpypi/bin/embedflow --help
|
|
@@ -79,11 +79,11 @@ Test optional integrations in a second clean environment:
|
|
|
79
79
|
/tmp/embedflow-testpypi/bin/python -m pip install \
|
|
80
80
|
--index-url https://test.pypi.org/simple/ \
|
|
81
81
|
--extra-index-url https://pypi.org/simple/ \
|
|
82
|
-
"embedflow[faiss,dashboard,pinecone,milvus,weaviate]==0.
|
|
82
|
+
"embedflow[faiss,dashboard,pinecone,milvus,weaviate]==0.6.0"
|
|
83
83
|
```
|
|
84
84
|
|
|
85
85
|
If the same filename already exists on TestPyPI, use a pre-release such as
|
|
86
|
-
`0.
|
|
86
|
+
`0.6.0rc1` for the TestPyPI-only trial. Keep production `0.6.0` unchanged.
|
|
87
87
|
|
|
88
88
|
## Trusted Publishing configuration
|
|
89
89
|
|
|
@@ -115,7 +115,7 @@ above keeps the test step explicit.
|
|
|
115
115
|
3. Run the final release gate and review the generated report.
|
|
116
116
|
4. Configure the PyPI pending publisher and protected `pypi` environment.
|
|
117
117
|
5. Create a Git tag and GitHub Release for the exact package version, for
|
|
118
|
-
example `v0.
|
|
118
|
+
example `v0.6.0`.
|
|
119
119
|
6. Approve the `pypi` environment when the release workflow is ready.
|
|
120
120
|
7. Verify the files and metadata on PyPI.
|
|
121
121
|
8. Install from production PyPI in a directory outside this checkout.
|
|
@@ -1,8 +1,14 @@
|
|
|
1
1
|
"""EmbedFlow: progressive embedding-model migration for existing indexes."""
|
|
2
2
|
|
|
3
|
-
__version__ = "0.
|
|
3
|
+
__version__ = "0.6.0"
|
|
4
4
|
|
|
5
|
-
from .config import EmbedFlowConfig, load_config
|
|
5
|
+
from .config import EmbedFlowConfig, PlannerConfig, load_config
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def plan(*args, **kwargs):
|
|
9
|
+
"""Generate an advisory migration plan without changing source traffic."""
|
|
10
|
+
from .planner import plan_migration
|
|
11
|
+
return plan_migration(*args, **kwargs)
|
|
6
12
|
|
|
7
13
|
|
|
8
14
|
def migrate(*args, **kwargs):
|
|
@@ -22,4 +28,4 @@ def analyze_migration(*args, **kwargs):
|
|
|
22
28
|
return _analyze_migration(*args, **kwargs)
|
|
23
29
|
|
|
24
30
|
|
|
25
|
-
__all__ = ["EmbedFlowConfig", "load_config", "migrate", "analyze_migration", "__version__"]
|
|
31
|
+
__all__ = ["EmbedFlowConfig", "PlannerConfig", "load_config", "migrate", "plan", "analyze_migration", "__version__"]
|
|
@@ -7,6 +7,7 @@ import os
|
|
|
7
7
|
import platform
|
|
8
8
|
import random
|
|
9
9
|
import sys
|
|
10
|
+
import tempfile
|
|
10
11
|
from pathlib import Path
|
|
11
12
|
from typing import Any
|
|
12
13
|
|
|
@@ -31,6 +32,7 @@ from .config import (
|
|
|
31
32
|
from .migration.compatibility import run_probe, save_probe
|
|
32
33
|
from .migration.state import DocumentStore
|
|
33
34
|
from .models import HashEmbeddingModel, load_embedding_model
|
|
35
|
+
from .planner import MigrationPlanner, render_plan
|
|
34
36
|
from .registry import (
|
|
35
37
|
MATCH_EXACT,
|
|
36
38
|
load_benchmark_profiles,
|
|
@@ -45,6 +47,26 @@ from .runtime import build_faiss_from_documents, load_documents, open_engine
|
|
|
45
47
|
def _json(value: Any) -> None: print(json.dumps(value, indent=2, ensure_ascii=False, default=float))
|
|
46
48
|
|
|
47
49
|
|
|
50
|
+
def _atomic_write_text(path: Path, text: str) -> None:
|
|
51
|
+
"""Replace a text artifact atomically in its destination directory."""
|
|
52
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
53
|
+
temporary: str | None = None
|
|
54
|
+
try:
|
|
55
|
+
fd, temporary = tempfile.mkstemp(prefix=f".{path.name}.", suffix=".tmp", dir=str(path.parent))
|
|
56
|
+
with os.fdopen(fd, "w", encoding="utf-8") as handle:
|
|
57
|
+
handle.write(text)
|
|
58
|
+
handle.flush()
|
|
59
|
+
os.fsync(handle.fileno())
|
|
60
|
+
os.replace(temporary, path)
|
|
61
|
+
temporary = None
|
|
62
|
+
finally:
|
|
63
|
+
if temporary is not None:
|
|
64
|
+
try:
|
|
65
|
+
os.unlink(temporary)
|
|
66
|
+
except FileNotFoundError:
|
|
67
|
+
pass
|
|
68
|
+
|
|
69
|
+
|
|
48
70
|
def _normalize_device(value: str | None) -> str | None:
|
|
49
71
|
"""Accept the common human spelling ``gpu`` while PyTorch uses ``cuda``."""
|
|
50
72
|
if value is None:
|
|
@@ -606,6 +628,53 @@ def cmd_analyze(args: argparse.Namespace) -> int:
|
|
|
606
628
|
return 0
|
|
607
629
|
|
|
608
630
|
|
|
631
|
+
def cmd_plan(args: argparse.Namespace) -> int:
|
|
632
|
+
"""Generate a structured, advisory migration plan.
|
|
633
|
+
|
|
634
|
+
Planning uses an isolated temporary target-cache/state directory and never
|
|
635
|
+
routes traffic or writes to the configured source index. Progress is sent
|
|
636
|
+
to stderr so JSON/YAML stdout remains machine-readable.
|
|
637
|
+
"""
|
|
638
|
+
planner = MigrationPlanner(args.config, device=_normalize_device(args.device), demo=args.demo,
|
|
639
|
+
model_root=args.model_root)
|
|
640
|
+
progress = None if args.quiet or args.format != "text" else lambda message: print(message, file=sys.stderr)
|
|
641
|
+
result = planner.plan(
|
|
642
|
+
probe_queries=args.queries,
|
|
643
|
+
max_probes=args.max_probes,
|
|
644
|
+
seed=args.seed,
|
|
645
|
+
k_grid=args.k_grid,
|
|
646
|
+
max_candidates=args.max_candidates,
|
|
647
|
+
max_target_encodes=args.max_target_encodes,
|
|
648
|
+
profile=args.profile,
|
|
649
|
+
gpu_hourly_cost=args.gpu_hourly_cost,
|
|
650
|
+
target_docs_per_second=args.target_docs_per_second,
|
|
651
|
+
queries_per_second=args.queries_per_second,
|
|
652
|
+
daily_queries=args.daily_queries,
|
|
653
|
+
cache_hit_rate=args.cache_hit_rate,
|
|
654
|
+
latency_budget_ms=args.latency_budget_ms,
|
|
655
|
+
access_trace=args.access_trace,
|
|
656
|
+
corpus_name=args.corpus_name,
|
|
657
|
+
corpus_fingerprint=args.corpus_fingerprint,
|
|
658
|
+
progress=progress,
|
|
659
|
+
)
|
|
660
|
+
if args.format == "json":
|
|
661
|
+
rendered = result.to_json()
|
|
662
|
+
elif args.format == "yaml":
|
|
663
|
+
rendered = result.to_yaml()
|
|
664
|
+
else:
|
|
665
|
+
rendered = render_plan(result)
|
|
666
|
+
if args.output:
|
|
667
|
+
destination = Path(args.output).expanduser()
|
|
668
|
+
_atomic_write_text(destination, rendered)
|
|
669
|
+
if not args.quiet:
|
|
670
|
+
print(f"wrote migration plan to {destination}", file=sys.stderr)
|
|
671
|
+
if not args.output or not args.quiet:
|
|
672
|
+
print(rendered, end="" if rendered.endswith("\n") else "\n")
|
|
673
|
+
# A DEFER/EXPAND result is a valid analytical outcome; only structural
|
|
674
|
+
# configuration/backend failures raise and become a non-zero CLI exit.
|
|
675
|
+
return 0
|
|
676
|
+
|
|
677
|
+
|
|
609
678
|
def cmd_evaluate(args: argparse.Namespace) -> int:
|
|
610
679
|
"""Run Mode A evaluation with qrels and native target evidence."""
|
|
611
680
|
cfg = load_config(args.config)
|
|
@@ -882,7 +951,11 @@ def cmd_doctor(args: argparse.Namespace) -> int:
|
|
|
882
951
|
checks.append({"name": "target_fingerprint", "ok": True, "detail": config.target.fingerprint[:16]})
|
|
883
952
|
except Exception as exc:
|
|
884
953
|
checks.append({"name": "config", "ok": False, "detail": str(exc)})
|
|
885
|
-
|
|
954
|
+
# The display label for the ``weaviate`` import is ``weaviate_client``
|
|
955
|
+
# (``weaviate-client`` is the distribution name). Keep both spellings
|
|
956
|
+
# optional so a base install does not fail ``doctor`` merely because an
|
|
957
|
+
# optional backend is absent.
|
|
958
|
+
optional_checks = {"faiss", "qdrant_client", "fastapi", "torch", "pytorch", "psycopg", "pinecone", "pymilvus", "weaviate", "weaviate_client"}
|
|
886
959
|
failed = [check for check in checks if not check["ok"] and check["name"] not in optional_checks]
|
|
887
960
|
if args.json:
|
|
888
961
|
_json({"checks": checks, "status": "FAIL" if failed else "PASS"})
|
|
@@ -1425,6 +1498,31 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
1425
1498
|
analyze.add_argument("--output")
|
|
1426
1499
|
analyze.add_argument("--output-dir")
|
|
1427
1500
|
analyze.set_defaults(func=cmd_analyze)
|
|
1501
|
+
plan_cmd = sub.add_parser("plan", help="build an advisory migration plan from source-index evidence")
|
|
1502
|
+
plan_cmd.add_argument("--config", required=True, help="EmbedFlow YAML configuration")
|
|
1503
|
+
plan_cmd.add_argument("--queries", help="JSONL probe queries; omit to run preflight/evidence/economics only")
|
|
1504
|
+
plan_cmd.add_argument("--max-probes", type=int, help="maximum deterministic probe sample size")
|
|
1505
|
+
plan_cmd.add_argument("--seed", type=int, help="deterministic probe sampling seed")
|
|
1506
|
+
plan_cmd.add_argument("--k-grid", help="comma-separated candidate depths, e.g. 20,50,100,200,500")
|
|
1507
|
+
plan_cmd.add_argument("--max-candidates", type=int, help="hard cap on candidate depth/work")
|
|
1508
|
+
plan_cmd.add_argument("--max-target-encodes", type=int, help="bound unique target document encodes")
|
|
1509
|
+
plan_cmd.add_argument("--format", choices=["text", "json", "yaml"], default="text")
|
|
1510
|
+
plan_cmd.add_argument("--output", help="optional plan artifact path")
|
|
1511
|
+
plan_cmd.add_argument("--profile", action="store_true", help="run a small optional local encode/search profile")
|
|
1512
|
+
plan_cmd.add_argument("--gpu-hourly-cost", type=float)
|
|
1513
|
+
plan_cmd.add_argument("--target-docs-per-second", type=float)
|
|
1514
|
+
plan_cmd.add_argument("--queries-per-second", type=float)
|
|
1515
|
+
plan_cmd.add_argument("--daily-queries", type=float)
|
|
1516
|
+
plan_cmd.add_argument("--cache-hit-rate", type=float)
|
|
1517
|
+
plan_cmd.add_argument("--latency-budget-ms", type=float)
|
|
1518
|
+
plan_cmd.add_argument("--access-trace", help="optional JSONL document access trace")
|
|
1519
|
+
plan_cmd.add_argument("--corpus-name", help="canonical registry corpus identifier, when known")
|
|
1520
|
+
plan_cmd.add_argument("--corpus-fingerprint", help="precomputed corpus fingerprint for exact registry matching")
|
|
1521
|
+
plan_cmd.add_argument("--model-root", help="directory containing staged model snapshots")
|
|
1522
|
+
plan_cmd.add_argument("--device", default=None, help="model device (cpu, cuda, or gpu)")
|
|
1523
|
+
plan_cmd.add_argument("--demo", action="store_true", help="use deterministic demo encoders")
|
|
1524
|
+
plan_cmd.add_argument("--quiet", action="store_true", help="suppress progress and stdout when --output is supplied")
|
|
1525
|
+
plan_cmd.set_defaults(func=cmd_plan)
|
|
1428
1526
|
evaluate = sub.add_parser("evaluate", help="compute qrels/native-target candidate gaps (Mode A)")
|
|
1429
1527
|
evaluate.add_argument("--config", required=True)
|
|
1430
1528
|
evaluate.add_argument("--queries")
|