embedflow 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. {embedflow-0.2.0 → embedflow-0.3.0}/CHANGELOG.md +8 -0
  2. {embedflow-0.2.0 → embedflow-0.3.0}/CITATION.cff +1 -1
  3. {embedflow-0.2.0 → embedflow-0.3.0}/CONTRIBUTING.md +1 -1
  4. {embedflow-0.2.0 → embedflow-0.3.0}/PKG-INFO +16 -5
  5. {embedflow-0.2.0 → embedflow-0.3.0}/README.md +10 -2
  6. {embedflow-0.2.0 → embedflow-0.3.0}/README_PYPI.md +11 -3
  7. {embedflow-0.2.0 → embedflow-0.3.0}/docs/api.md +3 -1
  8. {embedflow-0.2.0 → embedflow-0.3.0}/docs/configuration.md +8 -1
  9. {embedflow-0.2.0 → embedflow-0.3.0}/docs/installation.md +7 -0
  10. embedflow-0.3.0/docs/integrations/pinecone.md +157 -0
  11. {embedflow-0.2.0 → embedflow-0.3.0}/docs/limitations.md +1 -1
  12. {embedflow-0.2.0 → embedflow-0.3.0}/docs/releasing.md +7 -7
  13. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/__init__.py +2 -2
  14. embedflow-0.3.0/embedflow/api.py +10 -0
  15. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/cli.py +101 -31
  16. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/config.py +48 -7
  17. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/indexes/__init__.py +2 -0
  18. embedflow-0.3.0/embedflow/indexes/pinecone_backend.py +764 -0
  19. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/migration/facade.py +42 -15
  20. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/runtime.py +24 -4
  21. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow.egg-info/PKG-INFO +16 -5
  22. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow.egg-info/SOURCES.txt +6 -0
  23. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow.egg-info/requires.txt +4 -0
  24. embedflow-0.3.0/examples/pinecone/README.md +32 -0
  25. embedflow-0.3.0/examples/pinecone/run_smoke.sh +8 -0
  26. {embedflow-0.2.0 → embedflow-0.3.0}/pyproject.toml +4 -2
  27. embedflow-0.3.0/scripts/pinecone_smoke.py +54 -0
  28. {embedflow-0.2.0 → embedflow-0.3.0}/scripts/release_gate.py +9 -4
  29. {embedflow-0.2.0 → embedflow-0.3.0}/scripts/validate_pgvector_10k.py +2 -2
  30. {embedflow-0.2.0 → embedflow-0.3.0}/LICENSE +0 -0
  31. {embedflow-0.2.0 → embedflow-0.3.0}/MANIFEST.in +0 -0
  32. {embedflow-0.2.0 → embedflow-0.3.0}/SECURITY.md +0 -0
  33. {embedflow-0.2.0 → embedflow-0.3.0}/docs/assets/README.md +0 -0
  34. {embedflow-0.2.0 → embedflow-0.3.0}/docs/assets/candidate-gap-example.svg +0 -0
  35. {embedflow-0.2.0 → embedflow-0.3.0}/docs/assets/dashboard-screenshot.md +0 -0
  36. {embedflow-0.2.0 → embedflow-0.3.0}/docs/assets/terminal-demo.txt +0 -0
  37. {embedflow-0.2.0 → embedflow-0.3.0}/docs/cli.md +0 -0
  38. {embedflow-0.2.0 → embedflow-0.3.0}/docs/concepts.md +0 -0
  39. {embedflow-0.2.0 → embedflow-0.3.0}/docs/contributing-benchmarks.md +0 -0
  40. {embedflow-0.2.0 → embedflow-0.3.0}/docs/economics.md +0 -0
  41. {embedflow-0.2.0 → embedflow-0.3.0}/docs/integrations/faiss.md +0 -0
  42. {embedflow-0.2.0 → embedflow-0.3.0}/docs/integrations/pgvector.md +0 -0
  43. {embedflow-0.2.0 → embedflow-0.3.0}/docs/integrations/qdrant.md +0 -0
  44. {embedflow-0.2.0 → embedflow-0.3.0}/docs/methodology.md +0 -0
  45. {embedflow-0.2.0 → embedflow-0.3.0}/docs/quickstart.md +0 -0
  46. {embedflow-0.2.0 → embedflow-0.3.0}/docs/registry.md +0 -0
  47. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/__main__.py +0 -0
  48. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/analysis.py +0 -0
  49. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/cache/__init__.py +0 -0
  50. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/cache/base.py +0 -0
  51. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/cache/persistent_cache.py +0 -0
  52. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/compatibility/__init__.py +0 -0
  53. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/compatibility/candidate_gap.py +0 -0
  54. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/compatibility/containment.py +0 -0
  55. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/compatibility/evaluate.py +0 -0
  56. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/compatibility/metrics.py +0 -0
  57. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/compatibility/migration_depth.py +0 -0
  58. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/compatibility/probe.py +0 -0
  59. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/compatibility/report.py +0 -0
  60. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/compatibility/t2.py +0 -0
  61. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/data/__init__.py +0 -0
  62. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/data/registry/__init__.py +0 -0
  63. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/data/registry/benchmark_profiles.jsonl +0 -0
  64. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/data/registry/checksums.sha256 +0 -0
  65. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/data/registry/migrations.jsonl +0 -0
  66. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/data/registry/registry_manifest.json +0 -0
  67. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/data/registry/research_summaries.json +0 -0
  68. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/data/registry/schema_version.json +0 -0
  69. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/frozen/T2_V1_FROZEN_SPEC.md +0 -0
  70. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/frozen/T2_V1_FROZEN_SPEC.sha256 +0 -0
  71. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/indexes/base.py +0 -0
  72. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/indexes/faiss_backend.py +0 -0
  73. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/indexes/pgvector_backend.py +0 -0
  74. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/indexes/qdrant_backend.py +0 -0
  75. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/metrics/__init__.py +0 -0
  76. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/metrics/latency.py +0 -0
  77. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/migration/__init__.py +0 -0
  78. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/migration/compatibility.py +0 -0
  79. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/migration/materializer.py +0 -0
  80. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/migration/planner.py +0 -0
  81. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/migration/state.py +0 -0
  82. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/models/__init__.py +0 -0
  83. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/models/base.py +0 -0
  84. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/models/huggingface.py +0 -0
  85. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/registry/__init__.py +0 -0
  86. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/registry/loader.py +0 -0
  87. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/registry/matcher.py +0 -0
  88. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/registry/schema.py +0 -0
  89. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/serving/__init__.py +0 -0
  90. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/serving/api.py +0 -0
  91. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/serving/engine.py +0 -0
  92. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/serving/factory.py +0 -0
  93. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow/serving/schemas.py +0 -0
  94. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow.egg-info/dependency_links.txt +0 -0
  95. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow.egg-info/entry_points.txt +0 -0
  96. {embedflow-0.2.0 → embedflow-0.3.0}/embedflow.egg-info/top_level.txt +0 -0
  97. {embedflow-0.2.0 → embedflow-0.3.0}/examples/faiss/README.md +0 -0
  98. {embedflow-0.2.0 → embedflow-0.3.0}/examples/faiss/documents.jsonl +0 -0
  99. {embedflow-0.2.0 → embedflow-0.3.0}/examples/faiss/embedflow.yaml +0 -0
  100. {embedflow-0.2.0 → embedflow-0.3.0}/examples/faiss/queries.jsonl +0 -0
  101. {embedflow-0.2.0 → embedflow-0.3.0}/examples/pgvector/README.md +0 -0
  102. {embedflow-0.2.0 → embedflow-0.3.0}/examples/pgvector/build_index.py +0 -0
  103. {embedflow-0.2.0 → embedflow-0.3.0}/examples/pgvector/compose.yaml +0 -0
  104. {embedflow-0.2.0 → embedflow-0.3.0}/examples/pgvector/embedflow.yaml +0 -0
  105. {embedflow-0.2.0 → embedflow-0.3.0}/examples/pgvector/init.sql +0 -0
  106. {embedflow-0.2.0 → embedflow-0.3.0}/examples/pgvector/queries.jsonl +0 -0
  107. {embedflow-0.2.0 → embedflow-0.3.0}/examples/pgvector/run_demo.sh +0 -0
  108. {embedflow-0.2.0 → embedflow-0.3.0}/examples/qdrant/README.md +0 -0
  109. {embedflow-0.2.0 → embedflow-0.3.0}/examples/qdrant/build_index.py +0 -0
  110. {embedflow-0.2.0 → embedflow-0.3.0}/examples/qdrant/documents.jsonl +0 -0
  111. {embedflow-0.2.0 → embedflow-0.3.0}/examples/qdrant/embedflow.yaml +0 -0
  112. {embedflow-0.2.0 → embedflow-0.3.0}/examples/qdrant/queries.jsonl +0 -0
  113. {embedflow-0.2.0 → embedflow-0.3.0}/examples/research_analysis/README.md +0 -0
  114. {embedflow-0.2.0 → embedflow-0.3.0}/examples/research_analysis/documents.jsonl +0 -0
  115. {embedflow-0.2.0 → embedflow-0.3.0}/examples/research_analysis/embedflow.yaml +0 -0
  116. {embedflow-0.2.0 → embedflow-0.3.0}/examples/research_analysis/qrels.json +0 -0
  117. {embedflow-0.2.0 → embedflow-0.3.0}/examples/research_analysis/queries.jsonl +0 -0
  118. {embedflow-0.2.0 → embedflow-0.3.0}/frozen/T2_V1_FROZEN_SPEC.md +0 -0
  119. {embedflow-0.2.0 → embedflow-0.3.0}/frozen/T2_V1_FROZEN_SPEC.sha256 +0 -0
  120. {embedflow-0.2.0 → embedflow-0.3.0}/requirements-dev.txt +0 -0
  121. {embedflow-0.2.0 → embedflow-0.3.0}/requirements.txt +0 -0
  122. {embedflow-0.2.0 → embedflow-0.3.0}/scripts/real_qdrant_smoke.py +0 -0
  123. {embedflow-0.2.0 → embedflow-0.3.0}/scripts/run_demo.sh +0 -0
  124. {embedflow-0.2.0 → embedflow-0.3.0}/scripts/run_tests.sh +0 -0
  125. {embedflow-0.2.0 → embedflow-0.3.0}/setup.cfg +0 -0
  126. {embedflow-0.2.0 → embedflow-0.3.0}/src/__init__.py +0 -0
  127. {embedflow-0.2.0 → embedflow-0.3.0}/src/embed.py +0 -0
  128. {embedflow-0.2.0 → embedflow-0.3.0}/src/probe_features.py +0 -0
  129. {embedflow-0.2.0 → embedflow-0.3.0}/src/storage.py +0 -0
  130. {embedflow-0.2.0 → embedflow-0.3.0}/src/t2_v1.py +0 -0
  131. {embedflow-0.2.0 → embedflow-0.3.0}/src/utils.py +0 -0
@@ -1,5 +1,13 @@
1
1
  # Changelog
2
2
 
3
+ ## v0.3.0 — Pinecone backend
4
+
5
+ - Added a read-only Pinecone backend for existing dense indexes.
6
+ - Added host targeting, index-name resolution, namespace-aware retrieval, and
7
+ metadata or external-document text resolution.
8
+ - Added Pinecone status/audit integration, optional dependency packaging, and
9
+ unit/integration smoke fixtures.
10
+
3
11
  ## v0.2.0 — pgvector backend
4
12
 
5
13
  - Added a read-only pgvector backend for existing PostgreSQL vector tables.
@@ -2,7 +2,7 @@ cff-version: 1.2.0
2
2
  title: "EmbedFlow: Upgrading Legacy Embeddings Without Full Upfront Re-Embedding"
3
3
  message: "If EmbedFlow contributes to your work, please cite this software release."
4
4
  type: software
5
- version: 0.2.0
5
+ version: 0.3.0
6
6
  date-released: 2026-09-06
7
7
  repository-code: "https://github.com/arnsri33/embedflow"
8
8
  url: "https://github.com/arnsri33/embedflow"
@@ -11,7 +11,7 @@ python -m pip install -e '.[dev]'
11
11
  ```
12
12
 
13
13
  Optional integrations can be installed with `.[faiss]`, `.[qdrant]`,
14
- `.[models]`, or `.[dashboard]`.
14
+ `.[pgvector]`, `.[pinecone]`, `.[models]`, or `.[dashboard]`.
15
15
 
16
16
  ## Checks before opening a pull request
17
17
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: embedflow
3
- Version: 0.2.0
3
+ Version: 0.3.0
4
4
  Summary: Progressive embedding-model migration over existing vector indexes.
5
5
  Author: Arnav Srivastav
6
6
  License-Expression: AGPL-3.0-only
@@ -8,7 +8,7 @@ Project-URL: Homepage, https://embedflow.org
8
8
  Project-URL: Repository, https://github.com/arnsri33/embedflow
9
9
  Project-URL: Documentation, https://github.com/arnsri33/embedflow#readme
10
10
  Project-URL: Issues, https://github.com/arnsri33/embedflow/issues
11
- Keywords: embeddings,vector-search,rag,information-retrieval,faiss,qdrant,pgvector
11
+ Keywords: embeddings,vector-search,rag,information-retrieval,faiss,qdrant,pgvector,pinecone
12
12
  Classifier: Development Status :: 3 - Alpha
13
13
  Classifier: Intended Audience :: Developers
14
14
  Classifier: Intended Audience :: Science/Research
@@ -29,6 +29,8 @@ Provides-Extra: qdrant
29
29
  Requires-Dist: qdrant-client>=1.9; extra == "qdrant"
30
30
  Provides-Extra: pgvector
31
31
  Requires-Dist: psycopg[binary]>=3.2; extra == "pgvector"
32
+ Provides-Extra: pinecone
33
+ Requires-Dist: pinecone>=6.0; extra == "pinecone"
32
34
  Provides-Extra: models
33
35
  Requires-Dist: huggingface-hub<1.0,>=0.34; extra == "models"
34
36
  Requires-Dist: transformers<5.0,>=4.45; extra == "models"
@@ -42,6 +44,7 @@ Provides-Extra: all
42
44
  Requires-Dist: faiss-cpu>=1.8.0; extra == "all"
43
45
  Requires-Dist: qdrant-client>=1.9; extra == "all"
44
46
  Requires-Dist: psycopg[binary]>=3.2; extra == "all"
47
+ Requires-Dist: pinecone>=6.0; extra == "all"
45
48
  Requires-Dist: huggingface-hub<1.0,>=0.34; extra == "all"
46
49
  Requires-Dist: transformers<5.0,>=4.45; extra == "all"
47
50
  Requires-Dist: sentence-transformers>=3.0; extra == "all"
@@ -62,7 +65,7 @@ Dynamic: license-file
62
65
  EmbedFlow lets a new embedding model serve over candidates from an existing
63
66
  vector index while target document vectors are materialized progressively. It
64
67
  supports migration analysis, persistent caching, background work, FAISS,
65
- Qdrant, pgvector, a CLI, and FastAPI.
68
+ Qdrant, pgvector, Pinecone, a CLI, and FastAPI.
66
69
 
67
70
  The full project README and architecture diagram are on
68
71
  <https://github.com/arnsri33/embedflow>.
@@ -85,6 +88,12 @@ For an existing PostgreSQL/pgvector table:
85
88
  python -m pip install "embedflow[pgvector]"
86
89
  ```
87
90
 
91
+ For an existing Pinecone dense index:
92
+
93
+ ```bash
94
+ python -m pip install "embedflow[pinecone]"
95
+ ```
96
+
88
97
  Qdrant and model-runtime extras are documented in the
89
98
  [installation guide](https://github.com/arnsri33/embedflow/blob/main/docs/installation.md).
90
99
  For model-backed analysis, install `embedflow[faiss,models,dashboard]`.
@@ -186,10 +195,12 @@ for definitions and reproduction details.
186
195
  | FAISS | Supported |
187
196
  | Qdrant | Supported |
188
197
  | pgvector | Supported |
198
+ | Pinecone | Supported |
189
199
 
190
200
  See the [FAISS guide](https://github.com/arnsri33/embedflow/blob/main/docs/integrations/faiss.md),
191
201
  [Qdrant guide](https://github.com/arnsri33/embedflow/blob/main/docs/integrations/qdrant.md),
192
- and [pgvector guide](https://github.com/arnsri33/embedflow/blob/main/docs/integrations/pgvector.md).
202
+ and [pgvector guide](https://github.com/arnsri33/embedflow/blob/main/docs/integrations/pgvector.md),
203
+ and [Pinecone guide](https://github.com/arnsri33/embedflow/blob/main/docs/integrations/pinecone.md).
193
204
 
194
205
  ## CLI
195
206
 
@@ -209,7 +220,7 @@ cover the remaining commands and endpoints.
209
220
 
210
221
  ## Status
211
222
 
212
- EmbedFlow v0.2.0 is an alpha release for research and early real-world
223
+ EmbedFlow v0.3.0 is an alpha release for research and early real-world
213
224
  testing. T2-v1 is an empirical finite-tail diagnostic, partial rankings can
214
225
  differ from fully warm target reranking, and ANN fidelity needs a reference
215
226
  comparison to audit.
@@ -7,7 +7,7 @@
7
7
  EmbedFlow lets a new embedding model serve over candidates from an existing
8
8
  vector index while target document vectors are materialized progressively. It
9
9
  supports migration analysis, persistent caching, background work, and serving
10
- through FAISS, Qdrant, pgvector, a CLI, and FastAPI.
10
+ through FAISS, Qdrant, pgvector, Pinecone, a CLI, and FastAPI.
11
11
 
12
12
  [Quickstart](#try-it) · [Documentation](#documentation) · [Research](#research)
13
13
 
@@ -52,6 +52,12 @@ For FAISS and the dashboard, add the optional integrations:
52
52
  python -m pip install "embedflow[faiss,dashboard]"
53
53
  ```
54
54
 
55
+ For an existing Pinecone dense index:
56
+
57
+ ```bash
58
+ python -m pip install "embedflow[pinecone]"
59
+ ```
60
+
55
61
  Qdrant and model-runtime extras are documented in
56
62
  [`docs/installation.md`](https://github.com/arnsri33/embedflow/blob/main/docs/installation.md).
57
63
  For model-backed analysis, install `embedflow[faiss,models,dashboard]`.
@@ -190,12 +196,14 @@ is measured separately and is `UNKNOWN` until an exact reference is supplied.
190
196
  | FAISS | Supported |
191
197
  | Qdrant | Supported |
192
198
  | pgvector | Supported |
199
+ | Pinecone | Supported |
193
200
 
194
201
  Backend-specific setup and examples:
195
202
 
196
203
  - [FAISS](https://github.com/arnsri33/embedflow/blob/main/docs/integrations/faiss.md)
197
204
  - [Qdrant](https://github.com/arnsri33/embedflow/blob/main/docs/integrations/qdrant.md)
198
205
  - [pgvector](https://github.com/arnsri33/embedflow/blob/main/docs/integrations/pgvector.md)
206
+ - [Pinecone](https://github.com/arnsri33/embedflow/blob/main/docs/integrations/pinecone.md)
199
207
  - [Adding a backend](https://github.com/arnsri33/embedflow/blob/main/CONTRIBUTING.md)
200
208
 
201
209
  ## CLI
@@ -230,7 +238,7 @@ OpenAPI documentation; see
230
238
 
231
239
  ## Status
232
240
 
233
- EmbedFlow v0.2.0 is an alpha release for research and early real-world
241
+ EmbedFlow v0.3.0 is an alpha release for research and early real-world
234
242
  testing.
235
243
 
236
244
  - T2-v1 reports an empirical finite-tail diagnostic.
@@ -5,7 +5,7 @@
5
5
  EmbedFlow lets a new embedding model serve over candidates from an existing
6
6
  vector index while target document vectors are materialized progressively. It
7
7
  supports migration analysis, persistent caching, background work, FAISS,
8
- Qdrant, pgvector, a CLI, and FastAPI.
8
+ Qdrant, pgvector, Pinecone, a CLI, and FastAPI.
9
9
 
10
10
  The full project README and architecture diagram are on
11
11
  <https://github.com/arnsri33/embedflow>.
@@ -28,6 +28,12 @@ For an existing PostgreSQL/pgvector table:
28
28
  python -m pip install "embedflow[pgvector]"
29
29
  ```
30
30
 
31
+ For an existing Pinecone dense index:
32
+
33
+ ```bash
34
+ python -m pip install "embedflow[pinecone]"
35
+ ```
36
+
31
37
  Qdrant and model-runtime extras are documented in the
32
38
  [installation guide](https://github.com/arnsri33/embedflow/blob/main/docs/installation.md).
33
39
  For model-backed analysis, install `embedflow[faiss,models,dashboard]`.
@@ -129,10 +135,12 @@ for definitions and reproduction details.
129
135
  | FAISS | Supported |
130
136
  | Qdrant | Supported |
131
137
  | pgvector | Supported |
138
+ | Pinecone | Supported |
132
139
 
133
140
  See the [FAISS guide](https://github.com/arnsri33/embedflow/blob/main/docs/integrations/faiss.md),
134
141
  [Qdrant guide](https://github.com/arnsri33/embedflow/blob/main/docs/integrations/qdrant.md),
135
- and [pgvector guide](https://github.com/arnsri33/embedflow/blob/main/docs/integrations/pgvector.md).
142
+ and [pgvector guide](https://github.com/arnsri33/embedflow/blob/main/docs/integrations/pgvector.md),
143
+ and [Pinecone guide](https://github.com/arnsri33/embedflow/blob/main/docs/integrations/pinecone.md).
136
144
 
137
145
  ## CLI
138
146
 
@@ -152,7 +160,7 @@ cover the remaining commands and endpoints.
152
160
 
153
161
  ## Status
154
162
 
155
- EmbedFlow v0.2.0 is an alpha release for research and early real-world
163
+ EmbedFlow v0.3.0 is an alpha release for research and early real-world
156
164
  testing. T2-v1 is an empirical finite-tail diagnostic, partial rankings can
157
165
  differ from fully warm target reranking, and ANN fidelity needs a reference
158
166
  comparison to audit.
@@ -49,7 +49,9 @@ request. A partial response scores the available target vectors; it can differ
49
49
  from the fully warm ranking.
50
50
 
51
51
  `/status` includes safe backend metadata. For pgvector this names the
52
- schema/table and vector contract without returning the DSN or credentials;
52
+ schema/table and vector contract; for Pinecone it names the host/index,
53
+ namespace, dimension, metric, and safe vector counts. Neither backend returns
54
+ the DSN, API key, or credentials;
53
55
  `/health` remains a compact liveness response for probes and load balancers.
54
56
 
55
57
  ## Errors
@@ -14,7 +14,7 @@ target:
14
14
  device: cuda
15
15
 
16
16
  index:
17
- backend: faiss # faiss, qdrant, or pgvector
17
+ backend: faiss # faiss, qdrant, pgvector, or pinecone
18
18
  path: ./legacy.index
19
19
  ids: ./legacy.index.ids.json # FAISS sidecar
20
20
  metric: cosine
@@ -22,6 +22,8 @@ index:
22
22
  # Qdrant fields: url, collection, vector_name, api_key_env
23
23
  # pgvector fields: dsn_env, schema, table, id_column, vector_column, text_column
24
24
  # hnsw_ef_search, ivfflat_probes
25
+ # Pinecone fields: host (preferred) or index_name, api_key_env, namespace,
26
+ # text_metadata_field
25
27
 
26
28
  documents:
27
29
  path: ./documents.jsonl
@@ -88,6 +90,11 @@ EMBEDFLOW_PGVECTOR_VECTOR_COLUMN
88
90
  EMBEDFLOW_PGVECTOR_TEXT_COLUMN
89
91
  EMBEDFLOW_PGVECTOR_HNSW_EF_SEARCH
90
92
  EMBEDFLOW_PGVECTOR_IVFFLAT_PROBES
93
+ EMBEDFLOW_PINECONE_HOST
94
+ EMBEDFLOW_PINECONE_INDEX_NAME
95
+ EMBEDFLOW_PINECONE_NAMESPACE
96
+ EMBEDFLOW_PINECONE_TEXT_METADATA_FIELD
97
+ EMBEDFLOW_PINECONE_API_KEY_ENV
91
98
  EMBEDFLOW_INDEX_NPROBE
92
99
  EMBEDFLOW_DOCUMENTS_PATH
93
100
  EMBEDFLOW_CACHE_PATH
@@ -31,6 +31,7 @@ python -m pip install -e .
31
31
  | `faiss` | FAISS source-index adapter |
32
32
  | `qdrant` | Qdrant client and adapter |
33
33
  | `pgvector` | Psycopg 3 binary driver and pgvector adapter |
34
+ | `pinecone` | Official Pinecone Python SDK and adapter |
34
35
  | `models` | PyTorch, Transformers, Sentence Transformers, and Hub client |
35
36
  | `dashboard` | FastAPI, Uvicorn, and Pydantic |
36
37
  | `dev` | Pytest, Ruff, and build tooling |
@@ -59,6 +60,12 @@ Install the optional adapter with `python -m pip install "embedflow[pgvector]"`.
59
60
  The adapter connects to an existing table and reads the DSN from the
60
61
  environment; see [`integrations/pgvector.md`](integrations/pgvector.md).
61
62
 
63
+ ## Pinecone
64
+
65
+ Install the optional adapter with `python -m pip install "embedflow[pinecone]"`.
66
+ Set `PINECONE_API_KEY` in the environment and configure an existing dense
67
+ index host; see [`integrations/pinecone.md`](integrations/pinecone.md).
68
+
62
69
  ## CPU and GPU
63
70
 
64
71
  The deterministic demo runs on CPU. Real model serving accepts `--device cpu`
@@ -0,0 +1,157 @@
1
+ # Pinecone
2
+
3
+ EmbedFlow can use an existing Pinecone dense index as its source candidate
4
+ layer. It queries that index, reranks the returned documents with the target
5
+ embedding model, and materializes target vectors into EmbedFlow's local cache.
6
+ The source index is read-only during normal operation.
7
+
8
+ ## Install
9
+
10
+ ```bash
11
+ python -m pip install "embedflow[pinecone]"
12
+ ```
13
+
14
+ The extra installs the official `pinecone` Python SDK. The base package does
15
+ not import or require it.
16
+
17
+ ## Configure an existing index
18
+
19
+ Set the API key in the environment, never in YAML:
20
+
21
+ ```bash
22
+ export PINECONE_API_KEY='...'
23
+ ```
24
+
25
+ Use the data-plane host copied from the Pinecone console:
26
+
27
+ ```yaml
28
+ source:
29
+ model: sentence-transformers/all-MiniLM-L6-v2
30
+ dimension: 384
31
+ device: cuda
32
+
33
+ target:
34
+ model: Qwen/Qwen3-Embedding-0.6B
35
+ device: cuda
36
+
37
+ index:
38
+ backend: pinecone
39
+ api_key_env: PINECONE_API_KEY
40
+ host: my-index-xxxxx.svc.aped-xxxx.pinecone.io
41
+ namespace: production
42
+ metric: cosine
43
+ text_metadata_field: text
44
+
45
+ documents:
46
+ # Omit this file when text_metadata_field is present in Pinecone metadata.
47
+ # If present, it is used as the external text resolver instead.
48
+ path: ./documents.jsonl
49
+ id_field: id
50
+ text_field: text
51
+ ```
52
+
53
+ `host` takes precedence when both `host` and `index_name` are supplied. If a
54
+ host is unavailable, `index_name` can resolve through the control plane; host
55
+ targeting is recommended for production data-plane traffic. The configured
56
+ namespace is passed on every query and fetch. An empty namespace selects
57
+ Pinecone's default namespace.
58
+
59
+ The source index must be a dense vector index whose dimension and metric match
60
+ the source model contract. Pinecone IDs remain strings, including numeric-
61
+ looking IDs such as `"42"`, UUID-looking values, Unicode, and punctuation.
62
+
63
+ ## Document text
64
+
65
+ EmbedFlow needs candidate text for target-model encoding. There are two modes:
66
+
67
+ * Set `text_metadata_field` to read text from Pinecone metadata. Candidate
68
+ queries request metadata but never request vector values. Missing or
69
+ non-string text is reported with the candidate ID and field name.
70
+ * Provide the normal JSONL `documents` store when Pinecone contains IDs and
71
+ vectors only. The external store remains the source of text and avoids
72
+ duplicating a corpus in Pinecone.
73
+
74
+ ## Commands
75
+
76
+ ```bash
77
+ embedflow doctor --config embedflow.yaml
78
+ embedflow audit-index --config embedflow.yaml
79
+ embedflow analyze --config embedflow.yaml
80
+ embedflow serve --config embedflow.yaml
81
+ embedflow search --config embedflow.yaml "what causes auroras?"
82
+ embedflow status --config embedflow.yaml
83
+ ```
84
+
85
+ `audit-index` separates backend health from retrieval fidelity. It checks SDK
86
+ and credentials, reachability, namespace stats, dimension, metric, dense
87
+ vector type, a small candidate query, and text resolution where possible. A
88
+ successful API call is not an ANN recall or T2-v1 compatibility guarantee.
89
+
90
+ ## Read-only behavior
91
+
92
+ `doctor`, `audit-index`, `analyze`, `serve`, `search`, `status`, and `prewarm`
93
+ only query the configured index. EmbedFlow does not create or delete indexes,
94
+ upsert or delete vectors, alter metadata, or change namespaces. Test fixtures
95
+ may create temporary indexes explicitly, but production initialization never
96
+ does.
97
+
98
+ ## Metrics and limits
99
+
100
+ `cosine`, `dot`/`inner_product` (Pinecone `dotproduct`), and
101
+ `l2`/`euclidean` are accepted. Cosine and dot-product scores already rank
102
+ higher-is-better. Pinecone's Euclidean score is squared distance, so EmbedFlow
103
+ negates it to preserve the shared higher-is-better convention. Query `top_k`
104
+ is validated in the range 1–10,000, matching the Pinecone API limit.
105
+
106
+ ## Integrated embedding indexes
107
+
108
+ Pinecone indexes created with hosted/integrated embedding manage the source
109
+ embedding contract outside EmbedFlow. Unless the source model, dimension, and
110
+ metric can be verified, EmbedFlow reports that contract as unknown; it does not
111
+ assume Pinecone's hosted model is the configured source encoder. The first
112
+ adapter release is intended for standard dense-vector indexes.
113
+
114
+ ## Eventual consistency and tests
115
+
116
+ Pinecone is eventually consistent after writes. Any test setup that creates a
117
+ temporary fixture must poll `describe_index_stats` with a bounded timeout before
118
+ querying it. EmbedFlow's production adapter performs no writes and therefore
119
+ does not need a write-read delay.
120
+
121
+ ## Troubleshooting
122
+
123
+ * `Pinecone support requires ...`: install `embedflow[pinecone]` in the active
124
+ environment.
125
+ * `Environment variable PINECONE_API_KEY is not set.`: export the variable
126
+ named by `api_key_env`.
127
+ * `Unable to reach configured Pinecone index`: check the host, project, and
128
+ network policy. Credentials are not printed in this error.
129
+ * Dimension or metric mismatch: compare the source model contract with the
130
+ existing index configuration; EmbedFlow will not rewrite the index.
131
+ * Missing metadata text: set the correct `text_metadata_field` or provide an
132
+ external JSONL document store.
133
+
134
+ ## Local smoke fixture
135
+
136
+ The repository includes a non-destructive smoke helper for an existing index:
137
+
138
+ ```bash
139
+ export PINECONE_API_KEY='...'
140
+ python scripts/pinecone_smoke.py \
141
+ --host "$PINECONE_INDEX_HOST" \
142
+ --namespace production
143
+ ```
144
+
145
+ It only describes stats and, when given a query vector, performs a query. It
146
+ never upserts or deletes records. A remote create/query/delete integration is
147
+ opt-in and requires both `EMBEDFLOW_PINECONE_INTEGRATION_TEST=1` and
148
+ `EMBEDFLOW_PINECONE_CREATE_TEST_INDEX=1`; it uses a unique disposable index and
149
+ deletes it in teardown.
150
+
151
+ ## Limitations
152
+
153
+ The adapter uses one synchronous SDK client. The SDK call itself is safe for
154
+ the modest concurrent serving loads tested by EmbedFlow, but no connection pool
155
+ is introduced. Metadata filters are not exposed because the shared
156
+ `VectorIndex` contract has no portable filter field. Pinecone index provisioning
157
+ and namespace administration remain outside EmbedFlow.
@@ -1,6 +1,6 @@
1
1
  # Limitations and release scope
2
2
 
3
- EmbedFlow v0.2.0 is an alpha release for research and early real-world
3
+ EmbedFlow v0.3.0 is an alpha release for research and early real-world
4
4
  testing. The serving path is designed to make migration experiments concrete;
5
5
  production rollout still requires application-specific validation.
6
6
 
@@ -18,8 +18,8 @@ python -m twine check dist/*
18
18
  Inspect both archives before uploading:
19
19
 
20
20
  ```bash
21
- unzip -l dist/embedflow-0.2.0-py3-none-any.whl
22
- tar -tzf dist/embedflow-0.2.0.tar.gz
21
+ unzip -l dist/embedflow-0.3.0-py3-none-any.whl
22
+ tar -tzf dist/embedflow-0.3.0.tar.gz
23
23
  sha256sum dist/*
24
24
  ```
25
25
 
@@ -32,7 +32,7 @@ Test the wheel outside the source tree:
32
32
  ```bash
33
33
  python -m venv /tmp/embedflow-wheel-test
34
34
  /tmp/embedflow-wheel-test/bin/python -m pip install --upgrade pip
35
- /tmp/embedflow-wheel-test/bin/python -m pip install dist/embedflow-0.2.0-py3-none-any.whl
35
+ /tmp/embedflow-wheel-test/bin/python -m pip install dist/embedflow-0.3.0-py3-none-any.whl
36
36
  cd /tmp
37
37
  /tmp/embedflow-wheel-test/bin/python -c "import embedflow; print(embedflow.__version__)"
38
38
  /tmp/embedflow-wheel-test/bin/embedflow --help
@@ -65,7 +65,7 @@ python -m venv /tmp/embedflow-testpypi
65
65
  /tmp/embedflow-testpypi/bin/python -m pip install \
66
66
  --index-url https://test.pypi.org/simple/ \
67
67
  --extra-index-url https://pypi.org/simple/ \
68
- embedflow==0.2.0
68
+ embedflow==0.3.0
69
69
  cd /tmp
70
70
  /tmp/embedflow-testpypi/bin/python -c "import embedflow; print(embedflow.__version__)"
71
71
  /tmp/embedflow-testpypi/bin/embedflow --help
@@ -79,11 +79,11 @@ Test optional integrations in a second clean environment:
79
79
  /tmp/embedflow-testpypi/bin/python -m pip install \
80
80
  --index-url https://test.pypi.org/simple/ \
81
81
  --extra-index-url https://pypi.org/simple/ \
82
- "embedflow[faiss,dashboard]==0.2.0"
82
+ "embedflow[faiss,dashboard,pinecone]==0.3.0"
83
83
  ```
84
84
 
85
85
  If the same filename already exists on TestPyPI, use a pre-release such as
86
- `0.2.0rc1` for the TestPyPI-only trial. Keep production `0.2.0` unchanged.
86
+ `0.3.0rc1` for the TestPyPI-only trial. Keep production `0.3.0` unchanged.
87
87
 
88
88
  ## Trusted Publishing configuration
89
89
 
@@ -115,7 +115,7 @@ above keeps the test step explicit.
115
115
  3. Run the final release gate and review the generated report.
116
116
  4. Configure the PyPI pending publisher and protected `pypi` environment.
117
117
  5. Create a Git tag and GitHub Release for the exact package version, for
118
- example `v0.2.0`.
118
+ example `v0.3.0`.
119
119
  6. Approve the `pypi` environment when the release workflow is ready.
120
120
  7. Verify the files and metadata on PyPI.
121
121
  8. Install from production PyPI in a directory outside this checkout.
@@ -1,12 +1,12 @@
1
1
  """EmbedFlow: progressive embedding-model migration for existing indexes."""
2
2
 
3
- __version__ = "0.2.0"
3
+ __version__ = "0.3.0"
4
4
 
5
5
  from .config import EmbedFlowConfig, load_config
6
6
 
7
7
 
8
8
  def migrate(*args, **kwargs):
9
- """Start progressive migration over an existing FAISS, Qdrant, or pgvector index.
9
+ """Start progressive migration over an existing FAISS, Qdrant, pgvector, or Pinecone index.
10
10
 
11
11
  Imported lazily to keep the lightweight configuration package free of
12
12
  model-serving dependencies at import time. See ``embedflow.migration``
@@ -0,0 +1,10 @@
1
+ """Public compatibility exports for EmbedFlow's existing FastAPI app.
2
+
3
+ The implementation remains in :mod:`embedflow.serving.api`; this module keeps
4
+ the short ``embedflow.api`` import path lightweight and does not create an app
5
+ or import FastAPI until :func:`create_app` is called.
6
+ """
7
+
8
+ from .serving.api import create_app, dashboard_html
9
+
10
+ __all__ = ["create_app", "dashboard_html"]