embedflow 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. embedflow/__init__.py +25 -0
  2. embedflow/__main__.py +3 -0
  3. embedflow/analysis.py +192 -0
  4. embedflow/cache/__init__.py +4 -0
  5. embedflow/cache/base.py +28 -0
  6. embedflow/cache/persistent_cache.py +198 -0
  7. embedflow/cli.py +1200 -0
  8. embedflow/compatibility/__init__.py +28 -0
  9. embedflow/compatibility/candidate_gap.py +105 -0
  10. embedflow/compatibility/containment.py +17 -0
  11. embedflow/compatibility/evaluate.py +319 -0
  12. embedflow/compatibility/metrics.py +75 -0
  13. embedflow/compatibility/migration_depth.py +67 -0
  14. embedflow/compatibility/probe.py +34 -0
  15. embedflow/compatibility/report.py +102 -0
  16. embedflow/compatibility/t2.py +64 -0
  17. embedflow/config.py +455 -0
  18. embedflow/data/__init__.py +1 -0
  19. embedflow/data/registry/__init__.py +1 -0
  20. embedflow/data/registry/benchmark_profiles.jsonl +3 -0
  21. embedflow/data/registry/checksums.sha256 +4 -0
  22. embedflow/data/registry/migrations.jsonl +15 -0
  23. embedflow/data/registry/registry_manifest.json +16 -0
  24. embedflow/data/registry/research_summaries.json +55 -0
  25. embedflow/data/registry/schema_version.json +5 -0
  26. embedflow/frozen/T2_V1_FROZEN_SPEC.md +71 -0
  27. embedflow/frozen/T2_V1_FROZEN_SPEC.sha256 +1 -0
  28. embedflow/indexes/__init__.py +5 -0
  29. embedflow/indexes/base.py +60 -0
  30. embedflow/indexes/faiss_backend.py +240 -0
  31. embedflow/indexes/qdrant_backend.py +225 -0
  32. embedflow/metrics/__init__.py +3 -0
  33. embedflow/metrics/latency.py +50 -0
  34. embedflow/migration/__init__.py +3 -0
  35. embedflow/migration/compatibility.py +156 -0
  36. embedflow/migration/facade.py +312 -0
  37. embedflow/migration/materializer.py +190 -0
  38. embedflow/migration/planner.py +78 -0
  39. embedflow/migration/state.py +81 -0
  40. embedflow/models/__init__.py +4 -0
  41. embedflow/models/base.py +31 -0
  42. embedflow/models/huggingface.py +226 -0
  43. embedflow/registry/__init__.py +47 -0
  44. embedflow/registry/loader.py +785 -0
  45. embedflow/registry/matcher.py +197 -0
  46. embedflow/registry/schema.py +266 -0
  47. embedflow/runtime.py +115 -0
  48. embedflow/serving/__init__.py +3 -0
  49. embedflow/serving/api.py +161 -0
  50. embedflow/serving/engine.py +222 -0
  51. embedflow/serving/factory.py +3 -0
  52. embedflow/serving/schemas.py +39 -0
  53. embedflow-0.1.0.dist-info/METADATA +210 -0
  54. embedflow-0.1.0.dist-info/RECORD +64 -0
  55. embedflow-0.1.0.dist-info/WHEEL +5 -0
  56. embedflow-0.1.0.dist-info/entry_points.txt +2 -0
  57. embedflow-0.1.0.dist-info/licenses/LICENSE +178 -0
  58. embedflow-0.1.0.dist-info/top_level.txt +2 -0
  59. src/__init__.py +1 -0
  60. src/embed.py +123 -0
  61. src/probe_features.py +24 -0
  62. src/storage.py +51 -0
  63. src/t2_v1.py +21 -0
  64. src/utils.py +53 -0
@@ -0,0 +1,31 @@
1
+ from __future__ import annotations
2
+
3
+ from abc import ABC, abstractmethod
4
+ from typing import Any
5
+
6
+
7
+ class EmbeddingModel(ABC):
8
+ model_id: str
9
+ dimension: int
10
+ fingerprint: str
11
+
12
+ @abstractmethod
13
+ def encode_queries(self, texts: list[str]):
14
+ raise NotImplementedError
15
+
16
+ @abstractmethod
17
+ def encode_documents(self, texts: list[str], batch_size: int | None = None):
18
+ raise NotImplementedError
19
+
20
+ def encode_query(self, text: str):
21
+ return self.encode_queries([text])[0]
22
+
23
+ def encode_document(self, text: str):
24
+ return self.encode_documents([text])[0]
25
+
26
+ @abstractmethod
27
+ def close(self) -> None:
28
+ raise NotImplementedError
29
+
30
+ def metadata(self) -> dict[str, Any]:
31
+ return {"model_id": self.model_id, "dimension": self.dimension, "fingerprint": self.fingerprint}
@@ -0,0 +1,226 @@
1
+ from __future__ import annotations
2
+
3
+ import gc
4
+ import hashlib
5
+ import json
6
+ import re
7
+ from pathlib import Path
8
+
9
+ import numpy as np
10
+
11
+ from ..config import ModelConfig, hydrate_research_contract
12
+ from .base import EmbeddingModel
13
+
14
+
15
+ def _fingerprint(cfg: ModelConfig) -> str:
16
+ semantic = {key: value for key, value in cfg.contract().items() if key not in {"local_path", "device"}}
17
+ return hashlib.sha256(json.dumps(semantic, sort_keys=True, separators=(",", ":")).encode()).hexdigest()
18
+
19
+
20
+ class HashEmbeddingModel(EmbeddingModel):
21
+ """Deterministic tiny model used only by the self-contained demo/tests."""
22
+
23
+ def __init__(self, model_id: str = "embedflow/demo-hash", dimension: int = 64):
24
+ try:
25
+ parsed_dimension = int(dimension)
26
+ exact = float(dimension) == parsed_dimension
27
+ except (TypeError, ValueError, OverflowError) as exc:
28
+ raise ValueError("embedding dimension must be a positive integer") from exc
29
+ if isinstance(dimension, bool) or not exact or parsed_dimension < 1:
30
+ raise ValueError("embedding dimension must be a positive integer")
31
+ self.model_id, self.dimension = str(model_id), parsed_dimension
32
+ # Use the same semantic contract hashing as real adapters so demo
33
+ # indexes, state files, and target caches agree on the model identity.
34
+ self.fingerprint = ModelConfig(model_id, dimension=self.dimension).fingerprint
35
+
36
+ def _encode(self, texts: list[str], role: str) -> np.ndarray:
37
+ if not texts:
38
+ return np.empty((0, self.dimension), dtype="float32")
39
+ rows = []
40
+ for text in texts:
41
+ # Demo vectors intentionally share a token-derived space across
42
+ # query/document roles and model IDs, while fingerprints remain
43
+ # distinct for cache safety. This makes the tiny demo useful for
44
+ # retrieval without pretending to be a learned model.
45
+ tokens = re.findall(r"[a-z0-9]+", str(text).lower()) or [str(text).lower()]
46
+ pieces = []
47
+ for token in tokens:
48
+ seed = int.from_bytes(hashlib.sha256(token.encode()).digest()[:8], "little")
49
+ pieces.append(np.random.default_rng(seed).normal(size=self.dimension).astype("float32"))
50
+ v = np.mean(pieces, axis=0).astype("float32")
51
+ v /= max(float(np.linalg.norm(v)), 1e-12)
52
+ rows.append(v)
53
+ return np.asarray(rows, dtype="float32")
54
+
55
+ def encode_queries(self, texts: list[str]) -> np.ndarray:
56
+ return self._encode(texts, "query")
57
+
58
+ def encode_documents(self, texts: list[str], batch_size: int | None = None) -> np.ndarray:
59
+ return self._encode(texts, "document")
60
+
61
+ def close(self) -> None:
62
+ return None
63
+
64
+
65
+ class HuggingFaceEmbeddingModel(EmbeddingModel):
66
+ """HF/sentence-transformers adapter honoring the research contracts."""
67
+
68
+ def __init__(self, cfg: ModelConfig, model_root: str | Path | None = None, device: str = "cpu"):
69
+ self.cfg = hydrate_research_contract(cfg)
70
+ self.model_id = self.cfg.model
71
+ self.dimension = int(self.cfg.dimension or 0)
72
+ self.fingerprint = _fingerprint(self.cfg)
73
+ self.device = "cuda" if str(device).strip().lower() == "gpu" else str(device)
74
+ model_path = Path(self.cfg.local_path or self.model_id)
75
+ if model_root and not model_path.is_absolute():
76
+ # Known research checkpoints are staged under models/<registry key>.
77
+ registry = {
78
+ "sentence-transformers/all-MiniLM-L6-v2": "minilm_l6",
79
+ "Qwen/Qwen3-Embedding-0.6B": "qwen3_0_6b",
80
+ "Qwen/Qwen3-Embedding-4B": "qwen3_4b",
81
+ "Qwen/Qwen3-Embedding-8B": "qwen3_8b",
82
+ }
83
+ staged = Path(model_root) / registry.get(str(model_path), str(model_path))
84
+ if staged.exists():
85
+ model_path = staged
86
+ self.model_path = model_path
87
+ self._encoder = None
88
+ self._closed = False
89
+ self._kind = "sentence" if self.model_id.startswith("sentence-transformers/") else "transformers"
90
+ self._load()
91
+ self.cfg.dimension = self.dimension
92
+ self.fingerprint = _fingerprint(self.cfg)
93
+
94
+ def _load(self) -> None:
95
+ # Reuse the project's frozen implementation where available. It has
96
+ # the exact Qwen prompt, padding and last-token pooling behavior.
97
+ try:
98
+ known_contract = self.model_id in {
99
+ "sentence-transformers/all-MiniLM-L6-v2", "Qwen/Qwen3-Embedding-0.6B",
100
+ "Qwen/Qwen3-Embedding-4B", "Qwen/Qwen3-Embedding-8B"
101
+ }
102
+ if not known_contract or not self.model_path.exists():
103
+ raise ImportError("use generic Hugging Face loader")
104
+ from src.embed import make_encoder # type: ignore
105
+ contract = self.cfg.contract()
106
+ contract["model_id"] = self.model_id
107
+ contract["revision"] = self.cfg.revision
108
+ contract["dimension"] = self.dimension
109
+ self._encoder = make_encoder(contract, self.model_path, smoke=False, seed=0, device=self.device)
110
+ if not self.dimension and hasattr(self._encoder, "dimension"):
111
+ self.dimension = int(self._encoder.dimension)
112
+ if not self.dimension and hasattr(self._encoder, "model") and hasattr(self._encoder.model, "get_sentence_embedding_dimension"):
113
+ self.dimension = int(self._encoder.model.get_sentence_embedding_dimension())
114
+ if not self.dimension and hasattr(self._encoder, "net"):
115
+ self.dimension = int(getattr(getattr(self._encoder.net, "config", None), "hidden_size", 0) or 0)
116
+ if not self.dimension:
117
+ raise ValueError(f"could not infer embedding dimension for {self.model_id}; set source/target.dimension")
118
+ return
119
+ except (ImportError, ModuleNotFoundError):
120
+ pass
121
+ if self._kind == "sentence":
122
+ from sentence_transformers import SentenceTransformer
123
+ local_only = self.model_path.exists()
124
+ self._encoder = SentenceTransformer(str(self.model_path), device=self.device,
125
+ revision=self.cfg.revision, local_files_only=local_only)
126
+ self._encoder.max_seq_length = self.cfg.max_length
127
+ if not self.dimension:
128
+ self.dimension = int(self._encoder.get_sentence_embedding_dimension())
129
+ else:
130
+ import torch
131
+ from transformers import AutoModel, AutoTokenizer
132
+ local_only = self.model_path.exists()
133
+ self._tok = AutoTokenizer.from_pretrained(str(self.model_path), revision=self.cfg.revision,
134
+ local_files_only=local_only)
135
+ self._tok.padding_side = self.cfg.padding_side
136
+ self._tok.truncation_side = self.cfg.truncation_side
137
+ self._tok.pad_token = self._tok.pad_token or self._tok.eos_token
138
+ dtype = torch.bfloat16 if self.device.startswith("cuda") and torch.cuda.is_bf16_supported() else torch.float32
139
+ self._encoder = AutoModel.from_pretrained(str(self.model_path), revision=self.cfg.revision,
140
+ local_files_only=local_only, torch_dtype=dtype).eval().to(self.device)
141
+ if not self.dimension:
142
+ self.dimension = int(getattr(self._encoder.config, "hidden_size", 0) or getattr(self._encoder.config, "d_model", 0))
143
+ if not self.dimension:
144
+ raise ValueError(f"could not infer embedding dimension for {self.model_id}; set source/target.dimension")
145
+
146
+ def _format(self, texts: list[str], role: str) -> list[str]:
147
+ instruction = self.cfg.query_instruction if role == "query" else self.cfg.document_instruction
148
+ return [instruction.replace("{text}", str(t)) if instruction else str(t) for t in texts]
149
+
150
+ def _generic_encode(self, texts: list[str], role: str) -> np.ndarray:
151
+ if self._kind == "sentence":
152
+ vals = self._encoder.encode(self._format(texts, role), convert_to_numpy=True,
153
+ normalize_embeddings=self.cfg.normalization.lower() == "l2",
154
+ show_progress_bar=False, batch_size=min(32, max(1, len(texts))))
155
+ return np.asarray(vals, dtype="float32")
156
+ import torch
157
+ formatted = self._format(texts, role)
158
+ enc = self._tok(formatted, padding=True, truncation=True, max_length=self.cfg.max_length, return_tensors="pt").to(self.device)
159
+ with torch.inference_mode():
160
+ out = self._encoder(**enc).last_hidden_state
161
+ if self.cfg.pooling in {"last_token", "last_non_padding_token"}:
162
+ # For left padding the final sequence position is the last
163
+ # non-padding token; for right padding use the attention-mask
164
+ # length. This preserves the Qwen contract in both modes.
165
+ if self.cfg.padding_side == "left":
166
+ positions = torch.full((out.shape[0],), out.shape[1] - 1, dtype=torch.long, device=out.device)
167
+ else:
168
+ positions = enc["attention_mask"].sum(dim=1) - 1
169
+ vals = out[torch.arange(out.shape[0], device=out.device), positions]
170
+ else:
171
+ mask = enc["attention_mask"].unsqueeze(-1)
172
+ vals = (out * mask).sum(1) / mask.sum(1).clamp_min(1)
173
+ vals = vals.float()
174
+ if self.cfg.normalization.lower() == "l2":
175
+ vals = torch.nn.functional.normalize(vals, dim=1)
176
+ return vals.detach().cpu().numpy().astype("float32")
177
+
178
+ def _encode(self, texts: list[str], role: str, batch_size: int | None = None) -> np.ndarray:
179
+ if self._closed:
180
+ raise RuntimeError("embedding model is closed")
181
+ if not texts:
182
+ return np.empty((0, self.dimension), dtype="float32")
183
+ if self._encoder is not None and hasattr(self._encoder, "encode") and not hasattr(self, "_tok"):
184
+ # ``src.embed`` applies the frozen query/document instruction
185
+ # internally; formatting here would prepend it twice.
186
+ values, _ = self._encoder.encode(texts, role)
187
+ return np.asarray(values, dtype="float32")
188
+ return self._generic_encode(texts, role)
189
+
190
+ def encode_queries(self, texts: list[str]) -> np.ndarray:
191
+ return self._encode(texts, "query")
192
+
193
+ def encode_documents(self, texts: list[str], batch_size: int | None = None) -> np.ndarray:
194
+ if batch_size is None or len(texts) <= batch_size:
195
+ return self._encode(texts, "document", batch_size)
196
+ chunks = [self._encode(texts[i:i + batch_size], "document", batch_size) for i in range(0, len(texts), batch_size)]
197
+ return np.concatenate(chunks, axis=0)
198
+
199
+ def close(self) -> None:
200
+ if self._closed:
201
+ return
202
+ self._closed = True
203
+ try:
204
+ if hasattr(self, "_encoder"):
205
+ del self._encoder
206
+ if hasattr(self, "_tok"):
207
+ del self._tok
208
+ finally:
209
+ gc.collect()
210
+ try:
211
+ import torch
212
+ if torch.cuda.is_available(): torch.cuda.empty_cache()
213
+ except Exception:
214
+ pass
215
+
216
+
217
+ def load_embedding_model(cfg: ModelConfig, model_root: str | Path | None = None, device: str = "cpu",
218
+ demo: bool = False) -> EmbeddingModel:
219
+ if not isinstance(cfg, ModelConfig):
220
+ raise TypeError("cfg must be a ModelConfig")
221
+ device = str(device or cfg.device or "cpu")
222
+ if demo or cfg.model.startswith("embedflow/demo"):
223
+ return HashEmbeddingModel(cfg.model or "embedflow/demo-hash", cfg.dimension or 64)
224
+ if not cfg.model:
225
+ raise ValueError("model identifier is required")
226
+ return HuggingFaceEmbeddingModel(cfg, model_root=model_root, device=device)
@@ -0,0 +1,47 @@
1
+ """Versioned, provenance-aware migration evidence shipped with EmbedFlow.
2
+
3
+ The registry contains only retained research results. It is intentionally
4
+ separate from the runtime cache and never contains vectors or indexes.
5
+ """
6
+
7
+ from .loader import (
8
+ dataset_fingerprint,
9
+ find_evidence,
10
+ load_benchmark_profiles,
11
+ load_evidence,
12
+ load_manifest,
13
+ load_summaries,
14
+ verify_registry,
15
+ )
16
+ from .matcher import RegistryMatch, match_config, match_evidence
17
+ from .schema import (
18
+ MATCH_EXACT,
19
+ MATCH_NONE,
20
+ MATCH_PRIOR,
21
+ MATCH_RELATED,
22
+ BenchmarkProfile,
23
+ EvidenceRecord,
24
+ RegistryError,
25
+ contract_fingerprint,
26
+ )
27
+
28
+ __all__ = [
29
+ "BenchmarkProfile",
30
+ "EvidenceRecord",
31
+ "MATCH_EXACT",
32
+ "MATCH_NONE",
33
+ "MATCH_PRIOR",
34
+ "MATCH_RELATED",
35
+ "RegistryError",
36
+ "contract_fingerprint",
37
+ "RegistryMatch",
38
+ "dataset_fingerprint",
39
+ "find_evidence",
40
+ "load_benchmark_profiles",
41
+ "load_evidence",
42
+ "load_manifest",
43
+ "load_summaries",
44
+ "match_config",
45
+ "match_evidence",
46
+ "verify_registry",
47
+ ]