embedflow 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- embedflow/__init__.py +25 -0
- embedflow/__main__.py +3 -0
- embedflow/analysis.py +192 -0
- embedflow/cache/__init__.py +4 -0
- embedflow/cache/base.py +28 -0
- embedflow/cache/persistent_cache.py +198 -0
- embedflow/cli.py +1200 -0
- embedflow/compatibility/__init__.py +28 -0
- embedflow/compatibility/candidate_gap.py +105 -0
- embedflow/compatibility/containment.py +17 -0
- embedflow/compatibility/evaluate.py +319 -0
- embedflow/compatibility/metrics.py +75 -0
- embedflow/compatibility/migration_depth.py +67 -0
- embedflow/compatibility/probe.py +34 -0
- embedflow/compatibility/report.py +102 -0
- embedflow/compatibility/t2.py +64 -0
- embedflow/config.py +455 -0
- embedflow/data/__init__.py +1 -0
- embedflow/data/registry/__init__.py +1 -0
- embedflow/data/registry/benchmark_profiles.jsonl +3 -0
- embedflow/data/registry/checksums.sha256 +4 -0
- embedflow/data/registry/migrations.jsonl +15 -0
- embedflow/data/registry/registry_manifest.json +16 -0
- embedflow/data/registry/research_summaries.json +55 -0
- embedflow/data/registry/schema_version.json +5 -0
- embedflow/frozen/T2_V1_FROZEN_SPEC.md +71 -0
- embedflow/frozen/T2_V1_FROZEN_SPEC.sha256 +1 -0
- embedflow/indexes/__init__.py +5 -0
- embedflow/indexes/base.py +60 -0
- embedflow/indexes/faiss_backend.py +240 -0
- embedflow/indexes/qdrant_backend.py +225 -0
- embedflow/metrics/__init__.py +3 -0
- embedflow/metrics/latency.py +50 -0
- embedflow/migration/__init__.py +3 -0
- embedflow/migration/compatibility.py +156 -0
- embedflow/migration/facade.py +312 -0
- embedflow/migration/materializer.py +190 -0
- embedflow/migration/planner.py +78 -0
- embedflow/migration/state.py +81 -0
- embedflow/models/__init__.py +4 -0
- embedflow/models/base.py +31 -0
- embedflow/models/huggingface.py +226 -0
- embedflow/registry/__init__.py +47 -0
- embedflow/registry/loader.py +785 -0
- embedflow/registry/matcher.py +197 -0
- embedflow/registry/schema.py +266 -0
- embedflow/runtime.py +115 -0
- embedflow/serving/__init__.py +3 -0
- embedflow/serving/api.py +161 -0
- embedflow/serving/engine.py +222 -0
- embedflow/serving/factory.py +3 -0
- embedflow/serving/schemas.py +39 -0
- embedflow-0.1.0.dist-info/METADATA +210 -0
- embedflow-0.1.0.dist-info/RECORD +64 -0
- embedflow-0.1.0.dist-info/WHEEL +5 -0
- embedflow-0.1.0.dist-info/entry_points.txt +2 -0
- embedflow-0.1.0.dist-info/licenses/LICENSE +178 -0
- embedflow-0.1.0.dist-info/top_level.txt +2 -0
- src/__init__.py +1 -0
- src/embed.py +123 -0
- src/probe_features.py +24 -0
- src/storage.py +51 -0
- src/t2_v1.py +21 -0
- src/utils.py +53 -0
embedflow/models/base.py
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from abc import ABC, abstractmethod
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class EmbeddingModel(ABC):
|
|
8
|
+
model_id: str
|
|
9
|
+
dimension: int
|
|
10
|
+
fingerprint: str
|
|
11
|
+
|
|
12
|
+
@abstractmethod
|
|
13
|
+
def encode_queries(self, texts: list[str]):
|
|
14
|
+
raise NotImplementedError
|
|
15
|
+
|
|
16
|
+
@abstractmethod
|
|
17
|
+
def encode_documents(self, texts: list[str], batch_size: int | None = None):
|
|
18
|
+
raise NotImplementedError
|
|
19
|
+
|
|
20
|
+
def encode_query(self, text: str):
|
|
21
|
+
return self.encode_queries([text])[0]
|
|
22
|
+
|
|
23
|
+
def encode_document(self, text: str):
|
|
24
|
+
return self.encode_documents([text])[0]
|
|
25
|
+
|
|
26
|
+
@abstractmethod
|
|
27
|
+
def close(self) -> None:
|
|
28
|
+
raise NotImplementedError
|
|
29
|
+
|
|
30
|
+
def metadata(self) -> dict[str, Any]:
|
|
31
|
+
return {"model_id": self.model_id, "dimension": self.dimension, "fingerprint": self.fingerprint}
|
|
@@ -0,0 +1,226 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import gc
|
|
4
|
+
import hashlib
|
|
5
|
+
import json
|
|
6
|
+
import re
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
import numpy as np
|
|
10
|
+
|
|
11
|
+
from ..config import ModelConfig, hydrate_research_contract
|
|
12
|
+
from .base import EmbeddingModel
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _fingerprint(cfg: ModelConfig) -> str:
|
|
16
|
+
semantic = {key: value for key, value in cfg.contract().items() if key not in {"local_path", "device"}}
|
|
17
|
+
return hashlib.sha256(json.dumps(semantic, sort_keys=True, separators=(",", ":")).encode()).hexdigest()
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class HashEmbeddingModel(EmbeddingModel):
|
|
21
|
+
"""Deterministic tiny model used only by the self-contained demo/tests."""
|
|
22
|
+
|
|
23
|
+
def __init__(self, model_id: str = "embedflow/demo-hash", dimension: int = 64):
|
|
24
|
+
try:
|
|
25
|
+
parsed_dimension = int(dimension)
|
|
26
|
+
exact = float(dimension) == parsed_dimension
|
|
27
|
+
except (TypeError, ValueError, OverflowError) as exc:
|
|
28
|
+
raise ValueError("embedding dimension must be a positive integer") from exc
|
|
29
|
+
if isinstance(dimension, bool) or not exact or parsed_dimension < 1:
|
|
30
|
+
raise ValueError("embedding dimension must be a positive integer")
|
|
31
|
+
self.model_id, self.dimension = str(model_id), parsed_dimension
|
|
32
|
+
# Use the same semantic contract hashing as real adapters so demo
|
|
33
|
+
# indexes, state files, and target caches agree on the model identity.
|
|
34
|
+
self.fingerprint = ModelConfig(model_id, dimension=self.dimension).fingerprint
|
|
35
|
+
|
|
36
|
+
def _encode(self, texts: list[str], role: str) -> np.ndarray:
|
|
37
|
+
if not texts:
|
|
38
|
+
return np.empty((0, self.dimension), dtype="float32")
|
|
39
|
+
rows = []
|
|
40
|
+
for text in texts:
|
|
41
|
+
# Demo vectors intentionally share a token-derived space across
|
|
42
|
+
# query/document roles and model IDs, while fingerprints remain
|
|
43
|
+
# distinct for cache safety. This makes the tiny demo useful for
|
|
44
|
+
# retrieval without pretending to be a learned model.
|
|
45
|
+
tokens = re.findall(r"[a-z0-9]+", str(text).lower()) or [str(text).lower()]
|
|
46
|
+
pieces = []
|
|
47
|
+
for token in tokens:
|
|
48
|
+
seed = int.from_bytes(hashlib.sha256(token.encode()).digest()[:8], "little")
|
|
49
|
+
pieces.append(np.random.default_rng(seed).normal(size=self.dimension).astype("float32"))
|
|
50
|
+
v = np.mean(pieces, axis=0).astype("float32")
|
|
51
|
+
v /= max(float(np.linalg.norm(v)), 1e-12)
|
|
52
|
+
rows.append(v)
|
|
53
|
+
return np.asarray(rows, dtype="float32")
|
|
54
|
+
|
|
55
|
+
def encode_queries(self, texts: list[str]) -> np.ndarray:
|
|
56
|
+
return self._encode(texts, "query")
|
|
57
|
+
|
|
58
|
+
def encode_documents(self, texts: list[str], batch_size: int | None = None) -> np.ndarray:
|
|
59
|
+
return self._encode(texts, "document")
|
|
60
|
+
|
|
61
|
+
def close(self) -> None:
|
|
62
|
+
return None
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class HuggingFaceEmbeddingModel(EmbeddingModel):
|
|
66
|
+
"""HF/sentence-transformers adapter honoring the research contracts."""
|
|
67
|
+
|
|
68
|
+
def __init__(self, cfg: ModelConfig, model_root: str | Path | None = None, device: str = "cpu"):
|
|
69
|
+
self.cfg = hydrate_research_contract(cfg)
|
|
70
|
+
self.model_id = self.cfg.model
|
|
71
|
+
self.dimension = int(self.cfg.dimension or 0)
|
|
72
|
+
self.fingerprint = _fingerprint(self.cfg)
|
|
73
|
+
self.device = "cuda" if str(device).strip().lower() == "gpu" else str(device)
|
|
74
|
+
model_path = Path(self.cfg.local_path or self.model_id)
|
|
75
|
+
if model_root and not model_path.is_absolute():
|
|
76
|
+
# Known research checkpoints are staged under models/<registry key>.
|
|
77
|
+
registry = {
|
|
78
|
+
"sentence-transformers/all-MiniLM-L6-v2": "minilm_l6",
|
|
79
|
+
"Qwen/Qwen3-Embedding-0.6B": "qwen3_0_6b",
|
|
80
|
+
"Qwen/Qwen3-Embedding-4B": "qwen3_4b",
|
|
81
|
+
"Qwen/Qwen3-Embedding-8B": "qwen3_8b",
|
|
82
|
+
}
|
|
83
|
+
staged = Path(model_root) / registry.get(str(model_path), str(model_path))
|
|
84
|
+
if staged.exists():
|
|
85
|
+
model_path = staged
|
|
86
|
+
self.model_path = model_path
|
|
87
|
+
self._encoder = None
|
|
88
|
+
self._closed = False
|
|
89
|
+
self._kind = "sentence" if self.model_id.startswith("sentence-transformers/") else "transformers"
|
|
90
|
+
self._load()
|
|
91
|
+
self.cfg.dimension = self.dimension
|
|
92
|
+
self.fingerprint = _fingerprint(self.cfg)
|
|
93
|
+
|
|
94
|
+
def _load(self) -> None:
|
|
95
|
+
# Reuse the project's frozen implementation where available. It has
|
|
96
|
+
# the exact Qwen prompt, padding and last-token pooling behavior.
|
|
97
|
+
try:
|
|
98
|
+
known_contract = self.model_id in {
|
|
99
|
+
"sentence-transformers/all-MiniLM-L6-v2", "Qwen/Qwen3-Embedding-0.6B",
|
|
100
|
+
"Qwen/Qwen3-Embedding-4B", "Qwen/Qwen3-Embedding-8B"
|
|
101
|
+
}
|
|
102
|
+
if not known_contract or not self.model_path.exists():
|
|
103
|
+
raise ImportError("use generic Hugging Face loader")
|
|
104
|
+
from src.embed import make_encoder # type: ignore
|
|
105
|
+
contract = self.cfg.contract()
|
|
106
|
+
contract["model_id"] = self.model_id
|
|
107
|
+
contract["revision"] = self.cfg.revision
|
|
108
|
+
contract["dimension"] = self.dimension
|
|
109
|
+
self._encoder = make_encoder(contract, self.model_path, smoke=False, seed=0, device=self.device)
|
|
110
|
+
if not self.dimension and hasattr(self._encoder, "dimension"):
|
|
111
|
+
self.dimension = int(self._encoder.dimension)
|
|
112
|
+
if not self.dimension and hasattr(self._encoder, "model") and hasattr(self._encoder.model, "get_sentence_embedding_dimension"):
|
|
113
|
+
self.dimension = int(self._encoder.model.get_sentence_embedding_dimension())
|
|
114
|
+
if not self.dimension and hasattr(self._encoder, "net"):
|
|
115
|
+
self.dimension = int(getattr(getattr(self._encoder.net, "config", None), "hidden_size", 0) or 0)
|
|
116
|
+
if not self.dimension:
|
|
117
|
+
raise ValueError(f"could not infer embedding dimension for {self.model_id}; set source/target.dimension")
|
|
118
|
+
return
|
|
119
|
+
except (ImportError, ModuleNotFoundError):
|
|
120
|
+
pass
|
|
121
|
+
if self._kind == "sentence":
|
|
122
|
+
from sentence_transformers import SentenceTransformer
|
|
123
|
+
local_only = self.model_path.exists()
|
|
124
|
+
self._encoder = SentenceTransformer(str(self.model_path), device=self.device,
|
|
125
|
+
revision=self.cfg.revision, local_files_only=local_only)
|
|
126
|
+
self._encoder.max_seq_length = self.cfg.max_length
|
|
127
|
+
if not self.dimension:
|
|
128
|
+
self.dimension = int(self._encoder.get_sentence_embedding_dimension())
|
|
129
|
+
else:
|
|
130
|
+
import torch
|
|
131
|
+
from transformers import AutoModel, AutoTokenizer
|
|
132
|
+
local_only = self.model_path.exists()
|
|
133
|
+
self._tok = AutoTokenizer.from_pretrained(str(self.model_path), revision=self.cfg.revision,
|
|
134
|
+
local_files_only=local_only)
|
|
135
|
+
self._tok.padding_side = self.cfg.padding_side
|
|
136
|
+
self._tok.truncation_side = self.cfg.truncation_side
|
|
137
|
+
self._tok.pad_token = self._tok.pad_token or self._tok.eos_token
|
|
138
|
+
dtype = torch.bfloat16 if self.device.startswith("cuda") and torch.cuda.is_bf16_supported() else torch.float32
|
|
139
|
+
self._encoder = AutoModel.from_pretrained(str(self.model_path), revision=self.cfg.revision,
|
|
140
|
+
local_files_only=local_only, torch_dtype=dtype).eval().to(self.device)
|
|
141
|
+
if not self.dimension:
|
|
142
|
+
self.dimension = int(getattr(self._encoder.config, "hidden_size", 0) or getattr(self._encoder.config, "d_model", 0))
|
|
143
|
+
if not self.dimension:
|
|
144
|
+
raise ValueError(f"could not infer embedding dimension for {self.model_id}; set source/target.dimension")
|
|
145
|
+
|
|
146
|
+
def _format(self, texts: list[str], role: str) -> list[str]:
|
|
147
|
+
instruction = self.cfg.query_instruction if role == "query" else self.cfg.document_instruction
|
|
148
|
+
return [instruction.replace("{text}", str(t)) if instruction else str(t) for t in texts]
|
|
149
|
+
|
|
150
|
+
def _generic_encode(self, texts: list[str], role: str) -> np.ndarray:
|
|
151
|
+
if self._kind == "sentence":
|
|
152
|
+
vals = self._encoder.encode(self._format(texts, role), convert_to_numpy=True,
|
|
153
|
+
normalize_embeddings=self.cfg.normalization.lower() == "l2",
|
|
154
|
+
show_progress_bar=False, batch_size=min(32, max(1, len(texts))))
|
|
155
|
+
return np.asarray(vals, dtype="float32")
|
|
156
|
+
import torch
|
|
157
|
+
formatted = self._format(texts, role)
|
|
158
|
+
enc = self._tok(formatted, padding=True, truncation=True, max_length=self.cfg.max_length, return_tensors="pt").to(self.device)
|
|
159
|
+
with torch.inference_mode():
|
|
160
|
+
out = self._encoder(**enc).last_hidden_state
|
|
161
|
+
if self.cfg.pooling in {"last_token", "last_non_padding_token"}:
|
|
162
|
+
# For left padding the final sequence position is the last
|
|
163
|
+
# non-padding token; for right padding use the attention-mask
|
|
164
|
+
# length. This preserves the Qwen contract in both modes.
|
|
165
|
+
if self.cfg.padding_side == "left":
|
|
166
|
+
positions = torch.full((out.shape[0],), out.shape[1] - 1, dtype=torch.long, device=out.device)
|
|
167
|
+
else:
|
|
168
|
+
positions = enc["attention_mask"].sum(dim=1) - 1
|
|
169
|
+
vals = out[torch.arange(out.shape[0], device=out.device), positions]
|
|
170
|
+
else:
|
|
171
|
+
mask = enc["attention_mask"].unsqueeze(-1)
|
|
172
|
+
vals = (out * mask).sum(1) / mask.sum(1).clamp_min(1)
|
|
173
|
+
vals = vals.float()
|
|
174
|
+
if self.cfg.normalization.lower() == "l2":
|
|
175
|
+
vals = torch.nn.functional.normalize(vals, dim=1)
|
|
176
|
+
return vals.detach().cpu().numpy().astype("float32")
|
|
177
|
+
|
|
178
|
+
def _encode(self, texts: list[str], role: str, batch_size: int | None = None) -> np.ndarray:
|
|
179
|
+
if self._closed:
|
|
180
|
+
raise RuntimeError("embedding model is closed")
|
|
181
|
+
if not texts:
|
|
182
|
+
return np.empty((0, self.dimension), dtype="float32")
|
|
183
|
+
if self._encoder is not None and hasattr(self._encoder, "encode") and not hasattr(self, "_tok"):
|
|
184
|
+
# ``src.embed`` applies the frozen query/document instruction
|
|
185
|
+
# internally; formatting here would prepend it twice.
|
|
186
|
+
values, _ = self._encoder.encode(texts, role)
|
|
187
|
+
return np.asarray(values, dtype="float32")
|
|
188
|
+
return self._generic_encode(texts, role)
|
|
189
|
+
|
|
190
|
+
def encode_queries(self, texts: list[str]) -> np.ndarray:
|
|
191
|
+
return self._encode(texts, "query")
|
|
192
|
+
|
|
193
|
+
def encode_documents(self, texts: list[str], batch_size: int | None = None) -> np.ndarray:
|
|
194
|
+
if batch_size is None or len(texts) <= batch_size:
|
|
195
|
+
return self._encode(texts, "document", batch_size)
|
|
196
|
+
chunks = [self._encode(texts[i:i + batch_size], "document", batch_size) for i in range(0, len(texts), batch_size)]
|
|
197
|
+
return np.concatenate(chunks, axis=0)
|
|
198
|
+
|
|
199
|
+
def close(self) -> None:
|
|
200
|
+
if self._closed:
|
|
201
|
+
return
|
|
202
|
+
self._closed = True
|
|
203
|
+
try:
|
|
204
|
+
if hasattr(self, "_encoder"):
|
|
205
|
+
del self._encoder
|
|
206
|
+
if hasattr(self, "_tok"):
|
|
207
|
+
del self._tok
|
|
208
|
+
finally:
|
|
209
|
+
gc.collect()
|
|
210
|
+
try:
|
|
211
|
+
import torch
|
|
212
|
+
if torch.cuda.is_available(): torch.cuda.empty_cache()
|
|
213
|
+
except Exception:
|
|
214
|
+
pass
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def load_embedding_model(cfg: ModelConfig, model_root: str | Path | None = None, device: str = "cpu",
|
|
218
|
+
demo: bool = False) -> EmbeddingModel:
|
|
219
|
+
if not isinstance(cfg, ModelConfig):
|
|
220
|
+
raise TypeError("cfg must be a ModelConfig")
|
|
221
|
+
device = str(device or cfg.device or "cpu")
|
|
222
|
+
if demo or cfg.model.startswith("embedflow/demo"):
|
|
223
|
+
return HashEmbeddingModel(cfg.model or "embedflow/demo-hash", cfg.dimension or 64)
|
|
224
|
+
if not cfg.model:
|
|
225
|
+
raise ValueError("model identifier is required")
|
|
226
|
+
return HuggingFaceEmbeddingModel(cfg, model_root=model_root, device=device)
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""Versioned, provenance-aware migration evidence shipped with EmbedFlow.
|
|
2
|
+
|
|
3
|
+
The registry contains only retained research results. It is intentionally
|
|
4
|
+
separate from the runtime cache and never contains vectors or indexes.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from .loader import (
|
|
8
|
+
dataset_fingerprint,
|
|
9
|
+
find_evidence,
|
|
10
|
+
load_benchmark_profiles,
|
|
11
|
+
load_evidence,
|
|
12
|
+
load_manifest,
|
|
13
|
+
load_summaries,
|
|
14
|
+
verify_registry,
|
|
15
|
+
)
|
|
16
|
+
from .matcher import RegistryMatch, match_config, match_evidence
|
|
17
|
+
from .schema import (
|
|
18
|
+
MATCH_EXACT,
|
|
19
|
+
MATCH_NONE,
|
|
20
|
+
MATCH_PRIOR,
|
|
21
|
+
MATCH_RELATED,
|
|
22
|
+
BenchmarkProfile,
|
|
23
|
+
EvidenceRecord,
|
|
24
|
+
RegistryError,
|
|
25
|
+
contract_fingerprint,
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
__all__ = [
|
|
29
|
+
"BenchmarkProfile",
|
|
30
|
+
"EvidenceRecord",
|
|
31
|
+
"MATCH_EXACT",
|
|
32
|
+
"MATCH_NONE",
|
|
33
|
+
"MATCH_PRIOR",
|
|
34
|
+
"MATCH_RELATED",
|
|
35
|
+
"RegistryError",
|
|
36
|
+
"contract_fingerprint",
|
|
37
|
+
"RegistryMatch",
|
|
38
|
+
"dataset_fingerprint",
|
|
39
|
+
"find_evidence",
|
|
40
|
+
"load_benchmark_profiles",
|
|
41
|
+
"load_evidence",
|
|
42
|
+
"load_manifest",
|
|
43
|
+
"load_summaries",
|
|
44
|
+
"match_config",
|
|
45
|
+
"match_evidence",
|
|
46
|
+
"verify_registry",
|
|
47
|
+
]
|