ltcai 11.9.0 → 12.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +80 -59
- package/docs/CHANGELOG.md +92 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +267 -119
- package/docs/ENTERPRISE.md +1 -1
- package/docs/MULTI_AGENT_RUNTIME.md +4 -4
- package/docs/ONBOARDING.md +13 -3
- package/docs/OPERATIONS.md +13 -4
- package/docs/REALTIME_COLLABORATION.md +1 -1
- package/docs/ROADMAP.md +113 -0
- package/docs/TRUST_MODEL.md +16 -2
- package/docs/WHY_LATTICE.md +8 -2
- package/docs/WORKFLOW_DESIGNER.md +2 -2
- package/docs/kg-schema.md +51 -5
- package/docs/mcp-tools.md +17 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/extraction.py +459 -105
- package/lattice_brain/graph/_kg_common/normalize.py +305 -0
- package/lattice_brain/graph/_kg_common/patterns.py +275 -0
- package/lattice_brain/graph/_kg_common/relations.py +12 -3
- package/lattice_brain/graph/_kg_common/sections.py +107 -0
- package/lattice_brain/graph/_kg_constants.py +7 -0
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/agent_worker_seam.py +32 -0
- package/latticeai/api/worker_compute.py +78 -3
- package/latticeai/core/embedding_providers/__init__.py +16 -0
- package/latticeai/core/embedding_providers/autodetect.py +302 -0
- package/latticeai/core/embedding_providers/base.py +25 -0
- package/latticeai/core/embedding_providers/profiles.py +44 -0
- package/latticeai/core/embedding_providers/text.py +74 -8
- package/latticeai/core/vector_index/__init__.py +61 -0
- package/latticeai/core/vector_index/hnsw.py +383 -0
- package/latticeai/core/vector_index/sidecar.py +329 -0
- package/latticeai/models/router/generation.py +101 -22
- package/latticeai/models/router/loading.py +109 -4
- package/latticeai/runtime/brain_runtime.py +43 -9
- package/latticeai/runtime/build_phases/worker_profile.py +13 -4
- package/latticeai/services/architecture_readiness.py +2 -2
- package/latticeai/services/product_readiness.py +10 -5
- package/latticeai/services/search_service.py +7 -0
- package/latticeai/tools/__init__.py +6 -1
- package/latticeai/tools/documents.py +12 -0
- package/latticeai/tools/markup.py +152 -0
- package/package.json +2 -1
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/compose_openapi.py +2 -0
- package/scripts/openapi_route_families.json +7 -3
- package/scripts/publish_release.mjs +157 -0
- package/scripts/release_screen_claims.json +12 -0
- package/src-tauri/Cargo.lock +44 -10
- package/src-tauri/Cargo.toml +1 -1
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +47 -41
- package/static/app/assets/Act-Cf1L2709.js +2 -0
- package/static/app/assets/AdminConsole-DPAbLTYV.js +1 -0
- package/static/app/assets/Brain-DqamGrj-.js +2 -0
- package/static/app/assets/BrainHome-MHe2_RYs.js +2 -0
- package/static/app/assets/BrainSignals-CQPPfyyH.js +1 -0
- package/static/app/assets/Capture-DGdIH_Zc.js +1 -0
- package/static/app/assets/Chronicle-C-UlCJoJ.js +1 -0
- package/static/app/assets/CommandPalette-WNT4EqUX.js +1 -0
- package/static/app/assets/DigitalBrainExplorer-CEBH5Cwc.js +321 -0
- package/static/app/assets/Library-C6xd1dlf.js +1 -0
- package/static/app/assets/LivingBrain-BEk-0ohw.js +1 -0
- package/static/app/assets/ProductFlow-CZLm5iXh.js +1 -0
- package/static/app/assets/QueryClientProvider-B3OjqSyJ.js +1 -0
- package/static/app/assets/ReviewCard-CEHG6evf.js +3 -0
- package/static/app/assets/RunsListPanel-CLtEJSRW.js +1 -0
- package/static/app/assets/System-CAxwBUXw.js +1 -0
- package/static/app/assets/WorkflowGraph-Dj10RuGE.js +1 -0
- package/static/app/assets/WorkflowsPanel-Kyeh_LIT.js +2 -0
- package/static/app/assets/actHelpers-CtSmK9Dw.js +1 -0
- package/static/app/assets/arrow-left-CRl5EO4D.js +1 -0
- package/static/app/assets/{bot-Bc3Q27YR.js → bot-DhUGRel2.js} +1 -1
- package/static/app/assets/brain-CLkhHsHF.js +1 -0
- package/static/app/assets/button-CmaEqG1T.js +1 -0
- package/static/app/assets/circle-check-CFgejkOS.js +1 -0
- package/static/app/assets/{circle-pause-BGMiV8UU.js → circle-pause-l96izbxj.js} +1 -1
- package/static/app/assets/{circle-play-DoanLHnd.js → circle-play-CrZa25_q.js} +1 -1
- package/static/app/assets/{cpu-DwzNf82m.js → cpu-BaXudqwl.js} +1 -1
- package/static/app/assets/{download-Ddw49yCV.js → download-hCVFPiyc.js} +1 -1
- package/static/app/assets/{folder-open-Brd6Kvto.js → folder-open-CHL82Yp7.js} +1 -1
- package/static/app/assets/{hard-drive-Bu-DTJdB.js → hard-drive-DDzET7lk.js} +1 -1
- package/static/app/assets/{index-CGdg_aq9.css → index-CB93CZWW.css} +1 -1
- package/static/app/assets/index-D2H-wSl6.js +13 -0
- package/static/app/assets/input-Df1CAY_I.js +1 -0
- package/static/app/assets/jsx-runtime-bzQ4Vb5N.js +1 -0
- package/static/app/assets/{link-2-DZ4OA5tJ.js → link-2-xNnTIX1_.js} +1 -1
- package/static/app/assets/{permissionCopy-D9TR0F8b.js → permissionCopy-D3aWHco-.js} +1 -1
- package/static/app/assets/primitives-BioD2slS.js +1 -0
- package/static/app/assets/search-BzBw8YcW.js +1 -0
- package/static/app/assets/{share-2-BC5FirFv.js → share-2-FkzGf8Df.js} +1 -1
- package/static/app/assets/{shield-alert-DUbR2W2s.js → shield-alert-B3dwzik4.js} +1 -1
- package/static/app/assets/sourceMeta-DQSY_tah.js +1 -0
- package/static/app/assets/textarea-P8o6pvOP.js +1 -0
- package/static/app/assets/useFocusTrap-hswOIkXE.js +1 -0
- package/static/app/assets/useMutation-OJLrYSRA.js +1 -0
- package/static/app/assets/workspace-BCuk3Ku9.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/static/app/assets/Act-B4WT81kh.js +0 -1
- package/static/app/assets/AdminConsole--Wf71m-o.js +0 -1
- package/static/app/assets/Brain-CtYaa26c.js +0 -321
- package/static/app/assets/BrainHome-Be2VPJEc.js +0 -2
- package/static/app/assets/BrainSignals-C0__xgpG.js +0 -1
- package/static/app/assets/Capture-DdNi5Peb.js +0 -1
- package/static/app/assets/Chronicle-B3hNveeI.js +0 -1
- package/static/app/assets/CommandPalette-CIsnSsFL.js +0 -1
- package/static/app/assets/Library-Bl5XClFV.js +0 -1
- package/static/app/assets/LivingBrain-DjkTB_gh.js +0 -1
- package/static/app/assets/ProductFlow-DCUNRNHs.js +0 -1
- package/static/app/assets/ReviewCard-CD3yWvUB.js +0 -3
- package/static/app/assets/System-BJ6jQ_SL.js +0 -1
- package/static/app/assets/arrow-left-6_28Z0qH.js +0 -1
- package/static/app/assets/brain-B9BDrMTe.js +0 -1
- package/static/app/assets/button-D6JcpYcf.js +0 -1
- package/static/app/assets/circle-check-BFu9lD-3.js +0 -1
- package/static/app/assets/index-CWKRRsLW.js +0 -10
- package/static/app/assets/input-CEqsxtil.js +0 -1
- package/static/app/assets/primitives-DORg7Z_7.js +0 -1
- package/static/app/assets/search-0NQ21wXe.js +0 -1
- package/static/app/assets/textarea-jtQcRSXo.js +0 -1
- package/static/app/assets/useFocusTrap-HRemcWId.js +0 -1
- package/static/app/assets/useMutation-DqlFE-Bw.js +0 -1
- package/static/app/assets/useQuery-BizqBNGw.js +0 -1
- package/static/app/assets/utils-WgW4V69R.js +0 -4
- package/static/app/assets/workspace-DSek3jCY.js +0 -1
|
@@ -45,8 +45,18 @@ _KNOWN_DIMS = {
|
|
|
45
45
|
"gte-small": 384,
|
|
46
46
|
"gte-base": 768,
|
|
47
47
|
"gte-large": 1024,
|
|
48
|
+
"e5-small": 384,
|
|
49
|
+
"e5-base": 768,
|
|
48
50
|
"e5-large": 1024,
|
|
51
|
+
"multilingual-e5-small": 384,
|
|
52
|
+
"multilingual-e5-small-mlx": 384,
|
|
53
|
+
"multilingual-e5-base": 768,
|
|
54
|
+
"multilingual-e5-base-mlx": 768,
|
|
49
55
|
"multilingual-e5-large": 1024,
|
|
56
|
+
"multilingual-e5-large-mlx": 1024,
|
|
57
|
+
"snowflake-arctic-embed-l-v2.0-8bit": 1024,
|
|
58
|
+
"embeddinggemma-300m-4bit": 768,
|
|
59
|
+
"embeddinggemma-300m-8bit": 768,
|
|
50
60
|
"text-embedding-3-small": 1536,
|
|
51
61
|
"text-embedding-3-large": 3072,
|
|
52
62
|
"text-embedding-ada-002": 1536,
|
|
@@ -86,6 +96,21 @@ class EmbeddingProvider:
|
|
|
86
96
|
def embed_batch(self, texts: Sequence[str]) -> List[List[float]]:
|
|
87
97
|
raise NotImplementedError
|
|
88
98
|
|
|
99
|
+
# ── optional: asymmetric models ───────────────────────────────────────
|
|
100
|
+
def embed_batch_for(
|
|
101
|
+
self, texts: Sequence[str], kind: str = "passage"
|
|
102
|
+
) -> List[List[float]]:
|
|
103
|
+
"""Embed for a *role*: ``"query"`` or ``"passage"``.
|
|
104
|
+
|
|
105
|
+
Most embedders are symmetric and ignore the role — the default here
|
|
106
|
+
does. The E5 family is not: it was trained with a literal ``query: ``
|
|
107
|
+
or ``passage: `` in front of the text, and dropping the instruction
|
|
108
|
+
costs real retrieval accuracy. ``POST /worker/embed`` already carries
|
|
109
|
+
the role (``kind``), so the one provider that needs it can have it
|
|
110
|
+
without every caller learning about instructions.
|
|
111
|
+
"""
|
|
112
|
+
return self.embed_batch(texts)
|
|
113
|
+
|
|
89
114
|
# ── derived (shared) ──────────────────────────────────────────────────
|
|
90
115
|
def _model_id_with_dim(self, dim: int) -> str:
|
|
91
116
|
"""This provider's ``model_id`` restated at ``dim``.
|
|
@@ -10,6 +10,50 @@ from __future__ import annotations
|
|
|
10
10
|
from typing import Any, Dict, List
|
|
11
11
|
|
|
12
12
|
PRODUCTION_PROVIDER_PROFILES: Dict[str, Dict[str, Any]] = {
|
|
13
|
+
# ── one-click local profiles (v12.0.0) ────────────────────────────────
|
|
14
|
+
# These three carry `hf_repo_id` and `download_gb` because they are the
|
|
15
|
+
# ones a setup surface can *offer to fetch*: the id is what
|
|
16
|
+
# `huggingface_hub.snapshot_download` takes and what
|
|
17
|
+
# `autodetect.detect_local_mlx` looks for in the cache, so "offer",
|
|
18
|
+
# "download" and "detect" all name the same string. The rest of the table
|
|
19
|
+
# describes providers the user has to bring themselves (an Ollama server,
|
|
20
|
+
# an API key), which is why they have no repo id.
|
|
21
|
+
"local:multilingual-e5-small": {
|
|
22
|
+
"id": "local:multilingual-e5-small",
|
|
23
|
+
"provider": "mlx",
|
|
24
|
+
"model": "mlx-community/multilingual-e5-small-mlx",
|
|
25
|
+
"hf_repo_id": "mlx-community/multilingual-e5-small-mlx",
|
|
26
|
+
"download_gb": 0.24,
|
|
27
|
+
"dimensions": 384,
|
|
28
|
+
"grade": "production",
|
|
29
|
+
"family": "local",
|
|
30
|
+
"label": "Multilingual E5 Small (로컬)",
|
|
31
|
+
"detail": "한국어를 포함한 100여 개 언어. 240MB, 해시 임베더와 같은 384차원.",
|
|
32
|
+
},
|
|
33
|
+
"local:multilingual-e5-base": {
|
|
34
|
+
"id": "local:multilingual-e5-base",
|
|
35
|
+
"provider": "mlx",
|
|
36
|
+
"model": "mlx-community/multilingual-e5-base-mlx",
|
|
37
|
+
"hf_repo_id": "mlx-community/multilingual-e5-base-mlx",
|
|
38
|
+
"download_gb": 1.1,
|
|
39
|
+
"dimensions": 768,
|
|
40
|
+
"grade": "production",
|
|
41
|
+
"family": "local",
|
|
42
|
+
"label": "Multilingual E5 Base (로컬)",
|
|
43
|
+
"detail": "더 정확한 다국어 임베딩. 768차원이라 기존 색인은 다시 만들어야 합니다.",
|
|
44
|
+
},
|
|
45
|
+
"local:arctic-embed-l-v2": {
|
|
46
|
+
"id": "local:arctic-embed-l-v2",
|
|
47
|
+
"provider": "mlx",
|
|
48
|
+
"model": "mlx-community/snowflake-arctic-embed-l-v2.0-8bit",
|
|
49
|
+
"hf_repo_id": "mlx-community/snowflake-arctic-embed-l-v2.0-8bit",
|
|
50
|
+
"download_gb": 0.6,
|
|
51
|
+
"dimensions": 1024,
|
|
52
|
+
"grade": "production",
|
|
53
|
+
"family": "local",
|
|
54
|
+
"label": "Arctic Embed L v2 (로컬)",
|
|
55
|
+
"detail": "다국어 고품질 검색용 임베딩. 1024차원.",
|
|
56
|
+
},
|
|
13
57
|
"local:bge-m3": {
|
|
14
58
|
"id": "local:bge-m3",
|
|
15
59
|
"provider": "mlx",
|
|
@@ -57,7 +57,32 @@ def _as_float_list(value: Any) -> List[Any]:
|
|
|
57
57
|
return list(value)
|
|
58
58
|
|
|
59
59
|
|
|
60
|
+
#: Longest token sequence an encoder-family embedding model accepts. BERT and
|
|
61
|
+
#: XLM-R both stop at 512 including the two special tokens; a longer sequence
|
|
62
|
+
#: is a hard failure inside the model, not a slow answer, so the ids are cut
|
|
63
|
+
#: here rather than discovered at inference time.
|
|
64
|
+
MLX_MAX_TOKENS = 512
|
|
65
|
+
|
|
66
|
+
#: The E5 family was trained with these literal prefixes and loses accuracy
|
|
67
|
+
#: without them. Keyed on the ``kind`` ``POST /worker/embed`` already sends.
|
|
68
|
+
E5_PREFIXES: Dict[str, str] = {"query": "query: ", "passage": "passage: "}
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _wants_e5_prefix(model: str) -> bool:
|
|
72
|
+
"""Whether this model id names an E5-family checkpoint."""
|
|
73
|
+
return "e5" in str(model or "").lower()
|
|
74
|
+
|
|
75
|
+
|
|
60
76
|
class MLXEmbeddingProvider(_NetworkEmbeddingProvider):
|
|
77
|
+
"""A local Apple-Silicon sentence embedder through ``mlx_embeddings``.
|
|
78
|
+
|
|
79
|
+
Nothing here reaches the network: the model must already be in the local
|
|
80
|
+
Hugging Face cache (that is exactly what
|
|
81
|
+
:func:`~.autodetect.detect_local_mlx` looks for). A missing package or a
|
|
82
|
+
missing snapshot is :class:`EmbeddingUnavailable`, which
|
|
83
|
+
:func:`resolve_embedder` turns into the honest hash fallback.
|
|
84
|
+
"""
|
|
85
|
+
|
|
61
86
|
provider = "mlx"
|
|
62
87
|
|
|
63
88
|
def __init__(self, cfg: _RemoteConfig):
|
|
@@ -66,6 +91,7 @@ class MLXEmbeddingProvider(_NetworkEmbeddingProvider):
|
|
|
66
91
|
self.dim = _guess_dim(cfg.model, DEFAULT_EMBEDDING_DIM)
|
|
67
92
|
self.model_id = f"mlx:{cfg.model}:{self.dim}"
|
|
68
93
|
self._encoder: Optional[Tuple[str, Any, Any]] = None
|
|
94
|
+
self._prefixed = _wants_e5_prefix(cfg.model)
|
|
69
95
|
|
|
70
96
|
def _load(self):
|
|
71
97
|
if self._encoder is not None:
|
|
@@ -79,25 +105,37 @@ class MLXEmbeddingProvider(_NetworkEmbeddingProvider):
|
|
|
79
105
|
except Exception as exc: # pragma: no cover - environment dependent
|
|
80
106
|
raise EmbeddingUnavailable(f"MLX embedding model unavailable: {exc}") from exc
|
|
81
107
|
|
|
108
|
+
def embed_batch_for(
|
|
109
|
+
self, texts: Sequence[str], kind: str = "passage"
|
|
110
|
+
) -> List[List[float]]:
|
|
111
|
+
"""E5 asymmetry, honoured. Non-E5 models see the text unchanged."""
|
|
112
|
+
if not self._prefixed:
|
|
113
|
+
return self.embed_batch(texts)
|
|
114
|
+
prefix = E5_PREFIXES.get(str(kind or "passage").lower(), E5_PREFIXES["passage"])
|
|
115
|
+
return self.embed_batch([f"{prefix}{text}" for text in texts])
|
|
116
|
+
|
|
82
117
|
def _embed_raw(self, texts: Sequence[str]) -> List[List[float]]:
|
|
83
|
-
|
|
118
|
+
_, model, tokenizer = self._load()
|
|
84
119
|
try:
|
|
85
120
|
import mlx.core as mx # type: ignore
|
|
86
121
|
|
|
87
122
|
out: List[List[float]] = []
|
|
88
123
|
for text in texts:
|
|
89
|
-
ids = tokenizer.encode(text)
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
pooled = result[0] if isinstance(result, (tuple, list)) else result
|
|
93
|
-
vec = mx.mean(pooled, axis=1)[0] if pooled.ndim == 3 else pooled[0]
|
|
94
|
-
out.append([float(x) for x in _as_float_list(vec.tolist())])
|
|
124
|
+
ids = list(tokenizer.encode(text))[:MLX_MAX_TOKENS]
|
|
125
|
+
result = model(mx.array([ids]))
|
|
126
|
+
out.append([float(x) for x in _pool(mx, result)])
|
|
95
127
|
return out
|
|
96
128
|
except EmbeddingUnavailable:
|
|
97
129
|
raise
|
|
98
130
|
except Exception as exc: # pragma: no cover - environment dependent
|
|
99
131
|
raise EmbeddingUnavailable(f"MLX embedding failed: {exc}") from exc
|
|
100
132
|
|
|
133
|
+
def metadata(self) -> Dict[str, Any]:
|
|
134
|
+
info = super().metadata()
|
|
135
|
+
info["instruction_prefixes"] = self._prefixed
|
|
136
|
+
info["max_tokens"] = MLX_MAX_TOKENS
|
|
137
|
+
return info
|
|
138
|
+
|
|
101
139
|
def health(self) -> Dict[str, Any]:
|
|
102
140
|
try:
|
|
103
141
|
self._load()
|
|
@@ -106,6 +144,27 @@ class MLXEmbeddingProvider(_NetworkEmbeddingProvider):
|
|
|
106
144
|
return {"status": "unavailable", "detail": str(exc)}
|
|
107
145
|
|
|
108
146
|
|
|
147
|
+
def _pool(mx: Any, result: Any) -> List[Any]:
|
|
148
|
+
"""One sentence vector out of whatever shape the model handed back.
|
|
149
|
+
|
|
150
|
+
``mlx_embeddings`` models return an output object carrying a pooled
|
|
151
|
+
``text_embeds`` on newer versions and a bare hidden-state tensor on older
|
|
152
|
+
ones. Preferring the model's own pooled output matters: a checkpoint with a
|
|
153
|
+
trained pooler produces a better vector than a mean over its last layer,
|
|
154
|
+
and mean-pooling on top of an already-pooled row would be nonsense.
|
|
155
|
+
"""
|
|
156
|
+
for attribute in ("text_embeds", "pooler_output"):
|
|
157
|
+
pooled = getattr(result, attribute, None)
|
|
158
|
+
if pooled is not None:
|
|
159
|
+
row = pooled[0] if pooled.ndim > 1 else pooled
|
|
160
|
+
return _as_float_list(row.tolist())
|
|
161
|
+
hidden = getattr(result, "last_hidden_state", None)
|
|
162
|
+
if hidden is None:
|
|
163
|
+
hidden = result[0] if isinstance(result, (tuple, list)) else result
|
|
164
|
+
row = mx.mean(hidden, axis=1)[0] if hidden.ndim == 3 else hidden[0]
|
|
165
|
+
return _as_float_list(row.tolist())
|
|
166
|
+
|
|
167
|
+
|
|
109
168
|
class OllamaEmbeddingProvider(_NetworkEmbeddingProvider):
|
|
110
169
|
provider = "ollama"
|
|
111
170
|
|
|
@@ -285,9 +344,13 @@ class ResolvedEmbedder:
|
|
|
285
344
|
fell_back: bool
|
|
286
345
|
health: Dict[str, Any]
|
|
287
346
|
detail: str = ""
|
|
347
|
+
#: What :mod:`.autodetect` found on this machine, if anyone looked. It is
|
|
348
|
+
#: reported whether or not it was adopted, so a user running on the hash
|
|
349
|
+
#: fallback with a real embedder already downloaded can *see* that.
|
|
350
|
+
detected: Optional[Any] = None
|
|
288
351
|
|
|
289
352
|
def as_dict(self) -> Dict[str, Any]:
|
|
290
|
-
|
|
353
|
+
payload: Dict[str, Any] = {
|
|
291
354
|
"requested_provider": self.requested,
|
|
292
355
|
"active_provider": self.active,
|
|
293
356
|
"fell_back": self.fell_back,
|
|
@@ -295,6 +358,9 @@ class ResolvedEmbedder:
|
|
|
295
358
|
"detail": self.detail,
|
|
296
359
|
**self.provider.metadata(),
|
|
297
360
|
}
|
|
361
|
+
if self.detected is not None and hasattr(self.detected, "as_dict"):
|
|
362
|
+
payload["detected"] = self.detected.as_dict()
|
|
363
|
+
return payload
|
|
298
364
|
|
|
299
365
|
|
|
300
366
|
def resolve_embedder(
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""Search-side vector index backends (HNSW append + query).
|
|
2
|
+
|
|
3
|
+
``POST /worker/vector/query`` serves this sidecar. ``lattice-retrieval``
|
|
4
|
+
asks for top-k candidates when ``LATTICEAI_VECTOR_INDEX=hnsw`` and
|
|
5
|
+
re-scores them exactly. Default env stays the brute scan.
|
|
6
|
+
|
|
7
|
+
This package is the Python HNSW sidecar: a disposable ``.hnsw`` graph
|
|
8
|
+
next to the brain database, with incremental ``add_items`` so a write no
|
|
9
|
+
longer rebuilds the whole graph.
|
|
10
|
+
|
|
11
|
+
Rebuild (not append) when:
|
|
12
|
+
|
|
13
|
+
* the embedder identity (``model_id`` / dim / space) changed
|
|
14
|
+
* any previously indexed id is absent (deletion). Tombstones exist in
|
|
15
|
+
hnswlib (``mark_deleted``) but a deletion-heavy store (more than
|
|
16
|
+
:data:`DELETE_REBUILD_RATIO` of the graph, or any delete when the
|
|
17
|
+
sidecar was loaded without source vectors) is cheaper to rebuild
|
|
18
|
+
than to fragment.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from .hnsw import (
|
|
22
|
+
DELETE_REBUILD_RATIO,
|
|
23
|
+
HNSW_BACKEND,
|
|
24
|
+
HNSW_META_SUFFIX,
|
|
25
|
+
HNSW_SUFFIX,
|
|
26
|
+
HNSWLIB_MODULE,
|
|
27
|
+
HnswIndex,
|
|
28
|
+
hnswlib_available,
|
|
29
|
+
load_hnswlib,
|
|
30
|
+
sidecar_paths,
|
|
31
|
+
)
|
|
32
|
+
from .sidecar import (
|
|
33
|
+
GRAPH_DB_NAME,
|
|
34
|
+
VECTOR_QUERY_K_CAP,
|
|
35
|
+
decode_f32le,
|
|
36
|
+
query_sidecar,
|
|
37
|
+
reset_sidecar_cache,
|
|
38
|
+
resolve_graph_db,
|
|
39
|
+
sidecar_fingerprint,
|
|
40
|
+
sidecar_freshness,
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
__all__ = [
|
|
44
|
+
"DELETE_REBUILD_RATIO",
|
|
45
|
+
"GRAPH_DB_NAME",
|
|
46
|
+
"HNSW_BACKEND",
|
|
47
|
+
"HNSW_META_SUFFIX",
|
|
48
|
+
"HNSW_SUFFIX",
|
|
49
|
+
"HNSWLIB_MODULE",
|
|
50
|
+
"HnswIndex",
|
|
51
|
+
"VECTOR_QUERY_K_CAP",
|
|
52
|
+
"decode_f32le",
|
|
53
|
+
"hnswlib_available",
|
|
54
|
+
"load_hnswlib",
|
|
55
|
+
"query_sidecar",
|
|
56
|
+
"reset_sidecar_cache",
|
|
57
|
+
"resolve_graph_db",
|
|
58
|
+
"sidecar_fingerprint",
|
|
59
|
+
"sidecar_freshness",
|
|
60
|
+
"sidecar_paths",
|
|
61
|
+
]
|
|
@@ -0,0 +1,383 @@
|
|
|
1
|
+
"""Approximate nearest-neighbour index with incremental append.
|
|
2
|
+
|
|
3
|
+
``hnswlib`` is optional (``pip install hnswlib``; the published
|
|
4
|
+
``ltcai[hnsw]`` extra was retired in 11.6.0 when the write path moved to
|
|
5
|
+
Rust). When the compiled module is missing the import is caught here and
|
|
6
|
+
reported as a reason string; search then returns no ANN hits rather than
|
|
7
|
+
pretending.
|
|
8
|
+
|
|
9
|
+
Policy, also stated on :meth:`HnswIndex.add_items`:
|
|
10
|
+
|
|
11
|
+
* **Append** new ids onto the live graph (``resize_index`` + ``add_items``).
|
|
12
|
+
* **Full rebuild** on provider/dim/space change, on any deletion when the
|
|
13
|
+
graph was loaded from a sidecar without source vectors, or when deletes
|
|
14
|
+
exceed :data:`DELETE_REBUILD_RATIO` of the current size.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import json
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Any, Dict, Iterable, List, Mapping, Optional, Sequence, Tuple
|
|
22
|
+
|
|
23
|
+
HNSW_BACKEND = "hnsw"
|
|
24
|
+
HNSWLIB_MODULE = "hnswlib"
|
|
25
|
+
HNSW_SUFFIX = ".hnsw"
|
|
26
|
+
HNSW_META_SUFFIX = ".hnsw.meta.json"
|
|
27
|
+
#: Rebuild instead of tombstoning once this fraction of the graph would go.
|
|
28
|
+
DELETE_REBUILD_RATIO = 0.10
|
|
29
|
+
|
|
30
|
+
IndexItem = Tuple[str, Sequence[float], Mapping[str, Any]]
|
|
31
|
+
ScoredId = Tuple[str, float]
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def load_hnswlib() -> Tuple[Optional[Any], Optional[str]]:
|
|
35
|
+
"""Import ``hnswlib`` or explain why it is unavailable (never raises)."""
|
|
36
|
+
try:
|
|
37
|
+
import hnswlib # noqa: PLC0415 — guarded, optional, import-on-demand
|
|
38
|
+
except Exception as exc: # noqa: BLE001 — any import failure is "no ANN"
|
|
39
|
+
return None, (
|
|
40
|
+
f"{HNSWLIB_MODULE} is not available ({exc}); install it with "
|
|
41
|
+
"`pip install hnswlib` to use the approximate index"
|
|
42
|
+
)
|
|
43
|
+
return hnswlib, None
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def hnswlib_available() -> bool:
|
|
47
|
+
"""True when the optional ANN engine can actually be imported."""
|
|
48
|
+
module, _ = load_hnswlib()
|
|
49
|
+
return module is not None
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def sidecar_paths(db_path: Any) -> Tuple[Path, Path]:
|
|
53
|
+
"""``(index file, meta file)`` for a brain database path."""
|
|
54
|
+
base = Path(db_path)
|
|
55
|
+
return (
|
|
56
|
+
base.with_suffix(HNSW_SUFFIX),
|
|
57
|
+
base.with_suffix(HNSW_META_SUFFIX),
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class HnswIndex:
|
|
62
|
+
"""Small-M HNSW graph over the brain's embeddings."""
|
|
63
|
+
|
|
64
|
+
def __init__(
|
|
65
|
+
self,
|
|
66
|
+
*,
|
|
67
|
+
dim: int,
|
|
68
|
+
space: str = "cosine",
|
|
69
|
+
ef_construction: int = 400,
|
|
70
|
+
m: int = 32,
|
|
71
|
+
ef_search: int = 400,
|
|
72
|
+
model_id: str = "",
|
|
73
|
+
) -> None:
|
|
74
|
+
self._dim = int(dim)
|
|
75
|
+
self._space = str(space)
|
|
76
|
+
self._ef_construction = int(ef_construction)
|
|
77
|
+
self._m = int(m)
|
|
78
|
+
self._ef_search = int(ef_search)
|
|
79
|
+
self._model_id = str(model_id)
|
|
80
|
+
self._module, self._detail = load_hnswlib()
|
|
81
|
+
self._vectors: Dict[str, List[float]] = {}
|
|
82
|
+
self._metadata: Dict[str, Dict[str, Any]] = {}
|
|
83
|
+
self._labels: List[str] = []
|
|
84
|
+
self._ann: Any = None
|
|
85
|
+
self._dirty = True
|
|
86
|
+
self._loaded = False
|
|
87
|
+
|
|
88
|
+
@property
|
|
89
|
+
def backend(self) -> str:
|
|
90
|
+
return HNSW_BACKEND
|
|
91
|
+
|
|
92
|
+
@property
|
|
93
|
+
def approx(self) -> bool:
|
|
94
|
+
return True
|
|
95
|
+
|
|
96
|
+
@property
|
|
97
|
+
def exhaustive(self) -> bool:
|
|
98
|
+
return False
|
|
99
|
+
|
|
100
|
+
@property
|
|
101
|
+
def available(self) -> bool:
|
|
102
|
+
return self._module is not None
|
|
103
|
+
|
|
104
|
+
@property
|
|
105
|
+
def unavailable_detail(self) -> Optional[str]:
|
|
106
|
+
return self._detail
|
|
107
|
+
|
|
108
|
+
@property
|
|
109
|
+
def loaded_from_sidecar(self) -> bool:
|
|
110
|
+
"""True when the graph came off disk instead of being rebuilt."""
|
|
111
|
+
return self._loaded
|
|
112
|
+
|
|
113
|
+
@property
|
|
114
|
+
def dim(self) -> int:
|
|
115
|
+
return self._dim
|
|
116
|
+
|
|
117
|
+
@property
|
|
118
|
+
def model_id(self) -> str:
|
|
119
|
+
return self._model_id
|
|
120
|
+
|
|
121
|
+
def _identity_matches(self, dim: int, space: str, model_id: str) -> bool:
|
|
122
|
+
return (
|
|
123
|
+
int(dim) == self._dim
|
|
124
|
+
and str(space) == self._space
|
|
125
|
+
and str(model_id) == self._model_id
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
def add(
|
|
129
|
+
self,
|
|
130
|
+
id: str,
|
|
131
|
+
vector: Sequence[float],
|
|
132
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
133
|
+
) -> None:
|
|
134
|
+
if self._loaded:
|
|
135
|
+
raise RuntimeError(
|
|
136
|
+
"this HnswIndex was loaded from a sidecar and holds no source "
|
|
137
|
+
"vectors; rebuild it from vector_embeddings before mutating"
|
|
138
|
+
)
|
|
139
|
+
key = str(id)
|
|
140
|
+
self._vectors[key] = [float(value) for value in vector]
|
|
141
|
+
self._metadata[key] = dict(metadata or {})
|
|
142
|
+
self._dirty = True
|
|
143
|
+
|
|
144
|
+
def remove(self, id: str) -> None:
|
|
145
|
+
if self._loaded:
|
|
146
|
+
raise RuntimeError(
|
|
147
|
+
"this HnswIndex was loaded from a sidecar and holds no source "
|
|
148
|
+
"vectors; rebuild it from vector_embeddings before mutating"
|
|
149
|
+
)
|
|
150
|
+
key = str(id)
|
|
151
|
+
self._vectors.pop(key, None)
|
|
152
|
+
self._metadata.pop(key, None)
|
|
153
|
+
self._dirty = True
|
|
154
|
+
|
|
155
|
+
def rebuild(self, items: Iterable[IndexItem]) -> None:
|
|
156
|
+
"""Replace the held set and mark the graph for a full rebuild."""
|
|
157
|
+
self._vectors.clear()
|
|
158
|
+
self._metadata.clear()
|
|
159
|
+
self._labels.clear()
|
|
160
|
+
self._ann = None
|
|
161
|
+
self._loaded = False
|
|
162
|
+
for item_id, vector, metadata in items:
|
|
163
|
+
self.add(item_id, vector, metadata)
|
|
164
|
+
self._dirty = True
|
|
165
|
+
|
|
166
|
+
def add_items(
|
|
167
|
+
self,
|
|
168
|
+
items: Iterable[IndexItem],
|
|
169
|
+
*,
|
|
170
|
+
dim: Optional[int] = None,
|
|
171
|
+
space: Optional[str] = None,
|
|
172
|
+
model_id: Optional[str] = None,
|
|
173
|
+
) -> str:
|
|
174
|
+
"""Append new vectors, or rebuild when the policy says so.
|
|
175
|
+
|
|
176
|
+
Returns ``"append"`` or ``"rebuild"`` so a bench can time the two
|
|
177
|
+
doors without guessing.
|
|
178
|
+
|
|
179
|
+
Full rebuild when the embedder identity changed, when any currently
|
|
180
|
+
indexed id is missing from the incoming set (a deletion), when
|
|
181
|
+
deletes would exceed :data:`DELETE_REBUILD_RATIO`, or when this
|
|
182
|
+
instance was loaded from a sidecar and cannot append (no source
|
|
183
|
+
vectors to keep the label map honest).
|
|
184
|
+
"""
|
|
185
|
+
incoming: Dict[str, IndexItem] = {}
|
|
186
|
+
for item_id, vector, metadata in items:
|
|
187
|
+
incoming[str(item_id)] = (str(item_id), vector, metadata)
|
|
188
|
+
|
|
189
|
+
want_dim = self._dim if dim is None else int(dim)
|
|
190
|
+
want_space = self._space if space is None else str(space)
|
|
191
|
+
want_model = self._model_id if model_id is None else str(model_id)
|
|
192
|
+
if not self._identity_matches(want_dim, want_space, want_model):
|
|
193
|
+
self._dim = want_dim
|
|
194
|
+
self._space = want_space
|
|
195
|
+
self._model_id = want_model
|
|
196
|
+
self.rebuild(incoming.values())
|
|
197
|
+
return "rebuild"
|
|
198
|
+
|
|
199
|
+
if self._loaded:
|
|
200
|
+
self.rebuild(incoming.values())
|
|
201
|
+
return "rebuild"
|
|
202
|
+
|
|
203
|
+
current = set(self._vectors)
|
|
204
|
+
incoming_ids = set(incoming)
|
|
205
|
+
deleted = current - incoming_ids
|
|
206
|
+
if deleted:
|
|
207
|
+
ratio = len(deleted) / max(1, len(current))
|
|
208
|
+
if ratio >= DELETE_REBUILD_RATIO or not current:
|
|
209
|
+
self.rebuild(incoming.values())
|
|
210
|
+
return "rebuild"
|
|
211
|
+
# Sparse deletes: drop the gone ids and rebuild so we never
|
|
212
|
+
# leave a tombstoned hole the sidecar cannot describe.
|
|
213
|
+
self.rebuild(incoming.values())
|
|
214
|
+
return "rebuild"
|
|
215
|
+
|
|
216
|
+
new_ids = [item_id for item_id in incoming if item_id not in current]
|
|
217
|
+
if not new_ids and not self._dirty and self._ann is not None:
|
|
218
|
+
return "append"
|
|
219
|
+
for item_id in new_ids:
|
|
220
|
+
_, vector, metadata = incoming[item_id]
|
|
221
|
+
self.add(item_id, vector, metadata)
|
|
222
|
+
if self._ann is not None and new_ids and self._module is not None:
|
|
223
|
+
self._append_live(new_ids)
|
|
224
|
+
return "append"
|
|
225
|
+
self._dirty = True
|
|
226
|
+
return "append" if new_ids or self._ann is not None else "rebuild"
|
|
227
|
+
|
|
228
|
+
def _new_graph(self, module: Any, capacity: int) -> Any:
|
|
229
|
+
graph = module.Index(space=self._space, dim=self._dim)
|
|
230
|
+
graph.init_index(
|
|
231
|
+
max_elements=max(1, capacity),
|
|
232
|
+
ef_construction=self._ef_construction,
|
|
233
|
+
M=self._m,
|
|
234
|
+
)
|
|
235
|
+
return graph
|
|
236
|
+
|
|
237
|
+
def _append_live(self, new_ids: Sequence[str]) -> None:
|
|
238
|
+
"""``hnswlib`` ``resize_index`` + ``add_items`` for ``new_ids`` only."""
|
|
239
|
+
module = self._module
|
|
240
|
+
graph = self._ann
|
|
241
|
+
if module is None or graph is None:
|
|
242
|
+
self._dirty = True
|
|
243
|
+
return
|
|
244
|
+
start = len(self._labels)
|
|
245
|
+
needed = start + len(new_ids)
|
|
246
|
+
try:
|
|
247
|
+
current_max = int(graph.get_max_elements())
|
|
248
|
+
except Exception: # noqa: BLE001 — some builds expose no getter
|
|
249
|
+
current_max = start
|
|
250
|
+
if needed > current_max:
|
|
251
|
+
graph.resize_index(max(needed, current_max * 2 if current_max else needed))
|
|
252
|
+
graph.add_items(
|
|
253
|
+
[self._vectors[key] for key in new_ids],
|
|
254
|
+
list(range(start, start + len(new_ids))),
|
|
255
|
+
)
|
|
256
|
+
self._labels.extend(new_ids)
|
|
257
|
+
self._dirty = False
|
|
258
|
+
|
|
259
|
+
def _ensure_graph(self) -> Any:
|
|
260
|
+
"""Build the graph if the held vectors changed (None when disabled)."""
|
|
261
|
+
module = self._module
|
|
262
|
+
if module is None:
|
|
263
|
+
return None
|
|
264
|
+
if self._ann is not None and not self._dirty:
|
|
265
|
+
return self._ann
|
|
266
|
+
labels = list(self._vectors)
|
|
267
|
+
graph = self._new_graph(module, max(len(labels), 1))
|
|
268
|
+
if labels:
|
|
269
|
+
graph.add_items(
|
|
270
|
+
[self._vectors[key] for key in labels],
|
|
271
|
+
list(range(len(labels))),
|
|
272
|
+
)
|
|
273
|
+
self._ann = graph
|
|
274
|
+
self._labels = labels
|
|
275
|
+
self._dirty = False
|
|
276
|
+
return graph
|
|
277
|
+
|
|
278
|
+
def search(
|
|
279
|
+
self,
|
|
280
|
+
query: Sequence[float],
|
|
281
|
+
top_k: int,
|
|
282
|
+
filter: Optional[Mapping[str, Any]] = None,
|
|
283
|
+
) -> List[ScoredId]:
|
|
284
|
+
graph = self._ensure_graph()
|
|
285
|
+
if graph is None or not self._labels:
|
|
286
|
+
return []
|
|
287
|
+
floor = float("-inf")
|
|
288
|
+
if filter and filter.get("min_score") is not None:
|
|
289
|
+
try:
|
|
290
|
+
floor = float(filter["min_score"])
|
|
291
|
+
except (TypeError, ValueError):
|
|
292
|
+
floor = float("-inf")
|
|
293
|
+
wanted = max(1, min(int(top_k), len(self._labels)))
|
|
294
|
+
graph.set_ef(max(self._ef_search, wanted))
|
|
295
|
+
labels, distances = graph.knn_query([list(query)], k=wanted)
|
|
296
|
+
scored: List[ScoredId] = []
|
|
297
|
+
for label, distance in zip(labels[0], distances[0], strict=True):
|
|
298
|
+
score = 1.0 - float(distance)
|
|
299
|
+
if score < floor:
|
|
300
|
+
continue
|
|
301
|
+
scored.append((self._labels[int(label)], score))
|
|
302
|
+
return scored
|
|
303
|
+
|
|
304
|
+
def stats(self) -> Dict[str, Any]:
|
|
305
|
+
return {
|
|
306
|
+
"backend": self.backend,
|
|
307
|
+
"size": len(self._labels) if self._loaded else len(self._vectors),
|
|
308
|
+
"dim": self._dim,
|
|
309
|
+
"approx": True,
|
|
310
|
+
"exhaustive": False,
|
|
311
|
+
"available": self.available,
|
|
312
|
+
"detail": self._detail,
|
|
313
|
+
"model_id": self._model_id,
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
def save(self, db_path: Any, *, fingerprint: str) -> bool:
|
|
317
|
+
"""Write the graph + label map beside the brain database."""
|
|
318
|
+
graph = self._ensure_graph()
|
|
319
|
+
if graph is None or not self._labels:
|
|
320
|
+
return False
|
|
321
|
+
index_path, meta_path = sidecar_paths(db_path)
|
|
322
|
+
try:
|
|
323
|
+
index_path.parent.mkdir(parents=True, exist_ok=True)
|
|
324
|
+
graph.save_index(str(index_path))
|
|
325
|
+
meta_path.write_text(
|
|
326
|
+
json.dumps(
|
|
327
|
+
{
|
|
328
|
+
"fingerprint": str(fingerprint),
|
|
329
|
+
"dim": self._dim,
|
|
330
|
+
"space": self._space,
|
|
331
|
+
"model_id": self._model_id,
|
|
332
|
+
"labels": self._labels,
|
|
333
|
+
},
|
|
334
|
+
ensure_ascii=False,
|
|
335
|
+
),
|
|
336
|
+
encoding="utf-8",
|
|
337
|
+
)
|
|
338
|
+
except Exception: # noqa: BLE001 — persistence is best-effort
|
|
339
|
+
return False
|
|
340
|
+
return True
|
|
341
|
+
|
|
342
|
+
def load(self, db_path: Any, *, fingerprint: str) -> bool:
|
|
343
|
+
"""Adopt a sidecar graph when it provably matches ``fingerprint``."""
|
|
344
|
+
module = self._module
|
|
345
|
+
if module is None:
|
|
346
|
+
return False
|
|
347
|
+
index_path, meta_path = sidecar_paths(db_path)
|
|
348
|
+
try:
|
|
349
|
+
meta = json.loads(meta_path.read_text(encoding="utf-8"))
|
|
350
|
+
matches = str(meta["fingerprint"]) == str(fingerprint)
|
|
351
|
+
matches = matches and int(meta["dim"]) == self._dim
|
|
352
|
+
if meta.get("model_id") not in (None, "", self._model_id):
|
|
353
|
+
matches = False
|
|
354
|
+
labels = [str(label) for label in meta["labels"]]
|
|
355
|
+
except Exception: # noqa: BLE001 — absent/corrupt sidecar = rebuild
|
|
356
|
+
return False
|
|
357
|
+
if not (matches and labels):
|
|
358
|
+
return False
|
|
359
|
+
try:
|
|
360
|
+
graph = module.Index(space=self._space, dim=self._dim)
|
|
361
|
+
graph.load_index(str(index_path), max_elements=len(labels))
|
|
362
|
+
except Exception: # noqa: BLE001 — corrupt binary = rebuild
|
|
363
|
+
return False
|
|
364
|
+
self._ann = graph
|
|
365
|
+
self._labels = labels
|
|
366
|
+
self._vectors.clear()
|
|
367
|
+
self._metadata.clear()
|
|
368
|
+
self._dirty = False
|
|
369
|
+
self._loaded = True
|
|
370
|
+
return True
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
__all__ = [
|
|
374
|
+
"DELETE_REBUILD_RATIO",
|
|
375
|
+
"HNSW_BACKEND",
|
|
376
|
+
"HNSW_META_SUFFIX",
|
|
377
|
+
"HNSW_SUFFIX",
|
|
378
|
+
"HNSWLIB_MODULE",
|
|
379
|
+
"HnswIndex",
|
|
380
|
+
"hnswlib_available",
|
|
381
|
+
"load_hnswlib",
|
|
382
|
+
"sidecar_paths",
|
|
383
|
+
]
|