ltcai 11.9.0 → 12.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (127) hide show
  1. package/README.md +80 -59
  2. package/docs/CHANGELOG.md +92 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +267 -119
  5. package/docs/ENTERPRISE.md +1 -1
  6. package/docs/MULTI_AGENT_RUNTIME.md +4 -4
  7. package/docs/ONBOARDING.md +13 -3
  8. package/docs/OPERATIONS.md +13 -4
  9. package/docs/REALTIME_COLLABORATION.md +1 -1
  10. package/docs/ROADMAP.md +113 -0
  11. package/docs/TRUST_MODEL.md +16 -2
  12. package/docs/WHY_LATTICE.md +8 -2
  13. package/docs/WORKFLOW_DESIGNER.md +2 -2
  14. package/docs/kg-schema.md +51 -5
  15. package/docs/mcp-tools.md +17 -0
  16. package/lattice_brain/__init__.py +1 -1
  17. package/lattice_brain/graph/_kg_common/extraction.py +459 -105
  18. package/lattice_brain/graph/_kg_common/normalize.py +305 -0
  19. package/lattice_brain/graph/_kg_common/patterns.py +275 -0
  20. package/lattice_brain/graph/_kg_common/relations.py +12 -3
  21. package/lattice_brain/graph/_kg_common/sections.py +107 -0
  22. package/lattice_brain/graph/_kg_constants.py +7 -0
  23. package/latticeai/__init__.py +1 -1
  24. package/latticeai/api/agent_worker_seam.py +32 -0
  25. package/latticeai/api/worker_compute.py +78 -3
  26. package/latticeai/core/embedding_providers/__init__.py +16 -0
  27. package/latticeai/core/embedding_providers/autodetect.py +302 -0
  28. package/latticeai/core/embedding_providers/base.py +25 -0
  29. package/latticeai/core/embedding_providers/profiles.py +44 -0
  30. package/latticeai/core/embedding_providers/text.py +74 -8
  31. package/latticeai/core/vector_index/__init__.py +61 -0
  32. package/latticeai/core/vector_index/hnsw.py +383 -0
  33. package/latticeai/core/vector_index/sidecar.py +329 -0
  34. package/latticeai/models/router/generation.py +101 -22
  35. package/latticeai/models/router/loading.py +109 -4
  36. package/latticeai/runtime/brain_runtime.py +43 -9
  37. package/latticeai/runtime/build_phases/worker_profile.py +13 -4
  38. package/latticeai/services/architecture_readiness.py +2 -2
  39. package/latticeai/services/product_readiness.py +10 -5
  40. package/latticeai/services/search_service.py +7 -0
  41. package/latticeai/tools/__init__.py +6 -1
  42. package/latticeai/tools/documents.py +12 -0
  43. package/latticeai/tools/markup.py +152 -0
  44. package/package.json +2 -1
  45. package/scripts/check_current_release_docs.mjs +1 -1
  46. package/scripts/compose_openapi.py +2 -0
  47. package/scripts/openapi_route_families.json +7 -3
  48. package/scripts/publish_release.mjs +157 -0
  49. package/scripts/release_screen_claims.json +12 -0
  50. package/src-tauri/Cargo.lock +44 -10
  51. package/src-tauri/Cargo.toml +1 -1
  52. package/src-tauri/tauri.conf.json +1 -1
  53. package/static/app/asset-manifest.json +47 -41
  54. package/static/app/assets/Act-Cf1L2709.js +2 -0
  55. package/static/app/assets/AdminConsole-DPAbLTYV.js +1 -0
  56. package/static/app/assets/Brain-DqamGrj-.js +2 -0
  57. package/static/app/assets/BrainHome-MHe2_RYs.js +2 -0
  58. package/static/app/assets/BrainSignals-CQPPfyyH.js +1 -0
  59. package/static/app/assets/Capture-DGdIH_Zc.js +1 -0
  60. package/static/app/assets/Chronicle-C-UlCJoJ.js +1 -0
  61. package/static/app/assets/CommandPalette-WNT4EqUX.js +1 -0
  62. package/static/app/assets/DigitalBrainExplorer-CEBH5Cwc.js +321 -0
  63. package/static/app/assets/Library-C6xd1dlf.js +1 -0
  64. package/static/app/assets/LivingBrain-BEk-0ohw.js +1 -0
  65. package/static/app/assets/ProductFlow-CZLm5iXh.js +1 -0
  66. package/static/app/assets/QueryClientProvider-B3OjqSyJ.js +1 -0
  67. package/static/app/assets/ReviewCard-CEHG6evf.js +3 -0
  68. package/static/app/assets/RunsListPanel-CLtEJSRW.js +1 -0
  69. package/static/app/assets/System-CAxwBUXw.js +1 -0
  70. package/static/app/assets/WorkflowGraph-Dj10RuGE.js +1 -0
  71. package/static/app/assets/WorkflowsPanel-Kyeh_LIT.js +2 -0
  72. package/static/app/assets/actHelpers-CtSmK9Dw.js +1 -0
  73. package/static/app/assets/arrow-left-CRl5EO4D.js +1 -0
  74. package/static/app/assets/{bot-Bc3Q27YR.js → bot-DhUGRel2.js} +1 -1
  75. package/static/app/assets/brain-CLkhHsHF.js +1 -0
  76. package/static/app/assets/button-CmaEqG1T.js +1 -0
  77. package/static/app/assets/circle-check-CFgejkOS.js +1 -0
  78. package/static/app/assets/{circle-pause-BGMiV8UU.js → circle-pause-l96izbxj.js} +1 -1
  79. package/static/app/assets/{circle-play-DoanLHnd.js → circle-play-CrZa25_q.js} +1 -1
  80. package/static/app/assets/{cpu-DwzNf82m.js → cpu-BaXudqwl.js} +1 -1
  81. package/static/app/assets/{download-Ddw49yCV.js → download-hCVFPiyc.js} +1 -1
  82. package/static/app/assets/{folder-open-Brd6Kvto.js → folder-open-CHL82Yp7.js} +1 -1
  83. package/static/app/assets/{hard-drive-Bu-DTJdB.js → hard-drive-DDzET7lk.js} +1 -1
  84. package/static/app/assets/{index-CGdg_aq9.css → index-CB93CZWW.css} +1 -1
  85. package/static/app/assets/index-D2H-wSl6.js +13 -0
  86. package/static/app/assets/input-Df1CAY_I.js +1 -0
  87. package/static/app/assets/jsx-runtime-bzQ4Vb5N.js +1 -0
  88. package/static/app/assets/{link-2-DZ4OA5tJ.js → link-2-xNnTIX1_.js} +1 -1
  89. package/static/app/assets/{permissionCopy-D9TR0F8b.js → permissionCopy-D3aWHco-.js} +1 -1
  90. package/static/app/assets/primitives-BioD2slS.js +1 -0
  91. package/static/app/assets/search-BzBw8YcW.js +1 -0
  92. package/static/app/assets/{share-2-BC5FirFv.js → share-2-FkzGf8Df.js} +1 -1
  93. package/static/app/assets/{shield-alert-DUbR2W2s.js → shield-alert-B3dwzik4.js} +1 -1
  94. package/static/app/assets/sourceMeta-DQSY_tah.js +1 -0
  95. package/static/app/assets/textarea-P8o6pvOP.js +1 -0
  96. package/static/app/assets/useFocusTrap-hswOIkXE.js +1 -0
  97. package/static/app/assets/useMutation-OJLrYSRA.js +1 -0
  98. package/static/app/assets/workspace-BCuk3Ku9.js +1 -0
  99. package/static/app/index.html +4 -4
  100. package/static/sw.js +1 -1
  101. package/static/app/assets/Act-B4WT81kh.js +0 -1
  102. package/static/app/assets/AdminConsole--Wf71m-o.js +0 -1
  103. package/static/app/assets/Brain-CtYaa26c.js +0 -321
  104. package/static/app/assets/BrainHome-Be2VPJEc.js +0 -2
  105. package/static/app/assets/BrainSignals-C0__xgpG.js +0 -1
  106. package/static/app/assets/Capture-DdNi5Peb.js +0 -1
  107. package/static/app/assets/Chronicle-B3hNveeI.js +0 -1
  108. package/static/app/assets/CommandPalette-CIsnSsFL.js +0 -1
  109. package/static/app/assets/Library-Bl5XClFV.js +0 -1
  110. package/static/app/assets/LivingBrain-DjkTB_gh.js +0 -1
  111. package/static/app/assets/ProductFlow-DCUNRNHs.js +0 -1
  112. package/static/app/assets/ReviewCard-CD3yWvUB.js +0 -3
  113. package/static/app/assets/System-BJ6jQ_SL.js +0 -1
  114. package/static/app/assets/arrow-left-6_28Z0qH.js +0 -1
  115. package/static/app/assets/brain-B9BDrMTe.js +0 -1
  116. package/static/app/assets/button-D6JcpYcf.js +0 -1
  117. package/static/app/assets/circle-check-BFu9lD-3.js +0 -1
  118. package/static/app/assets/index-CWKRRsLW.js +0 -10
  119. package/static/app/assets/input-CEqsxtil.js +0 -1
  120. package/static/app/assets/primitives-DORg7Z_7.js +0 -1
  121. package/static/app/assets/search-0NQ21wXe.js +0 -1
  122. package/static/app/assets/textarea-jtQcRSXo.js +0 -1
  123. package/static/app/assets/useFocusTrap-HRemcWId.js +0 -1
  124. package/static/app/assets/useMutation-DqlFE-Bw.js +0 -1
  125. package/static/app/assets/useQuery-BizqBNGw.js +0 -1
  126. package/static/app/assets/utils-WgW4V69R.js +0 -4
  127. package/static/app/assets/workspace-DSek3jCY.js +0 -1
@@ -45,8 +45,18 @@ _KNOWN_DIMS = {
45
45
  "gte-small": 384,
46
46
  "gte-base": 768,
47
47
  "gte-large": 1024,
48
+ "e5-small": 384,
49
+ "e5-base": 768,
48
50
  "e5-large": 1024,
51
+ "multilingual-e5-small": 384,
52
+ "multilingual-e5-small-mlx": 384,
53
+ "multilingual-e5-base": 768,
54
+ "multilingual-e5-base-mlx": 768,
49
55
  "multilingual-e5-large": 1024,
56
+ "multilingual-e5-large-mlx": 1024,
57
+ "snowflake-arctic-embed-l-v2.0-8bit": 1024,
58
+ "embeddinggemma-300m-4bit": 768,
59
+ "embeddinggemma-300m-8bit": 768,
50
60
  "text-embedding-3-small": 1536,
51
61
  "text-embedding-3-large": 3072,
52
62
  "text-embedding-ada-002": 1536,
@@ -86,6 +96,21 @@ class EmbeddingProvider:
86
96
  def embed_batch(self, texts: Sequence[str]) -> List[List[float]]:
87
97
  raise NotImplementedError
88
98
 
99
+ # ── optional: asymmetric models ───────────────────────────────────────
100
+ def embed_batch_for(
101
+ self, texts: Sequence[str], kind: str = "passage"
102
+ ) -> List[List[float]]:
103
+ """Embed for a *role*: ``"query"`` or ``"passage"``.
104
+
105
+ Most embedders are symmetric and ignore the role — the default here
106
+ does. The E5 family is not: it was trained with a literal ``query: ``
107
+ or ``passage: `` in front of the text, and dropping the instruction
108
+ costs real retrieval accuracy. ``POST /worker/embed`` already carries
109
+ the role (``kind``), so the one provider that needs it can have it
110
+ without every caller learning about instructions.
111
+ """
112
+ return self.embed_batch(texts)
113
+
89
114
  # ── derived (shared) ──────────────────────────────────────────────────
90
115
  def _model_id_with_dim(self, dim: int) -> str:
91
116
  """This provider's ``model_id`` restated at ``dim``.
@@ -10,6 +10,50 @@ from __future__ import annotations
10
10
  from typing import Any, Dict, List
11
11
 
12
12
  PRODUCTION_PROVIDER_PROFILES: Dict[str, Dict[str, Any]] = {
13
+ # ── one-click local profiles (v12.0.0) ────────────────────────────────
14
+ # These three carry `hf_repo_id` and `download_gb` because they are the
15
+ # ones a setup surface can *offer to fetch*: the id is what
16
+ # `huggingface_hub.snapshot_download` takes and what
17
+ # `autodetect.detect_local_mlx` looks for in the cache, so "offer",
18
+ # "download" and "detect" all name the same string. The rest of the table
19
+ # describes providers the user has to bring themselves (an Ollama server,
20
+ # an API key), which is why they have no repo id.
21
+ "local:multilingual-e5-small": {
22
+ "id": "local:multilingual-e5-small",
23
+ "provider": "mlx",
24
+ "model": "mlx-community/multilingual-e5-small-mlx",
25
+ "hf_repo_id": "mlx-community/multilingual-e5-small-mlx",
26
+ "download_gb": 0.24,
27
+ "dimensions": 384,
28
+ "grade": "production",
29
+ "family": "local",
30
+ "label": "Multilingual E5 Small (로컬)",
31
+ "detail": "한국어를 포함한 100여 개 언어. 240MB, 해시 임베더와 같은 384차원.",
32
+ },
33
+ "local:multilingual-e5-base": {
34
+ "id": "local:multilingual-e5-base",
35
+ "provider": "mlx",
36
+ "model": "mlx-community/multilingual-e5-base-mlx",
37
+ "hf_repo_id": "mlx-community/multilingual-e5-base-mlx",
38
+ "download_gb": 1.1,
39
+ "dimensions": 768,
40
+ "grade": "production",
41
+ "family": "local",
42
+ "label": "Multilingual E5 Base (로컬)",
43
+ "detail": "더 정확한 다국어 임베딩. 768차원이라 기존 색인은 다시 만들어야 합니다.",
44
+ },
45
+ "local:arctic-embed-l-v2": {
46
+ "id": "local:arctic-embed-l-v2",
47
+ "provider": "mlx",
48
+ "model": "mlx-community/snowflake-arctic-embed-l-v2.0-8bit",
49
+ "hf_repo_id": "mlx-community/snowflake-arctic-embed-l-v2.0-8bit",
50
+ "download_gb": 0.6,
51
+ "dimensions": 1024,
52
+ "grade": "production",
53
+ "family": "local",
54
+ "label": "Arctic Embed L v2 (로컬)",
55
+ "detail": "다국어 고품질 검색용 임베딩. 1024차원.",
56
+ },
13
57
  "local:bge-m3": {
14
58
  "id": "local:bge-m3",
15
59
  "provider": "mlx",
@@ -57,7 +57,32 @@ def _as_float_list(value: Any) -> List[Any]:
57
57
  return list(value)
58
58
 
59
59
 
60
+ #: Longest token sequence an encoder-family embedding model accepts. BERT and
61
+ #: XLM-R both stop at 512 including the two special tokens; a longer sequence
62
+ #: is a hard failure inside the model, not a slow answer, so the ids are cut
63
+ #: here rather than discovered at inference time.
64
+ MLX_MAX_TOKENS = 512
65
+
66
+ #: The E5 family was trained with these literal prefixes and loses accuracy
67
+ #: without them. Keyed on the ``kind`` ``POST /worker/embed`` already sends.
68
+ E5_PREFIXES: Dict[str, str] = {"query": "query: ", "passage": "passage: "}
69
+
70
+
71
+ def _wants_e5_prefix(model: str) -> bool:
72
+ """Whether this model id names an E5-family checkpoint."""
73
+ return "e5" in str(model or "").lower()
74
+
75
+
60
76
  class MLXEmbeddingProvider(_NetworkEmbeddingProvider):
77
+ """A local Apple-Silicon sentence embedder through ``mlx_embeddings``.
78
+
79
+ Nothing here reaches the network: the model must already be in the local
80
+ Hugging Face cache (that is exactly what
81
+ :func:`~.autodetect.detect_local_mlx` looks for). A missing package or a
82
+ missing snapshot is :class:`EmbeddingUnavailable`, which
83
+ :func:`resolve_embedder` turns into the honest hash fallback.
84
+ """
85
+
61
86
  provider = "mlx"
62
87
 
63
88
  def __init__(self, cfg: _RemoteConfig):
@@ -66,6 +91,7 @@ class MLXEmbeddingProvider(_NetworkEmbeddingProvider):
66
91
  self.dim = _guess_dim(cfg.model, DEFAULT_EMBEDDING_DIM)
67
92
  self.model_id = f"mlx:{cfg.model}:{self.dim}"
68
93
  self._encoder: Optional[Tuple[str, Any, Any]] = None
94
+ self._prefixed = _wants_e5_prefix(cfg.model)
69
95
 
70
96
  def _load(self):
71
97
  if self._encoder is not None:
@@ -79,25 +105,37 @@ class MLXEmbeddingProvider(_NetworkEmbeddingProvider):
79
105
  except Exception as exc: # pragma: no cover - environment dependent
80
106
  raise EmbeddingUnavailable(f"MLX embedding model unavailable: {exc}") from exc
81
107
 
108
+ def embed_batch_for(
109
+ self, texts: Sequence[str], kind: str = "passage"
110
+ ) -> List[List[float]]:
111
+ """E5 asymmetry, honoured. Non-E5 models see the text unchanged."""
112
+ if not self._prefixed:
113
+ return self.embed_batch(texts)
114
+ prefix = E5_PREFIXES.get(str(kind or "passage").lower(), E5_PREFIXES["passage"])
115
+ return self.embed_batch([f"{prefix}{text}" for text in texts])
116
+
82
117
  def _embed_raw(self, texts: Sequence[str]) -> List[List[float]]:
83
- kind, model, tokenizer = self._load()
118
+ _, model, tokenizer = self._load()
84
119
  try:
85
120
  import mlx.core as mx # type: ignore
86
121
 
87
122
  out: List[List[float]] = []
88
123
  for text in texts:
89
- ids = tokenizer.encode(text)
90
- tokens = mx.array([ids])
91
- result = model(tokens)
92
- pooled = result[0] if isinstance(result, (tuple, list)) else result
93
- vec = mx.mean(pooled, axis=1)[0] if pooled.ndim == 3 else pooled[0]
94
- out.append([float(x) for x in _as_float_list(vec.tolist())])
124
+ ids = list(tokenizer.encode(text))[:MLX_MAX_TOKENS]
125
+ result = model(mx.array([ids]))
126
+ out.append([float(x) for x in _pool(mx, result)])
95
127
  return out
96
128
  except EmbeddingUnavailable:
97
129
  raise
98
130
  except Exception as exc: # pragma: no cover - environment dependent
99
131
  raise EmbeddingUnavailable(f"MLX embedding failed: {exc}") from exc
100
132
 
133
+ def metadata(self) -> Dict[str, Any]:
134
+ info = super().metadata()
135
+ info["instruction_prefixes"] = self._prefixed
136
+ info["max_tokens"] = MLX_MAX_TOKENS
137
+ return info
138
+
101
139
  def health(self) -> Dict[str, Any]:
102
140
  try:
103
141
  self._load()
@@ -106,6 +144,27 @@ class MLXEmbeddingProvider(_NetworkEmbeddingProvider):
106
144
  return {"status": "unavailable", "detail": str(exc)}
107
145
 
108
146
 
147
+ def _pool(mx: Any, result: Any) -> List[Any]:
148
+ """One sentence vector out of whatever shape the model handed back.
149
+
150
+ ``mlx_embeddings`` models return an output object carrying a pooled
151
+ ``text_embeds`` on newer versions and a bare hidden-state tensor on older
152
+ ones. Preferring the model's own pooled output matters: a checkpoint with a
153
+ trained pooler produces a better vector than a mean over its last layer,
154
+ and mean-pooling on top of an already-pooled row would be nonsense.
155
+ """
156
+ for attribute in ("text_embeds", "pooler_output"):
157
+ pooled = getattr(result, attribute, None)
158
+ if pooled is not None:
159
+ row = pooled[0] if pooled.ndim > 1 else pooled
160
+ return _as_float_list(row.tolist())
161
+ hidden = getattr(result, "last_hidden_state", None)
162
+ if hidden is None:
163
+ hidden = result[0] if isinstance(result, (tuple, list)) else result
164
+ row = mx.mean(hidden, axis=1)[0] if hidden.ndim == 3 else hidden[0]
165
+ return _as_float_list(row.tolist())
166
+
167
+
109
168
  class OllamaEmbeddingProvider(_NetworkEmbeddingProvider):
110
169
  provider = "ollama"
111
170
 
@@ -285,9 +344,13 @@ class ResolvedEmbedder:
285
344
  fell_back: bool
286
345
  health: Dict[str, Any]
287
346
  detail: str = ""
347
+ #: What :mod:`.autodetect` found on this machine, if anyone looked. It is
348
+ #: reported whether or not it was adopted, so a user running on the hash
349
+ #: fallback with a real embedder already downloaded can *see* that.
350
+ detected: Optional[Any] = None
288
351
 
289
352
  def as_dict(self) -> Dict[str, Any]:
290
- return {
353
+ payload: Dict[str, Any] = {
291
354
  "requested_provider": self.requested,
292
355
  "active_provider": self.active,
293
356
  "fell_back": self.fell_back,
@@ -295,6 +358,9 @@ class ResolvedEmbedder:
295
358
  "detail": self.detail,
296
359
  **self.provider.metadata(),
297
360
  }
361
+ if self.detected is not None and hasattr(self.detected, "as_dict"):
362
+ payload["detected"] = self.detected.as_dict()
363
+ return payload
298
364
 
299
365
 
300
366
  def resolve_embedder(
@@ -0,0 +1,61 @@
1
+ """Search-side vector index backends (HNSW append + query).
2
+
3
+ ``POST /worker/vector/query`` serves this sidecar. ``lattice-retrieval``
4
+ asks for top-k candidates when ``LATTICEAI_VECTOR_INDEX=hnsw`` and
5
+ re-scores them exactly. Default env stays the brute scan.
6
+
7
+ This package is the Python HNSW sidecar: a disposable ``.hnsw`` graph
8
+ next to the brain database, with incremental ``add_items`` so a write no
9
+ longer rebuilds the whole graph.
10
+
11
+ Rebuild (not append) when:
12
+
13
+ * the embedder identity (``model_id`` / dim / space) changed
14
+ * any previously indexed id is absent (deletion). Tombstones exist in
15
+ hnswlib (``mark_deleted``) but a deletion-heavy store (more than
16
+ :data:`DELETE_REBUILD_RATIO` of the graph, or any delete when the
17
+ sidecar was loaded without source vectors) is cheaper to rebuild
18
+ than to fragment.
19
+ """
20
+
21
+ from .hnsw import (
22
+ DELETE_REBUILD_RATIO,
23
+ HNSW_BACKEND,
24
+ HNSW_META_SUFFIX,
25
+ HNSW_SUFFIX,
26
+ HNSWLIB_MODULE,
27
+ HnswIndex,
28
+ hnswlib_available,
29
+ load_hnswlib,
30
+ sidecar_paths,
31
+ )
32
+ from .sidecar import (
33
+ GRAPH_DB_NAME,
34
+ VECTOR_QUERY_K_CAP,
35
+ decode_f32le,
36
+ query_sidecar,
37
+ reset_sidecar_cache,
38
+ resolve_graph_db,
39
+ sidecar_fingerprint,
40
+ sidecar_freshness,
41
+ )
42
+
43
+ __all__ = [
44
+ "DELETE_REBUILD_RATIO",
45
+ "GRAPH_DB_NAME",
46
+ "HNSW_BACKEND",
47
+ "HNSW_META_SUFFIX",
48
+ "HNSW_SUFFIX",
49
+ "HNSWLIB_MODULE",
50
+ "HnswIndex",
51
+ "VECTOR_QUERY_K_CAP",
52
+ "decode_f32le",
53
+ "hnswlib_available",
54
+ "load_hnswlib",
55
+ "query_sidecar",
56
+ "reset_sidecar_cache",
57
+ "resolve_graph_db",
58
+ "sidecar_fingerprint",
59
+ "sidecar_freshness",
60
+ "sidecar_paths",
61
+ ]
@@ -0,0 +1,383 @@
1
+ """Approximate nearest-neighbour index with incremental append.
2
+
3
+ ``hnswlib`` is optional (``pip install hnswlib``; the published
4
+ ``ltcai[hnsw]`` extra was retired in 11.6.0 when the write path moved to
5
+ Rust). When the compiled module is missing the import is caught here and
6
+ reported as a reason string; search then returns no ANN hits rather than
7
+ pretending.
8
+
9
+ Policy, also stated on :meth:`HnswIndex.add_items`:
10
+
11
+ * **Append** new ids onto the live graph (``resize_index`` + ``add_items``).
12
+ * **Full rebuild** on provider/dim/space change, on any deletion when the
13
+ graph was loaded from a sidecar without source vectors, or when deletes
14
+ exceed :data:`DELETE_REBUILD_RATIO` of the current size.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import json
20
+ from pathlib import Path
21
+ from typing import Any, Dict, Iterable, List, Mapping, Optional, Sequence, Tuple
22
+
23
+ HNSW_BACKEND = "hnsw"
24
+ HNSWLIB_MODULE = "hnswlib"
25
+ HNSW_SUFFIX = ".hnsw"
26
+ HNSW_META_SUFFIX = ".hnsw.meta.json"
27
+ #: Rebuild instead of tombstoning once this fraction of the graph would go.
28
+ DELETE_REBUILD_RATIO = 0.10
29
+
30
+ IndexItem = Tuple[str, Sequence[float], Mapping[str, Any]]
31
+ ScoredId = Tuple[str, float]
32
+
33
+
34
+ def load_hnswlib() -> Tuple[Optional[Any], Optional[str]]:
35
+ """Import ``hnswlib`` or explain why it is unavailable (never raises)."""
36
+ try:
37
+ import hnswlib # noqa: PLC0415 — guarded, optional, import-on-demand
38
+ except Exception as exc: # noqa: BLE001 — any import failure is "no ANN"
39
+ return None, (
40
+ f"{HNSWLIB_MODULE} is not available ({exc}); install it with "
41
+ "`pip install hnswlib` to use the approximate index"
42
+ )
43
+ return hnswlib, None
44
+
45
+
46
+ def hnswlib_available() -> bool:
47
+ """True when the optional ANN engine can actually be imported."""
48
+ module, _ = load_hnswlib()
49
+ return module is not None
50
+
51
+
52
+ def sidecar_paths(db_path: Any) -> Tuple[Path, Path]:
53
+ """``(index file, meta file)`` for a brain database path."""
54
+ base = Path(db_path)
55
+ return (
56
+ base.with_suffix(HNSW_SUFFIX),
57
+ base.with_suffix(HNSW_META_SUFFIX),
58
+ )
59
+
60
+
61
+ class HnswIndex:
62
+ """Small-M HNSW graph over the brain's embeddings."""
63
+
64
+ def __init__(
65
+ self,
66
+ *,
67
+ dim: int,
68
+ space: str = "cosine",
69
+ ef_construction: int = 400,
70
+ m: int = 32,
71
+ ef_search: int = 400,
72
+ model_id: str = "",
73
+ ) -> None:
74
+ self._dim = int(dim)
75
+ self._space = str(space)
76
+ self._ef_construction = int(ef_construction)
77
+ self._m = int(m)
78
+ self._ef_search = int(ef_search)
79
+ self._model_id = str(model_id)
80
+ self._module, self._detail = load_hnswlib()
81
+ self._vectors: Dict[str, List[float]] = {}
82
+ self._metadata: Dict[str, Dict[str, Any]] = {}
83
+ self._labels: List[str] = []
84
+ self._ann: Any = None
85
+ self._dirty = True
86
+ self._loaded = False
87
+
88
+ @property
89
+ def backend(self) -> str:
90
+ return HNSW_BACKEND
91
+
92
+ @property
93
+ def approx(self) -> bool:
94
+ return True
95
+
96
+ @property
97
+ def exhaustive(self) -> bool:
98
+ return False
99
+
100
+ @property
101
+ def available(self) -> bool:
102
+ return self._module is not None
103
+
104
+ @property
105
+ def unavailable_detail(self) -> Optional[str]:
106
+ return self._detail
107
+
108
+ @property
109
+ def loaded_from_sidecar(self) -> bool:
110
+ """True when the graph came off disk instead of being rebuilt."""
111
+ return self._loaded
112
+
113
+ @property
114
+ def dim(self) -> int:
115
+ return self._dim
116
+
117
+ @property
118
+ def model_id(self) -> str:
119
+ return self._model_id
120
+
121
+ def _identity_matches(self, dim: int, space: str, model_id: str) -> bool:
122
+ return (
123
+ int(dim) == self._dim
124
+ and str(space) == self._space
125
+ and str(model_id) == self._model_id
126
+ )
127
+
128
+ def add(
129
+ self,
130
+ id: str,
131
+ vector: Sequence[float],
132
+ metadata: Optional[Mapping[str, Any]] = None,
133
+ ) -> None:
134
+ if self._loaded:
135
+ raise RuntimeError(
136
+ "this HnswIndex was loaded from a sidecar and holds no source "
137
+ "vectors; rebuild it from vector_embeddings before mutating"
138
+ )
139
+ key = str(id)
140
+ self._vectors[key] = [float(value) for value in vector]
141
+ self._metadata[key] = dict(metadata or {})
142
+ self._dirty = True
143
+
144
+ def remove(self, id: str) -> None:
145
+ if self._loaded:
146
+ raise RuntimeError(
147
+ "this HnswIndex was loaded from a sidecar and holds no source "
148
+ "vectors; rebuild it from vector_embeddings before mutating"
149
+ )
150
+ key = str(id)
151
+ self._vectors.pop(key, None)
152
+ self._metadata.pop(key, None)
153
+ self._dirty = True
154
+
155
+ def rebuild(self, items: Iterable[IndexItem]) -> None:
156
+ """Replace the held set and mark the graph for a full rebuild."""
157
+ self._vectors.clear()
158
+ self._metadata.clear()
159
+ self._labels.clear()
160
+ self._ann = None
161
+ self._loaded = False
162
+ for item_id, vector, metadata in items:
163
+ self.add(item_id, vector, metadata)
164
+ self._dirty = True
165
+
166
+ def add_items(
167
+ self,
168
+ items: Iterable[IndexItem],
169
+ *,
170
+ dim: Optional[int] = None,
171
+ space: Optional[str] = None,
172
+ model_id: Optional[str] = None,
173
+ ) -> str:
174
+ """Append new vectors, or rebuild when the policy says so.
175
+
176
+ Returns ``"append"`` or ``"rebuild"`` so a bench can time the two
177
+ doors without guessing.
178
+
179
+ Full rebuild when the embedder identity changed, when any currently
180
+ indexed id is missing from the incoming set (a deletion), when
181
+ deletes would exceed :data:`DELETE_REBUILD_RATIO`, or when this
182
+ instance was loaded from a sidecar and cannot append (no source
183
+ vectors to keep the label map honest).
184
+ """
185
+ incoming: Dict[str, IndexItem] = {}
186
+ for item_id, vector, metadata in items:
187
+ incoming[str(item_id)] = (str(item_id), vector, metadata)
188
+
189
+ want_dim = self._dim if dim is None else int(dim)
190
+ want_space = self._space if space is None else str(space)
191
+ want_model = self._model_id if model_id is None else str(model_id)
192
+ if not self._identity_matches(want_dim, want_space, want_model):
193
+ self._dim = want_dim
194
+ self._space = want_space
195
+ self._model_id = want_model
196
+ self.rebuild(incoming.values())
197
+ return "rebuild"
198
+
199
+ if self._loaded:
200
+ self.rebuild(incoming.values())
201
+ return "rebuild"
202
+
203
+ current = set(self._vectors)
204
+ incoming_ids = set(incoming)
205
+ deleted = current - incoming_ids
206
+ if deleted:
207
+ ratio = len(deleted) / max(1, len(current))
208
+ if ratio >= DELETE_REBUILD_RATIO or not current:
209
+ self.rebuild(incoming.values())
210
+ return "rebuild"
211
+ # Sparse deletes: drop the gone ids and rebuild so we never
212
+ # leave a tombstoned hole the sidecar cannot describe.
213
+ self.rebuild(incoming.values())
214
+ return "rebuild"
215
+
216
+ new_ids = [item_id for item_id in incoming if item_id not in current]
217
+ if not new_ids and not self._dirty and self._ann is not None:
218
+ return "append"
219
+ for item_id in new_ids:
220
+ _, vector, metadata = incoming[item_id]
221
+ self.add(item_id, vector, metadata)
222
+ if self._ann is not None and new_ids and self._module is not None:
223
+ self._append_live(new_ids)
224
+ return "append"
225
+ self._dirty = True
226
+ return "append" if new_ids or self._ann is not None else "rebuild"
227
+
228
+ def _new_graph(self, module: Any, capacity: int) -> Any:
229
+ graph = module.Index(space=self._space, dim=self._dim)
230
+ graph.init_index(
231
+ max_elements=max(1, capacity),
232
+ ef_construction=self._ef_construction,
233
+ M=self._m,
234
+ )
235
+ return graph
236
+
237
+ def _append_live(self, new_ids: Sequence[str]) -> None:
238
+ """``hnswlib`` ``resize_index`` + ``add_items`` for ``new_ids`` only."""
239
+ module = self._module
240
+ graph = self._ann
241
+ if module is None or graph is None:
242
+ self._dirty = True
243
+ return
244
+ start = len(self._labels)
245
+ needed = start + len(new_ids)
246
+ try:
247
+ current_max = int(graph.get_max_elements())
248
+ except Exception: # noqa: BLE001 — some builds expose no getter
249
+ current_max = start
250
+ if needed > current_max:
251
+ graph.resize_index(max(needed, current_max * 2 if current_max else needed))
252
+ graph.add_items(
253
+ [self._vectors[key] for key in new_ids],
254
+ list(range(start, start + len(new_ids))),
255
+ )
256
+ self._labels.extend(new_ids)
257
+ self._dirty = False
258
+
259
+ def _ensure_graph(self) -> Any:
260
+ """Build the graph if the held vectors changed (None when disabled)."""
261
+ module = self._module
262
+ if module is None:
263
+ return None
264
+ if self._ann is not None and not self._dirty:
265
+ return self._ann
266
+ labels = list(self._vectors)
267
+ graph = self._new_graph(module, max(len(labels), 1))
268
+ if labels:
269
+ graph.add_items(
270
+ [self._vectors[key] for key in labels],
271
+ list(range(len(labels))),
272
+ )
273
+ self._ann = graph
274
+ self._labels = labels
275
+ self._dirty = False
276
+ return graph
277
+
278
+ def search(
279
+ self,
280
+ query: Sequence[float],
281
+ top_k: int,
282
+ filter: Optional[Mapping[str, Any]] = None,
283
+ ) -> List[ScoredId]:
284
+ graph = self._ensure_graph()
285
+ if graph is None or not self._labels:
286
+ return []
287
+ floor = float("-inf")
288
+ if filter and filter.get("min_score") is not None:
289
+ try:
290
+ floor = float(filter["min_score"])
291
+ except (TypeError, ValueError):
292
+ floor = float("-inf")
293
+ wanted = max(1, min(int(top_k), len(self._labels)))
294
+ graph.set_ef(max(self._ef_search, wanted))
295
+ labels, distances = graph.knn_query([list(query)], k=wanted)
296
+ scored: List[ScoredId] = []
297
+ for label, distance in zip(labels[0], distances[0], strict=True):
298
+ score = 1.0 - float(distance)
299
+ if score < floor:
300
+ continue
301
+ scored.append((self._labels[int(label)], score))
302
+ return scored
303
+
304
+ def stats(self) -> Dict[str, Any]:
305
+ return {
306
+ "backend": self.backend,
307
+ "size": len(self._labels) if self._loaded else len(self._vectors),
308
+ "dim": self._dim,
309
+ "approx": True,
310
+ "exhaustive": False,
311
+ "available": self.available,
312
+ "detail": self._detail,
313
+ "model_id": self._model_id,
314
+ }
315
+
316
+ def save(self, db_path: Any, *, fingerprint: str) -> bool:
317
+ """Write the graph + label map beside the brain database."""
318
+ graph = self._ensure_graph()
319
+ if graph is None or not self._labels:
320
+ return False
321
+ index_path, meta_path = sidecar_paths(db_path)
322
+ try:
323
+ index_path.parent.mkdir(parents=True, exist_ok=True)
324
+ graph.save_index(str(index_path))
325
+ meta_path.write_text(
326
+ json.dumps(
327
+ {
328
+ "fingerprint": str(fingerprint),
329
+ "dim": self._dim,
330
+ "space": self._space,
331
+ "model_id": self._model_id,
332
+ "labels": self._labels,
333
+ },
334
+ ensure_ascii=False,
335
+ ),
336
+ encoding="utf-8",
337
+ )
338
+ except Exception: # noqa: BLE001 — persistence is best-effort
339
+ return False
340
+ return True
341
+
342
+ def load(self, db_path: Any, *, fingerprint: str) -> bool:
343
+ """Adopt a sidecar graph when it provably matches ``fingerprint``."""
344
+ module = self._module
345
+ if module is None:
346
+ return False
347
+ index_path, meta_path = sidecar_paths(db_path)
348
+ try:
349
+ meta = json.loads(meta_path.read_text(encoding="utf-8"))
350
+ matches = str(meta["fingerprint"]) == str(fingerprint)
351
+ matches = matches and int(meta["dim"]) == self._dim
352
+ if meta.get("model_id") not in (None, "", self._model_id):
353
+ matches = False
354
+ labels = [str(label) for label in meta["labels"]]
355
+ except Exception: # noqa: BLE001 — absent/corrupt sidecar = rebuild
356
+ return False
357
+ if not (matches and labels):
358
+ return False
359
+ try:
360
+ graph = module.Index(space=self._space, dim=self._dim)
361
+ graph.load_index(str(index_path), max_elements=len(labels))
362
+ except Exception: # noqa: BLE001 — corrupt binary = rebuild
363
+ return False
364
+ self._ann = graph
365
+ self._labels = labels
366
+ self._vectors.clear()
367
+ self._metadata.clear()
368
+ self._dirty = False
369
+ self._loaded = True
370
+ return True
371
+
372
+
373
+ __all__ = [
374
+ "DELETE_REBUILD_RATIO",
375
+ "HNSW_BACKEND",
376
+ "HNSW_META_SUFFIX",
377
+ "HNSW_SUFFIX",
378
+ "HNSWLIB_MODULE",
379
+ "HnswIndex",
380
+ "hnswlib_available",
381
+ "load_hnswlib",
382
+ "sidecar_paths",
383
+ ]