ltcai 11.9.0 → 12.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +80 -59
- package/docs/CHANGELOG.md +92 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +267 -119
- package/docs/ENTERPRISE.md +1 -1
- package/docs/MULTI_AGENT_RUNTIME.md +4 -4
- package/docs/ONBOARDING.md +13 -3
- package/docs/OPERATIONS.md +13 -4
- package/docs/REALTIME_COLLABORATION.md +1 -1
- package/docs/ROADMAP.md +113 -0
- package/docs/TRUST_MODEL.md +16 -2
- package/docs/WHY_LATTICE.md +8 -2
- package/docs/WORKFLOW_DESIGNER.md +2 -2
- package/docs/kg-schema.md +51 -5
- package/docs/mcp-tools.md +17 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/extraction.py +459 -105
- package/lattice_brain/graph/_kg_common/normalize.py +305 -0
- package/lattice_brain/graph/_kg_common/patterns.py +275 -0
- package/lattice_brain/graph/_kg_common/relations.py +12 -3
- package/lattice_brain/graph/_kg_common/sections.py +107 -0
- package/lattice_brain/graph/_kg_constants.py +7 -0
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/agent_worker_seam.py +32 -0
- package/latticeai/api/worker_compute.py +78 -3
- package/latticeai/core/embedding_providers/__init__.py +16 -0
- package/latticeai/core/embedding_providers/autodetect.py +302 -0
- package/latticeai/core/embedding_providers/base.py +25 -0
- package/latticeai/core/embedding_providers/profiles.py +44 -0
- package/latticeai/core/embedding_providers/text.py +74 -8
- package/latticeai/core/vector_index/__init__.py +61 -0
- package/latticeai/core/vector_index/hnsw.py +383 -0
- package/latticeai/core/vector_index/sidecar.py +329 -0
- package/latticeai/models/router/generation.py +101 -22
- package/latticeai/models/router/loading.py +109 -4
- package/latticeai/runtime/brain_runtime.py +43 -9
- package/latticeai/runtime/build_phases/worker_profile.py +13 -4
- package/latticeai/services/architecture_readiness.py +2 -2
- package/latticeai/services/product_readiness.py +10 -5
- package/latticeai/services/search_service.py +7 -0
- package/latticeai/tools/__init__.py +6 -1
- package/latticeai/tools/documents.py +12 -0
- package/latticeai/tools/markup.py +152 -0
- package/package.json +2 -1
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/compose_openapi.py +2 -0
- package/scripts/openapi_route_families.json +7 -3
- package/scripts/publish_release.mjs +157 -0
- package/scripts/release_screen_claims.json +12 -0
- package/src-tauri/Cargo.lock +44 -10
- package/src-tauri/Cargo.toml +1 -1
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +47 -41
- package/static/app/assets/Act-Cf1L2709.js +2 -0
- package/static/app/assets/AdminConsole-DPAbLTYV.js +1 -0
- package/static/app/assets/Brain-DqamGrj-.js +2 -0
- package/static/app/assets/BrainHome-MHe2_RYs.js +2 -0
- package/static/app/assets/BrainSignals-CQPPfyyH.js +1 -0
- package/static/app/assets/Capture-DGdIH_Zc.js +1 -0
- package/static/app/assets/Chronicle-C-UlCJoJ.js +1 -0
- package/static/app/assets/CommandPalette-WNT4EqUX.js +1 -0
- package/static/app/assets/DigitalBrainExplorer-CEBH5Cwc.js +321 -0
- package/static/app/assets/Library-C6xd1dlf.js +1 -0
- package/static/app/assets/LivingBrain-BEk-0ohw.js +1 -0
- package/static/app/assets/ProductFlow-CZLm5iXh.js +1 -0
- package/static/app/assets/QueryClientProvider-B3OjqSyJ.js +1 -0
- package/static/app/assets/ReviewCard-CEHG6evf.js +3 -0
- package/static/app/assets/RunsListPanel-CLtEJSRW.js +1 -0
- package/static/app/assets/System-CAxwBUXw.js +1 -0
- package/static/app/assets/WorkflowGraph-Dj10RuGE.js +1 -0
- package/static/app/assets/WorkflowsPanel-Kyeh_LIT.js +2 -0
- package/static/app/assets/actHelpers-CtSmK9Dw.js +1 -0
- package/static/app/assets/arrow-left-CRl5EO4D.js +1 -0
- package/static/app/assets/{bot-Bc3Q27YR.js → bot-DhUGRel2.js} +1 -1
- package/static/app/assets/brain-CLkhHsHF.js +1 -0
- package/static/app/assets/button-CmaEqG1T.js +1 -0
- package/static/app/assets/circle-check-CFgejkOS.js +1 -0
- package/static/app/assets/{circle-pause-BGMiV8UU.js → circle-pause-l96izbxj.js} +1 -1
- package/static/app/assets/{circle-play-DoanLHnd.js → circle-play-CrZa25_q.js} +1 -1
- package/static/app/assets/{cpu-DwzNf82m.js → cpu-BaXudqwl.js} +1 -1
- package/static/app/assets/{download-Ddw49yCV.js → download-hCVFPiyc.js} +1 -1
- package/static/app/assets/{folder-open-Brd6Kvto.js → folder-open-CHL82Yp7.js} +1 -1
- package/static/app/assets/{hard-drive-Bu-DTJdB.js → hard-drive-DDzET7lk.js} +1 -1
- package/static/app/assets/{index-CGdg_aq9.css → index-CB93CZWW.css} +1 -1
- package/static/app/assets/index-D2H-wSl6.js +13 -0
- package/static/app/assets/input-Df1CAY_I.js +1 -0
- package/static/app/assets/jsx-runtime-bzQ4Vb5N.js +1 -0
- package/static/app/assets/{link-2-DZ4OA5tJ.js → link-2-xNnTIX1_.js} +1 -1
- package/static/app/assets/{permissionCopy-D9TR0F8b.js → permissionCopy-D3aWHco-.js} +1 -1
- package/static/app/assets/primitives-BioD2slS.js +1 -0
- package/static/app/assets/search-BzBw8YcW.js +1 -0
- package/static/app/assets/{share-2-BC5FirFv.js → share-2-FkzGf8Df.js} +1 -1
- package/static/app/assets/{shield-alert-DUbR2W2s.js → shield-alert-B3dwzik4.js} +1 -1
- package/static/app/assets/sourceMeta-DQSY_tah.js +1 -0
- package/static/app/assets/textarea-P8o6pvOP.js +1 -0
- package/static/app/assets/useFocusTrap-hswOIkXE.js +1 -0
- package/static/app/assets/useMutation-OJLrYSRA.js +1 -0
- package/static/app/assets/workspace-BCuk3Ku9.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/static/app/assets/Act-B4WT81kh.js +0 -1
- package/static/app/assets/AdminConsole--Wf71m-o.js +0 -1
- package/static/app/assets/Brain-CtYaa26c.js +0 -321
- package/static/app/assets/BrainHome-Be2VPJEc.js +0 -2
- package/static/app/assets/BrainSignals-C0__xgpG.js +0 -1
- package/static/app/assets/Capture-DdNi5Peb.js +0 -1
- package/static/app/assets/Chronicle-B3hNveeI.js +0 -1
- package/static/app/assets/CommandPalette-CIsnSsFL.js +0 -1
- package/static/app/assets/Library-Bl5XClFV.js +0 -1
- package/static/app/assets/LivingBrain-DjkTB_gh.js +0 -1
- package/static/app/assets/ProductFlow-DCUNRNHs.js +0 -1
- package/static/app/assets/ReviewCard-CD3yWvUB.js +0 -3
- package/static/app/assets/System-BJ6jQ_SL.js +0 -1
- package/static/app/assets/arrow-left-6_28Z0qH.js +0 -1
- package/static/app/assets/brain-B9BDrMTe.js +0 -1
- package/static/app/assets/button-D6JcpYcf.js +0 -1
- package/static/app/assets/circle-check-BFu9lD-3.js +0 -1
- package/static/app/assets/index-CWKRRsLW.js +0 -10
- package/static/app/assets/input-CEqsxtil.js +0 -1
- package/static/app/assets/primitives-DORg7Z_7.js +0 -1
- package/static/app/assets/search-0NQ21wXe.js +0 -1
- package/static/app/assets/textarea-jtQcRSXo.js +0 -1
- package/static/app/assets/useFocusTrap-HRemcWId.js +0 -1
- package/static/app/assets/useMutation-DqlFE-Bw.js +0 -1
- package/static/app/assets/useQuery-BizqBNGw.js +0 -1
- package/static/app/assets/utils-WgW4V69R.js +0 -4
- package/static/app/assets/workspace-DSek3jCY.js +0 -1
|
@@ -0,0 +1,329 @@
|
|
|
1
|
+
"""Load, refresh, and query the HNSW sidecar next to a brain database.
|
|
2
|
+
|
|
3
|
+
The live graph is cached per ``(db_path, model_id, dim)``. A missing sidecar
|
|
4
|
+
is ``index: "none"``. When the store has vectors the cache (or a rebuild from
|
|
5
|
+
``vector_embeddings``) can serve, the reply is ``index: "hnsw"``. Staleness is
|
|
6
|
+
sidecar size vs ``COUNT(*)`` for that identity — reported honestly, and
|
|
7
|
+
refreshed through :meth:`HnswIndex.add_items` so an ingest append does not
|
|
8
|
+
rebuild the whole graph.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import sqlite3
|
|
14
|
+
import struct
|
|
15
|
+
import threading
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import Any, Dict, List, Optional, Sequence, Tuple
|
|
18
|
+
|
|
19
|
+
from latticeai.core.config import default_data_dir
|
|
20
|
+
|
|
21
|
+
from .hnsw import HnswIndex, hnswlib_available, sidecar_paths
|
|
22
|
+
|
|
23
|
+
GRAPH_DB_NAME = "knowledge_graph.sqlite"
|
|
24
|
+
VECTOR_QUERY_K_CAP = 200
|
|
25
|
+
|
|
26
|
+
_CACHE: Dict[Tuple[str, str, int], HnswIndex] = {}
|
|
27
|
+
_LOCK = threading.Lock()
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def resolve_graph_db(workspace: Optional[str] = None) -> Path:
|
|
31
|
+
"""Brain sqlite path: explicit file, a data dir, or ``LATTICEAI_DATA_DIR``."""
|
|
32
|
+
if workspace:
|
|
33
|
+
raw = Path(str(workspace)).expanduser()
|
|
34
|
+
if raw.is_file():
|
|
35
|
+
return raw
|
|
36
|
+
if raw.suffix.lower() == ".sqlite":
|
|
37
|
+
return raw
|
|
38
|
+
return raw / GRAPH_DB_NAME
|
|
39
|
+
return default_data_dir() / GRAPH_DB_NAME
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def decode_f32le(blob: bytes, dim: int) -> List[float]:
|
|
43
|
+
"""Little-endian f32 payload — the same layout Rust ``encode`` writes."""
|
|
44
|
+
if not blob:
|
|
45
|
+
return []
|
|
46
|
+
count = dim if dim > 0 else len(blob) // 4
|
|
47
|
+
if len(blob) != count * 4:
|
|
48
|
+
count = len(blob) // 4
|
|
49
|
+
if count <= 0:
|
|
50
|
+
return []
|
|
51
|
+
return list(struct.unpack(f"<{count}f", blob[: count * 4]))
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def sidecar_fingerprint(model_id: str, dim: int, size: int) -> str:
|
|
55
|
+
"""Identity the ``.hnsw.meta.json`` file is keyed on."""
|
|
56
|
+
return f"{model_id}|{int(dim)}|{int(size)}"
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def store_vector_count(conn: sqlite3.Connection, model_id: str, dim: int) -> int:
|
|
60
|
+
row = conn.execute(
|
|
61
|
+
"SELECT COUNT(*) FROM vector_embeddings "
|
|
62
|
+
"WHERE embedding_model=? AND embedding_dim=?",
|
|
63
|
+
(model_id, int(dim)),
|
|
64
|
+
).fetchone()
|
|
65
|
+
return int(row[0]) if row else 0
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def load_store_items(
|
|
69
|
+
conn: sqlite3.Connection, model_id: str, dim: int
|
|
70
|
+
) -> List[Tuple[str, List[float], Dict[str, Any]]]:
|
|
71
|
+
items: List[Tuple[str, List[float], Dict[str, Any]]] = []
|
|
72
|
+
for item_id, blob in conn.execute(
|
|
73
|
+
"SELECT item_id, embedding FROM vector_embeddings "
|
|
74
|
+
"WHERE embedding_model=? AND embedding_dim=? ORDER BY indexed_at ASC, item_id ASC",
|
|
75
|
+
(model_id, int(dim)),
|
|
76
|
+
):
|
|
77
|
+
vector = decode_f32le(bytes(blob or b""), int(dim))
|
|
78
|
+
if len(vector) != int(dim):
|
|
79
|
+
continue
|
|
80
|
+
items.append((str(item_id), vector, {}))
|
|
81
|
+
return items
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def sidecar_meta_size(db_path: Path) -> Optional[int]:
|
|
85
|
+
"""Label count from the sidecar meta file, or ``None`` when it is absent."""
|
|
86
|
+
_, meta_path = sidecar_paths(db_path)
|
|
87
|
+
try:
|
|
88
|
+
import json
|
|
89
|
+
|
|
90
|
+
meta = json.loads(meta_path.read_text(encoding="utf-8"))
|
|
91
|
+
labels = meta.get("labels") or []
|
|
92
|
+
return len(labels)
|
|
93
|
+
except Exception: # noqa: BLE001 — absent/corrupt meta is "no sidecar"
|
|
94
|
+
return None
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _open_readonly(db_path: Path) -> Optional[sqlite3.Connection]:
|
|
98
|
+
if not db_path.is_file():
|
|
99
|
+
return None
|
|
100
|
+
try:
|
|
101
|
+
conn = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True)
|
|
102
|
+
except sqlite3.Error:
|
|
103
|
+
try:
|
|
104
|
+
conn = sqlite3.connect(str(db_path))
|
|
105
|
+
except sqlite3.Error:
|
|
106
|
+
return None
|
|
107
|
+
conn.row_factory = sqlite3.Row
|
|
108
|
+
return conn
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _none_reply(
|
|
112
|
+
*,
|
|
113
|
+
size: int = 0,
|
|
114
|
+
store_size: int = 0,
|
|
115
|
+
stale: bool = False,
|
|
116
|
+
detail: Optional[str] = None,
|
|
117
|
+
) -> Dict[str, Any]:
|
|
118
|
+
return {
|
|
119
|
+
"ids": [],
|
|
120
|
+
"scores": [],
|
|
121
|
+
"index": "none",
|
|
122
|
+
"size": int(size),
|
|
123
|
+
"store_size": int(store_size),
|
|
124
|
+
"stale": bool(stale),
|
|
125
|
+
"detail": detail,
|
|
126
|
+
"refreshed": None,
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _hnsw_reply(
|
|
131
|
+
ids: Sequence[str],
|
|
132
|
+
scores: Sequence[float],
|
|
133
|
+
*,
|
|
134
|
+
size: int,
|
|
135
|
+
store_size: int,
|
|
136
|
+
stale: bool,
|
|
137
|
+
detail: Optional[str],
|
|
138
|
+
refreshed: Optional[str],
|
|
139
|
+
) -> Dict[str, Any]:
|
|
140
|
+
return {
|
|
141
|
+
"ids": [str(item) for item in ids],
|
|
142
|
+
"scores": [float(score) for score in scores],
|
|
143
|
+
"index": "hnsw",
|
|
144
|
+
"size": int(size),
|
|
145
|
+
"store_size": int(store_size),
|
|
146
|
+
"stale": bool(stale),
|
|
147
|
+
"detail": detail,
|
|
148
|
+
"refreshed": refreshed,
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _refresh_index(
|
|
153
|
+
index: HnswIndex,
|
|
154
|
+
items: Sequence[Tuple[str, List[float], Dict[str, Any]]],
|
|
155
|
+
model_id: str,
|
|
156
|
+
dim: int,
|
|
157
|
+
) -> str:
|
|
158
|
+
return index.add_items(items, dim=dim, model_id=model_id)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def query_sidecar(
|
|
162
|
+
*,
|
|
163
|
+
workspace: Optional[str],
|
|
164
|
+
embedding_model: str,
|
|
165
|
+
embedding_dim: int,
|
|
166
|
+
vector: Sequence[float],
|
|
167
|
+
k: int,
|
|
168
|
+
db_path: Optional[Path] = None,
|
|
169
|
+
) -> Dict[str, Any]:
|
|
170
|
+
"""Serve one ANN query from the sidecar (or say the index is absent)."""
|
|
171
|
+
dim = int(embedding_dim)
|
|
172
|
+
model_id = str(embedding_model or "")
|
|
173
|
+
wanted = max(1, min(int(k), VECTOR_QUERY_K_CAP))
|
|
174
|
+
path = Path(db_path) if db_path is not None else resolve_graph_db(workspace)
|
|
175
|
+
if not hnswlib_available():
|
|
176
|
+
return _none_reply(detail="hnswlib is not available in this worker")
|
|
177
|
+
if dim <= 0 or len(vector) != dim:
|
|
178
|
+
return _none_reply(
|
|
179
|
+
detail=(
|
|
180
|
+
f"query vector width {len(vector)} does not match embedding_dim {dim}"
|
|
181
|
+
if dim > 0
|
|
182
|
+
else "embedding_dim must be a positive integer"
|
|
183
|
+
)
|
|
184
|
+
)
|
|
185
|
+
if not model_id:
|
|
186
|
+
return _none_reply(detail="embedding_model is required")
|
|
187
|
+
|
|
188
|
+
conn = _open_readonly(path)
|
|
189
|
+
store_size = 0
|
|
190
|
+
items: List[Tuple[str, List[float], Dict[str, Any]]] = []
|
|
191
|
+
if conn is not None:
|
|
192
|
+
try:
|
|
193
|
+
store_size = store_vector_count(conn, model_id, dim)
|
|
194
|
+
if store_size > 0:
|
|
195
|
+
items = load_store_items(conn, model_id, dim)
|
|
196
|
+
store_size = len(items)
|
|
197
|
+
finally:
|
|
198
|
+
conn.close()
|
|
199
|
+
|
|
200
|
+
key = (str(path), model_id, dim)
|
|
201
|
+
with _LOCK:
|
|
202
|
+
index = _CACHE.get(key)
|
|
203
|
+
refreshed: Optional[str] = None
|
|
204
|
+
if index is None:
|
|
205
|
+
index = HnswIndex(dim=dim, model_id=model_id)
|
|
206
|
+
loaded = index.load(path, fingerprint=sidecar_fingerprint(model_id, dim, store_size))
|
|
207
|
+
if not loaded:
|
|
208
|
+
if not items:
|
|
209
|
+
return _none_reply(
|
|
210
|
+
store_size=store_size,
|
|
211
|
+
detail="no HNSW sidecar for this embedding identity",
|
|
212
|
+
)
|
|
213
|
+
refreshed = _refresh_index(index, items, model_id, dim)
|
|
214
|
+
index.save(path, fingerprint=sidecar_fingerprint(model_id, dim, len(items)))
|
|
215
|
+
_CACHE[key] = index
|
|
216
|
+
|
|
217
|
+
cached_size = int(index.stats().get("size") or 0)
|
|
218
|
+
stale = cached_size != store_size
|
|
219
|
+
if stale:
|
|
220
|
+
if not items and store_size > 0:
|
|
221
|
+
conn = _open_readonly(path)
|
|
222
|
+
if conn is not None:
|
|
223
|
+
try:
|
|
224
|
+
items = load_store_items(conn, model_id, dim)
|
|
225
|
+
finally:
|
|
226
|
+
conn.close()
|
|
227
|
+
if items:
|
|
228
|
+
refreshed = _refresh_index(index, items, model_id, dim)
|
|
229
|
+
index.save(path, fingerprint=sidecar_fingerprint(model_id, dim, len(items)))
|
|
230
|
+
cached_size = int(index.stats().get("size") or 0)
|
|
231
|
+
stale = cached_size != len(items)
|
|
232
|
+
elif store_size == 0:
|
|
233
|
+
_CACHE.pop(key, None)
|
|
234
|
+
return _none_reply(
|
|
235
|
+
store_size=0,
|
|
236
|
+
detail="vector store is empty for this embedding identity",
|
|
237
|
+
)
|
|
238
|
+
|
|
239
|
+
if cached_size <= 0:
|
|
240
|
+
return _none_reply(
|
|
241
|
+
store_size=store_size,
|
|
242
|
+
stale=stale,
|
|
243
|
+
detail="HNSW sidecar has no labels for this identity",
|
|
244
|
+
)
|
|
245
|
+
|
|
246
|
+
index._ef_search = max(index._ef_search, wanted * 2, 200)
|
|
247
|
+
hits = index.search(list(vector), top_k=min(wanted, cached_size))
|
|
248
|
+
ids = [item_id for item_id, _ in hits]
|
|
249
|
+
scores = [score for _, score in hits]
|
|
250
|
+
detail = None
|
|
251
|
+
if stale:
|
|
252
|
+
detail = (
|
|
253
|
+
f"sidecar size {cached_size} != store count {store_size} "
|
|
254
|
+
"for this embedding identity"
|
|
255
|
+
)
|
|
256
|
+
return _hnsw_reply(
|
|
257
|
+
ids,
|
|
258
|
+
scores,
|
|
259
|
+
size=cached_size,
|
|
260
|
+
store_size=store_size,
|
|
261
|
+
stale=stale,
|
|
262
|
+
detail=detail,
|
|
263
|
+
refreshed=refreshed,
|
|
264
|
+
)
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def sidecar_freshness(
|
|
268
|
+
*,
|
|
269
|
+
workspace: Optional[str] = None,
|
|
270
|
+
embedding_model: Optional[str] = None,
|
|
271
|
+
embedding_dim: Optional[int] = None,
|
|
272
|
+
db_path: Optional[Path] = None,
|
|
273
|
+
) -> Dict[str, Any]:
|
|
274
|
+
"""Cheap on-disk vs store comparison — no ANN query, no rebuild."""
|
|
275
|
+
path = Path(db_path) if db_path is not None else resolve_graph_db(workspace)
|
|
276
|
+
meta_size = sidecar_meta_size(path)
|
|
277
|
+
store_size = 0
|
|
278
|
+
conn = _open_readonly(path)
|
|
279
|
+
if conn is not None:
|
|
280
|
+
try:
|
|
281
|
+
if embedding_model and embedding_dim:
|
|
282
|
+
store_size = store_vector_count(conn, embedding_model, int(embedding_dim))
|
|
283
|
+
else:
|
|
284
|
+
row = conn.execute("SELECT COUNT(*) FROM vector_embeddings").fetchone()
|
|
285
|
+
store_size = int(row[0]) if row else 0
|
|
286
|
+
except sqlite3.Error:
|
|
287
|
+
store_size = 0
|
|
288
|
+
finally:
|
|
289
|
+
conn.close()
|
|
290
|
+
if meta_size is None:
|
|
291
|
+
return {
|
|
292
|
+
"index": "none",
|
|
293
|
+
"size": 0,
|
|
294
|
+
"store_size": store_size,
|
|
295
|
+
"stale": store_size > 0,
|
|
296
|
+
"detail": "no HNSW sidecar next to the brain database",
|
|
297
|
+
}
|
|
298
|
+
stale = meta_size != store_size
|
|
299
|
+
return {
|
|
300
|
+
"index": "hnsw",
|
|
301
|
+
"size": meta_size,
|
|
302
|
+
"store_size": store_size,
|
|
303
|
+
"stale": stale,
|
|
304
|
+
"detail": (
|
|
305
|
+
f"sidecar size {meta_size} != store count {store_size}"
|
|
306
|
+
if stale
|
|
307
|
+
else "sidecar matches the vector store"
|
|
308
|
+
),
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def reset_sidecar_cache() -> None:
|
|
313
|
+
"""Drop the process cache — tests only."""
|
|
314
|
+
with _LOCK:
|
|
315
|
+
_CACHE.clear()
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
__all__ = [
|
|
319
|
+
"GRAPH_DB_NAME",
|
|
320
|
+
"VECTOR_QUERY_K_CAP",
|
|
321
|
+
"decode_f32le",
|
|
322
|
+
"query_sidecar",
|
|
323
|
+
"reset_sidecar_cache",
|
|
324
|
+
"resolve_graph_db",
|
|
325
|
+
"sidecar_fingerprint",
|
|
326
|
+
"sidecar_freshness",
|
|
327
|
+
"sidecar_meta_size",
|
|
328
|
+
"store_vector_count",
|
|
329
|
+
]
|
|
@@ -18,9 +18,57 @@ from latticeai.core.quiet import quiet
|
|
|
18
18
|
|
|
19
19
|
from ._contract import RouterCore as _Core
|
|
20
20
|
from .branding import SYSTEM_PROMPT, _compose_system, normalize_branding
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _system_for(context, cite_sources: bool) -> str:
|
|
24
|
+
"""The system prompt for one completion.
|
|
25
|
+
|
|
26
|
+
``cite_sources`` is what tells the two kinds of **caller** apart, and
|
|
27
|
+
v12.0.0 added it because nothing did.
|
|
28
|
+
|
|
29
|
+
A *chat* caller sends retrieved passages, and gets this product's own
|
|
30
|
+
system prompt with the Context block and
|
|
31
|
+
:data:`~latticeai.models.router.branding.CITATION_INSTRUCTION` appended.
|
|
32
|
+
:func:`_compose_system` used to do that for any non-empty context, which is
|
|
33
|
+
right there and wrong for the agent seam, whose context is the loop's own
|
|
34
|
+
executor prompt: every agent turn was being told its instructions were
|
|
35
|
+
"retrieved sources" to "cite inline as [1], [2]". A large model ignores
|
|
36
|
+
that. A small one obeys it — the acid-test 0.5B wrote
|
|
37
|
+
``[1] 인사말을 쓸 예시 코드: …`` into a file, and a 2B answered a tool call
|
|
38
|
+
with the citation instruction itself.
|
|
39
|
+
|
|
40
|
+
A *worker* caller sends **its own whole prompt**, and gets exactly that.
|
|
41
|
+
Same fact, one flag: a context that is not a corpus is an instruction, and
|
|
42
|
+
an instruction the caller wrote is the only instruction that turn should
|
|
43
|
+
carry. Until now the chat persona was still prepended to it, so six lines
|
|
44
|
+
of "You are Lattice AI … You are a Vision-Language Model … Be concise" sat
|
|
45
|
+
in front of every guided micro-turn — including the one that asks a model to
|
|
46
|
+
write a file's contents. Small models answer the nearest instruction: a live
|
|
47
|
+
2B asked to summarise a README wrote "I am a local AI assistant that can run
|
|
48
|
+
on Apple Silicon" into the file, and a gemma-4-e2b opened two of its three
|
|
49
|
+
files with "Identity: Lattice AI (Vision-Language Model on Apple Silicon)".
|
|
50
|
+
Nothing leaked verbatim; the *subject* leaked, which is the same defect one
|
|
51
|
+
paraphrase further on. The document path has always worked this way — its
|
|
52
|
+
own system prompt replaces the chat identity entirely — so this is the
|
|
53
|
+
existing rule reaching the second caller that has a prompt of its own.
|
|
54
|
+
|
|
55
|
+
Both agent prompts still say who they are (``You are the executor of an
|
|
56
|
+
agent loop``) and every guided block still carries the run's own answer
|
|
57
|
+
language, so nothing the loop depends on is being removed — only a second,
|
|
58
|
+
competing identity. Branding normalisation is untouched: it runs over the
|
|
59
|
+
context here and over every generated string on the way out.
|
|
60
|
+
|
|
61
|
+
A worker call with **no** context still gets the product prompt. A bare
|
|
62
|
+
completion carrying no instruction at all is the one case where the identity
|
|
63
|
+
is the only thing there is to say.
|
|
64
|
+
"""
|
|
65
|
+
context = normalize_branding(context)
|
|
66
|
+
if cite_sources:
|
|
67
|
+
return _compose_system(SYSTEM_PROMPT, context)
|
|
68
|
+
return context or SYSTEM_PROMPT
|
|
21
69
|
from .catalog import CloudModel
|
|
22
70
|
from .errors import ModelStreamError, _stream_failure
|
|
23
|
-
from .loading import _mlx_sampler, apply_stop_strings, executor
|
|
71
|
+
from .loading import _mlx_sampler, apply_prefix, apply_stop_strings, executor
|
|
24
72
|
|
|
25
73
|
|
|
26
74
|
def _stream_until_stop(
|
|
@@ -65,9 +113,10 @@ def _stream_until_stop(
|
|
|
65
113
|
class _GenerationMixin(_Core):
|
|
66
114
|
"""The chat generation half of :class:`LLMRouter`."""
|
|
67
115
|
|
|
68
|
-
def _build_prompt(
|
|
69
|
-
context =
|
|
70
|
-
|
|
116
|
+
def _build_prompt(
|
|
117
|
+
self, message: str, context: Optional[str], tokenizer, cite_sources: bool = True
|
|
118
|
+
) -> str:
|
|
119
|
+
system = _system_for(context, cite_sources)
|
|
71
120
|
if hasattr(tokenizer, "apply_chat_template"):
|
|
72
121
|
try:
|
|
73
122
|
msgs = [{"role": "system", "content": system}, {"role": "user", "content": message}]
|
|
@@ -76,9 +125,8 @@ class _GenerationMixin(_Core):
|
|
|
76
125
|
quiet()
|
|
77
126
|
return f"<|im_start|>system\n{system}<|im_end|>\n<|im_start|>user\n{message}<|im_end|>\n<|im_start|>assistant\n"
|
|
78
127
|
|
|
79
|
-
def _build_vlm_prompt(self, model, processor, message: str, context: Optional[str], num_images: int) -> str:
|
|
80
|
-
|
|
81
|
-
system = _compose_system(SYSTEM_PROMPT, context)
|
|
128
|
+
def _build_vlm_prompt(self, model, processor, message: str, context: Optional[str], num_images: int, cite_sources: bool = True) -> str:
|
|
129
|
+
system = _system_for(context, cite_sources)
|
|
82
130
|
try:
|
|
83
131
|
from mlx_vlm import apply_chat_template
|
|
84
132
|
|
|
@@ -94,7 +142,7 @@ class _GenerationMixin(_Core):
|
|
|
94
142
|
)
|
|
95
143
|
except Exception as e:
|
|
96
144
|
print(f"⚠️ VLM chat template fallback: {e}")
|
|
97
|
-
return self._build_prompt(message, context, processor)
|
|
145
|
+
return self._build_prompt(message, context, processor, cite_sources)
|
|
98
146
|
|
|
99
147
|
async def generate_as(
|
|
100
148
|
self,
|
|
@@ -105,6 +153,8 @@ class _GenerationMixin(_Core):
|
|
|
105
153
|
temperature: float = 0.2,
|
|
106
154
|
image_data: Optional[str] = None,
|
|
107
155
|
stop: Optional[List[str]] = None,
|
|
156
|
+
prefix: Optional[str] = None,
|
|
157
|
+
cite_sources: bool = True,
|
|
108
158
|
) -> str:
|
|
109
159
|
"""Generate with a request-scoped model without changing the default.
|
|
110
160
|
|
|
@@ -113,12 +163,23 @@ class _GenerationMixin(_Core):
|
|
|
113
163
|
after the stop are never produced rather than merely trimmed; the cloud
|
|
114
164
|
path forwards the list to the provider, which stops server-side. Absent
|
|
115
165
|
or empty means what it always meant — generate to ``max_tokens``.
|
|
166
|
+
|
|
167
|
+
``prefix`` **starts** the reply (v12.0.0): the characters are put in the
|
|
168
|
+
model's mouth rather than requested of it, so a caller that needs the
|
|
169
|
+
answer to begin ``{"thoughts": "`` gets that by construction instead of
|
|
170
|
+
by asking nicely and repairing the result. Locally this is a real
|
|
171
|
+
prefill — the text is appended to the templated prompt after the
|
|
172
|
+
generation marker, so the model continues from mid-token rather than
|
|
173
|
+
starting a fresh turn. The returned text always begins with it; see
|
|
174
|
+
:func:`~latticeai.models.router.loading.apply_prefix` for how the three
|
|
175
|
+
backends are reconciled.
|
|
116
176
|
"""
|
|
117
177
|
_selected, cached = self._model_snapshot(model_id)
|
|
118
178
|
if cached is None:
|
|
119
179
|
return "No model."
|
|
120
180
|
return await self._generate_cached(
|
|
121
|
-
cached, message, context, max_tokens, temperature, image_data, stop
|
|
181
|
+
cached, message, context, max_tokens, temperature, image_data, stop, prefix,
|
|
182
|
+
cite_sources,
|
|
122
183
|
)
|
|
123
184
|
|
|
124
185
|
async def generate(
|
|
@@ -129,9 +190,12 @@ class _GenerationMixin(_Core):
|
|
|
129
190
|
temperature: float = 0.2,
|
|
130
191
|
image_data: Optional[str] = None,
|
|
131
192
|
stop: Optional[List[str]] = None,
|
|
193
|
+
prefix: Optional[str] = None,
|
|
194
|
+
cite_sources: bool = True,
|
|
132
195
|
) -> str:
|
|
133
196
|
return await self.generate_as(
|
|
134
|
-
None, message, context, max_tokens, temperature, image_data, stop
|
|
197
|
+
None, message, context, max_tokens, temperature, image_data, stop, prefix,
|
|
198
|
+
cite_sources,
|
|
135
199
|
)
|
|
136
200
|
|
|
137
201
|
async def _generate_cached(
|
|
@@ -143,19 +207,28 @@ class _GenerationMixin(_Core):
|
|
|
143
207
|
temperature: float,
|
|
144
208
|
image_data: Optional[str],
|
|
145
209
|
stop: Optional[List[str]] = None,
|
|
210
|
+
prefix: Optional[str] = None,
|
|
211
|
+
cite_sources: bool = True,
|
|
146
212
|
) -> str:
|
|
147
213
|
if isinstance(cached, CloudModel):
|
|
148
214
|
return await self._cloud_generate(
|
|
149
|
-
cached, message, context, max_tokens, temperature, stop
|
|
215
|
+
cached, message, context, max_tokens, temperature, stop, prefix, cite_sources
|
|
150
216
|
)
|
|
151
217
|
|
|
152
218
|
model, tokenizer, draft_model, loader_kind = self._unpack_local_cache(cached)
|
|
153
219
|
use_vlm = loader_kind == "mlx_vlm"
|
|
154
220
|
prompt = (
|
|
155
|
-
self._build_vlm_prompt(
|
|
221
|
+
self._build_vlm_prompt(
|
|
222
|
+
model, tokenizer, message, context, 1 if image_data else 0, cite_sources
|
|
223
|
+
)
|
|
156
224
|
if use_vlm
|
|
157
|
-
else self._build_prompt(message, context, tokenizer)
|
|
225
|
+
else self._build_prompt(message, context, tokenizer, cite_sources)
|
|
158
226
|
)
|
|
227
|
+
# The prefill. Appended *after* the chat template's generation marker,
|
|
228
|
+
# so the first sampled token continues these characters instead of
|
|
229
|
+
# opening a reply. Nothing else in the pipeline changes.
|
|
230
|
+
if prefix:
|
|
231
|
+
prompt = f"{prompt}{prefix}"
|
|
159
232
|
stops = [marker for marker in (stop or []) if marker]
|
|
160
233
|
|
|
161
234
|
loop = asyncio.get_event_loop()
|
|
@@ -181,27 +254,33 @@ class _GenerationMixin(_Core):
|
|
|
181
254
|
result = await loop.run_in_executor(executor, _gen)
|
|
182
255
|
# mlx-vlm might return a GenerationResult object; extract the text
|
|
183
256
|
text = result.text if hasattr(result, "text") else str(result)
|
|
184
|
-
return normalize_branding(apply_stop_strings(text, stops))
|
|
257
|
+
return apply_prefix(prefix, normalize_branding(apply_stop_strings(text, stops)))
|
|
185
258
|
|
|
186
|
-
async def _cloud_generate(self, cloud: CloudModel, message: str, context: Optional[str], max_tokens: int, temperature: float, stop: Optional[List[str]] = None) -> str:
|
|
187
|
-
|
|
188
|
-
system = _compose_system(SYSTEM_PROMPT, context)
|
|
259
|
+
async def _cloud_generate(self, cloud: CloudModel, message: str, context: Optional[str], max_tokens: int, temperature: float, stop: Optional[List[str]] = None, prefix: Optional[str] = None, cite_sources: bool = True) -> str:
|
|
260
|
+
system = _system_for(context, cite_sources)
|
|
189
261
|
stops = [marker for marker in (stop or []) if marker]
|
|
190
262
|
extra = {"stop": stops} if stops else {}
|
|
263
|
+
messages = [
|
|
264
|
+
{"role": "system", "content": system},
|
|
265
|
+
{"role": "user", "content": message},
|
|
266
|
+
]
|
|
267
|
+
# Assistant prefill: the OpenAI-compatible way to say "continue this".
|
|
268
|
+
# Local servers (vLLM, llama.cpp, LM Studio, Ollama) honour it; a
|
|
269
|
+
# provider that does not simply reads it as context, and `apply_prefix`
|
|
270
|
+
# makes both answers the same shape for the caller.
|
|
271
|
+
if prefix:
|
|
272
|
+
messages.append({"role": "assistant", "content": prefix})
|
|
191
273
|
try:
|
|
192
274
|
response = await cloud.client.chat.completions.create(
|
|
193
275
|
model=cloud.model,
|
|
194
|
-
messages=
|
|
195
|
-
{"role": "system", "content": system},
|
|
196
|
-
{"role": "user", "content": message},
|
|
197
|
-
],
|
|
276
|
+
messages=messages,
|
|
198
277
|
max_tokens=max_tokens,
|
|
199
278
|
temperature=temperature,
|
|
200
279
|
**extra,
|
|
201
280
|
)
|
|
202
281
|
except Exception as e:
|
|
203
282
|
raise RuntimeError(self._local_server_error_hint(cloud, e)) from e
|
|
204
|
-
return normalize_branding(response.choices[0].message.content or "")
|
|
283
|
+
return apply_prefix(prefix, normalize_branding(response.choices[0].message.content or ""))
|
|
205
284
|
|
|
206
285
|
async def stream_generate_as(
|
|
207
286
|
self,
|
|
@@ -148,6 +148,104 @@ def _mlx_sampler(temperature: float):
|
|
|
148
148
|
return None
|
|
149
149
|
|
|
150
150
|
|
|
151
|
+
def _declares_vision(model_id: str) -> bool:
|
|
152
|
+
"""Whether this checkpoint's own config says it is multimodal.
|
|
153
|
+
|
|
154
|
+
Read from ``config.json`` rather than decided from the model's *name*: a
|
|
155
|
+
list of known vision families is a list that is wrong about every model
|
|
156
|
+
published after it was written, and "which models work" is exactly the
|
|
157
|
+
question this product must not answer with a hardcoded roster.
|
|
158
|
+
|
|
159
|
+
The three markers are the ones every MLX-VLM architecture carries — a
|
|
160
|
+
nested ``vision_config``, a ``vision_tower``, or an ``image_token_index``
|
|
161
|
+
for the placeholder the processor substitutes. A config that carries none
|
|
162
|
+
of them describes a text model.
|
|
163
|
+
"""
|
|
164
|
+
import json as _json
|
|
165
|
+
from pathlib import Path as _Path
|
|
166
|
+
|
|
167
|
+
from .local_models import hf_model_dir
|
|
168
|
+
|
|
169
|
+
raw = str(model_id or "").strip()
|
|
170
|
+
candidates = []
|
|
171
|
+
explicit = _Path(raw).expanduser()
|
|
172
|
+
if raw and explicit.exists():
|
|
173
|
+
candidates.append(explicit / "config.json")
|
|
174
|
+
try:
|
|
175
|
+
candidates.append(hf_model_dir(raw) / "config.json")
|
|
176
|
+
except Exception as error: # noqa: BLE001 — an unresolvable id is not a claim
|
|
177
|
+
quiet(f"vision probe skipped for {raw}: {error}")
|
|
178
|
+
for config_path in candidates:
|
|
179
|
+
try:
|
|
180
|
+
if not config_path.exists():
|
|
181
|
+
continue
|
|
182
|
+
data = _json.loads(config_path.read_text(encoding="utf-8"))
|
|
183
|
+
except Exception as error: # noqa: BLE001 — an unreadable config is not a claim
|
|
184
|
+
quiet(f"vision probe skipped for {config_path}: {error}")
|
|
185
|
+
continue
|
|
186
|
+
if any(key in data for key in ("vision_config", "vision_tower", "image_token_index")):
|
|
187
|
+
return True
|
|
188
|
+
return False
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _text_fallback_allowed(
|
|
192
|
+
model_id: str, target_model_id: str, model_type: Optional[str]
|
|
193
|
+
) -> bool:
|
|
194
|
+
"""Whether a failed multimodal load may be retried as a text load.
|
|
195
|
+
|
|
196
|
+
Every local model is routed to MLX-VLM first, because the recommended
|
|
197
|
+
defaults are multimodal. Until v12.0.0 the retry on the text path was
|
|
198
|
+
gated on the model *id* matching Gemma 4 — so a plain text model MLX-VLM
|
|
199
|
+
cannot open (a 0.5B Qwen, an AWQ checkpoint, any fine-tune with a text-only
|
|
200
|
+
architecture) failed to load at all, with MLX-LM sitting right there able
|
|
201
|
+
to open it. That is a roster deciding what runs, and the roster was one
|
|
202
|
+
family long. "Every small model works" cannot be true while loading is
|
|
203
|
+
gated on a name.
|
|
204
|
+
|
|
205
|
+
So the rule is now two rules, and the second is **additive**:
|
|
206
|
+
|
|
207
|
+
* the v11.9.0 Gemma 4 rule, unchanged — a Gemma 4 that is not the unified
|
|
208
|
+
(vision) architecture may always retry as text;
|
|
209
|
+
* and, new, **any model whose own ``config.json`` does not declare
|
|
210
|
+
vision.** Read from the checkpoint rather than decided from the name, so
|
|
211
|
+
a model published tomorrow is judged the same way as one published last
|
|
212
|
+
year.
|
|
213
|
+
|
|
214
|
+
A config that *does* declare vision is never silently downgraded: answering
|
|
215
|
+
an image request from a text-only load is worse than refusing it.
|
|
216
|
+
"""
|
|
217
|
+
if lm_load is None:
|
|
218
|
+
return False
|
|
219
|
+
if (model_type or "").strip().lower() == "gemma4_unified":
|
|
220
|
+
return False
|
|
221
|
+
if _is_gemma4_model_id(model_id):
|
|
222
|
+
return True
|
|
223
|
+
return not _declares_vision(target_model_id)
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def apply_prefix(prefix: Optional[str], text: str) -> str:
|
|
227
|
+
"""Guarantee ``text`` begins with ``prefix``, without doubling it.
|
|
228
|
+
|
|
229
|
+
The seam's contract for a forced prefix (v12.0.0) is one sentence: *the
|
|
230
|
+
reply starts with these characters*. How it got there differs by backend —
|
|
231
|
+
the local path really does prefill the prompt with them, so the model never
|
|
232
|
+
generated them and they have to be put back; a cloud provider handed an
|
|
233
|
+
assistant prefill message may echo them, may not, and may ignore the
|
|
234
|
+
message entirely. One normalisation covers all three, so the caller parses
|
|
235
|
+
one shape whichever answered.
|
|
236
|
+
|
|
237
|
+
Leading whitespace is dropped when the reply already carries the prefix:
|
|
238
|
+
a chat template that emits ``"\\n"`` before the continuation would otherwise
|
|
239
|
+
make ``startswith`` false for a reply that plainly does start with it.
|
|
240
|
+
"""
|
|
241
|
+
if not prefix:
|
|
242
|
+
return text
|
|
243
|
+
stripped = text.lstrip()
|
|
244
|
+
if stripped.startswith(prefix):
|
|
245
|
+
return stripped
|
|
246
|
+
return f"{prefix}{text}"
|
|
247
|
+
|
|
248
|
+
|
|
151
249
|
def apply_stop_strings(text: str, stop: Optional[List[str]]) -> str:
|
|
152
250
|
"""Cut ``text`` at the earliest stop string, if any is present.
|
|
153
251
|
|
|
@@ -206,7 +304,6 @@ class _LoadingMixin(_Core):
|
|
|
206
304
|
|
|
207
305
|
def _load():
|
|
208
306
|
mx.set_default_device(mx.gpu)
|
|
209
|
-
is_gemma4 = _is_gemma4_model_id(model_id)
|
|
210
307
|
model_type = _local_model_type(target_model_id) or _local_model_type(model_id)
|
|
211
308
|
loader_kind = "mlx_vlm"
|
|
212
309
|
|
|
@@ -216,11 +313,19 @@ class _LoadingMixin(_Core):
|
|
|
216
313
|
print(f"🔄 Loading Target (VLM Mode): {target_model_id}...")
|
|
217
314
|
model, tokenizer = vlm_load(target_model_id)
|
|
218
315
|
except Exception as vlm_error:
|
|
219
|
-
if not (
|
|
316
|
+
if not _text_fallback_allowed(model_id, target_model_id, model_type):
|
|
220
317
|
raise
|
|
221
|
-
print(f"⚠️
|
|
318
|
+
print(f"⚠️ MLX-VLM load failed; retrying the MLX-LM text path: {vlm_error}")
|
|
222
319
|
print(f"🔄 Loading Target (LM Mode): {target_model_id}...")
|
|
223
|
-
|
|
320
|
+
try:
|
|
321
|
+
model, tokenizer = lm_load(target_model_id)
|
|
322
|
+
except Exception:
|
|
323
|
+
# The text path is a *fallback*: when it fails too, the
|
|
324
|
+
# useful diagnosis is the first failure, from the loader
|
|
325
|
+
# this model was routed to. Raising the second one would
|
|
326
|
+
# tell an operator their multimodal model is not a text
|
|
327
|
+
# model, which they knew.
|
|
328
|
+
raise vlm_error from None
|
|
224
329
|
loader_kind = "mlx_lm"
|
|
225
330
|
|
|
226
331
|
draft_model = None
|