ltcai 11.9.0 → 12.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (127) hide show
  1. package/README.md +80 -59
  2. package/docs/CHANGELOG.md +92 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +267 -119
  5. package/docs/ENTERPRISE.md +1 -1
  6. package/docs/MULTI_AGENT_RUNTIME.md +4 -4
  7. package/docs/ONBOARDING.md +13 -3
  8. package/docs/OPERATIONS.md +13 -4
  9. package/docs/REALTIME_COLLABORATION.md +1 -1
  10. package/docs/ROADMAP.md +113 -0
  11. package/docs/TRUST_MODEL.md +16 -2
  12. package/docs/WHY_LATTICE.md +8 -2
  13. package/docs/WORKFLOW_DESIGNER.md +2 -2
  14. package/docs/kg-schema.md +51 -5
  15. package/docs/mcp-tools.md +17 -0
  16. package/lattice_brain/__init__.py +1 -1
  17. package/lattice_brain/graph/_kg_common/extraction.py +459 -105
  18. package/lattice_brain/graph/_kg_common/normalize.py +305 -0
  19. package/lattice_brain/graph/_kg_common/patterns.py +275 -0
  20. package/lattice_brain/graph/_kg_common/relations.py +12 -3
  21. package/lattice_brain/graph/_kg_common/sections.py +107 -0
  22. package/lattice_brain/graph/_kg_constants.py +7 -0
  23. package/latticeai/__init__.py +1 -1
  24. package/latticeai/api/agent_worker_seam.py +32 -0
  25. package/latticeai/api/worker_compute.py +78 -3
  26. package/latticeai/core/embedding_providers/__init__.py +16 -0
  27. package/latticeai/core/embedding_providers/autodetect.py +302 -0
  28. package/latticeai/core/embedding_providers/base.py +25 -0
  29. package/latticeai/core/embedding_providers/profiles.py +44 -0
  30. package/latticeai/core/embedding_providers/text.py +74 -8
  31. package/latticeai/core/vector_index/__init__.py +61 -0
  32. package/latticeai/core/vector_index/hnsw.py +383 -0
  33. package/latticeai/core/vector_index/sidecar.py +329 -0
  34. package/latticeai/models/router/generation.py +101 -22
  35. package/latticeai/models/router/loading.py +109 -4
  36. package/latticeai/runtime/brain_runtime.py +43 -9
  37. package/latticeai/runtime/build_phases/worker_profile.py +13 -4
  38. package/latticeai/services/architecture_readiness.py +2 -2
  39. package/latticeai/services/product_readiness.py +10 -5
  40. package/latticeai/services/search_service.py +7 -0
  41. package/latticeai/tools/__init__.py +6 -1
  42. package/latticeai/tools/documents.py +12 -0
  43. package/latticeai/tools/markup.py +152 -0
  44. package/package.json +2 -1
  45. package/scripts/check_current_release_docs.mjs +1 -1
  46. package/scripts/compose_openapi.py +2 -0
  47. package/scripts/openapi_route_families.json +7 -3
  48. package/scripts/publish_release.mjs +157 -0
  49. package/scripts/release_screen_claims.json +12 -0
  50. package/src-tauri/Cargo.lock +44 -10
  51. package/src-tauri/Cargo.toml +1 -1
  52. package/src-tauri/tauri.conf.json +1 -1
  53. package/static/app/asset-manifest.json +47 -41
  54. package/static/app/assets/Act-Cf1L2709.js +2 -0
  55. package/static/app/assets/AdminConsole-DPAbLTYV.js +1 -0
  56. package/static/app/assets/Brain-DqamGrj-.js +2 -0
  57. package/static/app/assets/BrainHome-MHe2_RYs.js +2 -0
  58. package/static/app/assets/BrainSignals-CQPPfyyH.js +1 -0
  59. package/static/app/assets/Capture-DGdIH_Zc.js +1 -0
  60. package/static/app/assets/Chronicle-C-UlCJoJ.js +1 -0
  61. package/static/app/assets/CommandPalette-WNT4EqUX.js +1 -0
  62. package/static/app/assets/DigitalBrainExplorer-CEBH5Cwc.js +321 -0
  63. package/static/app/assets/Library-C6xd1dlf.js +1 -0
  64. package/static/app/assets/LivingBrain-BEk-0ohw.js +1 -0
  65. package/static/app/assets/ProductFlow-CZLm5iXh.js +1 -0
  66. package/static/app/assets/QueryClientProvider-B3OjqSyJ.js +1 -0
  67. package/static/app/assets/ReviewCard-CEHG6evf.js +3 -0
  68. package/static/app/assets/RunsListPanel-CLtEJSRW.js +1 -0
  69. package/static/app/assets/System-CAxwBUXw.js +1 -0
  70. package/static/app/assets/WorkflowGraph-Dj10RuGE.js +1 -0
  71. package/static/app/assets/WorkflowsPanel-Kyeh_LIT.js +2 -0
  72. package/static/app/assets/actHelpers-CtSmK9Dw.js +1 -0
  73. package/static/app/assets/arrow-left-CRl5EO4D.js +1 -0
  74. package/static/app/assets/{bot-Bc3Q27YR.js → bot-DhUGRel2.js} +1 -1
  75. package/static/app/assets/brain-CLkhHsHF.js +1 -0
  76. package/static/app/assets/button-CmaEqG1T.js +1 -0
  77. package/static/app/assets/circle-check-CFgejkOS.js +1 -0
  78. package/static/app/assets/{circle-pause-BGMiV8UU.js → circle-pause-l96izbxj.js} +1 -1
  79. package/static/app/assets/{circle-play-DoanLHnd.js → circle-play-CrZa25_q.js} +1 -1
  80. package/static/app/assets/{cpu-DwzNf82m.js → cpu-BaXudqwl.js} +1 -1
  81. package/static/app/assets/{download-Ddw49yCV.js → download-hCVFPiyc.js} +1 -1
  82. package/static/app/assets/{folder-open-Brd6Kvto.js → folder-open-CHL82Yp7.js} +1 -1
  83. package/static/app/assets/{hard-drive-Bu-DTJdB.js → hard-drive-DDzET7lk.js} +1 -1
  84. package/static/app/assets/{index-CGdg_aq9.css → index-CB93CZWW.css} +1 -1
  85. package/static/app/assets/index-D2H-wSl6.js +13 -0
  86. package/static/app/assets/input-Df1CAY_I.js +1 -0
  87. package/static/app/assets/jsx-runtime-bzQ4Vb5N.js +1 -0
  88. package/static/app/assets/{link-2-DZ4OA5tJ.js → link-2-xNnTIX1_.js} +1 -1
  89. package/static/app/assets/{permissionCopy-D9TR0F8b.js → permissionCopy-D3aWHco-.js} +1 -1
  90. package/static/app/assets/primitives-BioD2slS.js +1 -0
  91. package/static/app/assets/search-BzBw8YcW.js +1 -0
  92. package/static/app/assets/{share-2-BC5FirFv.js → share-2-FkzGf8Df.js} +1 -1
  93. package/static/app/assets/{shield-alert-DUbR2W2s.js → shield-alert-B3dwzik4.js} +1 -1
  94. package/static/app/assets/sourceMeta-DQSY_tah.js +1 -0
  95. package/static/app/assets/textarea-P8o6pvOP.js +1 -0
  96. package/static/app/assets/useFocusTrap-hswOIkXE.js +1 -0
  97. package/static/app/assets/useMutation-OJLrYSRA.js +1 -0
  98. package/static/app/assets/workspace-BCuk3Ku9.js +1 -0
  99. package/static/app/index.html +4 -4
  100. package/static/sw.js +1 -1
  101. package/static/app/assets/Act-B4WT81kh.js +0 -1
  102. package/static/app/assets/AdminConsole--Wf71m-o.js +0 -1
  103. package/static/app/assets/Brain-CtYaa26c.js +0 -321
  104. package/static/app/assets/BrainHome-Be2VPJEc.js +0 -2
  105. package/static/app/assets/BrainSignals-C0__xgpG.js +0 -1
  106. package/static/app/assets/Capture-DdNi5Peb.js +0 -1
  107. package/static/app/assets/Chronicle-B3hNveeI.js +0 -1
  108. package/static/app/assets/CommandPalette-CIsnSsFL.js +0 -1
  109. package/static/app/assets/Library-Bl5XClFV.js +0 -1
  110. package/static/app/assets/LivingBrain-DjkTB_gh.js +0 -1
  111. package/static/app/assets/ProductFlow-DCUNRNHs.js +0 -1
  112. package/static/app/assets/ReviewCard-CD3yWvUB.js +0 -3
  113. package/static/app/assets/System-BJ6jQ_SL.js +0 -1
  114. package/static/app/assets/arrow-left-6_28Z0qH.js +0 -1
  115. package/static/app/assets/brain-B9BDrMTe.js +0 -1
  116. package/static/app/assets/button-D6JcpYcf.js +0 -1
  117. package/static/app/assets/circle-check-BFu9lD-3.js +0 -1
  118. package/static/app/assets/index-CWKRRsLW.js +0 -10
  119. package/static/app/assets/input-CEqsxtil.js +0 -1
  120. package/static/app/assets/primitives-DORg7Z_7.js +0 -1
  121. package/static/app/assets/search-0NQ21wXe.js +0 -1
  122. package/static/app/assets/textarea-jtQcRSXo.js +0 -1
  123. package/static/app/assets/useFocusTrap-HRemcWId.js +0 -1
  124. package/static/app/assets/useMutation-DqlFE-Bw.js +0 -1
  125. package/static/app/assets/useQuery-BizqBNGw.js +0 -1
  126. package/static/app/assets/utils-WgW4V69R.js +0 -4
  127. package/static/app/assets/workspace-DSek3jCY.js +0 -1
@@ -0,0 +1,329 @@
1
+ """Load, refresh, and query the HNSW sidecar next to a brain database.
2
+
3
+ The live graph is cached per ``(db_path, model_id, dim)``. A missing sidecar
4
+ is ``index: "none"``. When the store has vectors the cache (or a rebuild from
5
+ ``vector_embeddings``) can serve, the reply is ``index: "hnsw"``. Staleness is
6
+ sidecar size vs ``COUNT(*)`` for that identity — reported honestly, and
7
+ refreshed through :meth:`HnswIndex.add_items` so an ingest append does not
8
+ rebuild the whole graph.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import sqlite3
14
+ import struct
15
+ import threading
16
+ from pathlib import Path
17
+ from typing import Any, Dict, List, Optional, Sequence, Tuple
18
+
19
+ from latticeai.core.config import default_data_dir
20
+
21
+ from .hnsw import HnswIndex, hnswlib_available, sidecar_paths
22
+
23
+ GRAPH_DB_NAME = "knowledge_graph.sqlite"
24
+ VECTOR_QUERY_K_CAP = 200
25
+
26
+ _CACHE: Dict[Tuple[str, str, int], HnswIndex] = {}
27
+ _LOCK = threading.Lock()
28
+
29
+
30
+ def resolve_graph_db(workspace: Optional[str] = None) -> Path:
31
+ """Brain sqlite path: explicit file, a data dir, or ``LATTICEAI_DATA_DIR``."""
32
+ if workspace:
33
+ raw = Path(str(workspace)).expanduser()
34
+ if raw.is_file():
35
+ return raw
36
+ if raw.suffix.lower() == ".sqlite":
37
+ return raw
38
+ return raw / GRAPH_DB_NAME
39
+ return default_data_dir() / GRAPH_DB_NAME
40
+
41
+
42
+ def decode_f32le(blob: bytes, dim: int) -> List[float]:
43
+ """Little-endian f32 payload — the same layout Rust ``encode`` writes."""
44
+ if not blob:
45
+ return []
46
+ count = dim if dim > 0 else len(blob) // 4
47
+ if len(blob) != count * 4:
48
+ count = len(blob) // 4
49
+ if count <= 0:
50
+ return []
51
+ return list(struct.unpack(f"<{count}f", blob[: count * 4]))
52
+
53
+
54
+ def sidecar_fingerprint(model_id: str, dim: int, size: int) -> str:
55
+ """Identity the ``.hnsw.meta.json`` file is keyed on."""
56
+ return f"{model_id}|{int(dim)}|{int(size)}"
57
+
58
+
59
+ def store_vector_count(conn: sqlite3.Connection, model_id: str, dim: int) -> int:
60
+ row = conn.execute(
61
+ "SELECT COUNT(*) FROM vector_embeddings "
62
+ "WHERE embedding_model=? AND embedding_dim=?",
63
+ (model_id, int(dim)),
64
+ ).fetchone()
65
+ return int(row[0]) if row else 0
66
+
67
+
68
+ def load_store_items(
69
+ conn: sqlite3.Connection, model_id: str, dim: int
70
+ ) -> List[Tuple[str, List[float], Dict[str, Any]]]:
71
+ items: List[Tuple[str, List[float], Dict[str, Any]]] = []
72
+ for item_id, blob in conn.execute(
73
+ "SELECT item_id, embedding FROM vector_embeddings "
74
+ "WHERE embedding_model=? AND embedding_dim=? ORDER BY indexed_at ASC, item_id ASC",
75
+ (model_id, int(dim)),
76
+ ):
77
+ vector = decode_f32le(bytes(blob or b""), int(dim))
78
+ if len(vector) != int(dim):
79
+ continue
80
+ items.append((str(item_id), vector, {}))
81
+ return items
82
+
83
+
84
+ def sidecar_meta_size(db_path: Path) -> Optional[int]:
85
+ """Label count from the sidecar meta file, or ``None`` when it is absent."""
86
+ _, meta_path = sidecar_paths(db_path)
87
+ try:
88
+ import json
89
+
90
+ meta = json.loads(meta_path.read_text(encoding="utf-8"))
91
+ labels = meta.get("labels") or []
92
+ return len(labels)
93
+ except Exception: # noqa: BLE001 — absent/corrupt meta is "no sidecar"
94
+ return None
95
+
96
+
97
+ def _open_readonly(db_path: Path) -> Optional[sqlite3.Connection]:
98
+ if not db_path.is_file():
99
+ return None
100
+ try:
101
+ conn = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True)
102
+ except sqlite3.Error:
103
+ try:
104
+ conn = sqlite3.connect(str(db_path))
105
+ except sqlite3.Error:
106
+ return None
107
+ conn.row_factory = sqlite3.Row
108
+ return conn
109
+
110
+
111
+ def _none_reply(
112
+ *,
113
+ size: int = 0,
114
+ store_size: int = 0,
115
+ stale: bool = False,
116
+ detail: Optional[str] = None,
117
+ ) -> Dict[str, Any]:
118
+ return {
119
+ "ids": [],
120
+ "scores": [],
121
+ "index": "none",
122
+ "size": int(size),
123
+ "store_size": int(store_size),
124
+ "stale": bool(stale),
125
+ "detail": detail,
126
+ "refreshed": None,
127
+ }
128
+
129
+
130
+ def _hnsw_reply(
131
+ ids: Sequence[str],
132
+ scores: Sequence[float],
133
+ *,
134
+ size: int,
135
+ store_size: int,
136
+ stale: bool,
137
+ detail: Optional[str],
138
+ refreshed: Optional[str],
139
+ ) -> Dict[str, Any]:
140
+ return {
141
+ "ids": [str(item) for item in ids],
142
+ "scores": [float(score) for score in scores],
143
+ "index": "hnsw",
144
+ "size": int(size),
145
+ "store_size": int(store_size),
146
+ "stale": bool(stale),
147
+ "detail": detail,
148
+ "refreshed": refreshed,
149
+ }
150
+
151
+
152
+ def _refresh_index(
153
+ index: HnswIndex,
154
+ items: Sequence[Tuple[str, List[float], Dict[str, Any]]],
155
+ model_id: str,
156
+ dim: int,
157
+ ) -> str:
158
+ return index.add_items(items, dim=dim, model_id=model_id)
159
+
160
+
161
+ def query_sidecar(
162
+ *,
163
+ workspace: Optional[str],
164
+ embedding_model: str,
165
+ embedding_dim: int,
166
+ vector: Sequence[float],
167
+ k: int,
168
+ db_path: Optional[Path] = None,
169
+ ) -> Dict[str, Any]:
170
+ """Serve one ANN query from the sidecar (or say the index is absent)."""
171
+ dim = int(embedding_dim)
172
+ model_id = str(embedding_model or "")
173
+ wanted = max(1, min(int(k), VECTOR_QUERY_K_CAP))
174
+ path = Path(db_path) if db_path is not None else resolve_graph_db(workspace)
175
+ if not hnswlib_available():
176
+ return _none_reply(detail="hnswlib is not available in this worker")
177
+ if dim <= 0 or len(vector) != dim:
178
+ return _none_reply(
179
+ detail=(
180
+ f"query vector width {len(vector)} does not match embedding_dim {dim}"
181
+ if dim > 0
182
+ else "embedding_dim must be a positive integer"
183
+ )
184
+ )
185
+ if not model_id:
186
+ return _none_reply(detail="embedding_model is required")
187
+
188
+ conn = _open_readonly(path)
189
+ store_size = 0
190
+ items: List[Tuple[str, List[float], Dict[str, Any]]] = []
191
+ if conn is not None:
192
+ try:
193
+ store_size = store_vector_count(conn, model_id, dim)
194
+ if store_size > 0:
195
+ items = load_store_items(conn, model_id, dim)
196
+ store_size = len(items)
197
+ finally:
198
+ conn.close()
199
+
200
+ key = (str(path), model_id, dim)
201
+ with _LOCK:
202
+ index = _CACHE.get(key)
203
+ refreshed: Optional[str] = None
204
+ if index is None:
205
+ index = HnswIndex(dim=dim, model_id=model_id)
206
+ loaded = index.load(path, fingerprint=sidecar_fingerprint(model_id, dim, store_size))
207
+ if not loaded:
208
+ if not items:
209
+ return _none_reply(
210
+ store_size=store_size,
211
+ detail="no HNSW sidecar for this embedding identity",
212
+ )
213
+ refreshed = _refresh_index(index, items, model_id, dim)
214
+ index.save(path, fingerprint=sidecar_fingerprint(model_id, dim, len(items)))
215
+ _CACHE[key] = index
216
+
217
+ cached_size = int(index.stats().get("size") or 0)
218
+ stale = cached_size != store_size
219
+ if stale:
220
+ if not items and store_size > 0:
221
+ conn = _open_readonly(path)
222
+ if conn is not None:
223
+ try:
224
+ items = load_store_items(conn, model_id, dim)
225
+ finally:
226
+ conn.close()
227
+ if items:
228
+ refreshed = _refresh_index(index, items, model_id, dim)
229
+ index.save(path, fingerprint=sidecar_fingerprint(model_id, dim, len(items)))
230
+ cached_size = int(index.stats().get("size") or 0)
231
+ stale = cached_size != len(items)
232
+ elif store_size == 0:
233
+ _CACHE.pop(key, None)
234
+ return _none_reply(
235
+ store_size=0,
236
+ detail="vector store is empty for this embedding identity",
237
+ )
238
+
239
+ if cached_size <= 0:
240
+ return _none_reply(
241
+ store_size=store_size,
242
+ stale=stale,
243
+ detail="HNSW sidecar has no labels for this identity",
244
+ )
245
+
246
+ index._ef_search = max(index._ef_search, wanted * 2, 200)
247
+ hits = index.search(list(vector), top_k=min(wanted, cached_size))
248
+ ids = [item_id for item_id, _ in hits]
249
+ scores = [score for _, score in hits]
250
+ detail = None
251
+ if stale:
252
+ detail = (
253
+ f"sidecar size {cached_size} != store count {store_size} "
254
+ "for this embedding identity"
255
+ )
256
+ return _hnsw_reply(
257
+ ids,
258
+ scores,
259
+ size=cached_size,
260
+ store_size=store_size,
261
+ stale=stale,
262
+ detail=detail,
263
+ refreshed=refreshed,
264
+ )
265
+
266
+
267
+ def sidecar_freshness(
268
+ *,
269
+ workspace: Optional[str] = None,
270
+ embedding_model: Optional[str] = None,
271
+ embedding_dim: Optional[int] = None,
272
+ db_path: Optional[Path] = None,
273
+ ) -> Dict[str, Any]:
274
+ """Cheap on-disk vs store comparison — no ANN query, no rebuild."""
275
+ path = Path(db_path) if db_path is not None else resolve_graph_db(workspace)
276
+ meta_size = sidecar_meta_size(path)
277
+ store_size = 0
278
+ conn = _open_readonly(path)
279
+ if conn is not None:
280
+ try:
281
+ if embedding_model and embedding_dim:
282
+ store_size = store_vector_count(conn, embedding_model, int(embedding_dim))
283
+ else:
284
+ row = conn.execute("SELECT COUNT(*) FROM vector_embeddings").fetchone()
285
+ store_size = int(row[0]) if row else 0
286
+ except sqlite3.Error:
287
+ store_size = 0
288
+ finally:
289
+ conn.close()
290
+ if meta_size is None:
291
+ return {
292
+ "index": "none",
293
+ "size": 0,
294
+ "store_size": store_size,
295
+ "stale": store_size > 0,
296
+ "detail": "no HNSW sidecar next to the brain database",
297
+ }
298
+ stale = meta_size != store_size
299
+ return {
300
+ "index": "hnsw",
301
+ "size": meta_size,
302
+ "store_size": store_size,
303
+ "stale": stale,
304
+ "detail": (
305
+ f"sidecar size {meta_size} != store count {store_size}"
306
+ if stale
307
+ else "sidecar matches the vector store"
308
+ ),
309
+ }
310
+
311
+
312
+ def reset_sidecar_cache() -> None:
313
+ """Drop the process cache — tests only."""
314
+ with _LOCK:
315
+ _CACHE.clear()
316
+
317
+
318
+ __all__ = [
319
+ "GRAPH_DB_NAME",
320
+ "VECTOR_QUERY_K_CAP",
321
+ "decode_f32le",
322
+ "query_sidecar",
323
+ "reset_sidecar_cache",
324
+ "resolve_graph_db",
325
+ "sidecar_fingerprint",
326
+ "sidecar_freshness",
327
+ "sidecar_meta_size",
328
+ "store_vector_count",
329
+ ]
@@ -18,9 +18,57 @@ from latticeai.core.quiet import quiet
18
18
 
19
19
  from ._contract import RouterCore as _Core
20
20
  from .branding import SYSTEM_PROMPT, _compose_system, normalize_branding
21
+
22
+
23
+ def _system_for(context, cite_sources: bool) -> str:
24
+ """The system prompt for one completion.
25
+
26
+ ``cite_sources`` is what tells the two kinds of **caller** apart, and
27
+ v12.0.0 added it because nothing did.
28
+
29
+ A *chat* caller sends retrieved passages, and gets this product's own
30
+ system prompt with the Context block and
31
+ :data:`~latticeai.models.router.branding.CITATION_INSTRUCTION` appended.
32
+ :func:`_compose_system` used to do that for any non-empty context, which is
33
+ right there and wrong for the agent seam, whose context is the loop's own
34
+ executor prompt: every agent turn was being told its instructions were
35
+ "retrieved sources" to "cite inline as [1], [2]". A large model ignores
36
+ that. A small one obeys it — the acid-test 0.5B wrote
37
+ ``[1] 인사말을 쓸 예시 코드: …`` into a file, and a 2B answered a tool call
38
+ with the citation instruction itself.
39
+
40
+ A *worker* caller sends **its own whole prompt**, and gets exactly that.
41
+ Same fact, one flag: a context that is not a corpus is an instruction, and
42
+ an instruction the caller wrote is the only instruction that turn should
43
+ carry. Until now the chat persona was still prepended to it, so six lines
44
+ of "You are Lattice AI … You are a Vision-Language Model … Be concise" sat
45
+ in front of every guided micro-turn — including the one that asks a model to
46
+ write a file's contents. Small models answer the nearest instruction: a live
47
+ 2B asked to summarise a README wrote "I am a local AI assistant that can run
48
+ on Apple Silicon" into the file, and a gemma-4-e2b opened two of its three
49
+ files with "Identity: Lattice AI (Vision-Language Model on Apple Silicon)".
50
+ Nothing leaked verbatim; the *subject* leaked, which is the same defect one
51
+ paraphrase further on. The document path has always worked this way — its
52
+ own system prompt replaces the chat identity entirely — so this is the
53
+ existing rule reaching the second caller that has a prompt of its own.
54
+
55
+ Both agent prompts still say who they are (``You are the executor of an
56
+ agent loop``) and every guided block still carries the run's own answer
57
+ language, so nothing the loop depends on is being removed — only a second,
58
+ competing identity. Branding normalisation is untouched: it runs over the
59
+ context here and over every generated string on the way out.
60
+
61
+ A worker call with **no** context still gets the product prompt. A bare
62
+ completion carrying no instruction at all is the one case where the identity
63
+ is the only thing there is to say.
64
+ """
65
+ context = normalize_branding(context)
66
+ if cite_sources:
67
+ return _compose_system(SYSTEM_PROMPT, context)
68
+ return context or SYSTEM_PROMPT
21
69
  from .catalog import CloudModel
22
70
  from .errors import ModelStreamError, _stream_failure
23
- from .loading import _mlx_sampler, apply_stop_strings, executor
71
+ from .loading import _mlx_sampler, apply_prefix, apply_stop_strings, executor
24
72
 
25
73
 
26
74
  def _stream_until_stop(
@@ -65,9 +113,10 @@ def _stream_until_stop(
65
113
  class _GenerationMixin(_Core):
66
114
  """The chat generation half of :class:`LLMRouter`."""
67
115
 
68
- def _build_prompt(self, message: str, context: Optional[str], tokenizer) -> str:
69
- context = normalize_branding(context)
70
- system = _compose_system(SYSTEM_PROMPT, context)
116
+ def _build_prompt(
117
+ self, message: str, context: Optional[str], tokenizer, cite_sources: bool = True
118
+ ) -> str:
119
+ system = _system_for(context, cite_sources)
71
120
  if hasattr(tokenizer, "apply_chat_template"):
72
121
  try:
73
122
  msgs = [{"role": "system", "content": system}, {"role": "user", "content": message}]
@@ -76,9 +125,8 @@ class _GenerationMixin(_Core):
76
125
  quiet()
77
126
  return f"<|im_start|>system\n{system}<|im_end|>\n<|im_start|>user\n{message}<|im_end|>\n<|im_start|>assistant\n"
78
127
 
79
- def _build_vlm_prompt(self, model, processor, message: str, context: Optional[str], num_images: int) -> str:
80
- context = normalize_branding(context)
81
- system = _compose_system(SYSTEM_PROMPT, context)
128
+ def _build_vlm_prompt(self, model, processor, message: str, context: Optional[str], num_images: int, cite_sources: bool = True) -> str:
129
+ system = _system_for(context, cite_sources)
82
130
  try:
83
131
  from mlx_vlm import apply_chat_template
84
132
 
@@ -94,7 +142,7 @@ class _GenerationMixin(_Core):
94
142
  )
95
143
  except Exception as e:
96
144
  print(f"⚠️ VLM chat template fallback: {e}")
97
- return self._build_prompt(message, context, processor)
145
+ return self._build_prompt(message, context, processor, cite_sources)
98
146
 
99
147
  async def generate_as(
100
148
  self,
@@ -105,6 +153,8 @@ class _GenerationMixin(_Core):
105
153
  temperature: float = 0.2,
106
154
  image_data: Optional[str] = None,
107
155
  stop: Optional[List[str]] = None,
156
+ prefix: Optional[str] = None,
157
+ cite_sources: bool = True,
108
158
  ) -> str:
109
159
  """Generate with a request-scoped model without changing the default.
110
160
 
@@ -113,12 +163,23 @@ class _GenerationMixin(_Core):
113
163
  after the stop are never produced rather than merely trimmed; the cloud
114
164
  path forwards the list to the provider, which stops server-side. Absent
115
165
  or empty means what it always meant — generate to ``max_tokens``.
166
+
167
+ ``prefix`` **starts** the reply (v12.0.0): the characters are put in the
168
+ model's mouth rather than requested of it, so a caller that needs the
169
+ answer to begin ``{"thoughts": "`` gets that by construction instead of
170
+ by asking nicely and repairing the result. Locally this is a real
171
+ prefill — the text is appended to the templated prompt after the
172
+ generation marker, so the model continues from mid-token rather than
173
+ starting a fresh turn. The returned text always begins with it; see
174
+ :func:`~latticeai.models.router.loading.apply_prefix` for how the three
175
+ backends are reconciled.
116
176
  """
117
177
  _selected, cached = self._model_snapshot(model_id)
118
178
  if cached is None:
119
179
  return "No model."
120
180
  return await self._generate_cached(
121
- cached, message, context, max_tokens, temperature, image_data, stop
181
+ cached, message, context, max_tokens, temperature, image_data, stop, prefix,
182
+ cite_sources,
122
183
  )
123
184
 
124
185
  async def generate(
@@ -129,9 +190,12 @@ class _GenerationMixin(_Core):
129
190
  temperature: float = 0.2,
130
191
  image_data: Optional[str] = None,
131
192
  stop: Optional[List[str]] = None,
193
+ prefix: Optional[str] = None,
194
+ cite_sources: bool = True,
132
195
  ) -> str:
133
196
  return await self.generate_as(
134
- None, message, context, max_tokens, temperature, image_data, stop
197
+ None, message, context, max_tokens, temperature, image_data, stop, prefix,
198
+ cite_sources,
135
199
  )
136
200
 
137
201
  async def _generate_cached(
@@ -143,19 +207,28 @@ class _GenerationMixin(_Core):
143
207
  temperature: float,
144
208
  image_data: Optional[str],
145
209
  stop: Optional[List[str]] = None,
210
+ prefix: Optional[str] = None,
211
+ cite_sources: bool = True,
146
212
  ) -> str:
147
213
  if isinstance(cached, CloudModel):
148
214
  return await self._cloud_generate(
149
- cached, message, context, max_tokens, temperature, stop
215
+ cached, message, context, max_tokens, temperature, stop, prefix, cite_sources
150
216
  )
151
217
 
152
218
  model, tokenizer, draft_model, loader_kind = self._unpack_local_cache(cached)
153
219
  use_vlm = loader_kind == "mlx_vlm"
154
220
  prompt = (
155
- self._build_vlm_prompt(model, tokenizer, message, context, 1 if image_data else 0)
221
+ self._build_vlm_prompt(
222
+ model, tokenizer, message, context, 1 if image_data else 0, cite_sources
223
+ )
156
224
  if use_vlm
157
- else self._build_prompt(message, context, tokenizer)
225
+ else self._build_prompt(message, context, tokenizer, cite_sources)
158
226
  )
227
+ # The prefill. Appended *after* the chat template's generation marker,
228
+ # so the first sampled token continues these characters instead of
229
+ # opening a reply. Nothing else in the pipeline changes.
230
+ if prefix:
231
+ prompt = f"{prompt}{prefix}"
159
232
  stops = [marker for marker in (stop or []) if marker]
160
233
 
161
234
  loop = asyncio.get_event_loop()
@@ -181,27 +254,33 @@ class _GenerationMixin(_Core):
181
254
  result = await loop.run_in_executor(executor, _gen)
182
255
  # mlx-vlm might return a GenerationResult object; extract the text
183
256
  text = result.text if hasattr(result, "text") else str(result)
184
- return normalize_branding(apply_stop_strings(text, stops))
257
+ return apply_prefix(prefix, normalize_branding(apply_stop_strings(text, stops)))
185
258
 
186
- async def _cloud_generate(self, cloud: CloudModel, message: str, context: Optional[str], max_tokens: int, temperature: float, stop: Optional[List[str]] = None) -> str:
187
- context = normalize_branding(context)
188
- system = _compose_system(SYSTEM_PROMPT, context)
259
+ async def _cloud_generate(self, cloud: CloudModel, message: str, context: Optional[str], max_tokens: int, temperature: float, stop: Optional[List[str]] = None, prefix: Optional[str] = None, cite_sources: bool = True) -> str:
260
+ system = _system_for(context, cite_sources)
189
261
  stops = [marker for marker in (stop or []) if marker]
190
262
  extra = {"stop": stops} if stops else {}
263
+ messages = [
264
+ {"role": "system", "content": system},
265
+ {"role": "user", "content": message},
266
+ ]
267
+ # Assistant prefill: the OpenAI-compatible way to say "continue this".
268
+ # Local servers (vLLM, llama.cpp, LM Studio, Ollama) honour it; a
269
+ # provider that does not simply reads it as context, and `apply_prefix`
270
+ # makes both answers the same shape for the caller.
271
+ if prefix:
272
+ messages.append({"role": "assistant", "content": prefix})
191
273
  try:
192
274
  response = await cloud.client.chat.completions.create(
193
275
  model=cloud.model,
194
- messages=[
195
- {"role": "system", "content": system},
196
- {"role": "user", "content": message},
197
- ],
276
+ messages=messages,
198
277
  max_tokens=max_tokens,
199
278
  temperature=temperature,
200
279
  **extra,
201
280
  )
202
281
  except Exception as e:
203
282
  raise RuntimeError(self._local_server_error_hint(cloud, e)) from e
204
- return normalize_branding(response.choices[0].message.content or "")
283
+ return apply_prefix(prefix, normalize_branding(response.choices[0].message.content or ""))
205
284
 
206
285
  async def stream_generate_as(
207
286
  self,
@@ -148,6 +148,104 @@ def _mlx_sampler(temperature: float):
148
148
  return None
149
149
 
150
150
 
151
+ def _declares_vision(model_id: str) -> bool:
152
+ """Whether this checkpoint's own config says it is multimodal.
153
+
154
+ Read from ``config.json`` rather than decided from the model's *name*: a
155
+ list of known vision families is a list that is wrong about every model
156
+ published after it was written, and "which models work" is exactly the
157
+ question this product must not answer with a hardcoded roster.
158
+
159
+ The three markers are the ones every MLX-VLM architecture carries — a
160
+ nested ``vision_config``, a ``vision_tower``, or an ``image_token_index``
161
+ for the placeholder the processor substitutes. A config that carries none
162
+ of them describes a text model.
163
+ """
164
+ import json as _json
165
+ from pathlib import Path as _Path
166
+
167
+ from .local_models import hf_model_dir
168
+
169
+ raw = str(model_id or "").strip()
170
+ candidates = []
171
+ explicit = _Path(raw).expanduser()
172
+ if raw and explicit.exists():
173
+ candidates.append(explicit / "config.json")
174
+ try:
175
+ candidates.append(hf_model_dir(raw) / "config.json")
176
+ except Exception as error: # noqa: BLE001 — an unresolvable id is not a claim
177
+ quiet(f"vision probe skipped for {raw}: {error}")
178
+ for config_path in candidates:
179
+ try:
180
+ if not config_path.exists():
181
+ continue
182
+ data = _json.loads(config_path.read_text(encoding="utf-8"))
183
+ except Exception as error: # noqa: BLE001 — an unreadable config is not a claim
184
+ quiet(f"vision probe skipped for {config_path}: {error}")
185
+ continue
186
+ if any(key in data for key in ("vision_config", "vision_tower", "image_token_index")):
187
+ return True
188
+ return False
189
+
190
+
191
+ def _text_fallback_allowed(
192
+ model_id: str, target_model_id: str, model_type: Optional[str]
193
+ ) -> bool:
194
+ """Whether a failed multimodal load may be retried as a text load.
195
+
196
+ Every local model is routed to MLX-VLM first, because the recommended
197
+ defaults are multimodal. Until v12.0.0 the retry on the text path was
198
+ gated on the model *id* matching Gemma 4 — so a plain text model MLX-VLM
199
+ cannot open (a 0.5B Qwen, an AWQ checkpoint, any fine-tune with a text-only
200
+ architecture) failed to load at all, with MLX-LM sitting right there able
201
+ to open it. That is a roster deciding what runs, and the roster was one
202
+ family long. "Every small model works" cannot be true while loading is
203
+ gated on a name.
204
+
205
+ So the rule is now two rules, and the second is **additive**:
206
+
207
+ * the v11.9.0 Gemma 4 rule, unchanged — a Gemma 4 that is not the unified
208
+ (vision) architecture may always retry as text;
209
+ * and, new, **any model whose own ``config.json`` does not declare
210
+ vision.** Read from the checkpoint rather than decided from the name, so
211
+ a model published tomorrow is judged the same way as one published last
212
+ year.
213
+
214
+ A config that *does* declare vision is never silently downgraded: answering
215
+ an image request from a text-only load is worse than refusing it.
216
+ """
217
+ if lm_load is None:
218
+ return False
219
+ if (model_type or "").strip().lower() == "gemma4_unified":
220
+ return False
221
+ if _is_gemma4_model_id(model_id):
222
+ return True
223
+ return not _declares_vision(target_model_id)
224
+
225
+
226
+ def apply_prefix(prefix: Optional[str], text: str) -> str:
227
+ """Guarantee ``text`` begins with ``prefix``, without doubling it.
228
+
229
+ The seam's contract for a forced prefix (v12.0.0) is one sentence: *the
230
+ reply starts with these characters*. How it got there differs by backend —
231
+ the local path really does prefill the prompt with them, so the model never
232
+ generated them and they have to be put back; a cloud provider handed an
233
+ assistant prefill message may echo them, may not, and may ignore the
234
+ message entirely. One normalisation covers all three, so the caller parses
235
+ one shape whichever answered.
236
+
237
+ Leading whitespace is dropped when the reply already carries the prefix:
238
+ a chat template that emits ``"\\n"`` before the continuation would otherwise
239
+ make ``startswith`` false for a reply that plainly does start with it.
240
+ """
241
+ if not prefix:
242
+ return text
243
+ stripped = text.lstrip()
244
+ if stripped.startswith(prefix):
245
+ return stripped
246
+ return f"{prefix}{text}"
247
+
248
+
151
249
  def apply_stop_strings(text: str, stop: Optional[List[str]]) -> str:
152
250
  """Cut ``text`` at the earliest stop string, if any is present.
153
251
 
@@ -206,7 +304,6 @@ class _LoadingMixin(_Core):
206
304
 
207
305
  def _load():
208
306
  mx.set_default_device(mx.gpu)
209
- is_gemma4 = _is_gemma4_model_id(model_id)
210
307
  model_type = _local_model_type(target_model_id) or _local_model_type(model_id)
211
308
  loader_kind = "mlx_vlm"
212
309
 
@@ -216,11 +313,19 @@ class _LoadingMixin(_Core):
216
313
  print(f"🔄 Loading Target (VLM Mode): {target_model_id}...")
217
314
  model, tokenizer = vlm_load(target_model_id)
218
315
  except Exception as vlm_error:
219
- if not (is_gemma4 and model_type != "gemma4_unified" and lm_load is not None):
316
+ if not _text_fallback_allowed(model_id, target_model_id, model_type):
220
317
  raise
221
- print(f"⚠️ Gemma 4 MLX-VLM load failed; retrying MLX-LM text path: {vlm_error}")
318
+ print(f"⚠️ MLX-VLM load failed; retrying the MLX-LM text path: {vlm_error}")
222
319
  print(f"🔄 Loading Target (LM Mode): {target_model_id}...")
223
- model, tokenizer = lm_load(target_model_id)
320
+ try:
321
+ model, tokenizer = lm_load(target_model_id)
322
+ except Exception:
323
+ # The text path is a *fallback*: when it fails too, the
324
+ # useful diagnosis is the first failure, from the loader
325
+ # this model was routed to. Raising the second one would
326
+ # tell an operator their multimodal model is not a text
327
+ # model, which they knew.
328
+ raise vlm_error from None
224
329
  loader_kind = "mlx_lm"
225
330
 
226
331
  draft_model = None