ltcai 11.2.0 → 11.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (247) hide show
  1. package/README.md +46 -53
  2. package/docs/CHANGELOG.md +61 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/MULTI_AGENT_RUNTIME.md +1 -1
  6. package/docs/ONBOARDING.md +1 -1
  7. package/docs/OPERATIONS.md +6 -2
  8. package/docs/PERMISSION_MODE.md +1 -1
  9. package/docs/TRUST_MODEL.md +1 -1
  10. package/docs/WHY_LATTICE.md +1 -1
  11. package/docs/kg-schema.md +2 -2
  12. package/docs/v11.3.0_PLAN.md +202 -0
  13. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
  14. package/lattice_brain/__init__.py +1 -1
  15. package/lattice_brain/graph/_kg_common/__init__.py +287 -0
  16. package/lattice_brain/graph/_kg_common/extraction.py +516 -0
  17. package/lattice_brain/graph/_kg_common/relations.py +161 -0
  18. package/lattice_brain/graph/_kg_common/text.py +479 -0
  19. package/lattice_brain/graph/discovery_index/__init__.py +35 -0
  20. package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
  21. package/lattice_brain/graph/discovery_index/extract.py +137 -0
  22. package/lattice_brain/graph/discovery_index/scan.py +411 -0
  23. package/lattice_brain/graph/discovery_index/upsert.py +495 -0
  24. package/lattice_brain/graph/projection/__init__.py +42 -0
  25. package/lattice_brain/graph/projection/curation.py +500 -0
  26. package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
  27. package/lattice_brain/graph/retrieval/__init__.py +54 -0
  28. package/lattice_brain/graph/retrieval/context.py +197 -0
  29. package/lattice_brain/graph/retrieval/graph_view.py +319 -0
  30. package/lattice_brain/graph/retrieval/hybrid.py +488 -0
  31. package/lattice_brain/graph/retrieval/maintenance.py +121 -0
  32. package/lattice_brain/graph/retrieval/signals.py +95 -0
  33. package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
  34. package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
  35. package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
  36. package/lattice_brain/graph/retrieval_vector/search.py +560 -0
  37. package/lattice_brain/graph/retrieval_vector/status.py +374 -0
  38. package/lattice_brain/ingestion/__init__.py +130 -0
  39. package/lattice_brain/ingestion/_contract.py +90 -0
  40. package/lattice_brain/ingestion/constants.py +127 -0
  41. package/lattice_brain/ingestion/folder_scan.py +57 -0
  42. package/lattice_brain/ingestion/folders.py +258 -0
  43. package/lattice_brain/ingestion/hashing.py +26 -0
  44. package/lattice_brain/ingestion/jobs_api.py +107 -0
  45. package/lattice_brain/ingestion/models.py +80 -0
  46. package/lattice_brain/ingestion/pipeline.py +486 -0
  47. package/lattice_brain/ingestion/quality.py +209 -0
  48. package/lattice_brain/ingestion/routing.py +295 -0
  49. package/lattice_brain/multimodal/__init__.py +164 -0
  50. package/lattice_brain/multimodal/audio.py +77 -0
  51. package/lattice_brain/multimodal/common.py +118 -0
  52. package/lattice_brain/multimodal/images.py +498 -0
  53. package/lattice_brain/multimodal/ports.py +169 -0
  54. package/lattice_brain/multimodal/video.py +410 -0
  55. package/lattice_brain/portability/__init__.py +90 -0
  56. package/lattice_brain/portability/_contract.py +42 -0
  57. package/lattice_brain/portability/backups.py +338 -0
  58. package/lattice_brain/portability/bundles.py +136 -0
  59. package/lattice_brain/portability/constants.py +93 -0
  60. package/lattice_brain/portability/fsops.py +138 -0
  61. package/lattice_brain/portability/service.py +41 -0
  62. package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
  63. package/lattice_brain/runtime/__init__.py +1 -1
  64. package/lattice_brain/runtime/multi_agent.py +1 -1
  65. package/latticeai/__init__.py +1 -1
  66. package/latticeai/api/chronicle.py +63 -0
  67. package/latticeai/core/agent/__init__.py +93 -0
  68. package/latticeai/core/agent/_contract.py +79 -0
  69. package/latticeai/core/agent/context.py +57 -0
  70. package/latticeai/core/agent/deps.py +125 -0
  71. package/latticeai/core/agent/execution.py +622 -0
  72. package/latticeai/core/agent/planning.py +145 -0
  73. package/latticeai/core/agent/recovery.py +157 -0
  74. package/latticeai/core/agent/runtime.py +210 -0
  75. package/latticeai/core/agent/verification.py +231 -0
  76. package/latticeai/core/embedding_providers/__init__.py +151 -0
  77. package/latticeai/core/embedding_providers/base.py +199 -0
  78. package/latticeai/core/embedding_providers/captions.py +162 -0
  79. package/latticeai/core/embedding_providers/profiles.py +126 -0
  80. package/latticeai/core/embedding_providers/text.py +350 -0
  81. package/latticeai/core/embedding_providers/vision.py +352 -0
  82. package/latticeai/core/file_generation/__init__.py +115 -0
  83. package/latticeai/core/file_generation/bundles.py +76 -0
  84. package/latticeai/core/file_generation/extraction.py +154 -0
  85. package/latticeai/core/file_generation/inference.py +235 -0
  86. package/latticeai/core/file_generation/orchestration.py +152 -0
  87. package/latticeai/core/file_generation/prompting.py +117 -0
  88. package/latticeai/core/file_generation/repair.py +114 -0
  89. package/latticeai/core/file_generation/sanitize.py +61 -0
  90. package/latticeai/core/file_generation/validation.py +201 -0
  91. package/latticeai/core/legacy_compatibility.py +1 -1
  92. package/latticeai/core/marketplace.py +1 -1
  93. package/latticeai/core/messages.py +9 -0
  94. package/latticeai/core/workspace_os_constants.py +1 -1
  95. package/latticeai/integrations/telegram_bot/__init__.py +123 -0
  96. package/latticeai/integrations/telegram_bot/__main__.py +17 -0
  97. package/latticeai/integrations/telegram_bot/config.py +86 -0
  98. package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
  99. package/latticeai/integrations/telegram_bot/flows.py +478 -0
  100. package/latticeai/integrations/telegram_bot/helpers.py +322 -0
  101. package/latticeai/integrations/telegram_bot/screens.py +394 -0
  102. package/latticeai/models/router/__init__.py +88 -0
  103. package/latticeai/models/router/_contract.py +66 -0
  104. package/latticeai/models/router/branding.py +56 -0
  105. package/latticeai/models/router/catalog.py +69 -0
  106. package/latticeai/models/router/documents.py +199 -0
  107. package/latticeai/models/router/errors.py +37 -0
  108. package/latticeai/models/router/generation.py +258 -0
  109. package/latticeai/models/router/loading.py +291 -0
  110. package/latticeai/models/router/local_models.py +85 -0
  111. package/latticeai/models/router/registry.py +147 -0
  112. package/latticeai/runtime/build_phases/__init__.py +82 -0
  113. package/latticeai/runtime/build_phases/features.py +407 -0
  114. package/latticeai/runtime/build_phases/foundation.py +555 -0
  115. package/latticeai/runtime/build_phases/web.py +492 -0
  116. package/latticeai/runtime/runtime_context.py +1 -0
  117. package/latticeai/services/architecture_readiness.py +48 -19
  118. package/latticeai/services/brain_intelligence/__init__.py +58 -0
  119. package/latticeai/services/brain_intelligence/_contract.py +71 -0
  120. package/latticeai/services/brain_intelligence/consistency.py +193 -0
  121. package/latticeai/services/brain_intelligence/constants.py +47 -0
  122. package/latticeai/services/brain_intelligence/digest.py +258 -0
  123. package/latticeai/services/brain_intelligence/health.py +331 -0
  124. package/latticeai/services/brain_intelligence/proposals.py +264 -0
  125. package/latticeai/services/brain_intelligence/sampling.py +84 -0
  126. package/latticeai/services/brain_intelligence/service.py +48 -0
  127. package/latticeai/services/chronicle.py +557 -0
  128. package/latticeai/services/memory_service/__init__.py +52 -0
  129. package/latticeai/services/memory_service/_contract.py +100 -0
  130. package/latticeai/services/memory_service/brief.py +431 -0
  131. package/latticeai/services/memory_service/constants.py +57 -0
  132. package/latticeai/services/memory_service/maintenance.py +138 -0
  133. package/latticeai/services/memory_service/manager.py +186 -0
  134. package/latticeai/services/memory_service/proof.py +136 -0
  135. package/latticeai/services/memory_service/recall.py +225 -0
  136. package/latticeai/services/memory_service/service.py +48 -0
  137. package/latticeai/services/memory_service/stores.py +110 -0
  138. package/latticeai/services/model_runtime/__init__.py +322 -0
  139. package/latticeai/services/model_runtime/cloud.py +87 -0
  140. package/latticeai/services/model_runtime/download.py +282 -0
  141. package/latticeai/services/model_runtime/engines.py +341 -0
  142. package/latticeai/services/model_runtime/loading.py +178 -0
  143. package/latticeai/services/model_runtime/service.py +129 -0
  144. package/latticeai/services/model_runtime/state.py +131 -0
  145. package/latticeai/services/model_runtime/status.py +255 -0
  146. package/latticeai/services/product_readiness.py +15 -7
  147. package/latticeai/setup/wizard/__init__.py +126 -0
  148. package/latticeai/setup/wizard/catalog.py +172 -0
  149. package/latticeai/setup/wizard/detect.py +323 -0
  150. package/latticeai/setup/wizard/install.py +348 -0
  151. package/latticeai/setup/wizard/paths.py +168 -0
  152. package/latticeai/setup/wizard/plans.py +74 -0
  153. package/latticeai/setup/wizard/recommend.py +320 -0
  154. package/package.json +6 -2
  155. package/scripts/bump_version.py +14 -0
  156. package/scripts/capture_release_evidence.mjs +33 -21
  157. package/scripts/check_current_release_docs.mjs +1 -1
  158. package/scripts/check_i18n_namespace_coverage.mjs +41 -4
  159. package/scripts/check_max_file_lines.mjs +102 -0
  160. package/scripts/check_release_evidence_bound.mjs +30 -15
  161. package/scripts/check_screenshot_pixel_delta.py +34 -4
  162. package/scripts/check_server_i18n.mjs +1 -0
  163. package/scripts/generate_rust_parity_fixtures.py +562 -0
  164. package/scripts/lib/mock_server_fingerprint.mjs +94 -0
  165. package/scripts/release_screen_claims.json +31 -2
  166. package/src-tauri/Cargo.lock +361 -3
  167. package/src-tauri/Cargo.toml +6 -1
  168. package/src-tauri/src/backend.rs +349 -0
  169. package/src-tauri/src/folder.rs +33 -0
  170. package/src-tauri/src/main.rs +97 -399
  171. package/src-tauri/tauri.conf.json +1 -1
  172. package/static/app/asset-manifest.json +41 -37
  173. package/static/app/assets/Act-yYpYnn0v.js +1 -0
  174. package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
  175. package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
  176. package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
  177. package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
  178. package/static/app/assets/Capture-CFIRsFNE.js +1 -0
  179. package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
  180. package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
  181. package/static/app/assets/Library-DwO3yZST.js +1 -0
  182. package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
  183. package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
  184. package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
  185. package/static/app/assets/System-DW8F-2xL.js +1 -0
  186. package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
  187. package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
  188. package/static/app/assets/brain-Ci1CkWjM.js +1 -0
  189. package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
  190. package/static/app/assets/circle-check-DfInj-qD.js +1 -0
  191. package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
  192. package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
  193. package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
  194. package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
  195. package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
  196. package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
  197. package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
  198. package/static/app/assets/index-_u5iUHDr.js +10 -0
  199. package/static/app/assets/input-B0lPdRQZ.js +1 -0
  200. package/static/app/assets/link-2-CoFbooHS.js +1 -0
  201. package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
  202. package/static/app/assets/primitives-DEbN-d6p.js +1 -0
  203. package/static/app/assets/search-BybIWPNd.js +1 -0
  204. package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
  205. package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
  206. package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
  207. package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
  208. package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
  209. package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
  210. package/static/app/assets/utils-BlZr7Pd4.js +4 -0
  211. package/static/app/assets/workspace-jJY4RuAV.js +1 -0
  212. package/static/app/index.html +4 -4
  213. package/static/sw.js +1 -1
  214. package/lattice_brain/graph/_kg_common.py +0 -1331
  215. package/lattice_brain/graph/discovery_index.py +0 -1141
  216. package/lattice_brain/graph/retrieval.py +0 -1120
  217. package/lattice_brain/graph/retrieval_vector.py +0 -1293
  218. package/lattice_brain/ingestion.py +0 -1525
  219. package/lattice_brain/multimodal.py +0 -1258
  220. package/latticeai/core/agent.py +0 -1465
  221. package/latticeai/core/embedding_providers.py +0 -1196
  222. package/latticeai/core/file_generation.py +0 -1047
  223. package/latticeai/integrations/telegram_bot.py +0 -1390
  224. package/latticeai/models/router.py +0 -1007
  225. package/latticeai/runtime/build_phases.py +0 -1450
  226. package/latticeai/services/brain_intelligence.py +0 -1083
  227. package/latticeai/services/memory_service.py +0 -1177
  228. package/latticeai/services/model_runtime.py +0 -1281
  229. package/latticeai/setup/wizard.py +0 -1310
  230. package/static/app/assets/Act-AWf0SAKp.js +0 -1
  231. package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
  232. package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
  233. package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
  234. package/static/app/assets/Capture-CqOSzyPr.js +0 -1
  235. package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
  236. package/static/app/assets/Library-CX-bbhmK.js +0 -1
  237. package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
  238. package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
  239. package/static/app/assets/System-Bu2t5hn1.js +0 -1
  240. package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
  241. package/static/app/assets/brain-DJMoqrwx.js +0 -1
  242. package/static/app/assets/index-BpYkzcVm.js +0 -10
  243. package/static/app/assets/input-DSlJJxRs.js +0 -1
  244. package/static/app/assets/primitives-BCx6TvfG.js +0 -1
  245. package/static/app/assets/search-Cgy8cCFJ.js +0 -1
  246. package/static/app/assets/utils-zqPZJxdx.js +0 -4
  247. package/static/app/assets/workspace-DXTihhfU.js +0 -1
@@ -0,0 +1,199 @@
1
+ """Document generation — the same backends, driven by a specialized prompt.
2
+
3
+ Structurally parallel to :mod:`.generation` and deliberately separate: a
4
+ document run carries the caller's own system prompt, never the chat system
5
+ prompt, and never an image. Same executor, same stream-failure envelope, same
6
+ drain.
7
+ """
8
+
9
+ import asyncio
10
+ from typing import Any, AsyncIterator
11
+
12
+ from ._contract import RouterCore as _Core
13
+ from .branding import normalize_branding
14
+ from .catalog import CloudModel
15
+ from .errors import ModelStreamError, _stream_failure
16
+ from .loading import _mlx_sampler, executor
17
+
18
+
19
+ class _DocumentMixin(_Core):
20
+ """The document generation half of :class:`LLMRouter`."""
21
+
22
+ # ── Document Generation Pipeline ──────────────────────────────────────
23
+
24
+ async def generate_document(
25
+ self,
26
+ message: str,
27
+ system_prompt: str,
28
+ *,
29
+ max_tokens: int = 8192,
30
+ temperature: float = 0.3,
31
+ ) -> str:
32
+ """Generate a document using a specialized system prompt with graph context."""
33
+ return await self.generate_document_as(
34
+ None,
35
+ message,
36
+ system_prompt,
37
+ max_tokens=max_tokens,
38
+ temperature=temperature,
39
+ )
40
+
41
+ async def generate_document_as(
42
+ self,
43
+ model_id: str | None,
44
+ message: str,
45
+ system_prompt: str,
46
+ *,
47
+ max_tokens: int = 8192,
48
+ temperature: float = 0.3,
49
+ ) -> str:
50
+ """Generate a document with a request-scoped model."""
51
+ _selected, cached = self._model_snapshot(model_id)
52
+ if cached is None:
53
+ return "No model loaded."
54
+
55
+ if isinstance(cached, CloudModel):
56
+ return await self._cloud_generate_document(cached, message, system_prompt, max_tokens, temperature)
57
+
58
+ model, tokenizer, draft_model, loader_kind = self._unpack_local_cache(cached)
59
+ if hasattr(tokenizer, "apply_chat_template"):
60
+ try:
61
+ msgs = [
62
+ {"role": "system", "content": system_prompt},
63
+ {"role": "user", "content": message},
64
+ ]
65
+ prompt = tokenizer.apply_chat_template(msgs, tokenize=False, add_generation_prompt=True)
66
+ except Exception:
67
+ prompt = f"<|im_start|>system\n{system_prompt}<|im_end|>\n<|im_start|>user\n{message}<|im_end|>\n<|im_start|>assistant\n"
68
+ else:
69
+ prompt = f"<|im_start|>system\n{system_prompt}<|im_end|>\n<|im_start|>user\n{message}<|im_end|>\n<|im_start|>assistant\n"
70
+
71
+ loop = asyncio.get_event_loop()
72
+ def _gen():
73
+ import mlx.core as mx # type: ignore[no-redef]
74
+
75
+ mx.set_default_device(mx.gpu) # type: ignore[arg-type]
76
+ if loader_kind == "mlx_vlm":
77
+ from mlx_vlm import generate as vlm_gen
78
+ return vlm_gen(model, tokenizer, prompt=prompt, image=None, max_tokens=max_tokens, sampler=_mlx_sampler(temperature), draft_model=draft_model, draft_kind="mtp")
79
+ from mlx_lm import generate as lm_gen
80
+ return lm_gen(model, tokenizer, prompt=prompt, max_tokens=max_tokens, sampler=_mlx_sampler(temperature), draft_model=draft_model)
81
+ result = await loop.run_in_executor(executor, _gen)
82
+ if hasattr(result, "text"):
83
+ return normalize_branding(result.text)
84
+ return normalize_branding(str(result))
85
+
86
+ async def _cloud_generate_document(self, cloud: CloudModel, message: str, system_prompt: str, max_tokens: int, temperature: float) -> str:
87
+ try:
88
+ response = await cloud.client.chat.completions.create(
89
+ model=cloud.model,
90
+ messages=[
91
+ {"role": "system", "content": system_prompt},
92
+ {"role": "user", "content": message},
93
+ ],
94
+ max_tokens=max_tokens,
95
+ temperature=temperature,
96
+ )
97
+ except Exception as e:
98
+ raise RuntimeError(self._local_server_error_hint(cloud, e)) from e
99
+ return normalize_branding(response.choices[0].message.content or "")
100
+
101
+ async def stream_generate_document(
102
+ self,
103
+ message: str,
104
+ system_prompt: str,
105
+ *,
106
+ max_tokens: int = 8192,
107
+ temperature: float = 0.3,
108
+ ) -> AsyncIterator[str]:
109
+ """Stream document generation with specialized system prompt."""
110
+ async for chunk in self.stream_generate_document_as(
111
+ None,
112
+ message,
113
+ system_prompt,
114
+ max_tokens=max_tokens,
115
+ temperature=temperature,
116
+ ):
117
+ yield chunk
118
+
119
+ async def stream_generate_document_as(
120
+ self,
121
+ model_id: str | None,
122
+ message: str,
123
+ system_prompt: str,
124
+ *,
125
+ max_tokens: int = 8192,
126
+ temperature: float = 0.3,
127
+ ) -> AsyncIterator[str]:
128
+ """Stream a document with a request-scoped model."""
129
+ _selected, cached = self._model_snapshot(model_id)
130
+ if cached is None:
131
+ yield "No model loaded."
132
+ return
133
+
134
+ if isinstance(cached, CloudModel):
135
+ async for chunk in self._cloud_stream_document(cached, message, system_prompt, max_tokens, temperature):
136
+ yield chunk
137
+ return
138
+
139
+ model, tokenizer, draft_model, loader_kind = self._unpack_local_cache(cached)
140
+ if hasattr(tokenizer, "apply_chat_template"):
141
+ try:
142
+ msgs = [
143
+ {"role": "system", "content": system_prompt},
144
+ {"role": "user", "content": message},
145
+ ]
146
+ prompt = tokenizer.apply_chat_template(msgs, tokenize=False, add_generation_prompt=True)
147
+ except Exception:
148
+ prompt = f"<|im_start|>system\n{system_prompt}<|im_end|>\n<|im_start|>user\n{message}<|im_end|>\n<|im_start|>assistant\n"
149
+ else:
150
+ prompt = f"<|im_start|>system\n{system_prompt}<|im_end|>\n<|im_start|>user\n{message}<|im_end|>\n<|im_start|>assistant\n"
151
+
152
+ loop = asyncio.get_event_loop()
153
+ queue: "asyncio.Queue[Any]" = asyncio.Queue()
154
+
155
+ def _stream():
156
+ import mlx.core as mx # type: ignore[no-redef]
157
+
158
+ mx.set_default_device(mx.gpu) # type: ignore[arg-type]
159
+ try:
160
+ if loader_kind == "mlx_vlm":
161
+ from mlx_vlm import stream_generate as vlm_stream
162
+ gen = vlm_stream(model, tokenizer, prompt=prompt, image=None, max_tokens=max_tokens, sampler=_mlx_sampler(temperature), draft_model=draft_model, draft_kind="mtp")
163
+ else:
164
+ from mlx_lm import stream_generate as lm_stream
165
+ gen = lm_stream(model, tokenizer, prompt=prompt, max_tokens=max_tokens, sampler=_mlx_sampler(temperature), draft_model=draft_model)
166
+ for chunk in gen:
167
+ text = chunk.text if hasattr(chunk, "text") else (chunk[0] if isinstance(chunk, tuple) else str(chunk))
168
+ loop.call_soon_threadsafe(queue.put_nowait, text)
169
+ except Exception as exc:
170
+ loop.call_soon_threadsafe(
171
+ queue.put_nowait, _stream_failure("MLX document stream failed", exc)
172
+ )
173
+ finally:
174
+ loop.call_soon_threadsafe(queue.put_nowait, None)
175
+
176
+ loop.run_in_executor(executor, _stream)
177
+ async for chunk in self._drain_stream_queue(queue):
178
+ yield chunk
179
+
180
+ async def _cloud_stream_document(self, cloud: CloudModel, message: str, system_prompt: str, max_tokens: int, temperature: float) -> AsyncIterator[str]:
181
+ try:
182
+ stream = await cloud.client.chat.completions.create(
183
+ model=cloud.model,
184
+ messages=[
185
+ {"role": "system", "content": system_prompt},
186
+ {"role": "user", "content": message},
187
+ ],
188
+ max_tokens=max_tokens,
189
+ temperature=temperature,
190
+ stream=True,
191
+ )
192
+ except Exception as exc:
193
+ raise ModelStreamError(self._local_server_error_hint(cloud, exc)) from exc
194
+ async for event in stream:
195
+ if not event.choices:
196
+ continue
197
+ delta = event.choices[0].delta.content
198
+ if delta:
199
+ yield normalize_branding(delta)
@@ -0,0 +1,37 @@
1
+ """A backend that failed mid-stream produced an error, never model output.
2
+
3
+ Streaming backends used to hand their failure to the caller as a chunk of text,
4
+ which every consumer then treated as the model's answer. The failure now
5
+ travels as a typed exception, and ``_stream_failure`` is how a worker thread —
6
+ which cannot raise into the consuming coroutine — puts one on the chunk queue.
7
+ """
8
+
9
+
10
+ class ModelStreamError(RuntimeError):
11
+ """A backend failed mid-stream. This is an error, never model output.
12
+
13
+ Streaming backends used to hand their failure to the caller as a chunk of
14
+ text (``"⚠️ Error: ..."``), which every consumer then treated as the
15
+ model's answer: it was echoed to the client as content and persisted as a
16
+ successful turn. The failure now travels as this typed exception instead.
17
+
18
+ The MLX generators run on a worker thread that cannot raise into the
19
+ consuming coroutine, so the thread puts an instance on the chunk queue and
20
+ :meth:`LLMRouter._drain_stream_queue` re-raises it. The SSE endpoints
21
+ (``latticeai.api.chat_stream.stream_chat`` and the document stream in
22
+ ``latticeai.api.chat_documents``) already wrap their ``async for`` in
23
+ ``except Exception`` and emit an ``error`` frame plus a ``[stream_error]``
24
+ marker on the persisted answer, so the stream framing is unchanged.
25
+ """
26
+
27
+
28
+ def _stream_failure(stage: str, exc: BaseException) -> ModelStreamError:
29
+ """Envelope a backend exception for transport across the chunk queue.
30
+
31
+ ``raise ... from exc`` is unavailable on the worker thread (nothing there
32
+ consumes the traceback), so the cause is attached explicitly and stays
33
+ visible in logs when the consumer re-raises.
34
+ """
35
+ error = ModelStreamError(f"{stage}: {exc}")
36
+ error.__cause__ = exc
37
+ return error
@@ -0,0 +1,258 @@
1
+ """Chat generation — one answer, or a stream of tokens, local or cloud.
2
+
3
+ The MLX generators run on the dedicated single-thread executor so GPU streams
4
+ match across a request; they cannot raise into the consuming coroutine, so a
5
+ failure is enveloped onto the chunk queue and :meth:`_drain_stream_queue`
6
+ re-raises it. That is the whole reason a backend failure can no longer be
7
+ mistaken for the model's answer.
8
+ """
9
+
10
+ import asyncio
11
+ import base64
12
+ import io
13
+ from typing import Any, AsyncIterator, Optional
14
+
15
+ from PIL import Image
16
+
17
+ from latticeai.core.quiet import quiet
18
+
19
+ from ._contract import RouterCore as _Core
20
+ from .branding import SYSTEM_PROMPT, _compose_system, normalize_branding
21
+ from .catalog import CloudModel
22
+ from .errors import ModelStreamError, _stream_failure
23
+ from .loading import _mlx_sampler, executor
24
+
25
+
26
+ class _GenerationMixin(_Core):
27
+ """The chat generation half of :class:`LLMRouter`."""
28
+
29
+ def _build_prompt(self, message: str, context: Optional[str], tokenizer) -> str:
30
+ context = normalize_branding(context)
31
+ system = _compose_system(SYSTEM_PROMPT, context)
32
+ if hasattr(tokenizer, "apply_chat_template"):
33
+ try:
34
+ msgs = [{"role": "system", "content": system}, {"role": "user", "content": message}]
35
+ return tokenizer.apply_chat_template(msgs, tokenize=False, add_generation_prompt=True)
36
+ except Exception:
37
+ quiet()
38
+ return f"<|im_start|>system\n{system}<|im_end|>\n<|im_start|>user\n{message}<|im_end|>\n<|im_start|>assistant\n"
39
+
40
+ def _build_vlm_prompt(self, model, processor, message: str, context: Optional[str], num_images: int) -> str:
41
+ context = normalize_branding(context)
42
+ system = _compose_system(SYSTEM_PROMPT, context)
43
+ try:
44
+ from mlx_vlm import apply_chat_template
45
+
46
+ return apply_chat_template(
47
+ processor,
48
+ model.config,
49
+ [
50
+ {"role": "system", "content": system},
51
+ {"role": "user", "content": message},
52
+ ],
53
+ add_generation_prompt=True,
54
+ num_images=num_images,
55
+ )
56
+ except Exception as e:
57
+ print(f"⚠️ VLM chat template fallback: {e}")
58
+ return self._build_prompt(message, context, processor)
59
+
60
+ async def generate_as(
61
+ self,
62
+ model_id: str | None,
63
+ message: str,
64
+ context: Optional[str] = None,
65
+ max_tokens: int = 4096,
66
+ temperature: float = 0.2,
67
+ image_data: Optional[str] = None,
68
+ ) -> str:
69
+ """Generate with a request-scoped model without changing the default."""
70
+ _selected, cached = self._model_snapshot(model_id)
71
+ if cached is None:
72
+ return "No model."
73
+ return await self._generate_cached(cached, message, context, max_tokens, temperature, image_data)
74
+
75
+ async def generate(
76
+ self,
77
+ message: str,
78
+ context: Optional[str] = None,
79
+ max_tokens: int = 4096,
80
+ temperature: float = 0.2,
81
+ image_data: Optional[str] = None,
82
+ ) -> str:
83
+ return await self.generate_as(None, message, context, max_tokens, temperature, image_data)
84
+
85
+ async def _generate_cached(
86
+ self,
87
+ cached: object,
88
+ message: str,
89
+ context: Optional[str],
90
+ max_tokens: int,
91
+ temperature: float,
92
+ image_data: Optional[str],
93
+ ) -> str:
94
+ if isinstance(cached, CloudModel):
95
+ return await self._cloud_generate(cached, message, context, max_tokens, temperature)
96
+
97
+ model, tokenizer, draft_model, loader_kind = self._unpack_local_cache(cached)
98
+ use_vlm = loader_kind == "mlx_vlm"
99
+ prompt = (
100
+ self._build_vlm_prompt(model, tokenizer, message, context, 1 if image_data else 0)
101
+ if use_vlm
102
+ else self._build_prompt(message, context, tokenizer)
103
+ )
104
+
105
+ loop = asyncio.get_event_loop()
106
+
107
+ def _gen():
108
+ import mlx.core as mx # type: ignore[no-redef]
109
+
110
+ mx.set_default_device(mx.gpu) # type: ignore[arg-type]
111
+ if use_vlm:
112
+ from mlx_vlm import generate as vlm_gen
113
+ return vlm_gen(model, tokenizer, prompt=prompt, image=self._prep_image(image_data) if image_data else None, max_tokens=max_tokens, sampler=_mlx_sampler(temperature), draft_model=draft_model, draft_kind="mtp")
114
+ from mlx_lm import generate as lm_gen
115
+ return lm_gen(model, tokenizer, prompt=prompt, max_tokens=max_tokens, sampler=_mlx_sampler(temperature), draft_model=draft_model)
116
+ result = await loop.run_in_executor(executor, _gen)
117
+ # mlx-vlm might return a GenerationResult object; extract the text
118
+ if hasattr(result, "text"):
119
+ return normalize_branding(result.text)
120
+ return normalize_branding(str(result))
121
+
122
+ async def _cloud_generate(self, cloud: CloudModel, message: str, context: Optional[str], max_tokens: int, temperature: float) -> str:
123
+ context = normalize_branding(context)
124
+ system = _compose_system(SYSTEM_PROMPT, context)
125
+ try:
126
+ response = await cloud.client.chat.completions.create(
127
+ model=cloud.model,
128
+ messages=[
129
+ {"role": "system", "content": system},
130
+ {"role": "user", "content": message},
131
+ ],
132
+ max_tokens=max_tokens,
133
+ temperature=temperature,
134
+ )
135
+ except Exception as e:
136
+ raise RuntimeError(self._local_server_error_hint(cloud, e)) from e
137
+ return normalize_branding(response.choices[0].message.content or "")
138
+
139
+ async def stream_generate_as(
140
+ self,
141
+ model_id: str | None,
142
+ message: str,
143
+ context: Optional[str] = None,
144
+ max_tokens: int = 4096,
145
+ temperature: float = 0.2,
146
+ image_data: Optional[str] = None,
147
+ ) -> AsyncIterator[str]:
148
+ """Stream with a request-scoped model without changing the default."""
149
+ _selected, cached = self._model_snapshot(model_id)
150
+ if cached is None:
151
+ yield "No model."
152
+ return
153
+ if isinstance(cached, CloudModel):
154
+ async for chunk in self._cloud_stream_generate(cached, message, context, max_tokens, temperature):
155
+ yield chunk
156
+ return
157
+
158
+ model, tokenizer, draft_model, loader_kind = self._unpack_local_cache(cached)
159
+ use_vlm = loader_kind == "mlx_vlm"
160
+ prompt = (
161
+ self._build_vlm_prompt(model, tokenizer, message, context, 1 if image_data else 0)
162
+ if use_vlm
163
+ else self._build_prompt(message, context, tokenizer)
164
+ )
165
+ loop = asyncio.get_event_loop()
166
+ queue: "asyncio.Queue[Any]" = asyncio.Queue()
167
+
168
+ def _stream():
169
+ import mlx.core as mx # type: ignore[no-redef]
170
+
171
+ mx.set_default_device(mx.gpu) # type: ignore[arg-type]
172
+ try:
173
+ if use_vlm:
174
+ from mlx_vlm import stream_generate as vlm_stream
175
+ gen = vlm_stream(model, tokenizer, prompt=prompt, image=self._prep_image(image_data) if image_data else None, max_tokens=max_tokens, sampler=_mlx_sampler(temperature), draft_model=draft_model, draft_kind="mtp")
176
+ else:
177
+ from mlx_lm import stream_generate as lm_stream
178
+ gen = lm_stream(model, tokenizer, prompt=prompt, max_tokens=max_tokens, sampler=_mlx_sampler(temperature), draft_model=draft_model)
179
+
180
+ for chunk in gen:
181
+ text = chunk.text if hasattr(chunk, "text") else (chunk[0] if isinstance(chunk, tuple) else str(chunk))
182
+ loop.call_soon_threadsafe(queue.put_nowait, text)
183
+ except Exception as exc:
184
+ loop.call_soon_threadsafe(
185
+ queue.put_nowait, _stream_failure("MLX chat stream failed", exc)
186
+ )
187
+ finally:
188
+ loop.call_soon_threadsafe(queue.put_nowait, None)
189
+
190
+ loop.run_in_executor(executor, _stream)
191
+ async for chunk in self._drain_stream_queue(queue):
192
+ yield chunk
193
+
194
+ @staticmethod
195
+ async def _drain_stream_queue(queue: "asyncio.Queue[Any]") -> AsyncIterator[str]:
196
+ """Yield worker-thread chunks until the terminator; raise failures.
197
+
198
+ ``None`` terminates the stream. A :class:`ModelStreamError` on the
199
+ queue is a backend failure envelope, not model text, so it is raised
200
+ into the consuming coroutine — callers must never be able to mistake
201
+ it for an answer.
202
+ """
203
+ while True:
204
+ chunk = await queue.get()
205
+ if chunk is None:
206
+ return
207
+ if isinstance(chunk, ModelStreamError):
208
+ raise chunk
209
+ yield normalize_branding(chunk)
210
+
211
+ async def stream_generate(
212
+ self,
213
+ message: str,
214
+ context: Optional[str] = None,
215
+ max_tokens: int = 4096,
216
+ temperature: float = 0.2,
217
+ image_data: Optional[str] = None,
218
+ ) -> AsyncIterator[str]:
219
+ async for chunk in self.stream_generate_as(
220
+ None, message, context, max_tokens, temperature, image_data
221
+ ):
222
+ yield chunk
223
+
224
+ async def _cloud_stream_generate(self, cloud: CloudModel, message: str, context: Optional[str], max_tokens: int, temperature: float) -> AsyncIterator[str]:
225
+ context = normalize_branding(context)
226
+ system = _compose_system(SYSTEM_PROMPT, context)
227
+ try:
228
+ stream = await cloud.client.chat.completions.create(
229
+ model=cloud.model,
230
+ messages=[
231
+ {"role": "system", "content": system},
232
+ {"role": "user", "content": message},
233
+ ],
234
+ max_tokens=max_tokens,
235
+ temperature=temperature,
236
+ stream=True,
237
+ )
238
+ except Exception as exc:
239
+ # Same invariant as the MLX path: a backend that never produced a
240
+ # token failed, and that is an error — not the model's answer.
241
+ raise ModelStreamError(self._local_server_error_hint(cloud, exc)) from exc
242
+ async for event in stream:
243
+ if not event.choices:
244
+ continue
245
+ delta = event.choices[0].delta.content
246
+ if delta:
247
+ yield normalize_branding(delta)
248
+
249
+ def _prep_image(self, image_data: Optional[str]) -> Optional[Image.Image]:
250
+ if not image_data:
251
+ return None
252
+ try:
253
+ image = Image.open(io.BytesIO(base64.b64decode(image_data))).convert("RGB")
254
+ print(f"🖼️ VLM image decoded: {image.width}x{image.height}")
255
+ return image
256
+ except Exception as e:
257
+ print(f"⚠️ VLM image decode failed: {e}")
258
+ return None