ltcai 11.2.0 → 11.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (247) hide show
  1. package/README.md +46 -53
  2. package/docs/CHANGELOG.md +61 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/MULTI_AGENT_RUNTIME.md +1 -1
  6. package/docs/ONBOARDING.md +1 -1
  7. package/docs/OPERATIONS.md +6 -2
  8. package/docs/PERMISSION_MODE.md +1 -1
  9. package/docs/TRUST_MODEL.md +1 -1
  10. package/docs/WHY_LATTICE.md +1 -1
  11. package/docs/kg-schema.md +2 -2
  12. package/docs/v11.3.0_PLAN.md +202 -0
  13. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
  14. package/lattice_brain/__init__.py +1 -1
  15. package/lattice_brain/graph/_kg_common/__init__.py +287 -0
  16. package/lattice_brain/graph/_kg_common/extraction.py +516 -0
  17. package/lattice_brain/graph/_kg_common/relations.py +161 -0
  18. package/lattice_brain/graph/_kg_common/text.py +479 -0
  19. package/lattice_brain/graph/discovery_index/__init__.py +35 -0
  20. package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
  21. package/lattice_brain/graph/discovery_index/extract.py +137 -0
  22. package/lattice_brain/graph/discovery_index/scan.py +411 -0
  23. package/lattice_brain/graph/discovery_index/upsert.py +495 -0
  24. package/lattice_brain/graph/projection/__init__.py +42 -0
  25. package/lattice_brain/graph/projection/curation.py +500 -0
  26. package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
  27. package/lattice_brain/graph/retrieval/__init__.py +54 -0
  28. package/lattice_brain/graph/retrieval/context.py +197 -0
  29. package/lattice_brain/graph/retrieval/graph_view.py +319 -0
  30. package/lattice_brain/graph/retrieval/hybrid.py +488 -0
  31. package/lattice_brain/graph/retrieval/maintenance.py +121 -0
  32. package/lattice_brain/graph/retrieval/signals.py +95 -0
  33. package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
  34. package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
  35. package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
  36. package/lattice_brain/graph/retrieval_vector/search.py +560 -0
  37. package/lattice_brain/graph/retrieval_vector/status.py +374 -0
  38. package/lattice_brain/ingestion/__init__.py +130 -0
  39. package/lattice_brain/ingestion/_contract.py +90 -0
  40. package/lattice_brain/ingestion/constants.py +127 -0
  41. package/lattice_brain/ingestion/folder_scan.py +57 -0
  42. package/lattice_brain/ingestion/folders.py +258 -0
  43. package/lattice_brain/ingestion/hashing.py +26 -0
  44. package/lattice_brain/ingestion/jobs_api.py +107 -0
  45. package/lattice_brain/ingestion/models.py +80 -0
  46. package/lattice_brain/ingestion/pipeline.py +486 -0
  47. package/lattice_brain/ingestion/quality.py +209 -0
  48. package/lattice_brain/ingestion/routing.py +295 -0
  49. package/lattice_brain/multimodal/__init__.py +164 -0
  50. package/lattice_brain/multimodal/audio.py +77 -0
  51. package/lattice_brain/multimodal/common.py +118 -0
  52. package/lattice_brain/multimodal/images.py +498 -0
  53. package/lattice_brain/multimodal/ports.py +169 -0
  54. package/lattice_brain/multimodal/video.py +410 -0
  55. package/lattice_brain/portability/__init__.py +90 -0
  56. package/lattice_brain/portability/_contract.py +42 -0
  57. package/lattice_brain/portability/backups.py +338 -0
  58. package/lattice_brain/portability/bundles.py +136 -0
  59. package/lattice_brain/portability/constants.py +93 -0
  60. package/lattice_brain/portability/fsops.py +138 -0
  61. package/lattice_brain/portability/service.py +41 -0
  62. package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
  63. package/lattice_brain/runtime/__init__.py +1 -1
  64. package/lattice_brain/runtime/multi_agent.py +1 -1
  65. package/latticeai/__init__.py +1 -1
  66. package/latticeai/api/chronicle.py +63 -0
  67. package/latticeai/core/agent/__init__.py +93 -0
  68. package/latticeai/core/agent/_contract.py +79 -0
  69. package/latticeai/core/agent/context.py +57 -0
  70. package/latticeai/core/agent/deps.py +125 -0
  71. package/latticeai/core/agent/execution.py +622 -0
  72. package/latticeai/core/agent/planning.py +145 -0
  73. package/latticeai/core/agent/recovery.py +157 -0
  74. package/latticeai/core/agent/runtime.py +210 -0
  75. package/latticeai/core/agent/verification.py +231 -0
  76. package/latticeai/core/embedding_providers/__init__.py +151 -0
  77. package/latticeai/core/embedding_providers/base.py +199 -0
  78. package/latticeai/core/embedding_providers/captions.py +162 -0
  79. package/latticeai/core/embedding_providers/profiles.py +126 -0
  80. package/latticeai/core/embedding_providers/text.py +350 -0
  81. package/latticeai/core/embedding_providers/vision.py +352 -0
  82. package/latticeai/core/file_generation/__init__.py +115 -0
  83. package/latticeai/core/file_generation/bundles.py +76 -0
  84. package/latticeai/core/file_generation/extraction.py +154 -0
  85. package/latticeai/core/file_generation/inference.py +235 -0
  86. package/latticeai/core/file_generation/orchestration.py +152 -0
  87. package/latticeai/core/file_generation/prompting.py +117 -0
  88. package/latticeai/core/file_generation/repair.py +114 -0
  89. package/latticeai/core/file_generation/sanitize.py +61 -0
  90. package/latticeai/core/file_generation/validation.py +201 -0
  91. package/latticeai/core/legacy_compatibility.py +1 -1
  92. package/latticeai/core/marketplace.py +1 -1
  93. package/latticeai/core/messages.py +9 -0
  94. package/latticeai/core/workspace_os_constants.py +1 -1
  95. package/latticeai/integrations/telegram_bot/__init__.py +123 -0
  96. package/latticeai/integrations/telegram_bot/__main__.py +17 -0
  97. package/latticeai/integrations/telegram_bot/config.py +86 -0
  98. package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
  99. package/latticeai/integrations/telegram_bot/flows.py +478 -0
  100. package/latticeai/integrations/telegram_bot/helpers.py +322 -0
  101. package/latticeai/integrations/telegram_bot/screens.py +394 -0
  102. package/latticeai/models/router/__init__.py +88 -0
  103. package/latticeai/models/router/_contract.py +66 -0
  104. package/latticeai/models/router/branding.py +56 -0
  105. package/latticeai/models/router/catalog.py +69 -0
  106. package/latticeai/models/router/documents.py +199 -0
  107. package/latticeai/models/router/errors.py +37 -0
  108. package/latticeai/models/router/generation.py +258 -0
  109. package/latticeai/models/router/loading.py +291 -0
  110. package/latticeai/models/router/local_models.py +85 -0
  111. package/latticeai/models/router/registry.py +147 -0
  112. package/latticeai/runtime/build_phases/__init__.py +82 -0
  113. package/latticeai/runtime/build_phases/features.py +407 -0
  114. package/latticeai/runtime/build_phases/foundation.py +555 -0
  115. package/latticeai/runtime/build_phases/web.py +492 -0
  116. package/latticeai/runtime/runtime_context.py +1 -0
  117. package/latticeai/services/architecture_readiness.py +48 -19
  118. package/latticeai/services/brain_intelligence/__init__.py +58 -0
  119. package/latticeai/services/brain_intelligence/_contract.py +71 -0
  120. package/latticeai/services/brain_intelligence/consistency.py +193 -0
  121. package/latticeai/services/brain_intelligence/constants.py +47 -0
  122. package/latticeai/services/brain_intelligence/digest.py +258 -0
  123. package/latticeai/services/brain_intelligence/health.py +331 -0
  124. package/latticeai/services/brain_intelligence/proposals.py +264 -0
  125. package/latticeai/services/brain_intelligence/sampling.py +84 -0
  126. package/latticeai/services/brain_intelligence/service.py +48 -0
  127. package/latticeai/services/chronicle.py +557 -0
  128. package/latticeai/services/memory_service/__init__.py +52 -0
  129. package/latticeai/services/memory_service/_contract.py +100 -0
  130. package/latticeai/services/memory_service/brief.py +431 -0
  131. package/latticeai/services/memory_service/constants.py +57 -0
  132. package/latticeai/services/memory_service/maintenance.py +138 -0
  133. package/latticeai/services/memory_service/manager.py +186 -0
  134. package/latticeai/services/memory_service/proof.py +136 -0
  135. package/latticeai/services/memory_service/recall.py +225 -0
  136. package/latticeai/services/memory_service/service.py +48 -0
  137. package/latticeai/services/memory_service/stores.py +110 -0
  138. package/latticeai/services/model_runtime/__init__.py +322 -0
  139. package/latticeai/services/model_runtime/cloud.py +87 -0
  140. package/latticeai/services/model_runtime/download.py +282 -0
  141. package/latticeai/services/model_runtime/engines.py +341 -0
  142. package/latticeai/services/model_runtime/loading.py +178 -0
  143. package/latticeai/services/model_runtime/service.py +129 -0
  144. package/latticeai/services/model_runtime/state.py +131 -0
  145. package/latticeai/services/model_runtime/status.py +255 -0
  146. package/latticeai/services/product_readiness.py +15 -7
  147. package/latticeai/setup/wizard/__init__.py +126 -0
  148. package/latticeai/setup/wizard/catalog.py +172 -0
  149. package/latticeai/setup/wizard/detect.py +323 -0
  150. package/latticeai/setup/wizard/install.py +348 -0
  151. package/latticeai/setup/wizard/paths.py +168 -0
  152. package/latticeai/setup/wizard/plans.py +74 -0
  153. package/latticeai/setup/wizard/recommend.py +320 -0
  154. package/package.json +6 -2
  155. package/scripts/bump_version.py +14 -0
  156. package/scripts/capture_release_evidence.mjs +33 -21
  157. package/scripts/check_current_release_docs.mjs +1 -1
  158. package/scripts/check_i18n_namespace_coverage.mjs +41 -4
  159. package/scripts/check_max_file_lines.mjs +102 -0
  160. package/scripts/check_release_evidence_bound.mjs +30 -15
  161. package/scripts/check_screenshot_pixel_delta.py +34 -4
  162. package/scripts/check_server_i18n.mjs +1 -0
  163. package/scripts/generate_rust_parity_fixtures.py +562 -0
  164. package/scripts/lib/mock_server_fingerprint.mjs +94 -0
  165. package/scripts/release_screen_claims.json +31 -2
  166. package/src-tauri/Cargo.lock +361 -3
  167. package/src-tauri/Cargo.toml +6 -1
  168. package/src-tauri/src/backend.rs +349 -0
  169. package/src-tauri/src/folder.rs +33 -0
  170. package/src-tauri/src/main.rs +97 -399
  171. package/src-tauri/tauri.conf.json +1 -1
  172. package/static/app/asset-manifest.json +41 -37
  173. package/static/app/assets/Act-yYpYnn0v.js +1 -0
  174. package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
  175. package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
  176. package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
  177. package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
  178. package/static/app/assets/Capture-CFIRsFNE.js +1 -0
  179. package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
  180. package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
  181. package/static/app/assets/Library-DwO3yZST.js +1 -0
  182. package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
  183. package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
  184. package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
  185. package/static/app/assets/System-DW8F-2xL.js +1 -0
  186. package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
  187. package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
  188. package/static/app/assets/brain-Ci1CkWjM.js +1 -0
  189. package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
  190. package/static/app/assets/circle-check-DfInj-qD.js +1 -0
  191. package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
  192. package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
  193. package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
  194. package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
  195. package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
  196. package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
  197. package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
  198. package/static/app/assets/index-_u5iUHDr.js +10 -0
  199. package/static/app/assets/input-B0lPdRQZ.js +1 -0
  200. package/static/app/assets/link-2-CoFbooHS.js +1 -0
  201. package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
  202. package/static/app/assets/primitives-DEbN-d6p.js +1 -0
  203. package/static/app/assets/search-BybIWPNd.js +1 -0
  204. package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
  205. package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
  206. package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
  207. package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
  208. package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
  209. package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
  210. package/static/app/assets/utils-BlZr7Pd4.js +4 -0
  211. package/static/app/assets/workspace-jJY4RuAV.js +1 -0
  212. package/static/app/index.html +4 -4
  213. package/static/sw.js +1 -1
  214. package/lattice_brain/graph/_kg_common.py +0 -1331
  215. package/lattice_brain/graph/discovery_index.py +0 -1141
  216. package/lattice_brain/graph/retrieval.py +0 -1120
  217. package/lattice_brain/graph/retrieval_vector.py +0 -1293
  218. package/lattice_brain/ingestion.py +0 -1525
  219. package/lattice_brain/multimodal.py +0 -1258
  220. package/latticeai/core/agent.py +0 -1465
  221. package/latticeai/core/embedding_providers.py +0 -1196
  222. package/latticeai/core/file_generation.py +0 -1047
  223. package/latticeai/integrations/telegram_bot.py +0 -1390
  224. package/latticeai/models/router.py +0 -1007
  225. package/latticeai/runtime/build_phases.py +0 -1450
  226. package/latticeai/services/brain_intelligence.py +0 -1083
  227. package/latticeai/services/memory_service.py +0 -1177
  228. package/latticeai/services/model_runtime.py +0 -1281
  229. package/latticeai/setup/wizard.py +0 -1310
  230. package/static/app/assets/Act-AWf0SAKp.js +0 -1
  231. package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
  232. package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
  233. package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
  234. package/static/app/assets/Capture-CqOSzyPr.js +0 -1
  235. package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
  236. package/static/app/assets/Library-CX-bbhmK.js +0 -1
  237. package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
  238. package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
  239. package/static/app/assets/System-Bu2t5hn1.js +0 -1
  240. package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
  241. package/static/app/assets/brain-DJMoqrwx.js +0 -1
  242. package/static/app/assets/index-BpYkzcVm.js +0 -10
  243. package/static/app/assets/input-DSlJJxRs.js +0 -1
  244. package/static/app/assets/primitives-BCx6TvfG.js +0 -1
  245. package/static/app/assets/search-Cgy8cCFJ.js +0 -1
  246. package/static/app/assets/utils-zqPZJxdx.js +0 -4
  247. package/static/app/assets/workspace-DXTihhfU.js +0 -1
@@ -0,0 +1,291 @@
1
+ """The optional backends, and everything that loads a model through them.
2
+
3
+ Two things live here together, deliberately. First, the module-level runtime
4
+ bindings — ``mx`` / ``vlm_load`` / ``lm_load`` / ``VLM_AVAILABLE`` /
5
+ ``LM_AVAILABLE`` / ``AsyncOpenAI`` / ``executor`` — which are guarded imports
6
+ that ``ensure_mlx_runtime`` **rebinds** after an installer has run. Second,
7
+ every method that reads them: ``load_model``, ``_load_cloud_model`` and
8
+ ``_release_memory``.
9
+
10
+ They are one module because a rebindable global is only ever correct where it
11
+ is defined: a sibling module that did ``from .loading import mx`` would hold
12
+ the import-time value forever, and ``ensure_mlx_runtime`` would appear to do
13
+ nothing. For the same reason a test standing in for a backend patches
14
+ ``latticeai.models.router.loading``, not the package.
15
+ """
16
+
17
+ import os
18
+
19
+ # Default Gemma 4 assistant drafting to MTP without overriding an operator's
20
+ # explicit MLX runtime choice.
21
+ os.environ.setdefault("MLX_VLM_DRAFT_KIND", "mtp")
22
+
23
+ import asyncio
24
+ import gc
25
+ from concurrent.futures import ThreadPoolExecutor
26
+ from typing import Any, Dict, List, Optional
27
+
28
+ from latticeai.models.model_providers import (
29
+ OPENAI_COMPATIBLE_PROVIDERS,
30
+ PROVIDER_MODEL_CATALOG,
31
+ )
32
+
33
+ from ._contract import RouterCore as _Core
34
+ from .catalog import CloudModel, parse_model_ref, source_metadata_for_model
35
+ from .local_models import (
36
+ _is_gemma4_model_id,
37
+ _local_model_type,
38
+ _resolve_local_hf_model,
39
+ )
40
+
41
+ # Optional dependencies. Each is aliased on import and then re-exported as
42
+ # `Any`, so "installed" and "absent" are the same declared type and every
43
+ # call site keeps its historical name.
44
+ try:
45
+ from openai import AsyncOpenAI as _AsyncOpenAI
46
+ except Exception:
47
+ _AsyncOpenAI = None # type: ignore[assignment,misc]
48
+ AsyncOpenAI: Any = _AsyncOpenAI
49
+
50
+
51
+ # 추론 전용 싱글 스레드 워커 (GPU 스트림 보호용)
52
+ executor = ThreadPoolExecutor(max_workers=1)
53
+
54
+
55
+ try:
56
+ import mlx.core as _mx
57
+ except Exception as e:
58
+ _mx = None # type: ignore[assignment]
59
+ print(f"⚠️ MLX core unavailable: {e}")
60
+ mx: Any = _mx
61
+
62
+
63
+ try:
64
+ from mlx_vlm import load as _vlm_load
65
+ VLM_AVAILABLE = True
66
+ print("✅ MLX-VLM is ready for multimodal models.")
67
+ except Exception as e:
68
+ _vlm_load = None # type: ignore[assignment]
69
+ VLM_AVAILABLE = False
70
+ print(f"⚠️ MLX-VLM unavailable: {e}")
71
+ vlm_load: Any = _vlm_load
72
+
73
+
74
+ try:
75
+ from mlx_lm import load as _lm_load
76
+ LM_AVAILABLE = True
77
+ print("✅ MLX-LM is ready for text fallback models.")
78
+ except Exception as e:
79
+ _lm_load = None # type: ignore[assignment]
80
+ LM_AVAILABLE = False
81
+ print(f"⚠️ MLX-LM unavailable: {e}")
82
+ lm_load: Any = _lm_load
83
+
84
+
85
+ def ensure_mlx_runtime() -> None:
86
+ global mx, vlm_load, lm_load, VLM_AVAILABLE, LM_AVAILABLE
87
+ if mx is not None and (vlm_load is not None or lm_load is not None):
88
+ return
89
+ errors = []
90
+ try:
91
+ import mlx.core as mlx_core
92
+ mx = mlx_core
93
+ mx.set_default_device(mx.gpu)
94
+ except Exception as e:
95
+ errors.append(f"mlx: {e}")
96
+ mx = None
97
+
98
+ try:
99
+ from mlx_vlm import load as mlx_vlm_load
100
+ vlm_load = mlx_vlm_load
101
+ VLM_AVAILABLE = True
102
+ except Exception as e:
103
+ vlm_load = None
104
+ VLM_AVAILABLE = False
105
+ errors.append(f"mlx-vlm: {e}")
106
+
107
+ try:
108
+ from mlx_lm import load as mlx_lm_load
109
+ lm_load = mlx_lm_load
110
+ LM_AVAILABLE = True
111
+ except Exception as e:
112
+ lm_load = None
113
+ LM_AVAILABLE = False
114
+ errors.append(f"mlx-lm: {e}")
115
+
116
+ if mx is None or (vlm_load is None and lm_load is None):
117
+ raise RuntimeError(f"MLX runtime is not available after install: {'; '.join(errors)}")
118
+
119
+
120
+ def _mlx_sampler(temperature: float):
121
+ """Build an MLX sampler callable for the given temperature.
122
+
123
+ Lattice v2.2 keeps local execution on MLX-VLM only. Returning ``None`` lets
124
+ MLX-VLM use its bundled default sampler without pulling another generation
125
+ package into the runtime contract.
126
+ """
127
+ _ = temperature
128
+ return
129
+
130
+
131
+ class _LoadingMixin(_Core):
132
+ """Loading, unloading memory, and enumerating what could be loaded."""
133
+
134
+ def _release_memory(self) -> None:
135
+ gc.collect()
136
+ if mx is not None and hasattr(mx, "clear_cache"):
137
+ try:
138
+ mx.clear_cache()
139
+ except Exception as e:
140
+ print(f"⚠️ MLX cache clear skipped: {e}")
141
+
142
+ async def load_model(
143
+ self,
144
+ model_id: str,
145
+ adapter_path: Optional[str] = None,
146
+ draft_model_id: Optional[str] = None,
147
+ api_key_override: Optional[str] = None,
148
+ owner: Optional[str] = None,
149
+ ) -> str:
150
+ provider, provider_model = parse_model_ref(model_id)
151
+ if provider != "local_mlx":
152
+ return self._load_cloud_model(provider, provider_model, api_key_override=api_key_override, owner=owner)
153
+
154
+ ensure_mlx_runtime()
155
+ if mx is None or (vlm_load is None and lm_load is None):
156
+ raise RuntimeError("MLX is not available in this process. Run on Apple Silicon with Metal access.")
157
+
158
+ cache_key = f"{model_id}_{draft_model_id}" if draft_model_id else model_id
159
+ with self._lock:
160
+ if cache_key in self._cache:
161
+ self._current = cache_key
162
+ self._touch(cache_key)
163
+ return f"Cached: {cache_key}"
164
+
165
+ self._enforce_local_model_limit(cache_key)
166
+ print(f"⏳ Loading local model stack: {cache_key}...")
167
+ loop = asyncio.get_event_loop()
168
+ target_model_id = _resolve_local_hf_model(model_id)
169
+ target_draft_model_id = _resolve_local_hf_model(draft_model_id) if draft_model_id else None
170
+
171
+ def _load():
172
+ mx.set_default_device(mx.gpu)
173
+ is_gemma4 = _is_gemma4_model_id(model_id)
174
+ model_type = _local_model_type(target_model_id) or _local_model_type(model_id)
175
+ loader_kind = "mlx_vlm"
176
+
177
+ try:
178
+ if vlm_load is None:
179
+ raise RuntimeError("MLX-VLM is not installed.")
180
+ print(f"🔄 Loading Target (VLM Mode): {target_model_id}...")
181
+ model, tokenizer = vlm_load(target_model_id)
182
+ except Exception as vlm_error:
183
+ if not (is_gemma4 and model_type != "gemma4_unified" and lm_load is not None):
184
+ raise
185
+ print(f"⚠️ Gemma 4 MLX-VLM load failed; retrying MLX-LM text path: {vlm_error}")
186
+ print(f"🔄 Loading Target (LM Mode): {target_model_id}...")
187
+ model, tokenizer = lm_load(target_model_id)
188
+ loader_kind = "mlx_lm"
189
+
190
+ draft_model = None
191
+ if target_draft_model_id:
192
+ if loader_kind == "mlx_vlm":
193
+ print(f"🔄 Loading Assistant (VLM Mode): {target_draft_model_id}...")
194
+ draft_model, _ = vlm_load(target_draft_model_id)
195
+ elif lm_load is not None:
196
+ print(f"🔄 Loading Assistant (LM Mode): {target_draft_model_id}...")
197
+ draft_model, _ = lm_load(target_draft_model_id)
198
+ print("✅ Assistant Ready.")
199
+
200
+ return model, tokenizer, draft_model, loader_kind
201
+
202
+ try:
203
+ # Use the dedicated single-thread executor to ensure MLX GPU streams match during inference
204
+ model, tokenizer, draft_model, loader_kind = await loop.run_in_executor(executor, _load)
205
+ with self._lock:
206
+ self._cache[cache_key] = (model, tokenizer, draft_model, loader_kind)
207
+ self._current = cache_key
208
+ self._touch(cache_key)
209
+ print(f"✅ Fully Loaded: {cache_key} ({loader_kind})")
210
+ return f"Success: {cache_key} ({loader_kind})"
211
+ except Exception as e:
212
+ print(f"❌ Load Error: {e}")
213
+ raise e
214
+
215
+ def _load_cloud_model(self, provider: str, model: str, api_key_override: Optional[str] = None, owner: Optional[str] = None) -> str:
216
+ if AsyncOpenAI is None:
217
+ raise RuntimeError("openai package is not installed. Add it to requirements.txt and install dependencies.")
218
+ config = OPENAI_COMPATIBLE_PROVIDERS.get(provider)
219
+ if not config:
220
+ raise RuntimeError(f"Unsupported cloud provider: {provider}")
221
+
222
+ api_key = api_key_override or os.getenv(config["env_key"]) or config.get("api_key_fallback")
223
+ if not api_key:
224
+ raise RuntimeError(f"Missing API key env var: {config['env_key']}")
225
+
226
+ base_url = os.getenv(config.get("base_url_env", "")) if config.get("base_url_env") else None
227
+ base_url = base_url or config.get("base_url")
228
+ # base_url is passed only when configured: an explicit None is not
229
+ # the same as omitting the argument.
230
+ client = (
231
+ AsyncOpenAI(api_key=api_key, base_url=base_url)
232
+ if base_url
233
+ else AsyncOpenAI(api_key=api_key)
234
+ )
235
+
236
+ cache_owner = owner or "global"
237
+ cache_key = f"{provider}:{model}::{cache_owner}"
238
+ with self._lock:
239
+ self._cache[cache_key] = CloudModel(
240
+ provider=provider, model=model, client=client, cache_key=cache_key
241
+ )
242
+ self._current = cache_key
243
+ self._touch(cache_key)
244
+ return f"Cloud provider ready: {cache_key}"
245
+
246
+ def detected_cloud_models(self) -> List[Dict[str, str]]:
247
+ local_server_providers = {"ollama", "vllm", "lmstudio", "llamacpp"}
248
+ items: List[Dict[str, Any]] = []
249
+ for provider, config in OPENAI_COMPATIBLE_PROVIDERS.items():
250
+ has_key = bool(os.getenv(config["env_key"]) or config.get("api_key_fallback"))
251
+ provider_models = PROVIDER_MODEL_CATALOG.get(provider) or [{
252
+ "id": config["default_model"],
253
+ "name": f"{provider.title()} · {config['default_model']}",
254
+ "family": provider.title(),
255
+ }]
256
+ for model in provider_models:
257
+ model_id = model["id"]
258
+ local_server = provider in local_server_providers
259
+ items.append({
260
+ "id": f"{provider}:{model_id}",
261
+ "name": model.get("name") or f"{provider.title()} · {model_id}",
262
+ "provider": provider,
263
+ "family": model.get("family"),
264
+ "tag": "local-server" if local_server else "cloud",
265
+ "available": has_key,
266
+ "requires": config["env_key"] if not has_key else None,
267
+ **source_metadata_for_model(provider, model, local_server=local_server),
268
+ })
269
+ custom = os.getenv("LATTICEAI_CLOUD_MODELS") or ""
270
+ for raw in [item.strip() for item in custom.split(",") if item.strip()]:
271
+ provider, custom_model = parse_model_ref(raw)
272
+ if provider != "local_mlx" and provider in OPENAI_COMPATIBLE_PROVIDERS:
273
+ config = OPENAI_COMPATIBLE_PROVIDERS[provider]
274
+ items.append({
275
+ "id": f"{provider}:{custom_model}",
276
+ "name": f"{provider.title()} · {custom_model}",
277
+ "provider": provider,
278
+ "tag": "cloud",
279
+ "available": bool(os.getenv(config["env_key"]) or config.get("api_key_fallback")),
280
+ "requires": None,
281
+ **source_metadata_for_model(
282
+ provider,
283
+ {
284
+ "id": custom_model,
285
+ "name": f"{provider.title()} · {custom_model}",
286
+ "family": provider.title(),
287
+ },
288
+ local_server=provider in local_server_providers,
289
+ ),
290
+ })
291
+ return items
@@ -0,0 +1,85 @@
1
+ """Finding a locally downloaded model on disk.
2
+
3
+ Three places a model may already be: an explicit path, this app's own
4
+ ``~/.ltcai/hf-models`` directory, or the shared Hugging Face cache. A directory
5
+ only counts when it actually holds a config, weights and a tokenizer — a
6
+ half-finished download must not be handed to the loader as if it were a model.
7
+ """
8
+
9
+ import json
10
+ import re
11
+ from pathlib import Path
12
+ from typing import Optional
13
+
14
+ HF_MODELS_ROOT = Path.home() / ".ltcai" / "hf-models"
15
+
16
+
17
+ def hf_model_dir(repo_id: str) -> Path:
18
+ return HF_MODELS_ROOT / repo_id.replace("/", "__")
19
+
20
+
21
+ def hf_cache_model_dir(repo_id: str) -> Optional[Path]:
22
+ """Return a usable Hugging Face cache snapshot for an already-downloaded model."""
23
+ cache_root = Path.home() / ".cache" / "huggingface" / "hub" / f"models--{repo_id.replace('/', '--')}"
24
+ snapshots = cache_root / "snapshots"
25
+ if not snapshots.exists():
26
+ return None
27
+ candidates = sorted(
28
+ (item for item in snapshots.iterdir() if item.is_dir()),
29
+ key=lambda item: item.stat().st_mtime,
30
+ reverse=True,
31
+ )
32
+ for snapshot in candidates:
33
+ if _looks_like_hf_model_dir(snapshot):
34
+ return snapshot
35
+ return None
36
+
37
+
38
+ def _looks_like_hf_model_dir(path: Path) -> bool:
39
+ if not path.exists() or not path.is_dir():
40
+ return False
41
+ has_config = (path / "config.json").exists()
42
+ has_weights = any(path.glob("*.safetensors")) or any(path.glob("*.bin"))
43
+ has_tokenizer = (
44
+ (path / "tokenizer.json").exists()
45
+ or (path / "tokenizer.model").exists()
46
+ or (path / "tokenizer_config.json").exists()
47
+ )
48
+ return has_config and has_weights and has_tokenizer
49
+
50
+
51
+ def _resolve_local_hf_model(model_id: str) -> str:
52
+ explicit_path = Path(model_id).expanduser()
53
+ if explicit_path.exists():
54
+ return str(explicit_path)
55
+ local_dir = hf_model_dir(model_id)
56
+ if _looks_like_hf_model_dir(local_dir):
57
+ return str(local_dir)
58
+ cached_dir = hf_cache_model_dir(model_id)
59
+ if cached_dir is not None:
60
+ return str(cached_dir)
61
+ return model_id
62
+
63
+
64
+ def _is_gemma4_model_id(model_id: str) -> bool:
65
+ raw = str(model_id or "").lower()
66
+ return bool(re.search(r"gemma[-_/ ]?4|gemma4", raw))
67
+
68
+
69
+ def _local_model_type(path_or_model_id: str) -> Optional[str]:
70
+ raw = str(path_or_model_id or "").strip()
71
+ candidates = []
72
+ explicit = Path(raw).expanduser()
73
+ if raw and explicit.exists():
74
+ candidates.append(explicit / "config.json")
75
+ candidates.append(hf_model_dir(raw) / "config.json")
76
+ for config_path in candidates:
77
+ try:
78
+ if config_path.exists():
79
+ data = json.loads(config_path.read_text(encoding="utf-8"))
80
+ model_type = str(data.get("model_type") or "").strip().lower()
81
+ if model_type:
82
+ return model_type
83
+ except Exception as e:
84
+ print(f"⚠️ Model config read skipped for {config_path}: {e}")
85
+ return None
@@ -0,0 +1,147 @@
1
+ """The mutable model registry: what is loaded, what is current, what to evict.
2
+
3
+ Everything here is guarded by one reentrant lock, because the eviction path
4
+ nests (``_enforce_local_model_limit`` → ``unload_model`` → ``_release_memory``)
5
+ and because generation must never change ``_current`` — that value is a
6
+ UI/default preference shared by every request, so a request-scoped model is
7
+ taken as an immutable snapshot instead.
8
+ """
9
+
10
+ import os
11
+ import threading
12
+ import time
13
+ from typing import Any, Dict, List, Optional, Tuple
14
+
15
+ from latticeai.models.model_providers import OPENAI_COMPATIBLE_PROVIDERS
16
+
17
+ from ._contract import RouterCore as _Core
18
+ from .catalog import CloudModel
19
+
20
+
21
+ class _RegistryMixin(_Core):
22
+ """The loaded-model registry half of :class:`LLMRouter`."""
23
+
24
+ def __init__(self):
25
+ # A local entry is (model, tokenizer, draft_model, loader_kind); a
26
+ # cloud entry is a CloudModel. `_unpack_local_cache` splits them.
27
+ self._cache: Dict[str, Any] = {}
28
+ self._current: Optional[str] = None
29
+ self._last_used: Dict[str, float] = {}
30
+ self._max_local_models = max(1, int(os.getenv("LATTICEAI_MAX_LOCAL_MODELS", "1")))
31
+ # Guards the mutable model registry (_cache/_current/_last_used).
32
+ # Reentrant because the eviction path nests: _enforce_local_model_limit
33
+ # → unload_model → _release_memory. Never held across the heavy
34
+ # ``run_in_executor`` load (only the sync insert/read is guarded), so a
35
+ # long model load can't block a concurrent switch/unload from acquiring.
36
+ self._lock = threading.RLock()
37
+
38
+ @property
39
+ def current_model_id(self) -> Optional[str]:
40
+ with self._lock:
41
+ return self._current
42
+
43
+ @property
44
+ def loaded_model_ids(self) -> List[str]:
45
+ with self._lock:
46
+ return list(self._cache.keys())
47
+
48
+ def switch_model(self, model_id: str) -> None:
49
+ with self._lock:
50
+ if model_id not in self._cache:
51
+ raise KeyError(model_id)
52
+ self._current = model_id
53
+ self._touch(model_id)
54
+
55
+ def unload_model(self, model_id: str) -> None:
56
+ with self._lock:
57
+ self._cache.pop(model_id, None)
58
+ self._last_used.pop(model_id, None)
59
+ if self._current == model_id:
60
+ self._current = next(iter(self._cache), None)
61
+ self._release_memory()
62
+
63
+ def unload_all(self) -> None:
64
+ with self._lock:
65
+ self._cache.clear()
66
+ self._last_used.clear()
67
+ self._current = None
68
+ self._release_memory()
69
+
70
+ def unload_idle_models(self, idle_seconds: int) -> List[str]:
71
+ if idle_seconds <= 0:
72
+ return []
73
+ now = time.monotonic()
74
+ unloaded = []
75
+ with self._lock:
76
+ for model_id, last_used in list(self._last_used.items()):
77
+ if now - last_used >= idle_seconds:
78
+ self.unload_model(model_id)
79
+ unloaded.append(model_id)
80
+ return unloaded
81
+
82
+ def model_memory_policy(self) -> Dict[str, object]:
83
+ with self._lock:
84
+ return {
85
+ "max_local_models": self._max_local_models,
86
+ "loaded_count": len(self._cache),
87
+ "last_used": dict(self._last_used),
88
+ }
89
+
90
+ def _touch(self, model_id: Optional[str] = None) -> None:
91
+ model_id = model_id or self._current
92
+ if model_id:
93
+ self._last_used[model_id] = time.monotonic()
94
+
95
+ def _is_local_model(self, model_id: str) -> bool:
96
+ cached = self._cache.get(model_id)
97
+ return cached is not None and not isinstance(cached, CloudModel)
98
+
99
+ def _enforce_local_model_limit(self, incoming_key: str) -> None:
100
+ with self._lock:
101
+ local_ids = [model_id for model_id in self._cache if self._is_local_model(model_id)]
102
+ while len(local_ids) >= self._max_local_models:
103
+ victim = min(local_ids, key=lambda model_id: self._last_used.get(model_id, 0))
104
+ if victim == incoming_key:
105
+ break
106
+ print(f"🧹 Unloading local model to stay within memory policy: {victim}")
107
+ self.unload_model(victim)
108
+ local_ids = [model_id for model_id in self._cache if self._is_local_model(model_id)]
109
+
110
+ def _is_cloud_current(self) -> bool:
111
+ with self._lock:
112
+ return bool(self._current and isinstance(self._cache.get(self._current), CloudModel))
113
+
114
+ def _local_server_error_hint(self, cloud: CloudModel, error: Exception) -> str:
115
+ raw = str(error)
116
+ if cloud.provider == "lmstudio":
117
+ base_url = os.getenv("LMSTUDIO_BASE_URL") or OPENAI_COMPATIBLE_PROVIDERS["lmstudio"]["base_url"]
118
+ return (
119
+ f"LM Studio 연결 실패: {raw}\n\n"
120
+ f"- LM Studio의 Developer/Local Server를 켜고 모델을 로드했는지 확인하세요.\n"
121
+ f"- Lattice가 보는 주소는 {base_url} 입니다. 포트가 다르면 LMSTUDIO_BASE_URL을 맞춰주세요.\n"
122
+ f"- 모델 선택창에는 LM Studio /v1/models에서 감지된 모델만 표시됩니다."
123
+ )
124
+ return raw
125
+
126
+ def _unpack_local_cache(self, cached: Any) -> Tuple[Any, Any, Any, str]:
127
+ model, tokenizer, draft_model = cached[:3]
128
+ loader_kind = str(cached[3]) if len(cached) > 3 else "mlx_vlm"
129
+ return model, tokenizer, draft_model, loader_kind
130
+
131
+ def _model_snapshot(self, model_id: Optional[str] = None) -> tuple[Optional[str], object | None]:
132
+ """Return an immutable request-scoped view of a loaded model.
133
+
134
+ Generation must never change ``_current``: that value is a UI/default
135
+ preference shared by every request. Capturing the cache entry while
136
+ holding the registry lock prevents concurrent requests from selecting
137
+ or restoring each other's models.
138
+ """
139
+ with self._lock:
140
+ selected = model_id or self._current
141
+ if not selected:
142
+ return None, None
143
+ cached = self._cache.get(selected)
144
+ if cached is None:
145
+ raise ValueError(f"Model '{selected}' is not loaded. Load it first via /models/load.")
146
+ self._touch(selected)
147
+ return selected, cached
@@ -0,0 +1,82 @@
1
+ """The ordered phases that build the Lattice AI application.
2
+
3
+ Each phase reads what earlier phases published on the :class:`RuntimeContext`
4
+ and publishes its own results. The order below *is* the dependency order and is
5
+ fixed by ``tests/unit/test_runtime_context.py``:
6
+
7
+ 1. ``platform`` — MLX/GPU device selection (the only step that touches hardware)
8
+ 2. ``config`` — configuration, security settings, paths, filesystem layout
9
+ 3. ``identity`` — users, sessions, audit, access control, API keys, SSO/VPC config
10
+ 4. ``brain`` — embedder, knowledge graph, conversations, hooks, persistence, history
11
+ 5. ``domain`` — model router, garden, chat service (needed by the web phase)
12
+ 6. ``web`` — lifespan, the FastAPI app, model runtime, static + foundation routers
13
+ 7. ``services`` — retrieval/context, chat agent runtime, the typed AppContext
14
+ 8. ``foundation_routes`` — mount the foundation routers now that AppContext exists
15
+ 9. ``platform_features`` — workspace platform, automation, review, command centre
16
+ 10. ``interaction`` — model/chat/search/tools routers and the brain tail routers
17
+
18
+ Every heavy import lives *inside* a phase, never at module scope: importing
19
+ this module must stay free of GPU init, singleton construction, and filesystem
20
+ writes (``tests/unit/test_app_factory.py`` enforces that).
21
+
22
+ Why closures still appear here: several handlers must resolve a dependency at
23
+ call time rather than at construction time, because the dependency is built by
24
+ a later phase. Those read through ``ctx``, which is exactly the late binding
25
+ the original single function got from Python's closure rules.
26
+
27
+ v11.3.0 split the 1,450-line module into three stage submodules — ``foundation``
28
+ (1-4), ``web`` (5-8) and ``features`` (9-10). Nothing else moved: this package
29
+ re-exports every phase under its historical name and :data:`BUILD_PHASES` stays
30
+ defined here, in one place, so the order test and the factory read the same
31
+ tuple rather than a copy that can drift.
32
+ """
33
+
34
+ from __future__ import annotations
35
+
36
+ from latticeai.runtime.build_phases.features import (
37
+ phase_interaction as phase_interaction,
38
+ )
39
+ from latticeai.runtime.build_phases.features import (
40
+ phase_platform_features as phase_platform_features,
41
+ )
42
+ from latticeai.runtime.build_phases.foundation import phase_brain as phase_brain
43
+ from latticeai.runtime.build_phases.foundation import phase_config as phase_config
44
+ from latticeai.runtime.build_phases.foundation import phase_identity as phase_identity
45
+ from latticeai.runtime.build_phases.foundation import phase_platform as phase_platform
46
+ from latticeai.runtime.build_phases.web import phase_domain as phase_domain
47
+ from latticeai.runtime.build_phases.web import (
48
+ phase_foundation_routes as phase_foundation_routes,
49
+ )
50
+ from latticeai.runtime.build_phases.web import phase_services as phase_services
51
+ from latticeai.runtime.build_phases.web import phase_web as phase_web
52
+ from latticeai.runtime.build_phases.web import self_model_port as self_model_port
53
+
54
+ #: The build order. Exported so the ordering test reads the same list the
55
+ #: factory runs, rather than a copy that can drift.
56
+ BUILD_PHASES = (
57
+ phase_platform,
58
+ phase_config,
59
+ phase_identity,
60
+ phase_brain,
61
+ phase_domain,
62
+ phase_web,
63
+ phase_services,
64
+ phase_foundation_routes,
65
+ phase_platform_features,
66
+ phase_interaction,
67
+ )
68
+
69
+
70
+ __all__ = [
71
+ "BUILD_PHASES",
72
+ "phase_brain",
73
+ "phase_domain",
74
+ "phase_config",
75
+ "phase_foundation_routes",
76
+ "phase_identity",
77
+ "phase_interaction",
78
+ "phase_platform",
79
+ "phase_platform_features",
80
+ "phase_services",
81
+ "phase_web",
82
+ ]