ltcai 11.2.0 → 11.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (247) hide show
  1. package/README.md +46 -53
  2. package/docs/CHANGELOG.md +61 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/MULTI_AGENT_RUNTIME.md +1 -1
  6. package/docs/ONBOARDING.md +1 -1
  7. package/docs/OPERATIONS.md +6 -2
  8. package/docs/PERMISSION_MODE.md +1 -1
  9. package/docs/TRUST_MODEL.md +1 -1
  10. package/docs/WHY_LATTICE.md +1 -1
  11. package/docs/kg-schema.md +2 -2
  12. package/docs/v11.3.0_PLAN.md +202 -0
  13. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
  14. package/lattice_brain/__init__.py +1 -1
  15. package/lattice_brain/graph/_kg_common/__init__.py +287 -0
  16. package/lattice_brain/graph/_kg_common/extraction.py +516 -0
  17. package/lattice_brain/graph/_kg_common/relations.py +161 -0
  18. package/lattice_brain/graph/_kg_common/text.py +479 -0
  19. package/lattice_brain/graph/discovery_index/__init__.py +35 -0
  20. package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
  21. package/lattice_brain/graph/discovery_index/extract.py +137 -0
  22. package/lattice_brain/graph/discovery_index/scan.py +411 -0
  23. package/lattice_brain/graph/discovery_index/upsert.py +495 -0
  24. package/lattice_brain/graph/projection/__init__.py +42 -0
  25. package/lattice_brain/graph/projection/curation.py +500 -0
  26. package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
  27. package/lattice_brain/graph/retrieval/__init__.py +54 -0
  28. package/lattice_brain/graph/retrieval/context.py +197 -0
  29. package/lattice_brain/graph/retrieval/graph_view.py +319 -0
  30. package/lattice_brain/graph/retrieval/hybrid.py +488 -0
  31. package/lattice_brain/graph/retrieval/maintenance.py +121 -0
  32. package/lattice_brain/graph/retrieval/signals.py +95 -0
  33. package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
  34. package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
  35. package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
  36. package/lattice_brain/graph/retrieval_vector/search.py +560 -0
  37. package/lattice_brain/graph/retrieval_vector/status.py +374 -0
  38. package/lattice_brain/ingestion/__init__.py +130 -0
  39. package/lattice_brain/ingestion/_contract.py +90 -0
  40. package/lattice_brain/ingestion/constants.py +127 -0
  41. package/lattice_brain/ingestion/folder_scan.py +57 -0
  42. package/lattice_brain/ingestion/folders.py +258 -0
  43. package/lattice_brain/ingestion/hashing.py +26 -0
  44. package/lattice_brain/ingestion/jobs_api.py +107 -0
  45. package/lattice_brain/ingestion/models.py +80 -0
  46. package/lattice_brain/ingestion/pipeline.py +486 -0
  47. package/lattice_brain/ingestion/quality.py +209 -0
  48. package/lattice_brain/ingestion/routing.py +295 -0
  49. package/lattice_brain/multimodal/__init__.py +164 -0
  50. package/lattice_brain/multimodal/audio.py +77 -0
  51. package/lattice_brain/multimodal/common.py +118 -0
  52. package/lattice_brain/multimodal/images.py +498 -0
  53. package/lattice_brain/multimodal/ports.py +169 -0
  54. package/lattice_brain/multimodal/video.py +410 -0
  55. package/lattice_brain/portability/__init__.py +90 -0
  56. package/lattice_brain/portability/_contract.py +42 -0
  57. package/lattice_brain/portability/backups.py +338 -0
  58. package/lattice_brain/portability/bundles.py +136 -0
  59. package/lattice_brain/portability/constants.py +93 -0
  60. package/lattice_brain/portability/fsops.py +138 -0
  61. package/lattice_brain/portability/service.py +41 -0
  62. package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
  63. package/lattice_brain/runtime/__init__.py +1 -1
  64. package/lattice_brain/runtime/multi_agent.py +1 -1
  65. package/latticeai/__init__.py +1 -1
  66. package/latticeai/api/chronicle.py +63 -0
  67. package/latticeai/core/agent/__init__.py +93 -0
  68. package/latticeai/core/agent/_contract.py +79 -0
  69. package/latticeai/core/agent/context.py +57 -0
  70. package/latticeai/core/agent/deps.py +125 -0
  71. package/latticeai/core/agent/execution.py +622 -0
  72. package/latticeai/core/agent/planning.py +145 -0
  73. package/latticeai/core/agent/recovery.py +157 -0
  74. package/latticeai/core/agent/runtime.py +210 -0
  75. package/latticeai/core/agent/verification.py +231 -0
  76. package/latticeai/core/embedding_providers/__init__.py +151 -0
  77. package/latticeai/core/embedding_providers/base.py +199 -0
  78. package/latticeai/core/embedding_providers/captions.py +162 -0
  79. package/latticeai/core/embedding_providers/profiles.py +126 -0
  80. package/latticeai/core/embedding_providers/text.py +350 -0
  81. package/latticeai/core/embedding_providers/vision.py +352 -0
  82. package/latticeai/core/file_generation/__init__.py +115 -0
  83. package/latticeai/core/file_generation/bundles.py +76 -0
  84. package/latticeai/core/file_generation/extraction.py +154 -0
  85. package/latticeai/core/file_generation/inference.py +235 -0
  86. package/latticeai/core/file_generation/orchestration.py +152 -0
  87. package/latticeai/core/file_generation/prompting.py +117 -0
  88. package/latticeai/core/file_generation/repair.py +114 -0
  89. package/latticeai/core/file_generation/sanitize.py +61 -0
  90. package/latticeai/core/file_generation/validation.py +201 -0
  91. package/latticeai/core/legacy_compatibility.py +1 -1
  92. package/latticeai/core/marketplace.py +1 -1
  93. package/latticeai/core/messages.py +9 -0
  94. package/latticeai/core/workspace_os_constants.py +1 -1
  95. package/latticeai/integrations/telegram_bot/__init__.py +123 -0
  96. package/latticeai/integrations/telegram_bot/__main__.py +17 -0
  97. package/latticeai/integrations/telegram_bot/config.py +86 -0
  98. package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
  99. package/latticeai/integrations/telegram_bot/flows.py +478 -0
  100. package/latticeai/integrations/telegram_bot/helpers.py +322 -0
  101. package/latticeai/integrations/telegram_bot/screens.py +394 -0
  102. package/latticeai/models/router/__init__.py +88 -0
  103. package/latticeai/models/router/_contract.py +66 -0
  104. package/latticeai/models/router/branding.py +56 -0
  105. package/latticeai/models/router/catalog.py +69 -0
  106. package/latticeai/models/router/documents.py +199 -0
  107. package/latticeai/models/router/errors.py +37 -0
  108. package/latticeai/models/router/generation.py +258 -0
  109. package/latticeai/models/router/loading.py +291 -0
  110. package/latticeai/models/router/local_models.py +85 -0
  111. package/latticeai/models/router/registry.py +147 -0
  112. package/latticeai/runtime/build_phases/__init__.py +82 -0
  113. package/latticeai/runtime/build_phases/features.py +407 -0
  114. package/latticeai/runtime/build_phases/foundation.py +555 -0
  115. package/latticeai/runtime/build_phases/web.py +492 -0
  116. package/latticeai/runtime/runtime_context.py +1 -0
  117. package/latticeai/services/architecture_readiness.py +48 -19
  118. package/latticeai/services/brain_intelligence/__init__.py +58 -0
  119. package/latticeai/services/brain_intelligence/_contract.py +71 -0
  120. package/latticeai/services/brain_intelligence/consistency.py +193 -0
  121. package/latticeai/services/brain_intelligence/constants.py +47 -0
  122. package/latticeai/services/brain_intelligence/digest.py +258 -0
  123. package/latticeai/services/brain_intelligence/health.py +331 -0
  124. package/latticeai/services/brain_intelligence/proposals.py +264 -0
  125. package/latticeai/services/brain_intelligence/sampling.py +84 -0
  126. package/latticeai/services/brain_intelligence/service.py +48 -0
  127. package/latticeai/services/chronicle.py +557 -0
  128. package/latticeai/services/memory_service/__init__.py +52 -0
  129. package/latticeai/services/memory_service/_contract.py +100 -0
  130. package/latticeai/services/memory_service/brief.py +431 -0
  131. package/latticeai/services/memory_service/constants.py +57 -0
  132. package/latticeai/services/memory_service/maintenance.py +138 -0
  133. package/latticeai/services/memory_service/manager.py +186 -0
  134. package/latticeai/services/memory_service/proof.py +136 -0
  135. package/latticeai/services/memory_service/recall.py +225 -0
  136. package/latticeai/services/memory_service/service.py +48 -0
  137. package/latticeai/services/memory_service/stores.py +110 -0
  138. package/latticeai/services/model_runtime/__init__.py +322 -0
  139. package/latticeai/services/model_runtime/cloud.py +87 -0
  140. package/latticeai/services/model_runtime/download.py +282 -0
  141. package/latticeai/services/model_runtime/engines.py +341 -0
  142. package/latticeai/services/model_runtime/loading.py +178 -0
  143. package/latticeai/services/model_runtime/service.py +129 -0
  144. package/latticeai/services/model_runtime/state.py +131 -0
  145. package/latticeai/services/model_runtime/status.py +255 -0
  146. package/latticeai/services/product_readiness.py +15 -7
  147. package/latticeai/setup/wizard/__init__.py +126 -0
  148. package/latticeai/setup/wizard/catalog.py +172 -0
  149. package/latticeai/setup/wizard/detect.py +323 -0
  150. package/latticeai/setup/wizard/install.py +348 -0
  151. package/latticeai/setup/wizard/paths.py +168 -0
  152. package/latticeai/setup/wizard/plans.py +74 -0
  153. package/latticeai/setup/wizard/recommend.py +320 -0
  154. package/package.json +6 -2
  155. package/scripts/bump_version.py +14 -0
  156. package/scripts/capture_release_evidence.mjs +33 -21
  157. package/scripts/check_current_release_docs.mjs +1 -1
  158. package/scripts/check_i18n_namespace_coverage.mjs +41 -4
  159. package/scripts/check_max_file_lines.mjs +102 -0
  160. package/scripts/check_release_evidence_bound.mjs +30 -15
  161. package/scripts/check_screenshot_pixel_delta.py +34 -4
  162. package/scripts/check_server_i18n.mjs +1 -0
  163. package/scripts/generate_rust_parity_fixtures.py +562 -0
  164. package/scripts/lib/mock_server_fingerprint.mjs +94 -0
  165. package/scripts/release_screen_claims.json +31 -2
  166. package/src-tauri/Cargo.lock +361 -3
  167. package/src-tauri/Cargo.toml +6 -1
  168. package/src-tauri/src/backend.rs +349 -0
  169. package/src-tauri/src/folder.rs +33 -0
  170. package/src-tauri/src/main.rs +97 -399
  171. package/src-tauri/tauri.conf.json +1 -1
  172. package/static/app/asset-manifest.json +41 -37
  173. package/static/app/assets/Act-yYpYnn0v.js +1 -0
  174. package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
  175. package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
  176. package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
  177. package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
  178. package/static/app/assets/Capture-CFIRsFNE.js +1 -0
  179. package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
  180. package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
  181. package/static/app/assets/Library-DwO3yZST.js +1 -0
  182. package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
  183. package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
  184. package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
  185. package/static/app/assets/System-DW8F-2xL.js +1 -0
  186. package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
  187. package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
  188. package/static/app/assets/brain-Ci1CkWjM.js +1 -0
  189. package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
  190. package/static/app/assets/circle-check-DfInj-qD.js +1 -0
  191. package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
  192. package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
  193. package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
  194. package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
  195. package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
  196. package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
  197. package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
  198. package/static/app/assets/index-_u5iUHDr.js +10 -0
  199. package/static/app/assets/input-B0lPdRQZ.js +1 -0
  200. package/static/app/assets/link-2-CoFbooHS.js +1 -0
  201. package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
  202. package/static/app/assets/primitives-DEbN-d6p.js +1 -0
  203. package/static/app/assets/search-BybIWPNd.js +1 -0
  204. package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
  205. package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
  206. package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
  207. package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
  208. package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
  209. package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
  210. package/static/app/assets/utils-BlZr7Pd4.js +4 -0
  211. package/static/app/assets/workspace-jJY4RuAV.js +1 -0
  212. package/static/app/index.html +4 -4
  213. package/static/sw.js +1 -1
  214. package/lattice_brain/graph/_kg_common.py +0 -1331
  215. package/lattice_brain/graph/discovery_index.py +0 -1141
  216. package/lattice_brain/graph/retrieval.py +0 -1120
  217. package/lattice_brain/graph/retrieval_vector.py +0 -1293
  218. package/lattice_brain/ingestion.py +0 -1525
  219. package/lattice_brain/multimodal.py +0 -1258
  220. package/latticeai/core/agent.py +0 -1465
  221. package/latticeai/core/embedding_providers.py +0 -1196
  222. package/latticeai/core/file_generation.py +0 -1047
  223. package/latticeai/integrations/telegram_bot.py +0 -1390
  224. package/latticeai/models/router.py +0 -1007
  225. package/latticeai/runtime/build_phases.py +0 -1450
  226. package/latticeai/services/brain_intelligence.py +0 -1083
  227. package/latticeai/services/memory_service.py +0 -1177
  228. package/latticeai/services/model_runtime.py +0 -1281
  229. package/latticeai/setup/wizard.py +0 -1310
  230. package/static/app/assets/Act-AWf0SAKp.js +0 -1
  231. package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
  232. package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
  233. package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
  234. package/static/app/assets/Capture-CqOSzyPr.js +0 -1
  235. package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
  236. package/static/app/assets/Library-CX-bbhmK.js +0 -1
  237. package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
  238. package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
  239. package/static/app/assets/System-Bu2t5hn1.js +0 -1
  240. package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
  241. package/static/app/assets/brain-DJMoqrwx.js +0 -1
  242. package/static/app/assets/index-BpYkzcVm.js +0 -10
  243. package/static/app/assets/input-DSlJJxRs.js +0 -1
  244. package/static/app/assets/primitives-BCx6TvfG.js +0 -1
  245. package/static/app/assets/search-Cgy8cCFJ.js +0 -1
  246. package/static/app/assets/utils-zqPZJxdx.js +0 -4
  247. package/static/app/assets/workspace-DXTihhfU.js +0 -1
@@ -1,1281 +0,0 @@
1
- """Model runtime and provider helpers for Lattice AI.
2
-
3
- This module owns local/cloud model preparation, engine detection, model download,
4
- provider-specific server startup, smoke tests, and runtime feature payloads. It is
5
- configured by ``server_app`` with app-level state but has no FastAPI app import.
6
- """
7
-
8
- from __future__ import annotations
9
-
10
- import asyncio
11
- import importlib.util
12
- import json
13
- import logging
14
- import os
15
- import shutil
16
- import time
17
- import urllib.error
18
- import urllib.request
19
- from dataclasses import dataclass, field, fields
20
- from pathlib import Path
21
- from typing import Any, AsyncIterator, Callable, Dict, List, Optional
22
-
23
- from latticeai.core.model_compat import (
24
- SMOKE_PROMPT as _SMOKE_PROMPT,
25
- )
26
- from latticeai.core.model_compat import (
27
- friendly_model_runtime_error as _friendly_model_runtime_error,
28
- )
29
- from latticeai.core.model_compat import (
30
- model_runtime_compatibility as _model_runtime_compatibility,
31
- )
32
- from latticeai.core.model_resolution import ModelResolution as _ModelResolution
33
- from latticeai.models.router import (
34
- HF_MODELS_ROOT,
35
- OPENAI_COMPATIBLE_PROVIDERS,
36
- AsyncOpenAI,
37
- ensure_mlx_runtime,
38
- hf_cache_model_dir,
39
- hf_model_dir,
40
- parse_model_ref,
41
- )
42
-
43
- from .model_engines import (
44
- LOCAL_SERVER_PROCESSES as _LOCAL_SERVER_PROCESSES,
45
- )
46
- from .model_engines import (
47
- engine_install_plan as _engine_install_plan,
48
- )
49
- from .model_engines import (
50
- engine_support_status as _engine_support_status,
51
- )
52
- from .model_engines import (
53
- ensure_llamacpp_server as _ensure_llamacpp_server,
54
- )
55
- from .model_engines import (
56
- ensure_lmstudio_server as _ensure_lmstudio_server,
57
- )
58
- from .model_engines import (
59
- ensure_ollama_server as _ensure_ollama_server,
60
- )
61
- from .model_engines import (
62
- ensure_vllm_server as _ensure_vllm_server,
63
- )
64
- from .model_engines import (
65
- find_lmstudio_cli as _find_lmstudio_cli,
66
- )
67
- from .model_engines import (
68
- get_ollama_pulled_models as _get_ollama_pulled_models,
69
- )
70
- from .model_engines import (
71
- get_openai_compatible_server_models as _get_openai_compatible_server_models,
72
- )
73
- from .model_engines import (
74
- install_engine as _install_engine,
75
- )
76
- from .model_engines import (
77
- local_binary as _local_binary,
78
- )
79
- from .model_engines import (
80
- pull_ollama_model_with_progress as _pull_ollama_model_with_progress,
81
- )
82
- from .model_engines import (
83
- vllm_executable as _vllm_executable,
84
- )
85
- from .model_engines import (
86
- vllm_metal_python as _vllm_metal_python,
87
- )
88
- from .model_engines import (
89
- wait_for_openai_compatible_server as _wait_for_openai_compatible_server,
90
- )
91
- from .model_engines import (
92
- windows_binary_candidates as _windows_binary_candidates,
93
- )
94
- from .model_errors import ModelRuntimeError
95
-
96
- # ``model_loading._get_model_runtime_deps`` imports these private names from
97
- # this module to preserve the historical model_runtime wiring surface.
98
- _MODEL_LOADING_COMPAT_EXPORTS = (
99
- _friendly_model_runtime_error,
100
- _model_runtime_compatibility,
101
- _SMOKE_PROMPT,
102
- )
103
-
104
-
105
- def _missing_current_user(_request: Any) -> Optional[str]:
106
- return None
107
-
108
-
109
- def _missing_user_api_key(_email: Optional[str], _provider: str) -> Optional[str]:
110
- return None
111
-
112
-
113
- @dataclass(frozen=True, slots=True)
114
- class ModelRuntimeState:
115
- """Immutable application-owned dependencies for one model runtime.
116
-
117
- Upper-case configuration field names intentionally match the long-standing
118
- composition-root vocabulary. Unlike the former module ``STATE`` object,
119
- instances are explicit, immutable, and safe to create more than once in a
120
- process (for example in isolated tests or multiple ASGI applications).
121
- """
122
-
123
- router: Any = None
124
- APP_MODE: str = "local"
125
- DEFAULT_HOST: str = "127.0.0.1"
126
- DEFAULT_PORT: int = 4825
127
- DATA_DIR: Path = field(default_factory=lambda: Path.home() / ".latticeai")
128
- BASE_DIR: Path = field(default_factory=Path.cwd)
129
- ENABLE_TELEGRAM: bool = False
130
- ENABLE_GRAPH: bool = True
131
- AUTOLOAD_MODELS: bool = False
132
- MODEL_IDLE_UNLOAD_SECONDS: int = 0
133
- ALLOW_MODEL_DOWNLOADS: bool = False
134
- MODEL_DOWNLOAD_TIMEOUT: int = 300
135
- ALLOW_LOCAL_MODELS: bool = True
136
- REQUIRE_AUTH: bool = False
137
- INVITE_GATE_ENABLED: bool = False
138
- ALLOW_PLAINTEXT_API_KEYS: bool = False
139
- CORS_ALLOW_NETWORK: bool = False
140
- PUBLIC_MODEL: str = "openai:gpt-4o-mini"
141
- LOCAL_MODEL: str = "mlx-community/gemma-4-12B-it-4bit"
142
- IS_PUBLIC_MODE: bool = False
143
- keyring: Any = None
144
- get_current_user: Callable[[Any], Optional[str]] = _missing_current_user
145
- get_user_api_key: Callable[[Optional[str], str], Optional[str]] = _missing_user_api_key
146
-
147
-
148
- def create_model_runtime_state(**deps: Any) -> ModelRuntimeState:
149
- """Create an immutable runtime dependency set with strict key validation."""
150
-
151
- known = {item.name for item in fields(ModelRuntimeState)}
152
- unknown = sorted(set(deps) - known)
153
- if unknown:
154
- raise TypeError(f"unknown model runtime dependencies: {', '.join(unknown)}")
155
- return ModelRuntimeState(**deps)
156
-
157
- def _download_allowed(
158
- allow_download: bool = False, *, state: ModelRuntimeState
159
- ) -> bool:
160
- autoload = state.AUTOLOAD_MODELS
161
- configured = state.ALLOW_MODEL_DOWNLOADS
162
- return bool(allow_download) or bool(configured) or bool(autoload)
163
-
164
-
165
- def _download_block(provider: str, model_name: str) -> None:
166
- raise ModelRuntimeError(
167
- status_code=409,
168
- detail={
169
- "status": "unavailable",
170
- "capability": "model_download",
171
- "provider": provider,
172
- "model": model_name,
173
- "reason": (
174
- "Model files are not present locally. Lattice AI does not start "
175
- "outbound model downloads by default, and token/model presence "
176
- "alone never authorizes network activity."
177
- ),
178
- "action": "Use the explicit pull/prepare flow with download consent, or set LATTICEAI_ALLOW_MODEL_DOWNLOADS=true.",
179
- },
180
- )
181
-
182
-
183
- def _engine_install_block(engine: str) -> None:
184
- raise ModelRuntimeError(
185
- status_code=409,
186
- detail={
187
- "status": "unavailable",
188
- "capability": "engine_install",
189
- "engine": engine,
190
- "reason": (
191
- "The requested local runtime is not installed. Lattice AI does not "
192
- "run package-manager or installer commands from Model Load by default."
193
- ),
194
- "action": "Install the runtime explicitly from Library/System setup, or enable explicit download/install consent for this request.",
195
- },
196
- )
197
-
198
-
199
- def configure_model_runtime(**deps: Any) -> "ModelRuntimeService":
200
- """Compatibility factory returning an isolated, bound runtime service.
201
-
202
- The historical function mutated process-wide module globals. Keeping the
203
- import path while returning a service preserves practical construction
204
- compatibility without ambient state or cross-application leakage.
205
- """
206
-
207
- return ModelRuntimeService(create_model_runtime_state(**deps))
208
-
209
-
210
- # Catalog data + version-dedup helpers live in ``model_catalog``; re-exported
211
- # here so existing ``from ...model_runtime import ENGINE_MODEL_CATALOG`` imports
212
- # keep working.
213
- from latticeai.core.quiet import ( # noqa: E402 — re-export placed after the globals it documents
214
- quiet, # noqa: E402 — re-export placed after the globals it documents
215
- )
216
- from latticeai.services.model_catalog import ( # noqa: E402, F401 (re-export after the module globals it documents)
217
- _VERSIONED_MODEL_PATTERNS,
218
- ENGINE_INSTALLERS,
219
- ENGINE_MODEL_CATALOG,
220
- MODEL_ENGINE_ALIASES,
221
- _model_family_version,
222
- _version_tuple,
223
- filter_lower_family_versions,
224
- )
225
-
226
-
227
- def _update_env_file(env_file: Path, key: str, value: str) -> None:
228
- lines = []
229
- found = False
230
- if env_file.exists():
231
- for line in env_file.read_text(encoding="utf-8").splitlines():
232
- if line.startswith(f"{key}="):
233
- lines.append(f"{key}={value}")
234
- found = True
235
- else:
236
- lines.append(line)
237
- if not found:
238
- lines.append(f"{key}={value}")
239
- env_file.write_text("\n".join(lines) + "\n", encoding="utf-8")
240
-
241
-
242
- LOCAL_SERVER_PROCESSES = _LOCAL_SERVER_PROCESSES
243
- VLLM_METAL_ENV = Path.home() / ".venv-vllm-metal"
244
- VLLM_METAL_BIN = VLLM_METAL_ENV / "bin" / "vllm"
245
- VLLM_METAL_PYTHON = VLLM_METAL_ENV / "bin" / "python"
246
- LMSTUDIO_BUNDLED_CLI = Path("/Applications/LM Studio.app/Contents/Resources/app/.webpack/lms")
247
-
248
- def windows_binary_candidates(binary: str) -> List[Path]:
249
- return _windows_binary_candidates(binary)
250
-
251
-
252
- def local_binary(binary: str) -> Optional[str]:
253
- return _local_binary(binary)
254
-
255
-
256
- def find_lmstudio_cli() -> Optional[str]:
257
- return _find_lmstudio_cli()
258
-
259
-
260
- def vllm_executable() -> Optional[str]:
261
- return _vllm_executable()
262
-
263
-
264
- def vllm_metal_python() -> Optional[str]:
265
- return _vllm_metal_python()
266
-
267
-
268
- def _json_request(
269
- url: str,
270
- *,
271
- method: str = "GET",
272
- payload: Optional[Dict[str, Any]] = None,
273
- headers: Optional[Dict[str, str]] = None,
274
- timeout: float = 10.0,
275
- ) -> Dict[str, Any]:
276
- data = None
277
- req_headers = dict(headers or {})
278
- if payload is not None:
279
- data = json.dumps(payload).encode("utf-8")
280
- req_headers.setdefault("Content-Type", "application/json")
281
- req = urllib.request.Request(url, data=data, headers=req_headers, method=method)
282
- with urllib.request.urlopen(req, timeout=timeout) as res:
283
- raw = res.read().decode("utf-8", errors="replace")
284
- if not raw.strip():
285
- return {}
286
- return json.loads(raw)
287
-
288
-
289
- def lmstudio_api_base() -> str:
290
- return (os.getenv("LMSTUDIO_BASE_URL") or OPENAI_COMPATIBLE_PROVIDERS["lmstudio"]["base_url"]).rstrip("/")
291
-
292
-
293
- def lmstudio_native_api_base() -> str:
294
- base = lmstudio_api_base()
295
- return base[:-3] if base.endswith("/v1") else base
296
-
297
-
298
- def ensure_lmstudio_server() -> None:
299
- return _ensure_lmstudio_server()
300
-
301
-
302
- _LMSTUDIO_MODELS_CACHE: List[Dict[str, Any]] = []
303
- _LMSTUDIO_MODELS_CACHE_TS: float = 0.0
304
- _LMSTUDIO_MODELS_CACHE_TTL: float = 10.0
305
-
306
-
307
- def get_lmstudio_models(*, force: bool = False) -> List[Dict[str, Any]]:
308
- global _LMSTUDIO_MODELS_CACHE, _LMSTUDIO_MODELS_CACHE_TS
309
- if not force and time.monotonic() - _LMSTUDIO_MODELS_CACHE_TS < _LMSTUDIO_MODELS_CACHE_TTL:
310
- return _LMSTUDIO_MODELS_CACHE
311
- try:
312
- payload = _json_request(
313
- f"{lmstudio_native_api_base()}/api/v1/models",
314
- headers={"Authorization": f"Bearer {os.getenv('LMSTUDIO_API_KEY') or 'lmstudio'}"},
315
- timeout=2.5,
316
- )
317
- except Exception:
318
- return _LMSTUDIO_MODELS_CACHE
319
- models = payload.get("models")
320
- _LMSTUDIO_MODELS_CACHE = models if isinstance(models, list) else []
321
- _LMSTUDIO_MODELS_CACHE_TS = time.monotonic()
322
- return _LMSTUDIO_MODELS_CACHE
323
-
324
-
325
- def _lmstudio_candidate_keys(model_name: str) -> List[str]:
326
- raw = model_name.strip()
327
- if not raw:
328
- return []
329
- slug = raw.split("/")[-1].lower()
330
- slug = slug.replace("-gguf", "").replace("-awq", "")
331
- parts = [p for p in slug.split("-") if p]
332
- candidates = [raw.lower(), slug]
333
- if parts:
334
- candidates.append("-".join(parts[: min(4, len(parts))]))
335
- return list(dict.fromkeys(candidates))
336
-
337
-
338
- def _find_lmstudio_model_key(model_name: str, models: List[Dict[str, Any]]) -> Optional[str]:
339
- if not models:
340
- return None
341
- candidate_keys = _lmstudio_candidate_keys(model_name)
342
- exact = []
343
- fuzzy = []
344
- for item in models:
345
- key = str(item.get("key") or "").strip()
346
- display_name = str(item.get("display_name") or "").strip()
347
- haystacks = [key.lower(), display_name.lower()]
348
- if any(raw == key.lower() for raw in candidate_keys):
349
- exact.append(key)
350
- continue
351
- if any(token and token in hay for token in candidate_keys for hay in haystacks):
352
- fuzzy.append(key)
353
- return next(iter(exact or fuzzy), None)
354
-
355
-
356
- def ensure_lmstudio_model(model_name: str) -> Dict[str, Any]:
357
- ensure_lmstudio_server()
358
- auth_header = {"Authorization": f"Bearer {os.getenv('LMSTUDIO_API_KEY') or 'lmstudio'}"}
359
- models = get_lmstudio_models()
360
- found_key = _find_lmstudio_model_key(model_name, models)
361
- model_key = found_key or model_name
362
-
363
- if not found_key:
364
- try:
365
- job = _json_request(
366
- f"{lmstudio_native_api_base()}/api/v1/models/download",
367
- method="POST",
368
- payload={"model": model_name},
369
- headers=auth_header,
370
- timeout=30,
371
- )
372
- except urllib.error.HTTPError as e:
373
- detail = e.read().decode("utf-8", errors="replace")[-2000:]
374
- raise ModelRuntimeError(status_code=500, detail=f"LM Studio 모델 다운로드 실패: {detail or e.reason}")
375
- except Exception as e:
376
- raise ModelRuntimeError(status_code=500, detail=f"LM Studio 모델 다운로드 실패: {e}")
377
-
378
- status = str(job.get("status") or "")
379
- job_id = str(job.get("job_id") or "")
380
- if status not in {"completed", "already_downloaded"} and job_id:
381
- deadline = time.time() + 3600
382
- while time.time() < deadline:
383
- polled = _json_request(
384
- f"{lmstudio_native_api_base()}/api/v1/models/download/status/{job_id}",
385
- headers=auth_header,
386
- timeout=30,
387
- )
388
- polled_status = str(polled.get("status") or "")
389
- if polled_status == "completed":
390
- break
391
- if polled_status == "failed":
392
- raise ModelRuntimeError(status_code=500, detail=f"LM Studio 모델 다운로드 실패: {polled}")
393
- time.sleep(2)
394
- else:
395
- raise ModelRuntimeError(status_code=408, detail="LM Studio 모델 다운로드 시간이 초과되었습니다.")
396
-
397
- models = get_lmstudio_models(force=True)
398
- model_key = _find_lmstudio_model_key(model_name, models) or model_name
399
-
400
- target = next((item for item in models if isinstance(item, dict) and item.get("key") == model_key), None)
401
- loaded_instances = target.get("loaded_instances") if isinstance(target, dict) else None
402
- if loaded_instances:
403
- return {"provider": "lmstudio", "model": model_name, "resolved_model": model_key, "server_ready": True, "cached": True}
404
-
405
- try:
406
- loaded = _json_request(
407
- f"{lmstudio_native_api_base()}/api/v1/models/load",
408
- method="POST",
409
- payload={"model": model_key, "context_length": 4096},
410
- headers=auth_header,
411
- timeout=120,
412
- )
413
- except urllib.error.HTTPError as e:
414
- detail = e.read().decode("utf-8", errors="replace")[-2000:]
415
- raise ModelRuntimeError(status_code=500, detail=f"LM Studio 모델 로드 실패: {detail or e.reason}")
416
- except Exception as e:
417
- raise ModelRuntimeError(status_code=500, detail=f"LM Studio 모델 로드 실패: {e}")
418
-
419
- if str(loaded.get("status") or "") != "loaded":
420
- raise ModelRuntimeError(status_code=500, detail=f"LM Studio 모델 로드 실패: {loaded}")
421
-
422
- return {
423
- "provider": "lmstudio",
424
- "model": model_name,
425
- "resolved_model": model_key,
426
- "instance_id": loaded.get("instance_id"),
427
- "server_ready": True,
428
- "cached": False,
429
- }
430
-
431
- def engine_support_status(engine: str) -> Dict[str, Any]:
432
- return _engine_support_status(engine)
433
-
434
- def hf_model_ready(repo_id: str, provider: str = "local_mlx") -> bool:
435
- model_dir = hf_model_dir(repo_id)
436
- if provider in {"local_mlx", "vllm"} and (not model_dir.exists() or not model_dir.is_dir()):
437
- hf_cache_repo = Path.home() / ".cache" / "huggingface" / "hub" / f"models--{repo_id.replace('/', '--')}"
438
- if hf_cache_repo.exists() and any(hf_cache_repo.glob("snapshots/*")):
439
- if provider == "vllm":
440
- return True
441
- return hf_cache_model_dir(repo_id) is not None
442
- return False
443
- if not model_dir.exists() or not model_dir.is_dir():
444
- return False
445
- if provider == "llamacpp":
446
- return any(model_dir.rglob("*.gguf"))
447
- has_config = (model_dir / "config.json").exists()
448
- has_weights = any(model_dir.glob("*.safetensors")) or any(model_dir.glob("*.bin"))
449
- has_tokenizer = (
450
- (model_dir / "tokenizer.json").exists()
451
- or (model_dir / "tokenizer.model").exists()
452
- or (model_dir / "tokenizer_config.json").exists()
453
- )
454
- return has_config and has_weights and has_tokenizer
455
-
456
-
457
- def model_download_progress_payload(
458
- stage: str,
459
- message: str,
460
- *,
461
- percent: Optional[float] = None,
462
- detail: Optional[str] = None,
463
- downloaded_bytes: Optional[int] = None,
464
- total_bytes: Optional[int] = None,
465
- eta_seconds: Optional[float] = None,
466
- file: Optional[str] = None,
467
- indeterminate: bool = False,
468
- ) -> Dict[str, Any]:
469
- payload: Dict[str, Any] = {
470
- "stage": stage,
471
- "message": message,
472
- "indeterminate": indeterminate,
473
- "ts": time.time(),
474
- }
475
- if percent is not None:
476
- payload["percent"] = max(0, min(100, round(float(percent), 1)))
477
- if detail:
478
- payload["detail"] = detail
479
- if downloaded_bytes is not None:
480
- payload["downloaded_bytes"] = max(0, int(downloaded_bytes))
481
- if total_bytes is not None:
482
- payload["total_bytes"] = max(0, int(total_bytes))
483
- if eta_seconds is not None:
484
- payload["eta_seconds"] = max(0, round(float(eta_seconds)))
485
- if file:
486
- payload["file"] = file
487
- return payload
488
-
489
-
490
- def estimate_eta_seconds(started_at: float, percent: Optional[float]) -> Optional[float]:
491
- if percent is None or percent <= 0 or percent >= 100:
492
- return None
493
- elapsed = max(0.0, time.time() - started_at)
494
- return elapsed * (100.0 - percent) / percent
495
-
496
-
497
- def hf_repo_files_with_sizes(repo_id: str) -> List[Dict[str, Any]]:
498
- from huggingface_hub import HfApi
499
-
500
- api = HfApi()
501
- try:
502
- info = api.model_info(repo_id, files_metadata=True)
503
- files = []
504
- for sibling in getattr(info, "siblings", []) or []:
505
- name = str(getattr(sibling, "rfilename", "") or "").strip()
506
- if not name or name.endswith("/"):
507
- continue
508
- files.append({"name": name, "size": int(getattr(sibling, "size", 0) or 0)})
509
- if files:
510
- return files
511
- except TypeError:
512
- quiet()
513
- except Exception as e:
514
- logging.warning("huggingface model_info failed for %s: %s", repo_id, e)
515
-
516
- return [{"name": str(name), "size": 0} for name in api.list_repo_files(repo_id) if str(name).strip()]
517
-
518
-
519
- def download_hf_model(
520
- repo_id: str,
521
- provider: str = "local_mlx",
522
- progress_emit=None,
523
- ) -> Dict[str, Any]:
524
- if importlib.util.find_spec("huggingface_hub") is None:
525
- raise ModelRuntimeError(status_code=400, detail="huggingface_hub가 없습니다. 먼저 MLX runtime 설치를 진행해 주세요.")
526
-
527
- target_dir = hf_model_dir(repo_id)
528
- if hf_model_ready(repo_id, provider):
529
- cached_dir = hf_cache_model_dir(repo_id) if provider == "local_mlx" else None
530
- resolved_dir = cached_dir or target_dir
531
- if progress_emit:
532
- progress_emit(model_download_progress_payload(
533
- "download",
534
- "이미 다운로드된 모델을 확인했습니다.",
535
- percent=100,
536
- downloaded_bytes=0,
537
- total_bytes=0,
538
- eta_seconds=0,
539
- ))
540
- return {"model": repo_id, "path": str(resolved_dir), "cached": True}
541
-
542
- target_dir.mkdir(parents=True, exist_ok=True)
543
- try:
544
- from huggingface_hub import hf_hub_download
545
-
546
- started_at = time.time()
547
- all_files = hf_repo_files_with_sizes(repo_id)
548
- if provider == "llamacpp":
549
- ggufs = sorted(
550
- [item for item in all_files if str(item["name"]).lower().endswith(".gguf")],
551
- key=lambda item: str(item["name"]),
552
- )
553
- if not ggufs:
554
- raise RuntimeError("GGUF 파일을 찾지 못했습니다.")
555
- preference = ("q4_k_m", "q4_0", "q4_k_s", "q3_k_m", "q2_k")
556
- selected_files = [
557
- next(
558
- (item for pref in preference for item in ggufs if pref in str(item["name"]).lower()),
559
- ggufs[0],
560
- )
561
- ]
562
- else:
563
- selected_files = all_files
564
-
565
- total_bytes = sum(int(item.get("size") or 0) for item in selected_files) or None
566
- downloaded_bytes = 0
567
- total_files = max(1, len(selected_files))
568
- if progress_emit:
569
- progress_emit(model_download_progress_payload(
570
- "download",
571
- "모델 파일 정보를 확인했습니다.",
572
- percent=0,
573
- downloaded_bytes=0,
574
- total_bytes=total_bytes,
575
- indeterminate=total_bytes is None,
576
- ))
577
-
578
- for index, item in enumerate(selected_files, start=1):
579
- filename = str(item["name"])
580
- size = int(item.get("size") or 0)
581
- tqdm_class = None
582
- if progress_emit:
583
- current_percent = (
584
- (downloaded_bytes / total_bytes) * 100 if total_bytes else ((index - 1) / total_files) * 100
585
- )
586
- progress_emit(model_download_progress_payload(
587
- "download",
588
- "모델 다운로드 중입니다.",
589
- percent=current_percent,
590
- detail=filename,
591
- downloaded_bytes=downloaded_bytes,
592
- total_bytes=total_bytes,
593
- eta_seconds=estimate_eta_seconds(started_at, current_percent),
594
- file=filename,
595
- indeterminate=total_bytes is None and total_files <= 1,
596
- ))
597
- try:
598
- from tqdm.auto import tqdm as base_tqdm
599
-
600
- downloaded_before = downloaded_bytes
601
- last_emit = {"at": 0.0, "percent": -1.0}
602
-
603
- def emit_byte_progress(
604
- done_bytes: float,
605
- # Bound per iteration: this callback outlives the loop
606
- # body when a download runs long, and late binding
607
- # would report every file's progress against the last
608
- # file's offsets.
609
- downloaded_before: int = downloaded_before,
610
- size: Any = size,
611
- index: int = index,
612
- last_emit: dict = last_emit,
613
- filename: str = filename,
614
- ) -> None:
615
- done = max(0, int(done_bytes or 0))
616
- if total_bytes:
617
- aggregate = min(total_bytes, downloaded_before + done)
618
- percent = (aggregate / total_bytes) * 100
619
- else:
620
- file_total = size or done
621
- file_ratio = min(1.0, done / file_total) if file_total else 0.0
622
- aggregate = downloaded_before + done
623
- percent = ((index - 1) + file_ratio) / total_files * 100
624
- now = time.time()
625
- if percent < 100 and now - last_emit["at"] < 0.5 and percent - last_emit["percent"] < 0.3:
626
- return
627
- last_emit["at"] = now
628
- last_emit["percent"] = percent
629
- progress_emit(model_download_progress_payload(
630
- "download",
631
- "모델 다운로드 중입니다.",
632
- percent=percent,
633
- detail=filename,
634
- downloaded_bytes=aggregate,
635
- total_bytes=total_bytes,
636
- eta_seconds=estimate_eta_seconds(started_at, percent),
637
- file=filename,
638
- indeterminate=total_bytes is None and total_files <= 1,
639
- ))
640
-
641
- class ProgressTqdm(base_tqdm):
642
- def update(self, n=1):
643
- result = super().update(n)
644
- emit_byte_progress(float(getattr(self, "n", 0) or 0))
645
- return result
646
-
647
- tqdm_class = ProgressTqdm
648
- except Exception:
649
- tqdm_class = None
650
- local_path = hf_hub_download(
651
- repo_id=repo_id,
652
- filename=filename,
653
- local_dir=str(target_dir),
654
- tqdm_class=tqdm_class,
655
- )
656
- if size <= 0:
657
- try:
658
- size = Path(local_path).stat().st_size
659
- except OSError:
660
- size = 0
661
- downloaded_bytes += size
662
- if progress_emit:
663
- current_percent = (
664
- (downloaded_bytes / total_bytes) * 100 if total_bytes else (index / total_files) * 100
665
- )
666
- progress_emit(model_download_progress_payload(
667
- "download",
668
- "모델 다운로드 중입니다.",
669
- percent=current_percent,
670
- detail=filename,
671
- downloaded_bytes=downloaded_bytes,
672
- total_bytes=total_bytes,
673
- eta_seconds=estimate_eta_seconds(started_at, current_percent),
674
- file=filename,
675
- indeterminate=False,
676
- ))
677
-
678
- if progress_emit:
679
- progress_emit(model_download_progress_payload(
680
- "download",
681
- "모델 다운로드가 완료되었습니다.",
682
- percent=100,
683
- downloaded_bytes=downloaded_bytes,
684
- total_bytes=total_bytes or downloaded_bytes,
685
- eta_seconds=0,
686
- ))
687
- except Exception as e:
688
- raise ModelRuntimeError(status_code=500, detail=f"{repo_id} 다운로드 실패: {str(e)[-2000:]}")
689
-
690
- if not hf_model_ready(repo_id, provider):
691
- raise ModelRuntimeError(status_code=500, detail=f"{repo_id} 다운로드가 완료되지 않았습니다. 모델 파일을 찾지 못했습니다.")
692
-
693
- return {"model": repo_id, "path": str(target_dir), "cached": False}
694
-
695
-
696
- def pull_ollama_model_with_progress(model_name: str, progress_emit=None) -> Dict[str, Any]:
697
- return _pull_ollama_model_with_progress(model_name, progress_emit)
698
-
699
-
700
- def get_ollama_pulled_models() -> set:
701
- return _get_ollama_pulled_models()
702
-
703
-
704
- def get_openai_compatible_server_models(provider: str) -> List[str]:
705
- return _get_openai_compatible_server_models(provider)
706
-
707
-
708
- def ensure_ollama_server() -> None:
709
- return _ensure_ollama_server()
710
-
711
-
712
- def wait_for_openai_compatible_server(provider: str, model_name: Optional[str] = None, timeout: int = 45) -> bool:
713
- return _wait_for_openai_compatible_server(provider, model_name=model_name, timeout=timeout)
714
-
715
-
716
- def ensure_vllm_server(model_name: str) -> None:
717
- return _ensure_vllm_server(model_name)
718
-
719
-
720
- def ensure_llamacpp_server(model_name: str) -> None:
721
- return _ensure_llamacpp_server(model_name)
722
-
723
-
724
- def _safe_engine_install_plan(
725
- engine: str,
726
- *,
727
- base_dir: Path,
728
- ) -> Optional[Dict[str, Any]]:
729
- try:
730
- return _engine_install_plan(engine, base_dir=base_dir)
731
- except Exception:
732
- return None
733
-
734
-
735
- def engine_installed(engine: str) -> bool:
736
- if engine == "local_mlx":
737
- return bool(
738
- importlib.util.find_spec("mlx")
739
- and (importlib.util.find_spec("mlx_vlm") or importlib.util.find_spec("mlx_lm"))
740
- )
741
- if engine == "ollama":
742
- return local_binary("ollama") is not None
743
- if engine == "vllm":
744
- return vllm_metal_python() is not None or vllm_executable() is not None or importlib.util.find_spec("vllm") is not None
745
- if engine == "lmstudio":
746
- return find_lmstudio_cli() is not None or Path("/Applications/LM Studio.app").exists()
747
- if engine == "llamacpp":
748
- return shutil.which("llama-server") is not None
749
- if engine in {"openai", "openrouter", "groq", "together", "xai"}:
750
- return AsyncOpenAI is not None
751
- return False
752
-
753
- def engine_status(
754
- *,
755
- state: ModelRuntimeState,
756
- cloud_verify_cache: Optional[Dict[str, Dict[str, Any]]] = None,
757
- ) -> List[Dict]:
758
- r = state.router
759
- verify_cache = cloud_verify_cache or {}
760
- cloud_models = r.detected_cloud_models() if r else []
761
- cloud_by_provider: Dict[str, List[Dict[str, Any]]] = {}
762
- for model in cloud_models:
763
- cloud_by_provider.setdefault(model["provider"], []).append(model)
764
-
765
- ollama_installed = engine_installed("ollama")
766
- pulled = get_ollama_pulled_models() if ollama_installed else set()
767
- ollama_models = []
768
- for m in ENGINE_MODEL_CATALOG["ollama"]:
769
- pull_name = m["id"].removeprefix("ollama:")
770
- ollama_models.append({**m, "pulled": pull_name in pulled})
771
- ollama_models = filter_lower_family_versions(ollama_models)
772
-
773
- HF_MODELS_ROOT.mkdir(parents=True, exist_ok=True)
774
- mlx_models = []
775
- for m in ENGINE_MODEL_CATALOG.get("local_mlx", []):
776
- repo_id = m["id"]
777
- mlx_models.append({**m, "pulled": hf_model_ready(repo_id, "local_mlx")})
778
- mlx_models = filter_lower_family_versions(mlx_models)
779
-
780
- vllm_models = []
781
- for m in ENGINE_MODEL_CATALOG.get("vllm", []):
782
- repo_id = m["id"].removeprefix("vllm:")
783
- vllm_models.append({**m, "pulled": hf_model_ready(repo_id, "vllm")})
784
- vllm_models = filter_lower_family_versions(vllm_models)
785
-
786
- lmstudio_models = []
787
- downloaded_lmstudio = get_lmstudio_models()
788
- downloaded_by_key: Dict[str, Dict[str, Any]] = {}
789
- for item in downloaded_lmstudio:
790
- key = str(item.get("key") or "").strip()
791
- if not key:
792
- continue
793
- downloaded_by_key[key] = item
794
- loaded_instances = item.get("loaded_instances") or []
795
- lmstudio_models.append({
796
- "id": f"lmstudio:{key}",
797
- "name": item.get("display_name") or f"LM Studio · {key}",
798
- "family": item.get("architecture") or item.get("publisher") or "LM Studio",
799
- "tag": "loaded-server-model" if loaded_instances else "downloaded",
800
- "size": item.get("params_string") or item.get("format") or "LM Studio",
801
- "pullable": True,
802
- "pulled": True,
803
- })
804
-
805
- if not lmstudio_models:
806
- for m in ENGINE_MODEL_CATALOG.get("lmstudio", []):
807
- lmstudio_models.append({**m, "pulled": False})
808
- else:
809
- known_ids = {item["id"] for item in lmstudio_models}
810
- for m in ENGINE_MODEL_CATALOG.get("lmstudio", []):
811
- repo_id = m["id"].removeprefix("lmstudio:")
812
- if f"lmstudio:{repo_id}" not in known_ids and repo_id not in downloaded_by_key:
813
- lmstudio_models.append({**m, "pulled": False})
814
- lmstudio_models = filter_lower_family_versions(lmstudio_models)
815
-
816
- llamacpp_models = []
817
- for m in ENGINE_MODEL_CATALOG.get("llamacpp", []):
818
- repo_id = m["id"].removeprefix("llamacpp:")
819
- llamacpp_models.append({**m, "pulled": hf_model_ready(repo_id, "llamacpp")})
820
- llamacpp_models = filter_lower_family_versions(llamacpp_models)
821
-
822
- local_server_specs: List[Dict[str, Any]] = [
823
- {
824
- "id": "vllm",
825
- "name": "vLLM",
826
- "description": "vLLM OpenAI 호환 서버(예: http://localhost:8000/v1)에 연결합니다.",
827
- "requires": "VLLM_BASE_URL",
828
- "note": engine_support_status("vllm").get("reason"),
829
- },
830
- {
831
- "id": "lmstudio",
832
- "name": "LM Studio",
833
- "description": "LM Studio 로컬 OpenAI 호환 서버에 연결합니다.",
834
- "requires": "LMSTUDIO_BASE_URL",
835
- "note": (
836
- "다운로드된 모델은 자동 감지하고, 선택 시 필요하면 다운로드 후 바로 로드합니다."
837
- if downloaded_lmstudio else
838
- "LM Studio 설치 후 모델을 선택하면 Local Server 시작, 다운로드, 로드를 자동으로 진행합니다."
839
- ),
840
- "server_ready": bool(downloaded_lmstudio),
841
- },
842
- {
843
- "id": "llamacpp",
844
- "name": "llama.cpp",
845
- "description": "llama.cpp 서버(OpenAI 호환 /v1)에 연결합니다.",
846
- "requires": "LLAMACPP_BASE_URL",
847
- },
848
- ]
849
-
850
- engines = [
851
- {
852
- "id": "local_mlx",
853
- "name": "MLX",
854
- "kind": "local",
855
- "description": "Apple Silicon GPU에서 MLX-VLM 모델을 직접 실행하고, Gemma 4는 필요 시 MLX-LM 텍스트 경로로 재시도합니다.",
856
- "installed": engine_installed("local_mlx"),
857
- "installable": True,
858
- "install_label": ENGINE_INSTALLERS["local_mlx"]["label"],
859
- "install_plan": _safe_engine_install_plan("local_mlx", base_dir=state.BASE_DIR),
860
- "models": mlx_models,
861
- },
862
- {
863
- "id": "ollama",
864
- "name": "Ollama",
865
- "kind": "local-server",
866
- "description": "Ollama 로컬 서버를 OpenAI 호환 엔진처럼 사용합니다.",
867
- "installed": ollama_installed,
868
- "installable": True,
869
- "install_label": ENGINE_INSTALLERS["ollama"]["label"],
870
- "install_plan": _safe_engine_install_plan("ollama", base_dir=state.BASE_DIR),
871
- "models": ollama_models,
872
- },
873
- ]
874
- for spec in local_server_specs:
875
- support = engine_support_status(spec["id"])
876
- engines.append({
877
- "id": spec["id"],
878
- "name": spec["name"],
879
- "kind": "local-server",
880
- "description": spec["description"],
881
- "installed": engine_installed(spec["id"]),
882
- "supported": support["supported"],
883
- "support_reason": support["reason"],
884
- "installable": support["supported"] and spec["id"] in ENGINE_INSTALLERS,
885
- "install_label": ENGINE_INSTALLERS.get(spec["id"], {}).get("label"),
886
- "install_plan": (
887
- _safe_engine_install_plan(spec["id"], base_dir=state.BASE_DIR)
888
- if spec["id"] in ENGINE_INSTALLERS
889
- else None
890
- ),
891
- "requires": spec["requires"],
892
- "models": (
893
- vllm_models if spec["id"] == "vllm"
894
- else lmstudio_models if spec["id"] == "lmstudio"
895
- else llamacpp_models if spec["id"] == "llamacpp"
896
- else ENGINE_MODEL_CATALOG.get(spec["id"], [])
897
- ),
898
- "note": spec.get("note") or support["reason"] or f"{spec['requires']} 설정 시 활성화됩니다.",
899
- "server_ready": spec.get("server_ready"),
900
- })
901
- for provider in ["openai", "openrouter", "groq", "together", "xai"]:
902
- env_key = next((item.get("requires") for item in cloud_by_provider.get(provider, []) if item.get("requires")), None)
903
- provider_models = []
904
- for model in cloud_by_provider.get(provider, []):
905
- cache = verify_cache.get(str(model.get("id") or ""))
906
- provider_models.append({
907
- **model,
908
- "verified": cache.get("ok") if cache else None,
909
- "verify_reason": cache.get("reason") if cache else None,
910
- })
911
- engines.append({
912
- "id": provider,
913
- "name": provider.title(),
914
- "kind": "cloud",
915
- "description": "OpenAI 호환 Chat Completions API로 cloud LLM을 실행합니다.",
916
- "installed": engine_installed(provider),
917
- "installable": True,
918
- "install_label": ENGINE_INSTALLERS[provider]["label"],
919
- "install_plan": _safe_engine_install_plan(provider, base_dir=state.BASE_DIR),
920
- "requires": env_key,
921
- "models": provider_models,
922
- })
923
- return engines
924
-
925
- def runtime_features(*, state: ModelRuntimeState) -> Dict:
926
- s = state
927
- r = s.router
928
- return {
929
- "mode": s.APP_MODE,
930
- "public": s.IS_PUBLIC_MODE,
931
- "host": s.DEFAULT_HOST,
932
- "port": s.DEFAULT_PORT,
933
- "data_dir": str(s.DATA_DIR),
934
- "telegram_enabled": s.ENABLE_TELEGRAM,
935
- "graph_enabled": s.ENABLE_GRAPH,
936
- "autoload_models": s.AUTOLOAD_MODELS,
937
- "model_idle_unload_seconds": s.MODEL_IDLE_UNLOAD_SECONDS,
938
- "allow_model_downloads": s.ALLOW_MODEL_DOWNLOADS,
939
- "model_download_timeout": s.MODEL_DOWNLOAD_TIMEOUT,
940
- "model_memory_policy": r.model_memory_policy() if r else None,
941
- "allow_local_models": s.ALLOW_LOCAL_MODELS,
942
- "security": {
943
- "host": s.DEFAULT_HOST,
944
- "require_auth": s.REQUIRE_AUTH,
945
- "invite_gate_enabled": s.INVITE_GATE_ENABLED,
946
- "keyring_available": s.keyring is not None,
947
- "plaintext_api_keys_allowed": s.ALLOW_PLAINTEXT_API_KEYS,
948
- "cors_allow_network": s.CORS_ALLOW_NETWORK,
949
- },
950
- "default_model": s.PUBLIC_MODEL if s.IS_PUBLIC_MODE else s.LOCAL_MODEL,
951
- "local_only_features": {
952
- "mlx": s.ALLOW_LOCAL_MODELS and not s.IS_PUBLIC_MODE,
953
- "telegram_bridge": s.ENABLE_TELEGRAM,
954
- "desktop_chrome_bridge": not s.IS_PUBLIC_MODE,
955
- "computer_use_bridge": not s.IS_PUBLIC_MODE,
956
- },
957
- "public_features": {
958
- "web_ui": True,
959
- "openai_compatible_models": True,
960
- "persistent_data_dir": str(s.DATA_DIR),
961
- },
962
- }
963
-
964
- def install_engine(
965
- engine: str,
966
- confirmation_token: Optional[str] = None,
967
- *,
968
- state: ModelRuntimeState,
969
- ) -> Dict:
970
- return _install_engine(
971
- engine,
972
- confirmation_token=confirmation_token,
973
- base_dir=state.BASE_DIR,
974
- )
975
-
976
-
977
- def _resolve_model_alias(model_id: str, engine: Optional[str] = None) -> str:
978
- raw = model_id.strip()
979
- engine_hint = (engine or "").strip().lower()
980
- provider: Optional[str] = None
981
- model_name = raw
982
- if ":" in raw:
983
- prefix, rest = raw.split(":", 1)
984
- prefix = prefix.strip().lower()
985
- if prefix in {"ollama", "vllm", "lmstudio", "llamacpp", "local_mlx", "mlx"}:
986
- provider = "local_mlx" if prefix in {"local_mlx", "mlx"} else prefix
987
- model_name = rest.strip()
988
- provider = provider or ("local_mlx" if engine_hint in {"", "local_mlx", "mlx"} else engine_hint)
989
- aliases = MODEL_ENGINE_ALIASES.get(model_name.lower())
990
- if not aliases:
991
- return raw
992
- mapped = aliases.get(provider)
993
- if not mapped:
994
- return raw
995
- return mapped if provider == "local_mlx" else f"{provider}:{mapped}"
996
-
997
-
998
- def normalize_local_model_request(model_id: str, engine: Optional[str] = None) -> str:
999
- model_id = _resolve_model_alias(model_id, engine)
1000
- engine = (engine or "").strip().lower()
1001
- if engine in {"local_mlx", "mlx"} and model_id.startswith(("local_mlx:", "mlx:")):
1002
- return model_id.split(":", 1)[1].strip()
1003
- if engine and engine not in {"local_mlx", "mlx"} and ":" not in model_id:
1004
- return f"{engine}:{model_id}"
1005
- return model_id
1006
-
1007
-
1008
- def ensure_engine_ready(engine: str, *, state: ModelRuntimeState) -> Dict[str, Any]:
1009
- engine = "local_mlx" if engine == "mlx" else engine
1010
- if engine not in ENGINE_INSTALLERS and engine not in OPENAI_COMPATIBLE_PROVIDERS:
1011
- raise ModelRuntimeError(status_code=400, detail=f"지원하지 않는 엔진입니다: {engine}")
1012
- support = engine_support_status(engine)
1013
- if not support["supported"]:
1014
- raise ModelRuntimeError(status_code=400, detail=str(support["reason"]))
1015
-
1016
- if engine_installed(engine):
1017
- if engine == "local_mlx":
1018
- ensure_mlx_runtime()
1019
- return {"engine": engine, "installed": True, "installed_now": False}
1020
-
1021
- if engine not in ENGINE_INSTALLERS:
1022
- raise ModelRuntimeError(status_code=400, detail=f"{engine} 엔진 설치 방법이 등록되어 있지 않습니다.")
1023
-
1024
- result = install_engine(engine, state=state)
1025
- if result.get("returncode") not in (0, None) or not engine_installed(engine):
1026
- detail = result.get("stderr") or result.get("stdout") or f"{engine} 설치에 실패했습니다."
1027
- raise ModelRuntimeError(status_code=500, detail=str(detail)[-2000:])
1028
-
1029
- if engine == "local_mlx":
1030
- ensure_mlx_runtime()
1031
- return {"engine": engine, "installed": True, "installed_now": True, "install": result}
1032
-
1033
-
1034
- def build_model_resolution(
1035
- input_id: str,
1036
- engine: Optional[str],
1037
- *,
1038
- user_email: Optional[str] = None,
1039
- display_name: Optional[str] = None,
1040
- ) -> _ModelResolution:
1041
- """피드백 #1/#2 공용 ModelResolution 생성기.
1042
-
1043
- 사용자가 클릭한 input_id + engine 힌트를 받아 모든 단계가 공유할
1044
- canonical identity를 만든다.
1045
- """
1046
- normalized = normalize_local_model_request(input_id, engine)
1047
- return _ModelResolution.from_request(
1048
- normalized,
1049
- engine=engine,
1050
- user_email=user_email,
1051
- display_name=display_name or input_id,
1052
- engine_aliases=MODEL_ENGINE_ALIASES,
1053
- )
1054
-
1055
-
1056
- _LOCAL_SMOKE_ENGINES = {"local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"}
1057
-
1058
-
1059
- async def _smoke_test_loaded_model(
1060
- resolution: _ModelResolution,
1061
- *,
1062
- api_key_override: Optional[str] = None,
1063
- state: ModelRuntimeState,
1064
- ) -> Dict[str, Any]:
1065
- # Delegated to model_engines for server decomp
1066
- from .model_engines import _smoke_test_loaded_model as _impl_smoke
1067
- return await _impl_smoke(
1068
- resolution,
1069
- api_key_override=api_key_override,
1070
- model_router=state.router,
1071
- )
1072
-
1073
-
1074
- async def prepare_and_load_model(
1075
- model_id: str,
1076
- request: Any,
1077
- engine: Optional[str] = None,
1078
- user_email: Optional[str] = None,
1079
- adapter_path: Optional[str] = None,
1080
- draft_model_id: Optional[str] = None,
1081
- allow_download: bool = False,
1082
- *,
1083
- state: ModelRuntimeState,
1084
- ) -> Dict[str, Any]:
1085
- from .model_loading import prepare_and_load_model as _impl
1086
-
1087
- return await _impl(
1088
- model_id,
1089
- request,
1090
- engine=engine,
1091
- user_email=user_email,
1092
- adapter_path=adapter_path,
1093
- draft_model_id=draft_model_id,
1094
- allow_download=allow_download,
1095
- runtime_state=state,
1096
- )
1097
-
1098
-
1099
- def sse_event(event: str, data: Dict[str, Any]) -> str:
1100
- return f"event: {event}\ndata: {json.dumps(data, ensure_ascii=False)}\n\n"
1101
-
1102
-
1103
- async def prepare_and_load_model_stream(
1104
- model_id: str,
1105
- request: Any,
1106
- engine: Optional[str] = None,
1107
- user_email: Optional[str] = None,
1108
- allow_download: bool = False,
1109
- *,
1110
- state: ModelRuntimeState,
1111
- ) -> AsyncIterator[str]:
1112
- from .model_loading import prepare_and_load_model_stream as _impl
1113
-
1114
- async for event in _impl(
1115
- model_id,
1116
- request,
1117
- engine=engine,
1118
- user_email=user_email,
1119
- allow_download=allow_download,
1120
- runtime_state=state,
1121
- ):
1122
- yield event
1123
-
1124
-
1125
- CLOUD_VERIFY_TTL_SECONDS = 600
1126
-
1127
- async def _probe_cloud_model(model_ref: str) -> Dict[str, Any]:
1128
- provider, model_name = parse_model_ref(model_ref)
1129
- config = OPENAI_COMPATIBLE_PROVIDERS.get(provider)
1130
- if not config:
1131
- return {"ok": False, "reason": f"Unsupported provider: {provider}"}
1132
-
1133
- api_key = os.getenv(config["env_key"]) or config.get("api_key_fallback")
1134
- if not api_key:
1135
- return {"ok": False, "reason": f"Missing API key: {config['env_key']}"}
1136
-
1137
- base_url = os.getenv(config.get("base_url_env", "")) if config.get("base_url_env") else None
1138
- base_url = base_url or config.get("base_url")
1139
- try:
1140
- # base_url is passed only when configured: an explicit None is not
1141
- # the same as omitting the argument.
1142
- client = (
1143
- AsyncOpenAI(api_key=api_key, base_url=base_url)
1144
- if base_url
1145
- else AsyncOpenAI(api_key=api_key)
1146
- )
1147
- await asyncio.wait_for(
1148
- client.chat.completions.create(
1149
- model=model_name,
1150
- messages=[{"role": "user", "content": "ping"}],
1151
- max_tokens=1,
1152
- temperature=0,
1153
- ),
1154
- timeout=15,
1155
- )
1156
- return {"ok": True, "reason": "ok"}
1157
- except Exception as e:
1158
- return {"ok": False, "reason": str(e)[:220]}
1159
-
1160
-
1161
- async def verify_cloud_models(
1162
- force: bool = False,
1163
- provider_filter: Optional[str] = None,
1164
- *,
1165
- state: ModelRuntimeState,
1166
- cache: Dict[str, Dict[str, Any]],
1167
- ) -> Dict[str, Dict]:
1168
- now = time.time()
1169
- r = state.router
1170
- cloud_items = [item for item in (r.detected_cloud_models() if r else []) if item.get("tag") == "cloud"]
1171
- if provider_filter:
1172
- cloud_items = [item for item in cloud_items if item.get("provider") == provider_filter]
1173
-
1174
- results: Dict[str, Dict] = {}
1175
- for item in cloud_items:
1176
- model_ref = item["id"]
1177
- cached = cache.get(model_ref)
1178
- if not force and cached and (now - cached.get("ts", 0) <= CLOUD_VERIFY_TTL_SECONDS):
1179
- results[model_ref] = cached
1180
- continue
1181
- if item.get("available") is False:
1182
- record = {"ok": False, "reason": item.get("requires") or "API key missing", "ts": now}
1183
- cache[model_ref] = record
1184
- results[model_ref] = record
1185
- continue
1186
- probe = await _probe_cloud_model(model_ref)
1187
- record = {"ok": bool(probe.get("ok")), "reason": probe.get("reason", ""), "ts": now}
1188
- cache[model_ref] = record
1189
- results[model_ref] = record
1190
- return results
1191
-
1192
-
1193
- @dataclass(slots=True)
1194
- class ModelRuntimeService:
1195
- """Bound model operations for one explicitly configured application.
1196
-
1197
- All configuration and app-owned callables live on ``state``. Operational
1198
- verification cache data belongs to this service instance, so creating a
1199
- second ASGI app cannot inherit credentials, routers, or probe results from
1200
- the first one.
1201
- """
1202
-
1203
- state: ModelRuntimeState
1204
- _cloud_verify_cache: Dict[str, Dict[str, Any]] = field(default_factory=dict)
1205
-
1206
- def runtime_features(self) -> Dict[str, Any]:
1207
- return runtime_features(state=self.state)
1208
-
1209
- def engine_status(self) -> List[Dict[str, Any]]:
1210
- return engine_status(
1211
- state=self.state,
1212
- cloud_verify_cache=self._cloud_verify_cache,
1213
- )
1214
-
1215
- def install_engine(
1216
- self,
1217
- engine: str,
1218
- confirmation_token: Optional[str] = None,
1219
- ) -> Dict[str, Any]:
1220
- return install_engine(
1221
- engine,
1222
- confirmation_token=confirmation_token,
1223
- state=self.state,
1224
- )
1225
-
1226
- async def verify_cloud_models(
1227
- self,
1228
- force: bool = False,
1229
- provider_filter: Optional[str] = None,
1230
- ) -> Dict[str, Dict[str, Any]]:
1231
- return await verify_cloud_models(
1232
- force=force,
1233
- provider_filter=provider_filter,
1234
- state=self.state,
1235
- cache=self._cloud_verify_cache,
1236
- )
1237
-
1238
- async def prepare_and_load_model(
1239
- self,
1240
- model_id: str,
1241
- request: Any,
1242
- engine: Optional[str] = None,
1243
- user_email: Optional[str] = None,
1244
- adapter_path: Optional[str] = None,
1245
- draft_model_id: Optional[str] = None,
1246
- allow_download: bool = False,
1247
- ) -> Dict[str, Any]:
1248
- return await prepare_and_load_model(
1249
- model_id,
1250
- request,
1251
- engine=engine,
1252
- user_email=user_email,
1253
- adapter_path=adapter_path,
1254
- draft_model_id=draft_model_id,
1255
- allow_download=allow_download,
1256
- state=self.state,
1257
- )
1258
-
1259
- async def prepare_and_load_model_stream(
1260
- self,
1261
- model_id: str,
1262
- request: Any,
1263
- engine: Optional[str] = None,
1264
- user_email: Optional[str] = None,
1265
- allow_download: bool = False,
1266
- ) -> AsyncIterator[str]:
1267
- async for event in prepare_and_load_model_stream(
1268
- model_id,
1269
- request,
1270
- engine=engine,
1271
- user_email=user_email,
1272
- allow_download=allow_download,
1273
- state=self.state,
1274
- ):
1275
- yield event
1276
-
1277
-
1278
- def build_model_runtime(**deps: Any) -> ModelRuntimeService:
1279
- """Build the application's isolated model runtime service."""
1280
-
1281
- return ModelRuntimeService(create_model_runtime_state(**deps))