ltcai 11.2.0 → 11.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (247) hide show
  1. package/README.md +46 -53
  2. package/docs/CHANGELOG.md +61 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/MULTI_AGENT_RUNTIME.md +1 -1
  6. package/docs/ONBOARDING.md +1 -1
  7. package/docs/OPERATIONS.md +6 -2
  8. package/docs/PERMISSION_MODE.md +1 -1
  9. package/docs/TRUST_MODEL.md +1 -1
  10. package/docs/WHY_LATTICE.md +1 -1
  11. package/docs/kg-schema.md +2 -2
  12. package/docs/v11.3.0_PLAN.md +202 -0
  13. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
  14. package/lattice_brain/__init__.py +1 -1
  15. package/lattice_brain/graph/_kg_common/__init__.py +287 -0
  16. package/lattice_brain/graph/_kg_common/extraction.py +516 -0
  17. package/lattice_brain/graph/_kg_common/relations.py +161 -0
  18. package/lattice_brain/graph/_kg_common/text.py +479 -0
  19. package/lattice_brain/graph/discovery_index/__init__.py +35 -0
  20. package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
  21. package/lattice_brain/graph/discovery_index/extract.py +137 -0
  22. package/lattice_brain/graph/discovery_index/scan.py +411 -0
  23. package/lattice_brain/graph/discovery_index/upsert.py +495 -0
  24. package/lattice_brain/graph/projection/__init__.py +42 -0
  25. package/lattice_brain/graph/projection/curation.py +500 -0
  26. package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
  27. package/lattice_brain/graph/retrieval/__init__.py +54 -0
  28. package/lattice_brain/graph/retrieval/context.py +197 -0
  29. package/lattice_brain/graph/retrieval/graph_view.py +319 -0
  30. package/lattice_brain/graph/retrieval/hybrid.py +488 -0
  31. package/lattice_brain/graph/retrieval/maintenance.py +121 -0
  32. package/lattice_brain/graph/retrieval/signals.py +95 -0
  33. package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
  34. package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
  35. package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
  36. package/lattice_brain/graph/retrieval_vector/search.py +560 -0
  37. package/lattice_brain/graph/retrieval_vector/status.py +374 -0
  38. package/lattice_brain/ingestion/__init__.py +130 -0
  39. package/lattice_brain/ingestion/_contract.py +90 -0
  40. package/lattice_brain/ingestion/constants.py +127 -0
  41. package/lattice_brain/ingestion/folder_scan.py +57 -0
  42. package/lattice_brain/ingestion/folders.py +258 -0
  43. package/lattice_brain/ingestion/hashing.py +26 -0
  44. package/lattice_brain/ingestion/jobs_api.py +107 -0
  45. package/lattice_brain/ingestion/models.py +80 -0
  46. package/lattice_brain/ingestion/pipeline.py +486 -0
  47. package/lattice_brain/ingestion/quality.py +209 -0
  48. package/lattice_brain/ingestion/routing.py +295 -0
  49. package/lattice_brain/multimodal/__init__.py +164 -0
  50. package/lattice_brain/multimodal/audio.py +77 -0
  51. package/lattice_brain/multimodal/common.py +118 -0
  52. package/lattice_brain/multimodal/images.py +498 -0
  53. package/lattice_brain/multimodal/ports.py +169 -0
  54. package/lattice_brain/multimodal/video.py +410 -0
  55. package/lattice_brain/portability/__init__.py +90 -0
  56. package/lattice_brain/portability/_contract.py +42 -0
  57. package/lattice_brain/portability/backups.py +338 -0
  58. package/lattice_brain/portability/bundles.py +136 -0
  59. package/lattice_brain/portability/constants.py +93 -0
  60. package/lattice_brain/portability/fsops.py +138 -0
  61. package/lattice_brain/portability/service.py +41 -0
  62. package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
  63. package/lattice_brain/runtime/__init__.py +1 -1
  64. package/lattice_brain/runtime/multi_agent.py +1 -1
  65. package/latticeai/__init__.py +1 -1
  66. package/latticeai/api/chronicle.py +63 -0
  67. package/latticeai/core/agent/__init__.py +93 -0
  68. package/latticeai/core/agent/_contract.py +79 -0
  69. package/latticeai/core/agent/context.py +57 -0
  70. package/latticeai/core/agent/deps.py +125 -0
  71. package/latticeai/core/agent/execution.py +622 -0
  72. package/latticeai/core/agent/planning.py +145 -0
  73. package/latticeai/core/agent/recovery.py +157 -0
  74. package/latticeai/core/agent/runtime.py +210 -0
  75. package/latticeai/core/agent/verification.py +231 -0
  76. package/latticeai/core/embedding_providers/__init__.py +151 -0
  77. package/latticeai/core/embedding_providers/base.py +199 -0
  78. package/latticeai/core/embedding_providers/captions.py +162 -0
  79. package/latticeai/core/embedding_providers/profiles.py +126 -0
  80. package/latticeai/core/embedding_providers/text.py +350 -0
  81. package/latticeai/core/embedding_providers/vision.py +352 -0
  82. package/latticeai/core/file_generation/__init__.py +115 -0
  83. package/latticeai/core/file_generation/bundles.py +76 -0
  84. package/latticeai/core/file_generation/extraction.py +154 -0
  85. package/latticeai/core/file_generation/inference.py +235 -0
  86. package/latticeai/core/file_generation/orchestration.py +152 -0
  87. package/latticeai/core/file_generation/prompting.py +117 -0
  88. package/latticeai/core/file_generation/repair.py +114 -0
  89. package/latticeai/core/file_generation/sanitize.py +61 -0
  90. package/latticeai/core/file_generation/validation.py +201 -0
  91. package/latticeai/core/legacy_compatibility.py +1 -1
  92. package/latticeai/core/marketplace.py +1 -1
  93. package/latticeai/core/messages.py +9 -0
  94. package/latticeai/core/workspace_os_constants.py +1 -1
  95. package/latticeai/integrations/telegram_bot/__init__.py +123 -0
  96. package/latticeai/integrations/telegram_bot/__main__.py +17 -0
  97. package/latticeai/integrations/telegram_bot/config.py +86 -0
  98. package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
  99. package/latticeai/integrations/telegram_bot/flows.py +478 -0
  100. package/latticeai/integrations/telegram_bot/helpers.py +322 -0
  101. package/latticeai/integrations/telegram_bot/screens.py +394 -0
  102. package/latticeai/models/router/__init__.py +88 -0
  103. package/latticeai/models/router/_contract.py +66 -0
  104. package/latticeai/models/router/branding.py +56 -0
  105. package/latticeai/models/router/catalog.py +69 -0
  106. package/latticeai/models/router/documents.py +199 -0
  107. package/latticeai/models/router/errors.py +37 -0
  108. package/latticeai/models/router/generation.py +258 -0
  109. package/latticeai/models/router/loading.py +291 -0
  110. package/latticeai/models/router/local_models.py +85 -0
  111. package/latticeai/models/router/registry.py +147 -0
  112. package/latticeai/runtime/build_phases/__init__.py +82 -0
  113. package/latticeai/runtime/build_phases/features.py +407 -0
  114. package/latticeai/runtime/build_phases/foundation.py +555 -0
  115. package/latticeai/runtime/build_phases/web.py +492 -0
  116. package/latticeai/runtime/runtime_context.py +1 -0
  117. package/latticeai/services/architecture_readiness.py +48 -19
  118. package/latticeai/services/brain_intelligence/__init__.py +58 -0
  119. package/latticeai/services/brain_intelligence/_contract.py +71 -0
  120. package/latticeai/services/brain_intelligence/consistency.py +193 -0
  121. package/latticeai/services/brain_intelligence/constants.py +47 -0
  122. package/latticeai/services/brain_intelligence/digest.py +258 -0
  123. package/latticeai/services/brain_intelligence/health.py +331 -0
  124. package/latticeai/services/brain_intelligence/proposals.py +264 -0
  125. package/latticeai/services/brain_intelligence/sampling.py +84 -0
  126. package/latticeai/services/brain_intelligence/service.py +48 -0
  127. package/latticeai/services/chronicle.py +557 -0
  128. package/latticeai/services/memory_service/__init__.py +52 -0
  129. package/latticeai/services/memory_service/_contract.py +100 -0
  130. package/latticeai/services/memory_service/brief.py +431 -0
  131. package/latticeai/services/memory_service/constants.py +57 -0
  132. package/latticeai/services/memory_service/maintenance.py +138 -0
  133. package/latticeai/services/memory_service/manager.py +186 -0
  134. package/latticeai/services/memory_service/proof.py +136 -0
  135. package/latticeai/services/memory_service/recall.py +225 -0
  136. package/latticeai/services/memory_service/service.py +48 -0
  137. package/latticeai/services/memory_service/stores.py +110 -0
  138. package/latticeai/services/model_runtime/__init__.py +322 -0
  139. package/latticeai/services/model_runtime/cloud.py +87 -0
  140. package/latticeai/services/model_runtime/download.py +282 -0
  141. package/latticeai/services/model_runtime/engines.py +341 -0
  142. package/latticeai/services/model_runtime/loading.py +178 -0
  143. package/latticeai/services/model_runtime/service.py +129 -0
  144. package/latticeai/services/model_runtime/state.py +131 -0
  145. package/latticeai/services/model_runtime/status.py +255 -0
  146. package/latticeai/services/product_readiness.py +15 -7
  147. package/latticeai/setup/wizard/__init__.py +126 -0
  148. package/latticeai/setup/wizard/catalog.py +172 -0
  149. package/latticeai/setup/wizard/detect.py +323 -0
  150. package/latticeai/setup/wizard/install.py +348 -0
  151. package/latticeai/setup/wizard/paths.py +168 -0
  152. package/latticeai/setup/wizard/plans.py +74 -0
  153. package/latticeai/setup/wizard/recommend.py +320 -0
  154. package/package.json +6 -2
  155. package/scripts/bump_version.py +14 -0
  156. package/scripts/capture_release_evidence.mjs +33 -21
  157. package/scripts/check_current_release_docs.mjs +1 -1
  158. package/scripts/check_i18n_namespace_coverage.mjs +41 -4
  159. package/scripts/check_max_file_lines.mjs +102 -0
  160. package/scripts/check_release_evidence_bound.mjs +30 -15
  161. package/scripts/check_screenshot_pixel_delta.py +34 -4
  162. package/scripts/check_server_i18n.mjs +1 -0
  163. package/scripts/generate_rust_parity_fixtures.py +562 -0
  164. package/scripts/lib/mock_server_fingerprint.mjs +94 -0
  165. package/scripts/release_screen_claims.json +31 -2
  166. package/src-tauri/Cargo.lock +361 -3
  167. package/src-tauri/Cargo.toml +6 -1
  168. package/src-tauri/src/backend.rs +349 -0
  169. package/src-tauri/src/folder.rs +33 -0
  170. package/src-tauri/src/main.rs +97 -399
  171. package/src-tauri/tauri.conf.json +1 -1
  172. package/static/app/asset-manifest.json +41 -37
  173. package/static/app/assets/Act-yYpYnn0v.js +1 -0
  174. package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
  175. package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
  176. package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
  177. package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
  178. package/static/app/assets/Capture-CFIRsFNE.js +1 -0
  179. package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
  180. package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
  181. package/static/app/assets/Library-DwO3yZST.js +1 -0
  182. package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
  183. package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
  184. package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
  185. package/static/app/assets/System-DW8F-2xL.js +1 -0
  186. package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
  187. package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
  188. package/static/app/assets/brain-Ci1CkWjM.js +1 -0
  189. package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
  190. package/static/app/assets/circle-check-DfInj-qD.js +1 -0
  191. package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
  192. package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
  193. package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
  194. package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
  195. package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
  196. package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
  197. package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
  198. package/static/app/assets/index-_u5iUHDr.js +10 -0
  199. package/static/app/assets/input-B0lPdRQZ.js +1 -0
  200. package/static/app/assets/link-2-CoFbooHS.js +1 -0
  201. package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
  202. package/static/app/assets/primitives-DEbN-d6p.js +1 -0
  203. package/static/app/assets/search-BybIWPNd.js +1 -0
  204. package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
  205. package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
  206. package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
  207. package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
  208. package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
  209. package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
  210. package/static/app/assets/utils-BlZr7Pd4.js +4 -0
  211. package/static/app/assets/workspace-jJY4RuAV.js +1 -0
  212. package/static/app/index.html +4 -4
  213. package/static/sw.js +1 -1
  214. package/lattice_brain/graph/_kg_common.py +0 -1331
  215. package/lattice_brain/graph/discovery_index.py +0 -1141
  216. package/lattice_brain/graph/retrieval.py +0 -1120
  217. package/lattice_brain/graph/retrieval_vector.py +0 -1293
  218. package/lattice_brain/ingestion.py +0 -1525
  219. package/lattice_brain/multimodal.py +0 -1258
  220. package/latticeai/core/agent.py +0 -1465
  221. package/latticeai/core/embedding_providers.py +0 -1196
  222. package/latticeai/core/file_generation.py +0 -1047
  223. package/latticeai/integrations/telegram_bot.py +0 -1390
  224. package/latticeai/models/router.py +0 -1007
  225. package/latticeai/runtime/build_phases.py +0 -1450
  226. package/latticeai/services/brain_intelligence.py +0 -1083
  227. package/latticeai/services/memory_service.py +0 -1177
  228. package/latticeai/services/model_runtime.py +0 -1281
  229. package/latticeai/setup/wizard.py +0 -1310
  230. package/static/app/assets/Act-AWf0SAKp.js +0 -1
  231. package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
  232. package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
  233. package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
  234. package/static/app/assets/Capture-CqOSzyPr.js +0 -1
  235. package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
  236. package/static/app/assets/Library-CX-bbhmK.js +0 -1
  237. package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
  238. package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
  239. package/static/app/assets/System-Bu2t5hn1.js +0 -1
  240. package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
  241. package/static/app/assets/brain-DJMoqrwx.js +0 -1
  242. package/static/app/assets/index-BpYkzcVm.js +0 -10
  243. package/static/app/assets/input-DSlJJxRs.js +0 -1
  244. package/static/app/assets/primitives-BCx6TvfG.js +0 -1
  245. package/static/app/assets/search-Cgy8cCFJ.js +0 -1
  246. package/static/app/assets/utils-zqPZJxdx.js +0 -4
  247. package/static/app/assets/workspace-DXTihhfU.js +0 -1
@@ -0,0 +1,322 @@
1
+ """Model runtime and provider helpers for Lattice AI.
2
+
3
+ This module owns local/cloud model preparation, engine detection, model download,
4
+ provider-specific server startup, smoke tests, and runtime feature payloads. It is
5
+ configured by ``server_app`` with app-level state but has no FastAPI app import.
6
+
7
+ Split into cohesive submodules in v11.3.0 (no behaviour change): ``state`` (the
8
+ immutable ``ModelRuntimeState`` and the consent gates), ``engines`` (engine
9
+ wrappers + the LM Studio client), ``download`` (Hugging Face readiness and
10
+ fetch), ``status`` (``engine_status`` / ``runtime_features`` / ``install_engine``),
11
+ ``loading`` (identity resolution and the load entrypoints), ``cloud`` (cloud key
12
+ verification) and ``service`` (the bound ``ModelRuntimeService``). This module
13
+ re-exports every name the single file exposed, so
14
+ ``latticeai.services.model_runtime.X`` keeps working — including the private
15
+ names ``model_loading._get_model_runtime_deps`` imports.
16
+
17
+ Two deliberate omissions: ``engines._LMSTUDIO_MODELS_CACHE`` and
18
+ ``_LMSTUDIO_MODELS_CACHE_TS`` are *rebound* by ``get_lmstudio_models``, so a
19
+ re-export here would be a snapshot frozen at import time. They stay reachable
20
+ at ``latticeai.services.model_runtime.engines``, which is where the live values
21
+ are.
22
+
23
+ Stubbing note: rebinding a name *here* changes only this module's name — the
24
+ lazy ``from latticeai.services.model_runtime import …`` calls in
25
+ ``model_engines`` and ``model_loading`` read it, but a submodule that uses the
26
+ name holds its own reference. A test standing in for a helper patches the
27
+ submodule that uses it.
28
+ """
29
+
30
+ from __future__ import annotations
31
+
32
+ # The single file had no ``__all__``, so its public surface was "every module
33
+ # global" — including the names it imported for its own use. Every re-export
34
+ # below therefore uses the redundant-alias form: it reproduces exactly that
35
+ # surface, and it marks each name as deliberate rather than a leftover import.
36
+ from latticeai.core.quiet import quiet as quiet
37
+ from latticeai.models.router import HF_MODELS_ROOT as HF_MODELS_ROOT
38
+ from latticeai.models.router import (
39
+ OPENAI_COMPATIBLE_PROVIDERS as OPENAI_COMPATIBLE_PROVIDERS,
40
+ )
41
+ from latticeai.models.router import AsyncOpenAI as AsyncOpenAI
42
+ from latticeai.models.router import ensure_mlx_runtime as ensure_mlx_runtime
43
+ from latticeai.models.router import hf_cache_model_dir as hf_cache_model_dir
44
+ from latticeai.models.router import hf_model_dir as hf_model_dir
45
+ from latticeai.models.router import parse_model_ref as parse_model_ref
46
+
47
+ # Catalog data + version-dedup helpers live in ``model_catalog``; re-exported
48
+ # here so existing ``from ...model_runtime import ENGINE_MODEL_CATALOG`` imports
49
+ # keep working.
50
+ from latticeai.services.model_catalog import (
51
+ _VERSIONED_MODEL_PATTERNS as _VERSIONED_MODEL_PATTERNS,
52
+ )
53
+ from latticeai.services.model_catalog import (
54
+ ENGINE_INSTALLERS as ENGINE_INSTALLERS,
55
+ )
56
+ from latticeai.services.model_catalog import (
57
+ ENGINE_MODEL_CATALOG as ENGINE_MODEL_CATALOG,
58
+ )
59
+ from latticeai.services.model_catalog import (
60
+ MODEL_ENGINE_ALIASES as MODEL_ENGINE_ALIASES,
61
+ )
62
+ from latticeai.services.model_catalog import (
63
+ _model_family_version as _model_family_version,
64
+ )
65
+ from latticeai.services.model_catalog import (
66
+ _version_tuple as _version_tuple,
67
+ )
68
+ from latticeai.services.model_catalog import (
69
+ filter_lower_family_versions as filter_lower_family_versions,
70
+ )
71
+ from latticeai.services.model_errors import ModelRuntimeError as ModelRuntimeError
72
+ from latticeai.services.model_runtime.cloud import (
73
+ CLOUD_VERIFY_TTL_SECONDS as CLOUD_VERIFY_TTL_SECONDS,
74
+ )
75
+ from latticeai.services.model_runtime.cloud import (
76
+ _probe_cloud_model as _probe_cloud_model,
77
+ )
78
+ from latticeai.services.model_runtime.cloud import (
79
+ verify_cloud_models as verify_cloud_models,
80
+ )
81
+ from latticeai.services.model_runtime.download import (
82
+ download_hf_model as download_hf_model,
83
+ )
84
+ from latticeai.services.model_runtime.download import (
85
+ estimate_eta_seconds as estimate_eta_seconds,
86
+ )
87
+ from latticeai.services.model_runtime.download import (
88
+ hf_model_ready as hf_model_ready,
89
+ )
90
+ from latticeai.services.model_runtime.download import (
91
+ hf_repo_files_with_sizes as hf_repo_files_with_sizes,
92
+ )
93
+ from latticeai.services.model_runtime.download import (
94
+ model_download_progress_payload as model_download_progress_payload,
95
+ )
96
+ from latticeai.services.model_runtime.engines import (
97
+ _LMSTUDIO_MODELS_CACHE_TTL as _LMSTUDIO_MODELS_CACHE_TTL,
98
+ )
99
+
100
+ # The private aliases the historical module bound for its own use — including
101
+ # the ones ``model_loading._get_model_runtime_deps`` imports by name. They are
102
+ # re-exported from the submodule that already binds them under exactly these
103
+ # names, which is ruff's redundant-alias re-export form.
104
+ from latticeai.services.model_runtime.engines import (
105
+ _LOCAL_SERVER_PROCESSES as _LOCAL_SERVER_PROCESSES,
106
+ )
107
+ from latticeai.services.model_runtime.engines import (
108
+ LMSTUDIO_BUNDLED_CLI as LMSTUDIO_BUNDLED_CLI,
109
+ )
110
+ from latticeai.services.model_runtime.engines import (
111
+ LOCAL_SERVER_PROCESSES as LOCAL_SERVER_PROCESSES,
112
+ )
113
+ from latticeai.services.model_runtime.engines import (
114
+ VLLM_METAL_BIN as VLLM_METAL_BIN,
115
+ )
116
+ from latticeai.services.model_runtime.engines import (
117
+ VLLM_METAL_ENV as VLLM_METAL_ENV,
118
+ )
119
+ from latticeai.services.model_runtime.engines import (
120
+ VLLM_METAL_PYTHON as VLLM_METAL_PYTHON,
121
+ )
122
+ from latticeai.services.model_runtime.engines import (
123
+ _engine_install_plan as _engine_install_plan,
124
+ )
125
+ from latticeai.services.model_runtime.engines import (
126
+ _engine_support_status as _engine_support_status,
127
+ )
128
+ from latticeai.services.model_runtime.engines import (
129
+ _ensure_llamacpp_server as _ensure_llamacpp_server,
130
+ )
131
+ from latticeai.services.model_runtime.engines import (
132
+ _ensure_lmstudio_server as _ensure_lmstudio_server,
133
+ )
134
+ from latticeai.services.model_runtime.engines import (
135
+ _ensure_ollama_server as _ensure_ollama_server,
136
+ )
137
+ from latticeai.services.model_runtime.engines import (
138
+ _ensure_vllm_server as _ensure_vllm_server,
139
+ )
140
+ from latticeai.services.model_runtime.engines import (
141
+ _find_lmstudio_cli as _find_lmstudio_cli,
142
+ )
143
+ from latticeai.services.model_runtime.engines import (
144
+ _find_lmstudio_model_key as _find_lmstudio_model_key,
145
+ )
146
+ from latticeai.services.model_runtime.engines import (
147
+ _get_ollama_pulled_models as _get_ollama_pulled_models,
148
+ )
149
+ from latticeai.services.model_runtime.engines import (
150
+ _get_openai_compatible_server_models as _get_openai_compatible_server_models,
151
+ )
152
+ from latticeai.services.model_runtime.engines import (
153
+ _json_request as _json_request,
154
+ )
155
+ from latticeai.services.model_runtime.engines import (
156
+ _lmstudio_candidate_keys as _lmstudio_candidate_keys,
157
+ )
158
+ from latticeai.services.model_runtime.engines import (
159
+ _local_binary as _local_binary,
160
+ )
161
+ from latticeai.services.model_runtime.engines import (
162
+ _pull_ollama_model_with_progress as _pull_ollama_model_with_progress,
163
+ )
164
+ from latticeai.services.model_runtime.engines import (
165
+ _safe_engine_install_plan as _safe_engine_install_plan,
166
+ )
167
+ from latticeai.services.model_runtime.engines import (
168
+ _update_env_file as _update_env_file,
169
+ )
170
+ from latticeai.services.model_runtime.engines import (
171
+ _vllm_executable as _vllm_executable,
172
+ )
173
+ from latticeai.services.model_runtime.engines import (
174
+ _vllm_metal_python as _vllm_metal_python,
175
+ )
176
+ from latticeai.services.model_runtime.engines import (
177
+ _wait_for_openai_compatible_server as _wait_for_openai_compatible_server,
178
+ )
179
+ from latticeai.services.model_runtime.engines import (
180
+ _windows_binary_candidates as _windows_binary_candidates,
181
+ )
182
+ from latticeai.services.model_runtime.engines import (
183
+ engine_installed as engine_installed,
184
+ )
185
+ from latticeai.services.model_runtime.engines import (
186
+ engine_support_status as engine_support_status,
187
+ )
188
+ from latticeai.services.model_runtime.engines import (
189
+ ensure_llamacpp_server as ensure_llamacpp_server,
190
+ )
191
+ from latticeai.services.model_runtime.engines import (
192
+ ensure_lmstudio_model as ensure_lmstudio_model,
193
+ )
194
+ from latticeai.services.model_runtime.engines import (
195
+ ensure_lmstudio_server as ensure_lmstudio_server,
196
+ )
197
+ from latticeai.services.model_runtime.engines import (
198
+ ensure_ollama_server as ensure_ollama_server,
199
+ )
200
+ from latticeai.services.model_runtime.engines import (
201
+ ensure_vllm_server as ensure_vllm_server,
202
+ )
203
+ from latticeai.services.model_runtime.engines import (
204
+ find_lmstudio_cli as find_lmstudio_cli,
205
+ )
206
+ from latticeai.services.model_runtime.engines import (
207
+ get_lmstudio_models as get_lmstudio_models,
208
+ )
209
+ from latticeai.services.model_runtime.engines import (
210
+ get_ollama_pulled_models as get_ollama_pulled_models,
211
+ )
212
+ from latticeai.services.model_runtime.engines import (
213
+ get_openai_compatible_server_models as get_openai_compatible_server_models,
214
+ )
215
+ from latticeai.services.model_runtime.engines import (
216
+ lmstudio_api_base as lmstudio_api_base,
217
+ )
218
+ from latticeai.services.model_runtime.engines import (
219
+ lmstudio_native_api_base as lmstudio_native_api_base,
220
+ )
221
+ from latticeai.services.model_runtime.engines import (
222
+ local_binary as local_binary,
223
+ )
224
+ from latticeai.services.model_runtime.engines import (
225
+ pull_ollama_model_with_progress as pull_ollama_model_with_progress,
226
+ )
227
+ from latticeai.services.model_runtime.engines import (
228
+ vllm_executable as vllm_executable,
229
+ )
230
+ from latticeai.services.model_runtime.engines import (
231
+ vllm_metal_python as vllm_metal_python,
232
+ )
233
+ from latticeai.services.model_runtime.engines import (
234
+ wait_for_openai_compatible_server as wait_for_openai_compatible_server,
235
+ )
236
+ from latticeai.services.model_runtime.engines import (
237
+ windows_binary_candidates as windows_binary_candidates,
238
+ )
239
+ from latticeai.services.model_runtime.loading import (
240
+ _LOCAL_SMOKE_ENGINES as _LOCAL_SMOKE_ENGINES,
241
+ )
242
+ from latticeai.services.model_runtime.loading import (
243
+ _ModelResolution as _ModelResolution,
244
+ )
245
+ from latticeai.services.model_runtime.loading import (
246
+ _resolve_model_alias as _resolve_model_alias,
247
+ )
248
+ from latticeai.services.model_runtime.loading import (
249
+ _smoke_test_loaded_model as _smoke_test_loaded_model,
250
+ )
251
+ from latticeai.services.model_runtime.loading import (
252
+ build_model_resolution as build_model_resolution,
253
+ )
254
+ from latticeai.services.model_runtime.loading import (
255
+ ensure_engine_ready as ensure_engine_ready,
256
+ )
257
+ from latticeai.services.model_runtime.loading import (
258
+ normalize_local_model_request as normalize_local_model_request,
259
+ )
260
+ from latticeai.services.model_runtime.loading import (
261
+ prepare_and_load_model as prepare_and_load_model,
262
+ )
263
+ from latticeai.services.model_runtime.loading import (
264
+ prepare_and_load_model_stream as prepare_and_load_model_stream,
265
+ )
266
+ from latticeai.services.model_runtime.loading import (
267
+ sse_event as sse_event,
268
+ )
269
+ from latticeai.services.model_runtime.service import (
270
+ ModelRuntimeService as ModelRuntimeService,
271
+ )
272
+ from latticeai.services.model_runtime.service import (
273
+ build_model_runtime as build_model_runtime,
274
+ )
275
+ from latticeai.services.model_runtime.service import (
276
+ configure_model_runtime as configure_model_runtime,
277
+ )
278
+ from latticeai.services.model_runtime.state import (
279
+ _MODEL_LOADING_COMPAT_EXPORTS as _MODEL_LOADING_COMPAT_EXPORTS,
280
+ )
281
+ from latticeai.services.model_runtime.state import (
282
+ _SMOKE_PROMPT as _SMOKE_PROMPT,
283
+ )
284
+ from latticeai.services.model_runtime.state import (
285
+ ModelRuntimeState as ModelRuntimeState,
286
+ )
287
+ from latticeai.services.model_runtime.state import (
288
+ _download_allowed as _download_allowed,
289
+ )
290
+ from latticeai.services.model_runtime.state import (
291
+ _download_block as _download_block,
292
+ )
293
+ from latticeai.services.model_runtime.state import (
294
+ _engine_install_block as _engine_install_block,
295
+ )
296
+ from latticeai.services.model_runtime.state import (
297
+ _friendly_model_runtime_error as _friendly_model_runtime_error,
298
+ )
299
+ from latticeai.services.model_runtime.state import (
300
+ _missing_current_user as _missing_current_user,
301
+ )
302
+ from latticeai.services.model_runtime.state import (
303
+ _missing_user_api_key as _missing_user_api_key,
304
+ )
305
+ from latticeai.services.model_runtime.state import (
306
+ _model_runtime_compatibility as _model_runtime_compatibility,
307
+ )
308
+ from latticeai.services.model_runtime.state import (
309
+ create_model_runtime_state as create_model_runtime_state,
310
+ )
311
+ from latticeai.services.model_runtime.status import (
312
+ _install_engine as _install_engine,
313
+ )
314
+ from latticeai.services.model_runtime.status import (
315
+ engine_status as engine_status,
316
+ )
317
+ from latticeai.services.model_runtime.status import (
318
+ install_engine as install_engine,
319
+ )
320
+ from latticeai.services.model_runtime.status import (
321
+ runtime_features as runtime_features,
322
+ )
@@ -0,0 +1,87 @@
1
+ """Cloud model verification — does this key actually answer for this model?
2
+
3
+ A one-token chat completion per configured cloud model, cached per service
4
+ instance for :data:`CLOUD_VERIFY_TTL_SECONDS`. A model the router already knows
5
+ is unavailable is recorded as such without a network call.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import asyncio
11
+ import os
12
+ import time
13
+ from typing import Any, Dict, Optional
14
+
15
+ from latticeai.models.router import (
16
+ OPENAI_COMPATIBLE_PROVIDERS,
17
+ AsyncOpenAI,
18
+ parse_model_ref,
19
+ )
20
+ from latticeai.services.model_runtime.state import ModelRuntimeState
21
+
22
+ CLOUD_VERIFY_TTL_SECONDS = 600
23
+
24
+ async def _probe_cloud_model(model_ref: str) -> Dict[str, Any]:
25
+ provider, model_name = parse_model_ref(model_ref)
26
+ config = OPENAI_COMPATIBLE_PROVIDERS.get(provider)
27
+ if not config:
28
+ return {"ok": False, "reason": f"Unsupported provider: {provider}"}
29
+
30
+ api_key = os.getenv(config["env_key"]) or config.get("api_key_fallback")
31
+ if not api_key:
32
+ return {"ok": False, "reason": f"Missing API key: {config['env_key']}"}
33
+
34
+ base_url = os.getenv(config.get("base_url_env", "")) if config.get("base_url_env") else None
35
+ base_url = base_url or config.get("base_url")
36
+ try:
37
+ # base_url is passed only when configured: an explicit None is not
38
+ # the same as omitting the argument.
39
+ client = (
40
+ AsyncOpenAI(api_key=api_key, base_url=base_url)
41
+ if base_url
42
+ else AsyncOpenAI(api_key=api_key)
43
+ )
44
+ await asyncio.wait_for(
45
+ client.chat.completions.create(
46
+ model=model_name,
47
+ messages=[{"role": "user", "content": "ping"}],
48
+ max_tokens=1,
49
+ temperature=0,
50
+ ),
51
+ timeout=15,
52
+ )
53
+ return {"ok": True, "reason": "ok"}
54
+ except Exception as e:
55
+ return {"ok": False, "reason": str(e)[:220]}
56
+
57
+
58
+ async def verify_cloud_models(
59
+ force: bool = False,
60
+ provider_filter: Optional[str] = None,
61
+ *,
62
+ state: ModelRuntimeState,
63
+ cache: Dict[str, Dict[str, Any]],
64
+ ) -> Dict[str, Dict]:
65
+ now = time.time()
66
+ r = state.router
67
+ cloud_items = [item for item in (r.detected_cloud_models() if r else []) if item.get("tag") == "cloud"]
68
+ if provider_filter:
69
+ cloud_items = [item for item in cloud_items if item.get("provider") == provider_filter]
70
+
71
+ results: Dict[str, Dict] = {}
72
+ for item in cloud_items:
73
+ model_ref = item["id"]
74
+ cached = cache.get(model_ref)
75
+ if not force and cached and (now - cached.get("ts", 0) <= CLOUD_VERIFY_TTL_SECONDS):
76
+ results[model_ref] = cached
77
+ continue
78
+ if item.get("available") is False:
79
+ record = {"ok": False, "reason": item.get("requires") or "API key missing", "ts": now}
80
+ cache[model_ref] = record
81
+ results[model_ref] = record
82
+ continue
83
+ probe = await _probe_cloud_model(model_ref)
84
+ record = {"ok": bool(probe.get("ok")), "reason": probe.get("reason", ""), "ts": now}
85
+ cache[model_ref] = record
86
+ results[model_ref] = record
87
+ return results
@@ -0,0 +1,282 @@
1
+ """Hugging Face model presence, download, and download progress.
2
+
3
+ Answers one question — "are the weights actually on this disk?" — and, when
4
+ they are not and the caller has consent, fetches them while emitting a progress
5
+ payload per file. The readiness check is deliberately format-aware: a GGUF
6
+ repo, an MLX 4-bit repo and a vLLM bf16 repo each prove completeness
7
+ differently.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import importlib.util
13
+ import logging
14
+ import time
15
+ from pathlib import Path
16
+ from typing import Any, Dict, List, Optional
17
+
18
+ from latticeai.core.quiet import quiet
19
+ from latticeai.models.router import hf_cache_model_dir, hf_model_dir
20
+ from latticeai.services.model_errors import ModelRuntimeError
21
+
22
+
23
+ def hf_model_ready(repo_id: str, provider: str = "local_mlx") -> bool:
24
+ model_dir = hf_model_dir(repo_id)
25
+ if provider in {"local_mlx", "vllm"} and (not model_dir.exists() or not model_dir.is_dir()):
26
+ hf_cache_repo = Path.home() / ".cache" / "huggingface" / "hub" / f"models--{repo_id.replace('/', '--')}"
27
+ if hf_cache_repo.exists() and any(hf_cache_repo.glob("snapshots/*")):
28
+ if provider == "vllm":
29
+ return True
30
+ return hf_cache_model_dir(repo_id) is not None
31
+ return False
32
+ if not model_dir.exists() or not model_dir.is_dir():
33
+ return False
34
+ if provider == "llamacpp":
35
+ return any(model_dir.rglob("*.gguf"))
36
+ has_config = (model_dir / "config.json").exists()
37
+ has_weights = any(model_dir.glob("*.safetensors")) or any(model_dir.glob("*.bin"))
38
+ has_tokenizer = (
39
+ (model_dir / "tokenizer.json").exists()
40
+ or (model_dir / "tokenizer.model").exists()
41
+ or (model_dir / "tokenizer_config.json").exists()
42
+ )
43
+ return has_config and has_weights and has_tokenizer
44
+
45
+
46
+ def model_download_progress_payload(
47
+ stage: str,
48
+ message: str,
49
+ *,
50
+ percent: Optional[float] = None,
51
+ detail: Optional[str] = None,
52
+ downloaded_bytes: Optional[int] = None,
53
+ total_bytes: Optional[int] = None,
54
+ eta_seconds: Optional[float] = None,
55
+ file: Optional[str] = None,
56
+ indeterminate: bool = False,
57
+ ) -> Dict[str, Any]:
58
+ payload: Dict[str, Any] = {
59
+ "stage": stage,
60
+ "message": message,
61
+ "indeterminate": indeterminate,
62
+ "ts": time.time(),
63
+ }
64
+ if percent is not None:
65
+ payload["percent"] = max(0, min(100, round(float(percent), 1)))
66
+ if detail:
67
+ payload["detail"] = detail
68
+ if downloaded_bytes is not None:
69
+ payload["downloaded_bytes"] = max(0, int(downloaded_bytes))
70
+ if total_bytes is not None:
71
+ payload["total_bytes"] = max(0, int(total_bytes))
72
+ if eta_seconds is not None:
73
+ payload["eta_seconds"] = max(0, round(float(eta_seconds)))
74
+ if file:
75
+ payload["file"] = file
76
+ return payload
77
+
78
+
79
+ def estimate_eta_seconds(started_at: float, percent: Optional[float]) -> Optional[float]:
80
+ if percent is None or percent <= 0 or percent >= 100:
81
+ return None
82
+ elapsed = max(0.0, time.time() - started_at)
83
+ return elapsed * (100.0 - percent) / percent
84
+
85
+
86
+ def hf_repo_files_with_sizes(repo_id: str) -> List[Dict[str, Any]]:
87
+ from huggingface_hub import HfApi
88
+
89
+ api = HfApi()
90
+ try:
91
+ info = api.model_info(repo_id, files_metadata=True)
92
+ files = []
93
+ for sibling in getattr(info, "siblings", []) or []:
94
+ name = str(getattr(sibling, "rfilename", "") or "").strip()
95
+ if not name or name.endswith("/"):
96
+ continue
97
+ files.append({"name": name, "size": int(getattr(sibling, "size", 0) or 0)})
98
+ if files:
99
+ return files
100
+ except TypeError:
101
+ quiet()
102
+ except Exception as e:
103
+ logging.warning("huggingface model_info failed for %s: %s", repo_id, e)
104
+
105
+ return [{"name": str(name), "size": 0} for name in api.list_repo_files(repo_id) if str(name).strip()]
106
+
107
+
108
+ def download_hf_model(
109
+ repo_id: str,
110
+ provider: str = "local_mlx",
111
+ progress_emit=None,
112
+ ) -> Dict[str, Any]:
113
+ if importlib.util.find_spec("huggingface_hub") is None:
114
+ raise ModelRuntimeError(status_code=400, detail="huggingface_hub가 없습니다. 먼저 MLX runtime 설치를 진행해 주세요.")
115
+
116
+ target_dir = hf_model_dir(repo_id)
117
+ if hf_model_ready(repo_id, provider):
118
+ cached_dir = hf_cache_model_dir(repo_id) if provider == "local_mlx" else None
119
+ resolved_dir = cached_dir or target_dir
120
+ if progress_emit:
121
+ progress_emit(model_download_progress_payload(
122
+ "download",
123
+ "이미 다운로드된 모델을 확인했습니다.",
124
+ percent=100,
125
+ downloaded_bytes=0,
126
+ total_bytes=0,
127
+ eta_seconds=0,
128
+ ))
129
+ return {"model": repo_id, "path": str(resolved_dir), "cached": True}
130
+
131
+ target_dir.mkdir(parents=True, exist_ok=True)
132
+ try:
133
+ from huggingface_hub import hf_hub_download
134
+
135
+ started_at = time.time()
136
+ all_files = hf_repo_files_with_sizes(repo_id)
137
+ if provider == "llamacpp":
138
+ ggufs = sorted(
139
+ [item for item in all_files if str(item["name"]).lower().endswith(".gguf")],
140
+ key=lambda item: str(item["name"]),
141
+ )
142
+ if not ggufs:
143
+ raise RuntimeError("GGUF 파일을 찾지 못했습니다.")
144
+ preference = ("q4_k_m", "q4_0", "q4_k_s", "q3_k_m", "q2_k")
145
+ selected_files = [
146
+ next(
147
+ (item for pref in preference for item in ggufs if pref in str(item["name"]).lower()),
148
+ ggufs[0],
149
+ )
150
+ ]
151
+ else:
152
+ selected_files = all_files
153
+
154
+ total_bytes = sum(int(item.get("size") or 0) for item in selected_files) or None
155
+ downloaded_bytes = 0
156
+ total_files = max(1, len(selected_files))
157
+ if progress_emit:
158
+ progress_emit(model_download_progress_payload(
159
+ "download",
160
+ "모델 파일 정보를 확인했습니다.",
161
+ percent=0,
162
+ downloaded_bytes=0,
163
+ total_bytes=total_bytes,
164
+ indeterminate=total_bytes is None,
165
+ ))
166
+
167
+ for index, item in enumerate(selected_files, start=1):
168
+ filename = str(item["name"])
169
+ size = int(item.get("size") or 0)
170
+ tqdm_class = None
171
+ if progress_emit:
172
+ current_percent = (
173
+ (downloaded_bytes / total_bytes) * 100 if total_bytes else ((index - 1) / total_files) * 100
174
+ )
175
+ progress_emit(model_download_progress_payload(
176
+ "download",
177
+ "모델 다운로드 중입니다.",
178
+ percent=current_percent,
179
+ detail=filename,
180
+ downloaded_bytes=downloaded_bytes,
181
+ total_bytes=total_bytes,
182
+ eta_seconds=estimate_eta_seconds(started_at, current_percent),
183
+ file=filename,
184
+ indeterminate=total_bytes is None and total_files <= 1,
185
+ ))
186
+ try:
187
+ from tqdm.auto import tqdm as base_tqdm
188
+
189
+ downloaded_before = downloaded_bytes
190
+ last_emit = {"at": 0.0, "percent": -1.0}
191
+
192
+ def emit_byte_progress(
193
+ done_bytes: float,
194
+ # Bound per iteration: this callback outlives the loop
195
+ # body when a download runs long, and late binding
196
+ # would report every file's progress against the last
197
+ # file's offsets.
198
+ downloaded_before: int = downloaded_before,
199
+ size: Any = size,
200
+ index: int = index,
201
+ last_emit: dict = last_emit,
202
+ filename: str = filename,
203
+ ) -> None:
204
+ done = max(0, int(done_bytes or 0))
205
+ if total_bytes:
206
+ aggregate = min(total_bytes, downloaded_before + done)
207
+ percent = (aggregate / total_bytes) * 100
208
+ else:
209
+ file_total = size or done
210
+ file_ratio = min(1.0, done / file_total) if file_total else 0.0
211
+ aggregate = downloaded_before + done
212
+ percent = ((index - 1) + file_ratio) / total_files * 100
213
+ now = time.time()
214
+ if percent < 100 and now - last_emit["at"] < 0.5 and percent - last_emit["percent"] < 0.3:
215
+ return
216
+ last_emit["at"] = now
217
+ last_emit["percent"] = percent
218
+ progress_emit(model_download_progress_payload(
219
+ "download",
220
+ "모델 다운로드 중입니다.",
221
+ percent=percent,
222
+ detail=filename,
223
+ downloaded_bytes=aggregate,
224
+ total_bytes=total_bytes,
225
+ eta_seconds=estimate_eta_seconds(started_at, percent),
226
+ file=filename,
227
+ indeterminate=total_bytes is None and total_files <= 1,
228
+ ))
229
+
230
+ class ProgressTqdm(base_tqdm):
231
+ def update(self, n=1):
232
+ result = super().update(n)
233
+ emit_byte_progress(float(getattr(self, "n", 0) or 0))
234
+ return result
235
+
236
+ tqdm_class = ProgressTqdm
237
+ except Exception:
238
+ tqdm_class = None
239
+ local_path = hf_hub_download(
240
+ repo_id=repo_id,
241
+ filename=filename,
242
+ local_dir=str(target_dir),
243
+ tqdm_class=tqdm_class,
244
+ )
245
+ if size <= 0:
246
+ try:
247
+ size = Path(local_path).stat().st_size
248
+ except OSError:
249
+ size = 0
250
+ downloaded_bytes += size
251
+ if progress_emit:
252
+ current_percent = (
253
+ (downloaded_bytes / total_bytes) * 100 if total_bytes else (index / total_files) * 100
254
+ )
255
+ progress_emit(model_download_progress_payload(
256
+ "download",
257
+ "모델 다운로드 중입니다.",
258
+ percent=current_percent,
259
+ detail=filename,
260
+ downloaded_bytes=downloaded_bytes,
261
+ total_bytes=total_bytes,
262
+ eta_seconds=estimate_eta_seconds(started_at, current_percent),
263
+ file=filename,
264
+ indeterminate=False,
265
+ ))
266
+
267
+ if progress_emit:
268
+ progress_emit(model_download_progress_payload(
269
+ "download",
270
+ "모델 다운로드가 완료되었습니다.",
271
+ percent=100,
272
+ downloaded_bytes=downloaded_bytes,
273
+ total_bytes=total_bytes or downloaded_bytes,
274
+ eta_seconds=0,
275
+ ))
276
+ except Exception as e:
277
+ raise ModelRuntimeError(status_code=500, detail=f"{repo_id} 다운로드 실패: {str(e)[-2000:]}")
278
+
279
+ if not hf_model_ready(repo_id, provider):
280
+ raise ModelRuntimeError(status_code=500, detail=f"{repo_id} 다운로드가 완료되지 않았습니다. 모델 파일을 찾지 못했습니다.")
281
+
282
+ return {"model": repo_id, "path": str(target_dir), "cached": False}