ltcai 11.2.0 → 11.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (247) hide show
  1. package/README.md +46 -53
  2. package/docs/CHANGELOG.md +61 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/MULTI_AGENT_RUNTIME.md +1 -1
  6. package/docs/ONBOARDING.md +1 -1
  7. package/docs/OPERATIONS.md +6 -2
  8. package/docs/PERMISSION_MODE.md +1 -1
  9. package/docs/TRUST_MODEL.md +1 -1
  10. package/docs/WHY_LATTICE.md +1 -1
  11. package/docs/kg-schema.md +2 -2
  12. package/docs/v11.3.0_PLAN.md +202 -0
  13. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
  14. package/lattice_brain/__init__.py +1 -1
  15. package/lattice_brain/graph/_kg_common/__init__.py +287 -0
  16. package/lattice_brain/graph/_kg_common/extraction.py +516 -0
  17. package/lattice_brain/graph/_kg_common/relations.py +161 -0
  18. package/lattice_brain/graph/_kg_common/text.py +479 -0
  19. package/lattice_brain/graph/discovery_index/__init__.py +35 -0
  20. package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
  21. package/lattice_brain/graph/discovery_index/extract.py +137 -0
  22. package/lattice_brain/graph/discovery_index/scan.py +411 -0
  23. package/lattice_brain/graph/discovery_index/upsert.py +495 -0
  24. package/lattice_brain/graph/projection/__init__.py +42 -0
  25. package/lattice_brain/graph/projection/curation.py +500 -0
  26. package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
  27. package/lattice_brain/graph/retrieval/__init__.py +54 -0
  28. package/lattice_brain/graph/retrieval/context.py +197 -0
  29. package/lattice_brain/graph/retrieval/graph_view.py +319 -0
  30. package/lattice_brain/graph/retrieval/hybrid.py +488 -0
  31. package/lattice_brain/graph/retrieval/maintenance.py +121 -0
  32. package/lattice_brain/graph/retrieval/signals.py +95 -0
  33. package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
  34. package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
  35. package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
  36. package/lattice_brain/graph/retrieval_vector/search.py +560 -0
  37. package/lattice_brain/graph/retrieval_vector/status.py +374 -0
  38. package/lattice_brain/ingestion/__init__.py +130 -0
  39. package/lattice_brain/ingestion/_contract.py +90 -0
  40. package/lattice_brain/ingestion/constants.py +127 -0
  41. package/lattice_brain/ingestion/folder_scan.py +57 -0
  42. package/lattice_brain/ingestion/folders.py +258 -0
  43. package/lattice_brain/ingestion/hashing.py +26 -0
  44. package/lattice_brain/ingestion/jobs_api.py +107 -0
  45. package/lattice_brain/ingestion/models.py +80 -0
  46. package/lattice_brain/ingestion/pipeline.py +486 -0
  47. package/lattice_brain/ingestion/quality.py +209 -0
  48. package/lattice_brain/ingestion/routing.py +295 -0
  49. package/lattice_brain/multimodal/__init__.py +164 -0
  50. package/lattice_brain/multimodal/audio.py +77 -0
  51. package/lattice_brain/multimodal/common.py +118 -0
  52. package/lattice_brain/multimodal/images.py +498 -0
  53. package/lattice_brain/multimodal/ports.py +169 -0
  54. package/lattice_brain/multimodal/video.py +410 -0
  55. package/lattice_brain/portability/__init__.py +90 -0
  56. package/lattice_brain/portability/_contract.py +42 -0
  57. package/lattice_brain/portability/backups.py +338 -0
  58. package/lattice_brain/portability/bundles.py +136 -0
  59. package/lattice_brain/portability/constants.py +93 -0
  60. package/lattice_brain/portability/fsops.py +138 -0
  61. package/lattice_brain/portability/service.py +41 -0
  62. package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
  63. package/lattice_brain/runtime/__init__.py +1 -1
  64. package/lattice_brain/runtime/multi_agent.py +1 -1
  65. package/latticeai/__init__.py +1 -1
  66. package/latticeai/api/chronicle.py +63 -0
  67. package/latticeai/core/agent/__init__.py +93 -0
  68. package/latticeai/core/agent/_contract.py +79 -0
  69. package/latticeai/core/agent/context.py +57 -0
  70. package/latticeai/core/agent/deps.py +125 -0
  71. package/latticeai/core/agent/execution.py +622 -0
  72. package/latticeai/core/agent/planning.py +145 -0
  73. package/latticeai/core/agent/recovery.py +157 -0
  74. package/latticeai/core/agent/runtime.py +210 -0
  75. package/latticeai/core/agent/verification.py +231 -0
  76. package/latticeai/core/embedding_providers/__init__.py +151 -0
  77. package/latticeai/core/embedding_providers/base.py +199 -0
  78. package/latticeai/core/embedding_providers/captions.py +162 -0
  79. package/latticeai/core/embedding_providers/profiles.py +126 -0
  80. package/latticeai/core/embedding_providers/text.py +350 -0
  81. package/latticeai/core/embedding_providers/vision.py +352 -0
  82. package/latticeai/core/file_generation/__init__.py +115 -0
  83. package/latticeai/core/file_generation/bundles.py +76 -0
  84. package/latticeai/core/file_generation/extraction.py +154 -0
  85. package/latticeai/core/file_generation/inference.py +235 -0
  86. package/latticeai/core/file_generation/orchestration.py +152 -0
  87. package/latticeai/core/file_generation/prompting.py +117 -0
  88. package/latticeai/core/file_generation/repair.py +114 -0
  89. package/latticeai/core/file_generation/sanitize.py +61 -0
  90. package/latticeai/core/file_generation/validation.py +201 -0
  91. package/latticeai/core/legacy_compatibility.py +1 -1
  92. package/latticeai/core/marketplace.py +1 -1
  93. package/latticeai/core/messages.py +9 -0
  94. package/latticeai/core/workspace_os_constants.py +1 -1
  95. package/latticeai/integrations/telegram_bot/__init__.py +123 -0
  96. package/latticeai/integrations/telegram_bot/__main__.py +17 -0
  97. package/latticeai/integrations/telegram_bot/config.py +86 -0
  98. package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
  99. package/latticeai/integrations/telegram_bot/flows.py +478 -0
  100. package/latticeai/integrations/telegram_bot/helpers.py +322 -0
  101. package/latticeai/integrations/telegram_bot/screens.py +394 -0
  102. package/latticeai/models/router/__init__.py +88 -0
  103. package/latticeai/models/router/_contract.py +66 -0
  104. package/latticeai/models/router/branding.py +56 -0
  105. package/latticeai/models/router/catalog.py +69 -0
  106. package/latticeai/models/router/documents.py +199 -0
  107. package/latticeai/models/router/errors.py +37 -0
  108. package/latticeai/models/router/generation.py +258 -0
  109. package/latticeai/models/router/loading.py +291 -0
  110. package/latticeai/models/router/local_models.py +85 -0
  111. package/latticeai/models/router/registry.py +147 -0
  112. package/latticeai/runtime/build_phases/__init__.py +82 -0
  113. package/latticeai/runtime/build_phases/features.py +407 -0
  114. package/latticeai/runtime/build_phases/foundation.py +555 -0
  115. package/latticeai/runtime/build_phases/web.py +492 -0
  116. package/latticeai/runtime/runtime_context.py +1 -0
  117. package/latticeai/services/architecture_readiness.py +48 -19
  118. package/latticeai/services/brain_intelligence/__init__.py +58 -0
  119. package/latticeai/services/brain_intelligence/_contract.py +71 -0
  120. package/latticeai/services/brain_intelligence/consistency.py +193 -0
  121. package/latticeai/services/brain_intelligence/constants.py +47 -0
  122. package/latticeai/services/brain_intelligence/digest.py +258 -0
  123. package/latticeai/services/brain_intelligence/health.py +331 -0
  124. package/latticeai/services/brain_intelligence/proposals.py +264 -0
  125. package/latticeai/services/brain_intelligence/sampling.py +84 -0
  126. package/latticeai/services/brain_intelligence/service.py +48 -0
  127. package/latticeai/services/chronicle.py +557 -0
  128. package/latticeai/services/memory_service/__init__.py +52 -0
  129. package/latticeai/services/memory_service/_contract.py +100 -0
  130. package/latticeai/services/memory_service/brief.py +431 -0
  131. package/latticeai/services/memory_service/constants.py +57 -0
  132. package/latticeai/services/memory_service/maintenance.py +138 -0
  133. package/latticeai/services/memory_service/manager.py +186 -0
  134. package/latticeai/services/memory_service/proof.py +136 -0
  135. package/latticeai/services/memory_service/recall.py +225 -0
  136. package/latticeai/services/memory_service/service.py +48 -0
  137. package/latticeai/services/memory_service/stores.py +110 -0
  138. package/latticeai/services/model_runtime/__init__.py +322 -0
  139. package/latticeai/services/model_runtime/cloud.py +87 -0
  140. package/latticeai/services/model_runtime/download.py +282 -0
  141. package/latticeai/services/model_runtime/engines.py +341 -0
  142. package/latticeai/services/model_runtime/loading.py +178 -0
  143. package/latticeai/services/model_runtime/service.py +129 -0
  144. package/latticeai/services/model_runtime/state.py +131 -0
  145. package/latticeai/services/model_runtime/status.py +255 -0
  146. package/latticeai/services/product_readiness.py +15 -7
  147. package/latticeai/setup/wizard/__init__.py +126 -0
  148. package/latticeai/setup/wizard/catalog.py +172 -0
  149. package/latticeai/setup/wizard/detect.py +323 -0
  150. package/latticeai/setup/wizard/install.py +348 -0
  151. package/latticeai/setup/wizard/paths.py +168 -0
  152. package/latticeai/setup/wizard/plans.py +74 -0
  153. package/latticeai/setup/wizard/recommend.py +320 -0
  154. package/package.json +6 -2
  155. package/scripts/bump_version.py +14 -0
  156. package/scripts/capture_release_evidence.mjs +33 -21
  157. package/scripts/check_current_release_docs.mjs +1 -1
  158. package/scripts/check_i18n_namespace_coverage.mjs +41 -4
  159. package/scripts/check_max_file_lines.mjs +102 -0
  160. package/scripts/check_release_evidence_bound.mjs +30 -15
  161. package/scripts/check_screenshot_pixel_delta.py +34 -4
  162. package/scripts/check_server_i18n.mjs +1 -0
  163. package/scripts/generate_rust_parity_fixtures.py +562 -0
  164. package/scripts/lib/mock_server_fingerprint.mjs +94 -0
  165. package/scripts/release_screen_claims.json +31 -2
  166. package/src-tauri/Cargo.lock +361 -3
  167. package/src-tauri/Cargo.toml +6 -1
  168. package/src-tauri/src/backend.rs +349 -0
  169. package/src-tauri/src/folder.rs +33 -0
  170. package/src-tauri/src/main.rs +97 -399
  171. package/src-tauri/tauri.conf.json +1 -1
  172. package/static/app/asset-manifest.json +41 -37
  173. package/static/app/assets/Act-yYpYnn0v.js +1 -0
  174. package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
  175. package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
  176. package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
  177. package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
  178. package/static/app/assets/Capture-CFIRsFNE.js +1 -0
  179. package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
  180. package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
  181. package/static/app/assets/Library-DwO3yZST.js +1 -0
  182. package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
  183. package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
  184. package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
  185. package/static/app/assets/System-DW8F-2xL.js +1 -0
  186. package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
  187. package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
  188. package/static/app/assets/brain-Ci1CkWjM.js +1 -0
  189. package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
  190. package/static/app/assets/circle-check-DfInj-qD.js +1 -0
  191. package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
  192. package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
  193. package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
  194. package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
  195. package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
  196. package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
  197. package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
  198. package/static/app/assets/index-_u5iUHDr.js +10 -0
  199. package/static/app/assets/input-B0lPdRQZ.js +1 -0
  200. package/static/app/assets/link-2-CoFbooHS.js +1 -0
  201. package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
  202. package/static/app/assets/primitives-DEbN-d6p.js +1 -0
  203. package/static/app/assets/search-BybIWPNd.js +1 -0
  204. package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
  205. package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
  206. package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
  207. package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
  208. package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
  209. package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
  210. package/static/app/assets/utils-BlZr7Pd4.js +4 -0
  211. package/static/app/assets/workspace-jJY4RuAV.js +1 -0
  212. package/static/app/index.html +4 -4
  213. package/static/sw.js +1 -1
  214. package/lattice_brain/graph/_kg_common.py +0 -1331
  215. package/lattice_brain/graph/discovery_index.py +0 -1141
  216. package/lattice_brain/graph/retrieval.py +0 -1120
  217. package/lattice_brain/graph/retrieval_vector.py +0 -1293
  218. package/lattice_brain/ingestion.py +0 -1525
  219. package/lattice_brain/multimodal.py +0 -1258
  220. package/latticeai/core/agent.py +0 -1465
  221. package/latticeai/core/embedding_providers.py +0 -1196
  222. package/latticeai/core/file_generation.py +0 -1047
  223. package/latticeai/integrations/telegram_bot.py +0 -1390
  224. package/latticeai/models/router.py +0 -1007
  225. package/latticeai/runtime/build_phases.py +0 -1450
  226. package/latticeai/services/brain_intelligence.py +0 -1083
  227. package/latticeai/services/memory_service.py +0 -1177
  228. package/latticeai/services/model_runtime.py +0 -1281
  229. package/latticeai/setup/wizard.py +0 -1310
  230. package/static/app/assets/Act-AWf0SAKp.js +0 -1
  231. package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
  232. package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
  233. package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
  234. package/static/app/assets/Capture-CqOSzyPr.js +0 -1
  235. package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
  236. package/static/app/assets/Library-CX-bbhmK.js +0 -1
  237. package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
  238. package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
  239. package/static/app/assets/System-Bu2t5hn1.js +0 -1
  240. package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
  241. package/static/app/assets/brain-DJMoqrwx.js +0 -1
  242. package/static/app/assets/index-BpYkzcVm.js +0 -10
  243. package/static/app/assets/input-DSlJJxRs.js +0 -1
  244. package/static/app/assets/primitives-BCx6TvfG.js +0 -1
  245. package/static/app/assets/search-Cgy8cCFJ.js +0 -1
  246. package/static/app/assets/utils-zqPZJxdx.js +0 -4
  247. package/static/app/assets/workspace-DXTihhfU.js +0 -1
@@ -0,0 +1,394 @@
1
+ """The command screens — one Lattice-server read (or upload) rendered for a phone.
2
+
3
+ Everything reachable from the main menu that answers a question about the
4
+ running system rather than driving a conversation: server status, loaded
5
+ models, Knowledge Graph counts, a screenshot, chat history, the web-UI link,
6
+ the MCP tool list, and document upload into the graph.
7
+
8
+ Each screen follows the same shape: an optimistic chat action, one call through
9
+ :func:`~latticeai.integrations.telegram_bot.config._server_client`, and a plain
10
+ Korean rendering of exactly what came back — an unreachable server produces an
11
+ empty payload, never an invented one.
12
+
13
+ Stubbing note: these functions read ``_server_client``, ``_mac_ram_used_gb``,
14
+ ``send_photo``, ``send_message`` and ``download_telegram_file`` as *this*
15
+ module's globals, so a test standing in for any of them patches this module.
16
+ ``get_web_url``/``get_graph_url`` are read here but resolve
17
+ ``PUBLIC_WEB_URL``/``get_lan_ip``/``SERVER_PORT`` inside ``helpers``.
18
+ """
19
+
20
+ import asyncio
21
+ import os
22
+ import tempfile
23
+ from pathlib import Path
24
+
25
+ from latticeai.core.logging_safety import safe_log_text
26
+ from latticeai.core.quiet import quiet
27
+
28
+ from .config import (
29
+ API_URL,
30
+ BASE_URL,
31
+ GRAPH_STATS_URL,
32
+ HISTORY_URL,
33
+ MCP_TOOLS_URL,
34
+ MODELS_URL,
35
+ STATUS_URL,
36
+ UPLOAD_DOC_URL,
37
+ _server_client,
38
+ logger,
39
+ )
40
+ from .helpers import (
41
+ download_telegram_file,
42
+ get_graph_url,
43
+ get_web_url,
44
+ send_chat_action,
45
+ send_message,
46
+ send_photo,
47
+ )
48
+
49
+ # ── Main menu ─────────────────────────────────────────────────────────────────
50
+
51
+ MAIN_MENU = {
52
+ "inline_keyboard": [
53
+ [
54
+ {"text": "📊 서버 상태", "callback_data": "cmd:status"},
55
+ {"text": "🧠 현재 모델", "callback_data": "cmd:model"},
56
+ ],
57
+ [
58
+ {"text": "🕸 Knowledge Graph", "callback_data": "cmd:graph"},
59
+ {"text": "📸 스크린샷", "callback_data": "cmd:screenshot"},
60
+ ],
61
+ [
62
+ {"text": "📜 최근 대화 5건", "callback_data": "cmd:history"},
63
+ {"text": "🗑 기록 정리", "callback_data": "cmd:clear"},
64
+ ],
65
+ [
66
+ {"text": "🔗 웹 UI 열기", "callback_data": "cmd:web"},
67
+ {"text": "🔌 MCP 도구 목록", "callback_data": "cmd:mcp"},
68
+ ],
69
+ [
70
+ {"text": "🗂 변경 제안 검토", "callback_data": "cmd:review"},
71
+ ],
72
+ ]
73
+ }
74
+
75
+ async def show_menu(client, chat_id):
76
+ await send_message(client, chat_id, "📱 Lattice AI 원격 제어 메뉴입니다.", reply_markup=MAIN_MENU)
77
+
78
+ # ── Server status ─────────────────────────────────────────────────────────────
79
+
80
+ async def _mac_ram_used_gb() -> str:
81
+ try:
82
+ vm_proc = await asyncio.create_subprocess_exec(
83
+ "vm_stat", stdout=asyncio.subprocess.PIPE, stderr=asyncio.subprocess.DEVNULL
84
+ )
85
+ vm_out, _ = await vm_proc.communicate()
86
+ lines = vm_out.decode().splitlines()
87
+
88
+ # Parse page size from header line: "Mach Virtual Memory Statistics: (page size of 16384 bytes)"
89
+ page_size = 4096
90
+ if lines:
91
+ import re
92
+ m = re.search(r"page size of (\d+) bytes", lines[0])
93
+ if m:
94
+ page_size = int(m.group(1))
95
+
96
+ stats = {}
97
+ for line in lines[1:]:
98
+ if ":" in line:
99
+ k, _, v = line.partition(":")
100
+ try:
101
+ stats[k.strip()] = int(v.strip().rstrip(".")) * page_size
102
+ except ValueError:
103
+ quiet()
104
+
105
+ used = stats.get("Pages active", 0) + stats.get("Pages wired down", 0)
106
+
107
+ mem_proc = await asyncio.create_subprocess_exec(
108
+ "sysctl", "-n", "hw.memsize", stdout=asyncio.subprocess.PIPE
109
+ )
110
+ mem_out, _ = await mem_proc.communicate()
111
+ total = int(mem_out.strip())
112
+ return f"{used/1e9:.1f} GB / {total/1e9:.0f} GB"
113
+ except Exception:
114
+ return "N/A"
115
+
116
+ async def show_status(client, chat_id):
117
+ await send_chat_action(client, chat_id, "typing")
118
+ try:
119
+ async with _server_client() as lc:
120
+ res = await lc.get(STATUS_URL, timeout=5.0)
121
+ data = res.json() if res.status_code == 200 else {}
122
+ except Exception:
123
+ data = {}
124
+
125
+ ram = await _mac_ram_used_gb()
126
+ model = data.get("loaded_model") or "없음"
127
+ mode = data.get("mode") or "unknown"
128
+ state = "🟢 온라인" if data.get("status") == "online" else "🔴 오프라인"
129
+
130
+ text = (
131
+ f"📊 Lattice AI 서버 상태\n"
132
+ f"상태: {state}\n"
133
+ f"모드: {mode}\n"
134
+ f"모델: {model}\n"
135
+ f"RAM: {ram}"
136
+ )
137
+ await send_message(client, chat_id, text)
138
+
139
+ # ── Model info & unload ───────────────────────────────────────────────────────
140
+
141
+ async def show_model_info(client, chat_id):
142
+ await send_chat_action(client, chat_id, "typing")
143
+ try:
144
+ async with _server_client() as lc:
145
+ res = await lc.get(MODELS_URL, timeout=5.0)
146
+ data = res.json() if res.status_code == 200 else {}
147
+ except Exception:
148
+ data = {}
149
+
150
+ current = data.get("current") or "없음"
151
+ loaded = data.get("loaded") or []
152
+ loaded_str = "\n".join(f" - {m}" for m in loaded) if loaded else " 없음"
153
+ text = f"🧠 현재 모델: {current}\n\n로드된 모델:\n{loaded_str}"
154
+
155
+ markup = None
156
+ if loaded:
157
+ markup = {
158
+ "inline_keyboard": [
159
+ [{"text": f"🗑 {m} 언로드", "callback_data": f"model:unload:{m}"}]
160
+ for m in loaded
161
+ ] + [[{"text": "↩ 메뉴로", "callback_data": "cmd:menu"}]]
162
+ }
163
+ await send_message(client, chat_id, text, reply_markup=markup)
164
+
165
+ def _unload_all_report(results: list[tuple[str, int]]) -> str:
166
+ """Report an unload-all run from the statuses the server actually returned."""
167
+ failed = [(mid, code) for mid, code in results if code != 200]
168
+ if not failed:
169
+ return "✅ 모든 모델 언로드 완료. RAM이 해제되었습니다."
170
+ detail = ", ".join(f"{mid} ({code})" for mid, code in failed)
171
+ return (
172
+ f"일부 모델 언로드 실패: {detail}\n"
173
+ f"성공 {len(results) - len(failed)}개 / 실패 {len(failed)}개"
174
+ )
175
+
176
+ async def do_unload_model(client, chat_id, model_id: str = ""):
177
+ await send_chat_action(client, chat_id, "typing")
178
+ try:
179
+ results: list[tuple[str, int]] | None = None
180
+ async with _server_client() as lc:
181
+ if model_id:
182
+ res = await lc.delete(f"{BASE_URL}/models/unload/{model_id}", timeout=15.0)
183
+ else:
184
+ # Unload all: keep every delete's real status. Discarding them
185
+ # for a synthesized 200 reported "모든 모델 언로드 완료" even
186
+ # when a model refused to unload.
187
+ res = await lc.get(MODELS_URL, timeout=5.0)
188
+ if res.status_code == 200:
189
+ results = []
190
+ for mid in res.json().get("loaded") or []:
191
+ deleted = await lc.delete(f"{BASE_URL}/models/unload/{mid}", timeout=15.0)
192
+ results.append((mid, deleted.status_code))
193
+ if results is not None:
194
+ await send_message(client, chat_id, _unload_all_report(results))
195
+ elif res.status_code == 200:
196
+ await send_message(client, chat_id, f"✅ {model_id} 언로드 완료. RAM이 해제되었습니다.")
197
+ else:
198
+ await send_message(client, chat_id, f"언로드 실패 ({res.status_code})")
199
+ except Exception as e:
200
+ await send_message(client, chat_id, f"언로드 오류: {e}")
201
+
202
+ # ── Knowledge Graph stats ─────────────────────────────────────────────────────
203
+
204
+ async def show_graph_stats(client, chat_id):
205
+ await send_chat_action(client, chat_id, "typing")
206
+ try:
207
+ async with _server_client() as lc:
208
+ res = await lc.get(GRAPH_STATS_URL, timeout=5.0)
209
+ data = res.json() if res.status_code == 200 else {}
210
+ except Exception:
211
+ data = {}
212
+
213
+ nodes = data.get("nodes") or {}
214
+ edges = data.get("edges") or {}
215
+ total_nodes = sum(nodes.values())
216
+ total_edges = sum(edges.values())
217
+
218
+ node_lines = "\n".join(f" {t}: {c}" for t, c in sorted(nodes.items(), key=lambda x: -x[1])) or " 없음"
219
+ edge_lines = "\n".join(f" {t}: {c}" for t, c in sorted(edges.items(), key=lambda x: -x[1])[:8]) or " 없음"
220
+
221
+ text = (
222
+ f"🕸 Knowledge Graph 통계\n\n"
223
+ f"노드 총 {total_nodes}개:\n{node_lines}\n\n"
224
+ f"엣지 총 {total_edges}개:\n{edge_lines}\n\n"
225
+ f"그래프 보기: {get_graph_url()}"
226
+ )
227
+ markup = {
228
+ "inline_keyboard": [[
229
+ {"text": "🔗 그래프 열기", "url": get_graph_url()},
230
+ {"text": "↩ 메뉴로", "callback_data": "cmd:menu"},
231
+ ]]
232
+ }
233
+ await send_message(client, chat_id, text, reply_markup=markup)
234
+
235
+ # ── Screenshot ────────────────────────────────────────────────────────────────
236
+
237
+ async def take_screenshot(client, chat_id):
238
+ await send_chat_action(client, chat_id, "upload_photo")
239
+ # mkstemp, not mktemp: mktemp only predicts an unused name, leaving a
240
+ # window in which anything can create that path first. mkstemp creates
241
+ # the file atomically with 0600.
242
+ _fd, _name = tempfile.mkstemp(suffix=".jpg")
243
+ os.close(_fd)
244
+ tmp = Path(_name)
245
+ try:
246
+ proc = await asyncio.create_subprocess_exec(
247
+ "screencapture", "-x", str(tmp),
248
+ stdout=asyncio.subprocess.DEVNULL,
249
+ stderr=asyncio.subprocess.DEVNULL,
250
+ )
251
+ await asyncio.wait_for(proc.communicate(), timeout=10.0)
252
+ if tmp.exists() and tmp.stat().st_size > 0:
253
+ await send_photo(client, chat_id, tmp, caption="현재 화면입니다.")
254
+ else:
255
+ await send_message(client, chat_id, "스크린샷 파일이 생성되지 않았습니다. screencapture가 설치되어 있는지 확인하세요.")
256
+ except asyncio.TimeoutError:
257
+ await send_message(client, chat_id, "스크린샷 시간 초과")
258
+ except FileNotFoundError:
259
+ await send_message(client, chat_id, "screencapture 명령이 없습니다. macOS에서만 동작합니다.")
260
+ except Exception as e:
261
+ await send_message(client, chat_id, f"스크린샷 오류: {e}")
262
+ finally:
263
+ try:
264
+ tmp.unlink(missing_ok=True)
265
+ except Exception:
266
+ quiet()
267
+
268
+ # ── History ───────────────────────────────────────────────────────────────────
269
+
270
+ async def show_history_summary(client, chat_id, n: int = 5):
271
+ await send_chat_action(client, chat_id, "typing")
272
+ try:
273
+ async with _server_client() as lc:
274
+ res = await lc.get(HISTORY_URL, timeout=10.0)
275
+ items = res.json() if res.status_code == 200 else []
276
+ except Exception:
277
+ items = []
278
+
279
+ if not items:
280
+ await send_message(client, chat_id, "저장된 대화 기록이 없습니다.")
281
+ return
282
+
283
+ recent = [i for i in items if i.get("role") == "user"][-n:]
284
+ lines = [f"📜 최근 사용자 메시지 {len(recent)}건\n"]
285
+ for item in recent:
286
+ ts = str(item.get("timestamp", ""))[:16]
287
+ src = item.get("source", "web")
288
+ content = str(item.get("content", ""))[:120].replace("\n", " ")
289
+ lines.append(f"[{ts}] ({src}) {content}")
290
+ await send_message(client, chat_id, "\n".join(lines))
291
+
292
+ async def clear_server_history(client, chat_id, keep_last=0):
293
+ try:
294
+ async with _server_client() as lc:
295
+ res = await lc.delete(HISTORY_URL, params={"keep_last": keep_last}, timeout=10.0)
296
+ data = res.json() if res.headers.get("content-type", "").startswith("application/json") else {}
297
+ if res.status_code == 200:
298
+ await send_message(client, chat_id, f"대화 기록을 정리했습니다. 삭제 {data.get('removed', 0)}개, 유지 {data.get('kept', 0)}개.")
299
+ else:
300
+ await send_message(client, chat_id, f"대화 기록 정리 실패: {res.status_code}")
301
+ except Exception as e:
302
+ await send_message(client, chat_id, f"대화 기록 정리 오류: {e}")
303
+
304
+ # ── Web UI link ───────────────────────────────────────────────────────────────
305
+
306
+ async def send_web_link(client, chat_id):
307
+ web_url = get_web_url()
308
+ text = (
309
+ "웹 UI 링크입니다.\n"
310
+ f"{web_url}\n\n"
311
+ "핸드폰이 Mac과 같은 Wi-Fi에 있어야 바로 열립니다. "
312
+ "외부망에서 쓰려면 LATTICEAI_PUBLIC_URL에 터널 주소를 설정하세요."
313
+ )
314
+ payload = {
315
+ "chat_id": chat_id,
316
+ "text": text,
317
+ "reply_markup": {
318
+ "inline_keyboard": [[
319
+ {"text": "Lattice AI Web 열기", "url": web_url},
320
+ {"text": "Knowledge Graph", "url": get_graph_url()},
321
+ ]]
322
+ },
323
+ }
324
+ try:
325
+ # The Telegram client, like every other helper here. Sending this on the
326
+ # server client shipped the local bearer capability to api.telegram.org
327
+ # and failed outright whenever that token was unset.
328
+ await client.post(f"{API_URL}/sendMessage", json=payload)
329
+ except Exception as e:
330
+ logger.error("웹 링크 전송 실패: %s", safe_log_text(e))
331
+
332
+ # ── MCP tools ─────────────────────────────────────────────────────────────────
333
+
334
+ async def send_mcp_tools(client, chat_id):
335
+ try:
336
+ async with _server_client() as lc:
337
+ res = await lc.get(MCP_TOOLS_URL, timeout=10.0)
338
+ if res.status_code != 200:
339
+ await send_message(client, chat_id, f"MCP 도구 목록을 가져오지 못했습니다: {res.status_code}")
340
+ return
341
+ data = res.json()
342
+ names = [tool["name"] for tool in data.get("tools", [])]
343
+ await send_message(client, chat_id, "사용 가능한 MCP 도구:\n" + ("\n".join(f"- {n}" for n in names) or "없음"))
344
+ except Exception as e:
345
+ await send_message(client, chat_id, f"MCP 도구 조회 실패: {e}")
346
+
347
+ # ── Document upload → knowledge graph ────────────────────────────────────────
348
+
349
+ async def process_document_file(client, chat_id, file_id: str, filename: str, caption: str = ""):
350
+ await send_chat_action(client, chat_id, "upload_document")
351
+ raw = await download_telegram_file(client, file_id)
352
+ if not raw:
353
+ await send_message(client, chat_id, "파일 다운로드 실패")
354
+ return
355
+
356
+ suffix = Path(filename).suffix.lower()
357
+ allowed = {".pdf", ".docx", ".xlsx", ".pptx", ".txt", ".md", ".csv"}
358
+ if suffix not in allowed:
359
+ await send_message(client, chat_id,
360
+ f"지원하지 않는 파일 형식입니다({suffix}). "
361
+ f"지원 형식: {', '.join(sorted(allowed))}")
362
+ return
363
+
364
+ _fd, _name = tempfile.mkstemp(suffix=suffix) # see take_screenshot
365
+ os.close(_fd)
366
+ tmp = Path(_name)
367
+ try:
368
+ tmp.write_bytes(raw)
369
+ async with _server_client() as lc:
370
+ res = await lc.post(
371
+ UPLOAD_DOC_URL,
372
+ files={"file": (filename, raw)},
373
+ timeout=60.0,
374
+ )
375
+ if res.status_code == 200:
376
+ data = res.json()
377
+ chars = data.get("chars") or len(raw)
378
+ preview = str(data.get("preview") or "")[:300]
379
+ kg = data.get("knowledge_graph") or {}
380
+ node_id = kg.get("node_id", "")
381
+ text = (
382
+ f"✅ {filename} 수집 완료\n"
383
+ f"크기: {len(raw) // 1024} KB | 문자: {chars}\n"
384
+ f"노드: {node_id}\n"
385
+ f"\n미리보기:\n{preview}"
386
+ )
387
+ await send_message(client, chat_id, text)
388
+ else:
389
+ err = res.json().get("detail") if res.headers.get("content-type", "").startswith("application/json") else res.text
390
+ await send_message(client, chat_id, f"업로드 실패 ({res.status_code}): {err}")
391
+ except Exception as e:
392
+ await send_message(client, chat_id, f"문서 처리 오류: {e}")
393
+ finally:
394
+ tmp.unlink(missing_ok=True)
@@ -0,0 +1,88 @@
1
+ """
2
+ LLM Router — mlx-vlm 기반 Gemma 4 최적화 및 추측 디코딩(Speculative Decoding) 코어
3
+
4
+ v11.3.0 turned this module into a package. :class:`LLMRouter` is composed from
5
+ four cohesive mixins, each of which moved here verbatim:
6
+
7
+ * :mod:`.loading` — the guarded optional backends and every method that reads
8
+ them (``load_model``, ``_load_cloud_model``, ``_release_memory``);
9
+ * :mod:`.registry` — the locked model registry, eviction, and the immutable
10
+ request-scoped snapshot generation runs against;
11
+ * :mod:`.generation` — chat generation and streaming, local and cloud;
12
+ * :mod:`.documents` — the same backends driven by a caller-supplied system
13
+ prompt.
14
+
15
+ Around them: :mod:`.branding` (system prompt + legacy-alias rewrite),
16
+ :mod:`.errors` (the typed mid-stream failure), :mod:`.catalog` (model refs and
17
+ provenance), :mod:`.local_models` (finding a downloaded model on disk).
18
+
19
+ Every name this module exported still resolves from
20
+ ``latticeai.models.router`` — with one deliberate exception, spelled out
21
+ because it is the whole reason the loading half is one module:
22
+
23
+ ``mx`` / ``vlm_load`` / ``lm_load`` / ``VLM_AVAILABLE`` / ``LM_AVAILABLE``
24
+ are **rebound at runtime** by :func:`ensure_mlx_runtime` after an installer
25
+ has run. Re-exporting them here would publish the import-time value
26
+ forever, so ``ensure_mlx_runtime`` would appear to do nothing. Read them —
27
+ and stand in for them — on ``latticeai.models.router.loading``, where they
28
+ live.
29
+
30
+ Stubbing note, same shape: a name rebound *here* changes only this module's
31
+ binding. The submodule that calls it holds its own, so a test standing in for
32
+ a collaborator patches the submodule that reads it.
33
+ """
34
+
35
+ # The catalog data lives in .model_providers; re-exported here so
36
+ # ``from latticeai.models.router import OPENAI_COMPATIBLE_PROVIDERS`` (and the
37
+ # model_runtime re-export chain) resolve unchanged after the split.
38
+ from latticeai.core.quiet import quiet as quiet
39
+ from latticeai.models.model_providers import (
40
+ MODEL_SOURCE_BY_FAMILY as MODEL_SOURCE_BY_FAMILY,
41
+ )
42
+ from latticeai.models.model_providers import (
43
+ OPENAI_COMPATIBLE_PROVIDERS as OPENAI_COMPATIBLE_PROVIDERS,
44
+ )
45
+ from latticeai.models.model_providers import (
46
+ PROVIDER_MODEL_CATALOG as PROVIDER_MODEL_CATALOG,
47
+ )
48
+
49
+ from .branding import BRAND_NAME as BRAND_NAME
50
+ from .branding import CITATION_INSTRUCTION as CITATION_INSTRUCTION
51
+ from .branding import LEGACY_BRAND_PATTERNS as LEGACY_BRAND_PATTERNS
52
+ from .branding import SYSTEM_PROMPT as SYSTEM_PROMPT
53
+ from .branding import _compose_system as _compose_system
54
+ from .branding import normalize_branding as normalize_branding
55
+ from .catalog import CloudModel as CloudModel
56
+ from .catalog import parse_model_ref as parse_model_ref
57
+ from .catalog import source_metadata_for_model as source_metadata_for_model
58
+ from .documents import _DocumentMixin
59
+ from .errors import ModelStreamError as ModelStreamError
60
+ from .errors import _stream_failure as _stream_failure
61
+ from .generation import _GenerationMixin
62
+
63
+ # ``AsyncOpenAI`` and ``executor`` are bound once at import and never rebound,
64
+ # so a re-export is the same object the loading half uses — unlike the five MLX
65
+ # names named in the module docstring.
66
+ from .loading import AsyncOpenAI as AsyncOpenAI
67
+ from .loading import _LoadingMixin
68
+ from .loading import _mlx_sampler as _mlx_sampler
69
+ from .loading import ensure_mlx_runtime as ensure_mlx_runtime
70
+ from .loading import executor as executor
71
+ from .local_models import HF_MODELS_ROOT as HF_MODELS_ROOT
72
+ from .local_models import _is_gemma4_model_id as _is_gemma4_model_id
73
+ from .local_models import _local_model_type as _local_model_type
74
+ from .local_models import _looks_like_hf_model_dir as _looks_like_hf_model_dir
75
+ from .local_models import _resolve_local_hf_model as _resolve_local_hf_model
76
+ from .local_models import hf_cache_model_dir as hf_cache_model_dir
77
+ from .local_models import hf_model_dir as hf_model_dir
78
+ from .registry import _RegistryMixin
79
+
80
+
81
+ class LLMRouter(_LoadingMixin, _RegistryMixin, _GenerationMixin, _DocumentMixin):
82
+ """The multi-engine router, composed from its four cohesive halves.
83
+
84
+ The mixins define disjoint method sets, so resolution order changes nothing
85
+ at runtime: this class exposes exactly the methods it exposed when they all
86
+ lived in one 1,007-line module. ``__init__`` comes from the registry half,
87
+ which owns the state the other three read.
88
+ """
@@ -0,0 +1,66 @@
1
+ """The seam the four LLMRouter mixins share.
2
+
3
+ ``LLMRouter`` is assembled from the loading, registry, generation and document
4
+ mixins. Each reads state it does not own — the registry dict and its lock, the
5
+ snapshot helper, the cloud error hint — because the point of the split is that
6
+ "how a model is loaded" and "how a document is streamed" stop sharing a
7
+ 1,007-line file, not that they stop sharing ``self``.
8
+
9
+ Typing-only, exactly like :mod:`lattice_brain.ingestion._contract`: the
10
+ declarations below are never the implementation, so the MRO and every method
11
+ resolution stay byte-for-byte what the single-file class had. Adding a
12
+ cross-mixin call without declaring it here is a type error.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from typing import Any, AsyncIterator, Dict, Optional, Tuple
18
+
19
+ from .catalog import CloudModel
20
+
21
+
22
+ class RouterCore:
23
+ """What any router mixin may assume about ``self``.
24
+
25
+ Never instantiated directly. Members are declared, not implemented: the
26
+ implementation lives in whichever mixin owns it.
27
+ """
28
+
29
+ # ── State owned by _RegistryMixin.__init__ ───────────────────────────────
30
+ #: A local entry is ``(model, tokenizer, draft_model, loader_kind)``; a
31
+ #: cloud entry is a :class:`CloudModel`.
32
+ _cache: Dict[str, Any]
33
+ _current: Optional[str]
34
+ _last_used: Dict[str, float]
35
+ _max_local_models: int
36
+ #: ``threading.RLock``; annotated ``Any`` because the runtime lock type is
37
+ #: private in typeshed and nothing here depends on its identity.
38
+ _lock: Any
39
+
40
+ # ── registry.py: reached from the load path ──────────────────────────────
41
+ def _touch(self, model_id: Optional[str] = None) -> None:
42
+ raise NotImplementedError
43
+
44
+ def _enforce_local_model_limit(self, incoming_key: str) -> None:
45
+ raise NotImplementedError
46
+
47
+ # ── loading.py: reached from every unload path ───────────────────────────
48
+ def _release_memory(self) -> None:
49
+ raise NotImplementedError
50
+
51
+ # ── registry.py: reached from both generation halves ─────────────────────
52
+ def _model_snapshot(
53
+ self, model_id: Optional[str] = None
54
+ ) -> tuple[Optional[str], object | None]:
55
+ raise NotImplementedError
56
+
57
+ def _unpack_local_cache(self, cached: Any) -> Tuple[Any, Any, Any, str]:
58
+ raise NotImplementedError
59
+
60
+ def _local_server_error_hint(self, cloud: CloudModel, error: Exception) -> str:
61
+ raise NotImplementedError
62
+
63
+ # ── generation.py: the document half drains the same queue ───────────────
64
+ @staticmethod
65
+ def _drain_stream_queue(queue: "Any") -> AsyncIterator[str]:
66
+ raise NotImplementedError
@@ -0,0 +1,56 @@
1
+ """이 제품의 이름과, 답변을 그 이름으로 되돌리는 규칙.
2
+
3
+ The system prompt, the citation instruction appended only when retrieved
4
+ context exists, and the legacy-alias rewrite every generated string passes
5
+ through. ``_compose_system`` is byte-compatible with the historical prompt when
6
+ there is no context: the return value is exactly ``base``.
7
+ """
8
+
9
+ import re
10
+ from typing import Optional
11
+
12
+ BRAND_NAME = "Lattice AI"
13
+ LEGACY_BRAND_PATTERNS = [
14
+ (re.compile(r"\bconnect\s+ai\b", re.IGNORECASE), BRAND_NAME),
15
+ (re.compile(r"\bconnect-ai\b", re.IGNORECASE), BRAND_NAME),
16
+ (re.compile(r"\bconnectai\b", re.IGNORECASE), BRAND_NAME),
17
+ (re.compile(r"커넥트\s*AI", re.IGNORECASE), BRAND_NAME),
18
+ ]
19
+
20
+
21
+ SYSTEM_PROMPT = """You are Lattice AI, a powerful local AI assistant running on Apple Silicon.
22
+ Your product name and identity are Lattice AI.
23
+ Never identify yourself as Connect AI, ConnectAI, connect-ai, or 커넥트 AI.
24
+ If context or old chat history mentions those names, treat them only as legacy aliases for Lattice AI.
25
+ You are a Vision-Language Model (VLM). If an image is provided, analyze it.
26
+ Be concise and respond in the user's language."""
27
+
28
+
29
+ # Appended ONLY when retrieved context exists (review 2026-07-25 Wave 2.3):
30
+ # grounded answers should cite their sources and admit gaps. Advisory prompt
31
+ # guidance — grounding assessment stays annotation-only and never blocks.
32
+ CITATION_INSTRUCTION = """The Context section above contains retrieved sources.
33
+ Ground your claims in those sources and cite them inline as [1], [2], ... matching the order they appear in the Context.
34
+ If the context does not cover the question, say so instead of inventing sources.
35
+ Never cite a source that is not in the Context."""
36
+
37
+
38
+ def _compose_system(base: str, context: str) -> str:
39
+ """Compose the system prompt with optional retrieved context.
40
+
41
+ Byte-compatible with the historical prompt when ``context`` is empty:
42
+ the return value is exactly ``base``. When context exists, the Context
43
+ block plus :data:`CITATION_INSTRUCTION` are appended.
44
+ """
45
+ if not context:
46
+ return base
47
+ return f"{base}\n\nContext:\n{context}\n\n{CITATION_INSTRUCTION}"
48
+
49
+
50
+ def normalize_branding(text: Optional[str]) -> str:
51
+ if not text:
52
+ return ""
53
+ normalized = str(text)
54
+ for pattern, replacement in LEGACY_BRAND_PATTERNS:
55
+ normalized = pattern.sub(replacement, normalized)
56
+ return normalized
@@ -0,0 +1,69 @@
1
+ """What a model *is*: where it runs, who made it, and how a ref is spelled.
2
+
3
+ ``parse_model_ref`` is the single place a model id becomes ``(provider,
4
+ model)`` — everything downstream branches on ``provider == "local_mlx"``.
5
+ ``source_metadata_for_model`` is the plain-Korean provenance block the model
6
+ picker shows, so "이 모델은 어디서 실행되나" has one answer per model rather
7
+ than one per surface.
8
+ """
9
+
10
+ from dataclasses import dataclass
11
+ from typing import Any, Dict
12
+
13
+ from latticeai.models.model_providers import (
14
+ MODEL_SOURCE_BY_FAMILY,
15
+ OPENAI_COMPATIBLE_PROVIDERS,
16
+ )
17
+
18
+
19
+ # Returns a display payload whose `source_display_order` value is a list,
20
+ # so the value type is Any rather than str.
21
+ def source_metadata_for_model(
22
+ provider: str, model: Dict[str, Any], *, local_server: bool
23
+ ) -> Dict[str, Any]:
24
+ family = str(model.get("family") or "")
25
+ country, company = MODEL_SOURCE_BY_FAMILY.get(family, ("미상", provider.title()))
26
+ if local_server:
27
+ execution_method = "내 컴퓨터에서만 실행"
28
+ internet_requirement = "모델을 다운로드할 때만 인터넷 필요; 실행 중에는 필요 없음"
29
+ else:
30
+ execution_method = "인터넷 연결 후 사용"
31
+ internet_requirement = "내 파일이 인터넷으로 전송될 수 있음"
32
+ return {
33
+ "source_country": country,
34
+ "source_company": company,
35
+ "execution_method": execution_method,
36
+ "internet_requirement": internet_requirement,
37
+ "model_name": model.get("name") or model.get("id") or "",
38
+ "source_display_order": [
39
+ "source_country",
40
+ "source_company",
41
+ "execution_method",
42
+ "internet_requirement",
43
+ "model_name",
44
+ ],
45
+ }
46
+
47
+
48
+ @dataclass
49
+ class CloudModel:
50
+ provider: str
51
+ model: str
52
+ client: Any # AsyncOpenAI when the optional dependency is installed
53
+ cache_key: str
54
+
55
+
56
+ def parse_model_ref(model_id: str) -> tuple[str, str]:
57
+ """Return (provider, model). Unprefixed refs stay local MLX."""
58
+ if model_id.startswith("cloud:"):
59
+ _, provider, model = model_id.split(":", 2)
60
+ return provider, model
61
+ if ":" in model_id:
62
+ provider, model = model_id.split(":", 1)
63
+ if provider in OPENAI_COMPATIBLE_PROVIDERS:
64
+ return provider, model
65
+ if provider in {"local_mlx", "mlx"}:
66
+ return "local_mlx", model
67
+ if model_id.startswith("local_mlx:"):
68
+ return "local_mlx", model_id.split(":", 1)[1] # pragma: no cover — dead: a "local_mlx:" ref always has a ":" and returned above
69
+ return "local_mlx", model_id