ltcai 11.1.0 → 11.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (118) hide show
  1. package/README.md +53 -53
  2. package/docs/CHANGELOG.md +33 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/FEATURE_AUDIT_v11.2.0.md +393 -0
  6. package/docs/LAYOUT_REBUILD_SPEC.md +9 -1
  7. package/docs/ONBOARDING.md +1 -1
  8. package/docs/OPERATIONS.md +1 -1
  9. package/docs/TRUST_MODEL.md +1 -1
  10. package/docs/WHY_LATTICE.md +1 -1
  11. package/docs/architecture.md +6 -2
  12. package/docs/kg-schema.md +1 -1
  13. package/lattice_brain/__init__.py +1 -1
  14. package/lattice_brain/gates.py +125 -0
  15. package/lattice_brain/graph/fusion.py +35 -4
  16. package/lattice_brain/graph/projection.py +66 -8
  17. package/lattice_brain/graph/schema.py +9 -0
  18. package/lattice_brain/graph/store.py +9 -0
  19. package/lattice_brain/graph/vector_index/selector.py +32 -2
  20. package/lattice_brain/ingestion.py +175 -27
  21. package/lattice_brain/multimodal.py +525 -5
  22. package/lattice_brain/portability.py +169 -32
  23. package/lattice_brain/runtime/multi_agent.py +1 -1
  24. package/lattice_brain/sealed_box.py +244 -0
  25. package/lattice_brain/synthesis.py +24 -1
  26. package/latticeai/__init__.py +1 -1
  27. package/latticeai/api/brain_intelligence.py +4 -0
  28. package/latticeai/api/chat.py +11 -0
  29. package/latticeai/api/chat_helpers.py +16 -3
  30. package/latticeai/api/chat_hybrid.py +32 -1
  31. package/latticeai/api/features.py +70 -0
  32. package/latticeai/api/local_files.py +102 -0
  33. package/latticeai/api/portability.py +39 -4
  34. package/latticeai/api/review_queue.py +126 -0
  35. package/latticeai/api/search.py +16 -2
  36. package/latticeai/core/agent.py +55 -2
  37. package/latticeai/core/config.py +4 -1
  38. package/latticeai/core/context_builder.py +6 -3
  39. package/latticeai/core/legacy_compatibility.py +1 -1
  40. package/latticeai/core/marketplace.py +1 -1
  41. package/latticeai/core/messages.py +143 -0
  42. package/latticeai/core/model_compat.py +73 -2
  43. package/latticeai/core/workspace_os_constants.py +1 -1
  44. package/latticeai/models/model_providers.py +12 -4
  45. package/latticeai/runtime/build_phases.py +28 -0
  46. package/latticeai/runtime/chat_wiring.py +4 -0
  47. package/latticeai/runtime/feature_toggle_wiring.py +163 -0
  48. package/latticeai/runtime/router_registration.py +11 -0
  49. package/latticeai/services/app_context.py +8 -0
  50. package/latticeai/services/architecture_readiness.py +1 -1
  51. package/latticeai/services/automation_intelligence.py +22 -2
  52. package/latticeai/services/brain_intelligence.py +123 -7
  53. package/latticeai/services/command_center.py +10 -4
  54. package/latticeai/services/feature_toggles.py +502 -0
  55. package/latticeai/services/folder_watch.py +122 -1
  56. package/latticeai/services/hybrid_chat.py +56 -5
  57. package/latticeai/services/interop_bridges.py +978 -0
  58. package/latticeai/services/model_capability_registry.py +434 -261
  59. package/latticeai/services/model_catalog.py +95 -61
  60. package/latticeai/services/model_recommendation.py +18 -11
  61. package/latticeai/services/model_runtime.py +1 -1
  62. package/latticeai/services/multimodal_ports.py +26 -1
  63. package/latticeai/services/obsidian_bridge.py +16 -25
  64. package/latticeai/services/product_readiness.py +1 -1
  65. package/latticeai/services/search_service.py +149 -2
  66. package/latticeai/services/tool_dispatch.py +4 -0
  67. package/latticeai/setup/auto_setup.py +27 -30
  68. package/latticeai/setup/wizard.py +77 -44
  69. package/package.json +1 -1
  70. package/scripts/check_current_release_docs.mjs +1 -1
  71. package/scripts/check_server_i18n.mjs +1 -0
  72. package/scripts/release_screen_claims.json +13 -0
  73. package/scripts/verify_hf_model_registry.py +253 -218
  74. package/src-tauri/Cargo.lock +1 -1
  75. package/src-tauri/Cargo.toml +1 -1
  76. package/src-tauri/tauri.conf.json +1 -1
  77. package/static/app/asset-manifest.json +37 -37
  78. package/static/app/assets/{Act-D0HWqtn0.js → Act-AWf0SAKp.js} +1 -1
  79. package/static/app/assets/{AdminConsole-D-QDW-A4.js → AdminConsole-D0u8Tiyj.js} +1 -1
  80. package/static/app/assets/{Brain-CzCsI1mi.js → Brain-tuhI4sOC.js} +1 -1
  81. package/static/app/assets/BrainHome-Ts7G_Ila.js +2 -0
  82. package/static/app/assets/{BrainSignals-2dHQNkns.js → BrainSignals-jMYgQ2Ar.js} +1 -1
  83. package/static/app/assets/{Capture-CT8v1StE.js → Capture-CqOSzyPr.js} +1 -1
  84. package/static/app/assets/{CommandPalette-DoLXC2KH.js → CommandPalette-DC0Bzh-I.js} +1 -1
  85. package/static/app/assets/{Library-DDoxFE5c.js → Library-CX-bbhmK.js} +1 -1
  86. package/static/app/assets/{LivingBrain-BXMWIK_2.js → LivingBrain-DBwhto14.js} +1 -1
  87. package/static/app/assets/{ProductFlow-DOYf7JIs.js → ProductFlow-BHA2cfKI.js} +1 -1
  88. package/static/app/assets/{ReviewCard-COQsqidK.js → ReviewCard-BUhCKRNM.js} +1 -1
  89. package/static/app/assets/{System-BRllvYXd.js → System-Bu2t5hn1.js} +1 -1
  90. package/static/app/assets/arrow-left-Dzwa5zRb.js +1 -0
  91. package/static/app/assets/{bot-4BvN07ux.js → bot-Cia42c2h.js} +1 -1
  92. package/static/app/assets/brain-DJMoqrwx.js +1 -0
  93. package/static/app/assets/{button-CDjtnAoU.js → button-2j2Ijzgq.js} +1 -1
  94. package/static/app/assets/{circle-pause-D_RMn7tp.js → circle-pause-BEFeWpVW.js} +1 -1
  95. package/static/app/assets/{circle-play-B5OpB8ae.js → circle-play-ujXMcHxl.js} +1 -1
  96. package/static/app/assets/{cpu-BIlWInHf.js → cpu-k4awryFq.js} +1 -1
  97. package/static/app/assets/{download-BtjXfL3z.js → download-DFbLJ_ig.js} +1 -1
  98. package/static/app/assets/{folder-open-DefMpxI2.js → folder-open-7y_b6xkM.js} +1 -1
  99. package/static/app/assets/{hard-drive-BQ8NZVkw.js → hard-drive-Bidh02Kr.js} +1 -1
  100. package/static/app/assets/{index-0AvoEBzJ.js → index-BpYkzcVm.js} +3 -3
  101. package/static/app/assets/{index-vtEfYvQY.css → index-DwDl9-8Y.css} +1 -1
  102. package/static/app/assets/{input-B_5ZJ9oy.js → input-DSlJJxRs.js} +1 -1
  103. package/static/app/assets/{permissionCopy-BqZ5tsgL.js → permissionCopy-Bpb83Hx9.js} +1 -1
  104. package/static/app/assets/{primitives-CVwew78r.js → primitives-BCx6TvfG.js} +1 -1
  105. package/static/app/assets/search-Cgy8cCFJ.js +1 -0
  106. package/static/app/assets/{share-2-D5zg_0fY.js → share-2-BH1M-WNi.js} +1 -1
  107. package/static/app/assets/{shield-alert-B5pZzkUb.js → shield-alert-BlKdBXcG.js} +1 -1
  108. package/static/app/assets/{textarea-nEVIweKY.js → textarea-CCWbUfFB.js} +1 -1
  109. package/static/app/assets/{useFocusTrap-Cm99AHlz.js → useFocusTrap-YdHQ7pJ1.js} +1 -1
  110. package/static/app/assets/{useQuery-Dm__N6bL.js → useQuery-CXQiwbVT.js} +1 -1
  111. package/static/app/assets/{utils-DcDMoZIe.js → utils-zqPZJxdx.js} +2 -2
  112. package/static/app/assets/{workspace-LtRRSKTf.js → workspace-DXTihhfU.js} +1 -1
  113. package/static/app/index.html +4 -4
  114. package/static/sw.js +1 -1
  115. package/static/app/assets/BrainHome-Btns-_TA.js +0 -2
  116. package/static/app/assets/arrow-left-DnyMzss-.js +0 -1
  117. package/static/app/assets/brain-uMb_5hnO.js +0 -1
  118. package/static/app/assets/search-DkhnOKZt.js +0 -1
@@ -1,8 +1,8 @@
1
- """Structured Model Capability Registry for Lattice AI 5.2.0+.
1
+ """Structured Model Capability Registry for Lattice AI.
2
2
 
3
3
  User-focused, transparent model catalog with:
4
- - HF repo provenance
5
- - Modality / vision support
4
+ - HF repo provenance (exact-case repo id, pinned against the HF API)
5
+ - Modality / vision support and the HF config architecture (``model_type``)
6
6
  - Quantization, size, download/load strategies
7
7
  - Hardware notes (RAM estimates, Apple Silicon affinity)
8
8
  - License / safety notes
@@ -12,10 +12,30 @@ This replaces the flat ENGINE_MODEL_CATALOG construction with a richer,
12
12
  queryable source of truth while preserving exact legacy shapes for
13
13
  model_catalog / recommendation / API / frontend consumers.
14
14
 
15
- All entries are recommended multimodal (VLM) first. Text-only can be added later.
16
- Verification is honest: hf_exists + light metadata/config presence; full weights
17
- are never auto-fetched by the verifier. Large models (>12GB) explicitly note
18
- "local load practical only on high-RAM Apple Silicon or CUDA; expect long download".
15
+ 11.2.0 — two lists, one registry
16
+ ================================
17
+ A model can be *offered* or merely *understood*, and conflating the two either
18
+ pushes dead downloads at people or orphans the weights already on their disk.
19
+ Every entry therefore carries a :attr:`ModelCapability.lifecycle`:
20
+
21
+ ``RECOMMENDED``
22
+ Current generation. Appears in ENGINE_MODEL_CATALOG, the recommendation
23
+ tiers, and the download paths.
24
+ ``LEGACY``
25
+ Superseded, but real and still loadable. **Never** offered for download and
26
+ never recommended — it exists so a model a user already downloaded keeps its
27
+ name, size, family and runtime profile instead of showing up as an unknown
28
+ blob. See :func:`get_legacy_capabilities` / :func:`is_recognized_model`.
29
+
30
+ Models that no longer exist on the Hub, or that cannot be fetched without
31
+ credentials (gated), are **deleted outright** rather than demoted: pretending to
32
+ recognise something nobody can obtain is not compatibility, it is noise.
33
+
34
+ Verification is honest: hf_exists + metadata/siblings presence measured through
35
+ the public HF REST API; full weights are never auto-fetched by the verifier and
36
+ no model is ever loaded to produce these flags. Large models (>12GB) explicitly
37
+ note "local load practical only on high-RAM Apple Silicon or CUDA; expect long
38
+ download".
19
39
  """
20
40
 
21
41
  from __future__ import annotations
@@ -47,6 +67,13 @@ class VerificationStatus:
47
67
  verified_by: str = "hf-api-light" # or "local-load-test"
48
68
 
49
69
 
70
+ # ── Lifecycle vocabulary ──────────────────────────────────────────────────────
71
+ #: Current generation: offered for download, listed in the catalog, recommended.
72
+ RECOMMENDED = "recommended"
73
+ #: Superseded but still loadable: recognised only, never offered or recommended.
74
+ LEGACY = "legacy"
75
+
76
+
50
77
  @dataclass(frozen=True)
51
78
  class ModelCapability:
52
79
  """Rich capability entry. id is the canonical key used in ENGINE_MODEL_CATALOG."""
@@ -57,6 +84,15 @@ class ModelCapability:
57
84
  tag: str
58
85
  size: str # display string "7.6GB", "pull required"
59
86
  modality: str = "multimodal" # multimodal | vision | text | audio etc.
87
+ #: HF ``config.json`` ``model_type`` — the string mlx-lm / mlx-vlm dispatches
88
+ #: on. Pinned from the Hub API so the verifier can judge loadability
89
+ #: statically, without downloading or importing anything.
90
+ architecture: str = ""
91
+ #: Measured sum of the repo's sibling file sizes, in GB (10^9 bytes).
92
+ #: ``None`` for tool-managed engines that pull on demand.
93
+ download_size_gb: Optional[float] = None
94
+ #: RECOMMENDED (offered + recommended) or LEGACY (recognised only).
95
+ lifecycle: str = RECOMMENDED
60
96
  quantization: Optional[str] = None # "4bit", "Q4_K_M", "GGUF-Q4"
61
97
  provider_hints: List[str] = field(default_factory=lambda: ["local_mlx"]) # which engines this id primarily maps to
62
98
  download_strategy: str = "hf_hub" # hf_hub | ollama_pull | lmstudio_app | gguf_manual
@@ -85,6 +121,8 @@ class ModelCapability:
85
121
  "size": self.size,
86
122
  "pullable": True,
87
123
  "modality": self.modality,
124
+ "architecture": self.architecture,
125
+ "lifecycle": self.lifecycle,
88
126
  "source_country": self.source_country,
89
127
  "source_company": self.source_company,
90
128
  "execution_method": self.execution_method,
@@ -127,30 +165,48 @@ class ModelCapability:
127
165
  return base
128
166
 
129
167
 
130
- # ── Curated 5.2.0 registry (bold user-focused: transparent, multimodal-first, verified where practical) ──
131
- # Current Gemma-4 / Qwen3-VL / Llama-4 kept + modern additions (Gemma3, Qwen2.5-VL, Llama-3.2-Vision, Pixtral).
132
- # Presence verified via HF API (lightweight model_info) on 2026-06-14.
133
- # Full weight download is user-consent only; entries without config/tokenizer
134
- # hints are shown as available-but-not-load-verified.
168
+ # ── Curated registry ──────────────────────────────────────────────────────────
169
+ # Every field below was measured against the public HF REST API on 2026-08-10:
170
+ # repo existence, gated flag, library_name/tags, config ``model_type``, sibling
171
+ # file list and the exact byte sum of those siblings (``size`` and
172
+ # ``download_size_gb``). Nothing here was inferred from a model card, and no
173
+ # weights were downloaded to produce it — re-run
174
+ # ``scripts/verify_hf_model_registry.py`` to re-measure.
175
+ #
176
+ # Repo ids are stored in the Hub's own canonical casing (the API echoes the
177
+ # canonical id even when queried with different case, e.g. requesting
178
+ # ``gemma-4-12b-it-4bit`` answers with ``gemma-4-12B-it-4bit``). Storing the
179
+ # canonical form keeps the download path, the on-disk cache directory and the
180
+ # catalog key identical.
135
181
 
136
182
  _REGISTRY: List[ModelCapability] = [
137
- # Gemma 4 family (mlx-community 4-bit, Apple-first, excellent local VLM)
183
+ # ── Ultralight tier (≤8GB RAM) ───────────────────────────────────────────
138
184
  ModelCapability(
139
- id="mlx-community/gemma-4-e2b-4bit",
140
- hf_repo_id="mlx-community/gemma-4-e2b-4bit",
141
- name="Gemma 4 E2B Base",
142
- family="Gemma 4",
143
- tag="local-vlm",
144
- size="3.6GB",
185
+ id="mlx-community/LFM2.5-2.6B-4bit",
186
+ hf_repo_id="mlx-community/LFM2.5-2.6B-4bit",
187
+ name="LFM2.5 2.6B",
188
+ family="LFM2.5",
189
+ tag="local-llm",
190
+ size="1.5GB",
191
+ modality="text",
192
+ architecture="lfm2",
193
+ download_size_gb=1.54,
145
194
  quantization="4bit",
146
- provider_hints=["local_mlx"],
195
+ provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
147
196
  download_strategy="hf_hub",
148
- load_strategy="mlx_vlm",
149
- hardware=HardwareProfile(min_ram_gb=6.0, recommended_ram_gb=8.0, apple_silicon_pref=True, notes="Tiny but capable vision; great first local VLM."),
150
- source_country="미국", source_company="Google",
151
- verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="any-to-any", license="apache-2.0", verified_by="hf-api-light"),
197
+ load_strategy="mlx_lm",
198
+ hardware=HardwareProfile(
199
+ min_ram_gb=6.0, recommended_ram_gb=8.0, apple_silicon_pref=True,
200
+ notes="Smallest model that still answers in Korean. Text only — no image understanding.",
201
+ ),
202
+ source_country="미국", source_company="Liquid AI",
203
+ verification=VerificationStatus(
204
+ hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
205
+ has_weights_hint=True, pipeline_tag="text-generation", likes=3, license="apache-2.0",
206
+ notes="library_name=mlx, ungated, 1 safetensors shard.", verified_by="hf-api-light",
207
+ ),
152
208
  recommended_default=True,
153
- display_priority=10,
209
+ display_priority=5,
154
210
  ),
155
211
  ModelCapability(
156
212
  id="mlx-community/gemma-4-e2b-it-4bit",
@@ -159,79 +215,214 @@ _REGISTRY: List[ModelCapability] = [
159
215
  family="Gemma 4",
160
216
  tag="local-vlm",
161
217
  size="3.6GB",
218
+ architecture="gemma4",
219
+ download_size_gb=3.58,
162
220
  quantization="4bit",
163
- provider_hints=["local_mlx"],
221
+ provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
164
222
  download_strategy="hf_hub",
165
223
  load_strategy="mlx_vlm",
166
- hardware=HardwareProfile(min_ram_gb=6.0, recommended_ram_gb=8.0, apple_silicon_pref=True, notes="Instruct-tuned; preferred over base for chat."),
224
+ hardware=HardwareProfile(
225
+ min_ram_gb=8.0, recommended_ram_gb=8.0, apple_silicon_pref=True,
226
+ notes="Smallest model here that can still read pictures. Good first local VLM.",
227
+ ),
167
228
  source_country="미국", source_company="Google",
168
- verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="any-to-any", license="apache-2.0", verified_by="hf-api-light"),
229
+ verification=VerificationStatus(
230
+ hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
231
+ has_weights_hint=True, pipeline_tag="image-text-to-text", likes=25, license="apache-2.0",
232
+ notes="library_name=mlx, ungated, 1 safetensors shard.", verified_by="hf-api-light",
233
+ ),
169
234
  recommended_default=True,
170
- display_priority=11,
235
+ display_priority=10,
171
236
  ),
237
+
238
+ # ── Light tier (16GB RAM) ────────────────────────────────────────────────
172
239
  ModelCapability(
173
- id="mlx-community/gemma-4-e4b-4bit",
174
- hf_repo_id="mlx-community/gemma-4-e4b-4bit",
175
- name="Gemma 4 E4B Base",
240
+ id="mlx-community/gemma-4-e4b-it-4bit",
241
+ hf_repo_id="mlx-community/gemma-4-e4b-it-4bit",
242
+ name="Gemma 4 E4B Instruct",
176
243
  family="Gemma 4",
177
244
  tag="local-vlm",
178
245
  size="5.2GB",
246
+ architecture="gemma4",
247
+ download_size_gb=5.18,
179
248
  quantization="4bit",
180
- provider_hints=["local_mlx"],
249
+ provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
181
250
  download_strategy="hf_hub",
182
251
  load_strategy="mlx_vlm",
183
- hardware=HardwareProfile(min_ram_gb=8.0, recommended_ram_gb=10.0, apple_silicon_pref=True),
252
+ hardware=HardwareProfile(
253
+ min_ram_gb=10.0, recommended_ram_gb=16.0, apple_silicon_pref=True,
254
+ notes="Step up from E2B with the same tiny footprint discipline.",
255
+ ),
184
256
  source_country="미국", source_company="Google",
185
- verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="any-to-any", license="apache-2.0", verified_by="hf-api-light"),
257
+ verification=VerificationStatus(
258
+ hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
259
+ has_weights_hint=True, pipeline_tag="image-text-to-text", likes=35, license="apache-2.0",
260
+ notes="library_name=mlx, ungated, 1 safetensors shard.", verified_by="hf-api-light",
261
+ ),
262
+ display_priority=15,
186
263
  ),
264
+
265
+ # ── Mid tier (24GB RAM) ──────────────────────────────────────────────────
187
266
  ModelCapability(
188
- id="mlx-community/gemma-4-e4b-it-4bit",
189
- hf_repo_id="mlx-community/gemma-4-e4b-it-4bit",
190
- name="Gemma 4 E4B Instruct",
267
+ id="mlx-community/gemma-4-12B-it-4bit",
268
+ hf_repo_id="mlx-community/gemma-4-12B-it-4bit",
269
+ name="Gemma 4 12B Instruct",
191
270
  family="Gemma 4",
192
271
  tag="local-vlm",
193
- size="5.2GB",
272
+ size="6.8GB",
273
+ architecture="gemma4_unified",
274
+ download_size_gb=6.77,
194
275
  quantization="4bit",
195
- provider_hints=["local_mlx"],
276
+ provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
196
277
  download_strategy="hf_hub",
197
278
  load_strategy="mlx_vlm",
198
- hardware=HardwareProfile(min_ram_gb=8.0, recommended_ram_gb=10.0, apple_silicon_pref=True),
279
+ hardware=HardwareProfile(
280
+ min_ram_gb=12.0, recommended_ram_gb=16.0, apple_silicon_pref=True,
281
+ notes="Sweet spot for local multimodal on M-series 16GB+. Uses the gemma4_unified MLX loader.",
282
+ ),
199
283
  source_country="미국", source_company="Google",
200
- verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="any-to-any", license="apache-2.0", verified_by="hf-api-light"),
284
+ verification=VerificationStatus(
285
+ hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
286
+ has_weights_hint=True, pipeline_tag="image-text-to-text", likes=16, license="apache-2.0",
287
+ notes="library_name=mlx, ungated, 2 safetensors shards. Canonical id capitalises the B.",
288
+ verified_by="hf-api-light",
289
+ ),
290
+ recommended_default=True,
291
+ display_priority=25,
201
292
  ),
202
293
  ModelCapability(
203
- id="mlx-community/gemma-4-12b-it-4bit",
204
- hf_repo_id="mlx-community/gemma-4-12b-it-4bit",
205
- name="Gemma 4 12B Instruct",
206
- family="Gemma 4",
294
+ id="mlx-community/Qwen3.5-9B-MLX-4bit",
295
+ hf_repo_id="mlx-community/Qwen3.5-9B-MLX-4bit",
296
+ name="Qwen3.5 9B",
297
+ family="Qwen3.5",
207
298
  tag="local-vlm",
208
- size="7.6GB",
299
+ size="6.0GB",
300
+ architecture="qwen3_5",
301
+ download_size_gb=5.98,
209
302
  quantization="4bit",
210
- provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
303
+ provider_hints=["local_mlx", "vllm"],
211
304
  download_strategy="hf_hub",
212
305
  load_strategy="mlx_vlm",
213
- hardware=HardwareProfile(min_ram_gb=12.0, recommended_ram_gb=16.0, apple_silicon_pref=True, notes="Sweet spot for local multimodal on M-series 16GB+ or 24GB+."),
214
- source_country="미국", source_company="Google",
215
- verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="apache-2.0", verified_by="hf-api-light"),
306
+ hardware=HardwareProfile(
307
+ min_ram_gb=12.0, recommended_ram_gb=16.0, apple_silicon_pref=True,
308
+ notes="Mid-size vision-language model; replaces the retired Qwen2.5-VL / Llama-3.2-Vision slot.",
309
+ ),
310
+ source_country="중국", source_company="Alibaba",
311
+ verification=VerificationStatus(
312
+ hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
313
+ has_weights_hint=True, pipeline_tag="image-text-to-text", likes=153, license="apache-2.0",
314
+ notes="library_name=mlx, ungated, 2 safetensors shards.", verified_by="hf-api-light",
315
+ ),
216
316
  recommended_default=True,
217
317
  display_priority=20,
218
318
  ),
319
+
320
+ # ── General-purpose ──────────────────────────────────────────────────────
321
+ ModelCapability(
322
+ id="mlx-community/gpt-oss-20b-MXFP4-Q8",
323
+ hf_repo_id="mlx-community/gpt-oss-20b-MXFP4-Q8",
324
+ name="GPT-OSS 20B",
325
+ family="GPT-OSS",
326
+ tag="local-llm",
327
+ size="12.1GB",
328
+ modality="text",
329
+ architecture="gpt_oss",
330
+ download_size_gb=12.10,
331
+ quantization="MXFP4-Q8",
332
+ provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
333
+ download_strategy="hf_hub",
334
+ load_strategy="mlx_lm",
335
+ hardware=HardwareProfile(
336
+ min_ram_gb=18.0, recommended_ram_gb=24.0, apple_silicon_pref=True,
337
+ notes="Most-downloaded entry in this catalog. Text only — no image understanding.",
338
+ ),
339
+ source_country="미국", source_company="OpenAI",
340
+ verification=VerificationStatus(
341
+ hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
342
+ has_weights_hint=True, pipeline_tag="text-generation", likes=84, license="apache-2.0",
343
+ notes="library_name=mlx, ungated, 3 safetensors shards.", verified_by="hf-api-light",
344
+ ),
345
+ display_priority=30,
346
+ ),
347
+
348
+ # ── MoE tier (32GB+ RAM) ─────────────────────────────────────────────────
219
349
  ModelCapability(
220
350
  id="mlx-community/gemma-4-26b-a4b-it-4bit",
221
351
  hf_repo_id="mlx-community/gemma-4-26b-a4b-it-4bit",
222
352
  name="Gemma 4 26B A4B Instruct",
223
353
  family="Gemma 4",
224
354
  tag="local-vlm",
225
- size="15.6GB",
355
+ size="15.4GB",
356
+ architecture="gemma4",
357
+ download_size_gb=15.37,
226
358
  quantization="4bit",
227
- provider_hints=["local_mlx"],
359
+ provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
228
360
  download_strategy="hf_hub",
229
361
  load_strategy="mlx_vlm",
230
- hardware=HardwareProfile(min_ram_gb=20.0, recommended_ram_gb=28.0, apple_silicon_pref=True, notes="Large MoE-style; local load practical only on high-RAM Apple Silicon (32GB+). Long download expected."),
362
+ hardware=HardwareProfile(
363
+ min_ram_gb=22.0, recommended_ram_gb=32.0, apple_silicon_pref=True,
364
+ notes="Mixture-of-experts; practical on 32GB+ Apple Silicon. Long first download.",
365
+ ),
231
366
  source_country="미국", source_company="Google",
232
- verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="apache-2.0", verified_by="hf-api-light"),
367
+ verification=VerificationStatus(
368
+ hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
369
+ has_weights_hint=True, pipeline_tag="image-text-to-text", likes=80, license="apache-2.0",
370
+ notes="library_name=mlx, ungated, 3 safetensors shards.", verified_by="hf-api-light",
371
+ ),
233
372
  display_priority=50,
234
373
  ),
374
+ ModelCapability(
375
+ id="mlx-community/Qwen3.6-35B-A3B-4bit",
376
+ hf_repo_id="mlx-community/Qwen3.6-35B-A3B-4bit",
377
+ name="Qwen3.6 35B A3B",
378
+ family="Qwen3.6",
379
+ tag="local-vlm",
380
+ size="20.4GB",
381
+ architecture="qwen3_5_moe",
382
+ download_size_gb=20.43,
383
+ quantization="4bit",
384
+ provider_hints=["local_mlx", "vllm"],
385
+ download_strategy="hf_hub",
386
+ load_strategy="mlx_vlm",
387
+ hardware=HardwareProfile(
388
+ min_ram_gb=28.0, recommended_ram_gb=48.0, apple_silicon_pref=True,
389
+ notes="Mixture-of-experts; replaces the retired Llama 4 Scout slot. Long first download.",
390
+ ),
391
+ source_country="중국", source_company="Alibaba",
392
+ verification=VerificationStatus(
393
+ hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
394
+ has_weights_hint=True, pipeline_tag="image-text-to-text", likes=94, license="apache-2.0",
395
+ notes="library_name=mlx, ungated, 4 safetensors shards.", verified_by="hf-api-light",
396
+ ),
397
+ display_priority=65,
398
+ ),
399
+
400
+ # ── Large tier (48GB+ RAM) ───────────────────────────────────────────────
401
+ ModelCapability(
402
+ id="mlx-community/Qwen3.6-27B-4bit",
403
+ hf_repo_id="mlx-community/Qwen3.6-27B-4bit",
404
+ name="Qwen3.6 27B",
405
+ family="Qwen3.6",
406
+ tag="local-vlm",
407
+ size="16.1GB",
408
+ architecture="qwen3_5",
409
+ download_size_gb=16.08,
410
+ quantization="4bit",
411
+ provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
412
+ download_strategy="hf_hub",
413
+ load_strategy="mlx_vlm",
414
+ hardware=HardwareProfile(
415
+ min_ram_gb=24.0, recommended_ram_gb=48.0, apple_silicon_pref=True,
416
+ notes="Dense top tier — every parameter runs on every token, so it is slower but steadier than the MoE.",
417
+ ),
418
+ source_country="중국", source_company="Alibaba",
419
+ verification=VerificationStatus(
420
+ hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
421
+ has_weights_hint=True, pipeline_tag="image-text-to-text", likes=50, license="apache-2.0",
422
+ notes="library_name=mlx, ungated, 3 safetensors shards.", verified_by="hf-api-light",
423
+ ),
424
+ display_priority=55,
425
+ ),
235
426
  ModelCapability(
236
427
  id="mlx-community/gemma-4-31b-it-4bit",
237
428
  hf_repo_id="mlx-community/gemma-4-31b-it-4bit",
@@ -239,32 +430,93 @@ _REGISTRY: List[ModelCapability] = [
239
430
  family="Gemma 4",
240
431
  tag="local-vlm",
241
432
  size="18.4GB",
433
+ architecture="gemma4",
434
+ download_size_gb=18.44,
242
435
  quantization="4bit",
243
- provider_hints=["local_mlx", "ollama", "vllm"],
436
+ provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
244
437
  download_strategy="hf_hub",
245
438
  load_strategy="mlx_vlm",
246
- hardware=HardwareProfile(min_ram_gb=24.0, recommended_ram_gb=32.0, apple_silicon_pref=True, notes="Very large; high-end local only. Consider cloud fallback for lower RAM."),
439
+ hardware=HardwareProfile(
440
+ min_ram_gb=26.0, recommended_ram_gb=48.0, apple_silicon_pref=True,
441
+ notes="Largest Gemma 4 here; high-end local only. Consider a cloud path on lower RAM.",
442
+ ),
247
443
  source_country="미국", source_company="Google",
248
- verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="apache-2.0", verified_by="hf-api-light"),
444
+ verification=VerificationStatus(
445
+ hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
446
+ has_weights_hint=True, pipeline_tag="image-text-to-text", likes=46, license="apache-2.0",
447
+ notes="library_name=mlx, ungated, 4 safetensors shards.", verified_by="hf-api-light",
448
+ ),
449
+ display_priority=60,
249
450
  ),
451
+ ]
452
+
453
+
454
+ # ── Recognised-only entries (LEGACY) ──────────────────────────────────────────
455
+ # Superseded generations that are still on the Hub and still loadable. They are
456
+ # deliberately absent from ENGINE_MODEL_CATALOG, from the recommendation tiers
457
+ # and from every download path — their only job is to keep a model somebody
458
+ # already downloaded identifiable (name, family, size, runtime profile) instead
459
+ # of surfacing as an unknown blob. Verified on 2026-08-10 like the rest.
250
460
 
251
- # Qwen3-VL (strong real-world multimodal, good small sizes)
461
+ _LEGACY_REGISTRY: List[ModelCapability] = [
462
+ ModelCapability(
463
+ id="mlx-community/gemma-4-e2b-4bit",
464
+ hf_repo_id="mlx-community/gemma-4-e2b-4bit",
465
+ name="Gemma 4 E2B Base",
466
+ family="Gemma 4",
467
+ tag="local-vlm",
468
+ size="3.6GB",
469
+ architecture="gemma4",
470
+ download_size_gb=3.61,
471
+ lifecycle=LEGACY,
472
+ quantization="4bit",
473
+ hardware=HardwareProfile(min_ram_gb=8.0, recommended_ram_gb=8.0, apple_silicon_pref=True,
474
+ notes="Base (not instruction-tuned) — the Instruct build answers chat far better."),
475
+ source_country="미국", source_company="Google",
476
+ verification=VerificationStatus(
477
+ hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
478
+ has_weights_hint=True, pipeline_tag="any-to-any", likes=3, license="apache-2.0",
479
+ notes="Superseded by gemma-4-e2b-it-4bit for chat.", verified_by="hf-api-light",
480
+ ),
481
+ ),
482
+ ModelCapability(
483
+ id="mlx-community/gemma-4-e4b-4bit",
484
+ hf_repo_id="mlx-community/gemma-4-e4b-4bit",
485
+ name="Gemma 4 E4B Base",
486
+ family="Gemma 4",
487
+ tag="local-vlm",
488
+ size="5.3GB",
489
+ architecture="gemma4",
490
+ download_size_gb=5.25,
491
+ lifecycle=LEGACY,
492
+ quantization="4bit",
493
+ hardware=HardwareProfile(min_ram_gb=10.0, recommended_ram_gb=16.0, apple_silicon_pref=True,
494
+ notes="Base (not instruction-tuned) — the Instruct build answers chat far better."),
495
+ source_country="미국", source_company="Google",
496
+ verification=VerificationStatus(
497
+ hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
498
+ has_weights_hint=True, pipeline_tag="any-to-any", likes=6, license="apache-2.0",
499
+ notes="Superseded by gemma-4-e4b-it-4bit for chat.", verified_by="hf-api-light",
500
+ ),
501
+ ),
252
502
  ModelCapability(
253
503
  id="mlx-community/Qwen3-VL-4B-Instruct-4bit",
254
504
  hf_repo_id="mlx-community/Qwen3-VL-4B-Instruct-4bit",
255
505
  name="Qwen3-VL 4B",
256
506
  family="Qwen3-VL",
257
507
  tag="local-vlm",
258
- size="2.7GB",
508
+ size="3.1GB",
509
+ architecture="qwen3_vl",
510
+ download_size_gb=3.11,
511
+ lifecycle=LEGACY,
259
512
  quantization="4bit",
260
- provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
261
- download_strategy="hf_hub",
262
- load_strategy="mlx_vlm",
263
- hardware=HardwareProfile(min_ram_gb=5.0, recommended_ram_gb=8.0, apple_silicon_pref=True, notes="Extremely compact strong VLM. Best default for low-RAM Macs."),
513
+ hardware=HardwareProfile(min_ram_gb=5.0, recommended_ram_gb=8.0, apple_silicon_pref=True),
264
514
  source_country="중국", source_company="Alibaba",
265
- verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="apache-2.0", verified_by="hf-api-light"),
266
- recommended_default=True,
267
- display_priority=5,
515
+ verification=VerificationStatus(
516
+ hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
517
+ has_weights_hint=True, pipeline_tag="image-text-to-text", likes=7, license="apache-2.0",
518
+ notes="2025-10 generation; superseded by Qwen3.5 / Qwen3.6.", verified_by="hf-api-light",
519
+ ),
268
520
  ),
269
521
  ModelCapability(
270
522
  id="mlx-community/Qwen3-VL-8B-Instruct-4bit",
@@ -272,16 +524,18 @@ _REGISTRY: List[ModelCapability] = [
272
524
  name="Qwen3-VL 8B",
273
525
  family="Qwen3-VL",
274
526
  tag="local-vlm",
275
- size="4.8GB",
527
+ size="5.8GB",
528
+ architecture="qwen3_vl",
529
+ download_size_gb=5.78,
530
+ lifecycle=LEGACY,
276
531
  quantization="4bit",
277
- provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
278
- download_strategy="hf_hub",
279
- load_strategy="mlx_vlm",
280
532
  hardware=HardwareProfile(min_ram_gb=8.0, recommended_ram_gb=12.0, apple_silicon_pref=True),
281
533
  source_country="중국", source_company="Alibaba",
282
- verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="apache-2.0", verified_by="hf-api-light"),
283
- recommended_default=True,
284
- display_priority=15,
534
+ verification=VerificationStatus(
535
+ hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
536
+ has_weights_hint=True, pipeline_tag="image-text-to-text", likes=6, license="apache-2.0",
537
+ notes="2025-10 generation; superseded by Qwen3.5 9B.", verified_by="hf-api-light",
538
+ ),
285
539
  ),
286
540
  ModelCapability(
287
541
  id="mlx-community/Qwen3-VL-30B-A3B-Instruct-4bit",
@@ -289,215 +543,139 @@ _REGISTRY: List[ModelCapability] = [
289
543
  name="Qwen3-VL 30B A3B",
290
544
  family="Qwen3-VL",
291
545
  tag="local-vlm",
292
- size="18GB",
546
+ size="18.3GB",
547
+ architecture="qwen3_vl_moe",
548
+ download_size_gb=18.27,
549
+ lifecycle=LEGACY,
293
550
  quantization="4bit",
294
- provider_hints=["local_mlx", "ollama", "vllm"],
295
- download_strategy="hf_hub",
296
- load_strategy="mlx_vlm",
297
- hardware=HardwareProfile(min_ram_gb=24.0, recommended_ram_gb=32.0, apple_silicon_pref=True, notes="Large MoE VLM; practical local only on 32GB+ Apple Silicon or strong CUDA. Download is multi-GB."),
551
+ hardware=HardwareProfile(min_ram_gb=24.0, recommended_ram_gb=32.0, apple_silicon_pref=True),
298
552
  source_country="중국", source_company="Alibaba",
299
- verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="apache-2.0", verified_by="hf-api-light"),
300
- ),
301
-
302
- # Llama 4
303
- ModelCapability(
304
- id="mlx-community/Llama-4-Scout-17B-16E-Instruct-4bit",
305
- hf_repo_id="mlx-community/Llama-4-Scout-17B-16E-Instruct-4bit",
306
- name="Llama 4 Scout 17B 16E",
307
- family="Llama 4",
308
- tag="local-vlm",
309
- size="11.8GB",
310
- quantization="4bit",
311
- provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
312
- download_strategy="hf_hub",
313
- load_strategy="mlx_vlm",
314
- hardware=HardwareProfile(min_ram_gb=16.0, recommended_ram_gb=20.0, apple_silicon_pref=True),
315
- source_country="미국", source_company="Meta",
316
- verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="llama3.1-ish / meta-llama", verified_by="hf-api-light"),
317
- recommended_default=True,
318
- display_priority=25,
319
- ),
320
-
321
- # ── Modern additions for 5.2.0 (verified on HF, user choice expansion) ──
322
- # Gemma 3 (excellent real multimodal balance, smaller than 4 where present)
323
- ModelCapability(
324
- id="google/gemma-3-4b-it",
325
- hf_repo_id="google/gemma-3-4b-it",
326
- name="Gemma 3 4B Instruct (HF)",
327
- family="Gemma 3",
328
- tag="local-vlm",
329
- size="~5GB+",
330
- quantization="bf16 / 4bit variants",
331
- provider_hints=["local_mlx", "vllm", "ollama"],
332
- download_strategy="hf_hub",
333
- load_strategy="mlx_vlm",
334
- hardware=HardwareProfile(min_ram_gb=8.0, recommended_ram_gb=12.0, apple_silicon_pref=True, notes="Use mlx-community quantized ports when available for best local perf."),
335
- source_country="미국", source_company="Google",
336
- verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="gemma-terms", verified_by="hf-api-light"),
337
- display_priority=30,
338
- ),
339
- ModelCapability(
340
- id="google/gemma-3-12b-it",
341
- hf_repo_id="google/gemma-3-12b-it",
342
- name="Gemma 3 12B Instruct (HF)",
343
- family="Gemma 3",
344
- tag="local-vlm",
345
- size="~12GB+",
346
- quantization="bf16 / GGUF-4bit",
347
- provider_hints=["ollama", "vllm", "lmstudio", "llamacpp"],
348
- download_strategy="hf_hub",
349
- load_strategy="ollama",
350
- hardware=HardwareProfile(min_ram_gb=16.0, recommended_ram_gb=20.0, notes="Prefer quantized GGUF for llama.cpp / ollama on non-Apple or lower RAM."),
351
- source_country="미국", source_company="Google",
352
- verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="gemma-terms", verified_by="hf-api-light"),
553
+ verification=VerificationStatus(
554
+ hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
555
+ has_weights_hint=True, pipeline_tag="image-text-to-text", likes=8, license="apache-2.0",
556
+ notes="2025-10 generation; superseded by Qwen3.6 35B A3B.", verified_by="hf-api-light",
557
+ ),
353
558
  ),
354
-
355
- # Qwen2.5-VL (battle-tested, widely supported)
356
559
  ModelCapability(
357
- id="Qwen/Qwen2.5-VL-7B-Instruct",
358
- hf_repo_id="Qwen/Qwen2.5-VL-7B-Instruct",
359
- name="Qwen2.5-VL 7B Instruct",
560
+ id="mlx-community/Qwen2.5-VL-7B-Instruct-4bit",
561
+ hf_repo_id="mlx-community/Qwen2.5-VL-7B-Instruct-4bit",
562
+ name="Qwen2.5-VL 7B",
360
563
  family="Qwen2.5-VL",
361
564
  tag="local-vlm",
362
- size="~8-15GB (quant dependent)",
363
- quantization="AWQ / GGUF / 4bit ports",
364
- provider_hints=["vllm", "ollama", "lmstudio", "llamacpp"],
365
- download_strategy="hf_hub",
366
- load_strategy="vllm",
367
- hardware=HardwareProfile(min_ram_gb=12.0, recommended_ram_gb=16.0, cuda_pref=True, notes="Strong general VLM. mlx-community or GGUF ports recommended for local Apple."),
565
+ size="5.7GB",
566
+ architecture="qwen2_5_vl",
567
+ download_size_gb=5.65,
568
+ lifecycle=LEGACY,
569
+ quantization="4bit",
570
+ hardware=HardwareProfile(min_ram_gb=8.0, recommended_ram_gb=12.0, apple_silicon_pref=True),
368
571
  source_country="중국", source_company="Alibaba",
369
- verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="apache-2.0", verified_by="hf-api-light"),
370
- display_priority=35,
371
- ),
372
-
373
- # Llama 3.2 Vision (widely available, good ecosystem)
374
- ModelCapability(
375
- id="meta-llama/Llama-3.2-11B-Vision-Instruct",
376
- hf_repo_id="meta-llama/Llama-3.2-11B-Vision-Instruct",
377
- name="Llama 3.2 11B Vision Instruct",
378
- family="Llama 3.2 Vision",
379
- tag="local-vlm",
380
- size="~11-22GB (quant)",
381
- quantization="Q4_K_M GGUF widely available",
382
- provider_hints=["ollama", "llamacpp", "lmstudio", "vllm"],
383
- download_strategy="hf_hub",
384
- load_strategy="ollama",
385
- hardware=HardwareProfile(min_ram_gb=14.0, recommended_ram_gb=18.0, notes="Excellent GGUF support. Ollama / llama.cpp default path for most users."),
386
- source_country="미국", source_company="Meta",
387
- verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="llama3.2", verified_by="hf-api-light"),
388
- display_priority=40,
389
- ),
390
-
391
- # Pixtral (Mistral multimodal, strong)
392
- ModelCapability(
393
- id="mistralai/Pixtral-12B-2409",
394
- hf_repo_id="mistralai/Pixtral-12B-2409",
395
- name="Pixtral 12B (Mistral)",
396
- family="Pixtral",
397
- tag="local-vlm",
398
- size="~12-24GB",
399
- quantization="GGUF / AWQ ports",
400
- provider_hints=["vllm", "ollama", "lmstudio"],
401
- download_strategy="hf_hub",
402
- load_strategy="vllm",
403
- hardware=HardwareProfile(min_ram_gb=16.0, recommended_ram_gb=20.0, cuda_pref=True, notes="High quality vision-language. Best on CUDA / vLLM; GGUF for CPU/Apple via community ports."),
404
- source_country="프랑스", source_company="Mistral AI",
405
572
  verification=VerificationStatus(
406
- hf_exists=True,
407
- has_config=False,
408
- has_tokenizer=False,
409
- has_weights_hint=True,
410
- pipeline_tag=None,
411
- license="mistral-research",
412
- notes="HF repo and weights are present, but config/tokenizer files were not visible in the lightweight HF tree check; treat as available but not local-load verified.",
573
+ hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
574
+ has_weights_hint=True, pipeline_tag="image-text-to-text", likes=4, license="apache-2.0",
575
+ notes="2025-02 generation, library_name=transformers; superseded by Qwen3.5 9B.",
413
576
  verified_by="hf-api-light",
414
577
  ),
415
- display_priority=45,
416
578
  ),
417
-
418
- # Additional recent multimodal (2025-2026 era ports, MLX-first where possible)
419
579
  ModelCapability(
420
580
  id="mlx-community/Llama-3.2-11B-Vision-Instruct-4bit",
421
581
  hf_repo_id="mlx-community/Llama-3.2-11B-Vision-Instruct-4bit",
422
582
  name="Llama 3.2 11B Vision Instruct",
423
583
  family="Llama 3.2 Vision",
424
584
  tag="local-vlm",
425
- size="6.8GB",
585
+ size="6.0GB",
586
+ architecture="mllama",
587
+ download_size_gb=6.02,
588
+ lifecycle=LEGACY,
426
589
  quantization="4bit",
427
- provider_hints=["local_mlx", "ollama", "llamacpp"],
428
- download_strategy="hf_hub",
429
- load_strategy="mlx_vlm",
430
- hardware=HardwareProfile(min_ram_gb=10.0, recommended_ram_gb=14.0, apple_silicon_pref=True, notes="Excellent vision for its size. Great all-rounder multimodal."),
590
+ hardware=HardwareProfile(min_ram_gb=10.0, recommended_ram_gb=14.0, apple_silicon_pref=True),
431
591
  source_country="미국", source_company="Meta",
432
- verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="llama3.2", verified_by="hf-api-light"),
433
- recommended_default=True,
434
- display_priority=25,
435
- ),
436
- ModelCapability(
437
- id="mlx-community/phi-3.5-vision-4bit",
438
- hf_repo_id="mlx-community/phi-3.5-vision-4bit",
439
- name="Phi-3.5 Vision 4B",
440
- family="Phi Vision",
441
- tag="local-vlm",
442
- size="3.2GB",
443
- quantization="4bit",
444
- provider_hints=["local_mlx"],
445
- download_strategy="hf_hub",
446
- load_strategy="mlx_vlm",
447
- hardware=HardwareProfile(min_ram_gb=6.0, recommended_ram_gb=8.0, apple_silicon_pref=True, notes="Microsoft small VLM, fast and capable for on-device vision tasks."),
448
- source_country="미국", source_company="Microsoft",
449
- verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="mit", verified_by="hf-api-light"),
450
- display_priority=30,
451
- ),
452
- ModelCapability(
453
- id="mlx-community/Qwen2.5-VL-7B-Instruct-4bit",
454
- hf_repo_id="mlx-community/Qwen2.5-VL-7B-Instruct-4bit",
455
- name="Qwen2.5-VL 7B",
456
- family="Qwen2.5-VL",
457
- tag="local-vlm",
458
- size="4.5GB",
459
- quantization="4bit",
460
- provider_hints=["local_mlx", "ollama"],
461
- download_strategy="hf_hub",
462
- load_strategy="mlx_vlm",
463
- hardware=HardwareProfile(min_ram_gb=8.0, recommended_ram_gb=12.0, apple_silicon_pref=True, notes="Strong Chinese/English vision model, updated from Qwen2-VL."),
464
- source_country="중국", source_company="Alibaba",
465
- verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="apache-2.0", verified_by="hf-api-light"),
466
- recommended_default=True,
467
- display_priority=18,
592
+ license="llama3.2",
593
+ verification=VerificationStatus(
594
+ hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
595
+ has_weights_hint=True, pipeline_tag="image-text-to-text", likes=7, license="llama3.2",
596
+ notes="2024-10 generation, library_name=transformers; superseded by Qwen3.5 9B.",
597
+ verified_by="hf-api-light",
598
+ ),
468
599
  ),
469
600
  ModelCapability(
470
- id="mlx-community/moondream2-4bit",
471
- hf_repo_id="mlx-community/moondream2-4bit",
472
- name="Moondream2 (small VLM)",
473
- family="Moondream",
601
+ id="mlx-community/Llama-4-Scout-17B-16E-Instruct-4bit",
602
+ hf_repo_id="mlx-community/Llama-4-Scout-17B-16E-Instruct-4bit",
603
+ name="Llama 4 Scout 17B 16E",
604
+ family="Llama 4",
474
605
  tag="local-vlm",
475
- size="1.8GB",
606
+ size="61.1GB",
607
+ architecture="llama4",
608
+ download_size_gb=61.14,
609
+ lifecycle=LEGACY,
476
610
  quantization="4bit",
477
- provider_hints=["local_mlx"],
478
- download_strategy="hf_hub",
479
- load_strategy="mlx_vlm",
480
- hardware=HardwareProfile(min_ram_gb=4.0, recommended_ram_gb=6.0, apple_silicon_pref=True, notes="Ultra-light vision model for quick descriptions and simple QA. Ideal for low-resource or frequent use."),
481
- source_country="미국", source_company="vikhyatk",
482
- verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="apache-2.0", verified_by="hf-api-light"),
483
- display_priority=8,
611
+ hardware=HardwareProfile(
612
+ min_ram_gb=72.0, recommended_ram_gb=96.0, apple_silicon_pref=True,
613
+ notes="61GB on disk despite the \"4bit\" name — 12 shards. The registry claimed 11.8GB "
614
+ "until it was measured; only a very high-RAM machine can load it.",
615
+ ),
616
+ source_country="미국", source_company="Meta",
617
+ license="llama4",
618
+ verification=VerificationStatus(
619
+ hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
620
+ has_weights_hint=True, pipeline_tag="image-text-to-text", likes=12, license="llama4",
621
+ notes="2025-05 generation, library_name=transformers; the MoE slot is now Qwen3.6 35B A3B. "
622
+ "Meta's own repo is gated, so this port is the only anonymous route.",
623
+ verified_by="hf-api-light",
624
+ ),
484
625
  ),
485
626
  ]
486
627
 
487
628
 
488
629
  def get_all_capabilities() -> List[ModelCapability]:
630
+ """Every entry the registry knows — recommended *and* recognised-only.
631
+
632
+ This is the verification surface: ``scripts/verify_hf_model_registry.py``
633
+ re-measures all of it, because a legacy entry that quietly disappeared from
634
+ the Hub should be reported rather than kept as a comforting fiction.
635
+ """
636
+ return [*_REGISTRY, *_LEGACY_REGISTRY]
637
+
638
+
639
+ def get_recommended_capabilities() -> List[ModelCapability]:
640
+ """Current-generation entries: catalog, download and recommendation input."""
489
641
  return list(_REGISTRY)
490
642
 
491
643
 
644
+ def get_legacy_capabilities() -> List[ModelCapability]:
645
+ """Recognised-only entries: never offered, never recommended, still named."""
646
+ return list(_LEGACY_REGISTRY)
647
+
648
+
492
649
  def get_capability(model_id: str) -> Optional[ModelCapability]:
493
- for m in _REGISTRY:
650
+ """Look an id up across both lists — recognition covers legacy weights too."""
651
+ for m in get_all_capabilities():
494
652
  if m.id == model_id or m.hf_repo_id == model_id:
495
653
  return m
496
654
  return None
497
655
 
498
656
 
657
+ def is_recognized_model(model_id: str) -> bool:
658
+ """True when the registry can name this model, whatever its lifecycle.
659
+
660
+ Used by the load path: a model already on disk must keep working even after
661
+ its generation stops being offered.
662
+ """
663
+ return get_capability(model_id) is not None
664
+
665
+
666
+ def is_recommended_model(model_id: str) -> bool:
667
+ """True only for current-generation entries — the download/offer gate."""
668
+ cap = get_capability(model_id)
669
+ return cap is not None and cap.lifecycle == RECOMMENDED
670
+
671
+
499
672
  def build_engine_model_catalog() -> Dict[str, List[Dict[str, Any]]]:
500
- """Return legacy ENGINE_MODEL_CATALOG shape, enriched with 5.2 fields."""
673
+ """Return legacy ENGINE_MODEL_CATALOG shape, enriched with rich fields.
674
+
675
+ Built from the **recommended** list only: the catalog is what the product
676
+ offers, and offering a superseded generation is how users end up downloading
677
+ something we would not stand behind.
678
+ """
501
679
  from collections import defaultdict
502
680
  by_engine: Dict[str, List[Dict[str, Any]]] = defaultdict(list)
503
681
 
@@ -513,13 +691,14 @@ def build_engine_model_catalog() -> Dict[str, List[Dict[str, Any]]]:
513
691
  for eng_key, hints in engine_map.items():
514
692
  if any(h in cap.provider_hints for h in hints) or eng_key in cap.provider_hints:
515
693
  legacy = cap.to_legacy_dict()
516
- # Adapt id for non-mlx engines (match historical patterns)
694
+ # Adapt id for non-mlx engines. These are provisional prefixed
695
+ # ids; `model_catalog._normalize_engine_entry` then replaces them
696
+ # with the verified per-engine repo from MODEL_ENGINE_ALIASES.
697
+ # (Until 11.2.0 the ollama branch *fabricated* a repo path from
698
+ # the family name — `ggml-org/<family>-12B-it-GGUF` — which
699
+ # resolved to a 404 for anything but Gemma 4 12B.)
517
700
  if eng_key == "ollama" and not legacy["id"].startswith("ollama:"):
518
- # historical used prefixed or hf.co for some
519
- if "gguf" in cap.tag.lower() or "gguf" in (cap.quantization or "").lower():
520
- legacy["id"] = f"ollama:hf.co/ggml-org/{cap.family.lower().replace(' ', '')}-12B-it-GGUF:Q4_K_M" # fallback, overridden by aliases
521
- else:
522
- legacy["id"] = f"ollama:{cap.hf_repo_id.split('/')[-1].lower()}"
701
+ legacy["id"] = f"ollama:{cap.hf_repo_id.split('/')[-1].lower()}"
523
702
  elif eng_key == "vllm" and not legacy["id"].startswith("vllm:"):
524
703
  legacy["id"] = f"vllm:{cap.hf_repo_id}"
525
704
  elif eng_key == "lmstudio" and not legacy["id"].startswith("lmstudio:"):
@@ -528,23 +707,17 @@ def build_engine_model_catalog() -> Dict[str, List[Dict[str, Any]]]:
528
707
  legacy["id"] = f"llamacpp:{cap.hf_repo_id}-GGUF"
529
708
  by_engine[eng_key].append(legacy)
530
709
 
531
- # Ensure at least the primary local_mlx ones are present (exact historical)
532
- # If projection missed any, inject the original local_mlx entries enriched
533
- if not by_engine.get("local_mlx"):
534
- for cap in _REGISTRY:
535
- if "local_mlx" in cap.provider_hints:
536
- by_engine["local_mlx"].append(cap.to_legacy_dict())
537
-
538
- return {k: v for k, v in by_engine.items()}
710
+ return dict(by_engine)
539
711
 
540
712
 
541
713
  def get_verified_models() -> List[Dict[str, Any]]:
542
- """Return only load-verified HF entries with rich fields (for API/UI)."""
714
+ """Return only load-verified recommended entries with rich fields (API/UI)."""
543
715
  return [
544
716
  c.to_legacy_dict() for c in _REGISTRY
545
717
  if c.verification.hf_exists and c.verification.has_config and c.verification.has_tokenizer
546
718
  ]
547
719
 
548
720
 
549
- # Back-compat: expose a simple list mirroring the old top-level for mlx
721
+ # Back-compat: expose a simple list mirroring the old top-level for mlx.
722
+ # Recommended entries only — same reasoning as build_engine_model_catalog.
550
723
  LOCAL_MLX_MODELS = [c.to_legacy_dict() for c in _REGISTRY if "local_mlx" in c.provider_hints]