ltcai 11.1.0 → 11.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +53 -53
- package/docs/CHANGELOG.md +33 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/FEATURE_AUDIT_v11.2.0.md +393 -0
- package/docs/LAYOUT_REBUILD_SPEC.md +9 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/architecture.md +6 -2
- package/docs/kg-schema.md +1 -1
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/gates.py +125 -0
- package/lattice_brain/graph/fusion.py +35 -4
- package/lattice_brain/graph/projection.py +66 -8
- package/lattice_brain/graph/schema.py +9 -0
- package/lattice_brain/graph/store.py +9 -0
- package/lattice_brain/graph/vector_index/selector.py +32 -2
- package/lattice_brain/ingestion.py +175 -27
- package/lattice_brain/multimodal.py +525 -5
- package/lattice_brain/portability.py +169 -32
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/lattice_brain/sealed_box.py +244 -0
- package/lattice_brain/synthesis.py +24 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/brain_intelligence.py +4 -0
- package/latticeai/api/chat.py +11 -0
- package/latticeai/api/chat_helpers.py +16 -3
- package/latticeai/api/chat_hybrid.py +32 -1
- package/latticeai/api/features.py +70 -0
- package/latticeai/api/local_files.py +102 -0
- package/latticeai/api/portability.py +39 -4
- package/latticeai/api/review_queue.py +126 -0
- package/latticeai/api/search.py +16 -2
- package/latticeai/core/agent.py +55 -2
- package/latticeai/core/config.py +4 -1
- package/latticeai/core/context_builder.py +6 -3
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +143 -0
- package/latticeai/core/model_compat.py +73 -2
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/models/model_providers.py +12 -4
- package/latticeai/runtime/build_phases.py +28 -0
- package/latticeai/runtime/chat_wiring.py +4 -0
- package/latticeai/runtime/feature_toggle_wiring.py +163 -0
- package/latticeai/runtime/router_registration.py +11 -0
- package/latticeai/services/app_context.py +8 -0
- package/latticeai/services/architecture_readiness.py +1 -1
- package/latticeai/services/automation_intelligence.py +22 -2
- package/latticeai/services/brain_intelligence.py +123 -7
- package/latticeai/services/command_center.py +10 -4
- package/latticeai/services/feature_toggles.py +502 -0
- package/latticeai/services/folder_watch.py +122 -1
- package/latticeai/services/hybrid_chat.py +56 -5
- package/latticeai/services/interop_bridges.py +978 -0
- package/latticeai/services/model_capability_registry.py +434 -261
- package/latticeai/services/model_catalog.py +95 -61
- package/latticeai/services/model_recommendation.py +18 -11
- package/latticeai/services/model_runtime.py +1 -1
- package/latticeai/services/multimodal_ports.py +26 -1
- package/latticeai/services/obsidian_bridge.py +16 -25
- package/latticeai/services/product_readiness.py +1 -1
- package/latticeai/services/search_service.py +149 -2
- package/latticeai/services/tool_dispatch.py +4 -0
- package/latticeai/setup/auto_setup.py +27 -30
- package/latticeai/setup/wizard.py +77 -44
- package/package.json +1 -1
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_server_i18n.mjs +1 -0
- package/scripts/release_screen_claims.json +13 -0
- package/scripts/verify_hf_model_registry.py +253 -218
- package/src-tauri/Cargo.lock +1 -1
- package/src-tauri/Cargo.toml +1 -1
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +37 -37
- package/static/app/assets/{Act-D0HWqtn0.js → Act-AWf0SAKp.js} +1 -1
- package/static/app/assets/{AdminConsole-D-QDW-A4.js → AdminConsole-D0u8Tiyj.js} +1 -1
- package/static/app/assets/{Brain-CzCsI1mi.js → Brain-tuhI4sOC.js} +1 -1
- package/static/app/assets/BrainHome-Ts7G_Ila.js +2 -0
- package/static/app/assets/{BrainSignals-2dHQNkns.js → BrainSignals-jMYgQ2Ar.js} +1 -1
- package/static/app/assets/{Capture-CT8v1StE.js → Capture-CqOSzyPr.js} +1 -1
- package/static/app/assets/{CommandPalette-DoLXC2KH.js → CommandPalette-DC0Bzh-I.js} +1 -1
- package/static/app/assets/{Library-DDoxFE5c.js → Library-CX-bbhmK.js} +1 -1
- package/static/app/assets/{LivingBrain-BXMWIK_2.js → LivingBrain-DBwhto14.js} +1 -1
- package/static/app/assets/{ProductFlow-DOYf7JIs.js → ProductFlow-BHA2cfKI.js} +1 -1
- package/static/app/assets/{ReviewCard-COQsqidK.js → ReviewCard-BUhCKRNM.js} +1 -1
- package/static/app/assets/{System-BRllvYXd.js → System-Bu2t5hn1.js} +1 -1
- package/static/app/assets/arrow-left-Dzwa5zRb.js +1 -0
- package/static/app/assets/{bot-4BvN07ux.js → bot-Cia42c2h.js} +1 -1
- package/static/app/assets/brain-DJMoqrwx.js +1 -0
- package/static/app/assets/{button-CDjtnAoU.js → button-2j2Ijzgq.js} +1 -1
- package/static/app/assets/{circle-pause-D_RMn7tp.js → circle-pause-BEFeWpVW.js} +1 -1
- package/static/app/assets/{circle-play-B5OpB8ae.js → circle-play-ujXMcHxl.js} +1 -1
- package/static/app/assets/{cpu-BIlWInHf.js → cpu-k4awryFq.js} +1 -1
- package/static/app/assets/{download-BtjXfL3z.js → download-DFbLJ_ig.js} +1 -1
- package/static/app/assets/{folder-open-DefMpxI2.js → folder-open-7y_b6xkM.js} +1 -1
- package/static/app/assets/{hard-drive-BQ8NZVkw.js → hard-drive-Bidh02Kr.js} +1 -1
- package/static/app/assets/{index-0AvoEBzJ.js → index-BpYkzcVm.js} +3 -3
- package/static/app/assets/{index-vtEfYvQY.css → index-DwDl9-8Y.css} +1 -1
- package/static/app/assets/{input-B_5ZJ9oy.js → input-DSlJJxRs.js} +1 -1
- package/static/app/assets/{permissionCopy-BqZ5tsgL.js → permissionCopy-Bpb83Hx9.js} +1 -1
- package/static/app/assets/{primitives-CVwew78r.js → primitives-BCx6TvfG.js} +1 -1
- package/static/app/assets/search-Cgy8cCFJ.js +1 -0
- package/static/app/assets/{share-2-D5zg_0fY.js → share-2-BH1M-WNi.js} +1 -1
- package/static/app/assets/{shield-alert-B5pZzkUb.js → shield-alert-BlKdBXcG.js} +1 -1
- package/static/app/assets/{textarea-nEVIweKY.js → textarea-CCWbUfFB.js} +1 -1
- package/static/app/assets/{useFocusTrap-Cm99AHlz.js → useFocusTrap-YdHQ7pJ1.js} +1 -1
- package/static/app/assets/{useQuery-Dm__N6bL.js → useQuery-CXQiwbVT.js} +1 -1
- package/static/app/assets/{utils-DcDMoZIe.js → utils-zqPZJxdx.js} +2 -2
- package/static/app/assets/{workspace-LtRRSKTf.js → workspace-DXTihhfU.js} +1 -1
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/static/app/assets/BrainHome-Btns-_TA.js +0 -2
- package/static/app/assets/arrow-left-DnyMzss-.js +0 -1
- package/static/app/assets/brain-uMb_5hnO.js +0 -1
- package/static/app/assets/search-DkhnOKZt.js +0 -1
|
@@ -1,8 +1,8 @@
|
|
|
1
|
-
"""Structured Model Capability Registry for Lattice AI
|
|
1
|
+
"""Structured Model Capability Registry for Lattice AI.
|
|
2
2
|
|
|
3
3
|
User-focused, transparent model catalog with:
|
|
4
|
-
- HF repo provenance
|
|
5
|
-
- Modality / vision support
|
|
4
|
+
- HF repo provenance (exact-case repo id, pinned against the HF API)
|
|
5
|
+
- Modality / vision support and the HF config architecture (``model_type``)
|
|
6
6
|
- Quantization, size, download/load strategies
|
|
7
7
|
- Hardware notes (RAM estimates, Apple Silicon affinity)
|
|
8
8
|
- License / safety notes
|
|
@@ -12,10 +12,30 @@ This replaces the flat ENGINE_MODEL_CATALOG construction with a richer,
|
|
|
12
12
|
queryable source of truth while preserving exact legacy shapes for
|
|
13
13
|
model_catalog / recommendation / API / frontend consumers.
|
|
14
14
|
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
15
|
+
11.2.0 — two lists, one registry
|
|
16
|
+
================================
|
|
17
|
+
A model can be *offered* or merely *understood*, and conflating the two either
|
|
18
|
+
pushes dead downloads at people or orphans the weights already on their disk.
|
|
19
|
+
Every entry therefore carries a :attr:`ModelCapability.lifecycle`:
|
|
20
|
+
|
|
21
|
+
``RECOMMENDED``
|
|
22
|
+
Current generation. Appears in ENGINE_MODEL_CATALOG, the recommendation
|
|
23
|
+
tiers, and the download paths.
|
|
24
|
+
``LEGACY``
|
|
25
|
+
Superseded, but real and still loadable. **Never** offered for download and
|
|
26
|
+
never recommended — it exists so a model a user already downloaded keeps its
|
|
27
|
+
name, size, family and runtime profile instead of showing up as an unknown
|
|
28
|
+
blob. See :func:`get_legacy_capabilities` / :func:`is_recognized_model`.
|
|
29
|
+
|
|
30
|
+
Models that no longer exist on the Hub, or that cannot be fetched without
|
|
31
|
+
credentials (gated), are **deleted outright** rather than demoted: pretending to
|
|
32
|
+
recognise something nobody can obtain is not compatibility, it is noise.
|
|
33
|
+
|
|
34
|
+
Verification is honest: hf_exists + metadata/siblings presence measured through
|
|
35
|
+
the public HF REST API; full weights are never auto-fetched by the verifier and
|
|
36
|
+
no model is ever loaded to produce these flags. Large models (>12GB) explicitly
|
|
37
|
+
note "local load practical only on high-RAM Apple Silicon or CUDA; expect long
|
|
38
|
+
download".
|
|
19
39
|
"""
|
|
20
40
|
|
|
21
41
|
from __future__ import annotations
|
|
@@ -47,6 +67,13 @@ class VerificationStatus:
|
|
|
47
67
|
verified_by: str = "hf-api-light" # or "local-load-test"
|
|
48
68
|
|
|
49
69
|
|
|
70
|
+
# ── Lifecycle vocabulary ──────────────────────────────────────────────────────
|
|
71
|
+
#: Current generation: offered for download, listed in the catalog, recommended.
|
|
72
|
+
RECOMMENDED = "recommended"
|
|
73
|
+
#: Superseded but still loadable: recognised only, never offered or recommended.
|
|
74
|
+
LEGACY = "legacy"
|
|
75
|
+
|
|
76
|
+
|
|
50
77
|
@dataclass(frozen=True)
|
|
51
78
|
class ModelCapability:
|
|
52
79
|
"""Rich capability entry. id is the canonical key used in ENGINE_MODEL_CATALOG."""
|
|
@@ -57,6 +84,15 @@ class ModelCapability:
|
|
|
57
84
|
tag: str
|
|
58
85
|
size: str # display string "7.6GB", "pull required"
|
|
59
86
|
modality: str = "multimodal" # multimodal | vision | text | audio etc.
|
|
87
|
+
#: HF ``config.json`` ``model_type`` — the string mlx-lm / mlx-vlm dispatches
|
|
88
|
+
#: on. Pinned from the Hub API so the verifier can judge loadability
|
|
89
|
+
#: statically, without downloading or importing anything.
|
|
90
|
+
architecture: str = ""
|
|
91
|
+
#: Measured sum of the repo's sibling file sizes, in GB (10^9 bytes).
|
|
92
|
+
#: ``None`` for tool-managed engines that pull on demand.
|
|
93
|
+
download_size_gb: Optional[float] = None
|
|
94
|
+
#: RECOMMENDED (offered + recommended) or LEGACY (recognised only).
|
|
95
|
+
lifecycle: str = RECOMMENDED
|
|
60
96
|
quantization: Optional[str] = None # "4bit", "Q4_K_M", "GGUF-Q4"
|
|
61
97
|
provider_hints: List[str] = field(default_factory=lambda: ["local_mlx"]) # which engines this id primarily maps to
|
|
62
98
|
download_strategy: str = "hf_hub" # hf_hub | ollama_pull | lmstudio_app | gguf_manual
|
|
@@ -85,6 +121,8 @@ class ModelCapability:
|
|
|
85
121
|
"size": self.size,
|
|
86
122
|
"pullable": True,
|
|
87
123
|
"modality": self.modality,
|
|
124
|
+
"architecture": self.architecture,
|
|
125
|
+
"lifecycle": self.lifecycle,
|
|
88
126
|
"source_country": self.source_country,
|
|
89
127
|
"source_company": self.source_company,
|
|
90
128
|
"execution_method": self.execution_method,
|
|
@@ -127,30 +165,48 @@ class ModelCapability:
|
|
|
127
165
|
return base
|
|
128
166
|
|
|
129
167
|
|
|
130
|
-
# ── Curated
|
|
131
|
-
#
|
|
132
|
-
#
|
|
133
|
-
#
|
|
134
|
-
#
|
|
168
|
+
# ── Curated registry ──────────────────────────────────────────────────────────
|
|
169
|
+
# Every field below was measured against the public HF REST API on 2026-08-10:
|
|
170
|
+
# repo existence, gated flag, library_name/tags, config ``model_type``, sibling
|
|
171
|
+
# file list and the exact byte sum of those siblings (``size`` and
|
|
172
|
+
# ``download_size_gb``). Nothing here was inferred from a model card, and no
|
|
173
|
+
# weights were downloaded to produce it — re-run
|
|
174
|
+
# ``scripts/verify_hf_model_registry.py`` to re-measure.
|
|
175
|
+
#
|
|
176
|
+
# Repo ids are stored in the Hub's own canonical casing (the API echoes the
|
|
177
|
+
# canonical id even when queried with different case, e.g. requesting
|
|
178
|
+
# ``gemma-4-12b-it-4bit`` answers with ``gemma-4-12B-it-4bit``). Storing the
|
|
179
|
+
# canonical form keeps the download path, the on-disk cache directory and the
|
|
180
|
+
# catalog key identical.
|
|
135
181
|
|
|
136
182
|
_REGISTRY: List[ModelCapability] = [
|
|
137
|
-
#
|
|
183
|
+
# ── Ultralight tier (≤8GB RAM) ───────────────────────────────────────────
|
|
138
184
|
ModelCapability(
|
|
139
|
-
id="mlx-community/
|
|
140
|
-
hf_repo_id="mlx-community/
|
|
141
|
-
name="
|
|
142
|
-
family="
|
|
143
|
-
tag="local-
|
|
144
|
-
size="
|
|
185
|
+
id="mlx-community/LFM2.5-2.6B-4bit",
|
|
186
|
+
hf_repo_id="mlx-community/LFM2.5-2.6B-4bit",
|
|
187
|
+
name="LFM2.5 2.6B",
|
|
188
|
+
family="LFM2.5",
|
|
189
|
+
tag="local-llm",
|
|
190
|
+
size="1.5GB",
|
|
191
|
+
modality="text",
|
|
192
|
+
architecture="lfm2",
|
|
193
|
+
download_size_gb=1.54,
|
|
145
194
|
quantization="4bit",
|
|
146
|
-
provider_hints=["local_mlx"],
|
|
195
|
+
provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
|
|
147
196
|
download_strategy="hf_hub",
|
|
148
|
-
load_strategy="
|
|
149
|
-
hardware=HardwareProfile(
|
|
150
|
-
|
|
151
|
-
|
|
197
|
+
load_strategy="mlx_lm",
|
|
198
|
+
hardware=HardwareProfile(
|
|
199
|
+
min_ram_gb=6.0, recommended_ram_gb=8.0, apple_silicon_pref=True,
|
|
200
|
+
notes="Smallest model that still answers in Korean. Text only — no image understanding.",
|
|
201
|
+
),
|
|
202
|
+
source_country="미국", source_company="Liquid AI",
|
|
203
|
+
verification=VerificationStatus(
|
|
204
|
+
hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
|
|
205
|
+
has_weights_hint=True, pipeline_tag="text-generation", likes=3, license="apache-2.0",
|
|
206
|
+
notes="library_name=mlx, ungated, 1 safetensors shard.", verified_by="hf-api-light",
|
|
207
|
+
),
|
|
152
208
|
recommended_default=True,
|
|
153
|
-
display_priority=
|
|
209
|
+
display_priority=5,
|
|
154
210
|
),
|
|
155
211
|
ModelCapability(
|
|
156
212
|
id="mlx-community/gemma-4-e2b-it-4bit",
|
|
@@ -159,79 +215,214 @@ _REGISTRY: List[ModelCapability] = [
|
|
|
159
215
|
family="Gemma 4",
|
|
160
216
|
tag="local-vlm",
|
|
161
217
|
size="3.6GB",
|
|
218
|
+
architecture="gemma4",
|
|
219
|
+
download_size_gb=3.58,
|
|
162
220
|
quantization="4bit",
|
|
163
|
-
provider_hints=["local_mlx"],
|
|
221
|
+
provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
|
|
164
222
|
download_strategy="hf_hub",
|
|
165
223
|
load_strategy="mlx_vlm",
|
|
166
|
-
hardware=HardwareProfile(
|
|
224
|
+
hardware=HardwareProfile(
|
|
225
|
+
min_ram_gb=8.0, recommended_ram_gb=8.0, apple_silicon_pref=True,
|
|
226
|
+
notes="Smallest model here that can still read pictures. Good first local VLM.",
|
|
227
|
+
),
|
|
167
228
|
source_country="미국", source_company="Google",
|
|
168
|
-
verification=VerificationStatus(
|
|
229
|
+
verification=VerificationStatus(
|
|
230
|
+
hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
|
|
231
|
+
has_weights_hint=True, pipeline_tag="image-text-to-text", likes=25, license="apache-2.0",
|
|
232
|
+
notes="library_name=mlx, ungated, 1 safetensors shard.", verified_by="hf-api-light",
|
|
233
|
+
),
|
|
169
234
|
recommended_default=True,
|
|
170
|
-
display_priority=
|
|
235
|
+
display_priority=10,
|
|
171
236
|
),
|
|
237
|
+
|
|
238
|
+
# ── Light tier (16GB RAM) ────────────────────────────────────────────────
|
|
172
239
|
ModelCapability(
|
|
173
|
-
id="mlx-community/gemma-4-e4b-4bit",
|
|
174
|
-
hf_repo_id="mlx-community/gemma-4-e4b-4bit",
|
|
175
|
-
name="Gemma 4 E4B
|
|
240
|
+
id="mlx-community/gemma-4-e4b-it-4bit",
|
|
241
|
+
hf_repo_id="mlx-community/gemma-4-e4b-it-4bit",
|
|
242
|
+
name="Gemma 4 E4B Instruct",
|
|
176
243
|
family="Gemma 4",
|
|
177
244
|
tag="local-vlm",
|
|
178
245
|
size="5.2GB",
|
|
246
|
+
architecture="gemma4",
|
|
247
|
+
download_size_gb=5.18,
|
|
179
248
|
quantization="4bit",
|
|
180
|
-
provider_hints=["local_mlx"],
|
|
249
|
+
provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
|
|
181
250
|
download_strategy="hf_hub",
|
|
182
251
|
load_strategy="mlx_vlm",
|
|
183
|
-
hardware=HardwareProfile(
|
|
252
|
+
hardware=HardwareProfile(
|
|
253
|
+
min_ram_gb=10.0, recommended_ram_gb=16.0, apple_silicon_pref=True,
|
|
254
|
+
notes="Step up from E2B with the same tiny footprint discipline.",
|
|
255
|
+
),
|
|
184
256
|
source_country="미국", source_company="Google",
|
|
185
|
-
verification=VerificationStatus(
|
|
257
|
+
verification=VerificationStatus(
|
|
258
|
+
hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
|
|
259
|
+
has_weights_hint=True, pipeline_tag="image-text-to-text", likes=35, license="apache-2.0",
|
|
260
|
+
notes="library_name=mlx, ungated, 1 safetensors shard.", verified_by="hf-api-light",
|
|
261
|
+
),
|
|
262
|
+
display_priority=15,
|
|
186
263
|
),
|
|
264
|
+
|
|
265
|
+
# ── Mid tier (24GB RAM) ──────────────────────────────────────────────────
|
|
187
266
|
ModelCapability(
|
|
188
|
-
id="mlx-community/gemma-4-
|
|
189
|
-
hf_repo_id="mlx-community/gemma-4-
|
|
190
|
-
name="Gemma 4
|
|
267
|
+
id="mlx-community/gemma-4-12B-it-4bit",
|
|
268
|
+
hf_repo_id="mlx-community/gemma-4-12B-it-4bit",
|
|
269
|
+
name="Gemma 4 12B Instruct",
|
|
191
270
|
family="Gemma 4",
|
|
192
271
|
tag="local-vlm",
|
|
193
|
-
size="
|
|
272
|
+
size="6.8GB",
|
|
273
|
+
architecture="gemma4_unified",
|
|
274
|
+
download_size_gb=6.77,
|
|
194
275
|
quantization="4bit",
|
|
195
|
-
provider_hints=["local_mlx"],
|
|
276
|
+
provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
|
|
196
277
|
download_strategy="hf_hub",
|
|
197
278
|
load_strategy="mlx_vlm",
|
|
198
|
-
hardware=HardwareProfile(
|
|
279
|
+
hardware=HardwareProfile(
|
|
280
|
+
min_ram_gb=12.0, recommended_ram_gb=16.0, apple_silicon_pref=True,
|
|
281
|
+
notes="Sweet spot for local multimodal on M-series 16GB+. Uses the gemma4_unified MLX loader.",
|
|
282
|
+
),
|
|
199
283
|
source_country="미국", source_company="Google",
|
|
200
|
-
verification=VerificationStatus(
|
|
284
|
+
verification=VerificationStatus(
|
|
285
|
+
hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
|
|
286
|
+
has_weights_hint=True, pipeline_tag="image-text-to-text", likes=16, license="apache-2.0",
|
|
287
|
+
notes="library_name=mlx, ungated, 2 safetensors shards. Canonical id capitalises the B.",
|
|
288
|
+
verified_by="hf-api-light",
|
|
289
|
+
),
|
|
290
|
+
recommended_default=True,
|
|
291
|
+
display_priority=25,
|
|
201
292
|
),
|
|
202
293
|
ModelCapability(
|
|
203
|
-
id="mlx-community/
|
|
204
|
-
hf_repo_id="mlx-community/
|
|
205
|
-
name="
|
|
206
|
-
family="
|
|
294
|
+
id="mlx-community/Qwen3.5-9B-MLX-4bit",
|
|
295
|
+
hf_repo_id="mlx-community/Qwen3.5-9B-MLX-4bit",
|
|
296
|
+
name="Qwen3.5 9B",
|
|
297
|
+
family="Qwen3.5",
|
|
207
298
|
tag="local-vlm",
|
|
208
|
-
size="
|
|
299
|
+
size="6.0GB",
|
|
300
|
+
architecture="qwen3_5",
|
|
301
|
+
download_size_gb=5.98,
|
|
209
302
|
quantization="4bit",
|
|
210
|
-
provider_hints=["local_mlx", "
|
|
303
|
+
provider_hints=["local_mlx", "vllm"],
|
|
211
304
|
download_strategy="hf_hub",
|
|
212
305
|
load_strategy="mlx_vlm",
|
|
213
|
-
hardware=HardwareProfile(
|
|
214
|
-
|
|
215
|
-
|
|
306
|
+
hardware=HardwareProfile(
|
|
307
|
+
min_ram_gb=12.0, recommended_ram_gb=16.0, apple_silicon_pref=True,
|
|
308
|
+
notes="Mid-size vision-language model; replaces the retired Qwen2.5-VL / Llama-3.2-Vision slot.",
|
|
309
|
+
),
|
|
310
|
+
source_country="중국", source_company="Alibaba",
|
|
311
|
+
verification=VerificationStatus(
|
|
312
|
+
hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
|
|
313
|
+
has_weights_hint=True, pipeline_tag="image-text-to-text", likes=153, license="apache-2.0",
|
|
314
|
+
notes="library_name=mlx, ungated, 2 safetensors shards.", verified_by="hf-api-light",
|
|
315
|
+
),
|
|
216
316
|
recommended_default=True,
|
|
217
317
|
display_priority=20,
|
|
218
318
|
),
|
|
319
|
+
|
|
320
|
+
# ── General-purpose ──────────────────────────────────────────────────────
|
|
321
|
+
ModelCapability(
|
|
322
|
+
id="mlx-community/gpt-oss-20b-MXFP4-Q8",
|
|
323
|
+
hf_repo_id="mlx-community/gpt-oss-20b-MXFP4-Q8",
|
|
324
|
+
name="GPT-OSS 20B",
|
|
325
|
+
family="GPT-OSS",
|
|
326
|
+
tag="local-llm",
|
|
327
|
+
size="12.1GB",
|
|
328
|
+
modality="text",
|
|
329
|
+
architecture="gpt_oss",
|
|
330
|
+
download_size_gb=12.10,
|
|
331
|
+
quantization="MXFP4-Q8",
|
|
332
|
+
provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
|
|
333
|
+
download_strategy="hf_hub",
|
|
334
|
+
load_strategy="mlx_lm",
|
|
335
|
+
hardware=HardwareProfile(
|
|
336
|
+
min_ram_gb=18.0, recommended_ram_gb=24.0, apple_silicon_pref=True,
|
|
337
|
+
notes="Most-downloaded entry in this catalog. Text only — no image understanding.",
|
|
338
|
+
),
|
|
339
|
+
source_country="미국", source_company="OpenAI",
|
|
340
|
+
verification=VerificationStatus(
|
|
341
|
+
hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
|
|
342
|
+
has_weights_hint=True, pipeline_tag="text-generation", likes=84, license="apache-2.0",
|
|
343
|
+
notes="library_name=mlx, ungated, 3 safetensors shards.", verified_by="hf-api-light",
|
|
344
|
+
),
|
|
345
|
+
display_priority=30,
|
|
346
|
+
),
|
|
347
|
+
|
|
348
|
+
# ── MoE tier (32GB+ RAM) ─────────────────────────────────────────────────
|
|
219
349
|
ModelCapability(
|
|
220
350
|
id="mlx-community/gemma-4-26b-a4b-it-4bit",
|
|
221
351
|
hf_repo_id="mlx-community/gemma-4-26b-a4b-it-4bit",
|
|
222
352
|
name="Gemma 4 26B A4B Instruct",
|
|
223
353
|
family="Gemma 4",
|
|
224
354
|
tag="local-vlm",
|
|
225
|
-
size="15.
|
|
355
|
+
size="15.4GB",
|
|
356
|
+
architecture="gemma4",
|
|
357
|
+
download_size_gb=15.37,
|
|
226
358
|
quantization="4bit",
|
|
227
|
-
provider_hints=["local_mlx"],
|
|
359
|
+
provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
|
|
228
360
|
download_strategy="hf_hub",
|
|
229
361
|
load_strategy="mlx_vlm",
|
|
230
|
-
hardware=HardwareProfile(
|
|
362
|
+
hardware=HardwareProfile(
|
|
363
|
+
min_ram_gb=22.0, recommended_ram_gb=32.0, apple_silicon_pref=True,
|
|
364
|
+
notes="Mixture-of-experts; practical on 32GB+ Apple Silicon. Long first download.",
|
|
365
|
+
),
|
|
231
366
|
source_country="미국", source_company="Google",
|
|
232
|
-
verification=VerificationStatus(
|
|
367
|
+
verification=VerificationStatus(
|
|
368
|
+
hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
|
|
369
|
+
has_weights_hint=True, pipeline_tag="image-text-to-text", likes=80, license="apache-2.0",
|
|
370
|
+
notes="library_name=mlx, ungated, 3 safetensors shards.", verified_by="hf-api-light",
|
|
371
|
+
),
|
|
233
372
|
display_priority=50,
|
|
234
373
|
),
|
|
374
|
+
ModelCapability(
|
|
375
|
+
id="mlx-community/Qwen3.6-35B-A3B-4bit",
|
|
376
|
+
hf_repo_id="mlx-community/Qwen3.6-35B-A3B-4bit",
|
|
377
|
+
name="Qwen3.6 35B A3B",
|
|
378
|
+
family="Qwen3.6",
|
|
379
|
+
tag="local-vlm",
|
|
380
|
+
size="20.4GB",
|
|
381
|
+
architecture="qwen3_5_moe",
|
|
382
|
+
download_size_gb=20.43,
|
|
383
|
+
quantization="4bit",
|
|
384
|
+
provider_hints=["local_mlx", "vllm"],
|
|
385
|
+
download_strategy="hf_hub",
|
|
386
|
+
load_strategy="mlx_vlm",
|
|
387
|
+
hardware=HardwareProfile(
|
|
388
|
+
min_ram_gb=28.0, recommended_ram_gb=48.0, apple_silicon_pref=True,
|
|
389
|
+
notes="Mixture-of-experts; replaces the retired Llama 4 Scout slot. Long first download.",
|
|
390
|
+
),
|
|
391
|
+
source_country="중국", source_company="Alibaba",
|
|
392
|
+
verification=VerificationStatus(
|
|
393
|
+
hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
|
|
394
|
+
has_weights_hint=True, pipeline_tag="image-text-to-text", likes=94, license="apache-2.0",
|
|
395
|
+
notes="library_name=mlx, ungated, 4 safetensors shards.", verified_by="hf-api-light",
|
|
396
|
+
),
|
|
397
|
+
display_priority=65,
|
|
398
|
+
),
|
|
399
|
+
|
|
400
|
+
# ── Large tier (48GB+ RAM) ───────────────────────────────────────────────
|
|
401
|
+
ModelCapability(
|
|
402
|
+
id="mlx-community/Qwen3.6-27B-4bit",
|
|
403
|
+
hf_repo_id="mlx-community/Qwen3.6-27B-4bit",
|
|
404
|
+
name="Qwen3.6 27B",
|
|
405
|
+
family="Qwen3.6",
|
|
406
|
+
tag="local-vlm",
|
|
407
|
+
size="16.1GB",
|
|
408
|
+
architecture="qwen3_5",
|
|
409
|
+
download_size_gb=16.08,
|
|
410
|
+
quantization="4bit",
|
|
411
|
+
provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
|
|
412
|
+
download_strategy="hf_hub",
|
|
413
|
+
load_strategy="mlx_vlm",
|
|
414
|
+
hardware=HardwareProfile(
|
|
415
|
+
min_ram_gb=24.0, recommended_ram_gb=48.0, apple_silicon_pref=True,
|
|
416
|
+
notes="Dense top tier — every parameter runs on every token, so it is slower but steadier than the MoE.",
|
|
417
|
+
),
|
|
418
|
+
source_country="중국", source_company="Alibaba",
|
|
419
|
+
verification=VerificationStatus(
|
|
420
|
+
hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
|
|
421
|
+
has_weights_hint=True, pipeline_tag="image-text-to-text", likes=50, license="apache-2.0",
|
|
422
|
+
notes="library_name=mlx, ungated, 3 safetensors shards.", verified_by="hf-api-light",
|
|
423
|
+
),
|
|
424
|
+
display_priority=55,
|
|
425
|
+
),
|
|
235
426
|
ModelCapability(
|
|
236
427
|
id="mlx-community/gemma-4-31b-it-4bit",
|
|
237
428
|
hf_repo_id="mlx-community/gemma-4-31b-it-4bit",
|
|
@@ -239,32 +430,93 @@ _REGISTRY: List[ModelCapability] = [
|
|
|
239
430
|
family="Gemma 4",
|
|
240
431
|
tag="local-vlm",
|
|
241
432
|
size="18.4GB",
|
|
433
|
+
architecture="gemma4",
|
|
434
|
+
download_size_gb=18.44,
|
|
242
435
|
quantization="4bit",
|
|
243
|
-
provider_hints=["local_mlx", "ollama", "vllm"],
|
|
436
|
+
provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
|
|
244
437
|
download_strategy="hf_hub",
|
|
245
438
|
load_strategy="mlx_vlm",
|
|
246
|
-
hardware=HardwareProfile(
|
|
439
|
+
hardware=HardwareProfile(
|
|
440
|
+
min_ram_gb=26.0, recommended_ram_gb=48.0, apple_silicon_pref=True,
|
|
441
|
+
notes="Largest Gemma 4 here; high-end local only. Consider a cloud path on lower RAM.",
|
|
442
|
+
),
|
|
247
443
|
source_country="미국", source_company="Google",
|
|
248
|
-
verification=VerificationStatus(
|
|
444
|
+
verification=VerificationStatus(
|
|
445
|
+
hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
|
|
446
|
+
has_weights_hint=True, pipeline_tag="image-text-to-text", likes=46, license="apache-2.0",
|
|
447
|
+
notes="library_name=mlx, ungated, 4 safetensors shards.", verified_by="hf-api-light",
|
|
448
|
+
),
|
|
449
|
+
display_priority=60,
|
|
249
450
|
),
|
|
451
|
+
]
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
# ── Recognised-only entries (LEGACY) ──────────────────────────────────────────
|
|
455
|
+
# Superseded generations that are still on the Hub and still loadable. They are
|
|
456
|
+
# deliberately absent from ENGINE_MODEL_CATALOG, from the recommendation tiers
|
|
457
|
+
# and from every download path — their only job is to keep a model somebody
|
|
458
|
+
# already downloaded identifiable (name, family, size, runtime profile) instead
|
|
459
|
+
# of surfacing as an unknown blob. Verified on 2026-08-10 like the rest.
|
|
250
460
|
|
|
251
|
-
|
|
461
|
+
_LEGACY_REGISTRY: List[ModelCapability] = [
|
|
462
|
+
ModelCapability(
|
|
463
|
+
id="mlx-community/gemma-4-e2b-4bit",
|
|
464
|
+
hf_repo_id="mlx-community/gemma-4-e2b-4bit",
|
|
465
|
+
name="Gemma 4 E2B Base",
|
|
466
|
+
family="Gemma 4",
|
|
467
|
+
tag="local-vlm",
|
|
468
|
+
size="3.6GB",
|
|
469
|
+
architecture="gemma4",
|
|
470
|
+
download_size_gb=3.61,
|
|
471
|
+
lifecycle=LEGACY,
|
|
472
|
+
quantization="4bit",
|
|
473
|
+
hardware=HardwareProfile(min_ram_gb=8.0, recommended_ram_gb=8.0, apple_silicon_pref=True,
|
|
474
|
+
notes="Base (not instruction-tuned) — the Instruct build answers chat far better."),
|
|
475
|
+
source_country="미국", source_company="Google",
|
|
476
|
+
verification=VerificationStatus(
|
|
477
|
+
hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
|
|
478
|
+
has_weights_hint=True, pipeline_tag="any-to-any", likes=3, license="apache-2.0",
|
|
479
|
+
notes="Superseded by gemma-4-e2b-it-4bit for chat.", verified_by="hf-api-light",
|
|
480
|
+
),
|
|
481
|
+
),
|
|
482
|
+
ModelCapability(
|
|
483
|
+
id="mlx-community/gemma-4-e4b-4bit",
|
|
484
|
+
hf_repo_id="mlx-community/gemma-4-e4b-4bit",
|
|
485
|
+
name="Gemma 4 E4B Base",
|
|
486
|
+
family="Gemma 4",
|
|
487
|
+
tag="local-vlm",
|
|
488
|
+
size="5.3GB",
|
|
489
|
+
architecture="gemma4",
|
|
490
|
+
download_size_gb=5.25,
|
|
491
|
+
lifecycle=LEGACY,
|
|
492
|
+
quantization="4bit",
|
|
493
|
+
hardware=HardwareProfile(min_ram_gb=10.0, recommended_ram_gb=16.0, apple_silicon_pref=True,
|
|
494
|
+
notes="Base (not instruction-tuned) — the Instruct build answers chat far better."),
|
|
495
|
+
source_country="미국", source_company="Google",
|
|
496
|
+
verification=VerificationStatus(
|
|
497
|
+
hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
|
|
498
|
+
has_weights_hint=True, pipeline_tag="any-to-any", likes=6, license="apache-2.0",
|
|
499
|
+
notes="Superseded by gemma-4-e4b-it-4bit for chat.", verified_by="hf-api-light",
|
|
500
|
+
),
|
|
501
|
+
),
|
|
252
502
|
ModelCapability(
|
|
253
503
|
id="mlx-community/Qwen3-VL-4B-Instruct-4bit",
|
|
254
504
|
hf_repo_id="mlx-community/Qwen3-VL-4B-Instruct-4bit",
|
|
255
505
|
name="Qwen3-VL 4B",
|
|
256
506
|
family="Qwen3-VL",
|
|
257
507
|
tag="local-vlm",
|
|
258
|
-
size="
|
|
508
|
+
size="3.1GB",
|
|
509
|
+
architecture="qwen3_vl",
|
|
510
|
+
download_size_gb=3.11,
|
|
511
|
+
lifecycle=LEGACY,
|
|
259
512
|
quantization="4bit",
|
|
260
|
-
|
|
261
|
-
download_strategy="hf_hub",
|
|
262
|
-
load_strategy="mlx_vlm",
|
|
263
|
-
hardware=HardwareProfile(min_ram_gb=5.0, recommended_ram_gb=8.0, apple_silicon_pref=True, notes="Extremely compact strong VLM. Best default for low-RAM Macs."),
|
|
513
|
+
hardware=HardwareProfile(min_ram_gb=5.0, recommended_ram_gb=8.0, apple_silicon_pref=True),
|
|
264
514
|
source_country="중국", source_company="Alibaba",
|
|
265
|
-
verification=VerificationStatus(
|
|
266
|
-
|
|
267
|
-
|
|
515
|
+
verification=VerificationStatus(
|
|
516
|
+
hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
|
|
517
|
+
has_weights_hint=True, pipeline_tag="image-text-to-text", likes=7, license="apache-2.0",
|
|
518
|
+
notes="2025-10 generation; superseded by Qwen3.5 / Qwen3.6.", verified_by="hf-api-light",
|
|
519
|
+
),
|
|
268
520
|
),
|
|
269
521
|
ModelCapability(
|
|
270
522
|
id="mlx-community/Qwen3-VL-8B-Instruct-4bit",
|
|
@@ -272,16 +524,18 @@ _REGISTRY: List[ModelCapability] = [
|
|
|
272
524
|
name="Qwen3-VL 8B",
|
|
273
525
|
family="Qwen3-VL",
|
|
274
526
|
tag="local-vlm",
|
|
275
|
-
size="
|
|
527
|
+
size="5.8GB",
|
|
528
|
+
architecture="qwen3_vl",
|
|
529
|
+
download_size_gb=5.78,
|
|
530
|
+
lifecycle=LEGACY,
|
|
276
531
|
quantization="4bit",
|
|
277
|
-
provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
|
|
278
|
-
download_strategy="hf_hub",
|
|
279
|
-
load_strategy="mlx_vlm",
|
|
280
532
|
hardware=HardwareProfile(min_ram_gb=8.0, recommended_ram_gb=12.0, apple_silicon_pref=True),
|
|
281
533
|
source_country="중국", source_company="Alibaba",
|
|
282
|
-
verification=VerificationStatus(
|
|
283
|
-
|
|
284
|
-
|
|
534
|
+
verification=VerificationStatus(
|
|
535
|
+
hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
|
|
536
|
+
has_weights_hint=True, pipeline_tag="image-text-to-text", likes=6, license="apache-2.0",
|
|
537
|
+
notes="2025-10 generation; superseded by Qwen3.5 9B.", verified_by="hf-api-light",
|
|
538
|
+
),
|
|
285
539
|
),
|
|
286
540
|
ModelCapability(
|
|
287
541
|
id="mlx-community/Qwen3-VL-30B-A3B-Instruct-4bit",
|
|
@@ -289,215 +543,139 @@ _REGISTRY: List[ModelCapability] = [
|
|
|
289
543
|
name="Qwen3-VL 30B A3B",
|
|
290
544
|
family="Qwen3-VL",
|
|
291
545
|
tag="local-vlm",
|
|
292
|
-
size="
|
|
546
|
+
size="18.3GB",
|
|
547
|
+
architecture="qwen3_vl_moe",
|
|
548
|
+
download_size_gb=18.27,
|
|
549
|
+
lifecycle=LEGACY,
|
|
293
550
|
quantization="4bit",
|
|
294
|
-
|
|
295
|
-
download_strategy="hf_hub",
|
|
296
|
-
load_strategy="mlx_vlm",
|
|
297
|
-
hardware=HardwareProfile(min_ram_gb=24.0, recommended_ram_gb=32.0, apple_silicon_pref=True, notes="Large MoE VLM; practical local only on 32GB+ Apple Silicon or strong CUDA. Download is multi-GB."),
|
|
551
|
+
hardware=HardwareProfile(min_ram_gb=24.0, recommended_ram_gb=32.0, apple_silicon_pref=True),
|
|
298
552
|
source_country="중국", source_company="Alibaba",
|
|
299
|
-
verification=VerificationStatus(
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
id="mlx-community/Llama-4-Scout-17B-16E-Instruct-4bit",
|
|
305
|
-
hf_repo_id="mlx-community/Llama-4-Scout-17B-16E-Instruct-4bit",
|
|
306
|
-
name="Llama 4 Scout 17B 16E",
|
|
307
|
-
family="Llama 4",
|
|
308
|
-
tag="local-vlm",
|
|
309
|
-
size="11.8GB",
|
|
310
|
-
quantization="4bit",
|
|
311
|
-
provider_hints=["local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"],
|
|
312
|
-
download_strategy="hf_hub",
|
|
313
|
-
load_strategy="mlx_vlm",
|
|
314
|
-
hardware=HardwareProfile(min_ram_gb=16.0, recommended_ram_gb=20.0, apple_silicon_pref=True),
|
|
315
|
-
source_country="미국", source_company="Meta",
|
|
316
|
-
verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="llama3.1-ish / meta-llama", verified_by="hf-api-light"),
|
|
317
|
-
recommended_default=True,
|
|
318
|
-
display_priority=25,
|
|
319
|
-
),
|
|
320
|
-
|
|
321
|
-
# ── Modern additions for 5.2.0 (verified on HF, user choice expansion) ──
|
|
322
|
-
# Gemma 3 (excellent real multimodal balance, smaller than 4 where present)
|
|
323
|
-
ModelCapability(
|
|
324
|
-
id="google/gemma-3-4b-it",
|
|
325
|
-
hf_repo_id="google/gemma-3-4b-it",
|
|
326
|
-
name="Gemma 3 4B Instruct (HF)",
|
|
327
|
-
family="Gemma 3",
|
|
328
|
-
tag="local-vlm",
|
|
329
|
-
size="~5GB+",
|
|
330
|
-
quantization="bf16 / 4bit variants",
|
|
331
|
-
provider_hints=["local_mlx", "vllm", "ollama"],
|
|
332
|
-
download_strategy="hf_hub",
|
|
333
|
-
load_strategy="mlx_vlm",
|
|
334
|
-
hardware=HardwareProfile(min_ram_gb=8.0, recommended_ram_gb=12.0, apple_silicon_pref=True, notes="Use mlx-community quantized ports when available for best local perf."),
|
|
335
|
-
source_country="미국", source_company="Google",
|
|
336
|
-
verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="gemma-terms", verified_by="hf-api-light"),
|
|
337
|
-
display_priority=30,
|
|
338
|
-
),
|
|
339
|
-
ModelCapability(
|
|
340
|
-
id="google/gemma-3-12b-it",
|
|
341
|
-
hf_repo_id="google/gemma-3-12b-it",
|
|
342
|
-
name="Gemma 3 12B Instruct (HF)",
|
|
343
|
-
family="Gemma 3",
|
|
344
|
-
tag="local-vlm",
|
|
345
|
-
size="~12GB+",
|
|
346
|
-
quantization="bf16 / GGUF-4bit",
|
|
347
|
-
provider_hints=["ollama", "vllm", "lmstudio", "llamacpp"],
|
|
348
|
-
download_strategy="hf_hub",
|
|
349
|
-
load_strategy="ollama",
|
|
350
|
-
hardware=HardwareProfile(min_ram_gb=16.0, recommended_ram_gb=20.0, notes="Prefer quantized GGUF for llama.cpp / ollama on non-Apple or lower RAM."),
|
|
351
|
-
source_country="미국", source_company="Google",
|
|
352
|
-
verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="gemma-terms", verified_by="hf-api-light"),
|
|
553
|
+
verification=VerificationStatus(
|
|
554
|
+
hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
|
|
555
|
+
has_weights_hint=True, pipeline_tag="image-text-to-text", likes=8, license="apache-2.0",
|
|
556
|
+
notes="2025-10 generation; superseded by Qwen3.6 35B A3B.", verified_by="hf-api-light",
|
|
557
|
+
),
|
|
353
558
|
),
|
|
354
|
-
|
|
355
|
-
# Qwen2.5-VL (battle-tested, widely supported)
|
|
356
559
|
ModelCapability(
|
|
357
|
-
id="
|
|
358
|
-
hf_repo_id="
|
|
359
|
-
name="Qwen2.5-VL 7B
|
|
560
|
+
id="mlx-community/Qwen2.5-VL-7B-Instruct-4bit",
|
|
561
|
+
hf_repo_id="mlx-community/Qwen2.5-VL-7B-Instruct-4bit",
|
|
562
|
+
name="Qwen2.5-VL 7B",
|
|
360
563
|
family="Qwen2.5-VL",
|
|
361
564
|
tag="local-vlm",
|
|
362
|
-
size="
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
hardware=HardwareProfile(min_ram_gb=
|
|
565
|
+
size="5.7GB",
|
|
566
|
+
architecture="qwen2_5_vl",
|
|
567
|
+
download_size_gb=5.65,
|
|
568
|
+
lifecycle=LEGACY,
|
|
569
|
+
quantization="4bit",
|
|
570
|
+
hardware=HardwareProfile(min_ram_gb=8.0, recommended_ram_gb=12.0, apple_silicon_pref=True),
|
|
368
571
|
source_country="중국", source_company="Alibaba",
|
|
369
|
-
verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="apache-2.0", verified_by="hf-api-light"),
|
|
370
|
-
display_priority=35,
|
|
371
|
-
),
|
|
372
|
-
|
|
373
|
-
# Llama 3.2 Vision (widely available, good ecosystem)
|
|
374
|
-
ModelCapability(
|
|
375
|
-
id="meta-llama/Llama-3.2-11B-Vision-Instruct",
|
|
376
|
-
hf_repo_id="meta-llama/Llama-3.2-11B-Vision-Instruct",
|
|
377
|
-
name="Llama 3.2 11B Vision Instruct",
|
|
378
|
-
family="Llama 3.2 Vision",
|
|
379
|
-
tag="local-vlm",
|
|
380
|
-
size="~11-22GB (quant)",
|
|
381
|
-
quantization="Q4_K_M GGUF widely available",
|
|
382
|
-
provider_hints=["ollama", "llamacpp", "lmstudio", "vllm"],
|
|
383
|
-
download_strategy="hf_hub",
|
|
384
|
-
load_strategy="ollama",
|
|
385
|
-
hardware=HardwareProfile(min_ram_gb=14.0, recommended_ram_gb=18.0, notes="Excellent GGUF support. Ollama / llama.cpp default path for most users."),
|
|
386
|
-
source_country="미국", source_company="Meta",
|
|
387
|
-
verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="llama3.2", verified_by="hf-api-light"),
|
|
388
|
-
display_priority=40,
|
|
389
|
-
),
|
|
390
|
-
|
|
391
|
-
# Pixtral (Mistral multimodal, strong)
|
|
392
|
-
ModelCapability(
|
|
393
|
-
id="mistralai/Pixtral-12B-2409",
|
|
394
|
-
hf_repo_id="mistralai/Pixtral-12B-2409",
|
|
395
|
-
name="Pixtral 12B (Mistral)",
|
|
396
|
-
family="Pixtral",
|
|
397
|
-
tag="local-vlm",
|
|
398
|
-
size="~12-24GB",
|
|
399
|
-
quantization="GGUF / AWQ ports",
|
|
400
|
-
provider_hints=["vllm", "ollama", "lmstudio"],
|
|
401
|
-
download_strategy="hf_hub",
|
|
402
|
-
load_strategy="vllm",
|
|
403
|
-
hardware=HardwareProfile(min_ram_gb=16.0, recommended_ram_gb=20.0, cuda_pref=True, notes="High quality vision-language. Best on CUDA / vLLM; GGUF for CPU/Apple via community ports."),
|
|
404
|
-
source_country="프랑스", source_company="Mistral AI",
|
|
405
572
|
verification=VerificationStatus(
|
|
406
|
-
hf_exists=True,
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
has_weights_hint=True,
|
|
410
|
-
pipeline_tag=None,
|
|
411
|
-
license="mistral-research",
|
|
412
|
-
notes="HF repo and weights are present, but config/tokenizer files were not visible in the lightweight HF tree check; treat as available but not local-load verified.",
|
|
573
|
+
hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
|
|
574
|
+
has_weights_hint=True, pipeline_tag="image-text-to-text", likes=4, license="apache-2.0",
|
|
575
|
+
notes="2025-02 generation, library_name=transformers; superseded by Qwen3.5 9B.",
|
|
413
576
|
verified_by="hf-api-light",
|
|
414
577
|
),
|
|
415
|
-
display_priority=45,
|
|
416
578
|
),
|
|
417
|
-
|
|
418
|
-
# Additional recent multimodal (2025-2026 era ports, MLX-first where possible)
|
|
419
579
|
ModelCapability(
|
|
420
580
|
id="mlx-community/Llama-3.2-11B-Vision-Instruct-4bit",
|
|
421
581
|
hf_repo_id="mlx-community/Llama-3.2-11B-Vision-Instruct-4bit",
|
|
422
582
|
name="Llama 3.2 11B Vision Instruct",
|
|
423
583
|
family="Llama 3.2 Vision",
|
|
424
584
|
tag="local-vlm",
|
|
425
|
-
size="6.
|
|
585
|
+
size="6.0GB",
|
|
586
|
+
architecture="mllama",
|
|
587
|
+
download_size_gb=6.02,
|
|
588
|
+
lifecycle=LEGACY,
|
|
426
589
|
quantization="4bit",
|
|
427
|
-
|
|
428
|
-
download_strategy="hf_hub",
|
|
429
|
-
load_strategy="mlx_vlm",
|
|
430
|
-
hardware=HardwareProfile(min_ram_gb=10.0, recommended_ram_gb=14.0, apple_silicon_pref=True, notes="Excellent vision for its size. Great all-rounder multimodal."),
|
|
590
|
+
hardware=HardwareProfile(min_ram_gb=10.0, recommended_ram_gb=14.0, apple_silicon_pref=True),
|
|
431
591
|
source_country="미국", source_company="Meta",
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
name="Phi-3.5 Vision 4B",
|
|
440
|
-
family="Phi Vision",
|
|
441
|
-
tag="local-vlm",
|
|
442
|
-
size="3.2GB",
|
|
443
|
-
quantization="4bit",
|
|
444
|
-
provider_hints=["local_mlx"],
|
|
445
|
-
download_strategy="hf_hub",
|
|
446
|
-
load_strategy="mlx_vlm",
|
|
447
|
-
hardware=HardwareProfile(min_ram_gb=6.0, recommended_ram_gb=8.0, apple_silicon_pref=True, notes="Microsoft small VLM, fast and capable for on-device vision tasks."),
|
|
448
|
-
source_country="미국", source_company="Microsoft",
|
|
449
|
-
verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="mit", verified_by="hf-api-light"),
|
|
450
|
-
display_priority=30,
|
|
451
|
-
),
|
|
452
|
-
ModelCapability(
|
|
453
|
-
id="mlx-community/Qwen2.5-VL-7B-Instruct-4bit",
|
|
454
|
-
hf_repo_id="mlx-community/Qwen2.5-VL-7B-Instruct-4bit",
|
|
455
|
-
name="Qwen2.5-VL 7B",
|
|
456
|
-
family="Qwen2.5-VL",
|
|
457
|
-
tag="local-vlm",
|
|
458
|
-
size="4.5GB",
|
|
459
|
-
quantization="4bit",
|
|
460
|
-
provider_hints=["local_mlx", "ollama"],
|
|
461
|
-
download_strategy="hf_hub",
|
|
462
|
-
load_strategy="mlx_vlm",
|
|
463
|
-
hardware=HardwareProfile(min_ram_gb=8.0, recommended_ram_gb=12.0, apple_silicon_pref=True, notes="Strong Chinese/English vision model, updated from Qwen2-VL."),
|
|
464
|
-
source_country="중국", source_company="Alibaba",
|
|
465
|
-
verification=VerificationStatus(hf_exists=True, has_config=True, has_tokenizer=True, has_weights_hint=True, pipeline_tag="image-text-to-text", license="apache-2.0", verified_by="hf-api-light"),
|
|
466
|
-
recommended_default=True,
|
|
467
|
-
display_priority=18,
|
|
592
|
+
license="llama3.2",
|
|
593
|
+
verification=VerificationStatus(
|
|
594
|
+
hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
|
|
595
|
+
has_weights_hint=True, pipeline_tag="image-text-to-text", likes=7, license="llama3.2",
|
|
596
|
+
notes="2024-10 generation, library_name=transformers; superseded by Qwen3.5 9B.",
|
|
597
|
+
verified_by="hf-api-light",
|
|
598
|
+
),
|
|
468
599
|
),
|
|
469
600
|
ModelCapability(
|
|
470
|
-
id="mlx-community/
|
|
471
|
-
hf_repo_id="mlx-community/
|
|
472
|
-
name="
|
|
473
|
-
family="
|
|
601
|
+
id="mlx-community/Llama-4-Scout-17B-16E-Instruct-4bit",
|
|
602
|
+
hf_repo_id="mlx-community/Llama-4-Scout-17B-16E-Instruct-4bit",
|
|
603
|
+
name="Llama 4 Scout 17B 16E",
|
|
604
|
+
family="Llama 4",
|
|
474
605
|
tag="local-vlm",
|
|
475
|
-
size="
|
|
606
|
+
size="61.1GB",
|
|
607
|
+
architecture="llama4",
|
|
608
|
+
download_size_gb=61.14,
|
|
609
|
+
lifecycle=LEGACY,
|
|
476
610
|
quantization="4bit",
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
|
|
611
|
+
hardware=HardwareProfile(
|
|
612
|
+
min_ram_gb=72.0, recommended_ram_gb=96.0, apple_silicon_pref=True,
|
|
613
|
+
notes="61GB on disk despite the \"4bit\" name — 12 shards. The registry claimed 11.8GB "
|
|
614
|
+
"until it was measured; only a very high-RAM machine can load it.",
|
|
615
|
+
),
|
|
616
|
+
source_country="미국", source_company="Meta",
|
|
617
|
+
license="llama4",
|
|
618
|
+
verification=VerificationStatus(
|
|
619
|
+
hf_exists=True, hf_last_checked="2026-08-10", has_config=True, has_tokenizer=True,
|
|
620
|
+
has_weights_hint=True, pipeline_tag="image-text-to-text", likes=12, license="llama4",
|
|
621
|
+
notes="2025-05 generation, library_name=transformers; the MoE slot is now Qwen3.6 35B A3B. "
|
|
622
|
+
"Meta's own repo is gated, so this port is the only anonymous route.",
|
|
623
|
+
verified_by="hf-api-light",
|
|
624
|
+
),
|
|
484
625
|
),
|
|
485
626
|
]
|
|
486
627
|
|
|
487
628
|
|
|
488
629
|
def get_all_capabilities() -> List[ModelCapability]:
|
|
630
|
+
"""Every entry the registry knows — recommended *and* recognised-only.
|
|
631
|
+
|
|
632
|
+
This is the verification surface: ``scripts/verify_hf_model_registry.py``
|
|
633
|
+
re-measures all of it, because a legacy entry that quietly disappeared from
|
|
634
|
+
the Hub should be reported rather than kept as a comforting fiction.
|
|
635
|
+
"""
|
|
636
|
+
return [*_REGISTRY, *_LEGACY_REGISTRY]
|
|
637
|
+
|
|
638
|
+
|
|
639
|
+
def get_recommended_capabilities() -> List[ModelCapability]:
|
|
640
|
+
"""Current-generation entries: catalog, download and recommendation input."""
|
|
489
641
|
return list(_REGISTRY)
|
|
490
642
|
|
|
491
643
|
|
|
644
|
+
def get_legacy_capabilities() -> List[ModelCapability]:
|
|
645
|
+
"""Recognised-only entries: never offered, never recommended, still named."""
|
|
646
|
+
return list(_LEGACY_REGISTRY)
|
|
647
|
+
|
|
648
|
+
|
|
492
649
|
def get_capability(model_id: str) -> Optional[ModelCapability]:
|
|
493
|
-
|
|
650
|
+
"""Look an id up across both lists — recognition covers legacy weights too."""
|
|
651
|
+
for m in get_all_capabilities():
|
|
494
652
|
if m.id == model_id or m.hf_repo_id == model_id:
|
|
495
653
|
return m
|
|
496
654
|
return None
|
|
497
655
|
|
|
498
656
|
|
|
657
|
+
def is_recognized_model(model_id: str) -> bool:
|
|
658
|
+
"""True when the registry can name this model, whatever its lifecycle.
|
|
659
|
+
|
|
660
|
+
Used by the load path: a model already on disk must keep working even after
|
|
661
|
+
its generation stops being offered.
|
|
662
|
+
"""
|
|
663
|
+
return get_capability(model_id) is not None
|
|
664
|
+
|
|
665
|
+
|
|
666
|
+
def is_recommended_model(model_id: str) -> bool:
|
|
667
|
+
"""True only for current-generation entries — the download/offer gate."""
|
|
668
|
+
cap = get_capability(model_id)
|
|
669
|
+
return cap is not None and cap.lifecycle == RECOMMENDED
|
|
670
|
+
|
|
671
|
+
|
|
499
672
|
def build_engine_model_catalog() -> Dict[str, List[Dict[str, Any]]]:
|
|
500
|
-
"""Return legacy ENGINE_MODEL_CATALOG shape, enriched with
|
|
673
|
+
"""Return legacy ENGINE_MODEL_CATALOG shape, enriched with rich fields.
|
|
674
|
+
|
|
675
|
+
Built from the **recommended** list only: the catalog is what the product
|
|
676
|
+
offers, and offering a superseded generation is how users end up downloading
|
|
677
|
+
something we would not stand behind.
|
|
678
|
+
"""
|
|
501
679
|
from collections import defaultdict
|
|
502
680
|
by_engine: Dict[str, List[Dict[str, Any]]] = defaultdict(list)
|
|
503
681
|
|
|
@@ -513,13 +691,14 @@ def build_engine_model_catalog() -> Dict[str, List[Dict[str, Any]]]:
|
|
|
513
691
|
for eng_key, hints in engine_map.items():
|
|
514
692
|
if any(h in cap.provider_hints for h in hints) or eng_key in cap.provider_hints:
|
|
515
693
|
legacy = cap.to_legacy_dict()
|
|
516
|
-
# Adapt id for non-mlx engines
|
|
694
|
+
# Adapt id for non-mlx engines. These are provisional prefixed
|
|
695
|
+
# ids; `model_catalog._normalize_engine_entry` then replaces them
|
|
696
|
+
# with the verified per-engine repo from MODEL_ENGINE_ALIASES.
|
|
697
|
+
# (Until 11.2.0 the ollama branch *fabricated* a repo path from
|
|
698
|
+
# the family name — `ggml-org/<family>-12B-it-GGUF` — which
|
|
699
|
+
# resolved to a 404 for anything but Gemma 4 12B.)
|
|
517
700
|
if eng_key == "ollama" and not legacy["id"].startswith("ollama:"):
|
|
518
|
-
|
|
519
|
-
if "gguf" in cap.tag.lower() or "gguf" in (cap.quantization or "").lower():
|
|
520
|
-
legacy["id"] = f"ollama:hf.co/ggml-org/{cap.family.lower().replace(' ', '')}-12B-it-GGUF:Q4_K_M" # fallback, overridden by aliases
|
|
521
|
-
else:
|
|
522
|
-
legacy["id"] = f"ollama:{cap.hf_repo_id.split('/')[-1].lower()}"
|
|
701
|
+
legacy["id"] = f"ollama:{cap.hf_repo_id.split('/')[-1].lower()}"
|
|
523
702
|
elif eng_key == "vllm" and not legacy["id"].startswith("vllm:"):
|
|
524
703
|
legacy["id"] = f"vllm:{cap.hf_repo_id}"
|
|
525
704
|
elif eng_key == "lmstudio" and not legacy["id"].startswith("lmstudio:"):
|
|
@@ -528,23 +707,17 @@ def build_engine_model_catalog() -> Dict[str, List[Dict[str, Any]]]:
|
|
|
528
707
|
legacy["id"] = f"llamacpp:{cap.hf_repo_id}-GGUF"
|
|
529
708
|
by_engine[eng_key].append(legacy)
|
|
530
709
|
|
|
531
|
-
|
|
532
|
-
# If projection missed any, inject the original local_mlx entries enriched
|
|
533
|
-
if not by_engine.get("local_mlx"):
|
|
534
|
-
for cap in _REGISTRY:
|
|
535
|
-
if "local_mlx" in cap.provider_hints:
|
|
536
|
-
by_engine["local_mlx"].append(cap.to_legacy_dict())
|
|
537
|
-
|
|
538
|
-
return {k: v for k, v in by_engine.items()}
|
|
710
|
+
return dict(by_engine)
|
|
539
711
|
|
|
540
712
|
|
|
541
713
|
def get_verified_models() -> List[Dict[str, Any]]:
|
|
542
|
-
"""Return only load-verified
|
|
714
|
+
"""Return only load-verified recommended entries with rich fields (API/UI)."""
|
|
543
715
|
return [
|
|
544
716
|
c.to_legacy_dict() for c in _REGISTRY
|
|
545
717
|
if c.verification.hf_exists and c.verification.has_config and c.verification.has_tokenizer
|
|
546
718
|
]
|
|
547
719
|
|
|
548
720
|
|
|
549
|
-
# Back-compat: expose a simple list mirroring the old top-level for mlx
|
|
721
|
+
# Back-compat: expose a simple list mirroring the old top-level for mlx.
|
|
722
|
+
# Recommended entries only — same reasoning as build_engine_model_catalog.
|
|
550
723
|
LOCAL_MLX_MODELS = [c.to_legacy_dict() for c in _REGISTRY if "local_mlx" in c.provider_hints]
|