ltcai 11.2.0 → 11.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -53
- package/docs/CHANGELOG.md +61 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/MULTI_AGENT_RUNTIME.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +6 -2
- package/docs/PERMISSION_MODE.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +2 -2
- package/docs/v11.3.0_PLAN.md +202 -0
- package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +287 -0
- package/lattice_brain/graph/_kg_common/extraction.py +516 -0
- package/lattice_brain/graph/_kg_common/relations.py +161 -0
- package/lattice_brain/graph/_kg_common/text.py +479 -0
- package/lattice_brain/graph/discovery_index/__init__.py +35 -0
- package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
- package/lattice_brain/graph/discovery_index/extract.py +137 -0
- package/lattice_brain/graph/discovery_index/scan.py +411 -0
- package/lattice_brain/graph/discovery_index/upsert.py +495 -0
- package/lattice_brain/graph/projection/__init__.py +42 -0
- package/lattice_brain/graph/projection/curation.py +500 -0
- package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
- package/lattice_brain/graph/retrieval/__init__.py +54 -0
- package/lattice_brain/graph/retrieval/context.py +197 -0
- package/lattice_brain/graph/retrieval/graph_view.py +319 -0
- package/lattice_brain/graph/retrieval/hybrid.py +488 -0
- package/lattice_brain/graph/retrieval/maintenance.py +121 -0
- package/lattice_brain/graph/retrieval/signals.py +95 -0
- package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
- package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
- package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
- package/lattice_brain/graph/retrieval_vector/search.py +560 -0
- package/lattice_brain/graph/retrieval_vector/status.py +374 -0
- package/lattice_brain/ingestion/__init__.py +130 -0
- package/lattice_brain/ingestion/_contract.py +90 -0
- package/lattice_brain/ingestion/constants.py +127 -0
- package/lattice_brain/ingestion/folder_scan.py +57 -0
- package/lattice_brain/ingestion/folders.py +258 -0
- package/lattice_brain/ingestion/hashing.py +26 -0
- package/lattice_brain/ingestion/jobs_api.py +107 -0
- package/lattice_brain/ingestion/models.py +80 -0
- package/lattice_brain/ingestion/pipeline.py +486 -0
- package/lattice_brain/ingestion/quality.py +209 -0
- package/lattice_brain/ingestion/routing.py +295 -0
- package/lattice_brain/multimodal/__init__.py +164 -0
- package/lattice_brain/multimodal/audio.py +77 -0
- package/lattice_brain/multimodal/common.py +118 -0
- package/lattice_brain/multimodal/images.py +498 -0
- package/lattice_brain/multimodal/ports.py +169 -0
- package/lattice_brain/multimodal/video.py +410 -0
- package/lattice_brain/portability/__init__.py +90 -0
- package/lattice_brain/portability/_contract.py +42 -0
- package/lattice_brain/portability/backups.py +338 -0
- package/lattice_brain/portability/bundles.py +136 -0
- package/lattice_brain/portability/constants.py +93 -0
- package/lattice_brain/portability/fsops.py +138 -0
- package/lattice_brain/portability/service.py +41 -0
- package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
- package/lattice_brain/runtime/__init__.py +1 -1
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/chronicle.py +63 -0
- package/latticeai/core/agent/__init__.py +93 -0
- package/latticeai/core/agent/_contract.py +79 -0
- package/latticeai/core/agent/context.py +57 -0
- package/latticeai/core/agent/deps.py +125 -0
- package/latticeai/core/agent/execution.py +622 -0
- package/latticeai/core/agent/planning.py +145 -0
- package/latticeai/core/agent/recovery.py +157 -0
- package/latticeai/core/agent/runtime.py +210 -0
- package/latticeai/core/agent/verification.py +231 -0
- package/latticeai/core/embedding_providers/__init__.py +151 -0
- package/latticeai/core/embedding_providers/base.py +199 -0
- package/latticeai/core/embedding_providers/captions.py +162 -0
- package/latticeai/core/embedding_providers/profiles.py +126 -0
- package/latticeai/core/embedding_providers/text.py +350 -0
- package/latticeai/core/embedding_providers/vision.py +352 -0
- package/latticeai/core/file_generation/__init__.py +115 -0
- package/latticeai/core/file_generation/bundles.py +76 -0
- package/latticeai/core/file_generation/extraction.py +154 -0
- package/latticeai/core/file_generation/inference.py +235 -0
- package/latticeai/core/file_generation/orchestration.py +152 -0
- package/latticeai/core/file_generation/prompting.py +117 -0
- package/latticeai/core/file_generation/repair.py +114 -0
- package/latticeai/core/file_generation/sanitize.py +61 -0
- package/latticeai/core/file_generation/validation.py +201 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +9 -0
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/integrations/telegram_bot/__init__.py +123 -0
- package/latticeai/integrations/telegram_bot/__main__.py +17 -0
- package/latticeai/integrations/telegram_bot/config.py +86 -0
- package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
- package/latticeai/integrations/telegram_bot/flows.py +478 -0
- package/latticeai/integrations/telegram_bot/helpers.py +322 -0
- package/latticeai/integrations/telegram_bot/screens.py +394 -0
- package/latticeai/models/router/__init__.py +88 -0
- package/latticeai/models/router/_contract.py +66 -0
- package/latticeai/models/router/branding.py +56 -0
- package/latticeai/models/router/catalog.py +69 -0
- package/latticeai/models/router/documents.py +199 -0
- package/latticeai/models/router/errors.py +37 -0
- package/latticeai/models/router/generation.py +258 -0
- package/latticeai/models/router/loading.py +291 -0
- package/latticeai/models/router/local_models.py +85 -0
- package/latticeai/models/router/registry.py +147 -0
- package/latticeai/runtime/build_phases/__init__.py +82 -0
- package/latticeai/runtime/build_phases/features.py +407 -0
- package/latticeai/runtime/build_phases/foundation.py +555 -0
- package/latticeai/runtime/build_phases/web.py +492 -0
- package/latticeai/runtime/runtime_context.py +1 -0
- package/latticeai/services/architecture_readiness.py +48 -19
- package/latticeai/services/brain_intelligence/__init__.py +58 -0
- package/latticeai/services/brain_intelligence/_contract.py +71 -0
- package/latticeai/services/brain_intelligence/consistency.py +193 -0
- package/latticeai/services/brain_intelligence/constants.py +47 -0
- package/latticeai/services/brain_intelligence/digest.py +258 -0
- package/latticeai/services/brain_intelligence/health.py +331 -0
- package/latticeai/services/brain_intelligence/proposals.py +264 -0
- package/latticeai/services/brain_intelligence/sampling.py +84 -0
- package/latticeai/services/brain_intelligence/service.py +48 -0
- package/latticeai/services/chronicle.py +557 -0
- package/latticeai/services/memory_service/__init__.py +52 -0
- package/latticeai/services/memory_service/_contract.py +100 -0
- package/latticeai/services/memory_service/brief.py +431 -0
- package/latticeai/services/memory_service/constants.py +57 -0
- package/latticeai/services/memory_service/maintenance.py +138 -0
- package/latticeai/services/memory_service/manager.py +186 -0
- package/latticeai/services/memory_service/proof.py +136 -0
- package/latticeai/services/memory_service/recall.py +225 -0
- package/latticeai/services/memory_service/service.py +48 -0
- package/latticeai/services/memory_service/stores.py +110 -0
- package/latticeai/services/model_runtime/__init__.py +322 -0
- package/latticeai/services/model_runtime/cloud.py +87 -0
- package/latticeai/services/model_runtime/download.py +282 -0
- package/latticeai/services/model_runtime/engines.py +341 -0
- package/latticeai/services/model_runtime/loading.py +178 -0
- package/latticeai/services/model_runtime/service.py +129 -0
- package/latticeai/services/model_runtime/state.py +131 -0
- package/latticeai/services/model_runtime/status.py +255 -0
- package/latticeai/services/product_readiness.py +15 -7
- package/latticeai/setup/wizard/__init__.py +126 -0
- package/latticeai/setup/wizard/catalog.py +172 -0
- package/latticeai/setup/wizard/detect.py +323 -0
- package/latticeai/setup/wizard/install.py +348 -0
- package/latticeai/setup/wizard/paths.py +168 -0
- package/latticeai/setup/wizard/plans.py +74 -0
- package/latticeai/setup/wizard/recommend.py +320 -0
- package/package.json +6 -2
- package/scripts/bump_version.py +14 -0
- package/scripts/capture_release_evidence.mjs +33 -21
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_i18n_namespace_coverage.mjs +41 -4
- package/scripts/check_max_file_lines.mjs +102 -0
- package/scripts/check_release_evidence_bound.mjs +30 -15
- package/scripts/check_screenshot_pixel_delta.py +34 -4
- package/scripts/check_server_i18n.mjs +1 -0
- package/scripts/generate_rust_parity_fixtures.py +562 -0
- package/scripts/lib/mock_server_fingerprint.mjs +94 -0
- package/scripts/release_screen_claims.json +31 -2
- package/src-tauri/Cargo.lock +361 -3
- package/src-tauri/Cargo.toml +6 -1
- package/src-tauri/src/backend.rs +349 -0
- package/src-tauri/src/folder.rs +33 -0
- package/src-tauri/src/main.rs +97 -399
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +41 -37
- package/static/app/assets/Act-yYpYnn0v.js +1 -0
- package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
- package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
- package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
- package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
- package/static/app/assets/Capture-CFIRsFNE.js +1 -0
- package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
- package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
- package/static/app/assets/Library-DwO3yZST.js +1 -0
- package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
- package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
- package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
- package/static/app/assets/System-DW8F-2xL.js +1 -0
- package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
- package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
- package/static/app/assets/brain-Ci1CkWjM.js +1 -0
- package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
- package/static/app/assets/circle-check-DfInj-qD.js +1 -0
- package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
- package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
- package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
- package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
- package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
- package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
- package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
- package/static/app/assets/index-_u5iUHDr.js +10 -0
- package/static/app/assets/input-B0lPdRQZ.js +1 -0
- package/static/app/assets/link-2-CoFbooHS.js +1 -0
- package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
- package/static/app/assets/primitives-DEbN-d6p.js +1 -0
- package/static/app/assets/search-BybIWPNd.js +1 -0
- package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
- package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
- package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
- package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
- package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
- package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
- package/static/app/assets/utils-BlZr7Pd4.js +4 -0
- package/static/app/assets/workspace-jJY4RuAV.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/graph/_kg_common.py +0 -1331
- package/lattice_brain/graph/discovery_index.py +0 -1141
- package/lattice_brain/graph/retrieval.py +0 -1120
- package/lattice_brain/graph/retrieval_vector.py +0 -1293
- package/lattice_brain/ingestion.py +0 -1525
- package/lattice_brain/multimodal.py +0 -1258
- package/latticeai/core/agent.py +0 -1465
- package/latticeai/core/embedding_providers.py +0 -1196
- package/latticeai/core/file_generation.py +0 -1047
- package/latticeai/integrations/telegram_bot.py +0 -1390
- package/latticeai/models/router.py +0 -1007
- package/latticeai/runtime/build_phases.py +0 -1450
- package/latticeai/services/brain_intelligence.py +0 -1083
- package/latticeai/services/memory_service.py +0 -1177
- package/latticeai/services/model_runtime.py +0 -1281
- package/latticeai/setup/wizard.py +0 -1310
- package/static/app/assets/Act-AWf0SAKp.js +0 -1
- package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
- package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
- package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
- package/static/app/assets/Capture-CqOSzyPr.js +0 -1
- package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
- package/static/app/assets/Library-CX-bbhmK.js +0 -1
- package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
- package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
- package/static/app/assets/System-Bu2t5hn1.js +0 -1
- package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
- package/static/app/assets/brain-DJMoqrwx.js +0 -1
- package/static/app/assets/index-BpYkzcVm.js +0 -10
- package/static/app/assets/input-DSlJJxRs.js +0 -1
- package/static/app/assets/primitives-BCx6TvfG.js +0 -1
- package/static/app/assets/search-Cgy8cCFJ.js +0 -1
- package/static/app/assets/utils-zqPZJxdx.js +0 -4
- package/static/app/assets/workspace-DXTihhfU.js +0 -1
|
@@ -0,0 +1,322 @@
|
|
|
1
|
+
"""Model runtime and provider helpers for Lattice AI.
|
|
2
|
+
|
|
3
|
+
This module owns local/cloud model preparation, engine detection, model download,
|
|
4
|
+
provider-specific server startup, smoke tests, and runtime feature payloads. It is
|
|
5
|
+
configured by ``server_app`` with app-level state but has no FastAPI app import.
|
|
6
|
+
|
|
7
|
+
Split into cohesive submodules in v11.3.0 (no behaviour change): ``state`` (the
|
|
8
|
+
immutable ``ModelRuntimeState`` and the consent gates), ``engines`` (engine
|
|
9
|
+
wrappers + the LM Studio client), ``download`` (Hugging Face readiness and
|
|
10
|
+
fetch), ``status`` (``engine_status`` / ``runtime_features`` / ``install_engine``),
|
|
11
|
+
``loading`` (identity resolution and the load entrypoints), ``cloud`` (cloud key
|
|
12
|
+
verification) and ``service`` (the bound ``ModelRuntimeService``). This module
|
|
13
|
+
re-exports every name the single file exposed, so
|
|
14
|
+
``latticeai.services.model_runtime.X`` keeps working — including the private
|
|
15
|
+
names ``model_loading._get_model_runtime_deps`` imports.
|
|
16
|
+
|
|
17
|
+
Two deliberate omissions: ``engines._LMSTUDIO_MODELS_CACHE`` and
|
|
18
|
+
``_LMSTUDIO_MODELS_CACHE_TS`` are *rebound* by ``get_lmstudio_models``, so a
|
|
19
|
+
re-export here would be a snapshot frozen at import time. They stay reachable
|
|
20
|
+
at ``latticeai.services.model_runtime.engines``, which is where the live values
|
|
21
|
+
are.
|
|
22
|
+
|
|
23
|
+
Stubbing note: rebinding a name *here* changes only this module's name — the
|
|
24
|
+
lazy ``from latticeai.services.model_runtime import …`` calls in
|
|
25
|
+
``model_engines`` and ``model_loading`` read it, but a submodule that uses the
|
|
26
|
+
name holds its own reference. A test standing in for a helper patches the
|
|
27
|
+
submodule that uses it.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
# The single file had no ``__all__``, so its public surface was "every module
|
|
33
|
+
# global" — including the names it imported for its own use. Every re-export
|
|
34
|
+
# below therefore uses the redundant-alias form: it reproduces exactly that
|
|
35
|
+
# surface, and it marks each name as deliberate rather than a leftover import.
|
|
36
|
+
from latticeai.core.quiet import quiet as quiet
|
|
37
|
+
from latticeai.models.router import HF_MODELS_ROOT as HF_MODELS_ROOT
|
|
38
|
+
from latticeai.models.router import (
|
|
39
|
+
OPENAI_COMPATIBLE_PROVIDERS as OPENAI_COMPATIBLE_PROVIDERS,
|
|
40
|
+
)
|
|
41
|
+
from latticeai.models.router import AsyncOpenAI as AsyncOpenAI
|
|
42
|
+
from latticeai.models.router import ensure_mlx_runtime as ensure_mlx_runtime
|
|
43
|
+
from latticeai.models.router import hf_cache_model_dir as hf_cache_model_dir
|
|
44
|
+
from latticeai.models.router import hf_model_dir as hf_model_dir
|
|
45
|
+
from latticeai.models.router import parse_model_ref as parse_model_ref
|
|
46
|
+
|
|
47
|
+
# Catalog data + version-dedup helpers live in ``model_catalog``; re-exported
|
|
48
|
+
# here so existing ``from ...model_runtime import ENGINE_MODEL_CATALOG`` imports
|
|
49
|
+
# keep working.
|
|
50
|
+
from latticeai.services.model_catalog import (
|
|
51
|
+
_VERSIONED_MODEL_PATTERNS as _VERSIONED_MODEL_PATTERNS,
|
|
52
|
+
)
|
|
53
|
+
from latticeai.services.model_catalog import (
|
|
54
|
+
ENGINE_INSTALLERS as ENGINE_INSTALLERS,
|
|
55
|
+
)
|
|
56
|
+
from latticeai.services.model_catalog import (
|
|
57
|
+
ENGINE_MODEL_CATALOG as ENGINE_MODEL_CATALOG,
|
|
58
|
+
)
|
|
59
|
+
from latticeai.services.model_catalog import (
|
|
60
|
+
MODEL_ENGINE_ALIASES as MODEL_ENGINE_ALIASES,
|
|
61
|
+
)
|
|
62
|
+
from latticeai.services.model_catalog import (
|
|
63
|
+
_model_family_version as _model_family_version,
|
|
64
|
+
)
|
|
65
|
+
from latticeai.services.model_catalog import (
|
|
66
|
+
_version_tuple as _version_tuple,
|
|
67
|
+
)
|
|
68
|
+
from latticeai.services.model_catalog import (
|
|
69
|
+
filter_lower_family_versions as filter_lower_family_versions,
|
|
70
|
+
)
|
|
71
|
+
from latticeai.services.model_errors import ModelRuntimeError as ModelRuntimeError
|
|
72
|
+
from latticeai.services.model_runtime.cloud import (
|
|
73
|
+
CLOUD_VERIFY_TTL_SECONDS as CLOUD_VERIFY_TTL_SECONDS,
|
|
74
|
+
)
|
|
75
|
+
from latticeai.services.model_runtime.cloud import (
|
|
76
|
+
_probe_cloud_model as _probe_cloud_model,
|
|
77
|
+
)
|
|
78
|
+
from latticeai.services.model_runtime.cloud import (
|
|
79
|
+
verify_cloud_models as verify_cloud_models,
|
|
80
|
+
)
|
|
81
|
+
from latticeai.services.model_runtime.download import (
|
|
82
|
+
download_hf_model as download_hf_model,
|
|
83
|
+
)
|
|
84
|
+
from latticeai.services.model_runtime.download import (
|
|
85
|
+
estimate_eta_seconds as estimate_eta_seconds,
|
|
86
|
+
)
|
|
87
|
+
from latticeai.services.model_runtime.download import (
|
|
88
|
+
hf_model_ready as hf_model_ready,
|
|
89
|
+
)
|
|
90
|
+
from latticeai.services.model_runtime.download import (
|
|
91
|
+
hf_repo_files_with_sizes as hf_repo_files_with_sizes,
|
|
92
|
+
)
|
|
93
|
+
from latticeai.services.model_runtime.download import (
|
|
94
|
+
model_download_progress_payload as model_download_progress_payload,
|
|
95
|
+
)
|
|
96
|
+
from latticeai.services.model_runtime.engines import (
|
|
97
|
+
_LMSTUDIO_MODELS_CACHE_TTL as _LMSTUDIO_MODELS_CACHE_TTL,
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
# The private aliases the historical module bound for its own use — including
|
|
101
|
+
# the ones ``model_loading._get_model_runtime_deps`` imports by name. They are
|
|
102
|
+
# re-exported from the submodule that already binds them under exactly these
|
|
103
|
+
# names, which is ruff's redundant-alias re-export form.
|
|
104
|
+
from latticeai.services.model_runtime.engines import (
|
|
105
|
+
_LOCAL_SERVER_PROCESSES as _LOCAL_SERVER_PROCESSES,
|
|
106
|
+
)
|
|
107
|
+
from latticeai.services.model_runtime.engines import (
|
|
108
|
+
LMSTUDIO_BUNDLED_CLI as LMSTUDIO_BUNDLED_CLI,
|
|
109
|
+
)
|
|
110
|
+
from latticeai.services.model_runtime.engines import (
|
|
111
|
+
LOCAL_SERVER_PROCESSES as LOCAL_SERVER_PROCESSES,
|
|
112
|
+
)
|
|
113
|
+
from latticeai.services.model_runtime.engines import (
|
|
114
|
+
VLLM_METAL_BIN as VLLM_METAL_BIN,
|
|
115
|
+
)
|
|
116
|
+
from latticeai.services.model_runtime.engines import (
|
|
117
|
+
VLLM_METAL_ENV as VLLM_METAL_ENV,
|
|
118
|
+
)
|
|
119
|
+
from latticeai.services.model_runtime.engines import (
|
|
120
|
+
VLLM_METAL_PYTHON as VLLM_METAL_PYTHON,
|
|
121
|
+
)
|
|
122
|
+
from latticeai.services.model_runtime.engines import (
|
|
123
|
+
_engine_install_plan as _engine_install_plan,
|
|
124
|
+
)
|
|
125
|
+
from latticeai.services.model_runtime.engines import (
|
|
126
|
+
_engine_support_status as _engine_support_status,
|
|
127
|
+
)
|
|
128
|
+
from latticeai.services.model_runtime.engines import (
|
|
129
|
+
_ensure_llamacpp_server as _ensure_llamacpp_server,
|
|
130
|
+
)
|
|
131
|
+
from latticeai.services.model_runtime.engines import (
|
|
132
|
+
_ensure_lmstudio_server as _ensure_lmstudio_server,
|
|
133
|
+
)
|
|
134
|
+
from latticeai.services.model_runtime.engines import (
|
|
135
|
+
_ensure_ollama_server as _ensure_ollama_server,
|
|
136
|
+
)
|
|
137
|
+
from latticeai.services.model_runtime.engines import (
|
|
138
|
+
_ensure_vllm_server as _ensure_vllm_server,
|
|
139
|
+
)
|
|
140
|
+
from latticeai.services.model_runtime.engines import (
|
|
141
|
+
_find_lmstudio_cli as _find_lmstudio_cli,
|
|
142
|
+
)
|
|
143
|
+
from latticeai.services.model_runtime.engines import (
|
|
144
|
+
_find_lmstudio_model_key as _find_lmstudio_model_key,
|
|
145
|
+
)
|
|
146
|
+
from latticeai.services.model_runtime.engines import (
|
|
147
|
+
_get_ollama_pulled_models as _get_ollama_pulled_models,
|
|
148
|
+
)
|
|
149
|
+
from latticeai.services.model_runtime.engines import (
|
|
150
|
+
_get_openai_compatible_server_models as _get_openai_compatible_server_models,
|
|
151
|
+
)
|
|
152
|
+
from latticeai.services.model_runtime.engines import (
|
|
153
|
+
_json_request as _json_request,
|
|
154
|
+
)
|
|
155
|
+
from latticeai.services.model_runtime.engines import (
|
|
156
|
+
_lmstudio_candidate_keys as _lmstudio_candidate_keys,
|
|
157
|
+
)
|
|
158
|
+
from latticeai.services.model_runtime.engines import (
|
|
159
|
+
_local_binary as _local_binary,
|
|
160
|
+
)
|
|
161
|
+
from latticeai.services.model_runtime.engines import (
|
|
162
|
+
_pull_ollama_model_with_progress as _pull_ollama_model_with_progress,
|
|
163
|
+
)
|
|
164
|
+
from latticeai.services.model_runtime.engines import (
|
|
165
|
+
_safe_engine_install_plan as _safe_engine_install_plan,
|
|
166
|
+
)
|
|
167
|
+
from latticeai.services.model_runtime.engines import (
|
|
168
|
+
_update_env_file as _update_env_file,
|
|
169
|
+
)
|
|
170
|
+
from latticeai.services.model_runtime.engines import (
|
|
171
|
+
_vllm_executable as _vllm_executable,
|
|
172
|
+
)
|
|
173
|
+
from latticeai.services.model_runtime.engines import (
|
|
174
|
+
_vllm_metal_python as _vllm_metal_python,
|
|
175
|
+
)
|
|
176
|
+
from latticeai.services.model_runtime.engines import (
|
|
177
|
+
_wait_for_openai_compatible_server as _wait_for_openai_compatible_server,
|
|
178
|
+
)
|
|
179
|
+
from latticeai.services.model_runtime.engines import (
|
|
180
|
+
_windows_binary_candidates as _windows_binary_candidates,
|
|
181
|
+
)
|
|
182
|
+
from latticeai.services.model_runtime.engines import (
|
|
183
|
+
engine_installed as engine_installed,
|
|
184
|
+
)
|
|
185
|
+
from latticeai.services.model_runtime.engines import (
|
|
186
|
+
engine_support_status as engine_support_status,
|
|
187
|
+
)
|
|
188
|
+
from latticeai.services.model_runtime.engines import (
|
|
189
|
+
ensure_llamacpp_server as ensure_llamacpp_server,
|
|
190
|
+
)
|
|
191
|
+
from latticeai.services.model_runtime.engines import (
|
|
192
|
+
ensure_lmstudio_model as ensure_lmstudio_model,
|
|
193
|
+
)
|
|
194
|
+
from latticeai.services.model_runtime.engines import (
|
|
195
|
+
ensure_lmstudio_server as ensure_lmstudio_server,
|
|
196
|
+
)
|
|
197
|
+
from latticeai.services.model_runtime.engines import (
|
|
198
|
+
ensure_ollama_server as ensure_ollama_server,
|
|
199
|
+
)
|
|
200
|
+
from latticeai.services.model_runtime.engines import (
|
|
201
|
+
ensure_vllm_server as ensure_vllm_server,
|
|
202
|
+
)
|
|
203
|
+
from latticeai.services.model_runtime.engines import (
|
|
204
|
+
find_lmstudio_cli as find_lmstudio_cli,
|
|
205
|
+
)
|
|
206
|
+
from latticeai.services.model_runtime.engines import (
|
|
207
|
+
get_lmstudio_models as get_lmstudio_models,
|
|
208
|
+
)
|
|
209
|
+
from latticeai.services.model_runtime.engines import (
|
|
210
|
+
get_ollama_pulled_models as get_ollama_pulled_models,
|
|
211
|
+
)
|
|
212
|
+
from latticeai.services.model_runtime.engines import (
|
|
213
|
+
get_openai_compatible_server_models as get_openai_compatible_server_models,
|
|
214
|
+
)
|
|
215
|
+
from latticeai.services.model_runtime.engines import (
|
|
216
|
+
lmstudio_api_base as lmstudio_api_base,
|
|
217
|
+
)
|
|
218
|
+
from latticeai.services.model_runtime.engines import (
|
|
219
|
+
lmstudio_native_api_base as lmstudio_native_api_base,
|
|
220
|
+
)
|
|
221
|
+
from latticeai.services.model_runtime.engines import (
|
|
222
|
+
local_binary as local_binary,
|
|
223
|
+
)
|
|
224
|
+
from latticeai.services.model_runtime.engines import (
|
|
225
|
+
pull_ollama_model_with_progress as pull_ollama_model_with_progress,
|
|
226
|
+
)
|
|
227
|
+
from latticeai.services.model_runtime.engines import (
|
|
228
|
+
vllm_executable as vllm_executable,
|
|
229
|
+
)
|
|
230
|
+
from latticeai.services.model_runtime.engines import (
|
|
231
|
+
vllm_metal_python as vllm_metal_python,
|
|
232
|
+
)
|
|
233
|
+
from latticeai.services.model_runtime.engines import (
|
|
234
|
+
wait_for_openai_compatible_server as wait_for_openai_compatible_server,
|
|
235
|
+
)
|
|
236
|
+
from latticeai.services.model_runtime.engines import (
|
|
237
|
+
windows_binary_candidates as windows_binary_candidates,
|
|
238
|
+
)
|
|
239
|
+
from latticeai.services.model_runtime.loading import (
|
|
240
|
+
_LOCAL_SMOKE_ENGINES as _LOCAL_SMOKE_ENGINES,
|
|
241
|
+
)
|
|
242
|
+
from latticeai.services.model_runtime.loading import (
|
|
243
|
+
_ModelResolution as _ModelResolution,
|
|
244
|
+
)
|
|
245
|
+
from latticeai.services.model_runtime.loading import (
|
|
246
|
+
_resolve_model_alias as _resolve_model_alias,
|
|
247
|
+
)
|
|
248
|
+
from latticeai.services.model_runtime.loading import (
|
|
249
|
+
_smoke_test_loaded_model as _smoke_test_loaded_model,
|
|
250
|
+
)
|
|
251
|
+
from latticeai.services.model_runtime.loading import (
|
|
252
|
+
build_model_resolution as build_model_resolution,
|
|
253
|
+
)
|
|
254
|
+
from latticeai.services.model_runtime.loading import (
|
|
255
|
+
ensure_engine_ready as ensure_engine_ready,
|
|
256
|
+
)
|
|
257
|
+
from latticeai.services.model_runtime.loading import (
|
|
258
|
+
normalize_local_model_request as normalize_local_model_request,
|
|
259
|
+
)
|
|
260
|
+
from latticeai.services.model_runtime.loading import (
|
|
261
|
+
prepare_and_load_model as prepare_and_load_model,
|
|
262
|
+
)
|
|
263
|
+
from latticeai.services.model_runtime.loading import (
|
|
264
|
+
prepare_and_load_model_stream as prepare_and_load_model_stream,
|
|
265
|
+
)
|
|
266
|
+
from latticeai.services.model_runtime.loading import (
|
|
267
|
+
sse_event as sse_event,
|
|
268
|
+
)
|
|
269
|
+
from latticeai.services.model_runtime.service import (
|
|
270
|
+
ModelRuntimeService as ModelRuntimeService,
|
|
271
|
+
)
|
|
272
|
+
from latticeai.services.model_runtime.service import (
|
|
273
|
+
build_model_runtime as build_model_runtime,
|
|
274
|
+
)
|
|
275
|
+
from latticeai.services.model_runtime.service import (
|
|
276
|
+
configure_model_runtime as configure_model_runtime,
|
|
277
|
+
)
|
|
278
|
+
from latticeai.services.model_runtime.state import (
|
|
279
|
+
_MODEL_LOADING_COMPAT_EXPORTS as _MODEL_LOADING_COMPAT_EXPORTS,
|
|
280
|
+
)
|
|
281
|
+
from latticeai.services.model_runtime.state import (
|
|
282
|
+
_SMOKE_PROMPT as _SMOKE_PROMPT,
|
|
283
|
+
)
|
|
284
|
+
from latticeai.services.model_runtime.state import (
|
|
285
|
+
ModelRuntimeState as ModelRuntimeState,
|
|
286
|
+
)
|
|
287
|
+
from latticeai.services.model_runtime.state import (
|
|
288
|
+
_download_allowed as _download_allowed,
|
|
289
|
+
)
|
|
290
|
+
from latticeai.services.model_runtime.state import (
|
|
291
|
+
_download_block as _download_block,
|
|
292
|
+
)
|
|
293
|
+
from latticeai.services.model_runtime.state import (
|
|
294
|
+
_engine_install_block as _engine_install_block,
|
|
295
|
+
)
|
|
296
|
+
from latticeai.services.model_runtime.state import (
|
|
297
|
+
_friendly_model_runtime_error as _friendly_model_runtime_error,
|
|
298
|
+
)
|
|
299
|
+
from latticeai.services.model_runtime.state import (
|
|
300
|
+
_missing_current_user as _missing_current_user,
|
|
301
|
+
)
|
|
302
|
+
from latticeai.services.model_runtime.state import (
|
|
303
|
+
_missing_user_api_key as _missing_user_api_key,
|
|
304
|
+
)
|
|
305
|
+
from latticeai.services.model_runtime.state import (
|
|
306
|
+
_model_runtime_compatibility as _model_runtime_compatibility,
|
|
307
|
+
)
|
|
308
|
+
from latticeai.services.model_runtime.state import (
|
|
309
|
+
create_model_runtime_state as create_model_runtime_state,
|
|
310
|
+
)
|
|
311
|
+
from latticeai.services.model_runtime.status import (
|
|
312
|
+
_install_engine as _install_engine,
|
|
313
|
+
)
|
|
314
|
+
from latticeai.services.model_runtime.status import (
|
|
315
|
+
engine_status as engine_status,
|
|
316
|
+
)
|
|
317
|
+
from latticeai.services.model_runtime.status import (
|
|
318
|
+
install_engine as install_engine,
|
|
319
|
+
)
|
|
320
|
+
from latticeai.services.model_runtime.status import (
|
|
321
|
+
runtime_features as runtime_features,
|
|
322
|
+
)
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""Cloud model verification — does this key actually answer for this model?
|
|
2
|
+
|
|
3
|
+
A one-token chat completion per configured cloud model, cached per service
|
|
4
|
+
instance for :data:`CLOUD_VERIFY_TTL_SECONDS`. A model the router already knows
|
|
5
|
+
is unavailable is recorded as such without a network call.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import asyncio
|
|
11
|
+
import os
|
|
12
|
+
import time
|
|
13
|
+
from typing import Any, Dict, Optional
|
|
14
|
+
|
|
15
|
+
from latticeai.models.router import (
|
|
16
|
+
OPENAI_COMPATIBLE_PROVIDERS,
|
|
17
|
+
AsyncOpenAI,
|
|
18
|
+
parse_model_ref,
|
|
19
|
+
)
|
|
20
|
+
from latticeai.services.model_runtime.state import ModelRuntimeState
|
|
21
|
+
|
|
22
|
+
CLOUD_VERIFY_TTL_SECONDS = 600
|
|
23
|
+
|
|
24
|
+
async def _probe_cloud_model(model_ref: str) -> Dict[str, Any]:
|
|
25
|
+
provider, model_name = parse_model_ref(model_ref)
|
|
26
|
+
config = OPENAI_COMPATIBLE_PROVIDERS.get(provider)
|
|
27
|
+
if not config:
|
|
28
|
+
return {"ok": False, "reason": f"Unsupported provider: {provider}"}
|
|
29
|
+
|
|
30
|
+
api_key = os.getenv(config["env_key"]) or config.get("api_key_fallback")
|
|
31
|
+
if not api_key:
|
|
32
|
+
return {"ok": False, "reason": f"Missing API key: {config['env_key']}"}
|
|
33
|
+
|
|
34
|
+
base_url = os.getenv(config.get("base_url_env", "")) if config.get("base_url_env") else None
|
|
35
|
+
base_url = base_url or config.get("base_url")
|
|
36
|
+
try:
|
|
37
|
+
# base_url is passed only when configured: an explicit None is not
|
|
38
|
+
# the same as omitting the argument.
|
|
39
|
+
client = (
|
|
40
|
+
AsyncOpenAI(api_key=api_key, base_url=base_url)
|
|
41
|
+
if base_url
|
|
42
|
+
else AsyncOpenAI(api_key=api_key)
|
|
43
|
+
)
|
|
44
|
+
await asyncio.wait_for(
|
|
45
|
+
client.chat.completions.create(
|
|
46
|
+
model=model_name,
|
|
47
|
+
messages=[{"role": "user", "content": "ping"}],
|
|
48
|
+
max_tokens=1,
|
|
49
|
+
temperature=0,
|
|
50
|
+
),
|
|
51
|
+
timeout=15,
|
|
52
|
+
)
|
|
53
|
+
return {"ok": True, "reason": "ok"}
|
|
54
|
+
except Exception as e:
|
|
55
|
+
return {"ok": False, "reason": str(e)[:220]}
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
async def verify_cloud_models(
|
|
59
|
+
force: bool = False,
|
|
60
|
+
provider_filter: Optional[str] = None,
|
|
61
|
+
*,
|
|
62
|
+
state: ModelRuntimeState,
|
|
63
|
+
cache: Dict[str, Dict[str, Any]],
|
|
64
|
+
) -> Dict[str, Dict]:
|
|
65
|
+
now = time.time()
|
|
66
|
+
r = state.router
|
|
67
|
+
cloud_items = [item for item in (r.detected_cloud_models() if r else []) if item.get("tag") == "cloud"]
|
|
68
|
+
if provider_filter:
|
|
69
|
+
cloud_items = [item for item in cloud_items if item.get("provider") == provider_filter]
|
|
70
|
+
|
|
71
|
+
results: Dict[str, Dict] = {}
|
|
72
|
+
for item in cloud_items:
|
|
73
|
+
model_ref = item["id"]
|
|
74
|
+
cached = cache.get(model_ref)
|
|
75
|
+
if not force and cached and (now - cached.get("ts", 0) <= CLOUD_VERIFY_TTL_SECONDS):
|
|
76
|
+
results[model_ref] = cached
|
|
77
|
+
continue
|
|
78
|
+
if item.get("available") is False:
|
|
79
|
+
record = {"ok": False, "reason": item.get("requires") or "API key missing", "ts": now}
|
|
80
|
+
cache[model_ref] = record
|
|
81
|
+
results[model_ref] = record
|
|
82
|
+
continue
|
|
83
|
+
probe = await _probe_cloud_model(model_ref)
|
|
84
|
+
record = {"ok": bool(probe.get("ok")), "reason": probe.get("reason", ""), "ts": now}
|
|
85
|
+
cache[model_ref] = record
|
|
86
|
+
results[model_ref] = record
|
|
87
|
+
return results
|
|
@@ -0,0 +1,282 @@
|
|
|
1
|
+
"""Hugging Face model presence, download, and download progress.
|
|
2
|
+
|
|
3
|
+
Answers one question — "are the weights actually on this disk?" — and, when
|
|
4
|
+
they are not and the caller has consent, fetches them while emitting a progress
|
|
5
|
+
payload per file. The readiness check is deliberately format-aware: a GGUF
|
|
6
|
+
repo, an MLX 4-bit repo and a vLLM bf16 repo each prove completeness
|
|
7
|
+
differently.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import importlib.util
|
|
13
|
+
import logging
|
|
14
|
+
import time
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Any, Dict, List, Optional
|
|
17
|
+
|
|
18
|
+
from latticeai.core.quiet import quiet
|
|
19
|
+
from latticeai.models.router import hf_cache_model_dir, hf_model_dir
|
|
20
|
+
from latticeai.services.model_errors import ModelRuntimeError
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def hf_model_ready(repo_id: str, provider: str = "local_mlx") -> bool:
|
|
24
|
+
model_dir = hf_model_dir(repo_id)
|
|
25
|
+
if provider in {"local_mlx", "vllm"} and (not model_dir.exists() or not model_dir.is_dir()):
|
|
26
|
+
hf_cache_repo = Path.home() / ".cache" / "huggingface" / "hub" / f"models--{repo_id.replace('/', '--')}"
|
|
27
|
+
if hf_cache_repo.exists() and any(hf_cache_repo.glob("snapshots/*")):
|
|
28
|
+
if provider == "vllm":
|
|
29
|
+
return True
|
|
30
|
+
return hf_cache_model_dir(repo_id) is not None
|
|
31
|
+
return False
|
|
32
|
+
if not model_dir.exists() or not model_dir.is_dir():
|
|
33
|
+
return False
|
|
34
|
+
if provider == "llamacpp":
|
|
35
|
+
return any(model_dir.rglob("*.gguf"))
|
|
36
|
+
has_config = (model_dir / "config.json").exists()
|
|
37
|
+
has_weights = any(model_dir.glob("*.safetensors")) or any(model_dir.glob("*.bin"))
|
|
38
|
+
has_tokenizer = (
|
|
39
|
+
(model_dir / "tokenizer.json").exists()
|
|
40
|
+
or (model_dir / "tokenizer.model").exists()
|
|
41
|
+
or (model_dir / "tokenizer_config.json").exists()
|
|
42
|
+
)
|
|
43
|
+
return has_config and has_weights and has_tokenizer
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def model_download_progress_payload(
|
|
47
|
+
stage: str,
|
|
48
|
+
message: str,
|
|
49
|
+
*,
|
|
50
|
+
percent: Optional[float] = None,
|
|
51
|
+
detail: Optional[str] = None,
|
|
52
|
+
downloaded_bytes: Optional[int] = None,
|
|
53
|
+
total_bytes: Optional[int] = None,
|
|
54
|
+
eta_seconds: Optional[float] = None,
|
|
55
|
+
file: Optional[str] = None,
|
|
56
|
+
indeterminate: bool = False,
|
|
57
|
+
) -> Dict[str, Any]:
|
|
58
|
+
payload: Dict[str, Any] = {
|
|
59
|
+
"stage": stage,
|
|
60
|
+
"message": message,
|
|
61
|
+
"indeterminate": indeterminate,
|
|
62
|
+
"ts": time.time(),
|
|
63
|
+
}
|
|
64
|
+
if percent is not None:
|
|
65
|
+
payload["percent"] = max(0, min(100, round(float(percent), 1)))
|
|
66
|
+
if detail:
|
|
67
|
+
payload["detail"] = detail
|
|
68
|
+
if downloaded_bytes is not None:
|
|
69
|
+
payload["downloaded_bytes"] = max(0, int(downloaded_bytes))
|
|
70
|
+
if total_bytes is not None:
|
|
71
|
+
payload["total_bytes"] = max(0, int(total_bytes))
|
|
72
|
+
if eta_seconds is not None:
|
|
73
|
+
payload["eta_seconds"] = max(0, round(float(eta_seconds)))
|
|
74
|
+
if file:
|
|
75
|
+
payload["file"] = file
|
|
76
|
+
return payload
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def estimate_eta_seconds(started_at: float, percent: Optional[float]) -> Optional[float]:
|
|
80
|
+
if percent is None or percent <= 0 or percent >= 100:
|
|
81
|
+
return None
|
|
82
|
+
elapsed = max(0.0, time.time() - started_at)
|
|
83
|
+
return elapsed * (100.0 - percent) / percent
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def hf_repo_files_with_sizes(repo_id: str) -> List[Dict[str, Any]]:
|
|
87
|
+
from huggingface_hub import HfApi
|
|
88
|
+
|
|
89
|
+
api = HfApi()
|
|
90
|
+
try:
|
|
91
|
+
info = api.model_info(repo_id, files_metadata=True)
|
|
92
|
+
files = []
|
|
93
|
+
for sibling in getattr(info, "siblings", []) or []:
|
|
94
|
+
name = str(getattr(sibling, "rfilename", "") or "").strip()
|
|
95
|
+
if not name or name.endswith("/"):
|
|
96
|
+
continue
|
|
97
|
+
files.append({"name": name, "size": int(getattr(sibling, "size", 0) or 0)})
|
|
98
|
+
if files:
|
|
99
|
+
return files
|
|
100
|
+
except TypeError:
|
|
101
|
+
quiet()
|
|
102
|
+
except Exception as e:
|
|
103
|
+
logging.warning("huggingface model_info failed for %s: %s", repo_id, e)
|
|
104
|
+
|
|
105
|
+
return [{"name": str(name), "size": 0} for name in api.list_repo_files(repo_id) if str(name).strip()]
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def download_hf_model(
|
|
109
|
+
repo_id: str,
|
|
110
|
+
provider: str = "local_mlx",
|
|
111
|
+
progress_emit=None,
|
|
112
|
+
) -> Dict[str, Any]:
|
|
113
|
+
if importlib.util.find_spec("huggingface_hub") is None:
|
|
114
|
+
raise ModelRuntimeError(status_code=400, detail="huggingface_hub가 없습니다. 먼저 MLX runtime 설치를 진행해 주세요.")
|
|
115
|
+
|
|
116
|
+
target_dir = hf_model_dir(repo_id)
|
|
117
|
+
if hf_model_ready(repo_id, provider):
|
|
118
|
+
cached_dir = hf_cache_model_dir(repo_id) if provider == "local_mlx" else None
|
|
119
|
+
resolved_dir = cached_dir or target_dir
|
|
120
|
+
if progress_emit:
|
|
121
|
+
progress_emit(model_download_progress_payload(
|
|
122
|
+
"download",
|
|
123
|
+
"이미 다운로드된 모델을 확인했습니다.",
|
|
124
|
+
percent=100,
|
|
125
|
+
downloaded_bytes=0,
|
|
126
|
+
total_bytes=0,
|
|
127
|
+
eta_seconds=0,
|
|
128
|
+
))
|
|
129
|
+
return {"model": repo_id, "path": str(resolved_dir), "cached": True}
|
|
130
|
+
|
|
131
|
+
target_dir.mkdir(parents=True, exist_ok=True)
|
|
132
|
+
try:
|
|
133
|
+
from huggingface_hub import hf_hub_download
|
|
134
|
+
|
|
135
|
+
started_at = time.time()
|
|
136
|
+
all_files = hf_repo_files_with_sizes(repo_id)
|
|
137
|
+
if provider == "llamacpp":
|
|
138
|
+
ggufs = sorted(
|
|
139
|
+
[item for item in all_files if str(item["name"]).lower().endswith(".gguf")],
|
|
140
|
+
key=lambda item: str(item["name"]),
|
|
141
|
+
)
|
|
142
|
+
if not ggufs:
|
|
143
|
+
raise RuntimeError("GGUF 파일을 찾지 못했습니다.")
|
|
144
|
+
preference = ("q4_k_m", "q4_0", "q4_k_s", "q3_k_m", "q2_k")
|
|
145
|
+
selected_files = [
|
|
146
|
+
next(
|
|
147
|
+
(item for pref in preference for item in ggufs if pref in str(item["name"]).lower()),
|
|
148
|
+
ggufs[0],
|
|
149
|
+
)
|
|
150
|
+
]
|
|
151
|
+
else:
|
|
152
|
+
selected_files = all_files
|
|
153
|
+
|
|
154
|
+
total_bytes = sum(int(item.get("size") or 0) for item in selected_files) or None
|
|
155
|
+
downloaded_bytes = 0
|
|
156
|
+
total_files = max(1, len(selected_files))
|
|
157
|
+
if progress_emit:
|
|
158
|
+
progress_emit(model_download_progress_payload(
|
|
159
|
+
"download",
|
|
160
|
+
"모델 파일 정보를 확인했습니다.",
|
|
161
|
+
percent=0,
|
|
162
|
+
downloaded_bytes=0,
|
|
163
|
+
total_bytes=total_bytes,
|
|
164
|
+
indeterminate=total_bytes is None,
|
|
165
|
+
))
|
|
166
|
+
|
|
167
|
+
for index, item in enumerate(selected_files, start=1):
|
|
168
|
+
filename = str(item["name"])
|
|
169
|
+
size = int(item.get("size") or 0)
|
|
170
|
+
tqdm_class = None
|
|
171
|
+
if progress_emit:
|
|
172
|
+
current_percent = (
|
|
173
|
+
(downloaded_bytes / total_bytes) * 100 if total_bytes else ((index - 1) / total_files) * 100
|
|
174
|
+
)
|
|
175
|
+
progress_emit(model_download_progress_payload(
|
|
176
|
+
"download",
|
|
177
|
+
"모델 다운로드 중입니다.",
|
|
178
|
+
percent=current_percent,
|
|
179
|
+
detail=filename,
|
|
180
|
+
downloaded_bytes=downloaded_bytes,
|
|
181
|
+
total_bytes=total_bytes,
|
|
182
|
+
eta_seconds=estimate_eta_seconds(started_at, current_percent),
|
|
183
|
+
file=filename,
|
|
184
|
+
indeterminate=total_bytes is None and total_files <= 1,
|
|
185
|
+
))
|
|
186
|
+
try:
|
|
187
|
+
from tqdm.auto import tqdm as base_tqdm
|
|
188
|
+
|
|
189
|
+
downloaded_before = downloaded_bytes
|
|
190
|
+
last_emit = {"at": 0.0, "percent": -1.0}
|
|
191
|
+
|
|
192
|
+
def emit_byte_progress(
|
|
193
|
+
done_bytes: float,
|
|
194
|
+
# Bound per iteration: this callback outlives the loop
|
|
195
|
+
# body when a download runs long, and late binding
|
|
196
|
+
# would report every file's progress against the last
|
|
197
|
+
# file's offsets.
|
|
198
|
+
downloaded_before: int = downloaded_before,
|
|
199
|
+
size: Any = size,
|
|
200
|
+
index: int = index,
|
|
201
|
+
last_emit: dict = last_emit,
|
|
202
|
+
filename: str = filename,
|
|
203
|
+
) -> None:
|
|
204
|
+
done = max(0, int(done_bytes or 0))
|
|
205
|
+
if total_bytes:
|
|
206
|
+
aggregate = min(total_bytes, downloaded_before + done)
|
|
207
|
+
percent = (aggregate / total_bytes) * 100
|
|
208
|
+
else:
|
|
209
|
+
file_total = size or done
|
|
210
|
+
file_ratio = min(1.0, done / file_total) if file_total else 0.0
|
|
211
|
+
aggregate = downloaded_before + done
|
|
212
|
+
percent = ((index - 1) + file_ratio) / total_files * 100
|
|
213
|
+
now = time.time()
|
|
214
|
+
if percent < 100 and now - last_emit["at"] < 0.5 and percent - last_emit["percent"] < 0.3:
|
|
215
|
+
return
|
|
216
|
+
last_emit["at"] = now
|
|
217
|
+
last_emit["percent"] = percent
|
|
218
|
+
progress_emit(model_download_progress_payload(
|
|
219
|
+
"download",
|
|
220
|
+
"모델 다운로드 중입니다.",
|
|
221
|
+
percent=percent,
|
|
222
|
+
detail=filename,
|
|
223
|
+
downloaded_bytes=aggregate,
|
|
224
|
+
total_bytes=total_bytes,
|
|
225
|
+
eta_seconds=estimate_eta_seconds(started_at, percent),
|
|
226
|
+
file=filename,
|
|
227
|
+
indeterminate=total_bytes is None and total_files <= 1,
|
|
228
|
+
))
|
|
229
|
+
|
|
230
|
+
class ProgressTqdm(base_tqdm):
|
|
231
|
+
def update(self, n=1):
|
|
232
|
+
result = super().update(n)
|
|
233
|
+
emit_byte_progress(float(getattr(self, "n", 0) or 0))
|
|
234
|
+
return result
|
|
235
|
+
|
|
236
|
+
tqdm_class = ProgressTqdm
|
|
237
|
+
except Exception:
|
|
238
|
+
tqdm_class = None
|
|
239
|
+
local_path = hf_hub_download(
|
|
240
|
+
repo_id=repo_id,
|
|
241
|
+
filename=filename,
|
|
242
|
+
local_dir=str(target_dir),
|
|
243
|
+
tqdm_class=tqdm_class,
|
|
244
|
+
)
|
|
245
|
+
if size <= 0:
|
|
246
|
+
try:
|
|
247
|
+
size = Path(local_path).stat().st_size
|
|
248
|
+
except OSError:
|
|
249
|
+
size = 0
|
|
250
|
+
downloaded_bytes += size
|
|
251
|
+
if progress_emit:
|
|
252
|
+
current_percent = (
|
|
253
|
+
(downloaded_bytes / total_bytes) * 100 if total_bytes else (index / total_files) * 100
|
|
254
|
+
)
|
|
255
|
+
progress_emit(model_download_progress_payload(
|
|
256
|
+
"download",
|
|
257
|
+
"모델 다운로드 중입니다.",
|
|
258
|
+
percent=current_percent,
|
|
259
|
+
detail=filename,
|
|
260
|
+
downloaded_bytes=downloaded_bytes,
|
|
261
|
+
total_bytes=total_bytes,
|
|
262
|
+
eta_seconds=estimate_eta_seconds(started_at, current_percent),
|
|
263
|
+
file=filename,
|
|
264
|
+
indeterminate=False,
|
|
265
|
+
))
|
|
266
|
+
|
|
267
|
+
if progress_emit:
|
|
268
|
+
progress_emit(model_download_progress_payload(
|
|
269
|
+
"download",
|
|
270
|
+
"모델 다운로드가 완료되었습니다.",
|
|
271
|
+
percent=100,
|
|
272
|
+
downloaded_bytes=downloaded_bytes,
|
|
273
|
+
total_bytes=total_bytes or downloaded_bytes,
|
|
274
|
+
eta_seconds=0,
|
|
275
|
+
))
|
|
276
|
+
except Exception as e:
|
|
277
|
+
raise ModelRuntimeError(status_code=500, detail=f"{repo_id} 다운로드 실패: {str(e)[-2000:]}")
|
|
278
|
+
|
|
279
|
+
if not hf_model_ready(repo_id, provider):
|
|
280
|
+
raise ModelRuntimeError(status_code=500, detail=f"{repo_id} 다운로드가 완료되지 않았습니다. 모델 파일을 찾지 못했습니다.")
|
|
281
|
+
|
|
282
|
+
return {"model": repo_id, "path": str(target_dir), "cached": False}
|