ltcai 11.2.0 → 11.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -53
- package/docs/CHANGELOG.md +61 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/MULTI_AGENT_RUNTIME.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +6 -2
- package/docs/PERMISSION_MODE.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +2 -2
- package/docs/v11.3.0_PLAN.md +202 -0
- package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +287 -0
- package/lattice_brain/graph/_kg_common/extraction.py +516 -0
- package/lattice_brain/graph/_kg_common/relations.py +161 -0
- package/lattice_brain/graph/_kg_common/text.py +479 -0
- package/lattice_brain/graph/discovery_index/__init__.py +35 -0
- package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
- package/lattice_brain/graph/discovery_index/extract.py +137 -0
- package/lattice_brain/graph/discovery_index/scan.py +411 -0
- package/lattice_brain/graph/discovery_index/upsert.py +495 -0
- package/lattice_brain/graph/projection/__init__.py +42 -0
- package/lattice_brain/graph/projection/curation.py +500 -0
- package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
- package/lattice_brain/graph/retrieval/__init__.py +54 -0
- package/lattice_brain/graph/retrieval/context.py +197 -0
- package/lattice_brain/graph/retrieval/graph_view.py +319 -0
- package/lattice_brain/graph/retrieval/hybrid.py +488 -0
- package/lattice_brain/graph/retrieval/maintenance.py +121 -0
- package/lattice_brain/graph/retrieval/signals.py +95 -0
- package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
- package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
- package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
- package/lattice_brain/graph/retrieval_vector/search.py +560 -0
- package/lattice_brain/graph/retrieval_vector/status.py +374 -0
- package/lattice_brain/ingestion/__init__.py +130 -0
- package/lattice_brain/ingestion/_contract.py +90 -0
- package/lattice_brain/ingestion/constants.py +127 -0
- package/lattice_brain/ingestion/folder_scan.py +57 -0
- package/lattice_brain/ingestion/folders.py +258 -0
- package/lattice_brain/ingestion/hashing.py +26 -0
- package/lattice_brain/ingestion/jobs_api.py +107 -0
- package/lattice_brain/ingestion/models.py +80 -0
- package/lattice_brain/ingestion/pipeline.py +486 -0
- package/lattice_brain/ingestion/quality.py +209 -0
- package/lattice_brain/ingestion/routing.py +295 -0
- package/lattice_brain/multimodal/__init__.py +164 -0
- package/lattice_brain/multimodal/audio.py +77 -0
- package/lattice_brain/multimodal/common.py +118 -0
- package/lattice_brain/multimodal/images.py +498 -0
- package/lattice_brain/multimodal/ports.py +169 -0
- package/lattice_brain/multimodal/video.py +410 -0
- package/lattice_brain/portability/__init__.py +90 -0
- package/lattice_brain/portability/_contract.py +42 -0
- package/lattice_brain/portability/backups.py +338 -0
- package/lattice_brain/portability/bundles.py +136 -0
- package/lattice_brain/portability/constants.py +93 -0
- package/lattice_brain/portability/fsops.py +138 -0
- package/lattice_brain/portability/service.py +41 -0
- package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
- package/lattice_brain/runtime/__init__.py +1 -1
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/chronicle.py +63 -0
- package/latticeai/core/agent/__init__.py +93 -0
- package/latticeai/core/agent/_contract.py +79 -0
- package/latticeai/core/agent/context.py +57 -0
- package/latticeai/core/agent/deps.py +125 -0
- package/latticeai/core/agent/execution.py +622 -0
- package/latticeai/core/agent/planning.py +145 -0
- package/latticeai/core/agent/recovery.py +157 -0
- package/latticeai/core/agent/runtime.py +210 -0
- package/latticeai/core/agent/verification.py +231 -0
- package/latticeai/core/embedding_providers/__init__.py +151 -0
- package/latticeai/core/embedding_providers/base.py +199 -0
- package/latticeai/core/embedding_providers/captions.py +162 -0
- package/latticeai/core/embedding_providers/profiles.py +126 -0
- package/latticeai/core/embedding_providers/text.py +350 -0
- package/latticeai/core/embedding_providers/vision.py +352 -0
- package/latticeai/core/file_generation/__init__.py +115 -0
- package/latticeai/core/file_generation/bundles.py +76 -0
- package/latticeai/core/file_generation/extraction.py +154 -0
- package/latticeai/core/file_generation/inference.py +235 -0
- package/latticeai/core/file_generation/orchestration.py +152 -0
- package/latticeai/core/file_generation/prompting.py +117 -0
- package/latticeai/core/file_generation/repair.py +114 -0
- package/latticeai/core/file_generation/sanitize.py +61 -0
- package/latticeai/core/file_generation/validation.py +201 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +9 -0
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/integrations/telegram_bot/__init__.py +123 -0
- package/latticeai/integrations/telegram_bot/__main__.py +17 -0
- package/latticeai/integrations/telegram_bot/config.py +86 -0
- package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
- package/latticeai/integrations/telegram_bot/flows.py +478 -0
- package/latticeai/integrations/telegram_bot/helpers.py +322 -0
- package/latticeai/integrations/telegram_bot/screens.py +394 -0
- package/latticeai/models/router/__init__.py +88 -0
- package/latticeai/models/router/_contract.py +66 -0
- package/latticeai/models/router/branding.py +56 -0
- package/latticeai/models/router/catalog.py +69 -0
- package/latticeai/models/router/documents.py +199 -0
- package/latticeai/models/router/errors.py +37 -0
- package/latticeai/models/router/generation.py +258 -0
- package/latticeai/models/router/loading.py +291 -0
- package/latticeai/models/router/local_models.py +85 -0
- package/latticeai/models/router/registry.py +147 -0
- package/latticeai/runtime/build_phases/__init__.py +82 -0
- package/latticeai/runtime/build_phases/features.py +407 -0
- package/latticeai/runtime/build_phases/foundation.py +555 -0
- package/latticeai/runtime/build_phases/web.py +492 -0
- package/latticeai/runtime/runtime_context.py +1 -0
- package/latticeai/services/architecture_readiness.py +48 -19
- package/latticeai/services/brain_intelligence/__init__.py +58 -0
- package/latticeai/services/brain_intelligence/_contract.py +71 -0
- package/latticeai/services/brain_intelligence/consistency.py +193 -0
- package/latticeai/services/brain_intelligence/constants.py +47 -0
- package/latticeai/services/brain_intelligence/digest.py +258 -0
- package/latticeai/services/brain_intelligence/health.py +331 -0
- package/latticeai/services/brain_intelligence/proposals.py +264 -0
- package/latticeai/services/brain_intelligence/sampling.py +84 -0
- package/latticeai/services/brain_intelligence/service.py +48 -0
- package/latticeai/services/chronicle.py +557 -0
- package/latticeai/services/memory_service/__init__.py +52 -0
- package/latticeai/services/memory_service/_contract.py +100 -0
- package/latticeai/services/memory_service/brief.py +431 -0
- package/latticeai/services/memory_service/constants.py +57 -0
- package/latticeai/services/memory_service/maintenance.py +138 -0
- package/latticeai/services/memory_service/manager.py +186 -0
- package/latticeai/services/memory_service/proof.py +136 -0
- package/latticeai/services/memory_service/recall.py +225 -0
- package/latticeai/services/memory_service/service.py +48 -0
- package/latticeai/services/memory_service/stores.py +110 -0
- package/latticeai/services/model_runtime/__init__.py +322 -0
- package/latticeai/services/model_runtime/cloud.py +87 -0
- package/latticeai/services/model_runtime/download.py +282 -0
- package/latticeai/services/model_runtime/engines.py +341 -0
- package/latticeai/services/model_runtime/loading.py +178 -0
- package/latticeai/services/model_runtime/service.py +129 -0
- package/latticeai/services/model_runtime/state.py +131 -0
- package/latticeai/services/model_runtime/status.py +255 -0
- package/latticeai/services/product_readiness.py +15 -7
- package/latticeai/setup/wizard/__init__.py +126 -0
- package/latticeai/setup/wizard/catalog.py +172 -0
- package/latticeai/setup/wizard/detect.py +323 -0
- package/latticeai/setup/wizard/install.py +348 -0
- package/latticeai/setup/wizard/paths.py +168 -0
- package/latticeai/setup/wizard/plans.py +74 -0
- package/latticeai/setup/wizard/recommend.py +320 -0
- package/package.json +6 -2
- package/scripts/bump_version.py +14 -0
- package/scripts/capture_release_evidence.mjs +33 -21
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_i18n_namespace_coverage.mjs +41 -4
- package/scripts/check_max_file_lines.mjs +102 -0
- package/scripts/check_release_evidence_bound.mjs +30 -15
- package/scripts/check_screenshot_pixel_delta.py +34 -4
- package/scripts/check_server_i18n.mjs +1 -0
- package/scripts/generate_rust_parity_fixtures.py +562 -0
- package/scripts/lib/mock_server_fingerprint.mjs +94 -0
- package/scripts/release_screen_claims.json +31 -2
- package/src-tauri/Cargo.lock +361 -3
- package/src-tauri/Cargo.toml +6 -1
- package/src-tauri/src/backend.rs +349 -0
- package/src-tauri/src/folder.rs +33 -0
- package/src-tauri/src/main.rs +97 -399
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +41 -37
- package/static/app/assets/Act-yYpYnn0v.js +1 -0
- package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
- package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
- package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
- package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
- package/static/app/assets/Capture-CFIRsFNE.js +1 -0
- package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
- package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
- package/static/app/assets/Library-DwO3yZST.js +1 -0
- package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
- package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
- package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
- package/static/app/assets/System-DW8F-2xL.js +1 -0
- package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
- package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
- package/static/app/assets/brain-Ci1CkWjM.js +1 -0
- package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
- package/static/app/assets/circle-check-DfInj-qD.js +1 -0
- package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
- package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
- package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
- package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
- package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
- package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
- package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
- package/static/app/assets/index-_u5iUHDr.js +10 -0
- package/static/app/assets/input-B0lPdRQZ.js +1 -0
- package/static/app/assets/link-2-CoFbooHS.js +1 -0
- package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
- package/static/app/assets/primitives-DEbN-d6p.js +1 -0
- package/static/app/assets/search-BybIWPNd.js +1 -0
- package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
- package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
- package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
- package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
- package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
- package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
- package/static/app/assets/utils-BlZr7Pd4.js +4 -0
- package/static/app/assets/workspace-jJY4RuAV.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/graph/_kg_common.py +0 -1331
- package/lattice_brain/graph/discovery_index.py +0 -1141
- package/lattice_brain/graph/retrieval.py +0 -1120
- package/lattice_brain/graph/retrieval_vector.py +0 -1293
- package/lattice_brain/ingestion.py +0 -1525
- package/lattice_brain/multimodal.py +0 -1258
- package/latticeai/core/agent.py +0 -1465
- package/latticeai/core/embedding_providers.py +0 -1196
- package/latticeai/core/file_generation.py +0 -1047
- package/latticeai/integrations/telegram_bot.py +0 -1390
- package/latticeai/models/router.py +0 -1007
- package/latticeai/runtime/build_phases.py +0 -1450
- package/latticeai/services/brain_intelligence.py +0 -1083
- package/latticeai/services/memory_service.py +0 -1177
- package/latticeai/services/model_runtime.py +0 -1281
- package/latticeai/setup/wizard.py +0 -1310
- package/static/app/assets/Act-AWf0SAKp.js +0 -1
- package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
- package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
- package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
- package/static/app/assets/Capture-CqOSzyPr.js +0 -1
- package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
- package/static/app/assets/Library-CX-bbhmK.js +0 -1
- package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
- package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
- package/static/app/assets/System-Bu2t5hn1.js +0 -1
- package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
- package/static/app/assets/brain-DJMoqrwx.js +0 -1
- package/static/app/assets/index-BpYkzcVm.js +0 -10
- package/static/app/assets/input-DSlJJxRs.js +0 -1
- package/static/app/assets/primitives-BCx6TvfG.js +0 -1
- package/static/app/assets/search-Cgy8cCFJ.js +0 -1
- package/static/app/assets/utils-zqPZJxdx.js +0 -4
- package/static/app/assets/workspace-DXTihhfU.js +0 -1
|
@@ -1,1281 +0,0 @@
|
|
|
1
|
-
"""Model runtime and provider helpers for Lattice AI.
|
|
2
|
-
|
|
3
|
-
This module owns local/cloud model preparation, engine detection, model download,
|
|
4
|
-
provider-specific server startup, smoke tests, and runtime feature payloads. It is
|
|
5
|
-
configured by ``server_app`` with app-level state but has no FastAPI app import.
|
|
6
|
-
"""
|
|
7
|
-
|
|
8
|
-
from __future__ import annotations
|
|
9
|
-
|
|
10
|
-
import asyncio
|
|
11
|
-
import importlib.util
|
|
12
|
-
import json
|
|
13
|
-
import logging
|
|
14
|
-
import os
|
|
15
|
-
import shutil
|
|
16
|
-
import time
|
|
17
|
-
import urllib.error
|
|
18
|
-
import urllib.request
|
|
19
|
-
from dataclasses import dataclass, field, fields
|
|
20
|
-
from pathlib import Path
|
|
21
|
-
from typing import Any, AsyncIterator, Callable, Dict, List, Optional
|
|
22
|
-
|
|
23
|
-
from latticeai.core.model_compat import (
|
|
24
|
-
SMOKE_PROMPT as _SMOKE_PROMPT,
|
|
25
|
-
)
|
|
26
|
-
from latticeai.core.model_compat import (
|
|
27
|
-
friendly_model_runtime_error as _friendly_model_runtime_error,
|
|
28
|
-
)
|
|
29
|
-
from latticeai.core.model_compat import (
|
|
30
|
-
model_runtime_compatibility as _model_runtime_compatibility,
|
|
31
|
-
)
|
|
32
|
-
from latticeai.core.model_resolution import ModelResolution as _ModelResolution
|
|
33
|
-
from latticeai.models.router import (
|
|
34
|
-
HF_MODELS_ROOT,
|
|
35
|
-
OPENAI_COMPATIBLE_PROVIDERS,
|
|
36
|
-
AsyncOpenAI,
|
|
37
|
-
ensure_mlx_runtime,
|
|
38
|
-
hf_cache_model_dir,
|
|
39
|
-
hf_model_dir,
|
|
40
|
-
parse_model_ref,
|
|
41
|
-
)
|
|
42
|
-
|
|
43
|
-
from .model_engines import (
|
|
44
|
-
LOCAL_SERVER_PROCESSES as _LOCAL_SERVER_PROCESSES,
|
|
45
|
-
)
|
|
46
|
-
from .model_engines import (
|
|
47
|
-
engine_install_plan as _engine_install_plan,
|
|
48
|
-
)
|
|
49
|
-
from .model_engines import (
|
|
50
|
-
engine_support_status as _engine_support_status,
|
|
51
|
-
)
|
|
52
|
-
from .model_engines import (
|
|
53
|
-
ensure_llamacpp_server as _ensure_llamacpp_server,
|
|
54
|
-
)
|
|
55
|
-
from .model_engines import (
|
|
56
|
-
ensure_lmstudio_server as _ensure_lmstudio_server,
|
|
57
|
-
)
|
|
58
|
-
from .model_engines import (
|
|
59
|
-
ensure_ollama_server as _ensure_ollama_server,
|
|
60
|
-
)
|
|
61
|
-
from .model_engines import (
|
|
62
|
-
ensure_vllm_server as _ensure_vllm_server,
|
|
63
|
-
)
|
|
64
|
-
from .model_engines import (
|
|
65
|
-
find_lmstudio_cli as _find_lmstudio_cli,
|
|
66
|
-
)
|
|
67
|
-
from .model_engines import (
|
|
68
|
-
get_ollama_pulled_models as _get_ollama_pulled_models,
|
|
69
|
-
)
|
|
70
|
-
from .model_engines import (
|
|
71
|
-
get_openai_compatible_server_models as _get_openai_compatible_server_models,
|
|
72
|
-
)
|
|
73
|
-
from .model_engines import (
|
|
74
|
-
install_engine as _install_engine,
|
|
75
|
-
)
|
|
76
|
-
from .model_engines import (
|
|
77
|
-
local_binary as _local_binary,
|
|
78
|
-
)
|
|
79
|
-
from .model_engines import (
|
|
80
|
-
pull_ollama_model_with_progress as _pull_ollama_model_with_progress,
|
|
81
|
-
)
|
|
82
|
-
from .model_engines import (
|
|
83
|
-
vllm_executable as _vllm_executable,
|
|
84
|
-
)
|
|
85
|
-
from .model_engines import (
|
|
86
|
-
vllm_metal_python as _vllm_metal_python,
|
|
87
|
-
)
|
|
88
|
-
from .model_engines import (
|
|
89
|
-
wait_for_openai_compatible_server as _wait_for_openai_compatible_server,
|
|
90
|
-
)
|
|
91
|
-
from .model_engines import (
|
|
92
|
-
windows_binary_candidates as _windows_binary_candidates,
|
|
93
|
-
)
|
|
94
|
-
from .model_errors import ModelRuntimeError
|
|
95
|
-
|
|
96
|
-
# ``model_loading._get_model_runtime_deps`` imports these private names from
|
|
97
|
-
# this module to preserve the historical model_runtime wiring surface.
|
|
98
|
-
_MODEL_LOADING_COMPAT_EXPORTS = (
|
|
99
|
-
_friendly_model_runtime_error,
|
|
100
|
-
_model_runtime_compatibility,
|
|
101
|
-
_SMOKE_PROMPT,
|
|
102
|
-
)
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
def _missing_current_user(_request: Any) -> Optional[str]:
|
|
106
|
-
return None
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
def _missing_user_api_key(_email: Optional[str], _provider: str) -> Optional[str]:
|
|
110
|
-
return None
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
@dataclass(frozen=True, slots=True)
|
|
114
|
-
class ModelRuntimeState:
|
|
115
|
-
"""Immutable application-owned dependencies for one model runtime.
|
|
116
|
-
|
|
117
|
-
Upper-case configuration field names intentionally match the long-standing
|
|
118
|
-
composition-root vocabulary. Unlike the former module ``STATE`` object,
|
|
119
|
-
instances are explicit, immutable, and safe to create more than once in a
|
|
120
|
-
process (for example in isolated tests or multiple ASGI applications).
|
|
121
|
-
"""
|
|
122
|
-
|
|
123
|
-
router: Any = None
|
|
124
|
-
APP_MODE: str = "local"
|
|
125
|
-
DEFAULT_HOST: str = "127.0.0.1"
|
|
126
|
-
DEFAULT_PORT: int = 4825
|
|
127
|
-
DATA_DIR: Path = field(default_factory=lambda: Path.home() / ".latticeai")
|
|
128
|
-
BASE_DIR: Path = field(default_factory=Path.cwd)
|
|
129
|
-
ENABLE_TELEGRAM: bool = False
|
|
130
|
-
ENABLE_GRAPH: bool = True
|
|
131
|
-
AUTOLOAD_MODELS: bool = False
|
|
132
|
-
MODEL_IDLE_UNLOAD_SECONDS: int = 0
|
|
133
|
-
ALLOW_MODEL_DOWNLOADS: bool = False
|
|
134
|
-
MODEL_DOWNLOAD_TIMEOUT: int = 300
|
|
135
|
-
ALLOW_LOCAL_MODELS: bool = True
|
|
136
|
-
REQUIRE_AUTH: bool = False
|
|
137
|
-
INVITE_GATE_ENABLED: bool = False
|
|
138
|
-
ALLOW_PLAINTEXT_API_KEYS: bool = False
|
|
139
|
-
CORS_ALLOW_NETWORK: bool = False
|
|
140
|
-
PUBLIC_MODEL: str = "openai:gpt-4o-mini"
|
|
141
|
-
LOCAL_MODEL: str = "mlx-community/gemma-4-12B-it-4bit"
|
|
142
|
-
IS_PUBLIC_MODE: bool = False
|
|
143
|
-
keyring: Any = None
|
|
144
|
-
get_current_user: Callable[[Any], Optional[str]] = _missing_current_user
|
|
145
|
-
get_user_api_key: Callable[[Optional[str], str], Optional[str]] = _missing_user_api_key
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
def create_model_runtime_state(**deps: Any) -> ModelRuntimeState:
|
|
149
|
-
"""Create an immutable runtime dependency set with strict key validation."""
|
|
150
|
-
|
|
151
|
-
known = {item.name for item in fields(ModelRuntimeState)}
|
|
152
|
-
unknown = sorted(set(deps) - known)
|
|
153
|
-
if unknown:
|
|
154
|
-
raise TypeError(f"unknown model runtime dependencies: {', '.join(unknown)}")
|
|
155
|
-
return ModelRuntimeState(**deps)
|
|
156
|
-
|
|
157
|
-
def _download_allowed(
|
|
158
|
-
allow_download: bool = False, *, state: ModelRuntimeState
|
|
159
|
-
) -> bool:
|
|
160
|
-
autoload = state.AUTOLOAD_MODELS
|
|
161
|
-
configured = state.ALLOW_MODEL_DOWNLOADS
|
|
162
|
-
return bool(allow_download) or bool(configured) or bool(autoload)
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
def _download_block(provider: str, model_name: str) -> None:
|
|
166
|
-
raise ModelRuntimeError(
|
|
167
|
-
status_code=409,
|
|
168
|
-
detail={
|
|
169
|
-
"status": "unavailable",
|
|
170
|
-
"capability": "model_download",
|
|
171
|
-
"provider": provider,
|
|
172
|
-
"model": model_name,
|
|
173
|
-
"reason": (
|
|
174
|
-
"Model files are not present locally. Lattice AI does not start "
|
|
175
|
-
"outbound model downloads by default, and token/model presence "
|
|
176
|
-
"alone never authorizes network activity."
|
|
177
|
-
),
|
|
178
|
-
"action": "Use the explicit pull/prepare flow with download consent, or set LATTICEAI_ALLOW_MODEL_DOWNLOADS=true.",
|
|
179
|
-
},
|
|
180
|
-
)
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
def _engine_install_block(engine: str) -> None:
|
|
184
|
-
raise ModelRuntimeError(
|
|
185
|
-
status_code=409,
|
|
186
|
-
detail={
|
|
187
|
-
"status": "unavailable",
|
|
188
|
-
"capability": "engine_install",
|
|
189
|
-
"engine": engine,
|
|
190
|
-
"reason": (
|
|
191
|
-
"The requested local runtime is not installed. Lattice AI does not "
|
|
192
|
-
"run package-manager or installer commands from Model Load by default."
|
|
193
|
-
),
|
|
194
|
-
"action": "Install the runtime explicitly from Library/System setup, or enable explicit download/install consent for this request.",
|
|
195
|
-
},
|
|
196
|
-
)
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
def configure_model_runtime(**deps: Any) -> "ModelRuntimeService":
|
|
200
|
-
"""Compatibility factory returning an isolated, bound runtime service.
|
|
201
|
-
|
|
202
|
-
The historical function mutated process-wide module globals. Keeping the
|
|
203
|
-
import path while returning a service preserves practical construction
|
|
204
|
-
compatibility without ambient state or cross-application leakage.
|
|
205
|
-
"""
|
|
206
|
-
|
|
207
|
-
return ModelRuntimeService(create_model_runtime_state(**deps))
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
# Catalog data + version-dedup helpers live in ``model_catalog``; re-exported
|
|
211
|
-
# here so existing ``from ...model_runtime import ENGINE_MODEL_CATALOG`` imports
|
|
212
|
-
# keep working.
|
|
213
|
-
from latticeai.core.quiet import ( # noqa: E402 — re-export placed after the globals it documents
|
|
214
|
-
quiet, # noqa: E402 — re-export placed after the globals it documents
|
|
215
|
-
)
|
|
216
|
-
from latticeai.services.model_catalog import ( # noqa: E402, F401 (re-export after the module globals it documents)
|
|
217
|
-
_VERSIONED_MODEL_PATTERNS,
|
|
218
|
-
ENGINE_INSTALLERS,
|
|
219
|
-
ENGINE_MODEL_CATALOG,
|
|
220
|
-
MODEL_ENGINE_ALIASES,
|
|
221
|
-
_model_family_version,
|
|
222
|
-
_version_tuple,
|
|
223
|
-
filter_lower_family_versions,
|
|
224
|
-
)
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
def _update_env_file(env_file: Path, key: str, value: str) -> None:
|
|
228
|
-
lines = []
|
|
229
|
-
found = False
|
|
230
|
-
if env_file.exists():
|
|
231
|
-
for line in env_file.read_text(encoding="utf-8").splitlines():
|
|
232
|
-
if line.startswith(f"{key}="):
|
|
233
|
-
lines.append(f"{key}={value}")
|
|
234
|
-
found = True
|
|
235
|
-
else:
|
|
236
|
-
lines.append(line)
|
|
237
|
-
if not found:
|
|
238
|
-
lines.append(f"{key}={value}")
|
|
239
|
-
env_file.write_text("\n".join(lines) + "\n", encoding="utf-8")
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
LOCAL_SERVER_PROCESSES = _LOCAL_SERVER_PROCESSES
|
|
243
|
-
VLLM_METAL_ENV = Path.home() / ".venv-vllm-metal"
|
|
244
|
-
VLLM_METAL_BIN = VLLM_METAL_ENV / "bin" / "vllm"
|
|
245
|
-
VLLM_METAL_PYTHON = VLLM_METAL_ENV / "bin" / "python"
|
|
246
|
-
LMSTUDIO_BUNDLED_CLI = Path("/Applications/LM Studio.app/Contents/Resources/app/.webpack/lms")
|
|
247
|
-
|
|
248
|
-
def windows_binary_candidates(binary: str) -> List[Path]:
|
|
249
|
-
return _windows_binary_candidates(binary)
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
def local_binary(binary: str) -> Optional[str]:
|
|
253
|
-
return _local_binary(binary)
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
def find_lmstudio_cli() -> Optional[str]:
|
|
257
|
-
return _find_lmstudio_cli()
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
def vllm_executable() -> Optional[str]:
|
|
261
|
-
return _vllm_executable()
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
def vllm_metal_python() -> Optional[str]:
|
|
265
|
-
return _vllm_metal_python()
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
def _json_request(
|
|
269
|
-
url: str,
|
|
270
|
-
*,
|
|
271
|
-
method: str = "GET",
|
|
272
|
-
payload: Optional[Dict[str, Any]] = None,
|
|
273
|
-
headers: Optional[Dict[str, str]] = None,
|
|
274
|
-
timeout: float = 10.0,
|
|
275
|
-
) -> Dict[str, Any]:
|
|
276
|
-
data = None
|
|
277
|
-
req_headers = dict(headers or {})
|
|
278
|
-
if payload is not None:
|
|
279
|
-
data = json.dumps(payload).encode("utf-8")
|
|
280
|
-
req_headers.setdefault("Content-Type", "application/json")
|
|
281
|
-
req = urllib.request.Request(url, data=data, headers=req_headers, method=method)
|
|
282
|
-
with urllib.request.urlopen(req, timeout=timeout) as res:
|
|
283
|
-
raw = res.read().decode("utf-8", errors="replace")
|
|
284
|
-
if not raw.strip():
|
|
285
|
-
return {}
|
|
286
|
-
return json.loads(raw)
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
def lmstudio_api_base() -> str:
|
|
290
|
-
return (os.getenv("LMSTUDIO_BASE_URL") or OPENAI_COMPATIBLE_PROVIDERS["lmstudio"]["base_url"]).rstrip("/")
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
def lmstudio_native_api_base() -> str:
|
|
294
|
-
base = lmstudio_api_base()
|
|
295
|
-
return base[:-3] if base.endswith("/v1") else base
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
def ensure_lmstudio_server() -> None:
|
|
299
|
-
return _ensure_lmstudio_server()
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
_LMSTUDIO_MODELS_CACHE: List[Dict[str, Any]] = []
|
|
303
|
-
_LMSTUDIO_MODELS_CACHE_TS: float = 0.0
|
|
304
|
-
_LMSTUDIO_MODELS_CACHE_TTL: float = 10.0
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
def get_lmstudio_models(*, force: bool = False) -> List[Dict[str, Any]]:
|
|
308
|
-
global _LMSTUDIO_MODELS_CACHE, _LMSTUDIO_MODELS_CACHE_TS
|
|
309
|
-
if not force and time.monotonic() - _LMSTUDIO_MODELS_CACHE_TS < _LMSTUDIO_MODELS_CACHE_TTL:
|
|
310
|
-
return _LMSTUDIO_MODELS_CACHE
|
|
311
|
-
try:
|
|
312
|
-
payload = _json_request(
|
|
313
|
-
f"{lmstudio_native_api_base()}/api/v1/models",
|
|
314
|
-
headers={"Authorization": f"Bearer {os.getenv('LMSTUDIO_API_KEY') or 'lmstudio'}"},
|
|
315
|
-
timeout=2.5,
|
|
316
|
-
)
|
|
317
|
-
except Exception:
|
|
318
|
-
return _LMSTUDIO_MODELS_CACHE
|
|
319
|
-
models = payload.get("models")
|
|
320
|
-
_LMSTUDIO_MODELS_CACHE = models if isinstance(models, list) else []
|
|
321
|
-
_LMSTUDIO_MODELS_CACHE_TS = time.monotonic()
|
|
322
|
-
return _LMSTUDIO_MODELS_CACHE
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
def _lmstudio_candidate_keys(model_name: str) -> List[str]:
|
|
326
|
-
raw = model_name.strip()
|
|
327
|
-
if not raw:
|
|
328
|
-
return []
|
|
329
|
-
slug = raw.split("/")[-1].lower()
|
|
330
|
-
slug = slug.replace("-gguf", "").replace("-awq", "")
|
|
331
|
-
parts = [p for p in slug.split("-") if p]
|
|
332
|
-
candidates = [raw.lower(), slug]
|
|
333
|
-
if parts:
|
|
334
|
-
candidates.append("-".join(parts[: min(4, len(parts))]))
|
|
335
|
-
return list(dict.fromkeys(candidates))
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
def _find_lmstudio_model_key(model_name: str, models: List[Dict[str, Any]]) -> Optional[str]:
|
|
339
|
-
if not models:
|
|
340
|
-
return None
|
|
341
|
-
candidate_keys = _lmstudio_candidate_keys(model_name)
|
|
342
|
-
exact = []
|
|
343
|
-
fuzzy = []
|
|
344
|
-
for item in models:
|
|
345
|
-
key = str(item.get("key") or "").strip()
|
|
346
|
-
display_name = str(item.get("display_name") or "").strip()
|
|
347
|
-
haystacks = [key.lower(), display_name.lower()]
|
|
348
|
-
if any(raw == key.lower() for raw in candidate_keys):
|
|
349
|
-
exact.append(key)
|
|
350
|
-
continue
|
|
351
|
-
if any(token and token in hay for token in candidate_keys for hay in haystacks):
|
|
352
|
-
fuzzy.append(key)
|
|
353
|
-
return next(iter(exact or fuzzy), None)
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
def ensure_lmstudio_model(model_name: str) -> Dict[str, Any]:
|
|
357
|
-
ensure_lmstudio_server()
|
|
358
|
-
auth_header = {"Authorization": f"Bearer {os.getenv('LMSTUDIO_API_KEY') or 'lmstudio'}"}
|
|
359
|
-
models = get_lmstudio_models()
|
|
360
|
-
found_key = _find_lmstudio_model_key(model_name, models)
|
|
361
|
-
model_key = found_key or model_name
|
|
362
|
-
|
|
363
|
-
if not found_key:
|
|
364
|
-
try:
|
|
365
|
-
job = _json_request(
|
|
366
|
-
f"{lmstudio_native_api_base()}/api/v1/models/download",
|
|
367
|
-
method="POST",
|
|
368
|
-
payload={"model": model_name},
|
|
369
|
-
headers=auth_header,
|
|
370
|
-
timeout=30,
|
|
371
|
-
)
|
|
372
|
-
except urllib.error.HTTPError as e:
|
|
373
|
-
detail = e.read().decode("utf-8", errors="replace")[-2000:]
|
|
374
|
-
raise ModelRuntimeError(status_code=500, detail=f"LM Studio 모델 다운로드 실패: {detail or e.reason}")
|
|
375
|
-
except Exception as e:
|
|
376
|
-
raise ModelRuntimeError(status_code=500, detail=f"LM Studio 모델 다운로드 실패: {e}")
|
|
377
|
-
|
|
378
|
-
status = str(job.get("status") or "")
|
|
379
|
-
job_id = str(job.get("job_id") or "")
|
|
380
|
-
if status not in {"completed", "already_downloaded"} and job_id:
|
|
381
|
-
deadline = time.time() + 3600
|
|
382
|
-
while time.time() < deadline:
|
|
383
|
-
polled = _json_request(
|
|
384
|
-
f"{lmstudio_native_api_base()}/api/v1/models/download/status/{job_id}",
|
|
385
|
-
headers=auth_header,
|
|
386
|
-
timeout=30,
|
|
387
|
-
)
|
|
388
|
-
polled_status = str(polled.get("status") or "")
|
|
389
|
-
if polled_status == "completed":
|
|
390
|
-
break
|
|
391
|
-
if polled_status == "failed":
|
|
392
|
-
raise ModelRuntimeError(status_code=500, detail=f"LM Studio 모델 다운로드 실패: {polled}")
|
|
393
|
-
time.sleep(2)
|
|
394
|
-
else:
|
|
395
|
-
raise ModelRuntimeError(status_code=408, detail="LM Studio 모델 다운로드 시간이 초과되었습니다.")
|
|
396
|
-
|
|
397
|
-
models = get_lmstudio_models(force=True)
|
|
398
|
-
model_key = _find_lmstudio_model_key(model_name, models) or model_name
|
|
399
|
-
|
|
400
|
-
target = next((item for item in models if isinstance(item, dict) and item.get("key") == model_key), None)
|
|
401
|
-
loaded_instances = target.get("loaded_instances") if isinstance(target, dict) else None
|
|
402
|
-
if loaded_instances:
|
|
403
|
-
return {"provider": "lmstudio", "model": model_name, "resolved_model": model_key, "server_ready": True, "cached": True}
|
|
404
|
-
|
|
405
|
-
try:
|
|
406
|
-
loaded = _json_request(
|
|
407
|
-
f"{lmstudio_native_api_base()}/api/v1/models/load",
|
|
408
|
-
method="POST",
|
|
409
|
-
payload={"model": model_key, "context_length": 4096},
|
|
410
|
-
headers=auth_header,
|
|
411
|
-
timeout=120,
|
|
412
|
-
)
|
|
413
|
-
except urllib.error.HTTPError as e:
|
|
414
|
-
detail = e.read().decode("utf-8", errors="replace")[-2000:]
|
|
415
|
-
raise ModelRuntimeError(status_code=500, detail=f"LM Studio 모델 로드 실패: {detail or e.reason}")
|
|
416
|
-
except Exception as e:
|
|
417
|
-
raise ModelRuntimeError(status_code=500, detail=f"LM Studio 모델 로드 실패: {e}")
|
|
418
|
-
|
|
419
|
-
if str(loaded.get("status") or "") != "loaded":
|
|
420
|
-
raise ModelRuntimeError(status_code=500, detail=f"LM Studio 모델 로드 실패: {loaded}")
|
|
421
|
-
|
|
422
|
-
return {
|
|
423
|
-
"provider": "lmstudio",
|
|
424
|
-
"model": model_name,
|
|
425
|
-
"resolved_model": model_key,
|
|
426
|
-
"instance_id": loaded.get("instance_id"),
|
|
427
|
-
"server_ready": True,
|
|
428
|
-
"cached": False,
|
|
429
|
-
}
|
|
430
|
-
|
|
431
|
-
def engine_support_status(engine: str) -> Dict[str, Any]:
|
|
432
|
-
return _engine_support_status(engine)
|
|
433
|
-
|
|
434
|
-
def hf_model_ready(repo_id: str, provider: str = "local_mlx") -> bool:
|
|
435
|
-
model_dir = hf_model_dir(repo_id)
|
|
436
|
-
if provider in {"local_mlx", "vllm"} and (not model_dir.exists() or not model_dir.is_dir()):
|
|
437
|
-
hf_cache_repo = Path.home() / ".cache" / "huggingface" / "hub" / f"models--{repo_id.replace('/', '--')}"
|
|
438
|
-
if hf_cache_repo.exists() and any(hf_cache_repo.glob("snapshots/*")):
|
|
439
|
-
if provider == "vllm":
|
|
440
|
-
return True
|
|
441
|
-
return hf_cache_model_dir(repo_id) is not None
|
|
442
|
-
return False
|
|
443
|
-
if not model_dir.exists() or not model_dir.is_dir():
|
|
444
|
-
return False
|
|
445
|
-
if provider == "llamacpp":
|
|
446
|
-
return any(model_dir.rglob("*.gguf"))
|
|
447
|
-
has_config = (model_dir / "config.json").exists()
|
|
448
|
-
has_weights = any(model_dir.glob("*.safetensors")) or any(model_dir.glob("*.bin"))
|
|
449
|
-
has_tokenizer = (
|
|
450
|
-
(model_dir / "tokenizer.json").exists()
|
|
451
|
-
or (model_dir / "tokenizer.model").exists()
|
|
452
|
-
or (model_dir / "tokenizer_config.json").exists()
|
|
453
|
-
)
|
|
454
|
-
return has_config and has_weights and has_tokenizer
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
def model_download_progress_payload(
|
|
458
|
-
stage: str,
|
|
459
|
-
message: str,
|
|
460
|
-
*,
|
|
461
|
-
percent: Optional[float] = None,
|
|
462
|
-
detail: Optional[str] = None,
|
|
463
|
-
downloaded_bytes: Optional[int] = None,
|
|
464
|
-
total_bytes: Optional[int] = None,
|
|
465
|
-
eta_seconds: Optional[float] = None,
|
|
466
|
-
file: Optional[str] = None,
|
|
467
|
-
indeterminate: bool = False,
|
|
468
|
-
) -> Dict[str, Any]:
|
|
469
|
-
payload: Dict[str, Any] = {
|
|
470
|
-
"stage": stage,
|
|
471
|
-
"message": message,
|
|
472
|
-
"indeterminate": indeterminate,
|
|
473
|
-
"ts": time.time(),
|
|
474
|
-
}
|
|
475
|
-
if percent is not None:
|
|
476
|
-
payload["percent"] = max(0, min(100, round(float(percent), 1)))
|
|
477
|
-
if detail:
|
|
478
|
-
payload["detail"] = detail
|
|
479
|
-
if downloaded_bytes is not None:
|
|
480
|
-
payload["downloaded_bytes"] = max(0, int(downloaded_bytes))
|
|
481
|
-
if total_bytes is not None:
|
|
482
|
-
payload["total_bytes"] = max(0, int(total_bytes))
|
|
483
|
-
if eta_seconds is not None:
|
|
484
|
-
payload["eta_seconds"] = max(0, round(float(eta_seconds)))
|
|
485
|
-
if file:
|
|
486
|
-
payload["file"] = file
|
|
487
|
-
return payload
|
|
488
|
-
|
|
489
|
-
|
|
490
|
-
def estimate_eta_seconds(started_at: float, percent: Optional[float]) -> Optional[float]:
|
|
491
|
-
if percent is None or percent <= 0 or percent >= 100:
|
|
492
|
-
return None
|
|
493
|
-
elapsed = max(0.0, time.time() - started_at)
|
|
494
|
-
return elapsed * (100.0 - percent) / percent
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
def hf_repo_files_with_sizes(repo_id: str) -> List[Dict[str, Any]]:
|
|
498
|
-
from huggingface_hub import HfApi
|
|
499
|
-
|
|
500
|
-
api = HfApi()
|
|
501
|
-
try:
|
|
502
|
-
info = api.model_info(repo_id, files_metadata=True)
|
|
503
|
-
files = []
|
|
504
|
-
for sibling in getattr(info, "siblings", []) or []:
|
|
505
|
-
name = str(getattr(sibling, "rfilename", "") or "").strip()
|
|
506
|
-
if not name or name.endswith("/"):
|
|
507
|
-
continue
|
|
508
|
-
files.append({"name": name, "size": int(getattr(sibling, "size", 0) or 0)})
|
|
509
|
-
if files:
|
|
510
|
-
return files
|
|
511
|
-
except TypeError:
|
|
512
|
-
quiet()
|
|
513
|
-
except Exception as e:
|
|
514
|
-
logging.warning("huggingface model_info failed for %s: %s", repo_id, e)
|
|
515
|
-
|
|
516
|
-
return [{"name": str(name), "size": 0} for name in api.list_repo_files(repo_id) if str(name).strip()]
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
def download_hf_model(
|
|
520
|
-
repo_id: str,
|
|
521
|
-
provider: str = "local_mlx",
|
|
522
|
-
progress_emit=None,
|
|
523
|
-
) -> Dict[str, Any]:
|
|
524
|
-
if importlib.util.find_spec("huggingface_hub") is None:
|
|
525
|
-
raise ModelRuntimeError(status_code=400, detail="huggingface_hub가 없습니다. 먼저 MLX runtime 설치를 진행해 주세요.")
|
|
526
|
-
|
|
527
|
-
target_dir = hf_model_dir(repo_id)
|
|
528
|
-
if hf_model_ready(repo_id, provider):
|
|
529
|
-
cached_dir = hf_cache_model_dir(repo_id) if provider == "local_mlx" else None
|
|
530
|
-
resolved_dir = cached_dir or target_dir
|
|
531
|
-
if progress_emit:
|
|
532
|
-
progress_emit(model_download_progress_payload(
|
|
533
|
-
"download",
|
|
534
|
-
"이미 다운로드된 모델을 확인했습니다.",
|
|
535
|
-
percent=100,
|
|
536
|
-
downloaded_bytes=0,
|
|
537
|
-
total_bytes=0,
|
|
538
|
-
eta_seconds=0,
|
|
539
|
-
))
|
|
540
|
-
return {"model": repo_id, "path": str(resolved_dir), "cached": True}
|
|
541
|
-
|
|
542
|
-
target_dir.mkdir(parents=True, exist_ok=True)
|
|
543
|
-
try:
|
|
544
|
-
from huggingface_hub import hf_hub_download
|
|
545
|
-
|
|
546
|
-
started_at = time.time()
|
|
547
|
-
all_files = hf_repo_files_with_sizes(repo_id)
|
|
548
|
-
if provider == "llamacpp":
|
|
549
|
-
ggufs = sorted(
|
|
550
|
-
[item for item in all_files if str(item["name"]).lower().endswith(".gguf")],
|
|
551
|
-
key=lambda item: str(item["name"]),
|
|
552
|
-
)
|
|
553
|
-
if not ggufs:
|
|
554
|
-
raise RuntimeError("GGUF 파일을 찾지 못했습니다.")
|
|
555
|
-
preference = ("q4_k_m", "q4_0", "q4_k_s", "q3_k_m", "q2_k")
|
|
556
|
-
selected_files = [
|
|
557
|
-
next(
|
|
558
|
-
(item for pref in preference for item in ggufs if pref in str(item["name"]).lower()),
|
|
559
|
-
ggufs[0],
|
|
560
|
-
)
|
|
561
|
-
]
|
|
562
|
-
else:
|
|
563
|
-
selected_files = all_files
|
|
564
|
-
|
|
565
|
-
total_bytes = sum(int(item.get("size") or 0) for item in selected_files) or None
|
|
566
|
-
downloaded_bytes = 0
|
|
567
|
-
total_files = max(1, len(selected_files))
|
|
568
|
-
if progress_emit:
|
|
569
|
-
progress_emit(model_download_progress_payload(
|
|
570
|
-
"download",
|
|
571
|
-
"모델 파일 정보를 확인했습니다.",
|
|
572
|
-
percent=0,
|
|
573
|
-
downloaded_bytes=0,
|
|
574
|
-
total_bytes=total_bytes,
|
|
575
|
-
indeterminate=total_bytes is None,
|
|
576
|
-
))
|
|
577
|
-
|
|
578
|
-
for index, item in enumerate(selected_files, start=1):
|
|
579
|
-
filename = str(item["name"])
|
|
580
|
-
size = int(item.get("size") or 0)
|
|
581
|
-
tqdm_class = None
|
|
582
|
-
if progress_emit:
|
|
583
|
-
current_percent = (
|
|
584
|
-
(downloaded_bytes / total_bytes) * 100 if total_bytes else ((index - 1) / total_files) * 100
|
|
585
|
-
)
|
|
586
|
-
progress_emit(model_download_progress_payload(
|
|
587
|
-
"download",
|
|
588
|
-
"모델 다운로드 중입니다.",
|
|
589
|
-
percent=current_percent,
|
|
590
|
-
detail=filename,
|
|
591
|
-
downloaded_bytes=downloaded_bytes,
|
|
592
|
-
total_bytes=total_bytes,
|
|
593
|
-
eta_seconds=estimate_eta_seconds(started_at, current_percent),
|
|
594
|
-
file=filename,
|
|
595
|
-
indeterminate=total_bytes is None and total_files <= 1,
|
|
596
|
-
))
|
|
597
|
-
try:
|
|
598
|
-
from tqdm.auto import tqdm as base_tqdm
|
|
599
|
-
|
|
600
|
-
downloaded_before = downloaded_bytes
|
|
601
|
-
last_emit = {"at": 0.0, "percent": -1.0}
|
|
602
|
-
|
|
603
|
-
def emit_byte_progress(
|
|
604
|
-
done_bytes: float,
|
|
605
|
-
# Bound per iteration: this callback outlives the loop
|
|
606
|
-
# body when a download runs long, and late binding
|
|
607
|
-
# would report every file's progress against the last
|
|
608
|
-
# file's offsets.
|
|
609
|
-
downloaded_before: int = downloaded_before,
|
|
610
|
-
size: Any = size,
|
|
611
|
-
index: int = index,
|
|
612
|
-
last_emit: dict = last_emit,
|
|
613
|
-
filename: str = filename,
|
|
614
|
-
) -> None:
|
|
615
|
-
done = max(0, int(done_bytes or 0))
|
|
616
|
-
if total_bytes:
|
|
617
|
-
aggregate = min(total_bytes, downloaded_before + done)
|
|
618
|
-
percent = (aggregate / total_bytes) * 100
|
|
619
|
-
else:
|
|
620
|
-
file_total = size or done
|
|
621
|
-
file_ratio = min(1.0, done / file_total) if file_total else 0.0
|
|
622
|
-
aggregate = downloaded_before + done
|
|
623
|
-
percent = ((index - 1) + file_ratio) / total_files * 100
|
|
624
|
-
now = time.time()
|
|
625
|
-
if percent < 100 and now - last_emit["at"] < 0.5 and percent - last_emit["percent"] < 0.3:
|
|
626
|
-
return
|
|
627
|
-
last_emit["at"] = now
|
|
628
|
-
last_emit["percent"] = percent
|
|
629
|
-
progress_emit(model_download_progress_payload(
|
|
630
|
-
"download",
|
|
631
|
-
"모델 다운로드 중입니다.",
|
|
632
|
-
percent=percent,
|
|
633
|
-
detail=filename,
|
|
634
|
-
downloaded_bytes=aggregate,
|
|
635
|
-
total_bytes=total_bytes,
|
|
636
|
-
eta_seconds=estimate_eta_seconds(started_at, percent),
|
|
637
|
-
file=filename,
|
|
638
|
-
indeterminate=total_bytes is None and total_files <= 1,
|
|
639
|
-
))
|
|
640
|
-
|
|
641
|
-
class ProgressTqdm(base_tqdm):
|
|
642
|
-
def update(self, n=1):
|
|
643
|
-
result = super().update(n)
|
|
644
|
-
emit_byte_progress(float(getattr(self, "n", 0) or 0))
|
|
645
|
-
return result
|
|
646
|
-
|
|
647
|
-
tqdm_class = ProgressTqdm
|
|
648
|
-
except Exception:
|
|
649
|
-
tqdm_class = None
|
|
650
|
-
local_path = hf_hub_download(
|
|
651
|
-
repo_id=repo_id,
|
|
652
|
-
filename=filename,
|
|
653
|
-
local_dir=str(target_dir),
|
|
654
|
-
tqdm_class=tqdm_class,
|
|
655
|
-
)
|
|
656
|
-
if size <= 0:
|
|
657
|
-
try:
|
|
658
|
-
size = Path(local_path).stat().st_size
|
|
659
|
-
except OSError:
|
|
660
|
-
size = 0
|
|
661
|
-
downloaded_bytes += size
|
|
662
|
-
if progress_emit:
|
|
663
|
-
current_percent = (
|
|
664
|
-
(downloaded_bytes / total_bytes) * 100 if total_bytes else (index / total_files) * 100
|
|
665
|
-
)
|
|
666
|
-
progress_emit(model_download_progress_payload(
|
|
667
|
-
"download",
|
|
668
|
-
"모델 다운로드 중입니다.",
|
|
669
|
-
percent=current_percent,
|
|
670
|
-
detail=filename,
|
|
671
|
-
downloaded_bytes=downloaded_bytes,
|
|
672
|
-
total_bytes=total_bytes,
|
|
673
|
-
eta_seconds=estimate_eta_seconds(started_at, current_percent),
|
|
674
|
-
file=filename,
|
|
675
|
-
indeterminate=False,
|
|
676
|
-
))
|
|
677
|
-
|
|
678
|
-
if progress_emit:
|
|
679
|
-
progress_emit(model_download_progress_payload(
|
|
680
|
-
"download",
|
|
681
|
-
"모델 다운로드가 완료되었습니다.",
|
|
682
|
-
percent=100,
|
|
683
|
-
downloaded_bytes=downloaded_bytes,
|
|
684
|
-
total_bytes=total_bytes or downloaded_bytes,
|
|
685
|
-
eta_seconds=0,
|
|
686
|
-
))
|
|
687
|
-
except Exception as e:
|
|
688
|
-
raise ModelRuntimeError(status_code=500, detail=f"{repo_id} 다운로드 실패: {str(e)[-2000:]}")
|
|
689
|
-
|
|
690
|
-
if not hf_model_ready(repo_id, provider):
|
|
691
|
-
raise ModelRuntimeError(status_code=500, detail=f"{repo_id} 다운로드가 완료되지 않았습니다. 모델 파일을 찾지 못했습니다.")
|
|
692
|
-
|
|
693
|
-
return {"model": repo_id, "path": str(target_dir), "cached": False}
|
|
694
|
-
|
|
695
|
-
|
|
696
|
-
def pull_ollama_model_with_progress(model_name: str, progress_emit=None) -> Dict[str, Any]:
|
|
697
|
-
return _pull_ollama_model_with_progress(model_name, progress_emit)
|
|
698
|
-
|
|
699
|
-
|
|
700
|
-
def get_ollama_pulled_models() -> set:
|
|
701
|
-
return _get_ollama_pulled_models()
|
|
702
|
-
|
|
703
|
-
|
|
704
|
-
def get_openai_compatible_server_models(provider: str) -> List[str]:
|
|
705
|
-
return _get_openai_compatible_server_models(provider)
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
def ensure_ollama_server() -> None:
|
|
709
|
-
return _ensure_ollama_server()
|
|
710
|
-
|
|
711
|
-
|
|
712
|
-
def wait_for_openai_compatible_server(provider: str, model_name: Optional[str] = None, timeout: int = 45) -> bool:
|
|
713
|
-
return _wait_for_openai_compatible_server(provider, model_name=model_name, timeout=timeout)
|
|
714
|
-
|
|
715
|
-
|
|
716
|
-
def ensure_vllm_server(model_name: str) -> None:
|
|
717
|
-
return _ensure_vllm_server(model_name)
|
|
718
|
-
|
|
719
|
-
|
|
720
|
-
def ensure_llamacpp_server(model_name: str) -> None:
|
|
721
|
-
return _ensure_llamacpp_server(model_name)
|
|
722
|
-
|
|
723
|
-
|
|
724
|
-
def _safe_engine_install_plan(
|
|
725
|
-
engine: str,
|
|
726
|
-
*,
|
|
727
|
-
base_dir: Path,
|
|
728
|
-
) -> Optional[Dict[str, Any]]:
|
|
729
|
-
try:
|
|
730
|
-
return _engine_install_plan(engine, base_dir=base_dir)
|
|
731
|
-
except Exception:
|
|
732
|
-
return None
|
|
733
|
-
|
|
734
|
-
|
|
735
|
-
def engine_installed(engine: str) -> bool:
|
|
736
|
-
if engine == "local_mlx":
|
|
737
|
-
return bool(
|
|
738
|
-
importlib.util.find_spec("mlx")
|
|
739
|
-
and (importlib.util.find_spec("mlx_vlm") or importlib.util.find_spec("mlx_lm"))
|
|
740
|
-
)
|
|
741
|
-
if engine == "ollama":
|
|
742
|
-
return local_binary("ollama") is not None
|
|
743
|
-
if engine == "vllm":
|
|
744
|
-
return vllm_metal_python() is not None or vllm_executable() is not None or importlib.util.find_spec("vllm") is not None
|
|
745
|
-
if engine == "lmstudio":
|
|
746
|
-
return find_lmstudio_cli() is not None or Path("/Applications/LM Studio.app").exists()
|
|
747
|
-
if engine == "llamacpp":
|
|
748
|
-
return shutil.which("llama-server") is not None
|
|
749
|
-
if engine in {"openai", "openrouter", "groq", "together", "xai"}:
|
|
750
|
-
return AsyncOpenAI is not None
|
|
751
|
-
return False
|
|
752
|
-
|
|
753
|
-
def engine_status(
|
|
754
|
-
*,
|
|
755
|
-
state: ModelRuntimeState,
|
|
756
|
-
cloud_verify_cache: Optional[Dict[str, Dict[str, Any]]] = None,
|
|
757
|
-
) -> List[Dict]:
|
|
758
|
-
r = state.router
|
|
759
|
-
verify_cache = cloud_verify_cache or {}
|
|
760
|
-
cloud_models = r.detected_cloud_models() if r else []
|
|
761
|
-
cloud_by_provider: Dict[str, List[Dict[str, Any]]] = {}
|
|
762
|
-
for model in cloud_models:
|
|
763
|
-
cloud_by_provider.setdefault(model["provider"], []).append(model)
|
|
764
|
-
|
|
765
|
-
ollama_installed = engine_installed("ollama")
|
|
766
|
-
pulled = get_ollama_pulled_models() if ollama_installed else set()
|
|
767
|
-
ollama_models = []
|
|
768
|
-
for m in ENGINE_MODEL_CATALOG["ollama"]:
|
|
769
|
-
pull_name = m["id"].removeprefix("ollama:")
|
|
770
|
-
ollama_models.append({**m, "pulled": pull_name in pulled})
|
|
771
|
-
ollama_models = filter_lower_family_versions(ollama_models)
|
|
772
|
-
|
|
773
|
-
HF_MODELS_ROOT.mkdir(parents=True, exist_ok=True)
|
|
774
|
-
mlx_models = []
|
|
775
|
-
for m in ENGINE_MODEL_CATALOG.get("local_mlx", []):
|
|
776
|
-
repo_id = m["id"]
|
|
777
|
-
mlx_models.append({**m, "pulled": hf_model_ready(repo_id, "local_mlx")})
|
|
778
|
-
mlx_models = filter_lower_family_versions(mlx_models)
|
|
779
|
-
|
|
780
|
-
vllm_models = []
|
|
781
|
-
for m in ENGINE_MODEL_CATALOG.get("vllm", []):
|
|
782
|
-
repo_id = m["id"].removeprefix("vllm:")
|
|
783
|
-
vllm_models.append({**m, "pulled": hf_model_ready(repo_id, "vllm")})
|
|
784
|
-
vllm_models = filter_lower_family_versions(vllm_models)
|
|
785
|
-
|
|
786
|
-
lmstudio_models = []
|
|
787
|
-
downloaded_lmstudio = get_lmstudio_models()
|
|
788
|
-
downloaded_by_key: Dict[str, Dict[str, Any]] = {}
|
|
789
|
-
for item in downloaded_lmstudio:
|
|
790
|
-
key = str(item.get("key") or "").strip()
|
|
791
|
-
if not key:
|
|
792
|
-
continue
|
|
793
|
-
downloaded_by_key[key] = item
|
|
794
|
-
loaded_instances = item.get("loaded_instances") or []
|
|
795
|
-
lmstudio_models.append({
|
|
796
|
-
"id": f"lmstudio:{key}",
|
|
797
|
-
"name": item.get("display_name") or f"LM Studio · {key}",
|
|
798
|
-
"family": item.get("architecture") or item.get("publisher") or "LM Studio",
|
|
799
|
-
"tag": "loaded-server-model" if loaded_instances else "downloaded",
|
|
800
|
-
"size": item.get("params_string") or item.get("format") or "LM Studio",
|
|
801
|
-
"pullable": True,
|
|
802
|
-
"pulled": True,
|
|
803
|
-
})
|
|
804
|
-
|
|
805
|
-
if not lmstudio_models:
|
|
806
|
-
for m in ENGINE_MODEL_CATALOG.get("lmstudio", []):
|
|
807
|
-
lmstudio_models.append({**m, "pulled": False})
|
|
808
|
-
else:
|
|
809
|
-
known_ids = {item["id"] for item in lmstudio_models}
|
|
810
|
-
for m in ENGINE_MODEL_CATALOG.get("lmstudio", []):
|
|
811
|
-
repo_id = m["id"].removeprefix("lmstudio:")
|
|
812
|
-
if f"lmstudio:{repo_id}" not in known_ids and repo_id not in downloaded_by_key:
|
|
813
|
-
lmstudio_models.append({**m, "pulled": False})
|
|
814
|
-
lmstudio_models = filter_lower_family_versions(lmstudio_models)
|
|
815
|
-
|
|
816
|
-
llamacpp_models = []
|
|
817
|
-
for m in ENGINE_MODEL_CATALOG.get("llamacpp", []):
|
|
818
|
-
repo_id = m["id"].removeprefix("llamacpp:")
|
|
819
|
-
llamacpp_models.append({**m, "pulled": hf_model_ready(repo_id, "llamacpp")})
|
|
820
|
-
llamacpp_models = filter_lower_family_versions(llamacpp_models)
|
|
821
|
-
|
|
822
|
-
local_server_specs: List[Dict[str, Any]] = [
|
|
823
|
-
{
|
|
824
|
-
"id": "vllm",
|
|
825
|
-
"name": "vLLM",
|
|
826
|
-
"description": "vLLM OpenAI 호환 서버(예: http://localhost:8000/v1)에 연결합니다.",
|
|
827
|
-
"requires": "VLLM_BASE_URL",
|
|
828
|
-
"note": engine_support_status("vllm").get("reason"),
|
|
829
|
-
},
|
|
830
|
-
{
|
|
831
|
-
"id": "lmstudio",
|
|
832
|
-
"name": "LM Studio",
|
|
833
|
-
"description": "LM Studio 로컬 OpenAI 호환 서버에 연결합니다.",
|
|
834
|
-
"requires": "LMSTUDIO_BASE_URL",
|
|
835
|
-
"note": (
|
|
836
|
-
"다운로드된 모델은 자동 감지하고, 선택 시 필요하면 다운로드 후 바로 로드합니다."
|
|
837
|
-
if downloaded_lmstudio else
|
|
838
|
-
"LM Studio 설치 후 모델을 선택하면 Local Server 시작, 다운로드, 로드를 자동으로 진행합니다."
|
|
839
|
-
),
|
|
840
|
-
"server_ready": bool(downloaded_lmstudio),
|
|
841
|
-
},
|
|
842
|
-
{
|
|
843
|
-
"id": "llamacpp",
|
|
844
|
-
"name": "llama.cpp",
|
|
845
|
-
"description": "llama.cpp 서버(OpenAI 호환 /v1)에 연결합니다.",
|
|
846
|
-
"requires": "LLAMACPP_BASE_URL",
|
|
847
|
-
},
|
|
848
|
-
]
|
|
849
|
-
|
|
850
|
-
engines = [
|
|
851
|
-
{
|
|
852
|
-
"id": "local_mlx",
|
|
853
|
-
"name": "MLX",
|
|
854
|
-
"kind": "local",
|
|
855
|
-
"description": "Apple Silicon GPU에서 MLX-VLM 모델을 직접 실행하고, Gemma 4는 필요 시 MLX-LM 텍스트 경로로 재시도합니다.",
|
|
856
|
-
"installed": engine_installed("local_mlx"),
|
|
857
|
-
"installable": True,
|
|
858
|
-
"install_label": ENGINE_INSTALLERS["local_mlx"]["label"],
|
|
859
|
-
"install_plan": _safe_engine_install_plan("local_mlx", base_dir=state.BASE_DIR),
|
|
860
|
-
"models": mlx_models,
|
|
861
|
-
},
|
|
862
|
-
{
|
|
863
|
-
"id": "ollama",
|
|
864
|
-
"name": "Ollama",
|
|
865
|
-
"kind": "local-server",
|
|
866
|
-
"description": "Ollama 로컬 서버를 OpenAI 호환 엔진처럼 사용합니다.",
|
|
867
|
-
"installed": ollama_installed,
|
|
868
|
-
"installable": True,
|
|
869
|
-
"install_label": ENGINE_INSTALLERS["ollama"]["label"],
|
|
870
|
-
"install_plan": _safe_engine_install_plan("ollama", base_dir=state.BASE_DIR),
|
|
871
|
-
"models": ollama_models,
|
|
872
|
-
},
|
|
873
|
-
]
|
|
874
|
-
for spec in local_server_specs:
|
|
875
|
-
support = engine_support_status(spec["id"])
|
|
876
|
-
engines.append({
|
|
877
|
-
"id": spec["id"],
|
|
878
|
-
"name": spec["name"],
|
|
879
|
-
"kind": "local-server",
|
|
880
|
-
"description": spec["description"],
|
|
881
|
-
"installed": engine_installed(spec["id"]),
|
|
882
|
-
"supported": support["supported"],
|
|
883
|
-
"support_reason": support["reason"],
|
|
884
|
-
"installable": support["supported"] and spec["id"] in ENGINE_INSTALLERS,
|
|
885
|
-
"install_label": ENGINE_INSTALLERS.get(spec["id"], {}).get("label"),
|
|
886
|
-
"install_plan": (
|
|
887
|
-
_safe_engine_install_plan(spec["id"], base_dir=state.BASE_DIR)
|
|
888
|
-
if spec["id"] in ENGINE_INSTALLERS
|
|
889
|
-
else None
|
|
890
|
-
),
|
|
891
|
-
"requires": spec["requires"],
|
|
892
|
-
"models": (
|
|
893
|
-
vllm_models if spec["id"] == "vllm"
|
|
894
|
-
else lmstudio_models if spec["id"] == "lmstudio"
|
|
895
|
-
else llamacpp_models if spec["id"] == "llamacpp"
|
|
896
|
-
else ENGINE_MODEL_CATALOG.get(spec["id"], [])
|
|
897
|
-
),
|
|
898
|
-
"note": spec.get("note") or support["reason"] or f"{spec['requires']} 설정 시 활성화됩니다.",
|
|
899
|
-
"server_ready": spec.get("server_ready"),
|
|
900
|
-
})
|
|
901
|
-
for provider in ["openai", "openrouter", "groq", "together", "xai"]:
|
|
902
|
-
env_key = next((item.get("requires") for item in cloud_by_provider.get(provider, []) if item.get("requires")), None)
|
|
903
|
-
provider_models = []
|
|
904
|
-
for model in cloud_by_provider.get(provider, []):
|
|
905
|
-
cache = verify_cache.get(str(model.get("id") or ""))
|
|
906
|
-
provider_models.append({
|
|
907
|
-
**model,
|
|
908
|
-
"verified": cache.get("ok") if cache else None,
|
|
909
|
-
"verify_reason": cache.get("reason") if cache else None,
|
|
910
|
-
})
|
|
911
|
-
engines.append({
|
|
912
|
-
"id": provider,
|
|
913
|
-
"name": provider.title(),
|
|
914
|
-
"kind": "cloud",
|
|
915
|
-
"description": "OpenAI 호환 Chat Completions API로 cloud LLM을 실행합니다.",
|
|
916
|
-
"installed": engine_installed(provider),
|
|
917
|
-
"installable": True,
|
|
918
|
-
"install_label": ENGINE_INSTALLERS[provider]["label"],
|
|
919
|
-
"install_plan": _safe_engine_install_plan(provider, base_dir=state.BASE_DIR),
|
|
920
|
-
"requires": env_key,
|
|
921
|
-
"models": provider_models,
|
|
922
|
-
})
|
|
923
|
-
return engines
|
|
924
|
-
|
|
925
|
-
def runtime_features(*, state: ModelRuntimeState) -> Dict:
|
|
926
|
-
s = state
|
|
927
|
-
r = s.router
|
|
928
|
-
return {
|
|
929
|
-
"mode": s.APP_MODE,
|
|
930
|
-
"public": s.IS_PUBLIC_MODE,
|
|
931
|
-
"host": s.DEFAULT_HOST,
|
|
932
|
-
"port": s.DEFAULT_PORT,
|
|
933
|
-
"data_dir": str(s.DATA_DIR),
|
|
934
|
-
"telegram_enabled": s.ENABLE_TELEGRAM,
|
|
935
|
-
"graph_enabled": s.ENABLE_GRAPH,
|
|
936
|
-
"autoload_models": s.AUTOLOAD_MODELS,
|
|
937
|
-
"model_idle_unload_seconds": s.MODEL_IDLE_UNLOAD_SECONDS,
|
|
938
|
-
"allow_model_downloads": s.ALLOW_MODEL_DOWNLOADS,
|
|
939
|
-
"model_download_timeout": s.MODEL_DOWNLOAD_TIMEOUT,
|
|
940
|
-
"model_memory_policy": r.model_memory_policy() if r else None,
|
|
941
|
-
"allow_local_models": s.ALLOW_LOCAL_MODELS,
|
|
942
|
-
"security": {
|
|
943
|
-
"host": s.DEFAULT_HOST,
|
|
944
|
-
"require_auth": s.REQUIRE_AUTH,
|
|
945
|
-
"invite_gate_enabled": s.INVITE_GATE_ENABLED,
|
|
946
|
-
"keyring_available": s.keyring is not None,
|
|
947
|
-
"plaintext_api_keys_allowed": s.ALLOW_PLAINTEXT_API_KEYS,
|
|
948
|
-
"cors_allow_network": s.CORS_ALLOW_NETWORK,
|
|
949
|
-
},
|
|
950
|
-
"default_model": s.PUBLIC_MODEL if s.IS_PUBLIC_MODE else s.LOCAL_MODEL,
|
|
951
|
-
"local_only_features": {
|
|
952
|
-
"mlx": s.ALLOW_LOCAL_MODELS and not s.IS_PUBLIC_MODE,
|
|
953
|
-
"telegram_bridge": s.ENABLE_TELEGRAM,
|
|
954
|
-
"desktop_chrome_bridge": not s.IS_PUBLIC_MODE,
|
|
955
|
-
"computer_use_bridge": not s.IS_PUBLIC_MODE,
|
|
956
|
-
},
|
|
957
|
-
"public_features": {
|
|
958
|
-
"web_ui": True,
|
|
959
|
-
"openai_compatible_models": True,
|
|
960
|
-
"persistent_data_dir": str(s.DATA_DIR),
|
|
961
|
-
},
|
|
962
|
-
}
|
|
963
|
-
|
|
964
|
-
def install_engine(
|
|
965
|
-
engine: str,
|
|
966
|
-
confirmation_token: Optional[str] = None,
|
|
967
|
-
*,
|
|
968
|
-
state: ModelRuntimeState,
|
|
969
|
-
) -> Dict:
|
|
970
|
-
return _install_engine(
|
|
971
|
-
engine,
|
|
972
|
-
confirmation_token=confirmation_token,
|
|
973
|
-
base_dir=state.BASE_DIR,
|
|
974
|
-
)
|
|
975
|
-
|
|
976
|
-
|
|
977
|
-
def _resolve_model_alias(model_id: str, engine: Optional[str] = None) -> str:
|
|
978
|
-
raw = model_id.strip()
|
|
979
|
-
engine_hint = (engine or "").strip().lower()
|
|
980
|
-
provider: Optional[str] = None
|
|
981
|
-
model_name = raw
|
|
982
|
-
if ":" in raw:
|
|
983
|
-
prefix, rest = raw.split(":", 1)
|
|
984
|
-
prefix = prefix.strip().lower()
|
|
985
|
-
if prefix in {"ollama", "vllm", "lmstudio", "llamacpp", "local_mlx", "mlx"}:
|
|
986
|
-
provider = "local_mlx" if prefix in {"local_mlx", "mlx"} else prefix
|
|
987
|
-
model_name = rest.strip()
|
|
988
|
-
provider = provider or ("local_mlx" if engine_hint in {"", "local_mlx", "mlx"} else engine_hint)
|
|
989
|
-
aliases = MODEL_ENGINE_ALIASES.get(model_name.lower())
|
|
990
|
-
if not aliases:
|
|
991
|
-
return raw
|
|
992
|
-
mapped = aliases.get(provider)
|
|
993
|
-
if not mapped:
|
|
994
|
-
return raw
|
|
995
|
-
return mapped if provider == "local_mlx" else f"{provider}:{mapped}"
|
|
996
|
-
|
|
997
|
-
|
|
998
|
-
def normalize_local_model_request(model_id: str, engine: Optional[str] = None) -> str:
|
|
999
|
-
model_id = _resolve_model_alias(model_id, engine)
|
|
1000
|
-
engine = (engine or "").strip().lower()
|
|
1001
|
-
if engine in {"local_mlx", "mlx"} and model_id.startswith(("local_mlx:", "mlx:")):
|
|
1002
|
-
return model_id.split(":", 1)[1].strip()
|
|
1003
|
-
if engine and engine not in {"local_mlx", "mlx"} and ":" not in model_id:
|
|
1004
|
-
return f"{engine}:{model_id}"
|
|
1005
|
-
return model_id
|
|
1006
|
-
|
|
1007
|
-
|
|
1008
|
-
def ensure_engine_ready(engine: str, *, state: ModelRuntimeState) -> Dict[str, Any]:
|
|
1009
|
-
engine = "local_mlx" if engine == "mlx" else engine
|
|
1010
|
-
if engine not in ENGINE_INSTALLERS and engine not in OPENAI_COMPATIBLE_PROVIDERS:
|
|
1011
|
-
raise ModelRuntimeError(status_code=400, detail=f"지원하지 않는 엔진입니다: {engine}")
|
|
1012
|
-
support = engine_support_status(engine)
|
|
1013
|
-
if not support["supported"]:
|
|
1014
|
-
raise ModelRuntimeError(status_code=400, detail=str(support["reason"]))
|
|
1015
|
-
|
|
1016
|
-
if engine_installed(engine):
|
|
1017
|
-
if engine == "local_mlx":
|
|
1018
|
-
ensure_mlx_runtime()
|
|
1019
|
-
return {"engine": engine, "installed": True, "installed_now": False}
|
|
1020
|
-
|
|
1021
|
-
if engine not in ENGINE_INSTALLERS:
|
|
1022
|
-
raise ModelRuntimeError(status_code=400, detail=f"{engine} 엔진 설치 방법이 등록되어 있지 않습니다.")
|
|
1023
|
-
|
|
1024
|
-
result = install_engine(engine, state=state)
|
|
1025
|
-
if result.get("returncode") not in (0, None) or not engine_installed(engine):
|
|
1026
|
-
detail = result.get("stderr") or result.get("stdout") or f"{engine} 설치에 실패했습니다."
|
|
1027
|
-
raise ModelRuntimeError(status_code=500, detail=str(detail)[-2000:])
|
|
1028
|
-
|
|
1029
|
-
if engine == "local_mlx":
|
|
1030
|
-
ensure_mlx_runtime()
|
|
1031
|
-
return {"engine": engine, "installed": True, "installed_now": True, "install": result}
|
|
1032
|
-
|
|
1033
|
-
|
|
1034
|
-
def build_model_resolution(
|
|
1035
|
-
input_id: str,
|
|
1036
|
-
engine: Optional[str],
|
|
1037
|
-
*,
|
|
1038
|
-
user_email: Optional[str] = None,
|
|
1039
|
-
display_name: Optional[str] = None,
|
|
1040
|
-
) -> _ModelResolution:
|
|
1041
|
-
"""피드백 #1/#2 공용 ModelResolution 생성기.
|
|
1042
|
-
|
|
1043
|
-
사용자가 클릭한 input_id + engine 힌트를 받아 모든 단계가 공유할
|
|
1044
|
-
canonical identity를 만든다.
|
|
1045
|
-
"""
|
|
1046
|
-
normalized = normalize_local_model_request(input_id, engine)
|
|
1047
|
-
return _ModelResolution.from_request(
|
|
1048
|
-
normalized,
|
|
1049
|
-
engine=engine,
|
|
1050
|
-
user_email=user_email,
|
|
1051
|
-
display_name=display_name or input_id,
|
|
1052
|
-
engine_aliases=MODEL_ENGINE_ALIASES,
|
|
1053
|
-
)
|
|
1054
|
-
|
|
1055
|
-
|
|
1056
|
-
_LOCAL_SMOKE_ENGINES = {"local_mlx", "ollama", "vllm", "lmstudio", "llamacpp"}
|
|
1057
|
-
|
|
1058
|
-
|
|
1059
|
-
async def _smoke_test_loaded_model(
|
|
1060
|
-
resolution: _ModelResolution,
|
|
1061
|
-
*,
|
|
1062
|
-
api_key_override: Optional[str] = None,
|
|
1063
|
-
state: ModelRuntimeState,
|
|
1064
|
-
) -> Dict[str, Any]:
|
|
1065
|
-
# Delegated to model_engines for server decomp
|
|
1066
|
-
from .model_engines import _smoke_test_loaded_model as _impl_smoke
|
|
1067
|
-
return await _impl_smoke(
|
|
1068
|
-
resolution,
|
|
1069
|
-
api_key_override=api_key_override,
|
|
1070
|
-
model_router=state.router,
|
|
1071
|
-
)
|
|
1072
|
-
|
|
1073
|
-
|
|
1074
|
-
async def prepare_and_load_model(
|
|
1075
|
-
model_id: str,
|
|
1076
|
-
request: Any,
|
|
1077
|
-
engine: Optional[str] = None,
|
|
1078
|
-
user_email: Optional[str] = None,
|
|
1079
|
-
adapter_path: Optional[str] = None,
|
|
1080
|
-
draft_model_id: Optional[str] = None,
|
|
1081
|
-
allow_download: bool = False,
|
|
1082
|
-
*,
|
|
1083
|
-
state: ModelRuntimeState,
|
|
1084
|
-
) -> Dict[str, Any]:
|
|
1085
|
-
from .model_loading import prepare_and_load_model as _impl
|
|
1086
|
-
|
|
1087
|
-
return await _impl(
|
|
1088
|
-
model_id,
|
|
1089
|
-
request,
|
|
1090
|
-
engine=engine,
|
|
1091
|
-
user_email=user_email,
|
|
1092
|
-
adapter_path=adapter_path,
|
|
1093
|
-
draft_model_id=draft_model_id,
|
|
1094
|
-
allow_download=allow_download,
|
|
1095
|
-
runtime_state=state,
|
|
1096
|
-
)
|
|
1097
|
-
|
|
1098
|
-
|
|
1099
|
-
def sse_event(event: str, data: Dict[str, Any]) -> str:
|
|
1100
|
-
return f"event: {event}\ndata: {json.dumps(data, ensure_ascii=False)}\n\n"
|
|
1101
|
-
|
|
1102
|
-
|
|
1103
|
-
async def prepare_and_load_model_stream(
|
|
1104
|
-
model_id: str,
|
|
1105
|
-
request: Any,
|
|
1106
|
-
engine: Optional[str] = None,
|
|
1107
|
-
user_email: Optional[str] = None,
|
|
1108
|
-
allow_download: bool = False,
|
|
1109
|
-
*,
|
|
1110
|
-
state: ModelRuntimeState,
|
|
1111
|
-
) -> AsyncIterator[str]:
|
|
1112
|
-
from .model_loading import prepare_and_load_model_stream as _impl
|
|
1113
|
-
|
|
1114
|
-
async for event in _impl(
|
|
1115
|
-
model_id,
|
|
1116
|
-
request,
|
|
1117
|
-
engine=engine,
|
|
1118
|
-
user_email=user_email,
|
|
1119
|
-
allow_download=allow_download,
|
|
1120
|
-
runtime_state=state,
|
|
1121
|
-
):
|
|
1122
|
-
yield event
|
|
1123
|
-
|
|
1124
|
-
|
|
1125
|
-
CLOUD_VERIFY_TTL_SECONDS = 600
|
|
1126
|
-
|
|
1127
|
-
async def _probe_cloud_model(model_ref: str) -> Dict[str, Any]:
|
|
1128
|
-
provider, model_name = parse_model_ref(model_ref)
|
|
1129
|
-
config = OPENAI_COMPATIBLE_PROVIDERS.get(provider)
|
|
1130
|
-
if not config:
|
|
1131
|
-
return {"ok": False, "reason": f"Unsupported provider: {provider}"}
|
|
1132
|
-
|
|
1133
|
-
api_key = os.getenv(config["env_key"]) or config.get("api_key_fallback")
|
|
1134
|
-
if not api_key:
|
|
1135
|
-
return {"ok": False, "reason": f"Missing API key: {config['env_key']}"}
|
|
1136
|
-
|
|
1137
|
-
base_url = os.getenv(config.get("base_url_env", "")) if config.get("base_url_env") else None
|
|
1138
|
-
base_url = base_url or config.get("base_url")
|
|
1139
|
-
try:
|
|
1140
|
-
# base_url is passed only when configured: an explicit None is not
|
|
1141
|
-
# the same as omitting the argument.
|
|
1142
|
-
client = (
|
|
1143
|
-
AsyncOpenAI(api_key=api_key, base_url=base_url)
|
|
1144
|
-
if base_url
|
|
1145
|
-
else AsyncOpenAI(api_key=api_key)
|
|
1146
|
-
)
|
|
1147
|
-
await asyncio.wait_for(
|
|
1148
|
-
client.chat.completions.create(
|
|
1149
|
-
model=model_name,
|
|
1150
|
-
messages=[{"role": "user", "content": "ping"}],
|
|
1151
|
-
max_tokens=1,
|
|
1152
|
-
temperature=0,
|
|
1153
|
-
),
|
|
1154
|
-
timeout=15,
|
|
1155
|
-
)
|
|
1156
|
-
return {"ok": True, "reason": "ok"}
|
|
1157
|
-
except Exception as e:
|
|
1158
|
-
return {"ok": False, "reason": str(e)[:220]}
|
|
1159
|
-
|
|
1160
|
-
|
|
1161
|
-
async def verify_cloud_models(
|
|
1162
|
-
force: bool = False,
|
|
1163
|
-
provider_filter: Optional[str] = None,
|
|
1164
|
-
*,
|
|
1165
|
-
state: ModelRuntimeState,
|
|
1166
|
-
cache: Dict[str, Dict[str, Any]],
|
|
1167
|
-
) -> Dict[str, Dict]:
|
|
1168
|
-
now = time.time()
|
|
1169
|
-
r = state.router
|
|
1170
|
-
cloud_items = [item for item in (r.detected_cloud_models() if r else []) if item.get("tag") == "cloud"]
|
|
1171
|
-
if provider_filter:
|
|
1172
|
-
cloud_items = [item for item in cloud_items if item.get("provider") == provider_filter]
|
|
1173
|
-
|
|
1174
|
-
results: Dict[str, Dict] = {}
|
|
1175
|
-
for item in cloud_items:
|
|
1176
|
-
model_ref = item["id"]
|
|
1177
|
-
cached = cache.get(model_ref)
|
|
1178
|
-
if not force and cached and (now - cached.get("ts", 0) <= CLOUD_VERIFY_TTL_SECONDS):
|
|
1179
|
-
results[model_ref] = cached
|
|
1180
|
-
continue
|
|
1181
|
-
if item.get("available") is False:
|
|
1182
|
-
record = {"ok": False, "reason": item.get("requires") or "API key missing", "ts": now}
|
|
1183
|
-
cache[model_ref] = record
|
|
1184
|
-
results[model_ref] = record
|
|
1185
|
-
continue
|
|
1186
|
-
probe = await _probe_cloud_model(model_ref)
|
|
1187
|
-
record = {"ok": bool(probe.get("ok")), "reason": probe.get("reason", ""), "ts": now}
|
|
1188
|
-
cache[model_ref] = record
|
|
1189
|
-
results[model_ref] = record
|
|
1190
|
-
return results
|
|
1191
|
-
|
|
1192
|
-
|
|
1193
|
-
@dataclass(slots=True)
|
|
1194
|
-
class ModelRuntimeService:
|
|
1195
|
-
"""Bound model operations for one explicitly configured application.
|
|
1196
|
-
|
|
1197
|
-
All configuration and app-owned callables live on ``state``. Operational
|
|
1198
|
-
verification cache data belongs to this service instance, so creating a
|
|
1199
|
-
second ASGI app cannot inherit credentials, routers, or probe results from
|
|
1200
|
-
the first one.
|
|
1201
|
-
"""
|
|
1202
|
-
|
|
1203
|
-
state: ModelRuntimeState
|
|
1204
|
-
_cloud_verify_cache: Dict[str, Dict[str, Any]] = field(default_factory=dict)
|
|
1205
|
-
|
|
1206
|
-
def runtime_features(self) -> Dict[str, Any]:
|
|
1207
|
-
return runtime_features(state=self.state)
|
|
1208
|
-
|
|
1209
|
-
def engine_status(self) -> List[Dict[str, Any]]:
|
|
1210
|
-
return engine_status(
|
|
1211
|
-
state=self.state,
|
|
1212
|
-
cloud_verify_cache=self._cloud_verify_cache,
|
|
1213
|
-
)
|
|
1214
|
-
|
|
1215
|
-
def install_engine(
|
|
1216
|
-
self,
|
|
1217
|
-
engine: str,
|
|
1218
|
-
confirmation_token: Optional[str] = None,
|
|
1219
|
-
) -> Dict[str, Any]:
|
|
1220
|
-
return install_engine(
|
|
1221
|
-
engine,
|
|
1222
|
-
confirmation_token=confirmation_token,
|
|
1223
|
-
state=self.state,
|
|
1224
|
-
)
|
|
1225
|
-
|
|
1226
|
-
async def verify_cloud_models(
|
|
1227
|
-
self,
|
|
1228
|
-
force: bool = False,
|
|
1229
|
-
provider_filter: Optional[str] = None,
|
|
1230
|
-
) -> Dict[str, Dict[str, Any]]:
|
|
1231
|
-
return await verify_cloud_models(
|
|
1232
|
-
force=force,
|
|
1233
|
-
provider_filter=provider_filter,
|
|
1234
|
-
state=self.state,
|
|
1235
|
-
cache=self._cloud_verify_cache,
|
|
1236
|
-
)
|
|
1237
|
-
|
|
1238
|
-
async def prepare_and_load_model(
|
|
1239
|
-
self,
|
|
1240
|
-
model_id: str,
|
|
1241
|
-
request: Any,
|
|
1242
|
-
engine: Optional[str] = None,
|
|
1243
|
-
user_email: Optional[str] = None,
|
|
1244
|
-
adapter_path: Optional[str] = None,
|
|
1245
|
-
draft_model_id: Optional[str] = None,
|
|
1246
|
-
allow_download: bool = False,
|
|
1247
|
-
) -> Dict[str, Any]:
|
|
1248
|
-
return await prepare_and_load_model(
|
|
1249
|
-
model_id,
|
|
1250
|
-
request,
|
|
1251
|
-
engine=engine,
|
|
1252
|
-
user_email=user_email,
|
|
1253
|
-
adapter_path=adapter_path,
|
|
1254
|
-
draft_model_id=draft_model_id,
|
|
1255
|
-
allow_download=allow_download,
|
|
1256
|
-
state=self.state,
|
|
1257
|
-
)
|
|
1258
|
-
|
|
1259
|
-
async def prepare_and_load_model_stream(
|
|
1260
|
-
self,
|
|
1261
|
-
model_id: str,
|
|
1262
|
-
request: Any,
|
|
1263
|
-
engine: Optional[str] = None,
|
|
1264
|
-
user_email: Optional[str] = None,
|
|
1265
|
-
allow_download: bool = False,
|
|
1266
|
-
) -> AsyncIterator[str]:
|
|
1267
|
-
async for event in prepare_and_load_model_stream(
|
|
1268
|
-
model_id,
|
|
1269
|
-
request,
|
|
1270
|
-
engine=engine,
|
|
1271
|
-
user_email=user_email,
|
|
1272
|
-
allow_download=allow_download,
|
|
1273
|
-
state=self.state,
|
|
1274
|
-
):
|
|
1275
|
-
yield event
|
|
1276
|
-
|
|
1277
|
-
|
|
1278
|
-
def build_model_runtime(**deps: Any) -> ModelRuntimeService:
|
|
1279
|
-
"""Build the application's isolated model runtime service."""
|
|
1280
|
-
|
|
1281
|
-
return ModelRuntimeService(create_model_runtime_state(**deps))
|