ltcai 11.7.0 → 12.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +100 -76
- package/docs/BENCHMARKS.md +9 -2
- package/docs/CHANGELOG.md +249 -0
- package/docs/CI_AND_RELEASE_GATES.md +126 -41
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +271 -103
- package/docs/ENTERPRISE.md +1 -1
- package/docs/LEGACY_COMPATIBILITY.md +10 -6
- package/docs/MULTI_AGENT_RUNTIME.md +4 -4
- package/docs/ONBOARDING.md +16 -4
- package/docs/OPERATIONS.md +14 -1
- package/docs/PERMISSION_MODE.md +14 -9
- package/docs/REALTIME_COLLABORATION.md +1 -1
- package/docs/ROADMAP.md +113 -0
- package/docs/TRUST_MODEL.md +28 -7
- package/docs/USABILITY_AUDIT.md +5 -0
- package/docs/WHY_LATTICE.md +13 -5
- package/docs/WORKFLOW_DESIGNER.md +2 -2
- package/docs/kg-schema.md +57 -7
- package/docs/mcp-tools.md +93 -82
- package/docs/security-model.md +6 -3
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +1 -54
- package/lattice_brain/graph/_kg_common/extraction.py +459 -105
- package/lattice_brain/graph/_kg_common/normalize.py +305 -0
- package/lattice_brain/graph/_kg_common/patterns.py +275 -0
- package/lattice_brain/graph/_kg_common/relations.py +12 -3
- package/lattice_brain/graph/_kg_common/sections.py +107 -0
- package/lattice_brain/graph/_kg_common/text.py +14 -450
- package/lattice_brain/graph/_kg_constants.py +7 -0
- package/lattice_brain/ingestion/__init__.py +6 -3
- package/lattice_brain/multimodal/__init__.py +9 -3
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/agent_worker_seam.py +44 -1
- package/latticeai/api/models.py +18 -110
- package/latticeai/api/search.py +7 -30
- package/latticeai/api/worker_compute.py +127 -106
- package/latticeai/api/worker_seams.py +17 -2
- package/latticeai/core/embedding_providers/__init__.py +16 -0
- package/latticeai/core/embedding_providers/autodetect.py +302 -0
- package/latticeai/core/embedding_providers/base.py +25 -0
- package/latticeai/core/embedding_providers/profiles.py +44 -0
- package/latticeai/core/embedding_providers/text.py +74 -8
- package/latticeai/core/http_origin.py +3 -3
- package/latticeai/core/messages.py +0 -5
- package/latticeai/core/policy.py +1 -6
- package/latticeai/core/quiet.py +1 -20
- package/latticeai/core/security.py +29 -83
- package/latticeai/core/sessions.py +95 -4
- package/latticeai/core/users.py +0 -38
- package/latticeai/core/vector_index/__init__.py +61 -0
- package/latticeai/core/vector_index/hnsw.py +383 -0
- package/latticeai/core/vector_index/sidecar.py +329 -0
- package/latticeai/models/router/catalog.py +2 -2
- package/latticeai/models/router/generation.py +176 -30
- package/latticeai/models/router/loading.py +150 -9
- package/latticeai/runtime/access_runtime.py +7 -4
- package/latticeai/runtime/brain_runtime.py +43 -9
- package/latticeai/runtime/build_phases/features.py +8 -31
- package/latticeai/runtime/build_phases/foundation.py +7 -16
- package/latticeai/runtime/build_phases/web.py +3 -3
- package/latticeai/runtime/build_phases/worker_profile.py +29 -27
- package/latticeai/runtime/runtime_context.py +0 -2
- package/latticeai/services/architecture_readiness.py +18 -19
- package/latticeai/services/process_audit.py +1 -22
- package/latticeai/services/product_readiness.py +39 -12
- package/latticeai/services/search_service.py +7 -0
- package/latticeai/services/voice_capture.py +8 -28
- package/latticeai/tools/__init__.py +12 -47
- package/latticeai/tools/commands.py +9 -15
- package/latticeai/tools/documents.py +12 -0
- package/latticeai/tools/knowledge.py +0 -6
- package/latticeai/tools/markup.py +152 -0
- package/package.json +4 -5
- package/requirements.txt +0 -1
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_openapi_drift.mjs +3 -2
- package/scripts/check_server_i18n.mjs +5 -4
- package/scripts/compose_openapi.py +4 -1
- package/scripts/export_openapi.py +5 -4
- package/scripts/gen_worker_allowlist_fixture.py +2 -2
- package/scripts/openapi_route_families.json +19 -74
- package/scripts/publish_release.mjs +157 -0
- package/scripts/release_screen_claims.json +144 -28
- package/src-tauri/Cargo.lock +45 -10
- package/src-tauri/Cargo.toml +1 -1
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +47 -41
- package/static/app/assets/Act-Cf1L2709.js +2 -0
- package/static/app/assets/AdminConsole-DPAbLTYV.js +1 -0
- package/static/app/assets/Brain-DqamGrj-.js +2 -0
- package/static/app/assets/BrainHome-MHe2_RYs.js +2 -0
- package/static/app/assets/BrainSignals-CQPPfyyH.js +1 -0
- package/static/app/assets/Capture-DGdIH_Zc.js +1 -0
- package/static/app/assets/Chronicle-C-UlCJoJ.js +1 -0
- package/static/app/assets/CommandPalette-WNT4EqUX.js +1 -0
- package/static/app/assets/DigitalBrainExplorer-CEBH5Cwc.js +321 -0
- package/static/app/assets/Library-C6xd1dlf.js +1 -0
- package/static/app/assets/LivingBrain-BEk-0ohw.js +1 -0
- package/static/app/assets/ProductFlow-CZLm5iXh.js +1 -0
- package/static/app/assets/QueryClientProvider-B3OjqSyJ.js +1 -0
- package/static/app/assets/{ReviewCard-HXRle3qq.js → ReviewCard-CEHG6evf.js} +2 -2
- package/static/app/assets/RunsListPanel-CLtEJSRW.js +1 -0
- package/static/app/assets/System-CAxwBUXw.js +1 -0
- package/static/app/assets/WorkflowGraph-Dj10RuGE.js +1 -0
- package/static/app/assets/WorkflowsPanel-Kyeh_LIT.js +2 -0
- package/static/app/assets/actHelpers-CtSmK9Dw.js +1 -0
- package/static/app/assets/arrow-left-CRl5EO4D.js +1 -0
- package/static/app/assets/{bot-Cn8bWRuq.js → bot-DhUGRel2.js} +1 -1
- package/static/app/assets/brain-CLkhHsHF.js +1 -0
- package/static/app/assets/button-CmaEqG1T.js +1 -0
- package/static/app/assets/circle-check-CFgejkOS.js +1 -0
- package/static/app/assets/{circle-pause-CmzC_apg.js → circle-pause-l96izbxj.js} +1 -1
- package/static/app/assets/{circle-play-D8mW2aQ7.js → circle-play-CrZa25_q.js} +1 -1
- package/static/app/assets/{cpu-DZcdd0PZ.js → cpu-BaXudqwl.js} +1 -1
- package/static/app/assets/{download-bv1KEPGQ.js → download-hCVFPiyc.js} +1 -1
- package/static/app/assets/{folder-open-d-Pip5gr.js → folder-open-CHL82Yp7.js} +1 -1
- package/static/app/assets/{hard-drive-D20iavUb.js → hard-drive-DDzET7lk.js} +1 -1
- package/static/app/assets/index-CB93CZWW.css +2 -0
- package/static/app/assets/index-D2H-wSl6.js +13 -0
- package/static/app/assets/input-Df1CAY_I.js +1 -0
- package/static/app/assets/jsx-runtime-bzQ4Vb5N.js +1 -0
- package/static/app/assets/{link-2-BPJOFlAy.js → link-2-xNnTIX1_.js} +1 -1
- package/static/app/assets/{permissionCopy-ChdJd493.js → permissionCopy-D3aWHco-.js} +1 -1
- package/static/app/assets/primitives-BioD2slS.js +1 -0
- package/static/app/assets/search-BzBw8YcW.js +1 -0
- package/static/app/assets/{share-2-YNX_NtMU.js → share-2-FkzGf8Df.js} +1 -1
- package/static/app/assets/{shield-alert-DuQ3zrVL.js → shield-alert-B3dwzik4.js} +1 -1
- package/static/app/assets/sourceMeta-DQSY_tah.js +1 -0
- package/static/app/assets/textarea-P8o6pvOP.js +1 -0
- package/static/app/assets/useFocusTrap-hswOIkXE.js +1 -0
- package/static/app/assets/useMutation-OJLrYSRA.js +1 -0
- package/static/app/assets/workspace-BCuk3Ku9.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/ingestion/pipeline.py +0 -108
- package/latticeai/api/local_files.py +0 -44
- package/latticeai/api/tools.py +0 -126
- package/latticeai/api/voice_capture.py +0 -32
- package/latticeai/core/agent_permission.py +0 -85
- package/scripts/agent_eval.py +0 -34
- package/scripts/brain_quality_eval.py +0 -37
- package/scripts/check_legacy_debt.mjs +0 -91
- package/scripts/check_python.py +0 -100
- package/scripts/chunking_parity_corpus.py +0 -449
- package/scripts/generate_agent_parity_fixtures.py +0 -771
- package/scripts/generate_chunking_parity_fixtures.py +0 -259
- package/static/app/assets/Act-BPcVAbOL.js +0 -1
- package/static/app/assets/AdminConsole-Bw1ATQL0.js +0 -1
- package/static/app/assets/Brain-CT92Kos0.js +0 -321
- package/static/app/assets/BrainHome-CFBkt1K_.js +0 -2
- package/static/app/assets/BrainSignals-ReLWF2H8.js +0 -1
- package/static/app/assets/Capture-BsTokYkk.js +0 -1
- package/static/app/assets/Chronicle-B6f0T9id.js +0 -1
- package/static/app/assets/CommandPalette-CuvjTv1u.js +0 -1
- package/static/app/assets/Library-BGJbG9Hd.js +0 -1
- package/static/app/assets/LivingBrain-DGYK_Jsa.js +0 -1
- package/static/app/assets/ProductFlow-DXBC6brE.js +0 -1
- package/static/app/assets/System-CMHSO9qM.js +0 -1
- package/static/app/assets/arrow-left-BfmkskWx.js +0 -1
- package/static/app/assets/brain-CQJberbE.js +0 -1
- package/static/app/assets/button-Ct9f2_oT.js +0 -1
- package/static/app/assets/circle-check-DruOxB-4.js +0 -1
- package/static/app/assets/index-D9x-kSNy.css +0 -2
- package/static/app/assets/index-Do83hDzJ.js +0 -10
- package/static/app/assets/input-BLXVNmj1.js +0 -1
- package/static/app/assets/primitives-Cv5tbZBY.js +0 -1
- package/static/app/assets/search-CT9aho2j.js +0 -1
- package/static/app/assets/textarea-DqwLnli4.js +0 -1
- package/static/app/assets/useFocusTrap-ZVI98jaW.js +0 -1
- package/static/app/assets/useMutation-CVC4qv_D.js +0 -1
- package/static/app/assets/useQuery-C7BeG4HU.js +0 -1
- package/static/app/assets/utils-CiFtIdZq.js +0 -4
- package/static/app/assets/workspace-DQz9vIId.js +0 -1
|
@@ -53,7 +53,7 @@ from __future__ import annotations
|
|
|
53
53
|
|
|
54
54
|
import asyncio
|
|
55
55
|
import os
|
|
56
|
-
from typing import Any, Callable, Dict, Optional
|
|
56
|
+
from typing import Any, Callable, Dict, List, Optional
|
|
57
57
|
|
|
58
58
|
from fastapi import APIRouter, HTTPException, Request
|
|
59
59
|
from pydantic import BaseModel, Field
|
|
@@ -78,6 +78,16 @@ MAX_MAX_TOKENS = 8192
|
|
|
78
78
|
MIN_TEMPERATURE = 0.0
|
|
79
79
|
MAX_TEMPERATURE = 2.0
|
|
80
80
|
|
|
81
|
+
#: How many stop strings one completion may name. The kernel sends two.
|
|
82
|
+
MAX_STOP_STRINGS = 8
|
|
83
|
+
|
|
84
|
+
#: Longest forced completion prefix one call may name (v12.0.0). The prefix is
|
|
85
|
+
#: prepended to the prompt, so an unbounded one is an unbounded prompt. The
|
|
86
|
+
#: kernel sends two, and both are short: the executor's opening brace
|
|
87
|
+
#: (``{"thoughts": "``) and the guided menu's answer label — a few dozen
|
|
88
|
+
#: characters between them, and nothing legitimate needs a paragraph.
|
|
89
|
+
MAX_PREFIX_CHARS = 256
|
|
90
|
+
|
|
81
91
|
#: Rate-limit bucket. Deliberately *not* the ``"agent"`` bucket ``/agent`` uses:
|
|
82
92
|
#: that one is sized per *run* (10 burst, one refill per 10s) because one HTTP
|
|
83
93
|
#: call there is a whole agent run. Here one call is a single loop step, and a
|
|
@@ -101,6 +111,26 @@ class AgentLLMRequest(BaseModel):
|
|
|
101
111
|
context: Optional[str] = None
|
|
102
112
|
max_tokens: int = 4096
|
|
103
113
|
temperature: float = 0.2
|
|
114
|
+
#: Strings that end the reply (v11.9.0). Optional, and absent means what it
|
|
115
|
+
#: always meant: generate to ``max_tokens``. The kernel sends it on exactly
|
|
116
|
+
#: one call — the strict verification re-ask, whose reply is one short
|
|
117
|
+
#: closed object — because a stop string that is safe there ("\n```") would
|
|
118
|
+
#: truncate any reply carrying file content. Bounded so a caller cannot make
|
|
119
|
+
#: the sampler check a thousand needles per token.
|
|
120
|
+
stop: Optional[List[str]] = Field(default=None, max_length=MAX_STOP_STRINGS)
|
|
121
|
+
#: Characters the reply is **forced to begin with** (v12.0.0).
|
|
122
|
+
#:
|
|
123
|
+
#: The cheapest structural guarantee the seam can offer: instead of asking
|
|
124
|
+
#: a weak model to start at ``{`` and repairing the markdown fence it emits
|
|
125
|
+
#: instead, the worker prefills those characters and the model continues
|
|
126
|
+
#: from them. The response's ``text`` always starts with the prefix, so a
|
|
127
|
+
#: caller reads one shape whether the local prefill or a cloud provider's
|
|
128
|
+
#: assistant-prefill message produced it.
|
|
129
|
+
#:
|
|
130
|
+
#: Bounded because it is prepended to the prompt: an unbounded prefix would
|
|
131
|
+
#: be an unbounded prompt on the single MLX executor. The kernel sends
|
|
132
|
+
#: fourteen characters.
|
|
133
|
+
prefix: Optional[str] = Field(default=None, max_length=MAX_PREFIX_CHARS)
|
|
104
134
|
|
|
105
135
|
|
|
106
136
|
class AgentToolRequest(BaseModel):
|
|
@@ -247,6 +277,18 @@ def create_agent_worker_seam_router(
|
|
|
247
277
|
context=req.context,
|
|
248
278
|
max_tokens=req.max_tokens,
|
|
249
279
|
temperature=req.temperature,
|
|
280
|
+
stop=req.stop or None,
|
|
281
|
+
prefix=req.prefix or None,
|
|
282
|
+
# **The seam's ``context`` is a prompt, not a corpus** (v12.0.0).
|
|
283
|
+
# ``_compose_system`` appends the citation mandate to any non-empty
|
|
284
|
+
# context, which is right for chat (where it really is retrieved
|
|
285
|
+
# passages) and wrong here: the Rust kernel sends its whole executor
|
|
286
|
+
# prompt as ``context``, so every agent turn was being told its own
|
|
287
|
+
# instructions were sources to "cite inline as [1], [2]". Small
|
|
288
|
+
# models obey that — a 0.5B wrote ``[1] …`` into a file and a 2B
|
|
289
|
+
# answered a tool call with the citation instruction. A caller that
|
|
290
|
+
# wants citations puts them in its own prompt.
|
|
291
|
+
cite_sources=False,
|
|
250
292
|
)
|
|
251
293
|
return {"text": str(text)}
|
|
252
294
|
|
|
@@ -305,6 +347,7 @@ def create_agent_worker_seam_router(
|
|
|
305
347
|
|
|
306
348
|
__all__ = [
|
|
307
349
|
"MAX_MAX_TOKENS",
|
|
350
|
+
"MAX_PREFIX_CHARS",
|
|
308
351
|
"MAX_TEMPERATURE",
|
|
309
352
|
"MIN_MAX_TOKENS",
|
|
310
353
|
"MIN_TEMPERATURE",
|
package/latticeai/api/models.py
CHANGED
|
@@ -2,15 +2,24 @@
|
|
|
2
2
|
|
|
3
3
|
Extracted from ``server_app.py`` in v1.3.0 with the whole ``/models*`` +
|
|
4
4
|
``/engines*`` + ``/setup/set-api-key`` surface. v11.6.0 kept the eight routes
|
|
5
|
-
that are *this interpreter's* business
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
the two catalogue reads (``/models/compat-profiles``, ``/models/recommendations``)
|
|
10
|
-
are product views.
|
|
5
|
+
that are *this interpreter's* business and moved the rest to ``lattice-host``:
|
|
6
|
+
engine installation and cloud-key verification are host operations,
|
|
7
|
+
``/setup/set-api-key`` is account state, and the two catalogue reads
|
|
8
|
+
(``/models/compat-profiles``, ``/models/recommendations``) are product views.
|
|
11
9
|
|
|
12
|
-
|
|
13
|
-
|
|
10
|
+
v11.8.0 took three of those eight away, because nothing called them — not the
|
|
11
|
+
Rust surface, not the SPA client, not either extension:
|
|
12
|
+
|
|
13
|
+
* ``POST /engines/pull-model`` — every caller reaches a download through
|
|
14
|
+
``/engines/prepare-model``, which resolves, downloads on consent, loads and
|
|
15
|
+
smoke-tests in one step. The bare pull was the older door, and with it went
|
|
16
|
+
this module's only use of ``huggingface_hub`` / ``ollama`` (both still run
|
|
17
|
+
here, under ``services/model_loading.py``, for the prepare flow).
|
|
18
|
+
* ``POST /models/switch/{model_id:path}`` — ``/models/load`` is what every
|
|
19
|
+
surface sends, and it switches as part of loading.
|
|
20
|
+
* ``DELETE /models/unload-all`` — unloading is per-model everywhere.
|
|
21
|
+
|
|
22
|
+
What is left is list, load, unload-one and the two prepare flows.
|
|
14
23
|
|
|
15
24
|
Mirrors the established router-factory convention: the heavy provider/runtime
|
|
16
25
|
helpers are injected as bound service callables. This module owns the sole
|
|
@@ -21,14 +30,13 @@ from __future__ import annotations
|
|
|
21
30
|
|
|
22
31
|
import asyncio
|
|
23
32
|
import logging
|
|
24
|
-
import subprocess
|
|
25
33
|
from typing import Any, Callable, Dict, List, NoReturn, Optional
|
|
26
34
|
|
|
27
35
|
from fastapi import APIRouter, HTTPException, Request
|
|
28
36
|
from fastapi.responses import StreamingResponse
|
|
29
37
|
from pydantic import BaseModel
|
|
30
38
|
|
|
31
|
-
from latticeai.core.messages import http_error, resolve_language
|
|
39
|
+
from latticeai.core.messages import http_error, resolve_language
|
|
32
40
|
from latticeai.services.model_errors import ModelRuntimeError
|
|
33
41
|
|
|
34
42
|
|
|
@@ -76,11 +84,6 @@ class LoadModelRequest(BaseModel):
|
|
|
76
84
|
|
|
77
85
|
|
|
78
86
|
|
|
79
|
-
class PullModelRequest(BaseModel):
|
|
80
|
-
model: str
|
|
81
|
-
allow_download: bool = False
|
|
82
|
-
|
|
83
|
-
|
|
84
87
|
class PrepareModelRequest(BaseModel):
|
|
85
88
|
model: str
|
|
86
89
|
engine: Optional[str] = None
|
|
@@ -94,13 +97,9 @@ def create_models_router(
|
|
|
94
97
|
model_router: Any,
|
|
95
98
|
require_user: Callable[[Request], str],
|
|
96
99
|
require_admin: Callable[[Request], tuple],
|
|
97
|
-
normalize_local_model_request: Callable[..., str],
|
|
98
|
-
download_hf_model: Callable[..., Dict],
|
|
99
100
|
prepare_and_load_model: Callable[..., Any],
|
|
100
101
|
prepare_and_load_model_stream: Callable[..., Any],
|
|
101
102
|
sse_event: Callable[[str, Dict], str],
|
|
102
|
-
ensure_ollama_server: Callable[[], None],
|
|
103
|
-
local_binary: Callable[[str], Optional[str]],
|
|
104
103
|
engine_status: Callable[[], List[Dict]],
|
|
105
104
|
filter_lower_family_versions: Callable[[List[Dict]], List[Dict]],
|
|
106
105
|
list_compat_profiles: Callable[[], Any],
|
|
@@ -289,80 +288,6 @@ def create_models_router(
|
|
|
289
288
|
|
|
290
289
|
# ── Engines ───────────────────────────────────────────────────────────
|
|
291
290
|
|
|
292
|
-
|
|
293
|
-
@router.post("/engines/pull-model")
|
|
294
|
-
async def pull_ollama_model(req: PullModelRequest, request: Request):
|
|
295
|
-
_authorize_model_admin(request)
|
|
296
|
-
if not req.allow_download:
|
|
297
|
-
raise HTTPException(
|
|
298
|
-
status_code=403,
|
|
299
|
-
detail=translate("models.download_consent_required", resolve_language(request)),
|
|
300
|
-
)
|
|
301
|
-
model_ref = normalize_local_model_request(req.model, None)
|
|
302
|
-
if not model_ref:
|
|
303
|
-
raise http_error(400, "models.identifier_empty", resolve_language(request))
|
|
304
|
-
|
|
305
|
-
if ":" in model_ref and model_ref.split(":", 1)[0].strip().lower() in {"ollama", "vllm", "lmstudio", "llamacpp", "local_mlx", "mlx"}:
|
|
306
|
-
provider, model_name = model_ref.split(":", 1)
|
|
307
|
-
provider = provider.strip().lower()
|
|
308
|
-
model_name = model_name.strip()
|
|
309
|
-
else:
|
|
310
|
-
provider, model_name = "local_mlx", model_ref
|
|
311
|
-
|
|
312
|
-
if not model_name:
|
|
313
|
-
raise http_error(400, "models.name_empty", resolve_language(request))
|
|
314
|
-
|
|
315
|
-
if provider == "ollama":
|
|
316
|
-
try:
|
|
317
|
-
# Starts the daemon and waits for it — blocking, like the pull
|
|
318
|
-
# below, and for the same reason it may not run on the loop.
|
|
319
|
-
await asyncio.to_thread(ensure_ollama_server)
|
|
320
|
-
except ModelRuntimeError as exc:
|
|
321
|
-
_raise_model_http(exc)
|
|
322
|
-
ollama = local_binary("ollama")
|
|
323
|
-
if not ollama:
|
|
324
|
-
raise http_error(400, "models.ollama_missing", resolve_language(request))
|
|
325
|
-
try:
|
|
326
|
-
# A model pull is minutes of network I/O. Held on the event loop
|
|
327
|
-
# it stalled every other request — including the health check the
|
|
328
|
-
# UI uses to decide the server is alive — until the pull finished.
|
|
329
|
-
completed = await asyncio.to_thread(
|
|
330
|
-
lambda: subprocess.run(
|
|
331
|
-
[ollama, "pull", model_name],
|
|
332
|
-
capture_output=True, text=True, timeout=900, check=False,
|
|
333
|
-
)
|
|
334
|
-
)
|
|
335
|
-
except subprocess.TimeoutExpired:
|
|
336
|
-
raise http_error(408, "models.download_timeout", resolve_language(request))
|
|
337
|
-
if completed.returncode != 0:
|
|
338
|
-
raise HTTPException(
|
|
339
|
-
status_code=500,
|
|
340
|
-
detail=completed.stderr[-2000:]
|
|
341
|
-
or translate("models.pull_failed", resolve_language(request)),
|
|
342
|
-
)
|
|
343
|
-
return {"provider": provider, "model": model_name, "returncode": completed.returncode}
|
|
344
|
-
|
|
345
|
-
if provider == "lmstudio":
|
|
346
|
-
raise HTTPException(
|
|
347
|
-
status_code=400,
|
|
348
|
-
detail=(
|
|
349
|
-
"LM Studio 모델은 Lattice에서 Hugging Face로 pull하지 않습니다. "
|
|
350
|
-
"LM Studio 앱에서 모델을 다운로드하고 Local Server를 켠 뒤 모델을 로드하세요. "
|
|
351
|
-
"그러면 모델 선택창에 실제 /v1/models 항목이 표시됩니다."
|
|
352
|
-
),
|
|
353
|
-
)
|
|
354
|
-
|
|
355
|
-
if provider in {"vllm", "llamacpp", "local_mlx", "mlx"}:
|
|
356
|
-
download_provider = "local_mlx" if provider == "mlx" else provider
|
|
357
|
-
try:
|
|
358
|
-
# Multi-gigabyte Hugging Face download; same rule as the pull above.
|
|
359
|
-
result = await asyncio.to_thread(download_hf_model, model_name, download_provider)
|
|
360
|
-
except ModelRuntimeError as exc:
|
|
361
|
-
_raise_model_http(exc)
|
|
362
|
-
return {"provider": provider, "model": model_name, "returncode": 0, **result}
|
|
363
|
-
|
|
364
|
-
raise http_error(400, "models.download_not_automated", resolve_language(request), provider=provider) # pragma: no cover — every prefix the check above accepts is handled, and the fallback is local_mlx
|
|
365
|
-
|
|
366
291
|
@router.post("/engines/prepare-model")
|
|
367
292
|
async def engines_prepare_model(req: PrepareModelRequest, request: Request):
|
|
368
293
|
current_user = _authorize_model_admin(request, req.user_email)
|
|
@@ -484,27 +409,10 @@ def create_models_router(
|
|
|
484
409
|
detail=friendly_model_runtime_error(e, model_id=req.model_id, engine=req.engine),
|
|
485
410
|
)
|
|
486
411
|
|
|
487
|
-
@router.post("/models/switch/{model_id:path}")
|
|
488
|
-
async def switch_model(model_id: str, request: Request):
|
|
489
|
-
_authorize_model_admin(request)
|
|
490
|
-
try:
|
|
491
|
-
_router.switch_model(model_id)
|
|
492
|
-
return {"status": "ok", "current": _router.current_model_id}
|
|
493
|
-
except KeyError:
|
|
494
|
-
raise http_error(404, "chat.model_not_loaded", resolve_language(request), model=model_id)
|
|
495
|
-
|
|
496
412
|
@router.delete("/models/unload/{model_id:path}")
|
|
497
413
|
async def unload_model(model_id: str, request: Request):
|
|
498
414
|
_authorize_model_admin(request)
|
|
499
415
|
_router.unload_model(model_id)
|
|
500
416
|
return {"status": "ok", "unloaded": model_id}
|
|
501
417
|
|
|
502
|
-
@router.delete("/models/unload-all")
|
|
503
|
-
async def unload_all_models(request: Request):
|
|
504
|
-
_authorize_model_admin(request)
|
|
505
|
-
unloaded = _router.loaded_model_ids
|
|
506
|
-
_router.unload_all()
|
|
507
|
-
return {"status": "ok", "unloaded": unloaded}
|
|
508
|
-
|
|
509
|
-
|
|
510
418
|
return router
|
package/latticeai/api/search.py
CHANGED
|
@@ -10,6 +10,11 @@ it is the one that was asked for.
|
|
|
10
10
|
``GET /api/embeddings/status`` therefore reports the **embedder**, not the
|
|
11
11
|
index. Index completeness is a native jobs route now, and reporting it from
|
|
12
12
|
here would have meant re-opening a store the worker no longer holds.
|
|
13
|
+
|
|
14
|
+
It is now the module's only route. ``GET /api/embeddings/providers`` — the
|
|
15
|
+
static catalogue of provider ids and the env vars each needs — was removed in
|
|
16
|
+
v11.8.0: no surface in the tree asked for it, and a catalogue nothing reads is
|
|
17
|
+
a second place for the provider list to go stale.
|
|
13
18
|
"""
|
|
14
19
|
|
|
15
20
|
from __future__ import annotations
|
|
@@ -18,7 +23,6 @@ from typing import Any, Callable, Dict, NoReturn, Optional
|
|
|
18
23
|
|
|
19
24
|
from fastapi import APIRouter, HTTPException, Request
|
|
20
25
|
|
|
21
|
-
from latticeai.core.embedding_providers import embedding_provider_profiles
|
|
22
26
|
from latticeai.services.search_service import SearchService
|
|
23
27
|
|
|
24
28
|
|
|
@@ -31,8 +35,8 @@ def create_search_router(
|
|
|
31
35
|
router = APIRouter()
|
|
32
36
|
|
|
33
37
|
def _raise_embedder_error(exc: Exception) -> NoReturn:
|
|
34
|
-
# NoReturn, not None:
|
|
35
|
-
#
|
|
38
|
+
# NoReturn, not None: the handler ends with this in its except branch,
|
|
39
|
+
# and without it the function reads as a missing return.
|
|
36
40
|
raise HTTPException(status_code=404, detail=str(exc)) from exc
|
|
37
41
|
|
|
38
42
|
@router.get("/api/embeddings/status")
|
|
@@ -44,31 +48,4 @@ def create_search_router(
|
|
|
44
48
|
except ValueError as exc:
|
|
45
49
|
_raise_embedder_error(exc)
|
|
46
50
|
|
|
47
|
-
@router.get("/api/embeddings/providers")
|
|
48
|
-
async def embeddings_providers(request: Request) -> Dict[str, Any]:
|
|
49
|
-
require_user(request)
|
|
50
|
-
resolved = embedding_info() if embedding_info else {}
|
|
51
|
-
profiles = resolved.get("profiles") or embedding_provider_profiles()
|
|
52
|
-
return {
|
|
53
|
-
"active": resolved.get("active_provider"),
|
|
54
|
-
"requested": resolved.get("requested_provider"),
|
|
55
|
-
"profile": resolved.get("profile") or "",
|
|
56
|
-
"profiles": profiles,
|
|
57
|
-
"providers": [
|
|
58
|
-
{"id": "hash", "label": "Local hash (fallback)", "grade": "fallback",
|
|
59
|
-
"requires": [], "detail": "Deterministic offline vectors — always available."},
|
|
60
|
-
{"id": "mlx", "label": "MLX (Apple Silicon)", "grade": "production",
|
|
61
|
-
"requires": ["LATTICEAI_EMBEDDING_MODEL"], "detail": "Local embedding model via MLX."},
|
|
62
|
-
{"id": "ollama", "label": "Ollama", "grade": "production",
|
|
63
|
-
"requires": ["LATTICEAI_EMBEDDING_MODEL", "LATTICEAI_EMBEDDING_BASE_URL"],
|
|
64
|
-
"detail": "Local/remote Ollama embedding server."},
|
|
65
|
-
{"id": "openai", "label": "OpenAI-compatible", "grade": "production",
|
|
66
|
-
"requires": ["LATTICEAI_EMBEDDING_MODEL", "LATTICEAI_EMBEDDING_BASE_URL", "LATTICEAI_EMBEDDING_API_KEY"],
|
|
67
|
-
"detail": "Any /v1/embeddings endpoint (OpenAI, LM Studio, vLLM, …)."},
|
|
68
|
-
{"id": "custom", "label": "Custom callable", "grade": "production",
|
|
69
|
-
"requires": ["LATTICEAI_EMBEDDING_CUSTOM_TARGET"],
|
|
70
|
-
"detail": "User-supplied module:callable returning vectors."},
|
|
71
|
-
],
|
|
72
|
-
}
|
|
73
|
-
|
|
74
51
|
return router
|