ltcai 11.7.0 → 12.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/README.md +100 -76
  2. package/docs/BENCHMARKS.md +9 -2
  3. package/docs/CHANGELOG.md +249 -0
  4. package/docs/CI_AND_RELEASE_GATES.md +126 -41
  5. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  6. package/docs/DEVELOPMENT.md +271 -103
  7. package/docs/ENTERPRISE.md +1 -1
  8. package/docs/LEGACY_COMPATIBILITY.md +10 -6
  9. package/docs/MULTI_AGENT_RUNTIME.md +4 -4
  10. package/docs/ONBOARDING.md +16 -4
  11. package/docs/OPERATIONS.md +14 -1
  12. package/docs/PERMISSION_MODE.md +14 -9
  13. package/docs/REALTIME_COLLABORATION.md +1 -1
  14. package/docs/ROADMAP.md +113 -0
  15. package/docs/TRUST_MODEL.md +28 -7
  16. package/docs/USABILITY_AUDIT.md +5 -0
  17. package/docs/WHY_LATTICE.md +13 -5
  18. package/docs/WORKFLOW_DESIGNER.md +2 -2
  19. package/docs/kg-schema.md +57 -7
  20. package/docs/mcp-tools.md +93 -82
  21. package/docs/security-model.md +6 -3
  22. package/lattice_brain/__init__.py +1 -1
  23. package/lattice_brain/graph/_kg_common/__init__.py +1 -54
  24. package/lattice_brain/graph/_kg_common/extraction.py +459 -105
  25. package/lattice_brain/graph/_kg_common/normalize.py +305 -0
  26. package/lattice_brain/graph/_kg_common/patterns.py +275 -0
  27. package/lattice_brain/graph/_kg_common/relations.py +12 -3
  28. package/lattice_brain/graph/_kg_common/sections.py +107 -0
  29. package/lattice_brain/graph/_kg_common/text.py +14 -450
  30. package/lattice_brain/graph/_kg_constants.py +7 -0
  31. package/lattice_brain/ingestion/__init__.py +6 -3
  32. package/lattice_brain/multimodal/__init__.py +9 -3
  33. package/latticeai/__init__.py +1 -1
  34. package/latticeai/api/agent_worker_seam.py +44 -1
  35. package/latticeai/api/models.py +18 -110
  36. package/latticeai/api/search.py +7 -30
  37. package/latticeai/api/worker_compute.py +127 -106
  38. package/latticeai/api/worker_seams.py +17 -2
  39. package/latticeai/core/embedding_providers/__init__.py +16 -0
  40. package/latticeai/core/embedding_providers/autodetect.py +302 -0
  41. package/latticeai/core/embedding_providers/base.py +25 -0
  42. package/latticeai/core/embedding_providers/profiles.py +44 -0
  43. package/latticeai/core/embedding_providers/text.py +74 -8
  44. package/latticeai/core/http_origin.py +3 -3
  45. package/latticeai/core/messages.py +0 -5
  46. package/latticeai/core/policy.py +1 -6
  47. package/latticeai/core/quiet.py +1 -20
  48. package/latticeai/core/security.py +29 -83
  49. package/latticeai/core/sessions.py +95 -4
  50. package/latticeai/core/users.py +0 -38
  51. package/latticeai/core/vector_index/__init__.py +61 -0
  52. package/latticeai/core/vector_index/hnsw.py +383 -0
  53. package/latticeai/core/vector_index/sidecar.py +329 -0
  54. package/latticeai/models/router/catalog.py +2 -2
  55. package/latticeai/models/router/generation.py +176 -30
  56. package/latticeai/models/router/loading.py +150 -9
  57. package/latticeai/runtime/access_runtime.py +7 -4
  58. package/latticeai/runtime/brain_runtime.py +43 -9
  59. package/latticeai/runtime/build_phases/features.py +8 -31
  60. package/latticeai/runtime/build_phases/foundation.py +7 -16
  61. package/latticeai/runtime/build_phases/web.py +3 -3
  62. package/latticeai/runtime/build_phases/worker_profile.py +29 -27
  63. package/latticeai/runtime/runtime_context.py +0 -2
  64. package/latticeai/services/architecture_readiness.py +18 -19
  65. package/latticeai/services/process_audit.py +1 -22
  66. package/latticeai/services/product_readiness.py +39 -12
  67. package/latticeai/services/search_service.py +7 -0
  68. package/latticeai/services/voice_capture.py +8 -28
  69. package/latticeai/tools/__init__.py +12 -47
  70. package/latticeai/tools/commands.py +9 -15
  71. package/latticeai/tools/documents.py +12 -0
  72. package/latticeai/tools/knowledge.py +0 -6
  73. package/latticeai/tools/markup.py +152 -0
  74. package/package.json +4 -5
  75. package/requirements.txt +0 -1
  76. package/scripts/check_current_release_docs.mjs +1 -1
  77. package/scripts/check_openapi_drift.mjs +3 -2
  78. package/scripts/check_server_i18n.mjs +5 -4
  79. package/scripts/compose_openapi.py +4 -1
  80. package/scripts/export_openapi.py +5 -4
  81. package/scripts/gen_worker_allowlist_fixture.py +2 -2
  82. package/scripts/openapi_route_families.json +19 -74
  83. package/scripts/publish_release.mjs +157 -0
  84. package/scripts/release_screen_claims.json +144 -28
  85. package/src-tauri/Cargo.lock +45 -10
  86. package/src-tauri/Cargo.toml +1 -1
  87. package/src-tauri/tauri.conf.json +1 -1
  88. package/static/app/asset-manifest.json +47 -41
  89. package/static/app/assets/Act-Cf1L2709.js +2 -0
  90. package/static/app/assets/AdminConsole-DPAbLTYV.js +1 -0
  91. package/static/app/assets/Brain-DqamGrj-.js +2 -0
  92. package/static/app/assets/BrainHome-MHe2_RYs.js +2 -0
  93. package/static/app/assets/BrainSignals-CQPPfyyH.js +1 -0
  94. package/static/app/assets/Capture-DGdIH_Zc.js +1 -0
  95. package/static/app/assets/Chronicle-C-UlCJoJ.js +1 -0
  96. package/static/app/assets/CommandPalette-WNT4EqUX.js +1 -0
  97. package/static/app/assets/DigitalBrainExplorer-CEBH5Cwc.js +321 -0
  98. package/static/app/assets/Library-C6xd1dlf.js +1 -0
  99. package/static/app/assets/LivingBrain-BEk-0ohw.js +1 -0
  100. package/static/app/assets/ProductFlow-CZLm5iXh.js +1 -0
  101. package/static/app/assets/QueryClientProvider-B3OjqSyJ.js +1 -0
  102. package/static/app/assets/{ReviewCard-HXRle3qq.js → ReviewCard-CEHG6evf.js} +2 -2
  103. package/static/app/assets/RunsListPanel-CLtEJSRW.js +1 -0
  104. package/static/app/assets/System-CAxwBUXw.js +1 -0
  105. package/static/app/assets/WorkflowGraph-Dj10RuGE.js +1 -0
  106. package/static/app/assets/WorkflowsPanel-Kyeh_LIT.js +2 -0
  107. package/static/app/assets/actHelpers-CtSmK9Dw.js +1 -0
  108. package/static/app/assets/arrow-left-CRl5EO4D.js +1 -0
  109. package/static/app/assets/{bot-Cn8bWRuq.js → bot-DhUGRel2.js} +1 -1
  110. package/static/app/assets/brain-CLkhHsHF.js +1 -0
  111. package/static/app/assets/button-CmaEqG1T.js +1 -0
  112. package/static/app/assets/circle-check-CFgejkOS.js +1 -0
  113. package/static/app/assets/{circle-pause-CmzC_apg.js → circle-pause-l96izbxj.js} +1 -1
  114. package/static/app/assets/{circle-play-D8mW2aQ7.js → circle-play-CrZa25_q.js} +1 -1
  115. package/static/app/assets/{cpu-DZcdd0PZ.js → cpu-BaXudqwl.js} +1 -1
  116. package/static/app/assets/{download-bv1KEPGQ.js → download-hCVFPiyc.js} +1 -1
  117. package/static/app/assets/{folder-open-d-Pip5gr.js → folder-open-CHL82Yp7.js} +1 -1
  118. package/static/app/assets/{hard-drive-D20iavUb.js → hard-drive-DDzET7lk.js} +1 -1
  119. package/static/app/assets/index-CB93CZWW.css +2 -0
  120. package/static/app/assets/index-D2H-wSl6.js +13 -0
  121. package/static/app/assets/input-Df1CAY_I.js +1 -0
  122. package/static/app/assets/jsx-runtime-bzQ4Vb5N.js +1 -0
  123. package/static/app/assets/{link-2-BPJOFlAy.js → link-2-xNnTIX1_.js} +1 -1
  124. package/static/app/assets/{permissionCopy-ChdJd493.js → permissionCopy-D3aWHco-.js} +1 -1
  125. package/static/app/assets/primitives-BioD2slS.js +1 -0
  126. package/static/app/assets/search-BzBw8YcW.js +1 -0
  127. package/static/app/assets/{share-2-YNX_NtMU.js → share-2-FkzGf8Df.js} +1 -1
  128. package/static/app/assets/{shield-alert-DuQ3zrVL.js → shield-alert-B3dwzik4.js} +1 -1
  129. package/static/app/assets/sourceMeta-DQSY_tah.js +1 -0
  130. package/static/app/assets/textarea-P8o6pvOP.js +1 -0
  131. package/static/app/assets/useFocusTrap-hswOIkXE.js +1 -0
  132. package/static/app/assets/useMutation-OJLrYSRA.js +1 -0
  133. package/static/app/assets/workspace-BCuk3Ku9.js +1 -0
  134. package/static/app/index.html +4 -4
  135. package/static/sw.js +1 -1
  136. package/lattice_brain/ingestion/pipeline.py +0 -108
  137. package/latticeai/api/local_files.py +0 -44
  138. package/latticeai/api/tools.py +0 -126
  139. package/latticeai/api/voice_capture.py +0 -32
  140. package/latticeai/core/agent_permission.py +0 -85
  141. package/scripts/agent_eval.py +0 -34
  142. package/scripts/brain_quality_eval.py +0 -37
  143. package/scripts/check_legacy_debt.mjs +0 -91
  144. package/scripts/check_python.py +0 -100
  145. package/scripts/chunking_parity_corpus.py +0 -449
  146. package/scripts/generate_agent_parity_fixtures.py +0 -771
  147. package/scripts/generate_chunking_parity_fixtures.py +0 -259
  148. package/static/app/assets/Act-BPcVAbOL.js +0 -1
  149. package/static/app/assets/AdminConsole-Bw1ATQL0.js +0 -1
  150. package/static/app/assets/Brain-CT92Kos0.js +0 -321
  151. package/static/app/assets/BrainHome-CFBkt1K_.js +0 -2
  152. package/static/app/assets/BrainSignals-ReLWF2H8.js +0 -1
  153. package/static/app/assets/Capture-BsTokYkk.js +0 -1
  154. package/static/app/assets/Chronicle-B6f0T9id.js +0 -1
  155. package/static/app/assets/CommandPalette-CuvjTv1u.js +0 -1
  156. package/static/app/assets/Library-BGJbG9Hd.js +0 -1
  157. package/static/app/assets/LivingBrain-DGYK_Jsa.js +0 -1
  158. package/static/app/assets/ProductFlow-DXBC6brE.js +0 -1
  159. package/static/app/assets/System-CMHSO9qM.js +0 -1
  160. package/static/app/assets/arrow-left-BfmkskWx.js +0 -1
  161. package/static/app/assets/brain-CQJberbE.js +0 -1
  162. package/static/app/assets/button-Ct9f2_oT.js +0 -1
  163. package/static/app/assets/circle-check-DruOxB-4.js +0 -1
  164. package/static/app/assets/index-D9x-kSNy.css +0 -2
  165. package/static/app/assets/index-Do83hDzJ.js +0 -10
  166. package/static/app/assets/input-BLXVNmj1.js +0 -1
  167. package/static/app/assets/primitives-Cv5tbZBY.js +0 -1
  168. package/static/app/assets/search-CT9aho2j.js +0 -1
  169. package/static/app/assets/textarea-DqwLnli4.js +0 -1
  170. package/static/app/assets/useFocusTrap-ZVI98jaW.js +0 -1
  171. package/static/app/assets/useMutation-CVC4qv_D.js +0 -1
  172. package/static/app/assets/useQuery-C7BeG4HU.js +0 -1
  173. package/static/app/assets/utils-CiFtIdZq.js +0 -4
  174. package/static/app/assets/workspace-DQz9vIId.js +0 -1
@@ -53,7 +53,7 @@ from __future__ import annotations
53
53
 
54
54
  import asyncio
55
55
  import os
56
- from typing import Any, Callable, Dict, Optional
56
+ from typing import Any, Callable, Dict, List, Optional
57
57
 
58
58
  from fastapi import APIRouter, HTTPException, Request
59
59
  from pydantic import BaseModel, Field
@@ -78,6 +78,16 @@ MAX_MAX_TOKENS = 8192
78
78
  MIN_TEMPERATURE = 0.0
79
79
  MAX_TEMPERATURE = 2.0
80
80
 
81
+ #: How many stop strings one completion may name. The kernel sends two.
82
+ MAX_STOP_STRINGS = 8
83
+
84
+ #: Longest forced completion prefix one call may name (v12.0.0). The prefix is
85
+ #: prepended to the prompt, so an unbounded one is an unbounded prompt. The
86
+ #: kernel sends two, and both are short: the executor's opening brace
87
+ #: (``{"thoughts": "``) and the guided menu's answer label — a few dozen
88
+ #: characters between them, and nothing legitimate needs a paragraph.
89
+ MAX_PREFIX_CHARS = 256
90
+
81
91
  #: Rate-limit bucket. Deliberately *not* the ``"agent"`` bucket ``/agent`` uses:
82
92
  #: that one is sized per *run* (10 burst, one refill per 10s) because one HTTP
83
93
  #: call there is a whole agent run. Here one call is a single loop step, and a
@@ -101,6 +111,26 @@ class AgentLLMRequest(BaseModel):
101
111
  context: Optional[str] = None
102
112
  max_tokens: int = 4096
103
113
  temperature: float = 0.2
114
+ #: Strings that end the reply (v11.9.0). Optional, and absent means what it
115
+ #: always meant: generate to ``max_tokens``. The kernel sends it on exactly
116
+ #: one call — the strict verification re-ask, whose reply is one short
117
+ #: closed object — because a stop string that is safe there ("\n```") would
118
+ #: truncate any reply carrying file content. Bounded so a caller cannot make
119
+ #: the sampler check a thousand needles per token.
120
+ stop: Optional[List[str]] = Field(default=None, max_length=MAX_STOP_STRINGS)
121
+ #: Characters the reply is **forced to begin with** (v12.0.0).
122
+ #:
123
+ #: The cheapest structural guarantee the seam can offer: instead of asking
124
+ #: a weak model to start at ``{`` and repairing the markdown fence it emits
125
+ #: instead, the worker prefills those characters and the model continues
126
+ #: from them. The response's ``text`` always starts with the prefix, so a
127
+ #: caller reads one shape whether the local prefill or a cloud provider's
128
+ #: assistant-prefill message produced it.
129
+ #:
130
+ #: Bounded because it is prepended to the prompt: an unbounded prefix would
131
+ #: be an unbounded prompt on the single MLX executor. The kernel sends
132
+ #: fourteen characters.
133
+ prefix: Optional[str] = Field(default=None, max_length=MAX_PREFIX_CHARS)
104
134
 
105
135
 
106
136
  class AgentToolRequest(BaseModel):
@@ -247,6 +277,18 @@ def create_agent_worker_seam_router(
247
277
  context=req.context,
248
278
  max_tokens=req.max_tokens,
249
279
  temperature=req.temperature,
280
+ stop=req.stop or None,
281
+ prefix=req.prefix or None,
282
+ # **The seam's ``context`` is a prompt, not a corpus** (v12.0.0).
283
+ # ``_compose_system`` appends the citation mandate to any non-empty
284
+ # context, which is right for chat (where it really is retrieved
285
+ # passages) and wrong here: the Rust kernel sends its whole executor
286
+ # prompt as ``context``, so every agent turn was being told its own
287
+ # instructions were sources to "cite inline as [1], [2]". Small
288
+ # models obey that — a 0.5B wrote ``[1] …`` into a file and a 2B
289
+ # answered a tool call with the citation instruction. A caller that
290
+ # wants citations puts them in its own prompt.
291
+ cite_sources=False,
250
292
  )
251
293
  return {"text": str(text)}
252
294
 
@@ -305,6 +347,7 @@ def create_agent_worker_seam_router(
305
347
 
306
348
  __all__ = [
307
349
  "MAX_MAX_TOKENS",
350
+ "MAX_PREFIX_CHARS",
308
351
  "MAX_TEMPERATURE",
309
352
  "MIN_MAX_TOKENS",
310
353
  "MIN_TEMPERATURE",
@@ -2,15 +2,24 @@
2
2
 
3
3
  Extracted from ``server_app.py`` in v1.3.0 with the whole ``/models*`` +
4
4
  ``/engines*`` + ``/setup/set-api-key`` surface. v11.6.0 kept the eight routes
5
- that are *this interpreter's* business — loading, switching and unloading MLX
6
- models, and the resolve → download-if-consented → load → smoke prepare flow —
7
- and moved the rest to ``lattice-host``: engine installation and cloud-key
8
- verification are host operations, ``/setup/set-api-key`` is account state, and
9
- the two catalogue reads (``/models/compat-profiles``, ``/models/recommendations``)
10
- are product views.
5
+ that are *this interpreter's* business and moved the rest to ``lattice-host``:
6
+ engine installation and cloud-key verification are host operations,
7
+ ``/setup/set-api-key`` is account state, and the two catalogue reads
8
+ (``/models/compat-profiles``, ``/models/recommendations``) are product views.
11
9
 
12
- The download itself stays because it is ``huggingface_hub`` / ``ollama``
13
- running here (v11.6.0 gateway integration §4a).
10
+ v11.8.0 took three of those eight away, because nothing called them — not the
11
+ Rust surface, not the SPA client, not either extension:
12
+
13
+ * ``POST /engines/pull-model`` — every caller reaches a download through
14
+ ``/engines/prepare-model``, which resolves, downloads on consent, loads and
15
+ smoke-tests in one step. The bare pull was the older door, and with it went
16
+ this module's only use of ``huggingface_hub`` / ``ollama`` (both still run
17
+ here, under ``services/model_loading.py``, for the prepare flow).
18
+ * ``POST /models/switch/{model_id:path}`` — ``/models/load`` is what every
19
+ surface sends, and it switches as part of loading.
20
+ * ``DELETE /models/unload-all`` — unloading is per-model everywhere.
21
+
22
+ What is left is list, load, unload-one and the two prepare flows.
14
23
 
15
24
  Mirrors the established router-factory convention: the heavy provider/runtime
16
25
  helpers are injected as bound service callables. This module owns the sole
@@ -21,14 +30,13 @@ from __future__ import annotations
21
30
 
22
31
  import asyncio
23
32
  import logging
24
- import subprocess
25
33
  from typing import Any, Callable, Dict, List, NoReturn, Optional
26
34
 
27
35
  from fastapi import APIRouter, HTTPException, Request
28
36
  from fastapi.responses import StreamingResponse
29
37
  from pydantic import BaseModel
30
38
 
31
- from latticeai.core.messages import http_error, resolve_language, translate
39
+ from latticeai.core.messages import http_error, resolve_language
32
40
  from latticeai.services.model_errors import ModelRuntimeError
33
41
 
34
42
 
@@ -76,11 +84,6 @@ class LoadModelRequest(BaseModel):
76
84
 
77
85
 
78
86
 
79
- class PullModelRequest(BaseModel):
80
- model: str
81
- allow_download: bool = False
82
-
83
-
84
87
  class PrepareModelRequest(BaseModel):
85
88
  model: str
86
89
  engine: Optional[str] = None
@@ -94,13 +97,9 @@ def create_models_router(
94
97
  model_router: Any,
95
98
  require_user: Callable[[Request], str],
96
99
  require_admin: Callable[[Request], tuple],
97
- normalize_local_model_request: Callable[..., str],
98
- download_hf_model: Callable[..., Dict],
99
100
  prepare_and_load_model: Callable[..., Any],
100
101
  prepare_and_load_model_stream: Callable[..., Any],
101
102
  sse_event: Callable[[str, Dict], str],
102
- ensure_ollama_server: Callable[[], None],
103
- local_binary: Callable[[str], Optional[str]],
104
103
  engine_status: Callable[[], List[Dict]],
105
104
  filter_lower_family_versions: Callable[[List[Dict]], List[Dict]],
106
105
  list_compat_profiles: Callable[[], Any],
@@ -289,80 +288,6 @@ def create_models_router(
289
288
 
290
289
  # ── Engines ───────────────────────────────────────────────────────────
291
290
 
292
-
293
- @router.post("/engines/pull-model")
294
- async def pull_ollama_model(req: PullModelRequest, request: Request):
295
- _authorize_model_admin(request)
296
- if not req.allow_download:
297
- raise HTTPException(
298
- status_code=403,
299
- detail=translate("models.download_consent_required", resolve_language(request)),
300
- )
301
- model_ref = normalize_local_model_request(req.model, None)
302
- if not model_ref:
303
- raise http_error(400, "models.identifier_empty", resolve_language(request))
304
-
305
- if ":" in model_ref and model_ref.split(":", 1)[0].strip().lower() in {"ollama", "vllm", "lmstudio", "llamacpp", "local_mlx", "mlx"}:
306
- provider, model_name = model_ref.split(":", 1)
307
- provider = provider.strip().lower()
308
- model_name = model_name.strip()
309
- else:
310
- provider, model_name = "local_mlx", model_ref
311
-
312
- if not model_name:
313
- raise http_error(400, "models.name_empty", resolve_language(request))
314
-
315
- if provider == "ollama":
316
- try:
317
- # Starts the daemon and waits for it — blocking, like the pull
318
- # below, and for the same reason it may not run on the loop.
319
- await asyncio.to_thread(ensure_ollama_server)
320
- except ModelRuntimeError as exc:
321
- _raise_model_http(exc)
322
- ollama = local_binary("ollama")
323
- if not ollama:
324
- raise http_error(400, "models.ollama_missing", resolve_language(request))
325
- try:
326
- # A model pull is minutes of network I/O. Held on the event loop
327
- # it stalled every other request — including the health check the
328
- # UI uses to decide the server is alive — until the pull finished.
329
- completed = await asyncio.to_thread(
330
- lambda: subprocess.run(
331
- [ollama, "pull", model_name],
332
- capture_output=True, text=True, timeout=900, check=False,
333
- )
334
- )
335
- except subprocess.TimeoutExpired:
336
- raise http_error(408, "models.download_timeout", resolve_language(request))
337
- if completed.returncode != 0:
338
- raise HTTPException(
339
- status_code=500,
340
- detail=completed.stderr[-2000:]
341
- or translate("models.pull_failed", resolve_language(request)),
342
- )
343
- return {"provider": provider, "model": model_name, "returncode": completed.returncode}
344
-
345
- if provider == "lmstudio":
346
- raise HTTPException(
347
- status_code=400,
348
- detail=(
349
- "LM Studio 모델은 Lattice에서 Hugging Face로 pull하지 않습니다. "
350
- "LM Studio 앱에서 모델을 다운로드하고 Local Server를 켠 뒤 모델을 로드하세요. "
351
- "그러면 모델 선택창에 실제 /v1/models 항목이 표시됩니다."
352
- ),
353
- )
354
-
355
- if provider in {"vllm", "llamacpp", "local_mlx", "mlx"}:
356
- download_provider = "local_mlx" if provider == "mlx" else provider
357
- try:
358
- # Multi-gigabyte Hugging Face download; same rule as the pull above.
359
- result = await asyncio.to_thread(download_hf_model, model_name, download_provider)
360
- except ModelRuntimeError as exc:
361
- _raise_model_http(exc)
362
- return {"provider": provider, "model": model_name, "returncode": 0, **result}
363
-
364
- raise http_error(400, "models.download_not_automated", resolve_language(request), provider=provider) # pragma: no cover — every prefix the check above accepts is handled, and the fallback is local_mlx
365
-
366
291
  @router.post("/engines/prepare-model")
367
292
  async def engines_prepare_model(req: PrepareModelRequest, request: Request):
368
293
  current_user = _authorize_model_admin(request, req.user_email)
@@ -484,27 +409,10 @@ def create_models_router(
484
409
  detail=friendly_model_runtime_error(e, model_id=req.model_id, engine=req.engine),
485
410
  )
486
411
 
487
- @router.post("/models/switch/{model_id:path}")
488
- async def switch_model(model_id: str, request: Request):
489
- _authorize_model_admin(request)
490
- try:
491
- _router.switch_model(model_id)
492
- return {"status": "ok", "current": _router.current_model_id}
493
- except KeyError:
494
- raise http_error(404, "chat.model_not_loaded", resolve_language(request), model=model_id)
495
-
496
412
  @router.delete("/models/unload/{model_id:path}")
497
413
  async def unload_model(model_id: str, request: Request):
498
414
  _authorize_model_admin(request)
499
415
  _router.unload_model(model_id)
500
416
  return {"status": "ok", "unloaded": model_id}
501
417
 
502
- @router.delete("/models/unload-all")
503
- async def unload_all_models(request: Request):
504
- _authorize_model_admin(request)
505
- unloaded = _router.loaded_model_ids
506
- _router.unload_all()
507
- return {"status": "ok", "unloaded": unloaded}
508
-
509
-
510
418
  return router
@@ -10,6 +10,11 @@ it is the one that was asked for.
10
10
  ``GET /api/embeddings/status`` therefore reports the **embedder**, not the
11
11
  index. Index completeness is a native jobs route now, and reporting it from
12
12
  here would have meant re-opening a store the worker no longer holds.
13
+
14
+ It is now the module's only route. ``GET /api/embeddings/providers`` — the
15
+ static catalogue of provider ids and the env vars each needs — was removed in
16
+ v11.8.0: no surface in the tree asked for it, and a catalogue nothing reads is
17
+ a second place for the provider list to go stale.
13
18
  """
14
19
 
15
20
  from __future__ import annotations
@@ -18,7 +23,6 @@ from typing import Any, Callable, Dict, NoReturn, Optional
18
23
 
19
24
  from fastapi import APIRouter, HTTPException, Request
20
25
 
21
- from latticeai.core.embedding_providers import embedding_provider_profiles
22
26
  from latticeai.services.search_service import SearchService
23
27
 
24
28
 
@@ -31,8 +35,8 @@ def create_search_router(
31
35
  router = APIRouter()
32
36
 
33
37
  def _raise_embedder_error(exc: Exception) -> NoReturn:
34
- # NoReturn, not None: both handlers end with this in their except
35
- # branch, and without it each one reads as a missing return.
38
+ # NoReturn, not None: the handler ends with this in its except branch,
39
+ # and without it the function reads as a missing return.
36
40
  raise HTTPException(status_code=404, detail=str(exc)) from exc
37
41
 
38
42
  @router.get("/api/embeddings/status")
@@ -44,31 +48,4 @@ def create_search_router(
44
48
  except ValueError as exc:
45
49
  _raise_embedder_error(exc)
46
50
 
47
- @router.get("/api/embeddings/providers")
48
- async def embeddings_providers(request: Request) -> Dict[str, Any]:
49
- require_user(request)
50
- resolved = embedding_info() if embedding_info else {}
51
- profiles = resolved.get("profiles") or embedding_provider_profiles()
52
- return {
53
- "active": resolved.get("active_provider"),
54
- "requested": resolved.get("requested_provider"),
55
- "profile": resolved.get("profile") or "",
56
- "profiles": profiles,
57
- "providers": [
58
- {"id": "hash", "label": "Local hash (fallback)", "grade": "fallback",
59
- "requires": [], "detail": "Deterministic offline vectors — always available."},
60
- {"id": "mlx", "label": "MLX (Apple Silicon)", "grade": "production",
61
- "requires": ["LATTICEAI_EMBEDDING_MODEL"], "detail": "Local embedding model via MLX."},
62
- {"id": "ollama", "label": "Ollama", "grade": "production",
63
- "requires": ["LATTICEAI_EMBEDDING_MODEL", "LATTICEAI_EMBEDDING_BASE_URL"],
64
- "detail": "Local/remote Ollama embedding server."},
65
- {"id": "openai", "label": "OpenAI-compatible", "grade": "production",
66
- "requires": ["LATTICEAI_EMBEDDING_MODEL", "LATTICEAI_EMBEDDING_BASE_URL", "LATTICEAI_EMBEDDING_API_KEY"],
67
- "detail": "Any /v1/embeddings endpoint (OpenAI, LM Studio, vLLM, …)."},
68
- {"id": "custom", "label": "Custom callable", "grade": "production",
69
- "requires": ["LATTICEAI_EMBEDDING_CUSTOM_TARGET"],
70
- "detail": "User-supplied module:callable returning vectors."},
71
- ],
72
- }
73
-
74
51
  return router