ltcai 11.2.0 → 11.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -53
- package/docs/CHANGELOG.md +61 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/MULTI_AGENT_RUNTIME.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +6 -2
- package/docs/PERMISSION_MODE.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +2 -2
- package/docs/v11.3.0_PLAN.md +202 -0
- package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +287 -0
- package/lattice_brain/graph/_kg_common/extraction.py +516 -0
- package/lattice_brain/graph/_kg_common/relations.py +161 -0
- package/lattice_brain/graph/_kg_common/text.py +479 -0
- package/lattice_brain/graph/discovery_index/__init__.py +35 -0
- package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
- package/lattice_brain/graph/discovery_index/extract.py +137 -0
- package/lattice_brain/graph/discovery_index/scan.py +411 -0
- package/lattice_brain/graph/discovery_index/upsert.py +495 -0
- package/lattice_brain/graph/projection/__init__.py +42 -0
- package/lattice_brain/graph/projection/curation.py +500 -0
- package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
- package/lattice_brain/graph/retrieval/__init__.py +54 -0
- package/lattice_brain/graph/retrieval/context.py +197 -0
- package/lattice_brain/graph/retrieval/graph_view.py +319 -0
- package/lattice_brain/graph/retrieval/hybrid.py +488 -0
- package/lattice_brain/graph/retrieval/maintenance.py +121 -0
- package/lattice_brain/graph/retrieval/signals.py +95 -0
- package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
- package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
- package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
- package/lattice_brain/graph/retrieval_vector/search.py +560 -0
- package/lattice_brain/graph/retrieval_vector/status.py +374 -0
- package/lattice_brain/ingestion/__init__.py +130 -0
- package/lattice_brain/ingestion/_contract.py +90 -0
- package/lattice_brain/ingestion/constants.py +127 -0
- package/lattice_brain/ingestion/folder_scan.py +57 -0
- package/lattice_brain/ingestion/folders.py +258 -0
- package/lattice_brain/ingestion/hashing.py +26 -0
- package/lattice_brain/ingestion/jobs_api.py +107 -0
- package/lattice_brain/ingestion/models.py +80 -0
- package/lattice_brain/ingestion/pipeline.py +486 -0
- package/lattice_brain/ingestion/quality.py +209 -0
- package/lattice_brain/ingestion/routing.py +295 -0
- package/lattice_brain/multimodal/__init__.py +164 -0
- package/lattice_brain/multimodal/audio.py +77 -0
- package/lattice_brain/multimodal/common.py +118 -0
- package/lattice_brain/multimodal/images.py +498 -0
- package/lattice_brain/multimodal/ports.py +169 -0
- package/lattice_brain/multimodal/video.py +410 -0
- package/lattice_brain/portability/__init__.py +90 -0
- package/lattice_brain/portability/_contract.py +42 -0
- package/lattice_brain/portability/backups.py +338 -0
- package/lattice_brain/portability/bundles.py +136 -0
- package/lattice_brain/portability/constants.py +93 -0
- package/lattice_brain/portability/fsops.py +138 -0
- package/lattice_brain/portability/service.py +41 -0
- package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
- package/lattice_brain/runtime/__init__.py +1 -1
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/chronicle.py +63 -0
- package/latticeai/core/agent/__init__.py +93 -0
- package/latticeai/core/agent/_contract.py +79 -0
- package/latticeai/core/agent/context.py +57 -0
- package/latticeai/core/agent/deps.py +125 -0
- package/latticeai/core/agent/execution.py +622 -0
- package/latticeai/core/agent/planning.py +145 -0
- package/latticeai/core/agent/recovery.py +157 -0
- package/latticeai/core/agent/runtime.py +210 -0
- package/latticeai/core/agent/verification.py +231 -0
- package/latticeai/core/embedding_providers/__init__.py +151 -0
- package/latticeai/core/embedding_providers/base.py +199 -0
- package/latticeai/core/embedding_providers/captions.py +162 -0
- package/latticeai/core/embedding_providers/profiles.py +126 -0
- package/latticeai/core/embedding_providers/text.py +350 -0
- package/latticeai/core/embedding_providers/vision.py +352 -0
- package/latticeai/core/file_generation/__init__.py +115 -0
- package/latticeai/core/file_generation/bundles.py +76 -0
- package/latticeai/core/file_generation/extraction.py +154 -0
- package/latticeai/core/file_generation/inference.py +235 -0
- package/latticeai/core/file_generation/orchestration.py +152 -0
- package/latticeai/core/file_generation/prompting.py +117 -0
- package/latticeai/core/file_generation/repair.py +114 -0
- package/latticeai/core/file_generation/sanitize.py +61 -0
- package/latticeai/core/file_generation/validation.py +201 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +9 -0
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/integrations/telegram_bot/__init__.py +123 -0
- package/latticeai/integrations/telegram_bot/__main__.py +17 -0
- package/latticeai/integrations/telegram_bot/config.py +86 -0
- package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
- package/latticeai/integrations/telegram_bot/flows.py +478 -0
- package/latticeai/integrations/telegram_bot/helpers.py +322 -0
- package/latticeai/integrations/telegram_bot/screens.py +394 -0
- package/latticeai/models/router/__init__.py +88 -0
- package/latticeai/models/router/_contract.py +66 -0
- package/latticeai/models/router/branding.py +56 -0
- package/latticeai/models/router/catalog.py +69 -0
- package/latticeai/models/router/documents.py +199 -0
- package/latticeai/models/router/errors.py +37 -0
- package/latticeai/models/router/generation.py +258 -0
- package/latticeai/models/router/loading.py +291 -0
- package/latticeai/models/router/local_models.py +85 -0
- package/latticeai/models/router/registry.py +147 -0
- package/latticeai/runtime/build_phases/__init__.py +82 -0
- package/latticeai/runtime/build_phases/features.py +407 -0
- package/latticeai/runtime/build_phases/foundation.py +555 -0
- package/latticeai/runtime/build_phases/web.py +492 -0
- package/latticeai/runtime/runtime_context.py +1 -0
- package/latticeai/services/architecture_readiness.py +48 -19
- package/latticeai/services/brain_intelligence/__init__.py +58 -0
- package/latticeai/services/brain_intelligence/_contract.py +71 -0
- package/latticeai/services/brain_intelligence/consistency.py +193 -0
- package/latticeai/services/brain_intelligence/constants.py +47 -0
- package/latticeai/services/brain_intelligence/digest.py +258 -0
- package/latticeai/services/brain_intelligence/health.py +331 -0
- package/latticeai/services/brain_intelligence/proposals.py +264 -0
- package/latticeai/services/brain_intelligence/sampling.py +84 -0
- package/latticeai/services/brain_intelligence/service.py +48 -0
- package/latticeai/services/chronicle.py +557 -0
- package/latticeai/services/memory_service/__init__.py +52 -0
- package/latticeai/services/memory_service/_contract.py +100 -0
- package/latticeai/services/memory_service/brief.py +431 -0
- package/latticeai/services/memory_service/constants.py +57 -0
- package/latticeai/services/memory_service/maintenance.py +138 -0
- package/latticeai/services/memory_service/manager.py +186 -0
- package/latticeai/services/memory_service/proof.py +136 -0
- package/latticeai/services/memory_service/recall.py +225 -0
- package/latticeai/services/memory_service/service.py +48 -0
- package/latticeai/services/memory_service/stores.py +110 -0
- package/latticeai/services/model_runtime/__init__.py +322 -0
- package/latticeai/services/model_runtime/cloud.py +87 -0
- package/latticeai/services/model_runtime/download.py +282 -0
- package/latticeai/services/model_runtime/engines.py +341 -0
- package/latticeai/services/model_runtime/loading.py +178 -0
- package/latticeai/services/model_runtime/service.py +129 -0
- package/latticeai/services/model_runtime/state.py +131 -0
- package/latticeai/services/model_runtime/status.py +255 -0
- package/latticeai/services/product_readiness.py +15 -7
- package/latticeai/setup/wizard/__init__.py +126 -0
- package/latticeai/setup/wizard/catalog.py +172 -0
- package/latticeai/setup/wizard/detect.py +323 -0
- package/latticeai/setup/wizard/install.py +348 -0
- package/latticeai/setup/wizard/paths.py +168 -0
- package/latticeai/setup/wizard/plans.py +74 -0
- package/latticeai/setup/wizard/recommend.py +320 -0
- package/package.json +6 -2
- package/scripts/bump_version.py +14 -0
- package/scripts/capture_release_evidence.mjs +33 -21
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_i18n_namespace_coverage.mjs +41 -4
- package/scripts/check_max_file_lines.mjs +102 -0
- package/scripts/check_release_evidence_bound.mjs +30 -15
- package/scripts/check_screenshot_pixel_delta.py +34 -4
- package/scripts/check_server_i18n.mjs +1 -0
- package/scripts/generate_rust_parity_fixtures.py +562 -0
- package/scripts/lib/mock_server_fingerprint.mjs +94 -0
- package/scripts/release_screen_claims.json +31 -2
- package/src-tauri/Cargo.lock +361 -3
- package/src-tauri/Cargo.toml +6 -1
- package/src-tauri/src/backend.rs +349 -0
- package/src-tauri/src/folder.rs +33 -0
- package/src-tauri/src/main.rs +97 -399
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +41 -37
- package/static/app/assets/Act-yYpYnn0v.js +1 -0
- package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
- package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
- package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
- package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
- package/static/app/assets/Capture-CFIRsFNE.js +1 -0
- package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
- package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
- package/static/app/assets/Library-DwO3yZST.js +1 -0
- package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
- package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
- package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
- package/static/app/assets/System-DW8F-2xL.js +1 -0
- package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
- package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
- package/static/app/assets/brain-Ci1CkWjM.js +1 -0
- package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
- package/static/app/assets/circle-check-DfInj-qD.js +1 -0
- package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
- package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
- package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
- package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
- package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
- package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
- package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
- package/static/app/assets/index-_u5iUHDr.js +10 -0
- package/static/app/assets/input-B0lPdRQZ.js +1 -0
- package/static/app/assets/link-2-CoFbooHS.js +1 -0
- package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
- package/static/app/assets/primitives-DEbN-d6p.js +1 -0
- package/static/app/assets/search-BybIWPNd.js +1 -0
- package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
- package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
- package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
- package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
- package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
- package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
- package/static/app/assets/utils-BlZr7Pd4.js +4 -0
- package/static/app/assets/workspace-jJY4RuAV.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/graph/_kg_common.py +0 -1331
- package/lattice_brain/graph/discovery_index.py +0 -1141
- package/lattice_brain/graph/retrieval.py +0 -1120
- package/lattice_brain/graph/retrieval_vector.py +0 -1293
- package/lattice_brain/ingestion.py +0 -1525
- package/lattice_brain/multimodal.py +0 -1258
- package/latticeai/core/agent.py +0 -1465
- package/latticeai/core/embedding_providers.py +0 -1196
- package/latticeai/core/file_generation.py +0 -1047
- package/latticeai/integrations/telegram_bot.py +0 -1390
- package/latticeai/models/router.py +0 -1007
- package/latticeai/runtime/build_phases.py +0 -1450
- package/latticeai/services/brain_intelligence.py +0 -1083
- package/latticeai/services/memory_service.py +0 -1177
- package/latticeai/services/model_runtime.py +0 -1281
- package/latticeai/setup/wizard.py +0 -1310
- package/static/app/assets/Act-AWf0SAKp.js +0 -1
- package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
- package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
- package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
- package/static/app/assets/Capture-CqOSzyPr.js +0 -1
- package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
- package/static/app/assets/Library-CX-bbhmK.js +0 -1
- package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
- package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
- package/static/app/assets/System-Bu2t5hn1.js +0 -1
- package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
- package/static/app/assets/brain-DJMoqrwx.js +0 -1
- package/static/app/assets/index-BpYkzcVm.js +0 -10
- package/static/app/assets/input-DSlJJxRs.js +0 -1
- package/static/app/assets/primitives-BCx6TvfG.js +0 -1
- package/static/app/assets/search-Cgy8cCFJ.js +0 -1
- package/static/app/assets/utils-zqPZJxdx.js +0 -4
- package/static/app/assets/workspace-DXTihhfU.js +0 -1
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
"""Document generation — the same backends, driven by a specialized prompt.
|
|
2
|
+
|
|
3
|
+
Structurally parallel to :mod:`.generation` and deliberately separate: a
|
|
4
|
+
document run carries the caller's own system prompt, never the chat system
|
|
5
|
+
prompt, and never an image. Same executor, same stream-failure envelope, same
|
|
6
|
+
drain.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import asyncio
|
|
10
|
+
from typing import Any, AsyncIterator
|
|
11
|
+
|
|
12
|
+
from ._contract import RouterCore as _Core
|
|
13
|
+
from .branding import normalize_branding
|
|
14
|
+
from .catalog import CloudModel
|
|
15
|
+
from .errors import ModelStreamError, _stream_failure
|
|
16
|
+
from .loading import _mlx_sampler, executor
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class _DocumentMixin(_Core):
|
|
20
|
+
"""The document generation half of :class:`LLMRouter`."""
|
|
21
|
+
|
|
22
|
+
# ── Document Generation Pipeline ──────────────────────────────────────
|
|
23
|
+
|
|
24
|
+
async def generate_document(
|
|
25
|
+
self,
|
|
26
|
+
message: str,
|
|
27
|
+
system_prompt: str,
|
|
28
|
+
*,
|
|
29
|
+
max_tokens: int = 8192,
|
|
30
|
+
temperature: float = 0.3,
|
|
31
|
+
) -> str:
|
|
32
|
+
"""Generate a document using a specialized system prompt with graph context."""
|
|
33
|
+
return await self.generate_document_as(
|
|
34
|
+
None,
|
|
35
|
+
message,
|
|
36
|
+
system_prompt,
|
|
37
|
+
max_tokens=max_tokens,
|
|
38
|
+
temperature=temperature,
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
async def generate_document_as(
|
|
42
|
+
self,
|
|
43
|
+
model_id: str | None,
|
|
44
|
+
message: str,
|
|
45
|
+
system_prompt: str,
|
|
46
|
+
*,
|
|
47
|
+
max_tokens: int = 8192,
|
|
48
|
+
temperature: float = 0.3,
|
|
49
|
+
) -> str:
|
|
50
|
+
"""Generate a document with a request-scoped model."""
|
|
51
|
+
_selected, cached = self._model_snapshot(model_id)
|
|
52
|
+
if cached is None:
|
|
53
|
+
return "No model loaded."
|
|
54
|
+
|
|
55
|
+
if isinstance(cached, CloudModel):
|
|
56
|
+
return await self._cloud_generate_document(cached, message, system_prompt, max_tokens, temperature)
|
|
57
|
+
|
|
58
|
+
model, tokenizer, draft_model, loader_kind = self._unpack_local_cache(cached)
|
|
59
|
+
if hasattr(tokenizer, "apply_chat_template"):
|
|
60
|
+
try:
|
|
61
|
+
msgs = [
|
|
62
|
+
{"role": "system", "content": system_prompt},
|
|
63
|
+
{"role": "user", "content": message},
|
|
64
|
+
]
|
|
65
|
+
prompt = tokenizer.apply_chat_template(msgs, tokenize=False, add_generation_prompt=True)
|
|
66
|
+
except Exception:
|
|
67
|
+
prompt = f"<|im_start|>system\n{system_prompt}<|im_end|>\n<|im_start|>user\n{message}<|im_end|>\n<|im_start|>assistant\n"
|
|
68
|
+
else:
|
|
69
|
+
prompt = f"<|im_start|>system\n{system_prompt}<|im_end|>\n<|im_start|>user\n{message}<|im_end|>\n<|im_start|>assistant\n"
|
|
70
|
+
|
|
71
|
+
loop = asyncio.get_event_loop()
|
|
72
|
+
def _gen():
|
|
73
|
+
import mlx.core as mx # type: ignore[no-redef]
|
|
74
|
+
|
|
75
|
+
mx.set_default_device(mx.gpu) # type: ignore[arg-type]
|
|
76
|
+
if loader_kind == "mlx_vlm":
|
|
77
|
+
from mlx_vlm import generate as vlm_gen
|
|
78
|
+
return vlm_gen(model, tokenizer, prompt=prompt, image=None, max_tokens=max_tokens, sampler=_mlx_sampler(temperature), draft_model=draft_model, draft_kind="mtp")
|
|
79
|
+
from mlx_lm import generate as lm_gen
|
|
80
|
+
return lm_gen(model, tokenizer, prompt=prompt, max_tokens=max_tokens, sampler=_mlx_sampler(temperature), draft_model=draft_model)
|
|
81
|
+
result = await loop.run_in_executor(executor, _gen)
|
|
82
|
+
if hasattr(result, "text"):
|
|
83
|
+
return normalize_branding(result.text)
|
|
84
|
+
return normalize_branding(str(result))
|
|
85
|
+
|
|
86
|
+
async def _cloud_generate_document(self, cloud: CloudModel, message: str, system_prompt: str, max_tokens: int, temperature: float) -> str:
|
|
87
|
+
try:
|
|
88
|
+
response = await cloud.client.chat.completions.create(
|
|
89
|
+
model=cloud.model,
|
|
90
|
+
messages=[
|
|
91
|
+
{"role": "system", "content": system_prompt},
|
|
92
|
+
{"role": "user", "content": message},
|
|
93
|
+
],
|
|
94
|
+
max_tokens=max_tokens,
|
|
95
|
+
temperature=temperature,
|
|
96
|
+
)
|
|
97
|
+
except Exception as e:
|
|
98
|
+
raise RuntimeError(self._local_server_error_hint(cloud, e)) from e
|
|
99
|
+
return normalize_branding(response.choices[0].message.content or "")
|
|
100
|
+
|
|
101
|
+
async def stream_generate_document(
|
|
102
|
+
self,
|
|
103
|
+
message: str,
|
|
104
|
+
system_prompt: str,
|
|
105
|
+
*,
|
|
106
|
+
max_tokens: int = 8192,
|
|
107
|
+
temperature: float = 0.3,
|
|
108
|
+
) -> AsyncIterator[str]:
|
|
109
|
+
"""Stream document generation with specialized system prompt."""
|
|
110
|
+
async for chunk in self.stream_generate_document_as(
|
|
111
|
+
None,
|
|
112
|
+
message,
|
|
113
|
+
system_prompt,
|
|
114
|
+
max_tokens=max_tokens,
|
|
115
|
+
temperature=temperature,
|
|
116
|
+
):
|
|
117
|
+
yield chunk
|
|
118
|
+
|
|
119
|
+
async def stream_generate_document_as(
|
|
120
|
+
self,
|
|
121
|
+
model_id: str | None,
|
|
122
|
+
message: str,
|
|
123
|
+
system_prompt: str,
|
|
124
|
+
*,
|
|
125
|
+
max_tokens: int = 8192,
|
|
126
|
+
temperature: float = 0.3,
|
|
127
|
+
) -> AsyncIterator[str]:
|
|
128
|
+
"""Stream a document with a request-scoped model."""
|
|
129
|
+
_selected, cached = self._model_snapshot(model_id)
|
|
130
|
+
if cached is None:
|
|
131
|
+
yield "No model loaded."
|
|
132
|
+
return
|
|
133
|
+
|
|
134
|
+
if isinstance(cached, CloudModel):
|
|
135
|
+
async for chunk in self._cloud_stream_document(cached, message, system_prompt, max_tokens, temperature):
|
|
136
|
+
yield chunk
|
|
137
|
+
return
|
|
138
|
+
|
|
139
|
+
model, tokenizer, draft_model, loader_kind = self._unpack_local_cache(cached)
|
|
140
|
+
if hasattr(tokenizer, "apply_chat_template"):
|
|
141
|
+
try:
|
|
142
|
+
msgs = [
|
|
143
|
+
{"role": "system", "content": system_prompt},
|
|
144
|
+
{"role": "user", "content": message},
|
|
145
|
+
]
|
|
146
|
+
prompt = tokenizer.apply_chat_template(msgs, tokenize=False, add_generation_prompt=True)
|
|
147
|
+
except Exception:
|
|
148
|
+
prompt = f"<|im_start|>system\n{system_prompt}<|im_end|>\n<|im_start|>user\n{message}<|im_end|>\n<|im_start|>assistant\n"
|
|
149
|
+
else:
|
|
150
|
+
prompt = f"<|im_start|>system\n{system_prompt}<|im_end|>\n<|im_start|>user\n{message}<|im_end|>\n<|im_start|>assistant\n"
|
|
151
|
+
|
|
152
|
+
loop = asyncio.get_event_loop()
|
|
153
|
+
queue: "asyncio.Queue[Any]" = asyncio.Queue()
|
|
154
|
+
|
|
155
|
+
def _stream():
|
|
156
|
+
import mlx.core as mx # type: ignore[no-redef]
|
|
157
|
+
|
|
158
|
+
mx.set_default_device(mx.gpu) # type: ignore[arg-type]
|
|
159
|
+
try:
|
|
160
|
+
if loader_kind == "mlx_vlm":
|
|
161
|
+
from mlx_vlm import stream_generate as vlm_stream
|
|
162
|
+
gen = vlm_stream(model, tokenizer, prompt=prompt, image=None, max_tokens=max_tokens, sampler=_mlx_sampler(temperature), draft_model=draft_model, draft_kind="mtp")
|
|
163
|
+
else:
|
|
164
|
+
from mlx_lm import stream_generate as lm_stream
|
|
165
|
+
gen = lm_stream(model, tokenizer, prompt=prompt, max_tokens=max_tokens, sampler=_mlx_sampler(temperature), draft_model=draft_model)
|
|
166
|
+
for chunk in gen:
|
|
167
|
+
text = chunk.text if hasattr(chunk, "text") else (chunk[0] if isinstance(chunk, tuple) else str(chunk))
|
|
168
|
+
loop.call_soon_threadsafe(queue.put_nowait, text)
|
|
169
|
+
except Exception as exc:
|
|
170
|
+
loop.call_soon_threadsafe(
|
|
171
|
+
queue.put_nowait, _stream_failure("MLX document stream failed", exc)
|
|
172
|
+
)
|
|
173
|
+
finally:
|
|
174
|
+
loop.call_soon_threadsafe(queue.put_nowait, None)
|
|
175
|
+
|
|
176
|
+
loop.run_in_executor(executor, _stream)
|
|
177
|
+
async for chunk in self._drain_stream_queue(queue):
|
|
178
|
+
yield chunk
|
|
179
|
+
|
|
180
|
+
async def _cloud_stream_document(self, cloud: CloudModel, message: str, system_prompt: str, max_tokens: int, temperature: float) -> AsyncIterator[str]:
|
|
181
|
+
try:
|
|
182
|
+
stream = await cloud.client.chat.completions.create(
|
|
183
|
+
model=cloud.model,
|
|
184
|
+
messages=[
|
|
185
|
+
{"role": "system", "content": system_prompt},
|
|
186
|
+
{"role": "user", "content": message},
|
|
187
|
+
],
|
|
188
|
+
max_tokens=max_tokens,
|
|
189
|
+
temperature=temperature,
|
|
190
|
+
stream=True,
|
|
191
|
+
)
|
|
192
|
+
except Exception as exc:
|
|
193
|
+
raise ModelStreamError(self._local_server_error_hint(cloud, exc)) from exc
|
|
194
|
+
async for event in stream:
|
|
195
|
+
if not event.choices:
|
|
196
|
+
continue
|
|
197
|
+
delta = event.choices[0].delta.content
|
|
198
|
+
if delta:
|
|
199
|
+
yield normalize_branding(delta)
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""A backend that failed mid-stream produced an error, never model output.
|
|
2
|
+
|
|
3
|
+
Streaming backends used to hand their failure to the caller as a chunk of text,
|
|
4
|
+
which every consumer then treated as the model's answer. The failure now
|
|
5
|
+
travels as a typed exception, and ``_stream_failure`` is how a worker thread —
|
|
6
|
+
which cannot raise into the consuming coroutine — puts one on the chunk queue.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class ModelStreamError(RuntimeError):
|
|
11
|
+
"""A backend failed mid-stream. This is an error, never model output.
|
|
12
|
+
|
|
13
|
+
Streaming backends used to hand their failure to the caller as a chunk of
|
|
14
|
+
text (``"⚠️ Error: ..."``), which every consumer then treated as the
|
|
15
|
+
model's answer: it was echoed to the client as content and persisted as a
|
|
16
|
+
successful turn. The failure now travels as this typed exception instead.
|
|
17
|
+
|
|
18
|
+
The MLX generators run on a worker thread that cannot raise into the
|
|
19
|
+
consuming coroutine, so the thread puts an instance on the chunk queue and
|
|
20
|
+
:meth:`LLMRouter._drain_stream_queue` re-raises it. The SSE endpoints
|
|
21
|
+
(``latticeai.api.chat_stream.stream_chat`` and the document stream in
|
|
22
|
+
``latticeai.api.chat_documents``) already wrap their ``async for`` in
|
|
23
|
+
``except Exception`` and emit an ``error`` frame plus a ``[stream_error]``
|
|
24
|
+
marker on the persisted answer, so the stream framing is unchanged.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _stream_failure(stage: str, exc: BaseException) -> ModelStreamError:
|
|
29
|
+
"""Envelope a backend exception for transport across the chunk queue.
|
|
30
|
+
|
|
31
|
+
``raise ... from exc`` is unavailable on the worker thread (nothing there
|
|
32
|
+
consumes the traceback), so the cause is attached explicitly and stays
|
|
33
|
+
visible in logs when the consumer re-raises.
|
|
34
|
+
"""
|
|
35
|
+
error = ModelStreamError(f"{stage}: {exc}")
|
|
36
|
+
error.__cause__ = exc
|
|
37
|
+
return error
|
|
@@ -0,0 +1,258 @@
|
|
|
1
|
+
"""Chat generation — one answer, or a stream of tokens, local or cloud.
|
|
2
|
+
|
|
3
|
+
The MLX generators run on the dedicated single-thread executor so GPU streams
|
|
4
|
+
match across a request; they cannot raise into the consuming coroutine, so a
|
|
5
|
+
failure is enveloped onto the chunk queue and :meth:`_drain_stream_queue`
|
|
6
|
+
re-raises it. That is the whole reason a backend failure can no longer be
|
|
7
|
+
mistaken for the model's answer.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import asyncio
|
|
11
|
+
import base64
|
|
12
|
+
import io
|
|
13
|
+
from typing import Any, AsyncIterator, Optional
|
|
14
|
+
|
|
15
|
+
from PIL import Image
|
|
16
|
+
|
|
17
|
+
from latticeai.core.quiet import quiet
|
|
18
|
+
|
|
19
|
+
from ._contract import RouterCore as _Core
|
|
20
|
+
from .branding import SYSTEM_PROMPT, _compose_system, normalize_branding
|
|
21
|
+
from .catalog import CloudModel
|
|
22
|
+
from .errors import ModelStreamError, _stream_failure
|
|
23
|
+
from .loading import _mlx_sampler, executor
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class _GenerationMixin(_Core):
|
|
27
|
+
"""The chat generation half of :class:`LLMRouter`."""
|
|
28
|
+
|
|
29
|
+
def _build_prompt(self, message: str, context: Optional[str], tokenizer) -> str:
|
|
30
|
+
context = normalize_branding(context)
|
|
31
|
+
system = _compose_system(SYSTEM_PROMPT, context)
|
|
32
|
+
if hasattr(tokenizer, "apply_chat_template"):
|
|
33
|
+
try:
|
|
34
|
+
msgs = [{"role": "system", "content": system}, {"role": "user", "content": message}]
|
|
35
|
+
return tokenizer.apply_chat_template(msgs, tokenize=False, add_generation_prompt=True)
|
|
36
|
+
except Exception:
|
|
37
|
+
quiet()
|
|
38
|
+
return f"<|im_start|>system\n{system}<|im_end|>\n<|im_start|>user\n{message}<|im_end|>\n<|im_start|>assistant\n"
|
|
39
|
+
|
|
40
|
+
def _build_vlm_prompt(self, model, processor, message: str, context: Optional[str], num_images: int) -> str:
|
|
41
|
+
context = normalize_branding(context)
|
|
42
|
+
system = _compose_system(SYSTEM_PROMPT, context)
|
|
43
|
+
try:
|
|
44
|
+
from mlx_vlm import apply_chat_template
|
|
45
|
+
|
|
46
|
+
return apply_chat_template(
|
|
47
|
+
processor,
|
|
48
|
+
model.config,
|
|
49
|
+
[
|
|
50
|
+
{"role": "system", "content": system},
|
|
51
|
+
{"role": "user", "content": message},
|
|
52
|
+
],
|
|
53
|
+
add_generation_prompt=True,
|
|
54
|
+
num_images=num_images,
|
|
55
|
+
)
|
|
56
|
+
except Exception as e:
|
|
57
|
+
print(f"⚠️ VLM chat template fallback: {e}")
|
|
58
|
+
return self._build_prompt(message, context, processor)
|
|
59
|
+
|
|
60
|
+
async def generate_as(
|
|
61
|
+
self,
|
|
62
|
+
model_id: str | None,
|
|
63
|
+
message: str,
|
|
64
|
+
context: Optional[str] = None,
|
|
65
|
+
max_tokens: int = 4096,
|
|
66
|
+
temperature: float = 0.2,
|
|
67
|
+
image_data: Optional[str] = None,
|
|
68
|
+
) -> str:
|
|
69
|
+
"""Generate with a request-scoped model without changing the default."""
|
|
70
|
+
_selected, cached = self._model_snapshot(model_id)
|
|
71
|
+
if cached is None:
|
|
72
|
+
return "No model."
|
|
73
|
+
return await self._generate_cached(cached, message, context, max_tokens, temperature, image_data)
|
|
74
|
+
|
|
75
|
+
async def generate(
|
|
76
|
+
self,
|
|
77
|
+
message: str,
|
|
78
|
+
context: Optional[str] = None,
|
|
79
|
+
max_tokens: int = 4096,
|
|
80
|
+
temperature: float = 0.2,
|
|
81
|
+
image_data: Optional[str] = None,
|
|
82
|
+
) -> str:
|
|
83
|
+
return await self.generate_as(None, message, context, max_tokens, temperature, image_data)
|
|
84
|
+
|
|
85
|
+
async def _generate_cached(
|
|
86
|
+
self,
|
|
87
|
+
cached: object,
|
|
88
|
+
message: str,
|
|
89
|
+
context: Optional[str],
|
|
90
|
+
max_tokens: int,
|
|
91
|
+
temperature: float,
|
|
92
|
+
image_data: Optional[str],
|
|
93
|
+
) -> str:
|
|
94
|
+
if isinstance(cached, CloudModel):
|
|
95
|
+
return await self._cloud_generate(cached, message, context, max_tokens, temperature)
|
|
96
|
+
|
|
97
|
+
model, tokenizer, draft_model, loader_kind = self._unpack_local_cache(cached)
|
|
98
|
+
use_vlm = loader_kind == "mlx_vlm"
|
|
99
|
+
prompt = (
|
|
100
|
+
self._build_vlm_prompt(model, tokenizer, message, context, 1 if image_data else 0)
|
|
101
|
+
if use_vlm
|
|
102
|
+
else self._build_prompt(message, context, tokenizer)
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
loop = asyncio.get_event_loop()
|
|
106
|
+
|
|
107
|
+
def _gen():
|
|
108
|
+
import mlx.core as mx # type: ignore[no-redef]
|
|
109
|
+
|
|
110
|
+
mx.set_default_device(mx.gpu) # type: ignore[arg-type]
|
|
111
|
+
if use_vlm:
|
|
112
|
+
from mlx_vlm import generate as vlm_gen
|
|
113
|
+
return vlm_gen(model, tokenizer, prompt=prompt, image=self._prep_image(image_data) if image_data else None, max_tokens=max_tokens, sampler=_mlx_sampler(temperature), draft_model=draft_model, draft_kind="mtp")
|
|
114
|
+
from mlx_lm import generate as lm_gen
|
|
115
|
+
return lm_gen(model, tokenizer, prompt=prompt, max_tokens=max_tokens, sampler=_mlx_sampler(temperature), draft_model=draft_model)
|
|
116
|
+
result = await loop.run_in_executor(executor, _gen)
|
|
117
|
+
# mlx-vlm might return a GenerationResult object; extract the text
|
|
118
|
+
if hasattr(result, "text"):
|
|
119
|
+
return normalize_branding(result.text)
|
|
120
|
+
return normalize_branding(str(result))
|
|
121
|
+
|
|
122
|
+
async def _cloud_generate(self, cloud: CloudModel, message: str, context: Optional[str], max_tokens: int, temperature: float) -> str:
|
|
123
|
+
context = normalize_branding(context)
|
|
124
|
+
system = _compose_system(SYSTEM_PROMPT, context)
|
|
125
|
+
try:
|
|
126
|
+
response = await cloud.client.chat.completions.create(
|
|
127
|
+
model=cloud.model,
|
|
128
|
+
messages=[
|
|
129
|
+
{"role": "system", "content": system},
|
|
130
|
+
{"role": "user", "content": message},
|
|
131
|
+
],
|
|
132
|
+
max_tokens=max_tokens,
|
|
133
|
+
temperature=temperature,
|
|
134
|
+
)
|
|
135
|
+
except Exception as e:
|
|
136
|
+
raise RuntimeError(self._local_server_error_hint(cloud, e)) from e
|
|
137
|
+
return normalize_branding(response.choices[0].message.content or "")
|
|
138
|
+
|
|
139
|
+
async def stream_generate_as(
|
|
140
|
+
self,
|
|
141
|
+
model_id: str | None,
|
|
142
|
+
message: str,
|
|
143
|
+
context: Optional[str] = None,
|
|
144
|
+
max_tokens: int = 4096,
|
|
145
|
+
temperature: float = 0.2,
|
|
146
|
+
image_data: Optional[str] = None,
|
|
147
|
+
) -> AsyncIterator[str]:
|
|
148
|
+
"""Stream with a request-scoped model without changing the default."""
|
|
149
|
+
_selected, cached = self._model_snapshot(model_id)
|
|
150
|
+
if cached is None:
|
|
151
|
+
yield "No model."
|
|
152
|
+
return
|
|
153
|
+
if isinstance(cached, CloudModel):
|
|
154
|
+
async for chunk in self._cloud_stream_generate(cached, message, context, max_tokens, temperature):
|
|
155
|
+
yield chunk
|
|
156
|
+
return
|
|
157
|
+
|
|
158
|
+
model, tokenizer, draft_model, loader_kind = self._unpack_local_cache(cached)
|
|
159
|
+
use_vlm = loader_kind == "mlx_vlm"
|
|
160
|
+
prompt = (
|
|
161
|
+
self._build_vlm_prompt(model, tokenizer, message, context, 1 if image_data else 0)
|
|
162
|
+
if use_vlm
|
|
163
|
+
else self._build_prompt(message, context, tokenizer)
|
|
164
|
+
)
|
|
165
|
+
loop = asyncio.get_event_loop()
|
|
166
|
+
queue: "asyncio.Queue[Any]" = asyncio.Queue()
|
|
167
|
+
|
|
168
|
+
def _stream():
|
|
169
|
+
import mlx.core as mx # type: ignore[no-redef]
|
|
170
|
+
|
|
171
|
+
mx.set_default_device(mx.gpu) # type: ignore[arg-type]
|
|
172
|
+
try:
|
|
173
|
+
if use_vlm:
|
|
174
|
+
from mlx_vlm import stream_generate as vlm_stream
|
|
175
|
+
gen = vlm_stream(model, tokenizer, prompt=prompt, image=self._prep_image(image_data) if image_data else None, max_tokens=max_tokens, sampler=_mlx_sampler(temperature), draft_model=draft_model, draft_kind="mtp")
|
|
176
|
+
else:
|
|
177
|
+
from mlx_lm import stream_generate as lm_stream
|
|
178
|
+
gen = lm_stream(model, tokenizer, prompt=prompt, max_tokens=max_tokens, sampler=_mlx_sampler(temperature), draft_model=draft_model)
|
|
179
|
+
|
|
180
|
+
for chunk in gen:
|
|
181
|
+
text = chunk.text if hasattr(chunk, "text") else (chunk[0] if isinstance(chunk, tuple) else str(chunk))
|
|
182
|
+
loop.call_soon_threadsafe(queue.put_nowait, text)
|
|
183
|
+
except Exception as exc:
|
|
184
|
+
loop.call_soon_threadsafe(
|
|
185
|
+
queue.put_nowait, _stream_failure("MLX chat stream failed", exc)
|
|
186
|
+
)
|
|
187
|
+
finally:
|
|
188
|
+
loop.call_soon_threadsafe(queue.put_nowait, None)
|
|
189
|
+
|
|
190
|
+
loop.run_in_executor(executor, _stream)
|
|
191
|
+
async for chunk in self._drain_stream_queue(queue):
|
|
192
|
+
yield chunk
|
|
193
|
+
|
|
194
|
+
@staticmethod
|
|
195
|
+
async def _drain_stream_queue(queue: "asyncio.Queue[Any]") -> AsyncIterator[str]:
|
|
196
|
+
"""Yield worker-thread chunks until the terminator; raise failures.
|
|
197
|
+
|
|
198
|
+
``None`` terminates the stream. A :class:`ModelStreamError` on the
|
|
199
|
+
queue is a backend failure envelope, not model text, so it is raised
|
|
200
|
+
into the consuming coroutine — callers must never be able to mistake
|
|
201
|
+
it for an answer.
|
|
202
|
+
"""
|
|
203
|
+
while True:
|
|
204
|
+
chunk = await queue.get()
|
|
205
|
+
if chunk is None:
|
|
206
|
+
return
|
|
207
|
+
if isinstance(chunk, ModelStreamError):
|
|
208
|
+
raise chunk
|
|
209
|
+
yield normalize_branding(chunk)
|
|
210
|
+
|
|
211
|
+
async def stream_generate(
|
|
212
|
+
self,
|
|
213
|
+
message: str,
|
|
214
|
+
context: Optional[str] = None,
|
|
215
|
+
max_tokens: int = 4096,
|
|
216
|
+
temperature: float = 0.2,
|
|
217
|
+
image_data: Optional[str] = None,
|
|
218
|
+
) -> AsyncIterator[str]:
|
|
219
|
+
async for chunk in self.stream_generate_as(
|
|
220
|
+
None, message, context, max_tokens, temperature, image_data
|
|
221
|
+
):
|
|
222
|
+
yield chunk
|
|
223
|
+
|
|
224
|
+
async def _cloud_stream_generate(self, cloud: CloudModel, message: str, context: Optional[str], max_tokens: int, temperature: float) -> AsyncIterator[str]:
|
|
225
|
+
context = normalize_branding(context)
|
|
226
|
+
system = _compose_system(SYSTEM_PROMPT, context)
|
|
227
|
+
try:
|
|
228
|
+
stream = await cloud.client.chat.completions.create(
|
|
229
|
+
model=cloud.model,
|
|
230
|
+
messages=[
|
|
231
|
+
{"role": "system", "content": system},
|
|
232
|
+
{"role": "user", "content": message},
|
|
233
|
+
],
|
|
234
|
+
max_tokens=max_tokens,
|
|
235
|
+
temperature=temperature,
|
|
236
|
+
stream=True,
|
|
237
|
+
)
|
|
238
|
+
except Exception as exc:
|
|
239
|
+
# Same invariant as the MLX path: a backend that never produced a
|
|
240
|
+
# token failed, and that is an error — not the model's answer.
|
|
241
|
+
raise ModelStreamError(self._local_server_error_hint(cloud, exc)) from exc
|
|
242
|
+
async for event in stream:
|
|
243
|
+
if not event.choices:
|
|
244
|
+
continue
|
|
245
|
+
delta = event.choices[0].delta.content
|
|
246
|
+
if delta:
|
|
247
|
+
yield normalize_branding(delta)
|
|
248
|
+
|
|
249
|
+
def _prep_image(self, image_data: Optional[str]) -> Optional[Image.Image]:
|
|
250
|
+
if not image_data:
|
|
251
|
+
return None
|
|
252
|
+
try:
|
|
253
|
+
image = Image.open(io.BytesIO(base64.b64decode(image_data))).convert("RGB")
|
|
254
|
+
print(f"🖼️ VLM image decoded: {image.width}x{image.height}")
|
|
255
|
+
return image
|
|
256
|
+
except Exception as e:
|
|
257
|
+
print(f"⚠️ VLM image decode failed: {e}")
|
|
258
|
+
return None
|