ltcai 11.0.1 → 11.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (140) hide show
  1. package/README.md +55 -43
  2. package/docs/CHANGELOG.md +61 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/FEATURE_AUDIT_v11.2.0.md +393 -0
  6. package/docs/LAYOUT_REBUILD_SPEC.md +9 -1
  7. package/docs/ONBOARDING.md +1 -1
  8. package/docs/OPERATIONS.md +1 -1
  9. package/docs/PERFORMANCE.md +71 -18
  10. package/docs/TRUST_MODEL.md +1 -1
  11. package/docs/WHY_LATTICE.md +1 -1
  12. package/docs/architecture.md +6 -2
  13. package/docs/kg-schema.md +1 -1
  14. package/lattice_brain/__init__.py +1 -1
  15. package/lattice_brain/embeddings.py +12 -37
  16. package/lattice_brain/gates.py +125 -0
  17. package/lattice_brain/graph/discovery_index.py +30 -32
  18. package/lattice_brain/graph/fusion.py +35 -4
  19. package/lattice_brain/graph/image_vectors.py +230 -0
  20. package/lattice_brain/graph/ingest.py +11 -5
  21. package/lattice_brain/graph/projection.py +66 -8
  22. package/lattice_brain/graph/provenance.py +27 -2
  23. package/lattice_brain/graph/retrieval.py +113 -2
  24. package/lattice_brain/graph/retrieval_docgen.py +6 -6
  25. package/lattice_brain/graph/schema.py +18 -0
  26. package/lattice_brain/graph/store.py +9 -0
  27. package/lattice_brain/graph/vector_index/selector.py +32 -2
  28. package/lattice_brain/ingestion.py +363 -10
  29. package/lattice_brain/multimodal.py +1258 -0
  30. package/lattice_brain/portability.py +169 -32
  31. package/lattice_brain/runtime/multi_agent.py +1 -1
  32. package/lattice_brain/sealed_box.py +244 -0
  33. package/lattice_brain/self_model.py +77 -22
  34. package/lattice_brain/synthesis.py +24 -1
  35. package/latticeai/__init__.py +1 -1
  36. package/latticeai/api/brain_intelligence.py +4 -0
  37. package/latticeai/api/chat.py +11 -0
  38. package/latticeai/api/chat_helpers.py +16 -3
  39. package/latticeai/api/chat_hybrid.py +32 -1
  40. package/latticeai/api/features.py +70 -0
  41. package/latticeai/api/local_files.py +102 -0
  42. package/latticeai/api/memory.py +128 -1
  43. package/latticeai/api/portability.py +39 -4
  44. package/latticeai/api/review_queue.py +126 -0
  45. package/latticeai/api/search.py +16 -2
  46. package/latticeai/core/agent.py +59 -2
  47. package/latticeai/core/agent_prompts.py +66 -0
  48. package/latticeai/core/config.py +4 -1
  49. package/latticeai/core/context_builder.py +98 -11
  50. package/latticeai/core/embedding_providers.py +528 -0
  51. package/latticeai/core/legacy_compatibility.py +1 -1
  52. package/latticeai/core/marketplace.py +1 -1
  53. package/latticeai/core/messages.py +180 -0
  54. package/latticeai/core/model_compat.py +73 -2
  55. package/latticeai/core/workspace_os.py +43 -0
  56. package/latticeai/core/workspace_os_constants.py +1 -1
  57. package/latticeai/core/workspace_reorganization.py +335 -0
  58. package/latticeai/models/model_providers.py +12 -4
  59. package/latticeai/runtime/build_phases.py +33 -2
  60. package/latticeai/runtime/chat_wiring.py +4 -0
  61. package/latticeai/runtime/feature_toggle_wiring.py +163 -0
  62. package/latticeai/runtime/persistence_runtime.py +41 -4
  63. package/latticeai/runtime/router_registration.py +11 -0
  64. package/latticeai/runtime/runtime_context.py +1 -0
  65. package/latticeai/services/app_context.py +8 -0
  66. package/latticeai/services/architecture_readiness.py +1 -1
  67. package/latticeai/services/automation_intelligence.py +22 -2
  68. package/latticeai/services/brain_intelligence.py +123 -7
  69. package/latticeai/services/change_proposals.py +50 -10
  70. package/latticeai/services/command_center.py +10 -4
  71. package/latticeai/services/feature_toggles.py +502 -0
  72. package/latticeai/services/folder_watch.py +122 -1
  73. package/latticeai/services/hybrid_chat.py +56 -5
  74. package/latticeai/services/interop_bridges.py +978 -0
  75. package/latticeai/services/memory_service.py +34 -0
  76. package/latticeai/services/model_capability_registry.py +434 -261
  77. package/latticeai/services/model_catalog.py +95 -61
  78. package/latticeai/services/model_recommendation.py +18 -11
  79. package/latticeai/services/model_runtime.py +1 -1
  80. package/latticeai/services/multimodal_ports.py +112 -0
  81. package/latticeai/services/obsidian_bridge.py +16 -25
  82. package/latticeai/services/product_readiness.py +1 -1
  83. package/latticeai/services/search_service.py +149 -2
  84. package/latticeai/services/self_model_service.py +171 -0
  85. package/latticeai/services/tool_dispatch.py +4 -0
  86. package/latticeai/services/voice_capture.py +27 -1
  87. package/latticeai/setup/auto_setup.py +27 -30
  88. package/latticeai/setup/wizard.py +77 -44
  89. package/package.json +1 -1
  90. package/scripts/check_current_release_docs.mjs +1 -1
  91. package/scripts/check_server_i18n.mjs +1 -0
  92. package/scripts/release_screen_claims.json +22 -0
  93. package/scripts/verify_hf_model_registry.py +253 -218
  94. package/src-tauri/Cargo.lock +1 -1
  95. package/src-tauri/Cargo.toml +1 -1
  96. package/src-tauri/tauri.conf.json +1 -1
  97. package/static/app/asset-manifest.json +37 -37
  98. package/static/app/assets/{Act-D4zSxFR-.js → Act-AWf0SAKp.js} +1 -1
  99. package/static/app/assets/{AdminConsole-w5jBfPt2.js → AdminConsole-D0u8Tiyj.js} +1 -1
  100. package/static/app/assets/{Brain-C2EqQg74.js → Brain-tuhI4sOC.js} +1 -1
  101. package/static/app/assets/BrainHome-Ts7G_Ila.js +2 -0
  102. package/static/app/assets/BrainSignals-jMYgQ2Ar.js +1 -0
  103. package/static/app/assets/{Capture-DPqpGK8d.js → Capture-CqOSzyPr.js} +1 -1
  104. package/static/app/assets/{CommandPalette-CNf7h5fp.js → CommandPalette-DC0Bzh-I.js} +1 -1
  105. package/static/app/assets/{Library-BN0HYOfc.js → Library-CX-bbhmK.js} +1 -1
  106. package/static/app/assets/{LivingBrain-Dfq_wEDI.js → LivingBrain-DBwhto14.js} +1 -1
  107. package/static/app/assets/{ProductFlow-B-3O0rNV.js → ProductFlow-BHA2cfKI.js} +1 -1
  108. package/static/app/assets/{ReviewCard-gZ-tdqFM.js → ReviewCard-BUhCKRNM.js} +1 -1
  109. package/static/app/assets/{System-BElUcSSw.js → System-Bu2t5hn1.js} +1 -1
  110. package/static/app/assets/arrow-left-Dzwa5zRb.js +1 -0
  111. package/static/app/assets/{bot--qYHMtkP.js → bot-Cia42c2h.js} +1 -1
  112. package/static/app/assets/brain-DJMoqrwx.js +1 -0
  113. package/static/app/assets/{button-51Z3rsuv.js → button-2j2Ijzgq.js} +1 -1
  114. package/static/app/assets/{circle-pause-CMIiMaQl.js → circle-pause-BEFeWpVW.js} +1 -1
  115. package/static/app/assets/{circle-play-DZoO_cfG.js → circle-play-ujXMcHxl.js} +1 -1
  116. package/static/app/assets/{cpu-Bs6uc9W9.js → cpu-k4awryFq.js} +1 -1
  117. package/static/app/assets/{download-G-2olkWz.js → download-DFbLJ_ig.js} +1 -1
  118. package/static/app/assets/{folder-open-CTOspnmb.js → folder-open-7y_b6xkM.js} +1 -1
  119. package/static/app/assets/{hard-drive-CewHWJhn.js → hard-drive-Bidh02Kr.js} +1 -1
  120. package/static/app/assets/{index-D7Rr-J2Y.js → index-BpYkzcVm.js} +3 -3
  121. package/static/app/assets/index-DwDl9-8Y.css +2 -0
  122. package/static/app/assets/{input-D4w_BZWl.js → input-DSlJJxRs.js} +1 -1
  123. package/static/app/assets/{permissionCopy-CosBEXAZ.js → permissionCopy-Bpb83Hx9.js} +1 -1
  124. package/static/app/assets/{primitives-d0g9pvzS.js → primitives-BCx6TvfG.js} +1 -1
  125. package/static/app/assets/search-Cgy8cCFJ.js +1 -0
  126. package/static/app/assets/{share-2-NmD7e_oV.js → share-2-BH1M-WNi.js} +1 -1
  127. package/static/app/assets/{shield-alert-CcQeMuju.js → shield-alert-BlKdBXcG.js} +1 -1
  128. package/static/app/assets/{textarea-BPAJDc-0.js → textarea-CCWbUfFB.js} +1 -1
  129. package/static/app/assets/{useFocusTrap-C7YLdTBC.js → useFocusTrap-YdHQ7pJ1.js} +1 -1
  130. package/static/app/assets/{useQuery-DRyD9opW.js → useQuery-CXQiwbVT.js} +1 -1
  131. package/static/app/assets/{utils-DG1_ExrP.js → utils-zqPZJxdx.js} +2 -2
  132. package/static/app/assets/{workspace-CWVf3gsI.js → workspace-DXTihhfU.js} +1 -1
  133. package/static/app/index.html +4 -4
  134. package/static/sw.js +1 -1
  135. package/static/app/assets/BrainHome-CvXS6XiQ.js +0 -2
  136. package/static/app/assets/BrainSignals-DOE_KhOU.js +0 -1
  137. package/static/app/assets/arrow-left-CFNIMjhv.js +0 -1
  138. package/static/app/assets/brain-DDCLjRqO.js +0 -1
  139. package/static/app/assets/index-CkzokZAj.css +0 -2
  140. package/static/app/assets/search-BLCYt75v.js +0 -1
@@ -128,6 +128,17 @@ class NodeType(str, Enum):
128
128
  PREFERENCE = "PREFERENCE" # 선호 (좋아함/싫어함/선호 방식)
129
129
  HABIT = "HABIT" # 습관 (반복되는 행동)
130
130
  RELATIONSHIP = "RELATIONSHIP" # 관계 (동료/친구/멘토 등)
131
+ # v11.1.0 Multi-modal (Track 3) — a recording is its own noun. It used to
132
+ # enter through the text door as a DOCUMENT because "a transcript *is*
133
+ # text"; that is true of the transcript and false of the recording, which
134
+ # exists whether or not anyone could hear it. IMAGE has been first-class
135
+ # since 3.6.0 for the same reason, and AUDIO now sits beside it.
136
+ AUDIO = "AUDIO" # 녹음 / 음성 메모
137
+ # v11.2.0 — a video is its own noun for the same reason a recording is:
138
+ # keyframes and subtitles are *derived* from it (IMAGE children, text
139
+ # chunks), and calling the thing they came from a DOCUMENT would make the
140
+ # source of a memory indistinguishable from one of its pieces.
141
+ VIDEO = "VIDEO" # 영상 / 화면 녹화
131
142
 
132
143
  @classmethod
133
144
  def from_legacy(cls, label: str) -> "NodeType":
@@ -276,6 +287,13 @@ _LEGACY_NODE_MAP: Dict[str, NodeType] = {
276
287
  "relationship": NodeType.RELATIONSHIP,
277
288
  "관계": NodeType.RELATIONSHIP,
278
289
  "결정": NodeType.DECISION,
290
+ # v11.1.0 Multi-modal — recordings as a first-class noun.
291
+ "audio": NodeType.AUDIO,
292
+ "오디오": NodeType.AUDIO,
293
+ # v11.2.0 — videos likewise.
294
+ "video": NodeType.VIDEO,
295
+ "영상": NodeType.VIDEO,
296
+ "동영상": NodeType.VIDEO,
279
297
  }
280
298
 
281
299
  _LEGACY_EDGE_MAP: Dict[str, EdgeType] = {
@@ -241,6 +241,15 @@ class KnowledgeGraphStore(
241
241
  CREATE INDEX IF NOT EXISTS idx_vector_embeddings_type ON vector_embeddings(item_type);
242
242
  CREATE INDEX IF NOT EXISTS idx_vector_embeddings_source ON vector_embeddings(source_node);
243
243
  CREATE INDEX IF NOT EXISTS idx_vector_embeddings_model ON vector_embeddings(embedding_model);
244
+ -- v11.2.0: the ANN sidecar's freshness fingerprint is
245
+ -- COUNT(*) + MAX(indexed_at) filtered by (embedding_model,
246
+ -- embedding_dim), and it runs on *every* approximate
247
+ -- search. Against the single-column model index that was a
248
+ -- table scan of every row for the current model; this
249
+ -- covering index answers both aggregates from the index
250
+ -- alone. Additive and idempotent — no data is rewritten.
251
+ CREATE INDEX IF NOT EXISTS idx_vector_embeddings_model_dim_indexed
252
+ ON vector_embeddings(embedding_model, embedding_dim, indexed_at);
244
253
  CREATE INDEX IF NOT EXISTS idx_vector_index_operations_requested ON vector_index_operations(requested_at);
245
254
  CREATE INDEX IF NOT EXISTS idx_provenance_node ON ingestion_provenance(node_id);
246
255
  CREATE INDEX IF NOT EXISTS idx_provenance_source_type ON ingestion_provenance(source_type);
@@ -11,13 +11,21 @@ exactly like "search is a bit worse today":
11
11
 
12
12
  Both resolve to the exact brute-force scan and carry a ``detail`` string
13
13
  naming the cause, which ``index_status()`` and the search result surface.
14
+
15
+ The backend is a *choice*, not a switch, so it cannot ride a
16
+ :class:`~lattice_brain.gates.FeatureGate` (which answers booleans). It gets the
17
+ same shape of seam instead: :func:`bind_vector_index_resolver` installs a
18
+ caller-supplied resolver that is consulted ahead of the environment, which is
19
+ what lets the settings panel change the backend without a restart. With nothing
20
+ bound — the default — the environment variable is still the whole control
21
+ surface, exactly as before.
14
22
  """
15
23
 
16
24
  from __future__ import annotations
17
25
 
18
26
  import os
19
27
  from dataclasses import dataclass
20
- from typing import Any, Dict, Optional
28
+ from typing import Any, Callable, Dict, Optional
21
29
 
22
30
  from .base import Similarity, VectorIndex
23
31
  from .brute_force import BRUTE_FORCE_BACKEND, BruteForceIndex
@@ -37,6 +45,23 @@ _BACKEND_LABELS = {
37
45
  _APPROX = {"brute": False, "quantized": True, "hnsw": True}
38
46
  _EXHAUSTIVE = {"brute": True, "quantized": True, "hnsw": False}
39
47
 
48
+ #: App-layer resolver, consulted before the environment. ``None`` (the default)
49
+ #: leaves this module reading exactly the env var it always read.
50
+ _RESOLVER: Optional[Callable[[], Optional[str]]] = None
51
+
52
+
53
+ def bind_vector_index_resolver(
54
+ resolver: Optional[Callable[[], Optional[str]]],
55
+ ) -> None:
56
+ """Delegate backend selection to a callable (``None`` hands it back to env).
57
+
58
+ A resolver that returns ``None`` also falls through to the environment, so
59
+ "the settings service has no opinion" and "there is no settings service"
60
+ reach the same answer instead of two.
61
+ """
62
+ global _RESOLVER
63
+ _RESOLVER = resolver
64
+
40
65
 
41
66
  @dataclass(frozen=True)
42
67
  class BackendSelection:
@@ -79,7 +104,11 @@ def _selection(name: str, *, requested: str, detail: Optional[str]) -> BackendSe
79
104
 
80
105
  def resolve_vector_index(requested: Optional[str] = None) -> BackendSelection:
81
106
  """Resolve the configured backend (never raises, always falls back safe)."""
82
- raw = requested if requested is not None else os.getenv(VECTOR_INDEX_ENV, "")
107
+ raw = requested
108
+ if raw is None and _RESOLVER is not None:
109
+ raw = _RESOLVER()
110
+ if raw is None:
111
+ raw = os.getenv(VECTOR_INDEX_ENV, "")
83
112
  name = str(raw or "").strip().lower() or DEFAULT_VECTOR_INDEX
84
113
  if name not in VECTOR_INDEX_CHOICES:
85
114
  return _selection(
@@ -126,6 +155,7 @@ __all__ = [
126
155
  "VECTOR_INDEX_CHOICES",
127
156
  "VECTOR_INDEX_ENV",
128
157
  "BackendSelection",
158
+ "bind_vector_index_resolver",
129
159
  "build_index",
130
160
  "resolve_vector_index",
131
161
  ]
@@ -46,7 +46,31 @@ from dataclasses import dataclass, field
46
46
  from pathlib import Path
47
47
  from typing import Any, Dict, Iterable, List, Optional, Tuple
48
48
 
49
+ from .gates import FeatureGate
49
50
  from .graph.vector_index import DEFAULT_TICK_LIMIT as VECTOR_TICK_LIMIT
51
+ from .multimodal import (
52
+ AUDIO_EXTENSIONS,
53
+ DEFAULT_KEYFRAMES,
54
+ IMAGE_EXTENSIONS,
55
+ MODALITY_AUDIO,
56
+ MODALITY_IMAGE,
57
+ MODALITY_VIDEO,
58
+ VIDEO_EXTENSIONS,
59
+ VIDEO_UNAVAILABLE_DETAIL,
60
+ ImageFacts,
61
+ MultimodalPorts,
62
+ audio_quality_score,
63
+ detect_modality,
64
+ extract_image_facts,
65
+ ffmpeg_available,
66
+ image_quality_score,
67
+ read_video_facts,
68
+ transcribe_audio,
69
+ video_frame_dir,
70
+ video_quality_score,
71
+ write_image_memory,
72
+ write_video_memory,
73
+ )
50
74
  from .runtime.hooks import dispatch_tool
51
75
  from .utils import utc_now_iso
52
76
 
@@ -111,6 +135,59 @@ DEFAULT_MAX_FILE_BYTES = 4_000_000 # matches the local-index text/code budget
111
135
  LATTICEIGNORE_FILENAME = ".latticeignore"
112
136
  # Opt-out escape hatch for the post-ingest incremental vector sync.
113
137
  AUTO_VECTOR_INDEX_ENV = "LATTICEAI_AUTO_VECTOR_INDEX"
138
+ #: Default *on*, unlike every other gate here: new material has always been made
139
+ #: searchable straight away, and this exists so a settings surface can turn that
140
+ #: off (batch reindex later) without a restart. ``FeatureGate`` parses the env
141
+ #: var with the same words the hand-written opt-out check used, so an untouched
142
+ #: install — including one with a nonsense value — answers exactly as before.
143
+ AUTO_VECTOR_INDEX_GATE = FeatureGate(
144
+ AUTO_VECTOR_INDEX_ENV,
145
+ default=True,
146
+ name="auto_vector_index",
147
+ detail="New material is prepared for semantic search as soon as it lands.",
148
+ )
149
+
150
+ # ── Multi-modal ingestion (v11.1.0 Track 3) ──────────────────────────────────
151
+ # Opt-in, default off, on purpose. Turning it on changes what a folder scan
152
+ # *stores* (pictures and recordings, with OCR and — if a model is loaded —
153
+ # captions and vectors), and that is the user's call, not a default. With the
154
+ # flag off every routing decision below is skipped and behaviour is byte-for-
155
+ # byte what it was before this release.
156
+ ALLOW_MULTIMODAL_ENV = "LATTICEAI_ALLOW_MULTIMODAL"
157
+ #: The multi-modal switch, resolved when it is asked rather than frozen into
158
+ #: ``self`` at construction (v11.2.0). The environment variable is still the
159
+ #: answer for an untouched install — same var, same words, same default off —
160
+ #: but a settings surface can bind a resolver and move it without a restart.
161
+ MULTIMODAL_GATE = FeatureGate(
162
+ ALLOW_MULTIMODAL_ENV,
163
+ default=False,
164
+ name="allow_multimodal",
165
+ detail="Pictures and recordings are only ingested when this is turned on.",
166
+ )
167
+ #: Video is a *sub-switch* of the one above: with multi-modal off nothing about
168
+ #: video happens at all, and with it on video is included unless this is
169
+ #: explicitly turned off. The effective default is therefore still "no video",
170
+ #: and the seam exists so a settings screen can offer pictures without films.
171
+ ALLOW_VIDEO_ENV = "LATTICEAI_ALLOW_VIDEO"
172
+ VIDEO_GATE = FeatureGate(
173
+ ALLOW_VIDEO_ENV,
174
+ default=True,
175
+ name="allow_video",
176
+ detail="Videos are ingested as keyframes plus subtitles when multi-modal is on.",
177
+ )
178
+ #: Source types that name a modality outright (a caller who already knows).
179
+ IMAGE_SOURCE_TYPES = frozenset({"image", "screenshot", "photo"})
180
+ AUDIO_SOURCE_TYPES = frozenset({"audio", "voice_memo", "recording"})
181
+ VIDEO_SOURCE_TYPES = frozenset({"video", "screen_recording", "movie"})
182
+ #: Added to the folder-scan allow-list only while multimodal is enabled.
183
+ FOLDER_MULTIMODAL_EXTENSIONS = IMAGE_EXTENSIONS | AUDIO_EXTENSIONS
184
+ #: Videos join the folder allow-list only when this machine can decode one —
185
+ #: scanning a folder into a pile of refusals is not a feature.
186
+ FOLDER_VIDEO_EXTENSIONS = VIDEO_EXTENSIONS
187
+ #: Graph node type for a recording. ``NodeType.AUDIO`` normalizes this on the
188
+ #: KG v2 write side; the legacy tables keep the label verbatim, which is what
189
+ #: every type-aware read (graph view, context sections, doc-gen) matches on.
190
+ AUDIO_NODE_TYPE = "Audio"
114
191
 
115
192
  # ── Extraction quality heuristics (v9.8.0 A1) ────────────────────────────────
116
193
  # Pure heuristics over the extracted text — no model calls, no network. The
@@ -448,6 +525,8 @@ class IngestionPipeline:
448
525
  pipeline_name: str = "unified-ingestion",
449
526
  bg_queue: Optional[BackgroundIngestionQueue] = None,
450
527
  auto_vector_index: bool = True,
528
+ allow_multimodal: bool = False,
529
+ multimodal: Optional[MultimodalPorts] = None,
451
530
  ) -> None:
452
531
  self._kg = knowledge_graph
453
532
  self._hooks = hooks
@@ -464,16 +543,79 @@ class IngestionPipeline:
464
543
  db_path=getattr(knowledge_graph, "db_path", None)
465
544
  )
466
545
  # Incremental vector sync after each successful non-duplicate ingest.
467
- # Constructor opt-out AND env opt-out (LATTICEAI_AUTO_VECTOR_INDEX=0)
468
- # both disable it; a vector failure never fails the ingest.
469
- env_flag = os.getenv(AUTO_VECTOR_INDEX_ENV, "1").strip().lower() not in {
470
- "0", "false", "no", "off",
471
- }
472
- self._auto_vector_index = bool(auto_vector_index) and env_flag
546
+ # Constructor opt-out AND gate opt-out (LATTICEAI_AUTO_VECTOR_INDEX=0,
547
+ # or the settings toggle bound to it) both disable it; a vector failure
548
+ # never fails the ingest. The gate half is asked per ingest rather than
549
+ # frozen here, so turning it off takes effect on the next item.
550
+ self._auto_vector_index_opt_in = bool(auto_vector_index)
551
+ # Multi-modal routing. Off unless the caller asks for it *or* the gate
552
+ # says yes — the env behind that gate is the escape hatch for an
553
+ # install with no code path to the constructor (CLI, background
554
+ # worker), and the gate is now asked per call so a runtime toggle can
555
+ # reach it. A constructor ``True`` is still a permanent yes.
556
+ self._multimodal_opt_in = bool(allow_multimodal)
557
+ self._multimodal = multimodal or MultimodalPorts()
558
+ self._keyframes = DEFAULT_KEYFRAMES
559
+
560
+ @property
561
+ def _auto_vector_index(self) -> bool:
562
+ """Whether a landed ingest also syncs its vector, asked *now*."""
563
+ return self._auto_vector_index_opt_in and AUTO_VECTOR_INDEX_GATE.enabled()
564
+
565
+ @property
566
+ def _allow_multimodal(self) -> bool:
567
+ """Whether pictures and recordings route by modality, asked *now*."""
568
+ return self._multimodal_opt_in or MULTIMODAL_GATE.enabled()
569
+
570
+ @property
571
+ def _allow_video(self) -> bool:
572
+ """Video needs multi-modal on, its own sub-switch on, and a decoder."""
573
+ return self._allow_multimodal and VIDEO_GATE.enabled() and self._can_decode_video()
574
+
575
+ def _can_decode_video(self) -> bool:
576
+ """An injected keyframe port counts as a decoder; otherwise, ffmpeg."""
577
+ return self._multimodal.keyframe_extractor is not None or ffmpeg_available()
473
578
 
474
579
  def available(self) -> bool:
475
580
  return self._enable and self._kg is not None
476
581
 
582
+ def multimodal_status(self) -> Dict[str, Any]:
583
+ """What this pipeline will do with a picture or a recording, honestly.
584
+
585
+ ``enabled`` is the flag; the rest is which model-backed capabilities
586
+ were actually injected. Video reports whether it can really run — the
587
+ answer is no on a machine with no ffmpeg, and it says which of the two
588
+ reasons applies rather than leaving the surface to guess.
589
+ """
590
+ allowed = self._allow_multimodal
591
+ video = self._allow_video
592
+ return {
593
+ "enabled": allowed,
594
+ "image": allowed,
595
+ "audio": allowed,
596
+ "video": video,
597
+ "video_detail": None if video else self._video_refusal(),
598
+ "gates": {
599
+ "multimodal": MULTIMODAL_GATE.describe(),
600
+ "video": VIDEO_GATE.describe(),
601
+ },
602
+ **self._multimodal.describe(),
603
+ }
604
+
605
+ def _video_refusal(self) -> str:
606
+ """Why a video would be refused right now — never a stale reason."""
607
+ if not self._allow_multimodal:
608
+ return (
609
+ "multi-modal ingestion is off; pictures, recordings and videos "
610
+ f"are only stored when {ALLOW_MULTIMODAL_ENV} is on"
611
+ )
612
+ if not VIDEO_GATE.enabled():
613
+ return (
614
+ "video ingestion is turned off for this install "
615
+ f"({ALLOW_VIDEO_ENV}); pictures and recordings are unaffected"
616
+ )
617
+ return VIDEO_UNAVAILABLE_DETAIL
618
+
477
619
  # ── public API ───────────────────────────────────────────────────────────
478
620
  def ingest(self, item: IngestionItem, *, user_email: Optional[str] = None) -> IngestionResult:
479
621
  """Normalize, hash, route through dispatch_tool, and record provenance."""
@@ -485,6 +627,18 @@ class IngestionPipeline:
485
627
  detail="Knowledge Graph is disabled (LATTICEAI_ENABLE_GRAPH).",
486
628
  )
487
629
 
630
+ # Modality routing is a no-op while the flag is off: ``modality`` stays
631
+ # "text" and every branch below behaves exactly as it did before.
632
+ modality = self._modality_for(item, source_type)
633
+ if modality == MODALITY_VIDEO and not self._allow_video:
634
+ # Recognized and refused, with the reason that actually applies
635
+ # right now — a missing decoder is not the same answer as a
636
+ # switched-off feature, and the caller can act on the difference.
637
+ return IngestionResult(
638
+ status="unavailable", source_type=source_type,
639
+ indexing_status="skipped", detail=self._video_refusal(),
640
+ )
641
+
488
642
  captured_at = item.captured_at or utc_now_iso()
489
643
  owner = item.owner or user_email
490
644
  tool_name = f"kg_ingest.{source_type}"
@@ -501,6 +655,12 @@ class IngestionPipeline:
501
655
  return self._ingest_chat(item, source_type=source_type, owner=owner)
502
656
  if source_type in MEMORY_SOURCE_TYPES:
503
657
  return self._ingest_memory_record(item, source_type=source_type, owner=owner)
658
+ if modality == MODALITY_IMAGE:
659
+ return self._ingest_image(item, source_type=source_type, owner=owner, captured_at=captured_at)
660
+ if modality == MODALITY_AUDIO:
661
+ return self._ingest_audio(item, source_type=source_type, owner=owner, captured_at=captured_at)
662
+ if modality == MODALITY_VIDEO:
663
+ return self._ingest_video(item, source_type=source_type, owner=owner, captured_at=captured_at)
504
664
  if source_type in FILE_SOURCE_TYPES or (item.path and not item.text):
505
665
  return self._ingest_file(item, source_type=source_type, owner=owner, captured_at=captured_at)
506
666
  return self._ingest_text(item, source_type=source_type, owner=owner, captured_at=captured_at)
@@ -590,7 +750,11 @@ class IngestionPipeline:
590
750
  except Exception: # noqa: BLE001 — audit must never break ingestion
591
751
  quiet()
592
752
 
593
- extraction_quality = self._assess_item_quality(
753
+ # A modality-aware door scores its own extraction (a picture's quality
754
+ # is "how much of it can be retrieved", not "does the text read well"),
755
+ # so its verdict wins. Text/file doors never set the key and keep the
756
+ # historical scoring untouched.
757
+ extraction_quality = raw.get("extraction_quality") or self._assess_item_quality(
594
758
  item, source_type=source_type, text=quality_text, chunk_ids=chunk_ids,
595
759
  )
596
760
  warnings: List[str] = []
@@ -980,7 +1144,7 @@ class IngestionPipeline:
980
1144
  allowed_exts = (
981
1145
  frozenset(str(e).lower() if str(e).startswith(".") else f".{str(e).lower()}" for e in extensions)
982
1146
  if extensions
983
- else DEFAULT_FOLDER_EXTENSIONS
1147
+ else self._folder_extensions()
984
1148
  )
985
1149
  patterns = _load_latticeignore(root)
986
1150
  errors: List[Dict[str, Any]] = summary["errors"]
@@ -1036,7 +1200,11 @@ class IngestionPipeline:
1036
1200
  summary["truncated"] = True
1037
1201
  break
1038
1202
  item_metadata: Dict[str, Any] = {"relative_path": rel}
1039
- if ext in FOLDER_DOCUMENT_EXTENSIONS:
1203
+ if ext in (FOLDER_MULTIMODAL_EXTENSIONS | FOLDER_VIDEO_EXTENSIONS) and self._allow_multimodal:
1204
+ # Routed by modality inside ``ingest``; reading the bytes as
1205
+ # UTF-8 here would only produce mojibake.
1206
+ source_type = "file"
1207
+ elif ext in FOLDER_DOCUMENT_EXTENSIONS:
1040
1208
  source_type = "pdf"
1041
1209
  else:
1042
1210
  source_type = "file"
@@ -1080,6 +1248,19 @@ class IngestionPipeline:
1080
1248
  summary["status"] = "ok" if summary["failed"] == 0 else "partial"
1081
1249
  return summary
1082
1250
 
1251
+ def _folder_extensions(self) -> frozenset:
1252
+ """Folder-scan allow-list — pictures, recordings and films when enabled.
1253
+
1254
+ Video joins only when this machine can actually decode one, so a scan
1255
+ never fills the error list with files it was always going to refuse.
1256
+ """
1257
+ if not self._allow_multimodal:
1258
+ return DEFAULT_FOLDER_EXTENSIONS
1259
+ allowed = DEFAULT_FOLDER_EXTENSIONS | FOLDER_MULTIMODAL_EXTENSIONS
1260
+ if self._allow_video:
1261
+ return allowed | FOLDER_VIDEO_EXTENSIONS
1262
+ return allowed
1263
+
1083
1264
  # ── routing helpers ──────────────────────────────────────────────────────
1084
1265
  def _ingest_text(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
1085
1266
  text = item.text or ""
@@ -1142,7 +1323,26 @@ class IngestionPipeline:
1142
1323
  result.setdefault("title", item.title)
1143
1324
  return result
1144
1325
 
1145
- def _ingest_file(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
1326
+ # ── multi-modal routing (v11.1.0 Track 3) ────────────────────────────────
1327
+ def _modality_for(self, item: IngestionItem, source_type: str) -> str:
1328
+ """``image`` / ``audio`` / ``video`` / ``text`` for this item.
1329
+
1330
+ Always ``"text"`` while the flag is off, which is what makes "off" mean
1331
+ *unchanged* rather than *slightly different*.
1332
+ """
1333
+ if not self._allow_multimodal:
1334
+ return "text"
1335
+ if source_type in IMAGE_SOURCE_TYPES:
1336
+ return MODALITY_IMAGE
1337
+ if source_type in AUDIO_SOURCE_TYPES:
1338
+ return MODALITY_AUDIO
1339
+ if source_type in VIDEO_SOURCE_TYPES:
1340
+ return MODALITY_VIDEO
1341
+ if not item.path:
1342
+ return "text"
1343
+ return detect_modality(item.path, item.mime_type)
1344
+
1345
+ def _resolve_file_path(self, item: IngestionItem) -> Path:
1146
1346
  if not item.path:
1147
1347
  raise ValueError("File ingestion requires a path.")
1148
1348
  path = Path(item.path)
@@ -1150,6 +1350,150 @@ class IngestionPipeline:
1150
1350
  raise FileNotFoundError(f"File not found: {path}")
1151
1351
  if path.is_dir():
1152
1352
  raise ValueError(f"File ingestion requires a file, got a directory: {path}")
1353
+ return path
1354
+
1355
+ def _ingest_image(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
1356
+ """Store one picture as an ``Image`` node — OCR, caption, vector.
1357
+
1358
+ The image vector (when a vision model produced one) goes to its own
1359
+ index; the OCR/caption text rides the ordinary text index. That split
1360
+ is what lets a typed question find a screenshot without ever comparing
1361
+ a text vector to an image vector.
1362
+ """
1363
+ path = self._resolve_file_path(item)
1364
+ facts = extract_image_facts(str(path), ports=self._multimodal)
1365
+ result = write_image_memory(
1366
+ self._kg,
1367
+ path=path,
1368
+ facts=facts,
1369
+ title=item.title or path.name,
1370
+ source_type=source_type if source_type in IMAGE_SOURCE_TYPES else MODALITY_IMAGE,
1371
+ source_uri=item.source_uri,
1372
+ owner=owner,
1373
+ workspace_id=item.workspace_id,
1374
+ conversation_id=item.conversation_id,
1375
+ captured_at=captured_at,
1376
+ modified_at=item.modified_at,
1377
+ permissions=item.permissions,
1378
+ extra_metadata={"mime_type": item.mime_type, **(item.metadata or {})},
1379
+ )
1380
+ self._record_image_vector(result["node_id"], facts)
1381
+ quality = image_quality_score(facts)
1382
+ result["extraction_quality"] = {
1383
+ "score": quality["score"],
1384
+ "level": _quality_level(quality["score"]),
1385
+ "reasons": quality["reasons"],
1386
+ }
1387
+ return result
1388
+
1389
+ def _record_image_vector(self, node_id: str, facts: ImageFacts) -> None:
1390
+ """File the image-space vector, if a vision model actually made one."""
1391
+ if facts.embedding is None:
1392
+ return
1393
+ from .graph.image_vectors import record_image_vector
1394
+
1395
+ record_image_vector(
1396
+ self._kg,
1397
+ node_id=node_id,
1398
+ vector=facts.embedding,
1399
+ model_id=self._multimodal.vision_model_id or "vision:unnamed",
1400
+ space=self._multimodal.vision_space,
1401
+ updated_at=utc_now_iso(),
1402
+ )
1403
+
1404
+ def _ingest_audio(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
1405
+ """Store one recording as an ``Audio`` node, transcribed when possible.
1406
+
1407
+ The transcript is text and rides the ordinary text index — chunks,
1408
+ concepts, provenance, dedupe all unchanged — but the node itself is a
1409
+ recording, because that is what it is whether or not anyone could hear
1410
+ it. The recording's own facts stay in the metadata (``modality``,
1411
+ ``audio_path``, ``transcription``, ``searchable``). Without a
1412
+ transcriber the memory is still kept, and its body says plainly that
1413
+ the words were never recognized instead of leaving a blank note.
1414
+ """
1415
+ path = self._resolve_file_path(item)
1416
+ facts = transcribe_audio(str(path), ports=self._multimodal, transcript=item.text)
1417
+ title = item.title or path.stem
1418
+ body = facts.transcript or (
1419
+ f"[{MODALITY_AUDIO}] {title}\n"
1420
+ "이 녹음은 아직 글로 바뀌지 않았습니다 — 음성 인식기가 없어 내용 검색은 되지 않습니다."
1421
+ )
1422
+ result = self._kg.ingest_source(
1423
+ source_type=source_type,
1424
+ title=title,
1425
+ text=body,
1426
+ source_uri=item.source_uri or str(path),
1427
+ owner=owner,
1428
+ workspace_id=item.workspace_id,
1429
+ permissions=item.permissions,
1430
+ captured_at=captured_at,
1431
+ modified_at=item.modified_at,
1432
+ conversation_id=item.conversation_id,
1433
+ node_type=AUDIO_NODE_TYPE,
1434
+ metadata={
1435
+ "mime_type": item.mime_type,
1436
+ "modality": MODALITY_AUDIO,
1437
+ "audio_path": str(path),
1438
+ "audio_bytes": path.stat().st_size,
1439
+ "transcription": facts.transcription_status,
1440
+ "searchable": facts.searchable,
1441
+ **({"transcription_detail": facts.detail} if facts.detail else {}),
1442
+ **(item.metadata or {}),
1443
+ },
1444
+ )
1445
+ result.setdefault("title", title)
1446
+ quality = audio_quality_score(facts)
1447
+ result["extraction_quality"] = {
1448
+ "score": quality["score"],
1449
+ "level": _quality_level(quality["score"]),
1450
+ "reasons": quality["reasons"],
1451
+ }
1452
+ return result
1453
+
1454
+ def _ingest_video(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
1455
+ """Store one video as keyframes through the image door plus subtitles.
1456
+
1457
+ Nothing here is a new retrieval path: the stills become ordinary
1458
+ ``Image`` nodes (OCR, caption, vector, thumbnail) joined by
1459
+ ``CONTAINS_IMAGE``, and the subtitle text becomes ordinary chunks. What
1460
+ the ``Video`` node adds is the thing they belong to — and an honest
1461
+ body when there were no subtitles to read.
1462
+ """
1463
+ path = self._resolve_file_path(item)
1464
+ facts = read_video_facts(
1465
+ str(path),
1466
+ video_frame_dir(getattr(self._kg, "blob_dir", path.parent), _file_digest(path)),
1467
+ count=self._keyframes,
1468
+ ports=self._multimodal,
1469
+ subtitle_text=item.text,
1470
+ )
1471
+ result = write_video_memory(
1472
+ self._kg,
1473
+ path=path,
1474
+ facts=facts,
1475
+ title=item.title or path.stem,
1476
+ source_type=source_type if source_type in VIDEO_SOURCE_TYPES else MODALITY_VIDEO,
1477
+ source_uri=item.source_uri,
1478
+ owner=owner,
1479
+ workspace_id=item.workspace_id,
1480
+ conversation_id=item.conversation_id,
1481
+ captured_at=captured_at,
1482
+ modified_at=item.modified_at,
1483
+ permissions=item.permissions,
1484
+ extra_metadata={"mime_type": item.mime_type, **(item.metadata or {})},
1485
+ ports=self._multimodal,
1486
+ )
1487
+ quality = video_quality_score(facts)
1488
+ result["extraction_quality"] = {
1489
+ "score": quality["score"],
1490
+ "level": _quality_level(quality["score"]),
1491
+ "reasons": quality["reasons"],
1492
+ }
1493
+ return result
1494
+
1495
+ def _ingest_file(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
1496
+ path = self._resolve_file_path(item)
1153
1497
  return self._kg.ingest_document(
1154
1498
  path,
1155
1499
  original_filename=item.title or path.name,
@@ -1170,3 +1514,12 @@ class IngestionPipeline:
1170
1514
  def content_hash_text(text: str) -> str:
1171
1515
  """Canonical content hash for a text payload (matches store hashing scheme)."""
1172
1516
  return hashlib.sha256((text or "").encode("utf-8", "ignore")).hexdigest()
1517
+
1518
+
1519
+ def _file_digest(path: Path) -> str:
1520
+ """Streaming sha256 of a file — the key a video's frame folder is named by."""
1521
+ digest = hashlib.sha256()
1522
+ with path.open("rb") as handle:
1523
+ for block in iter(lambda: handle.read(1024 * 1024), b""):
1524
+ digest.update(block)
1525
+ return digest.hexdigest()