ltcai 11.2.0 → 11.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (247) hide show
  1. package/README.md +46 -53
  2. package/docs/CHANGELOG.md +61 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/MULTI_AGENT_RUNTIME.md +1 -1
  6. package/docs/ONBOARDING.md +1 -1
  7. package/docs/OPERATIONS.md +6 -2
  8. package/docs/PERMISSION_MODE.md +1 -1
  9. package/docs/TRUST_MODEL.md +1 -1
  10. package/docs/WHY_LATTICE.md +1 -1
  11. package/docs/kg-schema.md +2 -2
  12. package/docs/v11.3.0_PLAN.md +202 -0
  13. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
  14. package/lattice_brain/__init__.py +1 -1
  15. package/lattice_brain/graph/_kg_common/__init__.py +287 -0
  16. package/lattice_brain/graph/_kg_common/extraction.py +516 -0
  17. package/lattice_brain/graph/_kg_common/relations.py +161 -0
  18. package/lattice_brain/graph/_kg_common/text.py +479 -0
  19. package/lattice_brain/graph/discovery_index/__init__.py +35 -0
  20. package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
  21. package/lattice_brain/graph/discovery_index/extract.py +137 -0
  22. package/lattice_brain/graph/discovery_index/scan.py +411 -0
  23. package/lattice_brain/graph/discovery_index/upsert.py +495 -0
  24. package/lattice_brain/graph/projection/__init__.py +42 -0
  25. package/lattice_brain/graph/projection/curation.py +500 -0
  26. package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
  27. package/lattice_brain/graph/retrieval/__init__.py +54 -0
  28. package/lattice_brain/graph/retrieval/context.py +197 -0
  29. package/lattice_brain/graph/retrieval/graph_view.py +319 -0
  30. package/lattice_brain/graph/retrieval/hybrid.py +488 -0
  31. package/lattice_brain/graph/retrieval/maintenance.py +121 -0
  32. package/lattice_brain/graph/retrieval/signals.py +95 -0
  33. package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
  34. package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
  35. package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
  36. package/lattice_brain/graph/retrieval_vector/search.py +560 -0
  37. package/lattice_brain/graph/retrieval_vector/status.py +374 -0
  38. package/lattice_brain/ingestion/__init__.py +130 -0
  39. package/lattice_brain/ingestion/_contract.py +90 -0
  40. package/lattice_brain/ingestion/constants.py +127 -0
  41. package/lattice_brain/ingestion/folder_scan.py +57 -0
  42. package/lattice_brain/ingestion/folders.py +258 -0
  43. package/lattice_brain/ingestion/hashing.py +26 -0
  44. package/lattice_brain/ingestion/jobs_api.py +107 -0
  45. package/lattice_brain/ingestion/models.py +80 -0
  46. package/lattice_brain/ingestion/pipeline.py +486 -0
  47. package/lattice_brain/ingestion/quality.py +209 -0
  48. package/lattice_brain/ingestion/routing.py +295 -0
  49. package/lattice_brain/multimodal/__init__.py +164 -0
  50. package/lattice_brain/multimodal/audio.py +77 -0
  51. package/lattice_brain/multimodal/common.py +118 -0
  52. package/lattice_brain/multimodal/images.py +498 -0
  53. package/lattice_brain/multimodal/ports.py +169 -0
  54. package/lattice_brain/multimodal/video.py +410 -0
  55. package/lattice_brain/portability/__init__.py +90 -0
  56. package/lattice_brain/portability/_contract.py +42 -0
  57. package/lattice_brain/portability/backups.py +338 -0
  58. package/lattice_brain/portability/bundles.py +136 -0
  59. package/lattice_brain/portability/constants.py +93 -0
  60. package/lattice_brain/portability/fsops.py +138 -0
  61. package/lattice_brain/portability/service.py +41 -0
  62. package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
  63. package/lattice_brain/runtime/__init__.py +1 -1
  64. package/lattice_brain/runtime/multi_agent.py +1 -1
  65. package/latticeai/__init__.py +1 -1
  66. package/latticeai/api/chronicle.py +63 -0
  67. package/latticeai/core/agent/__init__.py +93 -0
  68. package/latticeai/core/agent/_contract.py +79 -0
  69. package/latticeai/core/agent/context.py +57 -0
  70. package/latticeai/core/agent/deps.py +125 -0
  71. package/latticeai/core/agent/execution.py +622 -0
  72. package/latticeai/core/agent/planning.py +145 -0
  73. package/latticeai/core/agent/recovery.py +157 -0
  74. package/latticeai/core/agent/runtime.py +210 -0
  75. package/latticeai/core/agent/verification.py +231 -0
  76. package/latticeai/core/embedding_providers/__init__.py +151 -0
  77. package/latticeai/core/embedding_providers/base.py +199 -0
  78. package/latticeai/core/embedding_providers/captions.py +162 -0
  79. package/latticeai/core/embedding_providers/profiles.py +126 -0
  80. package/latticeai/core/embedding_providers/text.py +350 -0
  81. package/latticeai/core/embedding_providers/vision.py +352 -0
  82. package/latticeai/core/file_generation/__init__.py +115 -0
  83. package/latticeai/core/file_generation/bundles.py +76 -0
  84. package/latticeai/core/file_generation/extraction.py +154 -0
  85. package/latticeai/core/file_generation/inference.py +235 -0
  86. package/latticeai/core/file_generation/orchestration.py +152 -0
  87. package/latticeai/core/file_generation/prompting.py +117 -0
  88. package/latticeai/core/file_generation/repair.py +114 -0
  89. package/latticeai/core/file_generation/sanitize.py +61 -0
  90. package/latticeai/core/file_generation/validation.py +201 -0
  91. package/latticeai/core/legacy_compatibility.py +1 -1
  92. package/latticeai/core/marketplace.py +1 -1
  93. package/latticeai/core/messages.py +9 -0
  94. package/latticeai/core/workspace_os_constants.py +1 -1
  95. package/latticeai/integrations/telegram_bot/__init__.py +123 -0
  96. package/latticeai/integrations/telegram_bot/__main__.py +17 -0
  97. package/latticeai/integrations/telegram_bot/config.py +86 -0
  98. package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
  99. package/latticeai/integrations/telegram_bot/flows.py +478 -0
  100. package/latticeai/integrations/telegram_bot/helpers.py +322 -0
  101. package/latticeai/integrations/telegram_bot/screens.py +394 -0
  102. package/latticeai/models/router/__init__.py +88 -0
  103. package/latticeai/models/router/_contract.py +66 -0
  104. package/latticeai/models/router/branding.py +56 -0
  105. package/latticeai/models/router/catalog.py +69 -0
  106. package/latticeai/models/router/documents.py +199 -0
  107. package/latticeai/models/router/errors.py +37 -0
  108. package/latticeai/models/router/generation.py +258 -0
  109. package/latticeai/models/router/loading.py +291 -0
  110. package/latticeai/models/router/local_models.py +85 -0
  111. package/latticeai/models/router/registry.py +147 -0
  112. package/latticeai/runtime/build_phases/__init__.py +82 -0
  113. package/latticeai/runtime/build_phases/features.py +407 -0
  114. package/latticeai/runtime/build_phases/foundation.py +555 -0
  115. package/latticeai/runtime/build_phases/web.py +492 -0
  116. package/latticeai/runtime/runtime_context.py +1 -0
  117. package/latticeai/services/architecture_readiness.py +48 -19
  118. package/latticeai/services/brain_intelligence/__init__.py +58 -0
  119. package/latticeai/services/brain_intelligence/_contract.py +71 -0
  120. package/latticeai/services/brain_intelligence/consistency.py +193 -0
  121. package/latticeai/services/brain_intelligence/constants.py +47 -0
  122. package/latticeai/services/brain_intelligence/digest.py +258 -0
  123. package/latticeai/services/brain_intelligence/health.py +331 -0
  124. package/latticeai/services/brain_intelligence/proposals.py +264 -0
  125. package/latticeai/services/brain_intelligence/sampling.py +84 -0
  126. package/latticeai/services/brain_intelligence/service.py +48 -0
  127. package/latticeai/services/chronicle.py +557 -0
  128. package/latticeai/services/memory_service/__init__.py +52 -0
  129. package/latticeai/services/memory_service/_contract.py +100 -0
  130. package/latticeai/services/memory_service/brief.py +431 -0
  131. package/latticeai/services/memory_service/constants.py +57 -0
  132. package/latticeai/services/memory_service/maintenance.py +138 -0
  133. package/latticeai/services/memory_service/manager.py +186 -0
  134. package/latticeai/services/memory_service/proof.py +136 -0
  135. package/latticeai/services/memory_service/recall.py +225 -0
  136. package/latticeai/services/memory_service/service.py +48 -0
  137. package/latticeai/services/memory_service/stores.py +110 -0
  138. package/latticeai/services/model_runtime/__init__.py +322 -0
  139. package/latticeai/services/model_runtime/cloud.py +87 -0
  140. package/latticeai/services/model_runtime/download.py +282 -0
  141. package/latticeai/services/model_runtime/engines.py +341 -0
  142. package/latticeai/services/model_runtime/loading.py +178 -0
  143. package/latticeai/services/model_runtime/service.py +129 -0
  144. package/latticeai/services/model_runtime/state.py +131 -0
  145. package/latticeai/services/model_runtime/status.py +255 -0
  146. package/latticeai/services/product_readiness.py +15 -7
  147. package/latticeai/setup/wizard/__init__.py +126 -0
  148. package/latticeai/setup/wizard/catalog.py +172 -0
  149. package/latticeai/setup/wizard/detect.py +323 -0
  150. package/latticeai/setup/wizard/install.py +348 -0
  151. package/latticeai/setup/wizard/paths.py +168 -0
  152. package/latticeai/setup/wizard/plans.py +74 -0
  153. package/latticeai/setup/wizard/recommend.py +320 -0
  154. package/package.json +6 -2
  155. package/scripts/bump_version.py +14 -0
  156. package/scripts/capture_release_evidence.mjs +33 -21
  157. package/scripts/check_current_release_docs.mjs +1 -1
  158. package/scripts/check_i18n_namespace_coverage.mjs +41 -4
  159. package/scripts/check_max_file_lines.mjs +102 -0
  160. package/scripts/check_release_evidence_bound.mjs +30 -15
  161. package/scripts/check_screenshot_pixel_delta.py +34 -4
  162. package/scripts/check_server_i18n.mjs +1 -0
  163. package/scripts/generate_rust_parity_fixtures.py +562 -0
  164. package/scripts/lib/mock_server_fingerprint.mjs +94 -0
  165. package/scripts/release_screen_claims.json +31 -2
  166. package/src-tauri/Cargo.lock +361 -3
  167. package/src-tauri/Cargo.toml +6 -1
  168. package/src-tauri/src/backend.rs +349 -0
  169. package/src-tauri/src/folder.rs +33 -0
  170. package/src-tauri/src/main.rs +97 -399
  171. package/src-tauri/tauri.conf.json +1 -1
  172. package/static/app/asset-manifest.json +41 -37
  173. package/static/app/assets/Act-yYpYnn0v.js +1 -0
  174. package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
  175. package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
  176. package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
  177. package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
  178. package/static/app/assets/Capture-CFIRsFNE.js +1 -0
  179. package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
  180. package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
  181. package/static/app/assets/Library-DwO3yZST.js +1 -0
  182. package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
  183. package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
  184. package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
  185. package/static/app/assets/System-DW8F-2xL.js +1 -0
  186. package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
  187. package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
  188. package/static/app/assets/brain-Ci1CkWjM.js +1 -0
  189. package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
  190. package/static/app/assets/circle-check-DfInj-qD.js +1 -0
  191. package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
  192. package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
  193. package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
  194. package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
  195. package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
  196. package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
  197. package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
  198. package/static/app/assets/index-_u5iUHDr.js +10 -0
  199. package/static/app/assets/input-B0lPdRQZ.js +1 -0
  200. package/static/app/assets/link-2-CoFbooHS.js +1 -0
  201. package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
  202. package/static/app/assets/primitives-DEbN-d6p.js +1 -0
  203. package/static/app/assets/search-BybIWPNd.js +1 -0
  204. package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
  205. package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
  206. package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
  207. package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
  208. package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
  209. package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
  210. package/static/app/assets/utils-BlZr7Pd4.js +4 -0
  211. package/static/app/assets/workspace-jJY4RuAV.js +1 -0
  212. package/static/app/index.html +4 -4
  213. package/static/sw.js +1 -1
  214. package/lattice_brain/graph/_kg_common.py +0 -1331
  215. package/lattice_brain/graph/discovery_index.py +0 -1141
  216. package/lattice_brain/graph/retrieval.py +0 -1120
  217. package/lattice_brain/graph/retrieval_vector.py +0 -1293
  218. package/lattice_brain/ingestion.py +0 -1525
  219. package/lattice_brain/multimodal.py +0 -1258
  220. package/latticeai/core/agent.py +0 -1465
  221. package/latticeai/core/embedding_providers.py +0 -1196
  222. package/latticeai/core/file_generation.py +0 -1047
  223. package/latticeai/integrations/telegram_bot.py +0 -1390
  224. package/latticeai/models/router.py +0 -1007
  225. package/latticeai/runtime/build_phases.py +0 -1450
  226. package/latticeai/services/brain_intelligence.py +0 -1083
  227. package/latticeai/services/memory_service.py +0 -1177
  228. package/latticeai/services/model_runtime.py +0 -1281
  229. package/latticeai/setup/wizard.py +0 -1310
  230. package/static/app/assets/Act-AWf0SAKp.js +0 -1
  231. package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
  232. package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
  233. package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
  234. package/static/app/assets/Capture-CqOSzyPr.js +0 -1
  235. package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
  236. package/static/app/assets/Library-CX-bbhmK.js +0 -1
  237. package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
  238. package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
  239. package/static/app/assets/System-Bu2t5hn1.js +0 -1
  240. package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
  241. package/static/app/assets/brain-DJMoqrwx.js +0 -1
  242. package/static/app/assets/index-BpYkzcVm.js +0 -10
  243. package/static/app/assets/input-DSlJJxRs.js +0 -1
  244. package/static/app/assets/primitives-BCx6TvfG.js +0 -1
  245. package/static/app/assets/search-Cgy8cCFJ.js +0 -1
  246. package/static/app/assets/utils-zqPZJxdx.js +0 -4
  247. package/static/app/assets/workspace-DXTihhfU.js +0 -1
@@ -1,1258 +0,0 @@
1
- """Images and audio as first-class memories (v11.1.0, Track 3).
2
-
3
- Before this module the Brain could only remember things that arrived as text.
4
- A screenshot of a whiteboard, a photo of a receipt, a voice memo — all of them
5
- either bounced off the ingestion pipeline or landed as an opaque ``Document``
6
- node whose only searchable content was its filename.
7
-
8
- What routes here
9
- ----------------
10
- :func:`detect_modality` reads the MIME type first and the extension second, and
11
- answers with one of ``text`` / ``image`` / ``audio`` / ``video``. ``video`` is
12
- deliberately a *recognized but unsupported* answer in this release: keyframe
13
- extraction needs a decoder this project does not ship, and returning "video,
14
- out of scope" is worth more than pretending a ``.mov`` is a picture.
15
-
16
- What an image memory contains
17
- -----------------------------
18
- :func:`extract_image_facts` gathers only what it can actually observe:
19
-
20
- * **dimensions/format** from Pillow (a core dependency);
21
- * **ocr_text** from ``pytesseract`` when it is installed — otherwise
22
- ``ocr_status="unavailable"`` and no text, never an empty string dressed up as
23
- a successful read;
24
- * **caption** from an injected vision-language port, and *only* from there. No
25
- VLM means ``caption is None``. Composing "Image IMG_2381.png (JPEG 3024x4032)"
26
- and storing it in the caption field would make metadata indistinguishable
27
- from a model's description forever after;
28
- * **embedding** from an injected vision port, which lives in its own vector
29
- space (see :mod:`latticeai.core.embedding_providers`) and therefore its own
30
- index — text queries reach images through OCR/caption text, not by scoring a
31
- BGE vector against CLIP vectors.
32
-
33
- Brain Core owns none of those models. Every heavy dependency arrives as an
34
- injected callable (:class:`MultimodalPorts`), which is also why this module
35
- imports nothing from ``latticeai``.
36
- """
37
-
38
- from __future__ import annotations
39
-
40
- import base64
41
- import hashlib
42
- import io
43
- import mimetypes
44
- import re
45
- import shutil
46
- import subprocess # noqa: S404 — one fixed binary, argv list, never a shell
47
- from dataclasses import dataclass, field
48
- from pathlib import Path
49
- from typing import Any, Callable, Dict, List, Optional
50
-
51
- from .quiet import quiet
52
- from .utils import utc_now_iso
53
-
54
- # ── modality taxonomy ────────────────────────────────────────────────────────
55
- MODALITY_TEXT = "text"
56
- MODALITY_IMAGE = "image"
57
- MODALITY_AUDIO = "audio"
58
- MODALITY_VIDEO = "video"
59
-
60
- IMAGE_EXTENSIONS = frozenset(
61
- {".png", ".jpg", ".jpeg", ".webp", ".gif", ".bmp", ".tif", ".tiff", ".heic"}
62
- )
63
- # Containers that are *only* ever audio. ``.mp4``/``.webm`` are deliberately
64
- # absent: by extension alone they are video, and a voice memo recorded in one
65
- # of them arrives through an explicit audio MIME type (or through
66
- # ``VoiceCaptureService``, where the user already said "this is a memo").
67
- # ``.mid``/``.midi`` are listed for a second reason: CPython's *built-in* mime
68
- # table has neither, so ``mimetypes`` answers "audio/midi" only on a host that
69
- # ships a system mime file (macOS reads /etc/apache2/mime.types; a slim Linux
70
- # container has nothing). Leaving them to the fallback let the platform decide
71
- # what a MIDI file is — a module table exists precisely so it does not.
72
- AUDIO_EXTENSIONS = frozenset(
73
- {".m4a", ".mp3", ".wav", ".aac", ".flac", ".ogg", ".opus", ".mid", ".midi"}
74
- )
75
- VIDEO_EXTENSIONS = frozenset({".mp4", ".webm", ".mov", ".mkv", ".avi", ".m4v"})
76
- #: Subtitle/caption files a video may arrive with. Same basename, so a
77
- #: ``standup.mp4`` next to a ``standup.srt`` is one memory, not two.
78
- SUBTITLE_EXTENSIONS = ("srt", "vtt")
79
-
80
- #: Why a video is recognized and still refused — surfaced to the caller. In
81
- #: 11.1.0 the reason was *scope* (nothing was implemented). Since 11.2.0 the
82
- #: implementation exists and the only remaining reason is a **runtime** one:
83
- #: this machine has no ``ffmpeg``, and inventing frames is not an option.
84
- VIDEO_UNAVAILABLE_DETAIL = (
85
- "video ingestion needs ffmpeg on this machine and none was found; the file "
86
- "was not stored (install ffmpeg to enable keyframe extraction)"
87
- )
88
- #: Kept under its 11.1.0 name so existing importers keep working; the reason it
89
- #: carries has changed from "out of scope" to "unavailable on this machine".
90
- VIDEO_OUT_OF_SCOPE = VIDEO_UNAVAILABLE_DETAIL
91
-
92
- #: Longest OCR/caption body kept on the node (a screenshot is not a novel).
93
- MAX_INDEX_TEXT_CHARS = 20_000
94
- #: Summary column budget, matching every other ingest door in the graph.
95
- SUMMARY_CHARS = 500
96
- #: Fixed-width chunking for OCR bodies that outgrow the summary.
97
- IMAGE_CHUNK_CHARS = 900
98
- #: Longest edge of the stored thumbnail, in pixels.
99
- THUMBNAIL_EDGE = 96
100
- #: A thumbnail is a UI affordance, not an archive — drop it past this size.
101
- MAX_THUMBNAIL_CHARS = 24_000
102
-
103
-
104
- def detect_modality(
105
- path: Optional[str] = None, mime_type: Optional[str] = None
106
- ) -> str:
107
- """``text`` | ``image`` | ``audio`` | ``video`` for one candidate file.
108
-
109
- The declared MIME type wins when it carries a usable top-level type: the
110
- capture surface saw the bytes, this function only sees a name. Otherwise
111
- the extension decides, and an unknown extension is ``text`` so existing
112
- behaviour is untouched.
113
- """
114
- declared = str(mime_type or "").strip().lower().split(";")[0].split("/")[0]
115
- if declared in {MODALITY_IMAGE, MODALITY_AUDIO, MODALITY_VIDEO}:
116
- return declared
117
- # The extension tables come before ``mimetypes`` on purpose: they are where
118
- # this module's decisions live (``.mp4`` is video unless someone who saw
119
- # the bytes says otherwise), and the stdlib table varies by platform.
120
- suffix = Path(str(path or "")).suffix.lower()
121
- if suffix in IMAGE_EXTENSIONS:
122
- return MODALITY_IMAGE
123
- if suffix in AUDIO_EXTENSIONS:
124
- return MODALITY_AUDIO
125
- if suffix in VIDEO_EXTENSIONS:
126
- return MODALITY_VIDEO
127
- if path:
128
- guessed, _ = mimetypes.guess_type(str(path))
129
- top = str(guessed or "").split("/")[0]
130
- if top in {MODALITY_IMAGE, MODALITY_AUDIO, MODALITY_VIDEO}:
131
- return top
132
- return MODALITY_TEXT
133
-
134
-
135
- # ── injected capability ports ────────────────────────────────────────────────
136
- @dataclass
137
- class MultimodalPorts:
138
- """The optional model-backed capabilities Brain Core cannot ship itself.
139
-
140
- Every field is a plain callable so the app layer can build it from
141
- ``latticeai.core.embedding_providers`` without Brain Core ever importing
142
- that package. ``None`` everywhere is the honest default: OCR still runs
143
- (``pytesseract`` is a local binary, not a model download), and everything
144
- else reports itself as unavailable.
145
- """
146
-
147
- #: ``(image_path) -> caption or None`` — a loaded VLM, or nothing.
148
- captioner: Optional[Callable[[str], Optional[str]]] = None
149
- #: ``(image_path) -> vector`` in the image space (raises when it cannot).
150
- vision_embedder: Optional[Callable[[str], List[float]]] = None
151
- #: ``(audio_path) -> transcript`` (raises/returns empty when it cannot).
152
- transcriber: Optional[Callable[[str], str]] = None
153
- #: ``(video_path, dest_dir, count) -> [frame paths]`` (v11.2.0). ``None``
154
- #: falls back to ffmpeg on PATH; absent ffmpeg is reported, never faked.
155
- keyframe_extractor: Optional[Callable[..., Any]] = None
156
- #: ``(query_text) -> vector`` in the *image* space (v11.2.0). Only a
157
- #: genuinely shared-space vision model can supply one, which is why it is
158
- #: its own port instead of being assumed from ``vision_embedder``.
159
- text_to_image_embedder: Optional[Callable[[str], List[float]]] = None
160
- #: Identity of the vision model, recorded next to every image vector.
161
- vision_model_id: str = ""
162
- #: ``image`` (own index + late fusion) or ``shared`` (same space as text).
163
- vision_space: str = MODALITY_IMAGE
164
-
165
- def describe(self) -> Dict[str, Any]:
166
- """What this install can honestly do with a picture or a recording."""
167
- return {
168
- "caption": self.captioner is not None,
169
- "vision_embedding": self.vision_embedder is not None,
170
- "transcription": self.transcriber is not None,
171
- "keyframes": self.keyframe_extractor is not None or ffmpeg_available(),
172
- "text_to_image_query": self.text_to_image_embedder is not None,
173
- "vision_model_id": self.vision_model_id,
174
- "vision_space": self.vision_space,
175
- }
176
-
177
-
178
- # ── image facts ──────────────────────────────────────────────────────────────
179
- @dataclass
180
- class ImageFacts:
181
- """Everything observed about one image, and the status of each attempt."""
182
-
183
- path: str
184
- width: Optional[int] = None
185
- height: Optional[int] = None
186
- image_format: Optional[str] = None
187
- mode: Optional[str] = None
188
- #: ``ok`` | ``empty`` | ``unavailable`` | ``failed`` | ``skipped``
189
- ocr_status: str = "skipped"
190
- ocr_text: str = ""
191
- ocr_detail: str = ""
192
- #: ``ok`` | ``unavailable``
193
- caption_status: str = "unavailable"
194
- caption: Optional[str] = None
195
- #: ``ok`` | ``unavailable`` | ``failed``
196
- embedding_status: str = "unavailable"
197
- embedding: Optional[List[float]] = None
198
- embedding_detail: str = ""
199
- thumbnail: Optional[str] = None
200
- #: Set when the file could not be opened as an image at all.
201
- error: str = ""
202
-
203
- @property
204
- def readable(self) -> bool:
205
- return not self.error
206
-
207
- def index_text(self) -> str:
208
- """The text a search engine can actually match this image on."""
209
- parts = [part for part in (self.caption, self.ocr_text) if part]
210
- return "\n".join(parts).strip()[:MAX_INDEX_TEXT_CHARS]
211
-
212
- def as_metadata(self) -> Dict[str, Any]:
213
- """Flat, JSON-safe view stored on the graph node."""
214
- payload: Dict[str, Any] = {
215
- "modality": MODALITY_IMAGE,
216
- "width": self.width,
217
- "height": self.height,
218
- "format": self.image_format,
219
- "mode": self.mode,
220
- "ocr_status": self.ocr_status,
221
- "ocr_chars": len(self.ocr_text),
222
- "caption_status": self.caption_status,
223
- "vision_embedding": self.embedding_status,
224
- }
225
- if self.ocr_text:
226
- payload["ocr_text"] = self.ocr_text
227
- if self.ocr_detail:
228
- payload["ocr_detail"] = self.ocr_detail
229
- if self.caption:
230
- payload["caption"] = self.caption
231
- if self.embedding_detail:
232
- payload["vision_embedding_detail"] = self.embedding_detail
233
- if self.thumbnail:
234
- payload["thumbnail"] = self.thumbnail
235
- if self.error:
236
- payload["image_error"] = self.error
237
- return payload
238
-
239
-
240
- def _open_image(path: str) -> Any:
241
- """Pillow's ``Image.open`` behind a guarded import."""
242
- from PIL import Image # local import: keeps the module importable without it
243
-
244
- return Image.open(str(path))
245
-
246
-
247
- def _thumbnail_data_uri(image: Any, edge: int = THUMBNAIL_EDGE) -> Optional[str]:
248
- """A tiny inline PNG the Evidence panel can render with no new route.
249
-
250
- Serving the original file would mean either a new static route over the
251
- user's disk or reusing ``/local/serve``, which exists precisely to make
252
- every read pass an explicit approval. A 96px data URI on the node dodges
253
- both: it is already inside the graph the user is looking at.
254
- """
255
- try:
256
- small = image.copy()
257
- small.thumbnail((edge, edge))
258
- if small.mode not in {"RGB", "L"}:
259
- small = small.convert("RGB")
260
- buffer = io.BytesIO()
261
- small.save(buffer, format="PNG")
262
- except Exception: # noqa: BLE001 — a missing thumbnail is not a failed ingest
263
- quiet()
264
- return None
265
- encoded = base64.b64encode(buffer.getvalue()).decode("ascii")
266
- if len(encoded) > MAX_THUMBNAIL_CHARS:
267
- return None
268
- return f"data:image/png;base64,{encoded}"
269
-
270
-
271
- def _run_ocr(image: Any) -> Dict[str, str]:
272
- """OCR through ``pytesseract`` when it is installed, honestly otherwise."""
273
- try:
274
- import pytesseract # optional local binary + wrapper
275
- except Exception as exc: # noqa: BLE001 — absence is a state, not an error
276
- return {"status": "unavailable", "text": "", "detail": str(exc)}
277
- try:
278
- text = str(pytesseract.image_to_string(image) or "").strip()
279
- except Exception as exc: # noqa: BLE001 — a broken OCR runtime is a state
280
- return {"status": "failed", "text": "", "detail": str(exc)}
281
- if not text:
282
- return {"status": "empty", "text": "", "detail": "no text found in the image"}
283
- return {"status": "ok", "text": text[:MAX_INDEX_TEXT_CHARS], "detail": ""}
284
-
285
-
286
- def extract_image_facts(
287
- path: str,
288
- *,
289
- ports: Optional[MultimodalPorts] = None,
290
- ocr: bool = True,
291
- thumbnail: bool = True,
292
- ) -> ImageFacts:
293
- """Observe one image: size, OCR, caption, vector — each with its status.
294
-
295
- Never raises. An unreadable file returns ``ImageFacts(error=...)`` so the
296
- caller can record "we saw this file and could not read it" instead of
297
- losing the memory entirely.
298
- """
299
- ports = ports or MultimodalPorts()
300
- facts = ImageFacts(path=str(path))
301
- try:
302
- with _open_image(path) as image:
303
- facts.width = int(image.width)
304
- facts.height = int(image.height)
305
- facts.image_format = image.format
306
- facts.mode = image.mode
307
- if ocr:
308
- result = _run_ocr(image)
309
- facts.ocr_status = result["status"]
310
- facts.ocr_text = result["text"]
311
- facts.ocr_detail = result["detail"]
312
- if thumbnail:
313
- facts.thumbnail = _thumbnail_data_uri(image)
314
- except Exception as exc: # noqa: BLE001 — an unreadable image is a state
315
- facts.error = str(exc)
316
- return facts
317
-
318
- if ports.captioner is not None:
319
- caption = _safe_caption(ports.captioner, facts.path)
320
- if caption:
321
- facts.caption = caption
322
- facts.caption_status = "ok"
323
-
324
- if ports.vision_embedder is not None:
325
- _apply_vision_embedding(facts, ports.vision_embedder)
326
- return facts
327
-
328
-
329
- def _safe_caption(
330
- captioner: Callable[[str], Optional[str]], path: str
331
- ) -> Optional[str]:
332
- """Ask the VLM; a failure means *no caption*, never an invented one."""
333
- try:
334
- caption = captioner(path)
335
- except Exception: # noqa: BLE001 — a broken captioner must not fail an ingest
336
- quiet()
337
- return None
338
- cleaned = str(caption or "").strip()
339
- return cleaned or None
340
-
341
-
342
- def _apply_vision_embedding(
343
- facts: ImageFacts, embedder: Callable[[str], List[float]]
344
- ) -> None:
345
- try:
346
- vector = [float(value) for value in embedder(facts.path)]
347
- except Exception as exc: # noqa: BLE001 — an absent model is not a failed ingest
348
- facts.embedding_status = "failed"
349
- facts.embedding_detail = str(exc)
350
- return
351
- if not vector:
352
- facts.embedding_status = "failed"
353
- facts.embedding_detail = "vision provider returned an empty vector"
354
- return
355
- facts.embedding = vector
356
- facts.embedding_status = "ok"
357
-
358
-
359
- # ── extraction quality for pictures ──────────────────────────────────────────
360
- def image_quality_score(facts: ImageFacts) -> Dict[str, Any]:
361
- """``{"score": float, "reasons": [...]}`` for an image memory.
362
-
363
- Deliberately *not* a judgement about the photograph: it scores how much of
364
- this image the Brain can actually retrieve later. Pixels alone are worth
365
- little to a text query, OCR text is worth the most, a caption is worth a
366
- lot, and a vector is worth something even without either.
367
- """
368
- if not facts.readable:
369
- return {"score": 0.0, "reasons": ["image_unreadable"]}
370
- reasons: List[str] = []
371
- score = 0.15 # we know it is an image and how big it is
372
- if facts.ocr_text:
373
- # 400+ characters of recognized text is a page, not a label.
374
- score += 0.45 * min(1.0, len(facts.ocr_text) / 400.0)
375
- reasons.append("ocr_text")
376
- elif facts.ocr_status == "unavailable":
377
- reasons.append("ocr_unavailable")
378
- elif facts.ocr_status == "skipped":
379
- reasons.append("ocr_skipped")
380
- else:
381
- reasons.append("no_ocr_text")
382
- if facts.caption:
383
- score += 0.3
384
- reasons.append("vision_caption")
385
- else:
386
- reasons.append("no_vision_caption")
387
- if facts.embedding_status == "ok":
388
- score += 0.1
389
- reasons.append("vision_embedding")
390
- return {"score": round(max(0.0, min(1.0, score)), 4), "reasons": reasons}
391
-
392
-
393
- # ── graph write ──────────────────────────────────────────────────────────────
394
- def _sha256_file(path: Path) -> str:
395
- digest = hashlib.sha256()
396
- with path.open("rb") as handle:
397
- for block in iter(lambda: handle.read(1024 * 1024), b""):
398
- digest.update(block)
399
- return digest.hexdigest()
400
-
401
-
402
- def _sha256_text(text: str) -> str:
403
- return hashlib.sha256(str(text).encode("utf-8", "ignore")).hexdigest()
404
-
405
-
406
- def _split_index_text(text: str) -> List[str]:
407
- """Fixed-width split for OCR bodies — no markdown or code structure here."""
408
- body = str(text or "").strip()
409
- if len(body) <= SUMMARY_CHARS:
410
- return []
411
- return [
412
- body[start : start + IMAGE_CHUNK_CHARS]
413
- for start in range(0, len(body), IMAGE_CHUNK_CHARS)
414
- ]
415
-
416
-
417
- def image_node_id(content_hash: str, workspace_id: Optional[str] = None) -> str:
418
- """Workspace-scoped, content-addressed id — re-ingesting is idempotent."""
419
- scoped = f"{workspace_id or 'legacy-global'}|{content_hash}"
420
- return f"image:{_sha256_text(scoped)[:24]}"
421
-
422
-
423
- def write_image_memory(
424
- store: Any,
425
- *,
426
- path: Path,
427
- facts: ImageFacts,
428
- title: str,
429
- source_type: str = MODALITY_IMAGE,
430
- source_uri: Optional[str] = None,
431
- owner: Optional[str] = None,
432
- workspace_id: Optional[str] = None,
433
- conversation_id: Optional[str] = None,
434
- captured_at: Optional[str] = None,
435
- modified_at: Optional[str] = None,
436
- permissions: Optional[Dict[str, Any]] = None,
437
- extra_metadata: Optional[Dict[str, Any]] = None,
438
- ) -> Dict[str, Any]:
439
- """Write one ``Image`` node (plus ``ImageText``/chunks) into the graph.
440
-
441
- The node is the image itself rather than a ``Document`` that happens to be
442
- a picture, because that is what the rest of the product reasons about: the
443
- graph already declares ``Image``/``ImageText``/``CONTAINS_IMAGE``, and
444
- ``hybrid_search`` already ranks ``Image`` as a first-class result type.
445
-
446
- Uses the same cross-mixin write door (``_upsert_node``/``_upsert_edge``/
447
- ``_upsert_chunk``) that every other ingest path uses — the store exposes no
448
- public node writer, and inventing a second one for images would be a
449
- parallel write path to keep in sync forever.
450
- """
451
- captured_at = captured_at or utc_now_iso()
452
- content_hash = _sha256_file(path)
453
- node_id = image_node_id(content_hash, workspace_id)
454
- index_text = facts.index_text()
455
- metadata: Dict[str, Any] = {
456
- "filename": path.name,
457
- "file_path": str(path),
458
- "ext": path.suffix.lower(),
459
- "bytes": path.stat().st_size,
460
- "sha256": content_hash,
461
- "content_hash": content_hash,
462
- "source_type": source_type,
463
- "source_uri": source_uri or str(path),
464
- "captured_at": captured_at,
465
- "modified_at": modified_at,
466
- "owner": owner,
467
- "workspace_id": workspace_id,
468
- "permissions": permissions or {},
469
- "conversation_id": conversation_id,
470
- **facts.as_metadata(),
471
- **(extra_metadata or {}),
472
- }
473
- # Honest summary: when nothing could be read out of the picture, say so
474
- # rather than leaving a blank card that looks like a failed render.
475
- summary = index_text[:SUMMARY_CHARS] or f"[{MODALITY_IMAGE}] {path.name}"
476
- chunk_ids: List[str] = []
477
-
478
- with store._connect() as conn:
479
- duplicate = (
480
- conn.execute("SELECT 1 FROM nodes WHERE id=? LIMIT 1", (node_id,)).fetchone()
481
- is not None
482
- )
483
- store._upsert_node(
484
- conn,
485
- node_id,
486
- "Image",
487
- title or path.name,
488
- summary=summary,
489
- metadata=metadata,
490
- raw=metadata,
491
- owner=owner,
492
- workspace_id=workspace_id,
493
- )
494
- if facts.ocr_text:
495
- image_text_id = f"imagetext:{_sha256_text(f'{node_id}:ocr')[:24]}"
496
- store._upsert_node(
497
- conn,
498
- image_text_id,
499
- "ImageText",
500
- f"{path.name} OCR",
501
- summary=facts.ocr_text[:700],
502
- metadata={
503
- "source_node": node_id,
504
- "chars": len(facts.ocr_text),
505
- "workspace_id": workspace_id,
506
- },
507
- owner=owner,
508
- workspace_id=workspace_id,
509
- )
510
- store._upsert_edge(
511
- conn,
512
- node_id,
513
- image_text_id,
514
- "포함함",
515
- weight=0.8,
516
- metadata={"source": "ocr", "workspace_id": workspace_id},
517
- )
518
- for index, piece in enumerate(_split_index_text(index_text)):
519
- chunk_id = f"chunk:{_sha256_text(f'{node_id}:{index}:{piece}')[:24]}"
520
- chunk_ids.append(chunk_id)
521
- chunk_meta = {
522
- "index": index,
523
- "source_node": node_id,
524
- "workspace_id": workspace_id,
525
- "modality": MODALITY_IMAGE,
526
- }
527
- store._upsert_node(
528
- conn,
529
- chunk_id,
530
- "Chunk",
531
- f"{path.name} chunk {index + 1}",
532
- summary=piece[:SUMMARY_CHARS],
533
- metadata=chunk_meta,
534
- owner=owner,
535
- workspace_id=workspace_id,
536
- )
537
- store._upsert_chunk(
538
- conn,
539
- chunk_id=chunk_id,
540
- source_node=node_id,
541
- text=piece,
542
- metadata=chunk_meta,
543
- )
544
- store._upsert_edge(conn, node_id, chunk_id, "포함함")
545
- # Concepts come from what is *in* the picture, never from its name:
546
- # "IMG_2381" is not a topic, and turning filenames into concept nodes
547
- # would fill the graph with hubs that mean nothing.
548
- concept_ids = _attach_concepts(
549
- store,
550
- conn,
551
- node_id=node_id,
552
- text=index_text,
553
- owner=owner,
554
- workspace_id=workspace_id,
555
- )
556
- source_node_id = _attach_source(
557
- store,
558
- conn,
559
- node_id=node_id,
560
- source_type=source_type,
561
- source_uri=source_uri or str(path),
562
- title=title or path.name,
563
- content_hash=content_hash,
564
- captured_at=captured_at,
565
- owner=owner,
566
- workspace_id=workspace_id,
567
- )
568
- metadata["concepts"] = concept_ids
569
- return {
570
- "node_id": node_id,
571
- "type": "Image",
572
- "title": title or path.name,
573
- "sha256": content_hash,
574
- "content_hash": content_hash,
575
- "source_node_id": source_node_id,
576
- "chunk_ids": chunk_ids,
577
- "chunk_count": len(chunk_ids),
578
- "duplicate": duplicate,
579
- "captured_at": captured_at,
580
- "metadata": metadata,
581
- }
582
-
583
-
584
- def _attach_concepts(
585
- store: Any,
586
- conn: Any,
587
- *,
588
- node_id: str,
589
- text: str,
590
- owner: Optional[str],
591
- workspace_id: Optional[str],
592
- ) -> List[str]:
593
- """Pull concepts out of the caption/OCR so the picture joins the graph.
594
-
595
- This is what "the caption contributes to the graph" means concretely: the
596
- same extractor every text door uses, over the same text, producing the same
597
- ``Concept``/``Feature``/… nodes and ``포함함`` edges — so a photo of a
598
- whiteboard about Q3 planning is one hop from every note about Q3 planning,
599
- instead of being an island that only exact search can reach.
600
-
601
- An image with nothing readable in it yields no concepts, which is correct:
602
- there is nothing to say about it.
603
- """
604
- body = str(text or "").strip()
605
- if not body:
606
- return []
607
- from .graph._kg_common import ( # local: keeps this module's import light
608
- _classify_node_type,
609
- _extract_concepts,
610
- )
611
-
612
- # The id derivation is imported rather than re-derived: two copies of a
613
- # node-id rule diverge, and a diverged id is a duplicate concept nobody
614
- # can see.
615
- from .graph.ingest import _scoped_slug_id
616
-
617
- concept_ids: List[str] = []
618
- for concept in _extract_concepts(body, limit=10):
619
- node_type = _classify_node_type(concept, body)
620
- concept_id = _scoped_slug_id(node_type.lower(), concept, workspace_id)
621
- store._upsert_node(
622
- conn,
623
- concept_id,
624
- node_type,
625
- concept,
626
- metadata={
627
- "auto_extracted": True,
628
- "source_node": node_id,
629
- "modality": MODALITY_IMAGE,
630
- "workspace_id": workspace_id,
631
- },
632
- owner=owner,
633
- workspace_id=workspace_id,
634
- )
635
- store._upsert_edge(conn, node_id, concept_id, "포함함", weight=0.8)
636
- concept_ids.append(concept_id)
637
- return concept_ids
638
-
639
-
640
- def _attach_source(
641
- store: Any,
642
- conn: Any,
643
- *,
644
- node_id: str,
645
- source_type: str,
646
- source_uri: str,
647
- title: str,
648
- content_hash: str,
649
- captured_at: str,
650
- owner: Optional[str],
651
- workspace_id: Optional[str],
652
- ) -> Optional[str]:
653
- """Link the image to a ``Source`` node when the store can make one."""
654
- attach = getattr(store, "_attach_source_node", None)
655
- if not callable(attach):
656
- return None
657
- return str(
658
- attach(
659
- conn,
660
- node_id,
661
- source_type=source_type,
662
- source_uri=source_uri,
663
- title=title,
664
- content_hash=content_hash,
665
- captured_at=captured_at,
666
- extra={"owner": owner, "workspace_id": workspace_id, "modality": MODALITY_IMAGE},
667
- )
668
- )
669
-
670
-
671
- # ── audio ────────────────────────────────────────────────────────────────────
672
- @dataclass
673
- class AudioFacts:
674
- """A recording, its transcript, and how honestly we got one."""
675
-
676
- path: str
677
- #: ``ok`` | ``unavailable`` | ``failed`` | ``supplied``
678
- transcription_status: str = "unavailable"
679
- transcript: str = ""
680
- detail: str = ""
681
- segments: List[Dict[str, Any]] = field(default_factory=list)
682
-
683
- @property
684
- def searchable(self) -> bool:
685
- return bool(self.transcript)
686
-
687
-
688
- def transcribe_audio(
689
- path: str,
690
- *,
691
- ports: Optional[MultimodalPorts] = None,
692
- transcript: Optional[str] = None,
693
- ) -> AudioFacts:
694
- """Transcribe a recording through the injected port, or say why not.
695
-
696
- ``transcript`` lets a caller that already has text (a phone's own
697
- dictation, ``VoiceCaptureService``) skip the local model entirely. An
698
- absent transcriber yields ``transcription_status="unavailable"`` and an
699
- empty transcript — the recording is still remembered by title and path,
700
- and the result never claims it is searchable.
701
- """
702
- ports = ports or MultimodalPorts()
703
- supplied = str(transcript or "").strip()
704
- if supplied:
705
- return AudioFacts(path=str(path), transcription_status="supplied", transcript=supplied)
706
- if ports.transcriber is None:
707
- return AudioFacts(
708
- path=str(path),
709
- transcription_status="unavailable",
710
- detail="no local transcriber is configured",
711
- )
712
- try:
713
- text = str(ports.transcriber(str(path)) or "").strip()
714
- except Exception as exc: # noqa: BLE001 — a broken transcriber is a state
715
- return AudioFacts(path=str(path), transcription_status="failed", detail=str(exc))
716
- if not text:
717
- return AudioFacts(
718
- path=str(path),
719
- transcription_status="failed",
720
- detail="the transcriber returned no text",
721
- )
722
- return AudioFacts(path=str(path), transcription_status="ok", transcript=text)
723
-
724
-
725
- def audio_quality_score(facts: AudioFacts) -> Dict[str, Any]:
726
- """How much of this recording is actually retrievable later."""
727
- if not facts.transcript:
728
- return {"score": 0.0, "reasons": ["no_transcript"]}
729
- # A transcript is text: length is the only honest extra signal here, and
730
- # the text pipeline scores the wording itself downstream.
731
- score = 0.5 + 0.5 * min(1.0, len(facts.transcript) / 400.0)
732
- return {"score": round(score, 4), "reasons": ["transcript"]}
733
-
734
-
735
- # ── video (v11.2.0) ──────────────────────────────────────────────────────────
736
- #: The decoder. Looked up by name on PATH and never bundled — a product that
737
- #: cannot decode a ``.mov`` says so instead of shipping a codec pack.
738
- FFMPEG_BINARY = "ffmpeg"
739
- #: Keyframes kept per video. Four is a memory of a video, not a copy of it.
740
- DEFAULT_KEYFRAMES = 4
741
- #: Frames ffmpeg's ``thumbnail`` filter considers before picking one. Larger
742
- #: windows mean more representative frames and a slower pass.
743
- KEYFRAME_WINDOW = 300
744
- KEYFRAME_TIMEOUT_SECONDS = 120
745
- #: Graph node type for a video. ``NodeType.VIDEO`` normalizes this on the KG v2
746
- #: write side; the legacy tables keep the label verbatim, like ``Audio``.
747
- VIDEO_NODE_TYPE = "Video"
748
- #: Source type stamped on the extracted stills so a keyframe is never mistaken
749
- #: for a photograph the user took.
750
- VIDEO_FRAME_SOURCE_TYPE = "video_keyframe"
751
- VIDEO_FRAME_RELATION = "CONTAINS_IMAGE"
752
- #: Longest subtitle body kept, matching the OCR/caption ceiling.
753
- MAX_SUBTITLE_CHARS = MAX_INDEX_TEXT_CHARS
754
-
755
- _SRT_INDEX_RE = re.compile(r"^\d+$")
756
- _TIMECODE_RE = re.compile(r"\d{1,2}:\d{2}:\d{2}[.,]\d{1,3}\s*-->")
757
- _CUE_TAG_RE = re.compile(r"</?[a-zA-Z][^>]*>")
758
-
759
-
760
- def _which_ffmpeg() -> Optional[str]:
761
- """Absolute path to ffmpeg, or ``None``. The one probe, seamed for tests."""
762
- return shutil.which(FFMPEG_BINARY)
763
-
764
-
765
- def ffmpeg_available() -> bool:
766
- """Whether this machine can decode a video at all (honest, never assumed)."""
767
- return _which_ffmpeg() is not None
768
-
769
-
770
- def _run_ffmpeg(binary: str, args: List[str]) -> int:
771
- """Run one fixed binary with an argv list — no shell, no user strings."""
772
- completed = subprocess.run( # noqa: S603 — argv list, fixed binary, no shell
773
- [binary, *args],
774
- stdout=subprocess.DEVNULL,
775
- stderr=subprocess.DEVNULL,
776
- timeout=KEYFRAME_TIMEOUT_SECONDS,
777
- check=False,
778
- )
779
- return int(completed.returncode)
780
-
781
-
782
- def find_subtitle(path: Any) -> Optional[Path]:
783
- """The ``.srt``/``.vtt`` sitting next to a video under the same basename."""
784
- video = Path(str(path))
785
- for suffix in SUBTITLE_EXTENSIONS:
786
- candidate = video.with_suffix(f".{suffix}")
787
- if candidate.is_file():
788
- return candidate
789
- return None
790
-
791
-
792
- def parse_subtitles(text: str) -> str:
793
- """Strip SRT/WebVTT scaffolding down to the words that were said.
794
-
795
- Cue numbers, timecodes, ``WEBVTT`` headers, ``NOTE`` blocks and inline
796
- ``<c>``/``<i>`` tags carry no recall value and would otherwise dominate a
797
- chunk. Consecutive duplicate lines (the usual rolling-caption artefact) are
798
- collapsed. Deliberately a small parser, not a dependency: the format is two
799
- rules deep and a library here would sit in the ingest path forever.
800
- """
801
- lines: List[str] = []
802
- for raw in str(text or "").splitlines():
803
- line = raw.strip().lstrip("")
804
- if not line or line.upper().startswith("WEBVTT") or line.startswith("NOTE"):
805
- continue
806
- if _SRT_INDEX_RE.match(line) or _TIMECODE_RE.search(line):
807
- continue
808
- cleaned = _CUE_TAG_RE.sub("", line).strip()
809
- if not cleaned:
810
- continue
811
- if lines and lines[-1] == cleaned:
812
- continue
813
- lines.append(cleaned)
814
- return "\n".join(lines)[:MAX_SUBTITLE_CHARS]
815
-
816
-
817
- @dataclass
818
- class VideoFacts:
819
- """What could actually be observed in one video, and how each attempt went."""
820
-
821
- path: str
822
- #: ``ok`` | ``unavailable`` | ``failed`` | ``empty``
823
- keyframe_status: str = "unavailable"
824
- keyframes: List[str] = field(default_factory=list)
825
- keyframe_detail: str = ""
826
- #: ``ok`` | ``absent`` | ``unreadable`` | ``empty``
827
- subtitle_status: str = "absent"
828
- subtitle_path: Optional[str] = None
829
- subtitle_text: str = ""
830
-
831
- @property
832
- def searchable(self) -> bool:
833
- """Whether anything in this video can be matched by a typed question."""
834
- return bool(self.subtitle_text)
835
-
836
- def as_metadata(self) -> Dict[str, Any]:
837
- payload: Dict[str, Any] = {
838
- "modality": MODALITY_VIDEO,
839
- "video_path": self.path,
840
- "keyframes": len(self.keyframes),
841
- "keyframe_status": self.keyframe_status,
842
- "subtitles": self.subtitle_status,
843
- "searchable": self.searchable,
844
- }
845
- if self.keyframe_detail:
846
- payload["keyframe_detail"] = self.keyframe_detail
847
- if self.subtitle_path:
848
- payload["subtitle_path"] = self.subtitle_path
849
- return payload
850
-
851
-
852
- def extract_keyframes(
853
- path: Any,
854
- dest_dir: Any,
855
- *,
856
- count: int = DEFAULT_KEYFRAMES,
857
- ports: Optional[MultimodalPorts] = None,
858
- ) -> Dict[str, Any]:
859
- """Pull up to ``count`` representative stills out of a video.
860
-
861
- An injected ``ports.keyframe_extractor`` wins outright — that is the seam
862
- an install with its own decoder (or a test) uses. Otherwise ffmpeg's
863
- ``thumbnail`` filter picks the most representative frame from each window
864
- of :data:`KEYFRAME_WINDOW` frames, which is one pass and no probing.
865
-
866
- Never raises. A missing decoder, a non-zero exit, and a video too short to
867
- yield a single frame are three different states and each says so.
868
- """
869
- ports = ports or MultimodalPorts()
870
- video = Path(str(path))
871
- dest = Path(str(dest_dir))
872
- wanted = max(1, int(count))
873
- if ports.keyframe_extractor is not None:
874
- return _injected_keyframes(ports.keyframe_extractor, video, dest, wanted)
875
- binary = _which_ffmpeg()
876
- if binary is None:
877
- return {"status": "unavailable", "frames": [], "detail": VIDEO_UNAVAILABLE_DETAIL}
878
- dest.mkdir(parents=True, exist_ok=True)
879
- args = [
880
- "-nostdin", "-loglevel", "error", "-y",
881
- "-i", str(video),
882
- "-vf", f"thumbnail={KEYFRAME_WINDOW}",
883
- "-frames:v", str(wanted),
884
- "-vsync", "vfr",
885
- str(dest / "keyframe-%03d.jpg"),
886
- ]
887
- try:
888
- code = _run_ffmpeg(binary, args)
889
- except Exception as exc: # noqa: BLE001 — a broken decoder is a state
890
- return {"status": "failed", "frames": [], "detail": f"ffmpeg failed: {exc}"}
891
- frames = sorted(str(p) for p in dest.glob("keyframe-*.jpg"))
892
- if code != 0 and not frames:
893
- return {
894
- "status": "failed",
895
- "frames": [],
896
- "detail": f"ffmpeg exited with status {code}",
897
- }
898
- if not frames:
899
- return {
900
- "status": "empty",
901
- "frames": [],
902
- "detail": "ffmpeg produced no frames from this video",
903
- }
904
- return {"status": "ok", "frames": frames[:wanted], "detail": ""}
905
-
906
-
907
- def _injected_keyframes(
908
- extractor: Callable[..., Any], video: Path, dest: Path, wanted: int
909
- ) -> Dict[str, Any]:
910
- """Run a caller-supplied extractor; a failure is reported, never raised."""
911
- try:
912
- produced = extractor(str(video), str(dest), wanted)
913
- except Exception as exc: # noqa: BLE001 — an injected port is not trusted more
914
- return {"status": "failed", "frames": [], "detail": f"keyframe port failed: {exc}"}
915
- frames = [str(item) for item in (produced or [])][:wanted]
916
- if not frames:
917
- return {
918
- "status": "empty",
919
- "frames": [],
920
- "detail": "the keyframe port produced no frames",
921
- }
922
- return {"status": "ok", "frames": frames, "detail": ""}
923
-
924
-
925
- def read_video_facts(
926
- path: Any,
927
- dest_dir: Any,
928
- *,
929
- count: int = DEFAULT_KEYFRAMES,
930
- ports: Optional[MultimodalPorts] = None,
931
- subtitle_text: Optional[str] = None,
932
- ) -> VideoFacts:
933
- """Observe one video: its keyframes and its companion subtitles."""
934
- facts = VideoFacts(path=str(path))
935
- outcome = extract_keyframes(path, dest_dir, count=count, ports=ports)
936
- facts.keyframe_status = str(outcome["status"])
937
- facts.keyframes = list(outcome["frames"])
938
- facts.keyframe_detail = str(outcome["detail"])
939
- supplied = str(subtitle_text or "").strip()
940
- if supplied:
941
- facts.subtitle_status = "ok"
942
- facts.subtitle_text = parse_subtitles(supplied)
943
- return facts
944
- companion = find_subtitle(path)
945
- if companion is None:
946
- return facts
947
- facts.subtitle_path = str(companion)
948
- try:
949
- raw = companion.read_text(encoding="utf-8", errors="ignore")
950
- except OSError as exc:
951
- facts.subtitle_status = "unreadable"
952
- facts.keyframe_detail = (facts.keyframe_detail or "").strip()
953
- facts.subtitle_text = ""
954
- quiet()
955
- facts.subtitle_path = f"{companion} ({exc.strerror or 'unreadable'})"
956
- return facts
957
- parsed = parse_subtitles(raw)
958
- facts.subtitle_status = "ok" if parsed else "empty"
959
- facts.subtitle_text = parsed
960
- return facts
961
-
962
-
963
- def video_quality_score(facts: VideoFacts) -> Dict[str, Any]:
964
- """How much of this video the Brain can actually retrieve later.
965
-
966
- Same principle as :func:`image_quality_score`: not a judgement about the
967
- footage. Subtitles are worth the most (they are the words), keyframes are
968
- worth something (they can be OCR'd and seen), and a video with neither is
969
- a filename.
970
- """
971
- reasons: List[str] = []
972
- score = 0.1 # we know it is a video and where it lives
973
- if facts.subtitle_text:
974
- score += 0.55 * min(1.0, len(facts.subtitle_text) / 400.0)
975
- reasons.append("subtitles")
976
- else:
977
- reasons.append(f"no_subtitles_{facts.subtitle_status}")
978
- if facts.keyframes:
979
- score += min(0.35, 0.1 * len(facts.keyframes))
980
- reasons.append("keyframes")
981
- else:
982
- reasons.append(f"no_keyframes_{facts.keyframe_status}")
983
- return {"score": round(max(0.0, min(1.0, score)), 4), "reasons": reasons}
984
-
985
-
986
- def video_node_id(content_hash: str, workspace_id: Optional[str] = None) -> str:
987
- """Workspace-scoped, content-addressed id — re-ingesting is idempotent."""
988
- scoped = f"{workspace_id or 'legacy-global'}|{content_hash}"
989
- return f"video:{_sha256_text(scoped)[:24]}"
990
-
991
-
992
- def video_frame_dir(blob_dir: Any, content_hash: str) -> Path:
993
- """Where this video's stills live: content-addressed, stable across runs.
994
-
995
- Deliberately under the Brain's own blob directory rather than a temp dir.
996
- A frame referenced by an ``Image`` node has to still be there the next time
997
- someone opens that memory, and a backup that copies the blobs copies these.
998
- """
999
- return Path(str(blob_dir)) / "video_frames" / str(content_hash)[:32]
1000
-
1001
-
1002
- def write_video_memory(
1003
- store: Any,
1004
- *,
1005
- path: Path,
1006
- facts: VideoFacts,
1007
- title: str,
1008
- source_type: str = MODALITY_VIDEO,
1009
- source_uri: Optional[str] = None,
1010
- owner: Optional[str] = None,
1011
- workspace_id: Optional[str] = None,
1012
- conversation_id: Optional[str] = None,
1013
- captured_at: Optional[str] = None,
1014
- modified_at: Optional[str] = None,
1015
- permissions: Optional[Dict[str, Any]] = None,
1016
- extra_metadata: Optional[Dict[str, Any]] = None,
1017
- ports: Optional[MultimodalPorts] = None,
1018
- ) -> Dict[str, Any]:
1019
- """Write one ``Video`` node, its keyframes, and its subtitles.
1020
-
1021
- Each keyframe goes through the **existing image path** — the same
1022
- :func:`extract_image_facts` and :func:`write_image_memory` a photograph
1023
- uses — so a still from a video is OCR'd, captioned, vectorized and made
1024
- searchable by exactly the machinery that already does that, and joined to
1025
- its video by ``CONTAINS_IMAGE``. Subtitles ride the ordinary text index as
1026
- chunks. Nothing about video gets its own retrieval path.
1027
-
1028
- The frames are written before the video node so their (short-lived) write
1029
- transactions never nest inside the video's.
1030
- """
1031
- ports = ports or MultimodalPorts()
1032
- captured_at = captured_at or utc_now_iso()
1033
- content_hash = _sha256_file(path)
1034
- node_id = video_node_id(content_hash, workspace_id)
1035
- frames = _write_keyframes(
1036
- store,
1037
- facts=facts,
1038
- title=title,
1039
- owner=owner,
1040
- workspace_id=workspace_id,
1041
- conversation_id=conversation_id,
1042
- captured_at=captured_at,
1043
- permissions=permissions,
1044
- ports=ports,
1045
- )
1046
- body = facts.subtitle_text or (
1047
- f"[{MODALITY_VIDEO}] {title}\n"
1048
- "이 영상에는 자막이 없어 말의 내용은 검색되지 않습니다 — "
1049
- "대신 대표 장면 이미지로 찾을 수 있습니다."
1050
- )
1051
- metadata: Dict[str, Any] = {
1052
- "filename": path.name,
1053
- "file_path": str(path),
1054
- "ext": path.suffix.lower(),
1055
- "bytes": path.stat().st_size,
1056
- "sha256": content_hash,
1057
- "content_hash": content_hash,
1058
- "source_type": source_type,
1059
- "source_uri": source_uri or str(path),
1060
- "captured_at": captured_at,
1061
- "modified_at": modified_at,
1062
- "owner": owner,
1063
- "workspace_id": workspace_id,
1064
- "permissions": permissions or {},
1065
- "conversation_id": conversation_id,
1066
- "keyframe_nodes": [frame["node_id"] for frame in frames],
1067
- **facts.as_metadata(),
1068
- **(extra_metadata or {}),
1069
- }
1070
- # Honest card: a video nobody captioned says so, rather than rendering as a
1071
- # blank summary that looks like a failed read.
1072
- summary = body[:SUMMARY_CHARS]
1073
- chunk_ids: List[str] = []
1074
- with store._connect() as conn:
1075
- duplicate = (
1076
- conn.execute("SELECT 1 FROM nodes WHERE id=? LIMIT 1", (node_id,)).fetchone()
1077
- is not None
1078
- )
1079
- store._upsert_node(
1080
- conn,
1081
- node_id,
1082
- VIDEO_NODE_TYPE,
1083
- title or path.name,
1084
- summary=summary,
1085
- metadata=metadata,
1086
- raw=metadata,
1087
- owner=owner,
1088
- workspace_id=workspace_id,
1089
- )
1090
- for frame in frames:
1091
- store._upsert_edge(
1092
- conn,
1093
- node_id,
1094
- frame["node_id"],
1095
- VIDEO_FRAME_RELATION,
1096
- weight=0.8,
1097
- metadata={
1098
- "source": VIDEO_FRAME_SOURCE_TYPE,
1099
- "index": frame["index"],
1100
- "workspace_id": workspace_id,
1101
- },
1102
- )
1103
- for index, piece in enumerate(_split_index_text(facts.subtitle_text)):
1104
- chunk_id = f"chunk:{_sha256_text(f'{node_id}:{index}:{piece}')[:24]}"
1105
- chunk_ids.append(chunk_id)
1106
- chunk_meta = {
1107
- "index": index,
1108
- "source_node": node_id,
1109
- "workspace_id": workspace_id,
1110
- "modality": MODALITY_VIDEO,
1111
- }
1112
- store._upsert_node(
1113
- conn,
1114
- chunk_id,
1115
- "Chunk",
1116
- f"{path.name} chunk {index + 1}",
1117
- summary=piece[:SUMMARY_CHARS],
1118
- metadata=chunk_meta,
1119
- owner=owner,
1120
- workspace_id=workspace_id,
1121
- )
1122
- store._upsert_chunk(
1123
- conn,
1124
- chunk_id=chunk_id,
1125
- source_node=node_id,
1126
- text=piece,
1127
- metadata=chunk_meta,
1128
- )
1129
- store._upsert_edge(conn, node_id, chunk_id, "포함함")
1130
- concept_ids = _attach_concepts(
1131
- store,
1132
- conn,
1133
- node_id=node_id,
1134
- text=facts.subtitle_text,
1135
- owner=owner,
1136
- workspace_id=workspace_id,
1137
- )
1138
- source_node_id = _attach_source(
1139
- store,
1140
- conn,
1141
- node_id=node_id,
1142
- source_type=source_type,
1143
- source_uri=source_uri or str(path),
1144
- title=title or path.name,
1145
- content_hash=content_hash,
1146
- captured_at=captured_at,
1147
- owner=owner,
1148
- workspace_id=workspace_id,
1149
- )
1150
- metadata["concepts"] = concept_ids
1151
- return {
1152
- "node_id": node_id,
1153
- "type": VIDEO_NODE_TYPE,
1154
- "title": title or path.name,
1155
- "sha256": content_hash,
1156
- "content_hash": content_hash,
1157
- "source_node_id": source_node_id,
1158
- "chunk_ids": chunk_ids,
1159
- "chunk_count": len(chunk_ids),
1160
- "duplicate": duplicate,
1161
- "captured_at": captured_at,
1162
- "keyframes": frames,
1163
- "metadata": metadata,
1164
- }
1165
-
1166
-
1167
- def _write_keyframes(
1168
- store: Any,
1169
- *,
1170
- facts: VideoFacts,
1171
- title: str,
1172
- owner: Optional[str],
1173
- workspace_id: Optional[str],
1174
- conversation_id: Optional[str],
1175
- captured_at: str,
1176
- permissions: Optional[Dict[str, Any]],
1177
- ports: MultimodalPorts,
1178
- ) -> List[Dict[str, Any]]:
1179
- """Every extracted still, through the ordinary image door."""
1180
- written: List[Dict[str, Any]] = []
1181
- for index, frame_path in enumerate(facts.keyframes):
1182
- frame = Path(frame_path)
1183
- image_facts = extract_image_facts(str(frame), ports=ports)
1184
- if not image_facts.readable:
1185
- # A frame ffmpeg wrote that Pillow cannot open is a state worth
1186
- # skipping, not worth failing the whole video over.
1187
- continue
1188
- result = write_image_memory(
1189
- store,
1190
- path=frame,
1191
- facts=image_facts,
1192
- title=f"{title} · 장면 {index + 1}",
1193
- source_type=VIDEO_FRAME_SOURCE_TYPE,
1194
- source_uri=str(frame),
1195
- owner=owner,
1196
- workspace_id=workspace_id,
1197
- conversation_id=conversation_id,
1198
- captured_at=captured_at,
1199
- permissions=permissions,
1200
- extra_metadata={
1201
- "modality": MODALITY_IMAGE,
1202
- "video_path": facts.path,
1203
- "keyframe_index": index,
1204
- },
1205
- )
1206
- written.append({
1207
- "node_id": result["node_id"],
1208
- "index": index,
1209
- "path": str(frame),
1210
- "ocr_status": image_facts.ocr_status,
1211
- "vision_embedding": image_facts.embedding_status,
1212
- })
1213
- return written
1214
-
1215
-
1216
- __all__ = [
1217
- "AUDIO_EXTENSIONS",
1218
- "DEFAULT_KEYFRAMES",
1219
- "FFMPEG_BINARY",
1220
- "IMAGE_CHUNK_CHARS",
1221
- "IMAGE_EXTENSIONS",
1222
- "MAX_INDEX_TEXT_CHARS",
1223
- "MAX_SUBTITLE_CHARS",
1224
- "MAX_THUMBNAIL_CHARS",
1225
- "MODALITY_AUDIO",
1226
- "MODALITY_IMAGE",
1227
- "MODALITY_TEXT",
1228
- "MODALITY_VIDEO",
1229
- "SUBTITLE_EXTENSIONS",
1230
- "SUMMARY_CHARS",
1231
- "THUMBNAIL_EDGE",
1232
- "VIDEO_EXTENSIONS",
1233
- "VIDEO_FRAME_RELATION",
1234
- "VIDEO_FRAME_SOURCE_TYPE",
1235
- "VIDEO_NODE_TYPE",
1236
- "VIDEO_OUT_OF_SCOPE",
1237
- "VIDEO_UNAVAILABLE_DETAIL",
1238
- "AudioFacts",
1239
- "ImageFacts",
1240
- "MultimodalPorts",
1241
- "VideoFacts",
1242
- "audio_quality_score",
1243
- "detect_modality",
1244
- "extract_image_facts",
1245
- "extract_keyframes",
1246
- "ffmpeg_available",
1247
- "find_subtitle",
1248
- "image_node_id",
1249
- "image_quality_score",
1250
- "parse_subtitles",
1251
- "read_video_facts",
1252
- "transcribe_audio",
1253
- "video_frame_dir",
1254
- "video_node_id",
1255
- "video_quality_score",
1256
- "write_image_memory",
1257
- "write_video_memory",
1258
- ]