ltcai 11.2.0 → 11.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -53
- package/docs/CHANGELOG.md +61 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/MULTI_AGENT_RUNTIME.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +6 -2
- package/docs/PERMISSION_MODE.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +2 -2
- package/docs/v11.3.0_PLAN.md +202 -0
- package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +287 -0
- package/lattice_brain/graph/_kg_common/extraction.py +516 -0
- package/lattice_brain/graph/_kg_common/relations.py +161 -0
- package/lattice_brain/graph/_kg_common/text.py +479 -0
- package/lattice_brain/graph/discovery_index/__init__.py +35 -0
- package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
- package/lattice_brain/graph/discovery_index/extract.py +137 -0
- package/lattice_brain/graph/discovery_index/scan.py +411 -0
- package/lattice_brain/graph/discovery_index/upsert.py +495 -0
- package/lattice_brain/graph/projection/__init__.py +42 -0
- package/lattice_brain/graph/projection/curation.py +500 -0
- package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
- package/lattice_brain/graph/retrieval/__init__.py +54 -0
- package/lattice_brain/graph/retrieval/context.py +197 -0
- package/lattice_brain/graph/retrieval/graph_view.py +319 -0
- package/lattice_brain/graph/retrieval/hybrid.py +488 -0
- package/lattice_brain/graph/retrieval/maintenance.py +121 -0
- package/lattice_brain/graph/retrieval/signals.py +95 -0
- package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
- package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
- package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
- package/lattice_brain/graph/retrieval_vector/search.py +560 -0
- package/lattice_brain/graph/retrieval_vector/status.py +374 -0
- package/lattice_brain/ingestion/__init__.py +130 -0
- package/lattice_brain/ingestion/_contract.py +90 -0
- package/lattice_brain/ingestion/constants.py +127 -0
- package/lattice_brain/ingestion/folder_scan.py +57 -0
- package/lattice_brain/ingestion/folders.py +258 -0
- package/lattice_brain/ingestion/hashing.py +26 -0
- package/lattice_brain/ingestion/jobs_api.py +107 -0
- package/lattice_brain/ingestion/models.py +80 -0
- package/lattice_brain/ingestion/pipeline.py +486 -0
- package/lattice_brain/ingestion/quality.py +209 -0
- package/lattice_brain/ingestion/routing.py +295 -0
- package/lattice_brain/multimodal/__init__.py +164 -0
- package/lattice_brain/multimodal/audio.py +77 -0
- package/lattice_brain/multimodal/common.py +118 -0
- package/lattice_brain/multimodal/images.py +498 -0
- package/lattice_brain/multimodal/ports.py +169 -0
- package/lattice_brain/multimodal/video.py +410 -0
- package/lattice_brain/portability/__init__.py +90 -0
- package/lattice_brain/portability/_contract.py +42 -0
- package/lattice_brain/portability/backups.py +338 -0
- package/lattice_brain/portability/bundles.py +136 -0
- package/lattice_brain/portability/constants.py +93 -0
- package/lattice_brain/portability/fsops.py +138 -0
- package/lattice_brain/portability/service.py +41 -0
- package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
- package/lattice_brain/runtime/__init__.py +1 -1
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/chronicle.py +63 -0
- package/latticeai/core/agent/__init__.py +93 -0
- package/latticeai/core/agent/_contract.py +79 -0
- package/latticeai/core/agent/context.py +57 -0
- package/latticeai/core/agent/deps.py +125 -0
- package/latticeai/core/agent/execution.py +622 -0
- package/latticeai/core/agent/planning.py +145 -0
- package/latticeai/core/agent/recovery.py +157 -0
- package/latticeai/core/agent/runtime.py +210 -0
- package/latticeai/core/agent/verification.py +231 -0
- package/latticeai/core/embedding_providers/__init__.py +151 -0
- package/latticeai/core/embedding_providers/base.py +199 -0
- package/latticeai/core/embedding_providers/captions.py +162 -0
- package/latticeai/core/embedding_providers/profiles.py +126 -0
- package/latticeai/core/embedding_providers/text.py +350 -0
- package/latticeai/core/embedding_providers/vision.py +352 -0
- package/latticeai/core/file_generation/__init__.py +115 -0
- package/latticeai/core/file_generation/bundles.py +76 -0
- package/latticeai/core/file_generation/extraction.py +154 -0
- package/latticeai/core/file_generation/inference.py +235 -0
- package/latticeai/core/file_generation/orchestration.py +152 -0
- package/latticeai/core/file_generation/prompting.py +117 -0
- package/latticeai/core/file_generation/repair.py +114 -0
- package/latticeai/core/file_generation/sanitize.py +61 -0
- package/latticeai/core/file_generation/validation.py +201 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +9 -0
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/integrations/telegram_bot/__init__.py +123 -0
- package/latticeai/integrations/telegram_bot/__main__.py +17 -0
- package/latticeai/integrations/telegram_bot/config.py +86 -0
- package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
- package/latticeai/integrations/telegram_bot/flows.py +478 -0
- package/latticeai/integrations/telegram_bot/helpers.py +322 -0
- package/latticeai/integrations/telegram_bot/screens.py +394 -0
- package/latticeai/models/router/__init__.py +88 -0
- package/latticeai/models/router/_contract.py +66 -0
- package/latticeai/models/router/branding.py +56 -0
- package/latticeai/models/router/catalog.py +69 -0
- package/latticeai/models/router/documents.py +199 -0
- package/latticeai/models/router/errors.py +37 -0
- package/latticeai/models/router/generation.py +258 -0
- package/latticeai/models/router/loading.py +291 -0
- package/latticeai/models/router/local_models.py +85 -0
- package/latticeai/models/router/registry.py +147 -0
- package/latticeai/runtime/build_phases/__init__.py +82 -0
- package/latticeai/runtime/build_phases/features.py +407 -0
- package/latticeai/runtime/build_phases/foundation.py +555 -0
- package/latticeai/runtime/build_phases/web.py +492 -0
- package/latticeai/runtime/runtime_context.py +1 -0
- package/latticeai/services/architecture_readiness.py +48 -19
- package/latticeai/services/brain_intelligence/__init__.py +58 -0
- package/latticeai/services/brain_intelligence/_contract.py +71 -0
- package/latticeai/services/brain_intelligence/consistency.py +193 -0
- package/latticeai/services/brain_intelligence/constants.py +47 -0
- package/latticeai/services/brain_intelligence/digest.py +258 -0
- package/latticeai/services/brain_intelligence/health.py +331 -0
- package/latticeai/services/brain_intelligence/proposals.py +264 -0
- package/latticeai/services/brain_intelligence/sampling.py +84 -0
- package/latticeai/services/brain_intelligence/service.py +48 -0
- package/latticeai/services/chronicle.py +557 -0
- package/latticeai/services/memory_service/__init__.py +52 -0
- package/latticeai/services/memory_service/_contract.py +100 -0
- package/latticeai/services/memory_service/brief.py +431 -0
- package/latticeai/services/memory_service/constants.py +57 -0
- package/latticeai/services/memory_service/maintenance.py +138 -0
- package/latticeai/services/memory_service/manager.py +186 -0
- package/latticeai/services/memory_service/proof.py +136 -0
- package/latticeai/services/memory_service/recall.py +225 -0
- package/latticeai/services/memory_service/service.py +48 -0
- package/latticeai/services/memory_service/stores.py +110 -0
- package/latticeai/services/model_runtime/__init__.py +322 -0
- package/latticeai/services/model_runtime/cloud.py +87 -0
- package/latticeai/services/model_runtime/download.py +282 -0
- package/latticeai/services/model_runtime/engines.py +341 -0
- package/latticeai/services/model_runtime/loading.py +178 -0
- package/latticeai/services/model_runtime/service.py +129 -0
- package/latticeai/services/model_runtime/state.py +131 -0
- package/latticeai/services/model_runtime/status.py +255 -0
- package/latticeai/services/product_readiness.py +15 -7
- package/latticeai/setup/wizard/__init__.py +126 -0
- package/latticeai/setup/wizard/catalog.py +172 -0
- package/latticeai/setup/wizard/detect.py +323 -0
- package/latticeai/setup/wizard/install.py +348 -0
- package/latticeai/setup/wizard/paths.py +168 -0
- package/latticeai/setup/wizard/plans.py +74 -0
- package/latticeai/setup/wizard/recommend.py +320 -0
- package/package.json +6 -2
- package/scripts/bump_version.py +14 -0
- package/scripts/capture_release_evidence.mjs +33 -21
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_i18n_namespace_coverage.mjs +41 -4
- package/scripts/check_max_file_lines.mjs +102 -0
- package/scripts/check_release_evidence_bound.mjs +30 -15
- package/scripts/check_screenshot_pixel_delta.py +34 -4
- package/scripts/check_server_i18n.mjs +1 -0
- package/scripts/generate_rust_parity_fixtures.py +562 -0
- package/scripts/lib/mock_server_fingerprint.mjs +94 -0
- package/scripts/release_screen_claims.json +31 -2
- package/src-tauri/Cargo.lock +361 -3
- package/src-tauri/Cargo.toml +6 -1
- package/src-tauri/src/backend.rs +349 -0
- package/src-tauri/src/folder.rs +33 -0
- package/src-tauri/src/main.rs +97 -399
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +41 -37
- package/static/app/assets/Act-yYpYnn0v.js +1 -0
- package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
- package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
- package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
- package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
- package/static/app/assets/Capture-CFIRsFNE.js +1 -0
- package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
- package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
- package/static/app/assets/Library-DwO3yZST.js +1 -0
- package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
- package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
- package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
- package/static/app/assets/System-DW8F-2xL.js +1 -0
- package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
- package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
- package/static/app/assets/brain-Ci1CkWjM.js +1 -0
- package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
- package/static/app/assets/circle-check-DfInj-qD.js +1 -0
- package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
- package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
- package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
- package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
- package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
- package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
- package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
- package/static/app/assets/index-_u5iUHDr.js +10 -0
- package/static/app/assets/input-B0lPdRQZ.js +1 -0
- package/static/app/assets/link-2-CoFbooHS.js +1 -0
- package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
- package/static/app/assets/primitives-DEbN-d6p.js +1 -0
- package/static/app/assets/search-BybIWPNd.js +1 -0
- package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
- package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
- package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
- package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
- package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
- package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
- package/static/app/assets/utils-BlZr7Pd4.js +4 -0
- package/static/app/assets/workspace-jJY4RuAV.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/graph/_kg_common.py +0 -1331
- package/lattice_brain/graph/discovery_index.py +0 -1141
- package/lattice_brain/graph/retrieval.py +0 -1120
- package/lattice_brain/graph/retrieval_vector.py +0 -1293
- package/lattice_brain/ingestion.py +0 -1525
- package/lattice_brain/multimodal.py +0 -1258
- package/latticeai/core/agent.py +0 -1465
- package/latticeai/core/embedding_providers.py +0 -1196
- package/latticeai/core/file_generation.py +0 -1047
- package/latticeai/integrations/telegram_bot.py +0 -1390
- package/latticeai/models/router.py +0 -1007
- package/latticeai/runtime/build_phases.py +0 -1450
- package/latticeai/services/brain_intelligence.py +0 -1083
- package/latticeai/services/memory_service.py +0 -1177
- package/latticeai/services/model_runtime.py +0 -1281
- package/latticeai/setup/wizard.py +0 -1310
- package/static/app/assets/Act-AWf0SAKp.js +0 -1
- package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
- package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
- package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
- package/static/app/assets/Capture-CqOSzyPr.js +0 -1
- package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
- package/static/app/assets/Library-CX-bbhmK.js +0 -1
- package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
- package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
- package/static/app/assets/System-Bu2t5hn1.js +0 -1
- package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
- package/static/app/assets/brain-DJMoqrwx.js +0 -1
- package/static/app/assets/index-BpYkzcVm.js +0 -10
- package/static/app/assets/input-DSlJJxRs.js +0 -1
- package/static/app/assets/primitives-BCx6TvfG.js +0 -1
- package/static/app/assets/search-Cgy8cCFJ.js +0 -1
- package/static/app/assets/utils-zqPZJxdx.js +0 -4
- package/static/app/assets/workspace-DXTihhfU.js +0 -1
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
"""The injected capability ports, and the ffmpeg fallback behind one of them.
|
|
2
|
+
|
|
3
|
+
:class:`MultimodalPorts` is how every heavy model reaches Brain Core: as a
|
|
4
|
+
plain callable the app layer supplies, never as an import. The keyframe port is
|
|
5
|
+
the one with a built-in fallback — ffmpeg on ``PATH`` — so the probe
|
|
6
|
+
(:func:`_which_ffmpeg`), the runner (:func:`_run_ffmpeg`) and the extraction
|
|
7
|
+
(:func:`extract_keyframes`) live here beside it rather than in ``video.py``.
|
|
8
|
+
Splitting them would put ``ffmpeg_available``'s view of the probe and
|
|
9
|
+
``extract_keyframes``'s view of it in two different module namespaces, which is
|
|
10
|
+
two things to stub instead of one.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import shutil
|
|
16
|
+
import subprocess # noqa: S404 — one fixed binary, argv list, never a shell
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from typing import Any, Callable, Dict, List, Optional
|
|
20
|
+
|
|
21
|
+
from .common import MODALITY_IMAGE, VIDEO_UNAVAILABLE_DETAIL
|
|
22
|
+
|
|
23
|
+
#: The decoder. Looked up by name on PATH and never bundled — a product that
|
|
24
|
+
#: cannot decode a ``.mov`` says so instead of shipping a codec pack.
|
|
25
|
+
FFMPEG_BINARY = "ffmpeg"
|
|
26
|
+
#: Keyframes kept per video. Four is a memory of a video, not a copy of it.
|
|
27
|
+
DEFAULT_KEYFRAMES = 4
|
|
28
|
+
#: Frames ffmpeg's ``thumbnail`` filter considers before picking one. Larger
|
|
29
|
+
#: windows mean more representative frames and a slower pass.
|
|
30
|
+
KEYFRAME_WINDOW = 300
|
|
31
|
+
KEYFRAME_TIMEOUT_SECONDS = 120
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
# ── injected capability ports ────────────────────────────────────────────────
|
|
35
|
+
@dataclass
|
|
36
|
+
class MultimodalPorts:
|
|
37
|
+
"""The optional model-backed capabilities Brain Core cannot ship itself.
|
|
38
|
+
|
|
39
|
+
Every field is a plain callable so the app layer can build it from
|
|
40
|
+
``latticeai.core.embedding_providers`` without Brain Core ever importing
|
|
41
|
+
that package. ``None`` everywhere is the honest default: OCR still runs
|
|
42
|
+
(``pytesseract`` is a local binary, not a model download), and everything
|
|
43
|
+
else reports itself as unavailable.
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
#: ``(image_path) -> caption or None`` — a loaded VLM, or nothing.
|
|
47
|
+
captioner: Optional[Callable[[str], Optional[str]]] = None
|
|
48
|
+
#: ``(image_path) -> vector`` in the image space (raises when it cannot).
|
|
49
|
+
vision_embedder: Optional[Callable[[str], List[float]]] = None
|
|
50
|
+
#: ``(audio_path) -> transcript`` (raises/returns empty when it cannot).
|
|
51
|
+
transcriber: Optional[Callable[[str], str]] = None
|
|
52
|
+
#: ``(video_path, dest_dir, count) -> [frame paths]`` (v11.2.0). ``None``
|
|
53
|
+
#: falls back to ffmpeg on PATH; absent ffmpeg is reported, never faked.
|
|
54
|
+
keyframe_extractor: Optional[Callable[..., Any]] = None
|
|
55
|
+
#: ``(query_text) -> vector`` in the *image* space (v11.2.0). Only a
|
|
56
|
+
#: genuinely shared-space vision model can supply one, which is why it is
|
|
57
|
+
#: its own port instead of being assumed from ``vision_embedder``.
|
|
58
|
+
text_to_image_embedder: Optional[Callable[[str], List[float]]] = None
|
|
59
|
+
#: Identity of the vision model, recorded next to every image vector.
|
|
60
|
+
vision_model_id: str = ""
|
|
61
|
+
#: ``image`` (own index + late fusion) or ``shared`` (same space as text).
|
|
62
|
+
vision_space: str = MODALITY_IMAGE
|
|
63
|
+
|
|
64
|
+
def describe(self) -> Dict[str, Any]:
|
|
65
|
+
"""What this install can honestly do with a picture or a recording."""
|
|
66
|
+
return {
|
|
67
|
+
"caption": self.captioner is not None,
|
|
68
|
+
"vision_embedding": self.vision_embedder is not None,
|
|
69
|
+
"transcription": self.transcriber is not None,
|
|
70
|
+
"keyframes": self.keyframe_extractor is not None or ffmpeg_available(),
|
|
71
|
+
"text_to_image_query": self.text_to_image_embedder is not None,
|
|
72
|
+
"vision_model_id": self.vision_model_id,
|
|
73
|
+
"vision_space": self.vision_space,
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _which_ffmpeg() -> Optional[str]:
|
|
78
|
+
"""Absolute path to ffmpeg, or ``None``. The one probe, seamed for tests."""
|
|
79
|
+
return shutil.which(FFMPEG_BINARY)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def ffmpeg_available() -> bool:
|
|
83
|
+
"""Whether this machine can decode a video at all (honest, never assumed)."""
|
|
84
|
+
return _which_ffmpeg() is not None
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _run_ffmpeg(binary: str, args: List[str]) -> int:
|
|
88
|
+
"""Run one fixed binary with an argv list — no shell, no user strings."""
|
|
89
|
+
completed = subprocess.run( # noqa: S603 — argv list, fixed binary, no shell
|
|
90
|
+
[binary, *args],
|
|
91
|
+
stdout=subprocess.DEVNULL,
|
|
92
|
+
stderr=subprocess.DEVNULL,
|
|
93
|
+
timeout=KEYFRAME_TIMEOUT_SECONDS,
|
|
94
|
+
check=False,
|
|
95
|
+
)
|
|
96
|
+
return int(completed.returncode)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def extract_keyframes(
|
|
100
|
+
path: Any,
|
|
101
|
+
dest_dir: Any,
|
|
102
|
+
*,
|
|
103
|
+
count: int = DEFAULT_KEYFRAMES,
|
|
104
|
+
ports: Optional[MultimodalPorts] = None,
|
|
105
|
+
) -> Dict[str, Any]:
|
|
106
|
+
"""Pull up to ``count`` representative stills out of a video.
|
|
107
|
+
|
|
108
|
+
An injected ``ports.keyframe_extractor`` wins outright — that is the seam
|
|
109
|
+
an install with its own decoder (or a test) uses. Otherwise ffmpeg's
|
|
110
|
+
``thumbnail`` filter picks the most representative frame from each window
|
|
111
|
+
of :data:`KEYFRAME_WINDOW` frames, which is one pass and no probing.
|
|
112
|
+
|
|
113
|
+
Never raises. A missing decoder, a non-zero exit, and a video too short to
|
|
114
|
+
yield a single frame are three different states and each says so.
|
|
115
|
+
"""
|
|
116
|
+
ports = ports or MultimodalPorts()
|
|
117
|
+
video = Path(str(path))
|
|
118
|
+
dest = Path(str(dest_dir))
|
|
119
|
+
wanted = max(1, int(count))
|
|
120
|
+
if ports.keyframe_extractor is not None:
|
|
121
|
+
return _injected_keyframes(ports.keyframe_extractor, video, dest, wanted)
|
|
122
|
+
binary = _which_ffmpeg()
|
|
123
|
+
if binary is None:
|
|
124
|
+
return {"status": "unavailable", "frames": [], "detail": VIDEO_UNAVAILABLE_DETAIL}
|
|
125
|
+
dest.mkdir(parents=True, exist_ok=True)
|
|
126
|
+
args = [
|
|
127
|
+
"-nostdin", "-loglevel", "error", "-y",
|
|
128
|
+
"-i", str(video),
|
|
129
|
+
"-vf", f"thumbnail={KEYFRAME_WINDOW}",
|
|
130
|
+
"-frames:v", str(wanted),
|
|
131
|
+
"-vsync", "vfr",
|
|
132
|
+
str(dest / "keyframe-%03d.jpg"),
|
|
133
|
+
]
|
|
134
|
+
try:
|
|
135
|
+
code = _run_ffmpeg(binary, args)
|
|
136
|
+
except Exception as exc: # noqa: BLE001 — a broken decoder is a state
|
|
137
|
+
return {"status": "failed", "frames": [], "detail": f"ffmpeg failed: {exc}"}
|
|
138
|
+
frames = sorted(str(p) for p in dest.glob("keyframe-*.jpg"))
|
|
139
|
+
if code != 0 and not frames:
|
|
140
|
+
return {
|
|
141
|
+
"status": "failed",
|
|
142
|
+
"frames": [],
|
|
143
|
+
"detail": f"ffmpeg exited with status {code}",
|
|
144
|
+
}
|
|
145
|
+
if not frames:
|
|
146
|
+
return {
|
|
147
|
+
"status": "empty",
|
|
148
|
+
"frames": [],
|
|
149
|
+
"detail": "ffmpeg produced no frames from this video",
|
|
150
|
+
}
|
|
151
|
+
return {"status": "ok", "frames": frames[:wanted], "detail": ""}
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _injected_keyframes(
|
|
155
|
+
extractor: Callable[..., Any], video: Path, dest: Path, wanted: int
|
|
156
|
+
) -> Dict[str, Any]:
|
|
157
|
+
"""Run a caller-supplied extractor; a failure is reported, never raised."""
|
|
158
|
+
try:
|
|
159
|
+
produced = extractor(str(video), str(dest), wanted)
|
|
160
|
+
except Exception as exc: # noqa: BLE001 — an injected port is not trusted more
|
|
161
|
+
return {"status": "failed", "frames": [], "detail": f"keyframe port failed: {exc}"}
|
|
162
|
+
frames = [str(item) for item in (produced or [])][:wanted]
|
|
163
|
+
if not frames:
|
|
164
|
+
return {
|
|
165
|
+
"status": "empty",
|
|
166
|
+
"frames": [],
|
|
167
|
+
"detail": "the keyframe port produced no frames",
|
|
168
|
+
}
|
|
169
|
+
return {"status": "ok", "frames": frames, "detail": ""}
|
|
@@ -0,0 +1,410 @@
|
|
|
1
|
+
"""A video as keyframes through the image door, plus its subtitles as text.
|
|
2
|
+
|
|
3
|
+
Nothing here is a new retrieval path. The stills go through
|
|
4
|
+
:func:`~lattice_brain.multimodal.images.write_image_memory` exactly as a
|
|
5
|
+
photograph does, the subtitles become ordinary chunks, and what the ``Video``
|
|
6
|
+
node adds is the thing they belong to. Keyframe *extraction* lives in
|
|
7
|
+
``ports.py`` beside the decoder it falls back to; this module only consumes it.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import re
|
|
13
|
+
from dataclasses import dataclass, field
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Any, Dict, List, Optional
|
|
16
|
+
|
|
17
|
+
from ..quiet import quiet
|
|
18
|
+
from ..utils import utc_now_iso
|
|
19
|
+
from .common import (
|
|
20
|
+
MAX_INDEX_TEXT_CHARS,
|
|
21
|
+
MODALITY_IMAGE,
|
|
22
|
+
MODALITY_VIDEO,
|
|
23
|
+
SUBTITLE_EXTENSIONS,
|
|
24
|
+
SUMMARY_CHARS,
|
|
25
|
+
_sha256_file,
|
|
26
|
+
_sha256_text,
|
|
27
|
+
_split_index_text,
|
|
28
|
+
)
|
|
29
|
+
from .images import (
|
|
30
|
+
_attach_concepts,
|
|
31
|
+
_attach_source,
|
|
32
|
+
extract_image_facts,
|
|
33
|
+
write_image_memory,
|
|
34
|
+
)
|
|
35
|
+
from .ports import DEFAULT_KEYFRAMES, MultimodalPorts, extract_keyframes
|
|
36
|
+
|
|
37
|
+
#: Graph node type for a video. ``NodeType.VIDEO`` normalizes this on the KG v2
|
|
38
|
+
#: write side; the legacy tables keep the label verbatim, like ``Audio``.
|
|
39
|
+
VIDEO_NODE_TYPE = "Video"
|
|
40
|
+
#: Source type stamped on the extracted stills so a keyframe is never mistaken
|
|
41
|
+
#: for a photograph the user took.
|
|
42
|
+
VIDEO_FRAME_SOURCE_TYPE = "video_keyframe"
|
|
43
|
+
VIDEO_FRAME_RELATION = "CONTAINS_IMAGE"
|
|
44
|
+
#: Longest subtitle body kept, matching the OCR/caption ceiling.
|
|
45
|
+
MAX_SUBTITLE_CHARS = MAX_INDEX_TEXT_CHARS
|
|
46
|
+
|
|
47
|
+
_SRT_INDEX_RE = re.compile(r"^\d+$")
|
|
48
|
+
_TIMECODE_RE = re.compile(r"\d{1,2}:\d{2}:\d{2}[.,]\d{1,3}\s*-->")
|
|
49
|
+
_CUE_TAG_RE = re.compile(r"</?[a-zA-Z][^>]*>")
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def find_subtitle(path: Any) -> Optional[Path]:
|
|
53
|
+
"""The ``.srt``/``.vtt`` sitting next to a video under the same basename."""
|
|
54
|
+
video = Path(str(path))
|
|
55
|
+
for suffix in SUBTITLE_EXTENSIONS:
|
|
56
|
+
candidate = video.with_suffix(f".{suffix}")
|
|
57
|
+
if candidate.is_file():
|
|
58
|
+
return candidate
|
|
59
|
+
return None
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def parse_subtitles(text: str) -> str:
|
|
63
|
+
"""Strip SRT/WebVTT scaffolding down to the words that were said.
|
|
64
|
+
|
|
65
|
+
Cue numbers, timecodes, ``WEBVTT`` headers, ``NOTE`` blocks and inline
|
|
66
|
+
``<c>``/``<i>`` tags carry no recall value and would otherwise dominate a
|
|
67
|
+
chunk. Consecutive duplicate lines (the usual rolling-caption artefact) are
|
|
68
|
+
collapsed. Deliberately a small parser, not a dependency: the format is two
|
|
69
|
+
rules deep and a library here would sit in the ingest path forever.
|
|
70
|
+
"""
|
|
71
|
+
lines: List[str] = []
|
|
72
|
+
for raw in str(text or "").splitlines():
|
|
73
|
+
line = raw.strip().lstrip("")
|
|
74
|
+
if not line or line.upper().startswith("WEBVTT") or line.startswith("NOTE"):
|
|
75
|
+
continue
|
|
76
|
+
if _SRT_INDEX_RE.match(line) or _TIMECODE_RE.search(line):
|
|
77
|
+
continue
|
|
78
|
+
cleaned = _CUE_TAG_RE.sub("", line).strip()
|
|
79
|
+
if not cleaned:
|
|
80
|
+
continue
|
|
81
|
+
if lines and lines[-1] == cleaned:
|
|
82
|
+
continue
|
|
83
|
+
lines.append(cleaned)
|
|
84
|
+
return "\n".join(lines)[:MAX_SUBTITLE_CHARS]
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
@dataclass
|
|
88
|
+
class VideoFacts:
|
|
89
|
+
"""What could actually be observed in one video, and how each attempt went."""
|
|
90
|
+
|
|
91
|
+
path: str
|
|
92
|
+
#: ``ok`` | ``unavailable`` | ``failed`` | ``empty``
|
|
93
|
+
keyframe_status: str = "unavailable"
|
|
94
|
+
keyframes: List[str] = field(default_factory=list)
|
|
95
|
+
keyframe_detail: str = ""
|
|
96
|
+
#: ``ok`` | ``absent`` | ``unreadable`` | ``empty``
|
|
97
|
+
subtitle_status: str = "absent"
|
|
98
|
+
subtitle_path: Optional[str] = None
|
|
99
|
+
subtitle_text: str = ""
|
|
100
|
+
|
|
101
|
+
@property
|
|
102
|
+
def searchable(self) -> bool:
|
|
103
|
+
"""Whether anything in this video can be matched by a typed question."""
|
|
104
|
+
return bool(self.subtitle_text)
|
|
105
|
+
|
|
106
|
+
def as_metadata(self) -> Dict[str, Any]:
|
|
107
|
+
payload: Dict[str, Any] = {
|
|
108
|
+
"modality": MODALITY_VIDEO,
|
|
109
|
+
"video_path": self.path,
|
|
110
|
+
"keyframes": len(self.keyframes),
|
|
111
|
+
"keyframe_status": self.keyframe_status,
|
|
112
|
+
"subtitles": self.subtitle_status,
|
|
113
|
+
"searchable": self.searchable,
|
|
114
|
+
}
|
|
115
|
+
if self.keyframe_detail:
|
|
116
|
+
payload["keyframe_detail"] = self.keyframe_detail
|
|
117
|
+
if self.subtitle_path:
|
|
118
|
+
payload["subtitle_path"] = self.subtitle_path
|
|
119
|
+
return payload
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def read_video_facts(
|
|
123
|
+
path: Any,
|
|
124
|
+
dest_dir: Any,
|
|
125
|
+
*,
|
|
126
|
+
count: int = DEFAULT_KEYFRAMES,
|
|
127
|
+
ports: Optional[MultimodalPorts] = None,
|
|
128
|
+
subtitle_text: Optional[str] = None,
|
|
129
|
+
) -> VideoFacts:
|
|
130
|
+
"""Observe one video: its keyframes and its companion subtitles."""
|
|
131
|
+
facts = VideoFacts(path=str(path))
|
|
132
|
+
outcome = extract_keyframes(path, dest_dir, count=count, ports=ports)
|
|
133
|
+
facts.keyframe_status = str(outcome["status"])
|
|
134
|
+
facts.keyframes = list(outcome["frames"])
|
|
135
|
+
facts.keyframe_detail = str(outcome["detail"])
|
|
136
|
+
supplied = str(subtitle_text or "").strip()
|
|
137
|
+
if supplied:
|
|
138
|
+
facts.subtitle_status = "ok"
|
|
139
|
+
facts.subtitle_text = parse_subtitles(supplied)
|
|
140
|
+
return facts
|
|
141
|
+
companion = find_subtitle(path)
|
|
142
|
+
if companion is None:
|
|
143
|
+
return facts
|
|
144
|
+
facts.subtitle_path = str(companion)
|
|
145
|
+
try:
|
|
146
|
+
raw = companion.read_text(encoding="utf-8", errors="ignore")
|
|
147
|
+
except OSError as exc:
|
|
148
|
+
facts.subtitle_status = "unreadable"
|
|
149
|
+
facts.keyframe_detail = (facts.keyframe_detail or "").strip()
|
|
150
|
+
facts.subtitle_text = ""
|
|
151
|
+
quiet()
|
|
152
|
+
facts.subtitle_path = f"{companion} ({exc.strerror or 'unreadable'})"
|
|
153
|
+
return facts
|
|
154
|
+
parsed = parse_subtitles(raw)
|
|
155
|
+
facts.subtitle_status = "ok" if parsed else "empty"
|
|
156
|
+
facts.subtitle_text = parsed
|
|
157
|
+
return facts
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def video_quality_score(facts: VideoFacts) -> Dict[str, Any]:
|
|
161
|
+
"""How much of this video the Brain can actually retrieve later.
|
|
162
|
+
|
|
163
|
+
Same principle as :func:`image_quality_score`: not a judgement about the
|
|
164
|
+
footage. Subtitles are worth the most (they are the words), keyframes are
|
|
165
|
+
worth something (they can be OCR'd and seen), and a video with neither is
|
|
166
|
+
a filename.
|
|
167
|
+
"""
|
|
168
|
+
reasons: List[str] = []
|
|
169
|
+
score = 0.1 # we know it is a video and where it lives
|
|
170
|
+
if facts.subtitle_text:
|
|
171
|
+
score += 0.55 * min(1.0, len(facts.subtitle_text) / 400.0)
|
|
172
|
+
reasons.append("subtitles")
|
|
173
|
+
else:
|
|
174
|
+
reasons.append(f"no_subtitles_{facts.subtitle_status}")
|
|
175
|
+
if facts.keyframes:
|
|
176
|
+
score += min(0.35, 0.1 * len(facts.keyframes))
|
|
177
|
+
reasons.append("keyframes")
|
|
178
|
+
else:
|
|
179
|
+
reasons.append(f"no_keyframes_{facts.keyframe_status}")
|
|
180
|
+
return {"score": round(max(0.0, min(1.0, score)), 4), "reasons": reasons}
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def video_node_id(content_hash: str, workspace_id: Optional[str] = None) -> str:
|
|
184
|
+
"""Workspace-scoped, content-addressed id — re-ingesting is idempotent."""
|
|
185
|
+
scoped = f"{workspace_id or 'legacy-global'}|{content_hash}"
|
|
186
|
+
return f"video:{_sha256_text(scoped)[:24]}"
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def video_frame_dir(blob_dir: Any, content_hash: str) -> Path:
|
|
190
|
+
"""Where this video's stills live: content-addressed, stable across runs.
|
|
191
|
+
|
|
192
|
+
Deliberately under the Brain's own blob directory rather than a temp dir.
|
|
193
|
+
A frame referenced by an ``Image`` node has to still be there the next time
|
|
194
|
+
someone opens that memory, and a backup that copies the blobs copies these.
|
|
195
|
+
"""
|
|
196
|
+
return Path(str(blob_dir)) / "video_frames" / str(content_hash)[:32]
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def write_video_memory(
|
|
200
|
+
store: Any,
|
|
201
|
+
*,
|
|
202
|
+
path: Path,
|
|
203
|
+
facts: VideoFacts,
|
|
204
|
+
title: str,
|
|
205
|
+
source_type: str = MODALITY_VIDEO,
|
|
206
|
+
source_uri: Optional[str] = None,
|
|
207
|
+
owner: Optional[str] = None,
|
|
208
|
+
workspace_id: Optional[str] = None,
|
|
209
|
+
conversation_id: Optional[str] = None,
|
|
210
|
+
captured_at: Optional[str] = None,
|
|
211
|
+
modified_at: Optional[str] = None,
|
|
212
|
+
permissions: Optional[Dict[str, Any]] = None,
|
|
213
|
+
extra_metadata: Optional[Dict[str, Any]] = None,
|
|
214
|
+
ports: Optional[MultimodalPorts] = None,
|
|
215
|
+
) -> Dict[str, Any]:
|
|
216
|
+
"""Write one ``Video`` node, its keyframes, and its subtitles.
|
|
217
|
+
|
|
218
|
+
Each keyframe goes through the **existing image path** — the same
|
|
219
|
+
:func:`extract_image_facts` and :func:`write_image_memory` a photograph
|
|
220
|
+
uses — so a still from a video is OCR'd, captioned, vectorized and made
|
|
221
|
+
searchable by exactly the machinery that already does that, and joined to
|
|
222
|
+
its video by ``CONTAINS_IMAGE``. Subtitles ride the ordinary text index as
|
|
223
|
+
chunks. Nothing about video gets its own retrieval path.
|
|
224
|
+
|
|
225
|
+
The frames are written before the video node so their (short-lived) write
|
|
226
|
+
transactions never nest inside the video's.
|
|
227
|
+
"""
|
|
228
|
+
ports = ports or MultimodalPorts()
|
|
229
|
+
captured_at = captured_at or utc_now_iso()
|
|
230
|
+
content_hash = _sha256_file(path)
|
|
231
|
+
node_id = video_node_id(content_hash, workspace_id)
|
|
232
|
+
frames = _write_keyframes(
|
|
233
|
+
store,
|
|
234
|
+
facts=facts,
|
|
235
|
+
title=title,
|
|
236
|
+
owner=owner,
|
|
237
|
+
workspace_id=workspace_id,
|
|
238
|
+
conversation_id=conversation_id,
|
|
239
|
+
captured_at=captured_at,
|
|
240
|
+
permissions=permissions,
|
|
241
|
+
ports=ports,
|
|
242
|
+
)
|
|
243
|
+
body = facts.subtitle_text or (
|
|
244
|
+
f"[{MODALITY_VIDEO}] {title}\n"
|
|
245
|
+
"이 영상에는 자막이 없어 말의 내용은 검색되지 않습니다 — "
|
|
246
|
+
"대신 대표 장면 이미지로 찾을 수 있습니다."
|
|
247
|
+
)
|
|
248
|
+
metadata: Dict[str, Any] = {
|
|
249
|
+
"filename": path.name,
|
|
250
|
+
"file_path": str(path),
|
|
251
|
+
"ext": path.suffix.lower(),
|
|
252
|
+
"bytes": path.stat().st_size,
|
|
253
|
+
"sha256": content_hash,
|
|
254
|
+
"content_hash": content_hash,
|
|
255
|
+
"source_type": source_type,
|
|
256
|
+
"source_uri": source_uri or str(path),
|
|
257
|
+
"captured_at": captured_at,
|
|
258
|
+
"modified_at": modified_at,
|
|
259
|
+
"owner": owner,
|
|
260
|
+
"workspace_id": workspace_id,
|
|
261
|
+
"permissions": permissions or {},
|
|
262
|
+
"conversation_id": conversation_id,
|
|
263
|
+
"keyframe_nodes": [frame["node_id"] for frame in frames],
|
|
264
|
+
**facts.as_metadata(),
|
|
265
|
+
**(extra_metadata or {}),
|
|
266
|
+
}
|
|
267
|
+
# Honest card: a video nobody captioned says so, rather than rendering as a
|
|
268
|
+
# blank summary that looks like a failed read.
|
|
269
|
+
summary = body[:SUMMARY_CHARS]
|
|
270
|
+
chunk_ids: List[str] = []
|
|
271
|
+
with store._connect() as conn:
|
|
272
|
+
duplicate = (
|
|
273
|
+
conn.execute("SELECT 1 FROM nodes WHERE id=? LIMIT 1", (node_id,)).fetchone()
|
|
274
|
+
is not None
|
|
275
|
+
)
|
|
276
|
+
store._upsert_node(
|
|
277
|
+
conn,
|
|
278
|
+
node_id,
|
|
279
|
+
VIDEO_NODE_TYPE,
|
|
280
|
+
title or path.name,
|
|
281
|
+
summary=summary,
|
|
282
|
+
metadata=metadata,
|
|
283
|
+
raw=metadata,
|
|
284
|
+
owner=owner,
|
|
285
|
+
workspace_id=workspace_id,
|
|
286
|
+
)
|
|
287
|
+
for frame in frames:
|
|
288
|
+
store._upsert_edge(
|
|
289
|
+
conn,
|
|
290
|
+
node_id,
|
|
291
|
+
frame["node_id"],
|
|
292
|
+
VIDEO_FRAME_RELATION,
|
|
293
|
+
weight=0.8,
|
|
294
|
+
metadata={
|
|
295
|
+
"source": VIDEO_FRAME_SOURCE_TYPE,
|
|
296
|
+
"index": frame["index"],
|
|
297
|
+
"workspace_id": workspace_id,
|
|
298
|
+
},
|
|
299
|
+
)
|
|
300
|
+
for index, piece in enumerate(_split_index_text(facts.subtitle_text)):
|
|
301
|
+
chunk_id = f"chunk:{_sha256_text(f'{node_id}:{index}:{piece}')[:24]}"
|
|
302
|
+
chunk_ids.append(chunk_id)
|
|
303
|
+
chunk_meta = {
|
|
304
|
+
"index": index,
|
|
305
|
+
"source_node": node_id,
|
|
306
|
+
"workspace_id": workspace_id,
|
|
307
|
+
"modality": MODALITY_VIDEO,
|
|
308
|
+
}
|
|
309
|
+
store._upsert_node(
|
|
310
|
+
conn,
|
|
311
|
+
chunk_id,
|
|
312
|
+
"Chunk",
|
|
313
|
+
f"{path.name} chunk {index + 1}",
|
|
314
|
+
summary=piece[:SUMMARY_CHARS],
|
|
315
|
+
metadata=chunk_meta,
|
|
316
|
+
owner=owner,
|
|
317
|
+
workspace_id=workspace_id,
|
|
318
|
+
)
|
|
319
|
+
store._upsert_chunk(
|
|
320
|
+
conn,
|
|
321
|
+
chunk_id=chunk_id,
|
|
322
|
+
source_node=node_id,
|
|
323
|
+
text=piece,
|
|
324
|
+
metadata=chunk_meta,
|
|
325
|
+
)
|
|
326
|
+
store._upsert_edge(conn, node_id, chunk_id, "포함함")
|
|
327
|
+
concept_ids = _attach_concepts(
|
|
328
|
+
store,
|
|
329
|
+
conn,
|
|
330
|
+
node_id=node_id,
|
|
331
|
+
text=facts.subtitle_text,
|
|
332
|
+
owner=owner,
|
|
333
|
+
workspace_id=workspace_id,
|
|
334
|
+
)
|
|
335
|
+
source_node_id = _attach_source(
|
|
336
|
+
store,
|
|
337
|
+
conn,
|
|
338
|
+
node_id=node_id,
|
|
339
|
+
source_type=source_type,
|
|
340
|
+
source_uri=source_uri or str(path),
|
|
341
|
+
title=title or path.name,
|
|
342
|
+
content_hash=content_hash,
|
|
343
|
+
captured_at=captured_at,
|
|
344
|
+
owner=owner,
|
|
345
|
+
workspace_id=workspace_id,
|
|
346
|
+
)
|
|
347
|
+
metadata["concepts"] = concept_ids
|
|
348
|
+
return {
|
|
349
|
+
"node_id": node_id,
|
|
350
|
+
"type": VIDEO_NODE_TYPE,
|
|
351
|
+
"title": title or path.name,
|
|
352
|
+
"sha256": content_hash,
|
|
353
|
+
"content_hash": content_hash,
|
|
354
|
+
"source_node_id": source_node_id,
|
|
355
|
+
"chunk_ids": chunk_ids,
|
|
356
|
+
"chunk_count": len(chunk_ids),
|
|
357
|
+
"duplicate": duplicate,
|
|
358
|
+
"captured_at": captured_at,
|
|
359
|
+
"keyframes": frames,
|
|
360
|
+
"metadata": metadata,
|
|
361
|
+
}
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
def _write_keyframes(
|
|
365
|
+
store: Any,
|
|
366
|
+
*,
|
|
367
|
+
facts: VideoFacts,
|
|
368
|
+
title: str,
|
|
369
|
+
owner: Optional[str],
|
|
370
|
+
workspace_id: Optional[str],
|
|
371
|
+
conversation_id: Optional[str],
|
|
372
|
+
captured_at: str,
|
|
373
|
+
permissions: Optional[Dict[str, Any]],
|
|
374
|
+
ports: MultimodalPorts,
|
|
375
|
+
) -> List[Dict[str, Any]]:
|
|
376
|
+
"""Every extracted still, through the ordinary image door."""
|
|
377
|
+
written: List[Dict[str, Any]] = []
|
|
378
|
+
for index, frame_path in enumerate(facts.keyframes):
|
|
379
|
+
frame = Path(frame_path)
|
|
380
|
+
image_facts = extract_image_facts(str(frame), ports=ports)
|
|
381
|
+
if not image_facts.readable:
|
|
382
|
+
# A frame ffmpeg wrote that Pillow cannot open is a state worth
|
|
383
|
+
# skipping, not worth failing the whole video over.
|
|
384
|
+
continue
|
|
385
|
+
result = write_image_memory(
|
|
386
|
+
store,
|
|
387
|
+
path=frame,
|
|
388
|
+
facts=image_facts,
|
|
389
|
+
title=f"{title} · 장면 {index + 1}",
|
|
390
|
+
source_type=VIDEO_FRAME_SOURCE_TYPE,
|
|
391
|
+
source_uri=str(frame),
|
|
392
|
+
owner=owner,
|
|
393
|
+
workspace_id=workspace_id,
|
|
394
|
+
conversation_id=conversation_id,
|
|
395
|
+
captured_at=captured_at,
|
|
396
|
+
permissions=permissions,
|
|
397
|
+
extra_metadata={
|
|
398
|
+
"modality": MODALITY_IMAGE,
|
|
399
|
+
"video_path": facts.path,
|
|
400
|
+
"keyframe_index": index,
|
|
401
|
+
},
|
|
402
|
+
)
|
|
403
|
+
written.append({
|
|
404
|
+
"node_id": result["node_id"],
|
|
405
|
+
"index": index,
|
|
406
|
+
"path": str(frame),
|
|
407
|
+
"ocr_status": image_facts.ocr_status,
|
|
408
|
+
"vision_embedding": image_facts.embedding_status,
|
|
409
|
+
})
|
|
410
|
+
return written
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
"""Knowledge Graph portability — local export / import / backup / restore.
|
|
2
|
+
|
|
3
|
+
The Knowledge Graph is the user's durable asset, so it must be portable without
|
|
4
|
+
any cloud service. Two complementary mechanisms, both fully local:
|
|
5
|
+
|
|
6
|
+
* **Logical export/import** (JSON): nodes/edges/chunks/sources/provenance with a
|
|
7
|
+
versioned header (schema + projection + embed-dim). Vectors are not in the
|
|
8
|
+
artifact; the importer re-embeds with its own embedder and reports the
|
|
9
|
+
resulting index state under ``result["index"]`` (``degraded: true`` means the
|
|
10
|
+
content landed but recall is lexical-only until a rebuild succeeds). That is
|
|
11
|
+
what makes it portable across machines.
|
|
12
|
+
* **Binary backup/restore** (ZIP): a faithful snapshot of the SQLite DB (incl.
|
|
13
|
+
vector embeddings) plus the blob directory, integrity-checked, for
|
|
14
|
+
same-machine recovery.
|
|
15
|
+
|
|
16
|
+
Split into cohesive submodules in v11.3.0 (no behaviour change): ``constants``
|
|
17
|
+
(formats + gates), ``fsops`` (atomic swaps), ``bundles`` (subgraph shaping),
|
|
18
|
+
``sharing`` and ``backups`` (the two halves of the service), ``service`` (the
|
|
19
|
+
composed class). This module re-exports every name the single file exposed, so
|
|
20
|
+
``lattice_brain.portability.X`` keeps working.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
# The single file had no ``__all__``, so its public surface was "every module
|
|
26
|
+
# global". The redundant-alias form reproduces exactly that surface and marks
|
|
27
|
+
# each name as a deliberate re-export rather than a leftover import.
|
|
28
|
+
#
|
|
29
|
+
# Stubbing note: rebinding one of these *here* changes only this module's name.
|
|
30
|
+
# The submodule that calls it holds its own reference, so a test standing in for
|
|
31
|
+
# ``_stamp`` patches ``lattice_brain.portability.fsops``, for ``_sha256_file``
|
|
32
|
+
# or ``SQLiteToPostgresMigrator`` ``…portability.backups``, and for
|
|
33
|
+
# ``load_recipient_identity`` ``…portability.sharing``.
|
|
34
|
+
from ..archive import KDF_ITERATIONS as KDF_ITERATIONS
|
|
35
|
+
from ..archive import BrainArchivePaths as BrainArchivePaths
|
|
36
|
+
from ..archive import EncryptedBrainArchive as EncryptedBrainArchive
|
|
37
|
+
from ..archive import _derive_key as _derive_key
|
|
38
|
+
from ..gates import FeatureGate as FeatureGate
|
|
39
|
+
from ..quiet import quiet as quiet
|
|
40
|
+
from ..sealed_box import SEALED_BOX_ALGORITHM as SEALED_BOX_ALGORITHM
|
|
41
|
+
from ..sealed_box import public_key_fingerprint as public_key_fingerprint
|
|
42
|
+
from ..sealed_box import seal as seal
|
|
43
|
+
from ..storage import DockerPostgresWizard as DockerPostgresWizard
|
|
44
|
+
from ..storage import PostgresEngine as PostgresEngine
|
|
45
|
+
from ..storage import SQLiteToPostgresMigrator as SQLiteToPostgresMigrator
|
|
46
|
+
from ..utils import utc_now_iso as utc_now_iso
|
|
47
|
+
from .backups import KGPortabilityBackupMixin as KGPortabilityBackupMixin
|
|
48
|
+
from .backups import _sha256_file as _sha256_file
|
|
49
|
+
from .bundles import _canonical_digest as _canonical_digest
|
|
50
|
+
from .bundles import _expand_neighbors as _expand_neighbors
|
|
51
|
+
from .bundles import _node_source_type as _node_source_type
|
|
52
|
+
from .bundles import _redact_node as _redact_node
|
|
53
|
+
from .bundles import _redact_provenance_row as _redact_provenance_row
|
|
54
|
+
from .bundles import _scope_node as _scope_node
|
|
55
|
+
from .bundles import _select_node_ids as _select_node_ids
|
|
56
|
+
from .bundles import _shared_node_summary as _shared_node_summary
|
|
57
|
+
from .bundles import _strip_fields as _strip_fields
|
|
58
|
+
from .constants import _NEIGHBOR_EXCLUDED_TYPES as _NEIGHBOR_EXCLUDED_TYPES
|
|
59
|
+
from .constants import _OPEN_REVIEW_STATUSES as _OPEN_REVIEW_STATUSES
|
|
60
|
+
from .constants import _PAYLOAD_KEYS as _PAYLOAD_KEYS
|
|
61
|
+
from .constants import _REDACTED_FIELDS as _REDACTED_FIELDS
|
|
62
|
+
from .constants import _TRUTHY as _TRUTHY
|
|
63
|
+
from .constants import BACKUP_FORMAT as BACKUP_FORMAT
|
|
64
|
+
from .constants import BRAIN_NETWORK_DISABLED_DETAIL as BRAIN_NETWORK_DISABLED_DETAIL
|
|
65
|
+
from .constants import BRAIN_NETWORK_ENV as BRAIN_NETWORK_ENV
|
|
66
|
+
from .constants import BRAIN_NETWORK_GATE as BRAIN_NETWORK_GATE
|
|
67
|
+
from .constants import ENCRYPTION_MODES as ENCRYPTION_MODES
|
|
68
|
+
from .constants import FORMAT as FORMAT
|
|
69
|
+
from .constants import FORMAT_VERSION as FORMAT_VERSION
|
|
70
|
+
from .constants import SUBGRAPH_ARCHIVE_FORMAT as SUBGRAPH_ARCHIVE_FORMAT
|
|
71
|
+
from .constants import SUBGRAPH_FORMAT as SUBGRAPH_FORMAT
|
|
72
|
+
from .constants import SUBGRAPH_FORMAT_VERSION as SUBGRAPH_FORMAT_VERSION
|
|
73
|
+
from .constants import SUBGRAPH_PROPOSAL_CAP as SUBGRAPH_PROPOSAL_CAP
|
|
74
|
+
from .constants import SUBGRAPH_REVIEW_KIND as SUBGRAPH_REVIEW_KIND
|
|
75
|
+
from .constants import SUBGRAPH_REVIEW_SOURCE as SUBGRAPH_REVIEW_SOURCE
|
|
76
|
+
from .constants import BrainNetworkDisabled as BrainNetworkDisabled
|
|
77
|
+
from .constants import _stamp as _stamp
|
|
78
|
+
from .constants import brain_network_enabled as brain_network_enabled
|
|
79
|
+
from .constants import require_brain_network as require_brain_network
|
|
80
|
+
from .fsops import _checkpoint_sqlite as _checkpoint_sqlite
|
|
81
|
+
from .fsops import _pre_restore_backup_dir as _pre_restore_backup_dir
|
|
82
|
+
from .fsops import _replace_sqlite_atomically as _replace_sqlite_atomically
|
|
83
|
+
from .fsops import _replace_tree_with_backup as _replace_tree_with_backup
|
|
84
|
+
from .fsops import _restore_sibling as _restore_sibling
|
|
85
|
+
from .fsops import _rollback_sqlite_from_backup as _rollback_sqlite_from_backup
|
|
86
|
+
from .fsops import _safe_zip_names as _safe_zip_names
|
|
87
|
+
from .fsops import _sqlite_siblings as _sqlite_siblings
|
|
88
|
+
from .service import KGPortabilityService as KGPortabilityService
|
|
89
|
+
from .sharing import KGPortabilitySharingMixin as KGPortabilitySharingMixin
|
|
90
|
+
from .sharing import load_recipient_identity as load_recipient_identity
|