ltcai 11.2.0 → 11.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (247) hide show
  1. package/README.md +46 -53
  2. package/docs/CHANGELOG.md +61 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/MULTI_AGENT_RUNTIME.md +1 -1
  6. package/docs/ONBOARDING.md +1 -1
  7. package/docs/OPERATIONS.md +6 -2
  8. package/docs/PERMISSION_MODE.md +1 -1
  9. package/docs/TRUST_MODEL.md +1 -1
  10. package/docs/WHY_LATTICE.md +1 -1
  11. package/docs/kg-schema.md +2 -2
  12. package/docs/v11.3.0_PLAN.md +202 -0
  13. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
  14. package/lattice_brain/__init__.py +1 -1
  15. package/lattice_brain/graph/_kg_common/__init__.py +287 -0
  16. package/lattice_brain/graph/_kg_common/extraction.py +516 -0
  17. package/lattice_brain/graph/_kg_common/relations.py +161 -0
  18. package/lattice_brain/graph/_kg_common/text.py +479 -0
  19. package/lattice_brain/graph/discovery_index/__init__.py +35 -0
  20. package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
  21. package/lattice_brain/graph/discovery_index/extract.py +137 -0
  22. package/lattice_brain/graph/discovery_index/scan.py +411 -0
  23. package/lattice_brain/graph/discovery_index/upsert.py +495 -0
  24. package/lattice_brain/graph/projection/__init__.py +42 -0
  25. package/lattice_brain/graph/projection/curation.py +500 -0
  26. package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
  27. package/lattice_brain/graph/retrieval/__init__.py +54 -0
  28. package/lattice_brain/graph/retrieval/context.py +197 -0
  29. package/lattice_brain/graph/retrieval/graph_view.py +319 -0
  30. package/lattice_brain/graph/retrieval/hybrid.py +488 -0
  31. package/lattice_brain/graph/retrieval/maintenance.py +121 -0
  32. package/lattice_brain/graph/retrieval/signals.py +95 -0
  33. package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
  34. package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
  35. package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
  36. package/lattice_brain/graph/retrieval_vector/search.py +560 -0
  37. package/lattice_brain/graph/retrieval_vector/status.py +374 -0
  38. package/lattice_brain/ingestion/__init__.py +130 -0
  39. package/lattice_brain/ingestion/_contract.py +90 -0
  40. package/lattice_brain/ingestion/constants.py +127 -0
  41. package/lattice_brain/ingestion/folder_scan.py +57 -0
  42. package/lattice_brain/ingestion/folders.py +258 -0
  43. package/lattice_brain/ingestion/hashing.py +26 -0
  44. package/lattice_brain/ingestion/jobs_api.py +107 -0
  45. package/lattice_brain/ingestion/models.py +80 -0
  46. package/lattice_brain/ingestion/pipeline.py +486 -0
  47. package/lattice_brain/ingestion/quality.py +209 -0
  48. package/lattice_brain/ingestion/routing.py +295 -0
  49. package/lattice_brain/multimodal/__init__.py +164 -0
  50. package/lattice_brain/multimodal/audio.py +77 -0
  51. package/lattice_brain/multimodal/common.py +118 -0
  52. package/lattice_brain/multimodal/images.py +498 -0
  53. package/lattice_brain/multimodal/ports.py +169 -0
  54. package/lattice_brain/multimodal/video.py +410 -0
  55. package/lattice_brain/portability/__init__.py +90 -0
  56. package/lattice_brain/portability/_contract.py +42 -0
  57. package/lattice_brain/portability/backups.py +338 -0
  58. package/lattice_brain/portability/bundles.py +136 -0
  59. package/lattice_brain/portability/constants.py +93 -0
  60. package/lattice_brain/portability/fsops.py +138 -0
  61. package/lattice_brain/portability/service.py +41 -0
  62. package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
  63. package/lattice_brain/runtime/__init__.py +1 -1
  64. package/lattice_brain/runtime/multi_agent.py +1 -1
  65. package/latticeai/__init__.py +1 -1
  66. package/latticeai/api/chronicle.py +63 -0
  67. package/latticeai/core/agent/__init__.py +93 -0
  68. package/latticeai/core/agent/_contract.py +79 -0
  69. package/latticeai/core/agent/context.py +57 -0
  70. package/latticeai/core/agent/deps.py +125 -0
  71. package/latticeai/core/agent/execution.py +622 -0
  72. package/latticeai/core/agent/planning.py +145 -0
  73. package/latticeai/core/agent/recovery.py +157 -0
  74. package/latticeai/core/agent/runtime.py +210 -0
  75. package/latticeai/core/agent/verification.py +231 -0
  76. package/latticeai/core/embedding_providers/__init__.py +151 -0
  77. package/latticeai/core/embedding_providers/base.py +199 -0
  78. package/latticeai/core/embedding_providers/captions.py +162 -0
  79. package/latticeai/core/embedding_providers/profiles.py +126 -0
  80. package/latticeai/core/embedding_providers/text.py +350 -0
  81. package/latticeai/core/embedding_providers/vision.py +352 -0
  82. package/latticeai/core/file_generation/__init__.py +115 -0
  83. package/latticeai/core/file_generation/bundles.py +76 -0
  84. package/latticeai/core/file_generation/extraction.py +154 -0
  85. package/latticeai/core/file_generation/inference.py +235 -0
  86. package/latticeai/core/file_generation/orchestration.py +152 -0
  87. package/latticeai/core/file_generation/prompting.py +117 -0
  88. package/latticeai/core/file_generation/repair.py +114 -0
  89. package/latticeai/core/file_generation/sanitize.py +61 -0
  90. package/latticeai/core/file_generation/validation.py +201 -0
  91. package/latticeai/core/legacy_compatibility.py +1 -1
  92. package/latticeai/core/marketplace.py +1 -1
  93. package/latticeai/core/messages.py +9 -0
  94. package/latticeai/core/workspace_os_constants.py +1 -1
  95. package/latticeai/integrations/telegram_bot/__init__.py +123 -0
  96. package/latticeai/integrations/telegram_bot/__main__.py +17 -0
  97. package/latticeai/integrations/telegram_bot/config.py +86 -0
  98. package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
  99. package/latticeai/integrations/telegram_bot/flows.py +478 -0
  100. package/latticeai/integrations/telegram_bot/helpers.py +322 -0
  101. package/latticeai/integrations/telegram_bot/screens.py +394 -0
  102. package/latticeai/models/router/__init__.py +88 -0
  103. package/latticeai/models/router/_contract.py +66 -0
  104. package/latticeai/models/router/branding.py +56 -0
  105. package/latticeai/models/router/catalog.py +69 -0
  106. package/latticeai/models/router/documents.py +199 -0
  107. package/latticeai/models/router/errors.py +37 -0
  108. package/latticeai/models/router/generation.py +258 -0
  109. package/latticeai/models/router/loading.py +291 -0
  110. package/latticeai/models/router/local_models.py +85 -0
  111. package/latticeai/models/router/registry.py +147 -0
  112. package/latticeai/runtime/build_phases/__init__.py +82 -0
  113. package/latticeai/runtime/build_phases/features.py +407 -0
  114. package/latticeai/runtime/build_phases/foundation.py +555 -0
  115. package/latticeai/runtime/build_phases/web.py +492 -0
  116. package/latticeai/runtime/runtime_context.py +1 -0
  117. package/latticeai/services/architecture_readiness.py +48 -19
  118. package/latticeai/services/brain_intelligence/__init__.py +58 -0
  119. package/latticeai/services/brain_intelligence/_contract.py +71 -0
  120. package/latticeai/services/brain_intelligence/consistency.py +193 -0
  121. package/latticeai/services/brain_intelligence/constants.py +47 -0
  122. package/latticeai/services/brain_intelligence/digest.py +258 -0
  123. package/latticeai/services/brain_intelligence/health.py +331 -0
  124. package/latticeai/services/brain_intelligence/proposals.py +264 -0
  125. package/latticeai/services/brain_intelligence/sampling.py +84 -0
  126. package/latticeai/services/brain_intelligence/service.py +48 -0
  127. package/latticeai/services/chronicle.py +557 -0
  128. package/latticeai/services/memory_service/__init__.py +52 -0
  129. package/latticeai/services/memory_service/_contract.py +100 -0
  130. package/latticeai/services/memory_service/brief.py +431 -0
  131. package/latticeai/services/memory_service/constants.py +57 -0
  132. package/latticeai/services/memory_service/maintenance.py +138 -0
  133. package/latticeai/services/memory_service/manager.py +186 -0
  134. package/latticeai/services/memory_service/proof.py +136 -0
  135. package/latticeai/services/memory_service/recall.py +225 -0
  136. package/latticeai/services/memory_service/service.py +48 -0
  137. package/latticeai/services/memory_service/stores.py +110 -0
  138. package/latticeai/services/model_runtime/__init__.py +322 -0
  139. package/latticeai/services/model_runtime/cloud.py +87 -0
  140. package/latticeai/services/model_runtime/download.py +282 -0
  141. package/latticeai/services/model_runtime/engines.py +341 -0
  142. package/latticeai/services/model_runtime/loading.py +178 -0
  143. package/latticeai/services/model_runtime/service.py +129 -0
  144. package/latticeai/services/model_runtime/state.py +131 -0
  145. package/latticeai/services/model_runtime/status.py +255 -0
  146. package/latticeai/services/product_readiness.py +15 -7
  147. package/latticeai/setup/wizard/__init__.py +126 -0
  148. package/latticeai/setup/wizard/catalog.py +172 -0
  149. package/latticeai/setup/wizard/detect.py +323 -0
  150. package/latticeai/setup/wizard/install.py +348 -0
  151. package/latticeai/setup/wizard/paths.py +168 -0
  152. package/latticeai/setup/wizard/plans.py +74 -0
  153. package/latticeai/setup/wizard/recommend.py +320 -0
  154. package/package.json +6 -2
  155. package/scripts/bump_version.py +14 -0
  156. package/scripts/capture_release_evidence.mjs +33 -21
  157. package/scripts/check_current_release_docs.mjs +1 -1
  158. package/scripts/check_i18n_namespace_coverage.mjs +41 -4
  159. package/scripts/check_max_file_lines.mjs +102 -0
  160. package/scripts/check_release_evidence_bound.mjs +30 -15
  161. package/scripts/check_screenshot_pixel_delta.py +34 -4
  162. package/scripts/check_server_i18n.mjs +1 -0
  163. package/scripts/generate_rust_parity_fixtures.py +562 -0
  164. package/scripts/lib/mock_server_fingerprint.mjs +94 -0
  165. package/scripts/release_screen_claims.json +31 -2
  166. package/src-tauri/Cargo.lock +361 -3
  167. package/src-tauri/Cargo.toml +6 -1
  168. package/src-tauri/src/backend.rs +349 -0
  169. package/src-tauri/src/folder.rs +33 -0
  170. package/src-tauri/src/main.rs +97 -399
  171. package/src-tauri/tauri.conf.json +1 -1
  172. package/static/app/asset-manifest.json +41 -37
  173. package/static/app/assets/Act-yYpYnn0v.js +1 -0
  174. package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
  175. package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
  176. package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
  177. package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
  178. package/static/app/assets/Capture-CFIRsFNE.js +1 -0
  179. package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
  180. package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
  181. package/static/app/assets/Library-DwO3yZST.js +1 -0
  182. package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
  183. package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
  184. package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
  185. package/static/app/assets/System-DW8F-2xL.js +1 -0
  186. package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
  187. package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
  188. package/static/app/assets/brain-Ci1CkWjM.js +1 -0
  189. package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
  190. package/static/app/assets/circle-check-DfInj-qD.js +1 -0
  191. package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
  192. package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
  193. package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
  194. package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
  195. package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
  196. package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
  197. package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
  198. package/static/app/assets/index-_u5iUHDr.js +10 -0
  199. package/static/app/assets/input-B0lPdRQZ.js +1 -0
  200. package/static/app/assets/link-2-CoFbooHS.js +1 -0
  201. package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
  202. package/static/app/assets/primitives-DEbN-d6p.js +1 -0
  203. package/static/app/assets/search-BybIWPNd.js +1 -0
  204. package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
  205. package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
  206. package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
  207. package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
  208. package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
  209. package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
  210. package/static/app/assets/utils-BlZr7Pd4.js +4 -0
  211. package/static/app/assets/workspace-jJY4RuAV.js +1 -0
  212. package/static/app/index.html +4 -4
  213. package/static/sw.js +1 -1
  214. package/lattice_brain/graph/_kg_common.py +0 -1331
  215. package/lattice_brain/graph/discovery_index.py +0 -1141
  216. package/lattice_brain/graph/retrieval.py +0 -1120
  217. package/lattice_brain/graph/retrieval_vector.py +0 -1293
  218. package/lattice_brain/ingestion.py +0 -1525
  219. package/lattice_brain/multimodal.py +0 -1258
  220. package/latticeai/core/agent.py +0 -1465
  221. package/latticeai/core/embedding_providers.py +0 -1196
  222. package/latticeai/core/file_generation.py +0 -1047
  223. package/latticeai/integrations/telegram_bot.py +0 -1390
  224. package/latticeai/models/router.py +0 -1007
  225. package/latticeai/runtime/build_phases.py +0 -1450
  226. package/latticeai/services/brain_intelligence.py +0 -1083
  227. package/latticeai/services/memory_service.py +0 -1177
  228. package/latticeai/services/model_runtime.py +0 -1281
  229. package/latticeai/setup/wizard.py +0 -1310
  230. package/static/app/assets/Act-AWf0SAKp.js +0 -1
  231. package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
  232. package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
  233. package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
  234. package/static/app/assets/Capture-CqOSzyPr.js +0 -1
  235. package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
  236. package/static/app/assets/Library-CX-bbhmK.js +0 -1
  237. package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
  238. package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
  239. package/static/app/assets/System-Bu2t5hn1.js +0 -1
  240. package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
  241. package/static/app/assets/brain-DJMoqrwx.js +0 -1
  242. package/static/app/assets/index-BpYkzcVm.js +0 -10
  243. package/static/app/assets/input-DSlJJxRs.js +0 -1
  244. package/static/app/assets/primitives-BCx6TvfG.js +0 -1
  245. package/static/app/assets/search-Cgy8cCFJ.js +0 -1
  246. package/static/app/assets/utils-zqPZJxdx.js +0 -4
  247. package/static/app/assets/workspace-DXTihhfU.js +0 -1
@@ -1,1525 +0,0 @@
1
- """Unified ingestion pipeline — the single write-side seam into the Knowledge Graph.
2
-
3
- v3.6.0 Knowledge Graph First principle: *no data source bypasses the Knowledge
4
- Graph and no source creates an isolated silo*. Every source — local files,
5
- connected folders, PDFs/Markdown/text/code, web URLs, browser tabs — is
6
- normalized into one :class:`IngestionItem` and pushed through one
7
- :meth:`IngestionPipeline.ingest` entrypoint:
8
-
9
- Source → normalize → content hash → (file | text) ingest → provenance
10
-
11
- The pipeline is deliberately thin. It owns normalization, idempotency reporting,
12
- provenance capture, and — crucially — routing every ingest through the shared
13
- ``dispatch_tool`` lifecycle so ``pre_tool``/``post_tool`` hooks fire on data
14
- ingestion exactly as they do on tool calls. The heavy graph construction lives in
15
- :class:`knowledge_graph.KnowledgeGraphStore` (``ingest_document`` for files,
16
- ``ingest_source`` for text/web), which this module composes rather than
17
- re-implements.
18
-
19
- Web ingestion seam
20
- ------------------
21
- The graph layer never fetches or parses the web. Fetching, rendering,
22
- readability extraction, and parse quality are the responsibility of the
23
- *upstream* capture surfaces (browser extension, tools layer, MCP servers):
24
- they hand this module already-extracted text. :meth:`IngestionPipeline.
25
- ingest_web_page` is the convenience wrapper for that hand-off — it normalizes
26
- ``(url, extracted_text)`` into an ``IngestionItem(source_type="web_url")`` and
27
- routes it through the exact same :meth:`IngestionPipeline.ingest` door as every
28
- other source. If the extracted text is bad, fix the extractor upstream; the
29
- pipeline will not attempt network access or HTML parsing.
30
-
31
- Folder ingestion (:meth:`IngestionPipeline.ingest_folder`) walks a local
32
- directory, honors a gitignore-like ``.latticeignore`` file at the root
33
- (blank lines, ``#`` comments, ``fnmatch`` glob patterns, ``dir/`` suffix for
34
- directories), always skips common noise (``.git``, ``node_modules``,
35
- ``__pycache__``, virtualenvs, ``dist``, hidden entries by default), applies
36
- size/extension filters, and either ingests inline or schedules through the
37
- existing :class:`BackgroundIngestionQueue`.
38
- """
39
-
40
- from __future__ import annotations
41
-
42
- import fnmatch
43
- import hashlib
44
- import os
45
- from dataclasses import dataclass, field
46
- from pathlib import Path
47
- from typing import Any, Dict, Iterable, List, Optional, Tuple
48
-
49
- from .gates import FeatureGate
50
- from .graph.vector_index import DEFAULT_TICK_LIMIT as VECTOR_TICK_LIMIT
51
- from .multimodal import (
52
- AUDIO_EXTENSIONS,
53
- DEFAULT_KEYFRAMES,
54
- IMAGE_EXTENSIONS,
55
- MODALITY_AUDIO,
56
- MODALITY_IMAGE,
57
- MODALITY_VIDEO,
58
- VIDEO_EXTENSIONS,
59
- VIDEO_UNAVAILABLE_DETAIL,
60
- ImageFacts,
61
- MultimodalPorts,
62
- audio_quality_score,
63
- detect_modality,
64
- extract_image_facts,
65
- ffmpeg_available,
66
- image_quality_score,
67
- read_video_facts,
68
- transcribe_audio,
69
- video_frame_dir,
70
- video_quality_score,
71
- write_image_memory,
72
- write_video_memory,
73
- )
74
- from .runtime.hooks import dispatch_tool
75
- from .utils import utc_now_iso
76
-
77
- # Source types that arrive as a file on disk (read via ingest_document).
78
- FILE_SOURCE_TYPES = frozenset({"file", "local_file", "upload", "pdf"})
79
- # Source types that arrive as extracted text (read via ingest_source).
80
- TEXT_SOURCE_TYPES = frozenset(
81
- {"web_url", "browser_tab", "text", "markdown", "note", "code", "clipboard"}
82
- )
83
- # Conversational exchanges (read via ingest_message — role/content semantics,
84
- # conversation chaining). v4: chat and MCP messages stop bypassing the
85
- # pipeline, so they carry provenance and fire the hook lifecycle like every
86
- # other source.
87
- CHAT_SOURCE_TYPES = frozenset({"chat_message", "mcp_message"})
88
- # Typed memory records (read via ingest_event → Decision/Experience/Event
89
- # nodes). The Memory System writes through the same door as everything else.
90
- MEMORY_SOURCE_TYPES = frozenset({"decision", "experience", "workspace_event"})
91
- _MEMORY_NODE_TYPES = {"decision": "Decision", "experience": "Experience", "workspace_event": "Event"}
92
-
93
- DEFAULT_MAX_TEXT_BYTES = 5 * 1024 * 1024 # 5 MB of extracted text per item
94
-
95
- # ── Folder ingestion (ingest_folder) filters ─────────────────────────────────
96
- # Directories that are always pruned regardless of .latticeignore.
97
- FOLDER_DEFAULT_SKIP_DIRS = frozenset(
98
- {
99
- ".git",
100
- "node_modules",
101
- "__pycache__",
102
- ".venv",
103
- "venv",
104
- "env",
105
- ".pytest_cache",
106
- ".mypy_cache",
107
- ".ruff_cache",
108
- "dist",
109
- "build",
110
- ".next",
111
- "target",
112
- ".cache",
113
- ".idea",
114
- ".vscode",
115
- }
116
- )
117
- # Extension filter matching FILE_SOURCE_TYPES conventions: text/markdown/code
118
- # are read inline (extracted content → chunks); .pdf routes as source_type
119
- # "pdf" through ingest_document (content extraction is upstream's concern).
120
- FOLDER_TEXT_EXTENSIONS = frozenset(
121
- {".txt", ".md", ".markdown", ".rst", ".csv", ".json", ".yaml", ".yml", ".toml", ".ini"}
122
- )
123
- FOLDER_CODE_EXTENSIONS = frozenset(
124
- {
125
- ".py", ".js", ".ts", ".tsx", ".jsx", ".html", ".css", ".go", ".rs",
126
- ".java", ".c", ".h", ".cpp", ".hpp", ".rb", ".php", ".swift", ".kt",
127
- ".sh", ".sql",
128
- }
129
- )
130
- FOLDER_DOCUMENT_EXTENSIONS = frozenset({".pdf"})
131
- DEFAULT_FOLDER_EXTENSIONS = (
132
- FOLDER_TEXT_EXTENSIONS | FOLDER_CODE_EXTENSIONS | FOLDER_DOCUMENT_EXTENSIONS
133
- )
134
- DEFAULT_MAX_FILE_BYTES = 4_000_000 # matches the local-index text/code budget
135
- LATTICEIGNORE_FILENAME = ".latticeignore"
136
- # Opt-out escape hatch for the post-ingest incremental vector sync.
137
- AUTO_VECTOR_INDEX_ENV = "LATTICEAI_AUTO_VECTOR_INDEX"
138
- #: Default *on*, unlike every other gate here: new material has always been made
139
- #: searchable straight away, and this exists so a settings surface can turn that
140
- #: off (batch reindex later) without a restart. ``FeatureGate`` parses the env
141
- #: var with the same words the hand-written opt-out check used, so an untouched
142
- #: install — including one with a nonsense value — answers exactly as before.
143
- AUTO_VECTOR_INDEX_GATE = FeatureGate(
144
- AUTO_VECTOR_INDEX_ENV,
145
- default=True,
146
- name="auto_vector_index",
147
- detail="New material is prepared for semantic search as soon as it lands.",
148
- )
149
-
150
- # ── Multi-modal ingestion (v11.1.0 Track 3) ──────────────────────────────────
151
- # Opt-in, default off, on purpose. Turning it on changes what a folder scan
152
- # *stores* (pictures and recordings, with OCR and — if a model is loaded —
153
- # captions and vectors), and that is the user's call, not a default. With the
154
- # flag off every routing decision below is skipped and behaviour is byte-for-
155
- # byte what it was before this release.
156
- ALLOW_MULTIMODAL_ENV = "LATTICEAI_ALLOW_MULTIMODAL"
157
- #: The multi-modal switch, resolved when it is asked rather than frozen into
158
- #: ``self`` at construction (v11.2.0). The environment variable is still the
159
- #: answer for an untouched install — same var, same words, same default off —
160
- #: but a settings surface can bind a resolver and move it without a restart.
161
- MULTIMODAL_GATE = FeatureGate(
162
- ALLOW_MULTIMODAL_ENV,
163
- default=False,
164
- name="allow_multimodal",
165
- detail="Pictures and recordings are only ingested when this is turned on.",
166
- )
167
- #: Video is a *sub-switch* of the one above: with multi-modal off nothing about
168
- #: video happens at all, and with it on video is included unless this is
169
- #: explicitly turned off. The effective default is therefore still "no video",
170
- #: and the seam exists so a settings screen can offer pictures without films.
171
- ALLOW_VIDEO_ENV = "LATTICEAI_ALLOW_VIDEO"
172
- VIDEO_GATE = FeatureGate(
173
- ALLOW_VIDEO_ENV,
174
- default=True,
175
- name="allow_video",
176
- detail="Videos are ingested as keyframes plus subtitles when multi-modal is on.",
177
- )
178
- #: Source types that name a modality outright (a caller who already knows).
179
- IMAGE_SOURCE_TYPES = frozenset({"image", "screenshot", "photo"})
180
- AUDIO_SOURCE_TYPES = frozenset({"audio", "voice_memo", "recording"})
181
- VIDEO_SOURCE_TYPES = frozenset({"video", "screen_recording", "movie"})
182
- #: Added to the folder-scan allow-list only while multimodal is enabled.
183
- FOLDER_MULTIMODAL_EXTENSIONS = IMAGE_EXTENSIONS | AUDIO_EXTENSIONS
184
- #: Videos join the folder allow-list only when this machine can decode one —
185
- #: scanning a folder into a pile of refusals is not a feature.
186
- FOLDER_VIDEO_EXTENSIONS = VIDEO_EXTENSIONS
187
- #: Graph node type for a recording. ``NodeType.AUDIO`` normalizes this on the
188
- #: KG v2 write side; the legacy tables keep the label verbatim, which is what
189
- #: every type-aware read (graph view, context sections, doc-gen) matches on.
190
- AUDIO_NODE_TYPE = "Audio"
191
-
192
- # ── Extraction quality heuristics (v9.8.0 A1) ────────────────────────────────
193
- # Pure heuristics over the extracted text — no model calls, no network. The
194
- # score is *advisory*: it never blocks an ingest, it only annotates the result
195
- # so capture surfaces (browser, folder scan) can surface low-quality warnings.
196
- QUALITY_HIGH_THRESHOLD = 0.7
197
- QUALITY_LOW_THRESHOLD = 0.4
198
- QUALITY_LOW_WARNING = "추출 품질이 낮습니다 — 원문 확인을 권장합니다."
199
- _WEB_SOURCE_TYPES = frozenset({"web_url", "browser_tab"})
200
- # Standalone short lines that smell like leftover site chrome (nav/menu/footer).
201
- _BOILERPLATE_LINE_MARKERS = frozenset(
202
- {
203
- "home", "menu", "nav", "navigation", "login", "log in", "sign in",
204
- "sign up", "register", "subscribe", "search", "about", "about us",
205
- "contact", "contact us", "privacy policy", "terms of service",
206
- "cookie policy", "accept cookies", "accept all cookies", "share",
207
- "skip to content", "copyright", "all rights reserved", "sitemap",
208
- "back to top", "footer", "read more", "next", "previous",
209
- }
210
- )
211
-
212
-
213
- def _quality_level(score: float) -> str:
214
- if score >= QUALITY_HIGH_THRESHOLD:
215
- return "high"
216
- if score >= QUALITY_LOW_THRESHOLD:
217
- return "medium"
218
- return "low"
219
-
220
-
221
- def assess_extraction_quality(
222
- text: Optional[str],
223
- *,
224
- source_type: Optional[str] = None,
225
- upstream_confidence: Optional[Any] = None,
226
- ) -> Dict[str, Any]:
227
- """Score extracted text 0..1 with reasons (pure heuristic, deterministic).
228
-
229
- Signals: text length, whitespace ratio, character/word diversity
230
- (repetition), sentence structure, and — for web sources — leftover
231
- nav/menu boilerplate. When the upstream extractor supplies its own
232
- confidence (``upstream_confidence``), that value wins verbatim: the
233
- extractor saw the raw document, this function only sees its output.
234
- """
235
- if upstream_confidence is not None:
236
- try:
237
- score = max(0.0, min(1.0, float(upstream_confidence)))
238
- except (TypeError, ValueError):
239
- score = None
240
- if score is not None:
241
- return {
242
- "score": round(score, 4),
243
- "level": _quality_level(score),
244
- "reasons": ["upstream_confidence"],
245
- }
246
-
247
- raw = str(text or "")
248
- stripped = raw.strip()
249
- if not stripped:
250
- return {"score": 0.0, "level": "low", "reasons": ["empty_text"]}
251
-
252
- reasons: List[str] = []
253
- length = len(stripped)
254
- sample = stripped[:4000]
255
- lines = [ln.strip() for ln in stripped.splitlines() if ln.strip()]
256
- words = stripped.split()
257
-
258
- # 1) Length — very short extractions rarely carry recall value.
259
- if length < 40:
260
- length_factor = 0.35
261
- reasons.append("very_short_text")
262
- elif length < 120:
263
- length_factor = 0.6
264
- reasons.append("short_text")
265
- elif length < 300:
266
- length_factor = 0.85
267
- else:
268
- length_factor = 1.0
269
-
270
- # 2) Sentence structure — prose has sentence-ending punctuation.
271
- sentence_marks = sum(sample.count(mark) for mark in (".", "!", "?", "…", "。", "!", "?"))
272
- if sentence_marks > 0:
273
- structure_factor = 1.0
274
- elif length < 200:
275
- structure_factor = 0.75 # titles/snippets legitimately lack periods
276
- else:
277
- structure_factor = 0.45
278
- reasons.append("no_sentence_structure")
279
-
280
- # 3) Diversity — repeated characters/lines/words indicate extraction junk.
281
- diversity_factor = 1.0
282
- distinct_chars = len(set(sample.lower()))
283
- if distinct_chars < 10:
284
- diversity_factor *= 0.2
285
- reasons.append("low_character_diversity")
286
- elif distinct_chars < 20:
287
- diversity_factor *= 0.7
288
- if len(lines) >= 6:
289
- top_count = max(lines.count(ln) for ln in set(lines))
290
- if top_count >= max(3, len(lines) // 4):
291
- diversity_factor *= 0.5
292
- reasons.append("repetitive_lines")
293
- if len(words) >= 30 and (len(set(w.lower() for w in words)) / len(words)) < 0.25:
294
- diversity_factor *= 0.5
295
- reasons.append("repetitive_words")
296
-
297
- # 4) Cleanliness — whitespace floods, fragmented lines, site chrome.
298
- cleanliness_factor = 1.0
299
- whitespace_ratio = sum(1 for ch in raw if ch.isspace()) / max(1, len(raw))
300
- if whitespace_ratio > 0.45:
301
- cleanliness_factor *= 0.6
302
- reasons.append("high_whitespace_ratio")
303
- if len(lines) >= 8:
304
- short_lines = sum(1 for ln in lines if len(ln.split()) <= 3)
305
- if short_lines / len(lines) > 0.6:
306
- cleanliness_factor *= 0.6
307
- reasons.append("fragmented_lines")
308
- boilerplate_hits = sum(
309
- 1 for ln in lines if ln.lower().strip(" .:>|•·-–—*") in _BOILERPLATE_LINE_MARKERS
310
- )
311
- if lines and boilerplate_hits >= 3 and (boilerplate_hits / len(lines)) > 0.2:
312
- cleanliness_factor *= 0.35
313
- if str(source_type or "").lower() in _WEB_SOURCE_TYPES:
314
- reasons.append("nav_menu_remnants")
315
- else:
316
- reasons.append("boilerplate_markers")
317
-
318
- score = length_factor * structure_factor * diversity_factor * cleanliness_factor
319
- score = max(0.0, min(1.0, score))
320
- if not reasons:
321
- reasons.append("clean_extraction")
322
- return {"score": round(score, 4), "level": _quality_level(score), "reasons": reasons}
323
-
324
-
325
- # ── capture quality CTA (backlog #9, review §7.2 C) ──────────────────────────
326
- # Structured verdict over the same extraction-quality schema the rest of the
327
- # pipeline uses, so capture surfaces (browser extension, read-url) can render
328
- # an honest "this capture is thin" CTA instead of silently storing junk.
329
- CAPTURE_SUGGESTIONS_THIN = ["recapture", "paste_manually", "highlight_source"]
330
- _CAPTURE_REASON_LABELS = {
331
- "empty_text": "추출된 본문이 비어 있습니다",
332
- "very_short_text": "추출된 본문이 매우 짧습니다",
333
- "short_text": "추출된 본문이 짧습니다",
334
- "no_sentence_structure": "문장 구조가 거의 없습니다",
335
- "low_character_diversity": "반복 문자가 대부분입니다",
336
- "repetitive_lines": "같은 줄이 반복됩니다",
337
- "repetitive_words": "같은 단어가 반복됩니다",
338
- "high_whitespace_ratio": "공백이 지나치게 많습니다",
339
- "fragmented_lines": "줄이 잘게 조각나 있습니다",
340
- "nav_menu_remnants": "메뉴/내비게이션 잔여물이 많습니다",
341
- "boilerplate_markers": "상용구 텍스트가 많습니다",
342
- "no_extracted_text": "추출된 텍스트가 없습니다",
343
- }
344
-
345
-
346
- def capture_quality_verdict(
347
- extraction_quality: Optional[Dict[str, Any]],
348
- *,
349
- source_type: Optional[str] = None,
350
- ) -> Dict[str, Any]:
351
- """Structured CTA verdict from a pipeline ``extraction_quality`` dict.
352
-
353
- ``{"status": "thin"|"ok", "reason": str|None, "suggestions": [...],
354
- "score": float|None, "level": str|None}``. ``thin`` (level == "low", the
355
- same threshold as the ingest warning) carries actionable suggestions —
356
- ``recapture`` / ``paste_manually`` / ``highlight_source`` — so the UI can
357
- offer the user a way to fix the capture instead of hiding the problem.
358
- Deterministic and never raises; ``None`` input yields an honest ``thin``.
359
- """
360
- if not isinstance(extraction_quality, dict):
361
- return {
362
- "status": "thin",
363
- "reason": _CAPTURE_REASON_LABELS["no_extracted_text"],
364
- "reason_codes": ["no_extracted_text"],
365
- "suggestions": list(CAPTURE_SUGGESTIONS_THIN),
366
- "score": None,
367
- "level": None,
368
- }
369
- level = str(extraction_quality.get("level") or "")
370
- score = extraction_quality.get("score")
371
- reasons = [str(item) for item in (extraction_quality.get("reasons") or [])]
372
- thin = level == "low"
373
- reason = None
374
- if thin:
375
- labeled = [
376
- _CAPTURE_REASON_LABELS[code]
377
- for code in reasons
378
- if code in _CAPTURE_REASON_LABELS
379
- ]
380
- reason = "; ".join(labeled) if labeled else QUALITY_LOW_WARNING
381
- return {
382
- "status": "thin" if thin else "ok",
383
- "reason": reason,
384
- "reason_codes": reasons if thin else [],
385
- "suggestions": list(CAPTURE_SUGGESTIONS_THIN) if thin else [],
386
- "score": score,
387
- "level": level or None,
388
- }
389
-
390
-
391
- def _load_latticeignore(root: Path) -> List[str]:
392
- """Parse ``root/.latticeignore`` → glob patterns (gitignore-like subset)."""
393
- ignore_file = root / LATTICEIGNORE_FILENAME
394
- patterns: List[str] = []
395
- if not ignore_file.is_file():
396
- return patterns
397
- try:
398
- lines = ignore_file.read_text(encoding="utf-8", errors="ignore").splitlines()
399
- except OSError:
400
- return patterns
401
- for raw in lines:
402
- line = raw.strip()
403
- if not line or line.startswith("#"):
404
- continue
405
- patterns.append(line)
406
- return patterns
407
-
408
-
409
- def _matches_ignore(
410
- rel_posix: str, name: str, *, is_dir: bool, patterns: Iterable[str]
411
- ) -> bool:
412
- """fnmatch-based .latticeignore matching.
413
-
414
- - ``pattern/`` matches directories only (files under it never appear
415
- because ignored directories are pruned during the walk).
416
- - Patterns match against both the root-relative posix path and the
417
- basename, so ``*.log`` and ``docs/draft.md`` both behave as expected.
418
- """
419
- for raw in patterns:
420
- pattern = raw
421
- if pattern.endswith("/"):
422
- if not is_dir:
423
- continue
424
- pattern = pattern.rstrip("/")
425
- pattern = pattern.lstrip("/")
426
- if not pattern:
427
- continue
428
- if fnmatch.fnmatch(rel_posix, pattern) or fnmatch.fnmatch(name, pattern):
429
- return True
430
- return False
431
-
432
-
433
- # Background job scheduling + progress lives in its own module (v9.9.6):
434
- # the pipeline owns "ingest one item", the queue owns "schedule many and
435
- # report progress". Re-exported so every existing import keeps working.
436
- from .ingestion_jobs import ( # noqa: E402,F401 — re-export for existing importers
437
- JOB_ERRORS_CAP,
438
- BackgroundIngestionJob,
439
- BackgroundIngestionQueue,
440
- )
441
- from .quiet import ( # noqa: E402 — imported after the module constants it depends on
442
- quiet, # noqa: E402 — imported after the module constants it depends on
443
- )
444
-
445
-
446
- @dataclass
447
- class IngestionItem:
448
- """A single thing to ingest, normalized across every source type."""
449
-
450
- source_type: str
451
- title: Optional[str] = None
452
- text: Optional[str] = None # text/web sources
453
- path: Optional[str] = None # file sources
454
- source_uri: Optional[str] = None
455
- mime_type: Optional[str] = None
456
- owner: Optional[str] = None
457
- workspace_id: Optional[str] = None
458
- permissions: Optional[Dict[str, Any]] = None
459
- captured_at: Optional[str] = None
460
- modified_at: Optional[str] = None
461
- conversation_id: Optional[str] = None
462
- agent_used: Optional[str] = None
463
- metadata: Dict[str, Any] = field(default_factory=dict)
464
-
465
-
466
- @dataclass
467
- class IngestionResult:
468
- """The outcome of one ingestion, including provenance and idempotency."""
469
-
470
- status: str # ok | unavailable | blocked | failed
471
- source_type: str
472
- node_id: Optional[str] = None
473
- source_node_id: Optional[str] = None
474
- content_hash: Optional[str] = None
475
- title: Optional[str] = None
476
- chunk_ids: List[str] = field(default_factory=list)
477
- chunk_count: int = 0
478
- duplicate: bool = False
479
- embedded: bool = False
480
- indexing_status: str = "pending" # indexed | skipped | failed | pending
481
- provenance_id: Optional[str] = None
482
- detail: Optional[str] = None
483
- # v9.8.0 additive quality fields — advisory only, never gate behavior.
484
- extraction_quality: Optional[Dict[str, Any]] = None
485
- warnings: List[str] = field(default_factory=list)
486
- quality_gate: Optional[Dict[str, Any]] = None
487
-
488
- def as_dict(self) -> Dict[str, Any]:
489
- payload: Dict[str, Any] = {
490
- "status": self.status,
491
- "source_type": self.source_type,
492
- "node_id": self.node_id,
493
- "source_node_id": self.source_node_id,
494
- "content_hash": self.content_hash,
495
- "title": self.title,
496
- "chunk_ids": self.chunk_ids,
497
- "chunk_count": self.chunk_count,
498
- "duplicate": self.duplicate,
499
- "embedded": self.embedded,
500
- "indexing_status": self.indexing_status,
501
- "provenance_id": self.provenance_id,
502
- "detail": self.detail,
503
- }
504
- # Additive keys only when populated so pre-v9.8 payloads are unchanged.
505
- if self.extraction_quality is not None:
506
- payload["extraction_quality"] = self.extraction_quality
507
- if self.warnings:
508
- payload["warnings"] = list(self.warnings)
509
- if self.quality_gate is not None:
510
- payload["quality_gate"] = self.quality_gate
511
- return payload
512
-
513
-
514
- class IngestionPipeline:
515
- """Single normalized entrypoint that feeds every source into the graph."""
516
-
517
- def __init__(
518
- self,
519
- knowledge_graph: Any,
520
- *,
521
- hooks: Any = None,
522
- enable_graph: bool = True,
523
- audit: Optional[Any] = None,
524
- max_text_bytes: int = DEFAULT_MAX_TEXT_BYTES,
525
- pipeline_name: str = "unified-ingestion",
526
- bg_queue: Optional[BackgroundIngestionQueue] = None,
527
- auto_vector_index: bool = True,
528
- allow_multimodal: bool = False,
529
- multimodal: Optional[MultimodalPorts] = None,
530
- ) -> None:
531
- self._kg = knowledge_graph
532
- self._hooks = hooks
533
- self._enable = bool(enable_graph)
534
- self._audit = audit
535
- self._max_text_bytes = int(max_text_bytes)
536
- self._pipeline_name = pipeline_name
537
- # Background job state lives in the graph database by default, so a
538
- # restart resumes from the last completed item instead of replaying the
539
- # whole corpus. A store without a usable ``db_path`` (mocks, disabled
540
- # graph) degrades to the historical in-memory queue, which reports
541
- # itself as non-durable through ``BackgroundIngestionQueue.describe()``.
542
- self._bg_queue = bg_queue or BackgroundIngestionQueue(
543
- db_path=getattr(knowledge_graph, "db_path", None)
544
- )
545
- # Incremental vector sync after each successful non-duplicate ingest.
546
- # Constructor opt-out AND gate opt-out (LATTICEAI_AUTO_VECTOR_INDEX=0,
547
- # or the settings toggle bound to it) both disable it; a vector failure
548
- # never fails the ingest. The gate half is asked per ingest rather than
549
- # frozen here, so turning it off takes effect on the next item.
550
- self._auto_vector_index_opt_in = bool(auto_vector_index)
551
- # Multi-modal routing. Off unless the caller asks for it *or* the gate
552
- # says yes — the env behind that gate is the escape hatch for an
553
- # install with no code path to the constructor (CLI, background
554
- # worker), and the gate is now asked per call so a runtime toggle can
555
- # reach it. A constructor ``True`` is still a permanent yes.
556
- self._multimodal_opt_in = bool(allow_multimodal)
557
- self._multimodal = multimodal or MultimodalPorts()
558
- self._keyframes = DEFAULT_KEYFRAMES
559
-
560
- @property
561
- def _auto_vector_index(self) -> bool:
562
- """Whether a landed ingest also syncs its vector, asked *now*."""
563
- return self._auto_vector_index_opt_in and AUTO_VECTOR_INDEX_GATE.enabled()
564
-
565
- @property
566
- def _allow_multimodal(self) -> bool:
567
- """Whether pictures and recordings route by modality, asked *now*."""
568
- return self._multimodal_opt_in or MULTIMODAL_GATE.enabled()
569
-
570
- @property
571
- def _allow_video(self) -> bool:
572
- """Video needs multi-modal on, its own sub-switch on, and a decoder."""
573
- return self._allow_multimodal and VIDEO_GATE.enabled() and self._can_decode_video()
574
-
575
- def _can_decode_video(self) -> bool:
576
- """An injected keyframe port counts as a decoder; otherwise, ffmpeg."""
577
- return self._multimodal.keyframe_extractor is not None or ffmpeg_available()
578
-
579
- def available(self) -> bool:
580
- return self._enable and self._kg is not None
581
-
582
- def multimodal_status(self) -> Dict[str, Any]:
583
- """What this pipeline will do with a picture or a recording, honestly.
584
-
585
- ``enabled`` is the flag; the rest is which model-backed capabilities
586
- were actually injected. Video reports whether it can really run — the
587
- answer is no on a machine with no ffmpeg, and it says which of the two
588
- reasons applies rather than leaving the surface to guess.
589
- """
590
- allowed = self._allow_multimodal
591
- video = self._allow_video
592
- return {
593
- "enabled": allowed,
594
- "image": allowed,
595
- "audio": allowed,
596
- "video": video,
597
- "video_detail": None if video else self._video_refusal(),
598
- "gates": {
599
- "multimodal": MULTIMODAL_GATE.describe(),
600
- "video": VIDEO_GATE.describe(),
601
- },
602
- **self._multimodal.describe(),
603
- }
604
-
605
- def _video_refusal(self) -> str:
606
- """Why a video would be refused right now — never a stale reason."""
607
- if not self._allow_multimodal:
608
- return (
609
- "multi-modal ingestion is off; pictures, recordings and videos "
610
- f"are only stored when {ALLOW_MULTIMODAL_ENV} is on"
611
- )
612
- if not VIDEO_GATE.enabled():
613
- return (
614
- "video ingestion is turned off for this install "
615
- f"({ALLOW_VIDEO_ENV}); pictures and recordings are unaffected"
616
- )
617
- return VIDEO_UNAVAILABLE_DETAIL
618
-
619
- # ── public API ───────────────────────────────────────────────────────────
620
- def ingest(self, item: IngestionItem, *, user_email: Optional[str] = None) -> IngestionResult:
621
- """Normalize, hash, route through dispatch_tool, and record provenance."""
622
- source_type = str(item.source_type or "text").strip().lower()
623
- if not self.available():
624
- return IngestionResult(
625
- status="unavailable", source_type=source_type,
626
- indexing_status="skipped",
627
- detail="Knowledge Graph is disabled (LATTICEAI_ENABLE_GRAPH).",
628
- )
629
-
630
- # Modality routing is a no-op while the flag is off: ``modality`` stays
631
- # "text" and every branch below behaves exactly as it did before.
632
- modality = self._modality_for(item, source_type)
633
- if modality == MODALITY_VIDEO and not self._allow_video:
634
- # Recognized and refused, with the reason that actually applies
635
- # right now — a missing decoder is not the same answer as a
636
- # switched-off feature, and the caller can act on the difference.
637
- return IngestionResult(
638
- status="unavailable", source_type=source_type,
639
- indexing_status="skipped", detail=self._video_refusal(),
640
- )
641
-
642
- captured_at = item.captured_at or utc_now_iso()
643
- owner = item.owner or user_email
644
- tool_name = f"kg_ingest.{source_type}"
645
- # Only the keys are read by the hook payload, so this dict is safe/cheap.
646
- args = {
647
- "source_type": source_type,
648
- "source_uri": item.source_uri,
649
- "owner": owner,
650
- "workspace_id": item.workspace_id,
651
- }
652
-
653
- def _run() -> Dict[str, Any]:
654
- if source_type in CHAT_SOURCE_TYPES:
655
- return self._ingest_chat(item, source_type=source_type, owner=owner)
656
- if source_type in MEMORY_SOURCE_TYPES:
657
- return self._ingest_memory_record(item, source_type=source_type, owner=owner)
658
- if modality == MODALITY_IMAGE:
659
- return self._ingest_image(item, source_type=source_type, owner=owner, captured_at=captured_at)
660
- if modality == MODALITY_AUDIO:
661
- return self._ingest_audio(item, source_type=source_type, owner=owner, captured_at=captured_at)
662
- if modality == MODALITY_VIDEO:
663
- return self._ingest_video(item, source_type=source_type, owner=owner, captured_at=captured_at)
664
- if source_type in FILE_SOURCE_TYPES or (item.path and not item.text):
665
- return self._ingest_file(item, source_type=source_type, owner=owner, captured_at=captured_at)
666
- return self._ingest_text(item, source_type=source_type, owner=owner, captured_at=captured_at)
667
-
668
- # v9.8.0 observation-only quality gate: computed *before* the write so
669
- # the search never matches the node we are about to create. It is
670
- # recorded on the result and never skips an ingest (behavior unchanged).
671
- quality_text = self._extractable_text(item)
672
- quality_gate = self._observe_quality_gate(
673
- item, source_type=source_type, text=quality_text,
674
- )
675
-
676
- try:
677
- raw = dispatch_tool(
678
- self._hooks, tool_name, args, _run,
679
- user_email=user_email, workspace_id=item.workspace_id, source="ingestion",
680
- )
681
- except PermissionError as exc:
682
- return IngestionResult(
683
- status="blocked", source_type=source_type,
684
- indexing_status="skipped", detail=str(exc),
685
- )
686
- except FileNotFoundError as exc:
687
- return IngestionResult(
688
- status="failed", source_type=source_type,
689
- indexing_status="failed", detail=str(exc),
690
- )
691
- except Exception as exc: # noqa: BLE001 — surface as a failed result, never crash the caller
692
- return IngestionResult(
693
- status="failed", source_type=source_type,
694
- indexing_status="failed", detail=str(exc),
695
- )
696
-
697
- node_id = raw.get("node_id")
698
- content_hash = raw.get("content_hash") or raw.get("sha256")
699
- chunk_ids = list(raw.get("chunk_ids") or [])
700
- title = raw.get("title") or item.title
701
-
702
- # Incremental vector-index sync (opt-in via auto_vector_index +
703
- # LATTICEAI_AUTO_VECTOR_INDEX). Exception-safe by contract: the graph
704
- # write above already landed, so a vector failure only downgrades
705
- # indexing_status to "pending" — index_status()/rebuild_vector_index()
706
- # discover the same node as backlog and pick it up later.
707
- indexing_status = "indexed"
708
- vector_detail: Optional[str] = None
709
- if node_id and self._auto_vector_index and not bool(raw.get("duplicate")):
710
- indexing_status, vector_detail = self._sync_vector_index(node_id)
711
- embedded = bool(self._kg.node_is_embedded(node_id)) if node_id else False
712
-
713
- # Provenance capture must never turn an already-persisted ingest into a
714
- # caller-visible failure: the graph write above succeeded, so a broken
715
- # provenance table degrades the result instead of raising.
716
- provenance_detail: Optional[str] = None
717
- try:
718
- prov = self._kg.record_provenance(
719
- node_id=node_id,
720
- source_type=source_type,
721
- pipeline=self._pipeline_name,
722
- source_uri=item.source_uri,
723
- content_hash=content_hash,
724
- title=title,
725
- owner=owner,
726
- workspace_id=item.workspace_id,
727
- captured_at=captured_at,
728
- modified_at=item.modified_at,
729
- embedded=embedded,
730
- linked=bool(raw.get("source_node_id")),
731
- duplicate=bool(raw.get("duplicate")),
732
- agent_used=item.agent_used,
733
- chunk_count=len(chunk_ids),
734
- permissions=item.permissions,
735
- metadata=item.metadata,
736
- )
737
- except Exception as exc: # noqa: BLE001 — the ingest itself already landed
738
- prov = {}
739
- provenance_detail = f"provenance capture failed: {exc}"
740
- if self._audit is not None:
741
- try:
742
- self._audit(
743
- "kg_ingest",
744
- {
745
- "source_type": source_type, "node_id": node_id,
746
- "content_hash": content_hash, "duplicate": bool(raw.get("duplicate")),
747
- },
748
- user_email,
749
- )
750
- except Exception: # noqa: BLE001 — audit must never break ingestion
751
- quiet()
752
-
753
- # A modality-aware door scores its own extraction (a picture's quality
754
- # is "how much of it can be retrieved", not "does the text read well"),
755
- # so its verdict wins. Text/file doors never set the key and keep the
756
- # historical scoring untouched.
757
- extraction_quality = raw.get("extraction_quality") or self._assess_item_quality(
758
- item, source_type=source_type, text=quality_text, chunk_ids=chunk_ids,
759
- )
760
- warnings: List[str] = []
761
- if extraction_quality is not None and extraction_quality.get("level") == "low":
762
- warnings.append(QUALITY_LOW_WARNING)
763
-
764
- details = [d for d in (provenance_detail, vector_detail) if d]
765
- return IngestionResult(
766
- status="ok",
767
- source_type=source_type,
768
- node_id=node_id,
769
- source_node_id=raw.get("source_node_id"),
770
- content_hash=content_hash,
771
- title=title,
772
- chunk_ids=chunk_ids,
773
- chunk_count=len(chunk_ids),
774
- duplicate=bool(raw.get("duplicate")),
775
- embedded=embedded,
776
- indexing_status=indexing_status,
777
- provenance_id=prov.get("id"),
778
- detail="; ".join(details) if details else None,
779
- extraction_quality=extraction_quality,
780
- warnings=warnings,
781
- quality_gate=quality_gate,
782
- )
783
-
784
- # ── extraction quality (v9.8.0 A1 — advisory, never gates) ───────────────
785
- @staticmethod
786
- def _extractable_text(item: IngestionItem) -> Optional[str]:
787
- """Best available extracted text for quality scoring/gating."""
788
- if item.text is not None:
789
- return item.text
790
- extracted = (item.metadata or {}).get("extracted")
791
- if isinstance(extracted, dict):
792
- content = extracted.get("content") or extracted.get("text")
793
- if content is not None:
794
- return str(content)
795
- return None
796
-
797
- @staticmethod
798
- def _upstream_confidence(item: IngestionItem) -> Optional[Any]:
799
- """Upstream extractor confidence, if the capture surface supplied one."""
800
- meta = item.metadata or {}
801
- extracted = meta.get("extracted")
802
- if isinstance(extracted, dict) and extracted.get("confidence") is not None:
803
- return extracted.get("confidence")
804
- if meta.get("extraction_confidence") is not None:
805
- return meta.get("extraction_confidence")
806
- return None
807
-
808
- def _assess_item_quality(
809
- self,
810
- item: IngestionItem,
811
- *,
812
- source_type: str,
813
- text: Optional[str],
814
- chunk_ids: List[str],
815
- ) -> Optional[Dict[str, Any]]:
816
- """Quality annotation for document-like sources (not chat/memory)."""
817
- if source_type in CHAT_SOURCE_TYPES or source_type in MEMORY_SOURCE_TYPES:
818
- return None
819
- confidence = self._upstream_confidence(item)
820
- if text is not None or confidence is not None:
821
- return assess_extraction_quality(
822
- text, source_type=source_type, upstream_confidence=confidence,
823
- )
824
- # File door without inline extraction (e.g. PDF): the pipeline never saw
825
- # the text, so score honestly from the chunk output instead of guessing.
826
- if chunk_ids:
827
- return {
828
- "score": 0.5,
829
- "level": "medium",
830
- "reasons": ["content_extracted_upstream_not_scored"],
831
- }
832
- return {"score": 0.0, "level": "low", "reasons": ["no_extracted_text"]}
833
-
834
- def _observe_quality_gate(
835
- self,
836
- item: IngestionItem,
837
- *,
838
- source_type: str,
839
- text: Optional[str],
840
- ) -> Optional[Dict[str, Any]]:
841
- """Observation-mode ``gate_ingest_candidate`` wiring.
842
-
843
- Records what the proactive gate *would* decide (ingest /
844
- skip_duplicate / review) without ever acting on it. Any failure —
845
- import, search, gate — yields ``None``; the ingest proceeds untouched.
846
- """
847
- if source_type in CHAT_SOURCE_TYPES or source_type in MEMORY_SOURCE_TYPES:
848
- return None
849
- body = str(text or "").strip()
850
- if not body:
851
- return None
852
- try:
853
- from .graph.proactive import gate_ingest_candidate
854
- except Exception: # noqa: BLE001 — optional observation, never required
855
- return None
856
-
857
- def _search(query: str) -> Any:
858
- snippet = str(query or "")[:400]
859
- try:
860
- if item.workspace_id:
861
- return self._kg.search(
862
- snippet, 20, allowed_workspaces={item.workspace_id},
863
- )
864
- return self._kg.search(snippet, 20)
865
- except TypeError:
866
- # Older store without workspace-scoped search.
867
- return self._kg.search(snippet, 20)
868
-
869
- try:
870
- gate = gate_ingest_candidate(body, _search)
871
- except Exception: # noqa: BLE001 — observation must never fail the ingest
872
- return None
873
- parts = [str(gate.get("reason") or "")]
874
- if gate.get("similarity") is not None:
875
- parts.append(f"similarity={gate.get('similarity')}")
876
- if gate.get("match_id"):
877
- parts.append(f"match={gate.get('match_id')}")
878
- return {
879
- "action": str(gate.get("action") or "review"),
880
- "detail": "; ".join(p for p in parts if p),
881
- }
882
-
883
- def _queue_pending_embed(self, node_id: str, detail: str) -> bool:
884
- """Hand a node the inline sync could not embed to the background queue.
885
-
886
- Before v11.1.0 ``indexing_status="pending"`` was the end of the story:
887
- honest, but nobody was coming back for it, so the node stayed
888
- unsearchable until a human ran a rebuild. The durable queue is who
889
- comes back. A store without one (older stores, mocks) just keeps the
890
- old behaviour — the node is still visible as ``index_status`` backlog.
891
- """
892
- queue = getattr(self._kg, "vector_queue", None)
893
- if queue is None:
894
- return False
895
- try:
896
- return bool(queue.schedule(node_id, detail=detail))
897
- except Exception: # noqa: BLE001 — queueing must never fail an ingest
898
- quiet()
899
- return False
900
-
901
- def _sync_vector_index(self, node_id: str) -> Tuple[str, Optional[str]]:
902
- """Best-effort incremental vector sync → (indexing_status, detail).
903
-
904
- Any failure — missing method on older stores, embedding provider down,
905
- storage error — yields ``("pending", detail)`` so a later
906
- ``rebuild_vector_index`` run picks the node up from the backlog, and
907
- the node is queued for background embedding so that pickup happens on
908
- its own.
909
- """
910
- sync = getattr(self._kg, "index_node_incremental", None)
911
- if not callable(sync):
912
- # Older store without the incremental path: the write-side already
913
- # embeds inline, so nothing extra to do.
914
- return "indexed", None
915
- try:
916
- outcome = sync(node_id) or {}
917
- except Exception as exc: # noqa: BLE001 — vector sync must never fail the ingest
918
- return "pending", self._pending_detail(node_id, f"vector index sync failed: {exc}")
919
- if str(outcome.get("status") or "") == "failed":
920
- reason = outcome.get("detail") or "unknown error"
921
- return "pending", self._pending_detail(
922
- node_id, f"vector index sync failed: {reason}"
923
- )
924
- return "indexed", None
925
-
926
- def _pending_detail(self, node_id: str, reason: str) -> str:
927
- """``reason``, plus whether a background retry was actually scheduled."""
928
- if self._queue_pending_embed(node_id, reason):
929
- return f"{reason}; queued for background embedding"
930
- return reason
931
-
932
- def drain_vector_queue(self, limit: int = VECTOR_TICK_LIMIT) -> Dict[str, Any]:
933
- """Run one background-embedding tick over the store's pending backlog.
934
-
935
- Deliberately caller-driven (a scheduler, a CLI, a test) rather than a
936
- thread this pipeline owns: the queue is durable, so "who runs it" is a
937
- deployment decision, not a property of having ingested something.
938
- """
939
- queue = getattr(self._kg, "vector_queue", None)
940
- if queue is None:
941
- return {
942
- "claimed": 0,
943
- "indexed": 0,
944
- "retried": 0,
945
- "failed": 0,
946
- "detail": "this store has no background vector queue",
947
- }
948
- return dict(queue.tick(limit))
949
-
950
- # --- Large candidate #1: background / incremental scheduling (slice) ---
951
- def schedule_background(
952
- self,
953
- items: List[IngestionItem],
954
- *,
955
- incremental: bool = True,
956
- user_email: Optional[str] = None,
957
- ) -> BackgroundIngestionJob:
958
- """Schedule items for background incremental indexing.
959
-
960
- Returns a job handle. Actual execution can be driven by caller
961
- (or future worker) calling pipeline.ingest on each — or through
962
- :meth:`run_background_job`. This seam enables large-corpus scale
963
- without blocking user requests.
964
- """
965
- job = self._bg_queue.schedule(items, incremental=incremental, user_email=user_email)
966
- # mark initial status on results concept (jobs track)
967
- return job
968
-
969
- def get_background_job(self, job_id: str) -> Optional[BackgroundIngestionJob]:
970
- return self._bg_queue.get(job_id)
971
-
972
- def list_background_jobs(self, limit: int = 20) -> List[Dict[str, Any]]:
973
- """Recent jobs (newest first) in the frozen ``/api/ingestion`` schema."""
974
- return [job.as_dict() for job in self._bg_queue.list_recent(limit=limit)]
975
-
976
- def run_background_job(
977
- self, job_id: str, *, user_email: Optional[str] = None
978
- ) -> Dict[str, Any]:
979
- """Execute a queued/interrupted job's remaining items.
980
-
981
- Per-item errors are recorded (capped) and never abort the job. The
982
- final status is ``completed`` (all done), ``partial`` (some done),
983
- or ``failed`` (nothing done). Already-completed items are skipped, so
984
- the same method safely powers both first-run and resume.
985
- """
986
- job = self._bg_queue.get(job_id)
987
- if job is None:
988
- return {"status": "not_found", "job_id": job_id}
989
- if job.status == "running":
990
- return job.as_dict()
991
- return self._execute_background_job(job, user_email=user_email)
992
-
993
- def resume_background_job(
994
- self, job_id: str, *, user_email: Optional[str] = None
995
- ) -> Dict[str, Any]:
996
- """Resume an interrupted/partial/failed job from its remaining items."""
997
- return self.run_background_job(job_id, user_email=user_email)
998
-
999
- def _execute_background_job(
1000
- self, job: BackgroundIngestionJob, *, user_email: Optional[str] = None
1001
- ) -> Dict[str, Any]:
1002
- job.status = "running"
1003
- # Retried items get a fresh verdict: reset failure state for this run.
1004
- job.failed = 0
1005
- job.errors = []
1006
- job.touch()
1007
- self._bg_queue.save(job)
1008
- runner_email = user_email or job.user_email
1009
- for index in job.remaining_indices():
1010
- item = job.items[index]
1011
- try:
1012
- result = self.ingest(item, user_email=runner_email or item.owner)
1013
- status, detail = result.status, result.detail
1014
- except Exception as exc: # noqa: BLE001 — per-item isolation: keep going
1015
- status, detail = "failed", str(exc)
1016
- if status == "ok":
1017
- job.done_indices.add(index)
1018
- else:
1019
- job.record_error(index, item, detail or status)
1020
- job.processed = len(job.done_indices)
1021
- job.touch()
1022
- # Checkpoint per item: a crash here must cost at most the item in
1023
- # flight, never the whole job's progress. One small UPDATE against
1024
- # an ingest (parse + chunk + embed) is noise.
1025
- self._bg_queue.save(job)
1026
- job.processed = len(job.done_indices)
1027
- if job.total == 0 or job.processed >= job.total:
1028
- job.status = "completed"
1029
- elif job.processed > 0:
1030
- job.status = "partial"
1031
- else:
1032
- job.status = "failed"
1033
- job.touch()
1034
- self._bg_queue.save(job)
1035
- return job.as_dict()
1036
-
1037
- def ingest_web_page(
1038
- self,
1039
- url: str,
1040
- extracted_text: str,
1041
- *,
1042
- title: Optional[str] = None,
1043
- metadata: Optional[Dict[str, Any]] = None,
1044
- owner: Optional[str] = None,
1045
- workspace_id: Optional[str] = None,
1046
- captured_at: Optional[str] = None,
1047
- user_email: Optional[str] = None,
1048
- ) -> IngestionResult:
1049
- """Ingest an *already-extracted* web page (see module docstring seam).
1050
-
1051
- Fetching/parsing is upstream's responsibility (browser extension /
1052
- tools layer); this wrapper only normalizes ``(url, extracted_text)``
1053
- into an ``IngestionItem(source_type="web_url")`` and routes it through
1054
- the standard :meth:`ingest` door.
1055
- """
1056
- url = str(url or "").strip()
1057
- if not url:
1058
- return IngestionResult(
1059
- status="failed", source_type="web_url",
1060
- indexing_status="skipped", detail="url required",
1061
- )
1062
- text = str(extracted_text or "")
1063
- if not text.strip():
1064
- return IngestionResult(
1065
- status="failed", source_type="web_url",
1066
- indexing_status="skipped",
1067
- detail=(
1068
- "extracted_text required — the graph layer does not fetch or "
1069
- "parse the web; extraction happens upstream."
1070
- ),
1071
- )
1072
- item = IngestionItem(
1073
- source_type="web_url",
1074
- title=title or url,
1075
- text=text,
1076
- source_uri=url,
1077
- owner=owner,
1078
- workspace_id=workspace_id,
1079
- captured_at=captured_at,
1080
- metadata=dict(metadata or {}),
1081
- )
1082
- return self.ingest(item, user_email=user_email or owner)
1083
-
1084
- def ingest_folder(
1085
- self,
1086
- root_path: Any,
1087
- *,
1088
- recursive: bool = True,
1089
- background: bool = False,
1090
- extensions: Optional[Iterable[str]] = None,
1091
- max_file_bytes: int = DEFAULT_MAX_FILE_BYTES,
1092
- include_hidden: bool = False,
1093
- max_files: int = 1000,
1094
- max_errors: int = 25,
1095
- owner: Optional[str] = None,
1096
- workspace_id: Optional[str] = None,
1097
- user_email: Optional[str] = None,
1098
- ) -> Dict[str, Any]:
1099
- """Walk ``root_path`` and ingest every eligible file through the pipeline.
1100
-
1101
- Filtering, in order: hard skip-list directories (``.git`` …), hidden
1102
- entries (unless ``include_hidden``), root ``.latticeignore`` patterns
1103
- (fnmatch globs; ``dir/`` suffix prunes directories), extension
1104
- allow-list, then ``max_file_bytes``. Text/code files are read inline so
1105
- their content is chunked; ``.pdf`` routes through the file door without
1106
- inline extraction.
1107
-
1108
- ``background=True`` schedules the built items on the existing
1109
- :class:`BackgroundIngestionQueue` instead of ingesting inline.
1110
- Returns a summary dict with counts and per-file errors (capped at
1111
- ``max_errors``).
1112
- """
1113
- summary: Dict[str, Any] = {
1114
- "root": str(root_path),
1115
- "recursive": bool(recursive),
1116
- "background": bool(background),
1117
- "scanned": 0,
1118
- "matched": 0,
1119
- "ingested": 0,
1120
- "duplicate": 0,
1121
- "failed": 0,
1122
- "skipped": {"ignored": 0, "extension": 0, "too_large": 0, "hidden": 0},
1123
- "truncated": False,
1124
- "errors": [],
1125
- }
1126
- try:
1127
- root = Path(root_path).expanduser()
1128
- except TypeError:
1129
- summary.update(status="failed", detail=f"invalid root path: {root_path!r}")
1130
- return summary
1131
- if not root.is_dir():
1132
- summary.update(status="failed", detail=f"not a directory: {root}")
1133
- return summary
1134
- if not self.available():
1135
- summary.update(
1136
- status="unavailable",
1137
- detail="Knowledge Graph is disabled (LATTICEAI_ENABLE_GRAPH).",
1138
- )
1139
- return summary
1140
- summary["root"] = str(root)
1141
- max_files = max(1, int(max_files))
1142
- max_errors = max(0, int(max_errors))
1143
- max_file_bytes = max(1, int(max_file_bytes))
1144
- allowed_exts = (
1145
- frozenset(str(e).lower() if str(e).startswith(".") else f".{str(e).lower()}" for e in extensions)
1146
- if extensions
1147
- else self._folder_extensions()
1148
- )
1149
- patterns = _load_latticeignore(root)
1150
- errors: List[Dict[str, Any]] = summary["errors"]
1151
- skipped = summary["skipped"]
1152
- items: List[IngestionItem] = []
1153
-
1154
- def _record_error(path: Path, detail: str, status: str = "failed") -> None:
1155
- summary["failed"] += 1
1156
- if len(errors) < max_errors:
1157
- errors.append({"path": str(path), "status": status, "detail": detail})
1158
-
1159
- for dirpath, dirnames, filenames in os.walk(root):
1160
- current = Path(dirpath)
1161
- rel_dir = current.relative_to(root)
1162
- kept_dirs: List[str] = []
1163
- for name in sorted(dirnames):
1164
- if name in FOLDER_DEFAULT_SKIP_DIRS:
1165
- continue
1166
- if name.startswith(".") and not include_hidden:
1167
- continue
1168
- rel = name if str(rel_dir) == "." else (rel_dir / name).as_posix()
1169
- if _matches_ignore(rel, name, is_dir=True, patterns=patterns):
1170
- skipped["ignored"] += 1
1171
- continue
1172
- kept_dirs.append(name)
1173
- dirnames[:] = kept_dirs if recursive else []
1174
-
1175
- for name in sorted(filenames):
1176
- if name == LATTICEIGNORE_FILENAME:
1177
- continue
1178
- summary["scanned"] += 1
1179
- path = current / name
1180
- rel = name if str(rel_dir) == "." else (rel_dir / name).as_posix()
1181
- if name.startswith(".") and not include_hidden:
1182
- skipped["hidden"] += 1
1183
- continue
1184
- if _matches_ignore(rel, name, is_dir=False, patterns=patterns):
1185
- skipped["ignored"] += 1
1186
- continue
1187
- ext = path.suffix.lower()
1188
- if ext not in allowed_exts:
1189
- skipped["extension"] += 1
1190
- continue
1191
- try:
1192
- size = path.stat().st_size
1193
- except OSError as exc:
1194
- _record_error(path, f"stat failed: {exc}")
1195
- continue
1196
- if size > max_file_bytes:
1197
- skipped["too_large"] += 1
1198
- continue
1199
- if len(items) >= max_files:
1200
- summary["truncated"] = True
1201
- break
1202
- item_metadata: Dict[str, Any] = {"relative_path": rel}
1203
- if ext in (FOLDER_MULTIMODAL_EXTENSIONS | FOLDER_VIDEO_EXTENSIONS) and self._allow_multimodal:
1204
- # Routed by modality inside ``ingest``; reading the bytes as
1205
- # UTF-8 here would only produce mojibake.
1206
- source_type = "file"
1207
- elif ext in FOLDER_DOCUMENT_EXTENSIONS:
1208
- source_type = "pdf"
1209
- else:
1210
- source_type = "file"
1211
- try:
1212
- content = path.read_text(encoding="utf-8", errors="ignore")
1213
- except OSError as exc:
1214
- _record_error(path, f"read failed: {exc}")
1215
- continue
1216
- item_metadata["extracted"] = {"content": content, "chars": len(content)}
1217
- items.append(
1218
- IngestionItem(
1219
- source_type=source_type,
1220
- title=name,
1221
- path=str(path),
1222
- source_uri=str(path),
1223
- owner=owner,
1224
- workspace_id=workspace_id,
1225
- metadata=item_metadata,
1226
- )
1227
- )
1228
- if summary["truncated"]:
1229
- break
1230
-
1231
- summary["matched"] = len(items)
1232
- if background:
1233
- job = self.schedule_background(
1234
- items, incremental=True, user_email=user_email or owner,
1235
- )
1236
- summary.update(status="scheduled", job_id=job.job_id, scheduled=len(items))
1237
- return summary
1238
-
1239
- for item in items:
1240
- result = self.ingest(item, user_email=user_email or owner)
1241
- if result.status == "ok":
1242
- if result.duplicate:
1243
- summary["duplicate"] += 1
1244
- else:
1245
- summary["ingested"] += 1
1246
- else:
1247
- _record_error(Path(item.path or ""), result.detail or result.status, result.status)
1248
- summary["status"] = "ok" if summary["failed"] == 0 else "partial"
1249
- return summary
1250
-
1251
- def _folder_extensions(self) -> frozenset:
1252
- """Folder-scan allow-list — pictures, recordings and films when enabled.
1253
-
1254
- Video joins only when this machine can actually decode one, so a scan
1255
- never fills the error list with files it was always going to refuse.
1256
- """
1257
- if not self._allow_multimodal:
1258
- return DEFAULT_FOLDER_EXTENSIONS
1259
- allowed = DEFAULT_FOLDER_EXTENSIONS | FOLDER_MULTIMODAL_EXTENSIONS
1260
- if self._allow_video:
1261
- return allowed | FOLDER_VIDEO_EXTENSIONS
1262
- return allowed
1263
-
1264
- # ── routing helpers ──────────────────────────────────────────────────────
1265
- def _ingest_text(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
1266
- text = item.text or ""
1267
- if not text.strip():
1268
- raise ValueError(
1269
- f"Empty content: {source_type} ingestion requires non-empty text."
1270
- )
1271
- if len(text.encode("utf-8", "ignore")) > self._max_text_bytes:
1272
- raise ValueError(
1273
- f"Text payload exceeds the {self._max_text_bytes // (1024 * 1024)}MB ingestion limit."
1274
- )
1275
- title = item.title or item.source_uri or source_type
1276
- return self._kg.ingest_source(
1277
- source_type=source_type,
1278
- title=title,
1279
- text=text,
1280
- source_uri=item.source_uri,
1281
- owner=owner,
1282
- workspace_id=item.workspace_id,
1283
- permissions=item.permissions,
1284
- captured_at=captured_at,
1285
- modified_at=item.modified_at,
1286
- conversation_id=item.conversation_id,
1287
- metadata={"mime_type": item.mime_type, **(item.metadata or {})},
1288
- )
1289
-
1290
- def _ingest_chat(self, item, *, source_type, owner) -> Dict[str, Any]:
1291
- text = item.text or ""
1292
- meta = item.metadata or {}
1293
- role = str(meta.get("role") or "user")
1294
- result = self._kg.ingest_message(
1295
- role,
1296
- text,
1297
- user_email=owner,
1298
- user_nickname=meta.get("user_nickname"),
1299
- source=meta.get("source") or source_type,
1300
- conversation_id=item.conversation_id,
1301
- workspace_id=item.workspace_id,
1302
- raw=meta.get("raw"),
1303
- )
1304
- # ingest_message reports message/response node ids; normalize the keys
1305
- # the provenance step expects.
1306
- result.setdefault("node_id", result.get("node_id") or result.get("message_node_id") or result.get("id"))
1307
- result.setdefault("title", item.title or text[:80])
1308
- return result
1309
-
1310
- def _ingest_memory_record(self, item, *, source_type, owner) -> Dict[str, Any]:
1311
- node_type = _MEMORY_NODE_TYPES[source_type]
1312
- meta = item.metadata or {}
1313
- result = self._kg.ingest_event(
1314
- node_type,
1315
- item.title or (item.text or node_type)[:120],
1316
- user_email=owner,
1317
- source=meta.get("source") or source_type,
1318
- conversation_id=item.conversation_id,
1319
- workspace_id=item.workspace_id,
1320
- metadata={**meta, "detail": (item.text or "")[:2000]},
1321
- )
1322
- result.setdefault("node_id", result.get("node_id") or result.get("id"))
1323
- result.setdefault("title", item.title)
1324
- return result
1325
-
1326
- # ── multi-modal routing (v11.1.0 Track 3) ────────────────────────────────
1327
- def _modality_for(self, item: IngestionItem, source_type: str) -> str:
1328
- """``image`` / ``audio`` / ``video`` / ``text`` for this item.
1329
-
1330
- Always ``"text"`` while the flag is off, which is what makes "off" mean
1331
- *unchanged* rather than *slightly different*.
1332
- """
1333
- if not self._allow_multimodal:
1334
- return "text"
1335
- if source_type in IMAGE_SOURCE_TYPES:
1336
- return MODALITY_IMAGE
1337
- if source_type in AUDIO_SOURCE_TYPES:
1338
- return MODALITY_AUDIO
1339
- if source_type in VIDEO_SOURCE_TYPES:
1340
- return MODALITY_VIDEO
1341
- if not item.path:
1342
- return "text"
1343
- return detect_modality(item.path, item.mime_type)
1344
-
1345
- def _resolve_file_path(self, item: IngestionItem) -> Path:
1346
- if not item.path:
1347
- raise ValueError("File ingestion requires a path.")
1348
- path = Path(item.path)
1349
- if not path.exists():
1350
- raise FileNotFoundError(f"File not found: {path}")
1351
- if path.is_dir():
1352
- raise ValueError(f"File ingestion requires a file, got a directory: {path}")
1353
- return path
1354
-
1355
- def _ingest_image(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
1356
- """Store one picture as an ``Image`` node — OCR, caption, vector.
1357
-
1358
- The image vector (when a vision model produced one) goes to its own
1359
- index; the OCR/caption text rides the ordinary text index. That split
1360
- is what lets a typed question find a screenshot without ever comparing
1361
- a text vector to an image vector.
1362
- """
1363
- path = self._resolve_file_path(item)
1364
- facts = extract_image_facts(str(path), ports=self._multimodal)
1365
- result = write_image_memory(
1366
- self._kg,
1367
- path=path,
1368
- facts=facts,
1369
- title=item.title or path.name,
1370
- source_type=source_type if source_type in IMAGE_SOURCE_TYPES else MODALITY_IMAGE,
1371
- source_uri=item.source_uri,
1372
- owner=owner,
1373
- workspace_id=item.workspace_id,
1374
- conversation_id=item.conversation_id,
1375
- captured_at=captured_at,
1376
- modified_at=item.modified_at,
1377
- permissions=item.permissions,
1378
- extra_metadata={"mime_type": item.mime_type, **(item.metadata or {})},
1379
- )
1380
- self._record_image_vector(result["node_id"], facts)
1381
- quality = image_quality_score(facts)
1382
- result["extraction_quality"] = {
1383
- "score": quality["score"],
1384
- "level": _quality_level(quality["score"]),
1385
- "reasons": quality["reasons"],
1386
- }
1387
- return result
1388
-
1389
- def _record_image_vector(self, node_id: str, facts: ImageFacts) -> None:
1390
- """File the image-space vector, if a vision model actually made one."""
1391
- if facts.embedding is None:
1392
- return
1393
- from .graph.image_vectors import record_image_vector
1394
-
1395
- record_image_vector(
1396
- self._kg,
1397
- node_id=node_id,
1398
- vector=facts.embedding,
1399
- model_id=self._multimodal.vision_model_id or "vision:unnamed",
1400
- space=self._multimodal.vision_space,
1401
- updated_at=utc_now_iso(),
1402
- )
1403
-
1404
- def _ingest_audio(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
1405
- """Store one recording as an ``Audio`` node, transcribed when possible.
1406
-
1407
- The transcript is text and rides the ordinary text index — chunks,
1408
- concepts, provenance, dedupe all unchanged — but the node itself is a
1409
- recording, because that is what it is whether or not anyone could hear
1410
- it. The recording's own facts stay in the metadata (``modality``,
1411
- ``audio_path``, ``transcription``, ``searchable``). Without a
1412
- transcriber the memory is still kept, and its body says plainly that
1413
- the words were never recognized instead of leaving a blank note.
1414
- """
1415
- path = self._resolve_file_path(item)
1416
- facts = transcribe_audio(str(path), ports=self._multimodal, transcript=item.text)
1417
- title = item.title or path.stem
1418
- body = facts.transcript or (
1419
- f"[{MODALITY_AUDIO}] {title}\n"
1420
- "이 녹음은 아직 글로 바뀌지 않았습니다 — 음성 인식기가 없어 내용 검색은 되지 않습니다."
1421
- )
1422
- result = self._kg.ingest_source(
1423
- source_type=source_type,
1424
- title=title,
1425
- text=body,
1426
- source_uri=item.source_uri or str(path),
1427
- owner=owner,
1428
- workspace_id=item.workspace_id,
1429
- permissions=item.permissions,
1430
- captured_at=captured_at,
1431
- modified_at=item.modified_at,
1432
- conversation_id=item.conversation_id,
1433
- node_type=AUDIO_NODE_TYPE,
1434
- metadata={
1435
- "mime_type": item.mime_type,
1436
- "modality": MODALITY_AUDIO,
1437
- "audio_path": str(path),
1438
- "audio_bytes": path.stat().st_size,
1439
- "transcription": facts.transcription_status,
1440
- "searchable": facts.searchable,
1441
- **({"transcription_detail": facts.detail} if facts.detail else {}),
1442
- **(item.metadata or {}),
1443
- },
1444
- )
1445
- result.setdefault("title", title)
1446
- quality = audio_quality_score(facts)
1447
- result["extraction_quality"] = {
1448
- "score": quality["score"],
1449
- "level": _quality_level(quality["score"]),
1450
- "reasons": quality["reasons"],
1451
- }
1452
- return result
1453
-
1454
- def _ingest_video(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
1455
- """Store one video as keyframes through the image door plus subtitles.
1456
-
1457
- Nothing here is a new retrieval path: the stills become ordinary
1458
- ``Image`` nodes (OCR, caption, vector, thumbnail) joined by
1459
- ``CONTAINS_IMAGE``, and the subtitle text becomes ordinary chunks. What
1460
- the ``Video`` node adds is the thing they belong to — and an honest
1461
- body when there were no subtitles to read.
1462
- """
1463
- path = self._resolve_file_path(item)
1464
- facts = read_video_facts(
1465
- str(path),
1466
- video_frame_dir(getattr(self._kg, "blob_dir", path.parent), _file_digest(path)),
1467
- count=self._keyframes,
1468
- ports=self._multimodal,
1469
- subtitle_text=item.text,
1470
- )
1471
- result = write_video_memory(
1472
- self._kg,
1473
- path=path,
1474
- facts=facts,
1475
- title=item.title or path.stem,
1476
- source_type=source_type if source_type in VIDEO_SOURCE_TYPES else MODALITY_VIDEO,
1477
- source_uri=item.source_uri,
1478
- owner=owner,
1479
- workspace_id=item.workspace_id,
1480
- conversation_id=item.conversation_id,
1481
- captured_at=captured_at,
1482
- modified_at=item.modified_at,
1483
- permissions=item.permissions,
1484
- extra_metadata={"mime_type": item.mime_type, **(item.metadata or {})},
1485
- ports=self._multimodal,
1486
- )
1487
- quality = video_quality_score(facts)
1488
- result["extraction_quality"] = {
1489
- "score": quality["score"],
1490
- "level": _quality_level(quality["score"]),
1491
- "reasons": quality["reasons"],
1492
- }
1493
- return result
1494
-
1495
- def _ingest_file(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
1496
- path = self._resolve_file_path(item)
1497
- return self._kg.ingest_document(
1498
- path,
1499
- original_filename=item.title or path.name,
1500
- mime_type=item.mime_type,
1501
- uploader=owner,
1502
- conversation_id=item.conversation_id,
1503
- extracted=item.metadata.get("extracted") if item.metadata else None,
1504
- source_type=source_type,
1505
- source_uri=item.source_uri or str(path),
1506
- captured_at=captured_at,
1507
- modified_at=item.modified_at,
1508
- owner=owner,
1509
- workspace_id=item.workspace_id,
1510
- permissions=item.permissions,
1511
- )
1512
-
1513
-
1514
- def content_hash_text(text: str) -> str:
1515
- """Canonical content hash for a text payload (matches store hashing scheme)."""
1516
- return hashlib.sha256((text or "").encode("utf-8", "ignore")).hexdigest()
1517
-
1518
-
1519
- def _file_digest(path: Path) -> str:
1520
- """Streaming sha256 of a file — the key a video's frame folder is named by."""
1521
- digest = hashlib.sha256()
1522
- with path.open("rb") as handle:
1523
- for block in iter(lambda: handle.read(1024 * 1024), b""):
1524
- digest.update(block)
1525
- return digest.hexdigest()