ltcai 11.1.0 → 11.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (289) hide show
  1. package/README.md +47 -54
  2. package/docs/CHANGELOG.md +94 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/FEATURE_AUDIT_v11.2.0.md +393 -0
  6. package/docs/LAYOUT_REBUILD_SPEC.md +9 -1
  7. package/docs/MULTI_AGENT_RUNTIME.md +1 -1
  8. package/docs/ONBOARDING.md +1 -1
  9. package/docs/OPERATIONS.md +6 -2
  10. package/docs/PERMISSION_MODE.md +1 -1
  11. package/docs/TRUST_MODEL.md +1 -1
  12. package/docs/WHY_LATTICE.md +1 -1
  13. package/docs/architecture.md +6 -2
  14. package/docs/kg-schema.md +2 -2
  15. package/docs/v11.3.0_PLAN.md +202 -0
  16. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
  17. package/lattice_brain/__init__.py +1 -1
  18. package/lattice_brain/gates.py +125 -0
  19. package/lattice_brain/graph/_kg_common/__init__.py +287 -0
  20. package/lattice_brain/graph/_kg_common/extraction.py +516 -0
  21. package/lattice_brain/graph/_kg_common/relations.py +161 -0
  22. package/lattice_brain/graph/_kg_common/text.py +479 -0
  23. package/lattice_brain/graph/discovery_index/__init__.py +35 -0
  24. package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
  25. package/lattice_brain/graph/discovery_index/extract.py +137 -0
  26. package/lattice_brain/graph/discovery_index/scan.py +411 -0
  27. package/lattice_brain/graph/discovery_index/upsert.py +495 -0
  28. package/lattice_brain/graph/fusion.py +35 -4
  29. package/lattice_brain/graph/projection/__init__.py +42 -0
  30. package/lattice_brain/graph/projection/curation.py +500 -0
  31. package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +81 -485
  32. package/lattice_brain/graph/retrieval/__init__.py +54 -0
  33. package/lattice_brain/graph/retrieval/context.py +197 -0
  34. package/lattice_brain/graph/retrieval/graph_view.py +319 -0
  35. package/lattice_brain/graph/retrieval/hybrid.py +488 -0
  36. package/lattice_brain/graph/retrieval/maintenance.py +121 -0
  37. package/lattice_brain/graph/retrieval/signals.py +95 -0
  38. package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
  39. package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
  40. package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
  41. package/lattice_brain/graph/retrieval_vector/search.py +560 -0
  42. package/lattice_brain/graph/retrieval_vector/status.py +374 -0
  43. package/lattice_brain/graph/schema.py +9 -0
  44. package/lattice_brain/graph/store.py +9 -0
  45. package/lattice_brain/graph/vector_index/selector.py +32 -2
  46. package/lattice_brain/ingestion/__init__.py +130 -0
  47. package/lattice_brain/ingestion/_contract.py +90 -0
  48. package/lattice_brain/ingestion/constants.py +127 -0
  49. package/lattice_brain/ingestion/folder_scan.py +57 -0
  50. package/lattice_brain/ingestion/folders.py +258 -0
  51. package/lattice_brain/ingestion/hashing.py +26 -0
  52. package/lattice_brain/ingestion/jobs_api.py +107 -0
  53. package/lattice_brain/ingestion/models.py +80 -0
  54. package/lattice_brain/ingestion/pipeline.py +486 -0
  55. package/lattice_brain/ingestion/quality.py +209 -0
  56. package/lattice_brain/ingestion/routing.py +295 -0
  57. package/lattice_brain/multimodal/__init__.py +164 -0
  58. package/lattice_brain/multimodal/audio.py +77 -0
  59. package/lattice_brain/multimodal/common.py +118 -0
  60. package/lattice_brain/{multimodal.py → multimodal/images.py} +22 -262
  61. package/lattice_brain/multimodal/ports.py +169 -0
  62. package/lattice_brain/multimodal/video.py +410 -0
  63. package/lattice_brain/portability/__init__.py +90 -0
  64. package/lattice_brain/portability/_contract.py +42 -0
  65. package/lattice_brain/portability/backups.py +338 -0
  66. package/lattice_brain/portability/bundles.py +136 -0
  67. package/lattice_brain/portability/constants.py +93 -0
  68. package/lattice_brain/portability/fsops.py +138 -0
  69. package/lattice_brain/portability/service.py +41 -0
  70. package/lattice_brain/portability/sharing.py +714 -0
  71. package/lattice_brain/runtime/__init__.py +1 -1
  72. package/lattice_brain/runtime/multi_agent.py +1 -1
  73. package/lattice_brain/sealed_box.py +244 -0
  74. package/lattice_brain/synthesis.py +24 -1
  75. package/latticeai/__init__.py +1 -1
  76. package/latticeai/api/brain_intelligence.py +4 -0
  77. package/latticeai/api/chat.py +11 -0
  78. package/latticeai/api/chat_helpers.py +16 -3
  79. package/latticeai/api/chat_hybrid.py +32 -1
  80. package/latticeai/api/chronicle.py +63 -0
  81. package/latticeai/api/features.py +70 -0
  82. package/latticeai/api/local_files.py +102 -0
  83. package/latticeai/api/portability.py +39 -4
  84. package/latticeai/api/review_queue.py +126 -0
  85. package/latticeai/api/search.py +16 -2
  86. package/latticeai/core/agent/__init__.py +93 -0
  87. package/latticeai/core/agent/_contract.py +79 -0
  88. package/latticeai/core/agent/context.py +57 -0
  89. package/latticeai/core/agent/deps.py +125 -0
  90. package/latticeai/core/agent/execution.py +622 -0
  91. package/latticeai/core/agent/planning.py +145 -0
  92. package/latticeai/core/agent/recovery.py +157 -0
  93. package/latticeai/core/agent/runtime.py +210 -0
  94. package/latticeai/core/agent/verification.py +231 -0
  95. package/latticeai/core/config.py +4 -1
  96. package/latticeai/core/context_builder.py +6 -3
  97. package/latticeai/core/embedding_providers/__init__.py +151 -0
  98. package/latticeai/core/embedding_providers/base.py +199 -0
  99. package/latticeai/core/embedding_providers/captions.py +162 -0
  100. package/latticeai/core/embedding_providers/profiles.py +126 -0
  101. package/latticeai/core/embedding_providers/text.py +350 -0
  102. package/latticeai/core/embedding_providers/vision.py +352 -0
  103. package/latticeai/core/file_generation/__init__.py +115 -0
  104. package/latticeai/core/file_generation/bundles.py +76 -0
  105. package/latticeai/core/file_generation/extraction.py +154 -0
  106. package/latticeai/core/file_generation/inference.py +235 -0
  107. package/latticeai/core/file_generation/orchestration.py +152 -0
  108. package/latticeai/core/file_generation/prompting.py +117 -0
  109. package/latticeai/core/file_generation/repair.py +114 -0
  110. package/latticeai/core/file_generation/sanitize.py +61 -0
  111. package/latticeai/core/file_generation/validation.py +201 -0
  112. package/latticeai/core/legacy_compatibility.py +1 -1
  113. package/latticeai/core/marketplace.py +1 -1
  114. package/latticeai/core/messages.py +152 -0
  115. package/latticeai/core/model_compat.py +73 -2
  116. package/latticeai/core/workspace_os_constants.py +1 -1
  117. package/latticeai/integrations/telegram_bot/__init__.py +123 -0
  118. package/latticeai/integrations/telegram_bot/__main__.py +17 -0
  119. package/latticeai/integrations/telegram_bot/config.py +86 -0
  120. package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
  121. package/latticeai/integrations/telegram_bot/flows.py +478 -0
  122. package/latticeai/integrations/telegram_bot/helpers.py +322 -0
  123. package/latticeai/integrations/telegram_bot/screens.py +394 -0
  124. package/latticeai/models/model_providers.py +12 -4
  125. package/latticeai/models/router/__init__.py +88 -0
  126. package/latticeai/models/router/_contract.py +66 -0
  127. package/latticeai/models/router/branding.py +56 -0
  128. package/latticeai/models/router/catalog.py +69 -0
  129. package/latticeai/models/router/documents.py +199 -0
  130. package/latticeai/models/router/errors.py +37 -0
  131. package/latticeai/models/router/generation.py +258 -0
  132. package/latticeai/models/router/loading.py +291 -0
  133. package/latticeai/models/router/local_models.py +85 -0
  134. package/latticeai/models/router/registry.py +147 -0
  135. package/latticeai/runtime/build_phases/__init__.py +82 -0
  136. package/latticeai/runtime/build_phases/features.py +407 -0
  137. package/latticeai/runtime/build_phases/foundation.py +555 -0
  138. package/latticeai/runtime/build_phases/web.py +492 -0
  139. package/latticeai/runtime/chat_wiring.py +4 -0
  140. package/latticeai/runtime/feature_toggle_wiring.py +163 -0
  141. package/latticeai/runtime/router_registration.py +11 -0
  142. package/latticeai/runtime/runtime_context.py +1 -0
  143. package/latticeai/services/app_context.py +8 -0
  144. package/latticeai/services/architecture_readiness.py +48 -19
  145. package/latticeai/services/automation_intelligence.py +22 -2
  146. package/latticeai/services/brain_intelligence/__init__.py +58 -0
  147. package/latticeai/services/brain_intelligence/_contract.py +71 -0
  148. package/latticeai/services/brain_intelligence/consistency.py +193 -0
  149. package/latticeai/services/brain_intelligence/constants.py +47 -0
  150. package/latticeai/services/brain_intelligence/digest.py +258 -0
  151. package/latticeai/services/brain_intelligence/health.py +331 -0
  152. package/latticeai/services/brain_intelligence/proposals.py +264 -0
  153. package/latticeai/services/brain_intelligence/sampling.py +84 -0
  154. package/latticeai/services/brain_intelligence/service.py +48 -0
  155. package/latticeai/services/chronicle.py +557 -0
  156. package/latticeai/services/command_center.py +10 -4
  157. package/latticeai/services/feature_toggles.py +502 -0
  158. package/latticeai/services/folder_watch.py +122 -1
  159. package/latticeai/services/hybrid_chat.py +56 -5
  160. package/latticeai/services/interop_bridges.py +978 -0
  161. package/latticeai/services/memory_service/__init__.py +52 -0
  162. package/latticeai/services/memory_service/_contract.py +100 -0
  163. package/latticeai/services/memory_service/brief.py +431 -0
  164. package/latticeai/services/memory_service/constants.py +57 -0
  165. package/latticeai/services/memory_service/maintenance.py +138 -0
  166. package/latticeai/services/memory_service/manager.py +186 -0
  167. package/latticeai/services/memory_service/proof.py +136 -0
  168. package/latticeai/services/memory_service/recall.py +225 -0
  169. package/latticeai/services/memory_service/service.py +48 -0
  170. package/latticeai/services/memory_service/stores.py +110 -0
  171. package/latticeai/services/model_capability_registry.py +434 -261
  172. package/latticeai/services/model_catalog.py +95 -61
  173. package/latticeai/services/model_recommendation.py +18 -11
  174. package/latticeai/services/model_runtime/__init__.py +322 -0
  175. package/latticeai/services/model_runtime/cloud.py +87 -0
  176. package/latticeai/services/model_runtime/download.py +282 -0
  177. package/latticeai/services/model_runtime/engines.py +341 -0
  178. package/latticeai/services/model_runtime/loading.py +178 -0
  179. package/latticeai/services/model_runtime/service.py +129 -0
  180. package/latticeai/services/model_runtime/state.py +131 -0
  181. package/latticeai/services/model_runtime/status.py +255 -0
  182. package/latticeai/services/multimodal_ports.py +26 -1
  183. package/latticeai/services/obsidian_bridge.py +16 -25
  184. package/latticeai/services/product_readiness.py +15 -7
  185. package/latticeai/services/search_service.py +149 -2
  186. package/latticeai/services/tool_dispatch.py +4 -0
  187. package/latticeai/setup/auto_setup.py +27 -30
  188. package/latticeai/setup/wizard/__init__.py +126 -0
  189. package/latticeai/setup/wizard/catalog.py +172 -0
  190. package/latticeai/setup/wizard/detect.py +323 -0
  191. package/latticeai/setup/wizard/install.py +348 -0
  192. package/latticeai/setup/wizard/paths.py +168 -0
  193. package/latticeai/setup/wizard/plans.py +74 -0
  194. package/latticeai/setup/wizard/recommend.py +320 -0
  195. package/package.json +6 -2
  196. package/scripts/bump_version.py +14 -0
  197. package/scripts/capture_release_evidence.mjs +33 -21
  198. package/scripts/check_current_release_docs.mjs +1 -1
  199. package/scripts/check_i18n_namespace_coverage.mjs +41 -4
  200. package/scripts/check_max_file_lines.mjs +102 -0
  201. package/scripts/check_release_evidence_bound.mjs +30 -15
  202. package/scripts/check_screenshot_pixel_delta.py +34 -4
  203. package/scripts/check_server_i18n.mjs +2 -0
  204. package/scripts/generate_rust_parity_fixtures.py +562 -0
  205. package/scripts/lib/mock_server_fingerprint.mjs +94 -0
  206. package/scripts/release_screen_claims.json +44 -2
  207. package/scripts/verify_hf_model_registry.py +253 -218
  208. package/src-tauri/Cargo.lock +361 -3
  209. package/src-tauri/Cargo.toml +6 -1
  210. package/src-tauri/src/backend.rs +349 -0
  211. package/src-tauri/src/folder.rs +33 -0
  212. package/src-tauri/src/main.rs +97 -399
  213. package/src-tauri/tauri.conf.json +1 -1
  214. package/static/app/asset-manifest.json +41 -37
  215. package/static/app/assets/Act-yYpYnn0v.js +1 -0
  216. package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
  217. package/static/app/assets/{Brain-CzCsI1mi.js → Brain-C1HBN0Wf.js} +2 -2
  218. package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
  219. package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
  220. package/static/app/assets/Capture-CFIRsFNE.js +1 -0
  221. package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
  222. package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
  223. package/static/app/assets/Library-DwO3yZST.js +1 -0
  224. package/static/app/assets/{LivingBrain-BXMWIK_2.js → LivingBrain-Jn1GK0-S.js} +1 -1
  225. package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
  226. package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
  227. package/static/app/assets/System-DW8F-2xL.js +1 -0
  228. package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
  229. package/static/app/assets/{bot-4BvN07ux.js → bot-IM_E_Y12.js} +1 -1
  230. package/static/app/assets/brain-Ci1CkWjM.js +1 -0
  231. package/static/app/assets/{button-CDjtnAoU.js → button-COwyqfHM.js} +1 -1
  232. package/static/app/assets/circle-check-DfInj-qD.js +1 -0
  233. package/static/app/assets/{circle-pause-D_RMn7tp.js → circle-pause-DEM4A1Y5.js} +1 -1
  234. package/static/app/assets/{circle-play-B5OpB8ae.js → circle-play-C9djDuLd.js} +1 -1
  235. package/static/app/assets/{cpu-BIlWInHf.js → cpu-DFdo1gw-.js} +1 -1
  236. package/static/app/assets/{download-BtjXfL3z.js → download-SnJL6oqk.js} +1 -1
  237. package/static/app/assets/{folder-open-DefMpxI2.js → folder-open-CqZeDkjE.js} +1 -1
  238. package/static/app/assets/{hard-drive-BQ8NZVkw.js → hard-drive-j1jJXYYf.js} +1 -1
  239. package/static/app/assets/{index-vtEfYvQY.css → index-BLPb5lmE.css} +1 -1
  240. package/static/app/assets/index-_u5iUHDr.js +10 -0
  241. package/static/app/assets/input-B0lPdRQZ.js +1 -0
  242. package/static/app/assets/link-2-CoFbooHS.js +1 -0
  243. package/static/app/assets/{permissionCopy-BqZ5tsgL.js → permissionCopy-BsyLxtao.js} +1 -1
  244. package/static/app/assets/primitives-DEbN-d6p.js +1 -0
  245. package/static/app/assets/search-BybIWPNd.js +1 -0
  246. package/static/app/assets/{share-2-D5zg_0fY.js → share-2-CVtZ_ewX.js} +1 -1
  247. package/static/app/assets/{shield-alert-B5pZzkUb.js → shield-alert-CBi2GNWM.js} +1 -1
  248. package/static/app/assets/{textarea-nEVIweKY.js → textarea-DNMpB5ih.js} +1 -1
  249. package/static/app/assets/{useFocusTrap-Cm99AHlz.js → useFocusTrap-C83t3GXF.js} +1 -1
  250. package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
  251. package/static/app/assets/{useQuery-Dm__N6bL.js → useQuery-Dcp1OChy.js} +1 -1
  252. package/static/app/assets/utils-BlZr7Pd4.js +4 -0
  253. package/static/app/assets/workspace-jJY4RuAV.js +1 -0
  254. package/static/app/index.html +4 -4
  255. package/static/sw.js +1 -1
  256. package/lattice_brain/graph/_kg_common.py +0 -1331
  257. package/lattice_brain/graph/discovery_index.py +0 -1141
  258. package/lattice_brain/graph/retrieval.py +0 -1120
  259. package/lattice_brain/graph/retrieval_vector.py +0 -1293
  260. package/lattice_brain/ingestion.py +0 -1377
  261. package/lattice_brain/portability.py +0 -1210
  262. package/latticeai/core/agent.py +0 -1412
  263. package/latticeai/core/embedding_providers.py +0 -1196
  264. package/latticeai/core/file_generation.py +0 -1047
  265. package/latticeai/integrations/telegram_bot.py +0 -1390
  266. package/latticeai/models/router.py +0 -1007
  267. package/latticeai/runtime/build_phases.py +0 -1422
  268. package/latticeai/services/brain_intelligence.py +0 -967
  269. package/latticeai/services/memory_service.py +0 -1177
  270. package/latticeai/services/model_runtime.py +0 -1281
  271. package/latticeai/setup/wizard.py +0 -1277
  272. package/static/app/assets/Act-D0HWqtn0.js +0 -1
  273. package/static/app/assets/AdminConsole-D-QDW-A4.js +0 -1
  274. package/static/app/assets/BrainHome-Btns-_TA.js +0 -2
  275. package/static/app/assets/BrainSignals-2dHQNkns.js +0 -1
  276. package/static/app/assets/Capture-CT8v1StE.js +0 -1
  277. package/static/app/assets/CommandPalette-DoLXC2KH.js +0 -1
  278. package/static/app/assets/Library-DDoxFE5c.js +0 -1
  279. package/static/app/assets/ProductFlow-DOYf7JIs.js +0 -1
  280. package/static/app/assets/ReviewCard-COQsqidK.js +0 -3
  281. package/static/app/assets/System-BRllvYXd.js +0 -1
  282. package/static/app/assets/arrow-left-DnyMzss-.js +0 -1
  283. package/static/app/assets/brain-uMb_5hnO.js +0 -1
  284. package/static/app/assets/index-0AvoEBzJ.js +0 -10
  285. package/static/app/assets/input-B_5ZJ9oy.js +0 -1
  286. package/static/app/assets/primitives-CVwew78r.js +0 -1
  287. package/static/app/assets/search-DkhnOKZt.js +0 -1
  288. package/static/app/assets/utils-DcDMoZIe.js +0 -4
  289. package/static/app/assets/workspace-LtRRSKTf.js +0 -1
@@ -1,1377 +0,0 @@
1
- """Unified ingestion pipeline — the single write-side seam into the Knowledge Graph.
2
-
3
- v3.6.0 Knowledge Graph First principle: *no data source bypasses the Knowledge
4
- Graph and no source creates an isolated silo*. Every source — local files,
5
- connected folders, PDFs/Markdown/text/code, web URLs, browser tabs — is
6
- normalized into one :class:`IngestionItem` and pushed through one
7
- :meth:`IngestionPipeline.ingest` entrypoint:
8
-
9
- Source → normalize → content hash → (file | text) ingest → provenance
10
-
11
- The pipeline is deliberately thin. It owns normalization, idempotency reporting,
12
- provenance capture, and — crucially — routing every ingest through the shared
13
- ``dispatch_tool`` lifecycle so ``pre_tool``/``post_tool`` hooks fire on data
14
- ingestion exactly as they do on tool calls. The heavy graph construction lives in
15
- :class:`knowledge_graph.KnowledgeGraphStore` (``ingest_document`` for files,
16
- ``ingest_source`` for text/web), which this module composes rather than
17
- re-implements.
18
-
19
- Web ingestion seam
20
- ------------------
21
- The graph layer never fetches or parses the web. Fetching, rendering,
22
- readability extraction, and parse quality are the responsibility of the
23
- *upstream* capture surfaces (browser extension, tools layer, MCP servers):
24
- they hand this module already-extracted text. :meth:`IngestionPipeline.
25
- ingest_web_page` is the convenience wrapper for that hand-off — it normalizes
26
- ``(url, extracted_text)`` into an ``IngestionItem(source_type="web_url")`` and
27
- routes it through the exact same :meth:`IngestionPipeline.ingest` door as every
28
- other source. If the extracted text is bad, fix the extractor upstream; the
29
- pipeline will not attempt network access or HTML parsing.
30
-
31
- Folder ingestion (:meth:`IngestionPipeline.ingest_folder`) walks a local
32
- directory, honors a gitignore-like ``.latticeignore`` file at the root
33
- (blank lines, ``#`` comments, ``fnmatch`` glob patterns, ``dir/`` suffix for
34
- directories), always skips common noise (``.git``, ``node_modules``,
35
- ``__pycache__``, virtualenvs, ``dist``, hidden entries by default), applies
36
- size/extension filters, and either ingests inline or schedules through the
37
- existing :class:`BackgroundIngestionQueue`.
38
- """
39
-
40
- from __future__ import annotations
41
-
42
- import fnmatch
43
- import hashlib
44
- import os
45
- from dataclasses import dataclass, field
46
- from pathlib import Path
47
- from typing import Any, Dict, Iterable, List, Optional, Tuple
48
-
49
- from .graph.vector_index import DEFAULT_TICK_LIMIT as VECTOR_TICK_LIMIT
50
- from .multimodal import (
51
- AUDIO_EXTENSIONS,
52
- IMAGE_EXTENSIONS,
53
- MODALITY_AUDIO,
54
- MODALITY_IMAGE,
55
- MODALITY_VIDEO,
56
- VIDEO_OUT_OF_SCOPE,
57
- ImageFacts,
58
- MultimodalPorts,
59
- audio_quality_score,
60
- detect_modality,
61
- extract_image_facts,
62
- image_quality_score,
63
- transcribe_audio,
64
- write_image_memory,
65
- )
66
- from .runtime.hooks import dispatch_tool
67
- from .utils import utc_now_iso
68
-
69
- # Source types that arrive as a file on disk (read via ingest_document).
70
- FILE_SOURCE_TYPES = frozenset({"file", "local_file", "upload", "pdf"})
71
- # Source types that arrive as extracted text (read via ingest_source).
72
- TEXT_SOURCE_TYPES = frozenset(
73
- {"web_url", "browser_tab", "text", "markdown", "note", "code", "clipboard"}
74
- )
75
- # Conversational exchanges (read via ingest_message — role/content semantics,
76
- # conversation chaining). v4: chat and MCP messages stop bypassing the
77
- # pipeline, so they carry provenance and fire the hook lifecycle like every
78
- # other source.
79
- CHAT_SOURCE_TYPES = frozenset({"chat_message", "mcp_message"})
80
- # Typed memory records (read via ingest_event → Decision/Experience/Event
81
- # nodes). The Memory System writes through the same door as everything else.
82
- MEMORY_SOURCE_TYPES = frozenset({"decision", "experience", "workspace_event"})
83
- _MEMORY_NODE_TYPES = {"decision": "Decision", "experience": "Experience", "workspace_event": "Event"}
84
-
85
- DEFAULT_MAX_TEXT_BYTES = 5 * 1024 * 1024 # 5 MB of extracted text per item
86
-
87
- # ── Folder ingestion (ingest_folder) filters ─────────────────────────────────
88
- # Directories that are always pruned regardless of .latticeignore.
89
- FOLDER_DEFAULT_SKIP_DIRS = frozenset(
90
- {
91
- ".git",
92
- "node_modules",
93
- "__pycache__",
94
- ".venv",
95
- "venv",
96
- "env",
97
- ".pytest_cache",
98
- ".mypy_cache",
99
- ".ruff_cache",
100
- "dist",
101
- "build",
102
- ".next",
103
- "target",
104
- ".cache",
105
- ".idea",
106
- ".vscode",
107
- }
108
- )
109
- # Extension filter matching FILE_SOURCE_TYPES conventions: text/markdown/code
110
- # are read inline (extracted content → chunks); .pdf routes as source_type
111
- # "pdf" through ingest_document (content extraction is upstream's concern).
112
- FOLDER_TEXT_EXTENSIONS = frozenset(
113
- {".txt", ".md", ".markdown", ".rst", ".csv", ".json", ".yaml", ".yml", ".toml", ".ini"}
114
- )
115
- FOLDER_CODE_EXTENSIONS = frozenset(
116
- {
117
- ".py", ".js", ".ts", ".tsx", ".jsx", ".html", ".css", ".go", ".rs",
118
- ".java", ".c", ".h", ".cpp", ".hpp", ".rb", ".php", ".swift", ".kt",
119
- ".sh", ".sql",
120
- }
121
- )
122
- FOLDER_DOCUMENT_EXTENSIONS = frozenset({".pdf"})
123
- DEFAULT_FOLDER_EXTENSIONS = (
124
- FOLDER_TEXT_EXTENSIONS | FOLDER_CODE_EXTENSIONS | FOLDER_DOCUMENT_EXTENSIONS
125
- )
126
- DEFAULT_MAX_FILE_BYTES = 4_000_000 # matches the local-index text/code budget
127
- LATTICEIGNORE_FILENAME = ".latticeignore"
128
- # Opt-out escape hatch for the post-ingest incremental vector sync.
129
- AUTO_VECTOR_INDEX_ENV = "LATTICEAI_AUTO_VECTOR_INDEX"
130
-
131
- # ── Multi-modal ingestion (v11.1.0 Track 3) ──────────────────────────────────
132
- # Opt-in, default off, on purpose. Turning it on changes what a folder scan
133
- # *stores* (pictures and recordings, with OCR and — if a model is loaded —
134
- # captions and vectors), and that is the user's call, not a default. With the
135
- # flag off every routing decision below is skipped and behaviour is byte-for-
136
- # byte what it was before this release.
137
- ALLOW_MULTIMODAL_ENV = "LATTICEAI_ALLOW_MULTIMODAL"
138
- #: Source types that name a modality outright (a caller who already knows).
139
- IMAGE_SOURCE_TYPES = frozenset({"image", "screenshot", "photo"})
140
- AUDIO_SOURCE_TYPES = frozenset({"audio", "voice_memo", "recording"})
141
- #: Added to the folder-scan allow-list only while multimodal is enabled.
142
- FOLDER_MULTIMODAL_EXTENSIONS = IMAGE_EXTENSIONS | AUDIO_EXTENSIONS
143
- #: Graph node type for a recording. ``NodeType.AUDIO`` normalizes this on the
144
- #: KG v2 write side; the legacy tables keep the label verbatim, which is what
145
- #: every type-aware read (graph view, context sections, doc-gen) matches on.
146
- AUDIO_NODE_TYPE = "Audio"
147
-
148
- # ── Extraction quality heuristics (v9.8.0 A1) ────────────────────────────────
149
- # Pure heuristics over the extracted text — no model calls, no network. The
150
- # score is *advisory*: it never blocks an ingest, it only annotates the result
151
- # so capture surfaces (browser, folder scan) can surface low-quality warnings.
152
- QUALITY_HIGH_THRESHOLD = 0.7
153
- QUALITY_LOW_THRESHOLD = 0.4
154
- QUALITY_LOW_WARNING = "추출 품질이 낮습니다 — 원문 확인을 권장합니다."
155
- _WEB_SOURCE_TYPES = frozenset({"web_url", "browser_tab"})
156
- # Standalone short lines that smell like leftover site chrome (nav/menu/footer).
157
- _BOILERPLATE_LINE_MARKERS = frozenset(
158
- {
159
- "home", "menu", "nav", "navigation", "login", "log in", "sign in",
160
- "sign up", "register", "subscribe", "search", "about", "about us",
161
- "contact", "contact us", "privacy policy", "terms of service",
162
- "cookie policy", "accept cookies", "accept all cookies", "share",
163
- "skip to content", "copyright", "all rights reserved", "sitemap",
164
- "back to top", "footer", "read more", "next", "previous",
165
- }
166
- )
167
-
168
-
169
- def _quality_level(score: float) -> str:
170
- if score >= QUALITY_HIGH_THRESHOLD:
171
- return "high"
172
- if score >= QUALITY_LOW_THRESHOLD:
173
- return "medium"
174
- return "low"
175
-
176
-
177
- def assess_extraction_quality(
178
- text: Optional[str],
179
- *,
180
- source_type: Optional[str] = None,
181
- upstream_confidence: Optional[Any] = None,
182
- ) -> Dict[str, Any]:
183
- """Score extracted text 0..1 with reasons (pure heuristic, deterministic).
184
-
185
- Signals: text length, whitespace ratio, character/word diversity
186
- (repetition), sentence structure, and — for web sources — leftover
187
- nav/menu boilerplate. When the upstream extractor supplies its own
188
- confidence (``upstream_confidence``), that value wins verbatim: the
189
- extractor saw the raw document, this function only sees its output.
190
- """
191
- if upstream_confidence is not None:
192
- try:
193
- score = max(0.0, min(1.0, float(upstream_confidence)))
194
- except (TypeError, ValueError):
195
- score = None
196
- if score is not None:
197
- return {
198
- "score": round(score, 4),
199
- "level": _quality_level(score),
200
- "reasons": ["upstream_confidence"],
201
- }
202
-
203
- raw = str(text or "")
204
- stripped = raw.strip()
205
- if not stripped:
206
- return {"score": 0.0, "level": "low", "reasons": ["empty_text"]}
207
-
208
- reasons: List[str] = []
209
- length = len(stripped)
210
- sample = stripped[:4000]
211
- lines = [ln.strip() for ln in stripped.splitlines() if ln.strip()]
212
- words = stripped.split()
213
-
214
- # 1) Length — very short extractions rarely carry recall value.
215
- if length < 40:
216
- length_factor = 0.35
217
- reasons.append("very_short_text")
218
- elif length < 120:
219
- length_factor = 0.6
220
- reasons.append("short_text")
221
- elif length < 300:
222
- length_factor = 0.85
223
- else:
224
- length_factor = 1.0
225
-
226
- # 2) Sentence structure — prose has sentence-ending punctuation.
227
- sentence_marks = sum(sample.count(mark) for mark in (".", "!", "?", "…", "。", "!", "?"))
228
- if sentence_marks > 0:
229
- structure_factor = 1.0
230
- elif length < 200:
231
- structure_factor = 0.75 # titles/snippets legitimately lack periods
232
- else:
233
- structure_factor = 0.45
234
- reasons.append("no_sentence_structure")
235
-
236
- # 3) Diversity — repeated characters/lines/words indicate extraction junk.
237
- diversity_factor = 1.0
238
- distinct_chars = len(set(sample.lower()))
239
- if distinct_chars < 10:
240
- diversity_factor *= 0.2
241
- reasons.append("low_character_diversity")
242
- elif distinct_chars < 20:
243
- diversity_factor *= 0.7
244
- if len(lines) >= 6:
245
- top_count = max(lines.count(ln) for ln in set(lines))
246
- if top_count >= max(3, len(lines) // 4):
247
- diversity_factor *= 0.5
248
- reasons.append("repetitive_lines")
249
- if len(words) >= 30 and (len(set(w.lower() for w in words)) / len(words)) < 0.25:
250
- diversity_factor *= 0.5
251
- reasons.append("repetitive_words")
252
-
253
- # 4) Cleanliness — whitespace floods, fragmented lines, site chrome.
254
- cleanliness_factor = 1.0
255
- whitespace_ratio = sum(1 for ch in raw if ch.isspace()) / max(1, len(raw))
256
- if whitespace_ratio > 0.45:
257
- cleanliness_factor *= 0.6
258
- reasons.append("high_whitespace_ratio")
259
- if len(lines) >= 8:
260
- short_lines = sum(1 for ln in lines if len(ln.split()) <= 3)
261
- if short_lines / len(lines) > 0.6:
262
- cleanliness_factor *= 0.6
263
- reasons.append("fragmented_lines")
264
- boilerplate_hits = sum(
265
- 1 for ln in lines if ln.lower().strip(" .:>|•·-–—*") in _BOILERPLATE_LINE_MARKERS
266
- )
267
- if lines and boilerplate_hits >= 3 and (boilerplate_hits / len(lines)) > 0.2:
268
- cleanliness_factor *= 0.35
269
- if str(source_type or "").lower() in _WEB_SOURCE_TYPES:
270
- reasons.append("nav_menu_remnants")
271
- else:
272
- reasons.append("boilerplate_markers")
273
-
274
- score = length_factor * structure_factor * diversity_factor * cleanliness_factor
275
- score = max(0.0, min(1.0, score))
276
- if not reasons:
277
- reasons.append("clean_extraction")
278
- return {"score": round(score, 4), "level": _quality_level(score), "reasons": reasons}
279
-
280
-
281
- # ── capture quality CTA (backlog #9, review §7.2 C) ──────────────────────────
282
- # Structured verdict over the same extraction-quality schema the rest of the
283
- # pipeline uses, so capture surfaces (browser extension, read-url) can render
284
- # an honest "this capture is thin" CTA instead of silently storing junk.
285
- CAPTURE_SUGGESTIONS_THIN = ["recapture", "paste_manually", "highlight_source"]
286
- _CAPTURE_REASON_LABELS = {
287
- "empty_text": "추출된 본문이 비어 있습니다",
288
- "very_short_text": "추출된 본문이 매우 짧습니다",
289
- "short_text": "추출된 본문이 짧습니다",
290
- "no_sentence_structure": "문장 구조가 거의 없습니다",
291
- "low_character_diversity": "반복 문자가 대부분입니다",
292
- "repetitive_lines": "같은 줄이 반복됩니다",
293
- "repetitive_words": "같은 단어가 반복됩니다",
294
- "high_whitespace_ratio": "공백이 지나치게 많습니다",
295
- "fragmented_lines": "줄이 잘게 조각나 있습니다",
296
- "nav_menu_remnants": "메뉴/내비게이션 잔여물이 많습니다",
297
- "boilerplate_markers": "상용구 텍스트가 많습니다",
298
- "no_extracted_text": "추출된 텍스트가 없습니다",
299
- }
300
-
301
-
302
- def capture_quality_verdict(
303
- extraction_quality: Optional[Dict[str, Any]],
304
- *,
305
- source_type: Optional[str] = None,
306
- ) -> Dict[str, Any]:
307
- """Structured CTA verdict from a pipeline ``extraction_quality`` dict.
308
-
309
- ``{"status": "thin"|"ok", "reason": str|None, "suggestions": [...],
310
- "score": float|None, "level": str|None}``. ``thin`` (level == "low", the
311
- same threshold as the ingest warning) carries actionable suggestions —
312
- ``recapture`` / ``paste_manually`` / ``highlight_source`` — so the UI can
313
- offer the user a way to fix the capture instead of hiding the problem.
314
- Deterministic and never raises; ``None`` input yields an honest ``thin``.
315
- """
316
- if not isinstance(extraction_quality, dict):
317
- return {
318
- "status": "thin",
319
- "reason": _CAPTURE_REASON_LABELS["no_extracted_text"],
320
- "reason_codes": ["no_extracted_text"],
321
- "suggestions": list(CAPTURE_SUGGESTIONS_THIN),
322
- "score": None,
323
- "level": None,
324
- }
325
- level = str(extraction_quality.get("level") or "")
326
- score = extraction_quality.get("score")
327
- reasons = [str(item) for item in (extraction_quality.get("reasons") or [])]
328
- thin = level == "low"
329
- reason = None
330
- if thin:
331
- labeled = [
332
- _CAPTURE_REASON_LABELS[code]
333
- for code in reasons
334
- if code in _CAPTURE_REASON_LABELS
335
- ]
336
- reason = "; ".join(labeled) if labeled else QUALITY_LOW_WARNING
337
- return {
338
- "status": "thin" if thin else "ok",
339
- "reason": reason,
340
- "reason_codes": reasons if thin else [],
341
- "suggestions": list(CAPTURE_SUGGESTIONS_THIN) if thin else [],
342
- "score": score,
343
- "level": level or None,
344
- }
345
-
346
-
347
- def _load_latticeignore(root: Path) -> List[str]:
348
- """Parse ``root/.latticeignore`` → glob patterns (gitignore-like subset)."""
349
- ignore_file = root / LATTICEIGNORE_FILENAME
350
- patterns: List[str] = []
351
- if not ignore_file.is_file():
352
- return patterns
353
- try:
354
- lines = ignore_file.read_text(encoding="utf-8", errors="ignore").splitlines()
355
- except OSError:
356
- return patterns
357
- for raw in lines:
358
- line = raw.strip()
359
- if not line or line.startswith("#"):
360
- continue
361
- patterns.append(line)
362
- return patterns
363
-
364
-
365
- def _matches_ignore(
366
- rel_posix: str, name: str, *, is_dir: bool, patterns: Iterable[str]
367
- ) -> bool:
368
- """fnmatch-based .latticeignore matching.
369
-
370
- - ``pattern/`` matches directories only (files under it never appear
371
- because ignored directories are pruned during the walk).
372
- - Patterns match against both the root-relative posix path and the
373
- basename, so ``*.log`` and ``docs/draft.md`` both behave as expected.
374
- """
375
- for raw in patterns:
376
- pattern = raw
377
- if pattern.endswith("/"):
378
- if not is_dir:
379
- continue
380
- pattern = pattern.rstrip("/")
381
- pattern = pattern.lstrip("/")
382
- if not pattern:
383
- continue
384
- if fnmatch.fnmatch(rel_posix, pattern) or fnmatch.fnmatch(name, pattern):
385
- return True
386
- return False
387
-
388
-
389
- # Background job scheduling + progress lives in its own module (v9.9.6):
390
- # the pipeline owns "ingest one item", the queue owns "schedule many and
391
- # report progress". Re-exported so every existing import keeps working.
392
- from .ingestion_jobs import ( # noqa: E402,F401 — re-export for existing importers
393
- JOB_ERRORS_CAP,
394
- BackgroundIngestionJob,
395
- BackgroundIngestionQueue,
396
- )
397
- from .quiet import ( # noqa: E402 — imported after the module constants it depends on
398
- quiet, # noqa: E402 — imported after the module constants it depends on
399
- )
400
-
401
-
402
- @dataclass
403
- class IngestionItem:
404
- """A single thing to ingest, normalized across every source type."""
405
-
406
- source_type: str
407
- title: Optional[str] = None
408
- text: Optional[str] = None # text/web sources
409
- path: Optional[str] = None # file sources
410
- source_uri: Optional[str] = None
411
- mime_type: Optional[str] = None
412
- owner: Optional[str] = None
413
- workspace_id: Optional[str] = None
414
- permissions: Optional[Dict[str, Any]] = None
415
- captured_at: Optional[str] = None
416
- modified_at: Optional[str] = None
417
- conversation_id: Optional[str] = None
418
- agent_used: Optional[str] = None
419
- metadata: Dict[str, Any] = field(default_factory=dict)
420
-
421
-
422
- @dataclass
423
- class IngestionResult:
424
- """The outcome of one ingestion, including provenance and idempotency."""
425
-
426
- status: str # ok | unavailable | blocked | failed
427
- source_type: str
428
- node_id: Optional[str] = None
429
- source_node_id: Optional[str] = None
430
- content_hash: Optional[str] = None
431
- title: Optional[str] = None
432
- chunk_ids: List[str] = field(default_factory=list)
433
- chunk_count: int = 0
434
- duplicate: bool = False
435
- embedded: bool = False
436
- indexing_status: str = "pending" # indexed | skipped | failed | pending
437
- provenance_id: Optional[str] = None
438
- detail: Optional[str] = None
439
- # v9.8.0 additive quality fields — advisory only, never gate behavior.
440
- extraction_quality: Optional[Dict[str, Any]] = None
441
- warnings: List[str] = field(default_factory=list)
442
- quality_gate: Optional[Dict[str, Any]] = None
443
-
444
- def as_dict(self) -> Dict[str, Any]:
445
- payload: Dict[str, Any] = {
446
- "status": self.status,
447
- "source_type": self.source_type,
448
- "node_id": self.node_id,
449
- "source_node_id": self.source_node_id,
450
- "content_hash": self.content_hash,
451
- "title": self.title,
452
- "chunk_ids": self.chunk_ids,
453
- "chunk_count": self.chunk_count,
454
- "duplicate": self.duplicate,
455
- "embedded": self.embedded,
456
- "indexing_status": self.indexing_status,
457
- "provenance_id": self.provenance_id,
458
- "detail": self.detail,
459
- }
460
- # Additive keys only when populated so pre-v9.8 payloads are unchanged.
461
- if self.extraction_quality is not None:
462
- payload["extraction_quality"] = self.extraction_quality
463
- if self.warnings:
464
- payload["warnings"] = list(self.warnings)
465
- if self.quality_gate is not None:
466
- payload["quality_gate"] = self.quality_gate
467
- return payload
468
-
469
-
470
- class IngestionPipeline:
471
- """Single normalized entrypoint that feeds every source into the graph."""
472
-
473
- def __init__(
474
- self,
475
- knowledge_graph: Any,
476
- *,
477
- hooks: Any = None,
478
- enable_graph: bool = True,
479
- audit: Optional[Any] = None,
480
- max_text_bytes: int = DEFAULT_MAX_TEXT_BYTES,
481
- pipeline_name: str = "unified-ingestion",
482
- bg_queue: Optional[BackgroundIngestionQueue] = None,
483
- auto_vector_index: bool = True,
484
- allow_multimodal: bool = False,
485
- multimodal: Optional[MultimodalPorts] = None,
486
- ) -> None:
487
- self._kg = knowledge_graph
488
- self._hooks = hooks
489
- self._enable = bool(enable_graph)
490
- self._audit = audit
491
- self._max_text_bytes = int(max_text_bytes)
492
- self._pipeline_name = pipeline_name
493
- # Background job state lives in the graph database by default, so a
494
- # restart resumes from the last completed item instead of replaying the
495
- # whole corpus. A store without a usable ``db_path`` (mocks, disabled
496
- # graph) degrades to the historical in-memory queue, which reports
497
- # itself as non-durable through ``BackgroundIngestionQueue.describe()``.
498
- self._bg_queue = bg_queue or BackgroundIngestionQueue(
499
- db_path=getattr(knowledge_graph, "db_path", None)
500
- )
501
- # Incremental vector sync after each successful non-duplicate ingest.
502
- # Constructor opt-out AND env opt-out (LATTICEAI_AUTO_VECTOR_INDEX=0)
503
- # both disable it; a vector failure never fails the ingest.
504
- env_flag = os.getenv(AUTO_VECTOR_INDEX_ENV, "1").strip().lower() not in {
505
- "0", "false", "no", "off",
506
- }
507
- self._auto_vector_index = bool(auto_vector_index) and env_flag
508
- # Multi-modal routing. Off unless the caller asks for it *or* the env
509
- # flag is set — the env is the escape hatch for an install that has no
510
- # code path to the constructor (CLI, background worker).
511
- self._allow_multimodal = bool(allow_multimodal) or os.getenv(
512
- ALLOW_MULTIMODAL_ENV, "0"
513
- ).strip().lower() in {"1", "true", "yes", "on"}
514
- self._multimodal = multimodal or MultimodalPorts()
515
-
516
- def available(self) -> bool:
517
- return self._enable and self._kg is not None
518
-
519
- def multimodal_status(self) -> Dict[str, Any]:
520
- """What this pipeline will do with a picture or a recording, honestly.
521
-
522
- ``enabled`` is the flag; the rest is which model-backed capabilities
523
- were actually injected. Video is listed as unsupported rather than
524
- omitted, so the surface can say *why* a ``.mov`` was refused.
525
- """
526
- return {
527
- "enabled": self._allow_multimodal,
528
- "image": self._allow_multimodal,
529
- "audio": self._allow_multimodal,
530
- "video": False,
531
- "video_detail": VIDEO_OUT_OF_SCOPE,
532
- **self._multimodal.describe(),
533
- }
534
-
535
- # ── public API ───────────────────────────────────────────────────────────
536
- def ingest(self, item: IngestionItem, *, user_email: Optional[str] = None) -> IngestionResult:
537
- """Normalize, hash, route through dispatch_tool, and record provenance."""
538
- source_type = str(item.source_type or "text").strip().lower()
539
- if not self.available():
540
- return IngestionResult(
541
- status="unavailable", source_type=source_type,
542
- indexing_status="skipped",
543
- detail="Knowledge Graph is disabled (LATTICEAI_ENABLE_GRAPH).",
544
- )
545
-
546
- # Modality routing is a no-op while the flag is off: ``modality`` stays
547
- # "text" and every branch below behaves exactly as it did before.
548
- modality = self._modality_for(item, source_type)
549
- if modality == MODALITY_VIDEO:
550
- return IngestionResult(
551
- status="unavailable", source_type=source_type,
552
- indexing_status="skipped", detail=VIDEO_OUT_OF_SCOPE,
553
- )
554
-
555
- captured_at = item.captured_at or utc_now_iso()
556
- owner = item.owner or user_email
557
- tool_name = f"kg_ingest.{source_type}"
558
- # Only the keys are read by the hook payload, so this dict is safe/cheap.
559
- args = {
560
- "source_type": source_type,
561
- "source_uri": item.source_uri,
562
- "owner": owner,
563
- "workspace_id": item.workspace_id,
564
- }
565
-
566
- def _run() -> Dict[str, Any]:
567
- if source_type in CHAT_SOURCE_TYPES:
568
- return self._ingest_chat(item, source_type=source_type, owner=owner)
569
- if source_type in MEMORY_SOURCE_TYPES:
570
- return self._ingest_memory_record(item, source_type=source_type, owner=owner)
571
- if modality == MODALITY_IMAGE:
572
- return self._ingest_image(item, source_type=source_type, owner=owner, captured_at=captured_at)
573
- if modality == MODALITY_AUDIO:
574
- return self._ingest_audio(item, source_type=source_type, owner=owner, captured_at=captured_at)
575
- if source_type in FILE_SOURCE_TYPES or (item.path and not item.text):
576
- return self._ingest_file(item, source_type=source_type, owner=owner, captured_at=captured_at)
577
- return self._ingest_text(item, source_type=source_type, owner=owner, captured_at=captured_at)
578
-
579
- # v9.8.0 observation-only quality gate: computed *before* the write so
580
- # the search never matches the node we are about to create. It is
581
- # recorded on the result and never skips an ingest (behavior unchanged).
582
- quality_text = self._extractable_text(item)
583
- quality_gate = self._observe_quality_gate(
584
- item, source_type=source_type, text=quality_text,
585
- )
586
-
587
- try:
588
- raw = dispatch_tool(
589
- self._hooks, tool_name, args, _run,
590
- user_email=user_email, workspace_id=item.workspace_id, source="ingestion",
591
- )
592
- except PermissionError as exc:
593
- return IngestionResult(
594
- status="blocked", source_type=source_type,
595
- indexing_status="skipped", detail=str(exc),
596
- )
597
- except FileNotFoundError as exc:
598
- return IngestionResult(
599
- status="failed", source_type=source_type,
600
- indexing_status="failed", detail=str(exc),
601
- )
602
- except Exception as exc: # noqa: BLE001 — surface as a failed result, never crash the caller
603
- return IngestionResult(
604
- status="failed", source_type=source_type,
605
- indexing_status="failed", detail=str(exc),
606
- )
607
-
608
- node_id = raw.get("node_id")
609
- content_hash = raw.get("content_hash") or raw.get("sha256")
610
- chunk_ids = list(raw.get("chunk_ids") or [])
611
- title = raw.get("title") or item.title
612
-
613
- # Incremental vector-index sync (opt-in via auto_vector_index +
614
- # LATTICEAI_AUTO_VECTOR_INDEX). Exception-safe by contract: the graph
615
- # write above already landed, so a vector failure only downgrades
616
- # indexing_status to "pending" — index_status()/rebuild_vector_index()
617
- # discover the same node as backlog and pick it up later.
618
- indexing_status = "indexed"
619
- vector_detail: Optional[str] = None
620
- if node_id and self._auto_vector_index and not bool(raw.get("duplicate")):
621
- indexing_status, vector_detail = self._sync_vector_index(node_id)
622
- embedded = bool(self._kg.node_is_embedded(node_id)) if node_id else False
623
-
624
- # Provenance capture must never turn an already-persisted ingest into a
625
- # caller-visible failure: the graph write above succeeded, so a broken
626
- # provenance table degrades the result instead of raising.
627
- provenance_detail: Optional[str] = None
628
- try:
629
- prov = self._kg.record_provenance(
630
- node_id=node_id,
631
- source_type=source_type,
632
- pipeline=self._pipeline_name,
633
- source_uri=item.source_uri,
634
- content_hash=content_hash,
635
- title=title,
636
- owner=owner,
637
- workspace_id=item.workspace_id,
638
- captured_at=captured_at,
639
- modified_at=item.modified_at,
640
- embedded=embedded,
641
- linked=bool(raw.get("source_node_id")),
642
- duplicate=bool(raw.get("duplicate")),
643
- agent_used=item.agent_used,
644
- chunk_count=len(chunk_ids),
645
- permissions=item.permissions,
646
- metadata=item.metadata,
647
- )
648
- except Exception as exc: # noqa: BLE001 — the ingest itself already landed
649
- prov = {}
650
- provenance_detail = f"provenance capture failed: {exc}"
651
- if self._audit is not None:
652
- try:
653
- self._audit(
654
- "kg_ingest",
655
- {
656
- "source_type": source_type, "node_id": node_id,
657
- "content_hash": content_hash, "duplicate": bool(raw.get("duplicate")),
658
- },
659
- user_email,
660
- )
661
- except Exception: # noqa: BLE001 — audit must never break ingestion
662
- quiet()
663
-
664
- # A modality-aware door scores its own extraction (a picture's quality
665
- # is "how much of it can be retrieved", not "does the text read well"),
666
- # so its verdict wins. Text/file doors never set the key and keep the
667
- # historical scoring untouched.
668
- extraction_quality = raw.get("extraction_quality") or self._assess_item_quality(
669
- item, source_type=source_type, text=quality_text, chunk_ids=chunk_ids,
670
- )
671
- warnings: List[str] = []
672
- if extraction_quality is not None and extraction_quality.get("level") == "low":
673
- warnings.append(QUALITY_LOW_WARNING)
674
-
675
- details = [d for d in (provenance_detail, vector_detail) if d]
676
- return IngestionResult(
677
- status="ok",
678
- source_type=source_type,
679
- node_id=node_id,
680
- source_node_id=raw.get("source_node_id"),
681
- content_hash=content_hash,
682
- title=title,
683
- chunk_ids=chunk_ids,
684
- chunk_count=len(chunk_ids),
685
- duplicate=bool(raw.get("duplicate")),
686
- embedded=embedded,
687
- indexing_status=indexing_status,
688
- provenance_id=prov.get("id"),
689
- detail="; ".join(details) if details else None,
690
- extraction_quality=extraction_quality,
691
- warnings=warnings,
692
- quality_gate=quality_gate,
693
- )
694
-
695
- # ── extraction quality (v9.8.0 A1 — advisory, never gates) ───────────────
696
- @staticmethod
697
- def _extractable_text(item: IngestionItem) -> Optional[str]:
698
- """Best available extracted text for quality scoring/gating."""
699
- if item.text is not None:
700
- return item.text
701
- extracted = (item.metadata or {}).get("extracted")
702
- if isinstance(extracted, dict):
703
- content = extracted.get("content") or extracted.get("text")
704
- if content is not None:
705
- return str(content)
706
- return None
707
-
708
- @staticmethod
709
- def _upstream_confidence(item: IngestionItem) -> Optional[Any]:
710
- """Upstream extractor confidence, if the capture surface supplied one."""
711
- meta = item.metadata or {}
712
- extracted = meta.get("extracted")
713
- if isinstance(extracted, dict) and extracted.get("confidence") is not None:
714
- return extracted.get("confidence")
715
- if meta.get("extraction_confidence") is not None:
716
- return meta.get("extraction_confidence")
717
- return None
718
-
719
- def _assess_item_quality(
720
- self,
721
- item: IngestionItem,
722
- *,
723
- source_type: str,
724
- text: Optional[str],
725
- chunk_ids: List[str],
726
- ) -> Optional[Dict[str, Any]]:
727
- """Quality annotation for document-like sources (not chat/memory)."""
728
- if source_type in CHAT_SOURCE_TYPES or source_type in MEMORY_SOURCE_TYPES:
729
- return None
730
- confidence = self._upstream_confidence(item)
731
- if text is not None or confidence is not None:
732
- return assess_extraction_quality(
733
- text, source_type=source_type, upstream_confidence=confidence,
734
- )
735
- # File door without inline extraction (e.g. PDF): the pipeline never saw
736
- # the text, so score honestly from the chunk output instead of guessing.
737
- if chunk_ids:
738
- return {
739
- "score": 0.5,
740
- "level": "medium",
741
- "reasons": ["content_extracted_upstream_not_scored"],
742
- }
743
- return {"score": 0.0, "level": "low", "reasons": ["no_extracted_text"]}
744
-
745
- def _observe_quality_gate(
746
- self,
747
- item: IngestionItem,
748
- *,
749
- source_type: str,
750
- text: Optional[str],
751
- ) -> Optional[Dict[str, Any]]:
752
- """Observation-mode ``gate_ingest_candidate`` wiring.
753
-
754
- Records what the proactive gate *would* decide (ingest /
755
- skip_duplicate / review) without ever acting on it. Any failure —
756
- import, search, gate — yields ``None``; the ingest proceeds untouched.
757
- """
758
- if source_type in CHAT_SOURCE_TYPES or source_type in MEMORY_SOURCE_TYPES:
759
- return None
760
- body = str(text or "").strip()
761
- if not body:
762
- return None
763
- try:
764
- from .graph.proactive import gate_ingest_candidate
765
- except Exception: # noqa: BLE001 — optional observation, never required
766
- return None
767
-
768
- def _search(query: str) -> Any:
769
- snippet = str(query or "")[:400]
770
- try:
771
- if item.workspace_id:
772
- return self._kg.search(
773
- snippet, 20, allowed_workspaces={item.workspace_id},
774
- )
775
- return self._kg.search(snippet, 20)
776
- except TypeError:
777
- # Older store without workspace-scoped search.
778
- return self._kg.search(snippet, 20)
779
-
780
- try:
781
- gate = gate_ingest_candidate(body, _search)
782
- except Exception: # noqa: BLE001 — observation must never fail the ingest
783
- return None
784
- parts = [str(gate.get("reason") or "")]
785
- if gate.get("similarity") is not None:
786
- parts.append(f"similarity={gate.get('similarity')}")
787
- if gate.get("match_id"):
788
- parts.append(f"match={gate.get('match_id')}")
789
- return {
790
- "action": str(gate.get("action") or "review"),
791
- "detail": "; ".join(p for p in parts if p),
792
- }
793
-
794
- def _queue_pending_embed(self, node_id: str, detail: str) -> bool:
795
- """Hand a node the inline sync could not embed to the background queue.
796
-
797
- Before v11.1.0 ``indexing_status="pending"`` was the end of the story:
798
- honest, but nobody was coming back for it, so the node stayed
799
- unsearchable until a human ran a rebuild. The durable queue is who
800
- comes back. A store without one (older stores, mocks) just keeps the
801
- old behaviour — the node is still visible as ``index_status`` backlog.
802
- """
803
- queue = getattr(self._kg, "vector_queue", None)
804
- if queue is None:
805
- return False
806
- try:
807
- return bool(queue.schedule(node_id, detail=detail))
808
- except Exception: # noqa: BLE001 — queueing must never fail an ingest
809
- quiet()
810
- return False
811
-
812
- def _sync_vector_index(self, node_id: str) -> Tuple[str, Optional[str]]:
813
- """Best-effort incremental vector sync → (indexing_status, detail).
814
-
815
- Any failure — missing method on older stores, embedding provider down,
816
- storage error — yields ``("pending", detail)`` so a later
817
- ``rebuild_vector_index`` run picks the node up from the backlog, and
818
- the node is queued for background embedding so that pickup happens on
819
- its own.
820
- """
821
- sync = getattr(self._kg, "index_node_incremental", None)
822
- if not callable(sync):
823
- # Older store without the incremental path: the write-side already
824
- # embeds inline, so nothing extra to do.
825
- return "indexed", None
826
- try:
827
- outcome = sync(node_id) or {}
828
- except Exception as exc: # noqa: BLE001 — vector sync must never fail the ingest
829
- return "pending", self._pending_detail(node_id, f"vector index sync failed: {exc}")
830
- if str(outcome.get("status") or "") == "failed":
831
- reason = outcome.get("detail") or "unknown error"
832
- return "pending", self._pending_detail(
833
- node_id, f"vector index sync failed: {reason}"
834
- )
835
- return "indexed", None
836
-
837
- def _pending_detail(self, node_id: str, reason: str) -> str:
838
- """``reason``, plus whether a background retry was actually scheduled."""
839
- if self._queue_pending_embed(node_id, reason):
840
- return f"{reason}; queued for background embedding"
841
- return reason
842
-
843
- def drain_vector_queue(self, limit: int = VECTOR_TICK_LIMIT) -> Dict[str, Any]:
844
- """Run one background-embedding tick over the store's pending backlog.
845
-
846
- Deliberately caller-driven (a scheduler, a CLI, a test) rather than a
847
- thread this pipeline owns: the queue is durable, so "who runs it" is a
848
- deployment decision, not a property of having ingested something.
849
- """
850
- queue = getattr(self._kg, "vector_queue", None)
851
- if queue is None:
852
- return {
853
- "claimed": 0,
854
- "indexed": 0,
855
- "retried": 0,
856
- "failed": 0,
857
- "detail": "this store has no background vector queue",
858
- }
859
- return dict(queue.tick(limit))
860
-
861
- # --- Large candidate #1: background / incremental scheduling (slice) ---
862
- def schedule_background(
863
- self,
864
- items: List[IngestionItem],
865
- *,
866
- incremental: bool = True,
867
- user_email: Optional[str] = None,
868
- ) -> BackgroundIngestionJob:
869
- """Schedule items for background incremental indexing.
870
-
871
- Returns a job handle. Actual execution can be driven by caller
872
- (or future worker) calling pipeline.ingest on each — or through
873
- :meth:`run_background_job`. This seam enables large-corpus scale
874
- without blocking user requests.
875
- """
876
- job = self._bg_queue.schedule(items, incremental=incremental, user_email=user_email)
877
- # mark initial status on results concept (jobs track)
878
- return job
879
-
880
- def get_background_job(self, job_id: str) -> Optional[BackgroundIngestionJob]:
881
- return self._bg_queue.get(job_id)
882
-
883
- def list_background_jobs(self, limit: int = 20) -> List[Dict[str, Any]]:
884
- """Recent jobs (newest first) in the frozen ``/api/ingestion`` schema."""
885
- return [job.as_dict() for job in self._bg_queue.list_recent(limit=limit)]
886
-
887
- def run_background_job(
888
- self, job_id: str, *, user_email: Optional[str] = None
889
- ) -> Dict[str, Any]:
890
- """Execute a queued/interrupted job's remaining items.
891
-
892
- Per-item errors are recorded (capped) and never abort the job. The
893
- final status is ``completed`` (all done), ``partial`` (some done),
894
- or ``failed`` (nothing done). Already-completed items are skipped, so
895
- the same method safely powers both first-run and resume.
896
- """
897
- job = self._bg_queue.get(job_id)
898
- if job is None:
899
- return {"status": "not_found", "job_id": job_id}
900
- if job.status == "running":
901
- return job.as_dict()
902
- return self._execute_background_job(job, user_email=user_email)
903
-
904
- def resume_background_job(
905
- self, job_id: str, *, user_email: Optional[str] = None
906
- ) -> Dict[str, Any]:
907
- """Resume an interrupted/partial/failed job from its remaining items."""
908
- return self.run_background_job(job_id, user_email=user_email)
909
-
910
- def _execute_background_job(
911
- self, job: BackgroundIngestionJob, *, user_email: Optional[str] = None
912
- ) -> Dict[str, Any]:
913
- job.status = "running"
914
- # Retried items get a fresh verdict: reset failure state for this run.
915
- job.failed = 0
916
- job.errors = []
917
- job.touch()
918
- self._bg_queue.save(job)
919
- runner_email = user_email or job.user_email
920
- for index in job.remaining_indices():
921
- item = job.items[index]
922
- try:
923
- result = self.ingest(item, user_email=runner_email or item.owner)
924
- status, detail = result.status, result.detail
925
- except Exception as exc: # noqa: BLE001 — per-item isolation: keep going
926
- status, detail = "failed", str(exc)
927
- if status == "ok":
928
- job.done_indices.add(index)
929
- else:
930
- job.record_error(index, item, detail or status)
931
- job.processed = len(job.done_indices)
932
- job.touch()
933
- # Checkpoint per item: a crash here must cost at most the item in
934
- # flight, never the whole job's progress. One small UPDATE against
935
- # an ingest (parse + chunk + embed) is noise.
936
- self._bg_queue.save(job)
937
- job.processed = len(job.done_indices)
938
- if job.total == 0 or job.processed >= job.total:
939
- job.status = "completed"
940
- elif job.processed > 0:
941
- job.status = "partial"
942
- else:
943
- job.status = "failed"
944
- job.touch()
945
- self._bg_queue.save(job)
946
- return job.as_dict()
947
-
948
- def ingest_web_page(
949
- self,
950
- url: str,
951
- extracted_text: str,
952
- *,
953
- title: Optional[str] = None,
954
- metadata: Optional[Dict[str, Any]] = None,
955
- owner: Optional[str] = None,
956
- workspace_id: Optional[str] = None,
957
- captured_at: Optional[str] = None,
958
- user_email: Optional[str] = None,
959
- ) -> IngestionResult:
960
- """Ingest an *already-extracted* web page (see module docstring seam).
961
-
962
- Fetching/parsing is upstream's responsibility (browser extension /
963
- tools layer); this wrapper only normalizes ``(url, extracted_text)``
964
- into an ``IngestionItem(source_type="web_url")`` and routes it through
965
- the standard :meth:`ingest` door.
966
- """
967
- url = str(url or "").strip()
968
- if not url:
969
- return IngestionResult(
970
- status="failed", source_type="web_url",
971
- indexing_status="skipped", detail="url required",
972
- )
973
- text = str(extracted_text or "")
974
- if not text.strip():
975
- return IngestionResult(
976
- status="failed", source_type="web_url",
977
- indexing_status="skipped",
978
- detail=(
979
- "extracted_text required — the graph layer does not fetch or "
980
- "parse the web; extraction happens upstream."
981
- ),
982
- )
983
- item = IngestionItem(
984
- source_type="web_url",
985
- title=title or url,
986
- text=text,
987
- source_uri=url,
988
- owner=owner,
989
- workspace_id=workspace_id,
990
- captured_at=captured_at,
991
- metadata=dict(metadata or {}),
992
- )
993
- return self.ingest(item, user_email=user_email or owner)
994
-
995
- def ingest_folder(
996
- self,
997
- root_path: Any,
998
- *,
999
- recursive: bool = True,
1000
- background: bool = False,
1001
- extensions: Optional[Iterable[str]] = None,
1002
- max_file_bytes: int = DEFAULT_MAX_FILE_BYTES,
1003
- include_hidden: bool = False,
1004
- max_files: int = 1000,
1005
- max_errors: int = 25,
1006
- owner: Optional[str] = None,
1007
- workspace_id: Optional[str] = None,
1008
- user_email: Optional[str] = None,
1009
- ) -> Dict[str, Any]:
1010
- """Walk ``root_path`` and ingest every eligible file through the pipeline.
1011
-
1012
- Filtering, in order: hard skip-list directories (``.git`` …), hidden
1013
- entries (unless ``include_hidden``), root ``.latticeignore`` patterns
1014
- (fnmatch globs; ``dir/`` suffix prunes directories), extension
1015
- allow-list, then ``max_file_bytes``. Text/code files are read inline so
1016
- their content is chunked; ``.pdf`` routes through the file door without
1017
- inline extraction.
1018
-
1019
- ``background=True`` schedules the built items on the existing
1020
- :class:`BackgroundIngestionQueue` instead of ingesting inline.
1021
- Returns a summary dict with counts and per-file errors (capped at
1022
- ``max_errors``).
1023
- """
1024
- summary: Dict[str, Any] = {
1025
- "root": str(root_path),
1026
- "recursive": bool(recursive),
1027
- "background": bool(background),
1028
- "scanned": 0,
1029
- "matched": 0,
1030
- "ingested": 0,
1031
- "duplicate": 0,
1032
- "failed": 0,
1033
- "skipped": {"ignored": 0, "extension": 0, "too_large": 0, "hidden": 0},
1034
- "truncated": False,
1035
- "errors": [],
1036
- }
1037
- try:
1038
- root = Path(root_path).expanduser()
1039
- except TypeError:
1040
- summary.update(status="failed", detail=f"invalid root path: {root_path!r}")
1041
- return summary
1042
- if not root.is_dir():
1043
- summary.update(status="failed", detail=f"not a directory: {root}")
1044
- return summary
1045
- if not self.available():
1046
- summary.update(
1047
- status="unavailable",
1048
- detail="Knowledge Graph is disabled (LATTICEAI_ENABLE_GRAPH).",
1049
- )
1050
- return summary
1051
- summary["root"] = str(root)
1052
- max_files = max(1, int(max_files))
1053
- max_errors = max(0, int(max_errors))
1054
- max_file_bytes = max(1, int(max_file_bytes))
1055
- allowed_exts = (
1056
- frozenset(str(e).lower() if str(e).startswith(".") else f".{str(e).lower()}" for e in extensions)
1057
- if extensions
1058
- else self._folder_extensions()
1059
- )
1060
- patterns = _load_latticeignore(root)
1061
- errors: List[Dict[str, Any]] = summary["errors"]
1062
- skipped = summary["skipped"]
1063
- items: List[IngestionItem] = []
1064
-
1065
- def _record_error(path: Path, detail: str, status: str = "failed") -> None:
1066
- summary["failed"] += 1
1067
- if len(errors) < max_errors:
1068
- errors.append({"path": str(path), "status": status, "detail": detail})
1069
-
1070
- for dirpath, dirnames, filenames in os.walk(root):
1071
- current = Path(dirpath)
1072
- rel_dir = current.relative_to(root)
1073
- kept_dirs: List[str] = []
1074
- for name in sorted(dirnames):
1075
- if name in FOLDER_DEFAULT_SKIP_DIRS:
1076
- continue
1077
- if name.startswith(".") and not include_hidden:
1078
- continue
1079
- rel = name if str(rel_dir) == "." else (rel_dir / name).as_posix()
1080
- if _matches_ignore(rel, name, is_dir=True, patterns=patterns):
1081
- skipped["ignored"] += 1
1082
- continue
1083
- kept_dirs.append(name)
1084
- dirnames[:] = kept_dirs if recursive else []
1085
-
1086
- for name in sorted(filenames):
1087
- if name == LATTICEIGNORE_FILENAME:
1088
- continue
1089
- summary["scanned"] += 1
1090
- path = current / name
1091
- rel = name if str(rel_dir) == "." else (rel_dir / name).as_posix()
1092
- if name.startswith(".") and not include_hidden:
1093
- skipped["hidden"] += 1
1094
- continue
1095
- if _matches_ignore(rel, name, is_dir=False, patterns=patterns):
1096
- skipped["ignored"] += 1
1097
- continue
1098
- ext = path.suffix.lower()
1099
- if ext not in allowed_exts:
1100
- skipped["extension"] += 1
1101
- continue
1102
- try:
1103
- size = path.stat().st_size
1104
- except OSError as exc:
1105
- _record_error(path, f"stat failed: {exc}")
1106
- continue
1107
- if size > max_file_bytes:
1108
- skipped["too_large"] += 1
1109
- continue
1110
- if len(items) >= max_files:
1111
- summary["truncated"] = True
1112
- break
1113
- item_metadata: Dict[str, Any] = {"relative_path": rel}
1114
- if ext in FOLDER_MULTIMODAL_EXTENSIONS and self._allow_multimodal:
1115
- # Routed by modality inside ``ingest``; reading the bytes as
1116
- # UTF-8 here would only produce mojibake.
1117
- source_type = "file"
1118
- elif ext in FOLDER_DOCUMENT_EXTENSIONS:
1119
- source_type = "pdf"
1120
- else:
1121
- source_type = "file"
1122
- try:
1123
- content = path.read_text(encoding="utf-8", errors="ignore")
1124
- except OSError as exc:
1125
- _record_error(path, f"read failed: {exc}")
1126
- continue
1127
- item_metadata["extracted"] = {"content": content, "chars": len(content)}
1128
- items.append(
1129
- IngestionItem(
1130
- source_type=source_type,
1131
- title=name,
1132
- path=str(path),
1133
- source_uri=str(path),
1134
- owner=owner,
1135
- workspace_id=workspace_id,
1136
- metadata=item_metadata,
1137
- )
1138
- )
1139
- if summary["truncated"]:
1140
- break
1141
-
1142
- summary["matched"] = len(items)
1143
- if background:
1144
- job = self.schedule_background(
1145
- items, incremental=True, user_email=user_email or owner,
1146
- )
1147
- summary.update(status="scheduled", job_id=job.job_id, scheduled=len(items))
1148
- return summary
1149
-
1150
- for item in items:
1151
- result = self.ingest(item, user_email=user_email or owner)
1152
- if result.status == "ok":
1153
- if result.duplicate:
1154
- summary["duplicate"] += 1
1155
- else:
1156
- summary["ingested"] += 1
1157
- else:
1158
- _record_error(Path(item.path or ""), result.detail or result.status, result.status)
1159
- summary["status"] = "ok" if summary["failed"] == 0 else "partial"
1160
- return summary
1161
-
1162
- def _folder_extensions(self) -> frozenset:
1163
- """Folder-scan allow-list — pictures and recordings only when enabled."""
1164
- if self._allow_multimodal:
1165
- return DEFAULT_FOLDER_EXTENSIONS | FOLDER_MULTIMODAL_EXTENSIONS
1166
- return DEFAULT_FOLDER_EXTENSIONS
1167
-
1168
- # ── routing helpers ──────────────────────────────────────────────────────
1169
- def _ingest_text(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
1170
- text = item.text or ""
1171
- if not text.strip():
1172
- raise ValueError(
1173
- f"Empty content: {source_type} ingestion requires non-empty text."
1174
- )
1175
- if len(text.encode("utf-8", "ignore")) > self._max_text_bytes:
1176
- raise ValueError(
1177
- f"Text payload exceeds the {self._max_text_bytes // (1024 * 1024)}MB ingestion limit."
1178
- )
1179
- title = item.title or item.source_uri or source_type
1180
- return self._kg.ingest_source(
1181
- source_type=source_type,
1182
- title=title,
1183
- text=text,
1184
- source_uri=item.source_uri,
1185
- owner=owner,
1186
- workspace_id=item.workspace_id,
1187
- permissions=item.permissions,
1188
- captured_at=captured_at,
1189
- modified_at=item.modified_at,
1190
- conversation_id=item.conversation_id,
1191
- metadata={"mime_type": item.mime_type, **(item.metadata or {})},
1192
- )
1193
-
1194
- def _ingest_chat(self, item, *, source_type, owner) -> Dict[str, Any]:
1195
- text = item.text or ""
1196
- meta = item.metadata or {}
1197
- role = str(meta.get("role") or "user")
1198
- result = self._kg.ingest_message(
1199
- role,
1200
- text,
1201
- user_email=owner,
1202
- user_nickname=meta.get("user_nickname"),
1203
- source=meta.get("source") or source_type,
1204
- conversation_id=item.conversation_id,
1205
- workspace_id=item.workspace_id,
1206
- raw=meta.get("raw"),
1207
- )
1208
- # ingest_message reports message/response node ids; normalize the keys
1209
- # the provenance step expects.
1210
- result.setdefault("node_id", result.get("node_id") or result.get("message_node_id") or result.get("id"))
1211
- result.setdefault("title", item.title or text[:80])
1212
- return result
1213
-
1214
- def _ingest_memory_record(self, item, *, source_type, owner) -> Dict[str, Any]:
1215
- node_type = _MEMORY_NODE_TYPES[source_type]
1216
- meta = item.metadata or {}
1217
- result = self._kg.ingest_event(
1218
- node_type,
1219
- item.title or (item.text or node_type)[:120],
1220
- user_email=owner,
1221
- source=meta.get("source") or source_type,
1222
- conversation_id=item.conversation_id,
1223
- workspace_id=item.workspace_id,
1224
- metadata={**meta, "detail": (item.text or "")[:2000]},
1225
- )
1226
- result.setdefault("node_id", result.get("node_id") or result.get("id"))
1227
- result.setdefault("title", item.title)
1228
- return result
1229
-
1230
- # ── multi-modal routing (v11.1.0 Track 3) ────────────────────────────────
1231
- def _modality_for(self, item: IngestionItem, source_type: str) -> str:
1232
- """``image`` / ``audio`` / ``video`` / ``text`` for this item.
1233
-
1234
- Always ``"text"`` while the flag is off, which is what makes "off" mean
1235
- *unchanged* rather than *slightly different*.
1236
- """
1237
- if not self._allow_multimodal:
1238
- return "text"
1239
- if source_type in IMAGE_SOURCE_TYPES:
1240
- return MODALITY_IMAGE
1241
- if source_type in AUDIO_SOURCE_TYPES:
1242
- return MODALITY_AUDIO
1243
- if not item.path:
1244
- return "text"
1245
- return detect_modality(item.path, item.mime_type)
1246
-
1247
- def _resolve_file_path(self, item: IngestionItem) -> Path:
1248
- if not item.path:
1249
- raise ValueError("File ingestion requires a path.")
1250
- path = Path(item.path)
1251
- if not path.exists():
1252
- raise FileNotFoundError(f"File not found: {path}")
1253
- if path.is_dir():
1254
- raise ValueError(f"File ingestion requires a file, got a directory: {path}")
1255
- return path
1256
-
1257
- def _ingest_image(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
1258
- """Store one picture as an ``Image`` node — OCR, caption, vector.
1259
-
1260
- The image vector (when a vision model produced one) goes to its own
1261
- index; the OCR/caption text rides the ordinary text index. That split
1262
- is what lets a typed question find a screenshot without ever comparing
1263
- a text vector to an image vector.
1264
- """
1265
- path = self._resolve_file_path(item)
1266
- facts = extract_image_facts(str(path), ports=self._multimodal)
1267
- result = write_image_memory(
1268
- self._kg,
1269
- path=path,
1270
- facts=facts,
1271
- title=item.title or path.name,
1272
- source_type=source_type if source_type in IMAGE_SOURCE_TYPES else MODALITY_IMAGE,
1273
- source_uri=item.source_uri,
1274
- owner=owner,
1275
- workspace_id=item.workspace_id,
1276
- conversation_id=item.conversation_id,
1277
- captured_at=captured_at,
1278
- modified_at=item.modified_at,
1279
- permissions=item.permissions,
1280
- extra_metadata={"mime_type": item.mime_type, **(item.metadata or {})},
1281
- )
1282
- self._record_image_vector(result["node_id"], facts)
1283
- quality = image_quality_score(facts)
1284
- result["extraction_quality"] = {
1285
- "score": quality["score"],
1286
- "level": _quality_level(quality["score"]),
1287
- "reasons": quality["reasons"],
1288
- }
1289
- return result
1290
-
1291
- def _record_image_vector(self, node_id: str, facts: ImageFacts) -> None:
1292
- """File the image-space vector, if a vision model actually made one."""
1293
- if facts.embedding is None:
1294
- return
1295
- from .graph.image_vectors import record_image_vector
1296
-
1297
- record_image_vector(
1298
- self._kg,
1299
- node_id=node_id,
1300
- vector=facts.embedding,
1301
- model_id=self._multimodal.vision_model_id or "vision:unnamed",
1302
- space=self._multimodal.vision_space,
1303
- updated_at=utc_now_iso(),
1304
- )
1305
-
1306
- def _ingest_audio(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
1307
- """Store one recording as an ``Audio`` node, transcribed when possible.
1308
-
1309
- The transcript is text and rides the ordinary text index — chunks,
1310
- concepts, provenance, dedupe all unchanged — but the node itself is a
1311
- recording, because that is what it is whether or not anyone could hear
1312
- it. The recording's own facts stay in the metadata (``modality``,
1313
- ``audio_path``, ``transcription``, ``searchable``). Without a
1314
- transcriber the memory is still kept, and its body says plainly that
1315
- the words were never recognized instead of leaving a blank note.
1316
- """
1317
- path = self._resolve_file_path(item)
1318
- facts = transcribe_audio(str(path), ports=self._multimodal, transcript=item.text)
1319
- title = item.title or path.stem
1320
- body = facts.transcript or (
1321
- f"[{MODALITY_AUDIO}] {title}\n"
1322
- "이 녹음은 아직 글로 바뀌지 않았습니다 — 음성 인식기가 없어 내용 검색은 되지 않습니다."
1323
- )
1324
- result = self._kg.ingest_source(
1325
- source_type=source_type,
1326
- title=title,
1327
- text=body,
1328
- source_uri=item.source_uri or str(path),
1329
- owner=owner,
1330
- workspace_id=item.workspace_id,
1331
- permissions=item.permissions,
1332
- captured_at=captured_at,
1333
- modified_at=item.modified_at,
1334
- conversation_id=item.conversation_id,
1335
- node_type=AUDIO_NODE_TYPE,
1336
- metadata={
1337
- "mime_type": item.mime_type,
1338
- "modality": MODALITY_AUDIO,
1339
- "audio_path": str(path),
1340
- "audio_bytes": path.stat().st_size,
1341
- "transcription": facts.transcription_status,
1342
- "searchable": facts.searchable,
1343
- **({"transcription_detail": facts.detail} if facts.detail else {}),
1344
- **(item.metadata or {}),
1345
- },
1346
- )
1347
- result.setdefault("title", title)
1348
- quality = audio_quality_score(facts)
1349
- result["extraction_quality"] = {
1350
- "score": quality["score"],
1351
- "level": _quality_level(quality["score"]),
1352
- "reasons": quality["reasons"],
1353
- }
1354
- return result
1355
-
1356
- def _ingest_file(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
1357
- path = self._resolve_file_path(item)
1358
- return self._kg.ingest_document(
1359
- path,
1360
- original_filename=item.title or path.name,
1361
- mime_type=item.mime_type,
1362
- uploader=owner,
1363
- conversation_id=item.conversation_id,
1364
- extracted=item.metadata.get("extracted") if item.metadata else None,
1365
- source_type=source_type,
1366
- source_uri=item.source_uri or str(path),
1367
- captured_at=captured_at,
1368
- modified_at=item.modified_at,
1369
- owner=owner,
1370
- workspace_id=item.workspace_id,
1371
- permissions=item.permissions,
1372
- )
1373
-
1374
-
1375
- def content_hash_text(text: str) -> str:
1376
- """Canonical content hash for a text payload (matches store hashing scheme)."""
1377
- return hashlib.sha256((text or "").encode("utf-8", "ignore")).hexdigest()