ltcai 11.2.0 → 11.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (247) hide show
  1. package/README.md +46 -53
  2. package/docs/CHANGELOG.md +61 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/MULTI_AGENT_RUNTIME.md +1 -1
  6. package/docs/ONBOARDING.md +1 -1
  7. package/docs/OPERATIONS.md +6 -2
  8. package/docs/PERMISSION_MODE.md +1 -1
  9. package/docs/TRUST_MODEL.md +1 -1
  10. package/docs/WHY_LATTICE.md +1 -1
  11. package/docs/kg-schema.md +2 -2
  12. package/docs/v11.3.0_PLAN.md +202 -0
  13. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
  14. package/lattice_brain/__init__.py +1 -1
  15. package/lattice_brain/graph/_kg_common/__init__.py +287 -0
  16. package/lattice_brain/graph/_kg_common/extraction.py +516 -0
  17. package/lattice_brain/graph/_kg_common/relations.py +161 -0
  18. package/lattice_brain/graph/_kg_common/text.py +479 -0
  19. package/lattice_brain/graph/discovery_index/__init__.py +35 -0
  20. package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
  21. package/lattice_brain/graph/discovery_index/extract.py +137 -0
  22. package/lattice_brain/graph/discovery_index/scan.py +411 -0
  23. package/lattice_brain/graph/discovery_index/upsert.py +495 -0
  24. package/lattice_brain/graph/projection/__init__.py +42 -0
  25. package/lattice_brain/graph/projection/curation.py +500 -0
  26. package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
  27. package/lattice_brain/graph/retrieval/__init__.py +54 -0
  28. package/lattice_brain/graph/retrieval/context.py +197 -0
  29. package/lattice_brain/graph/retrieval/graph_view.py +319 -0
  30. package/lattice_brain/graph/retrieval/hybrid.py +488 -0
  31. package/lattice_brain/graph/retrieval/maintenance.py +121 -0
  32. package/lattice_brain/graph/retrieval/signals.py +95 -0
  33. package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
  34. package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
  35. package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
  36. package/lattice_brain/graph/retrieval_vector/search.py +560 -0
  37. package/lattice_brain/graph/retrieval_vector/status.py +374 -0
  38. package/lattice_brain/ingestion/__init__.py +130 -0
  39. package/lattice_brain/ingestion/_contract.py +90 -0
  40. package/lattice_brain/ingestion/constants.py +127 -0
  41. package/lattice_brain/ingestion/folder_scan.py +57 -0
  42. package/lattice_brain/ingestion/folders.py +258 -0
  43. package/lattice_brain/ingestion/hashing.py +26 -0
  44. package/lattice_brain/ingestion/jobs_api.py +107 -0
  45. package/lattice_brain/ingestion/models.py +80 -0
  46. package/lattice_brain/ingestion/pipeline.py +486 -0
  47. package/lattice_brain/ingestion/quality.py +209 -0
  48. package/lattice_brain/ingestion/routing.py +295 -0
  49. package/lattice_brain/multimodal/__init__.py +164 -0
  50. package/lattice_brain/multimodal/audio.py +77 -0
  51. package/lattice_brain/multimodal/common.py +118 -0
  52. package/lattice_brain/multimodal/images.py +498 -0
  53. package/lattice_brain/multimodal/ports.py +169 -0
  54. package/lattice_brain/multimodal/video.py +410 -0
  55. package/lattice_brain/portability/__init__.py +90 -0
  56. package/lattice_brain/portability/_contract.py +42 -0
  57. package/lattice_brain/portability/backups.py +338 -0
  58. package/lattice_brain/portability/bundles.py +136 -0
  59. package/lattice_brain/portability/constants.py +93 -0
  60. package/lattice_brain/portability/fsops.py +138 -0
  61. package/lattice_brain/portability/service.py +41 -0
  62. package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
  63. package/lattice_brain/runtime/__init__.py +1 -1
  64. package/lattice_brain/runtime/multi_agent.py +1 -1
  65. package/latticeai/__init__.py +1 -1
  66. package/latticeai/api/chronicle.py +63 -0
  67. package/latticeai/core/agent/__init__.py +93 -0
  68. package/latticeai/core/agent/_contract.py +79 -0
  69. package/latticeai/core/agent/context.py +57 -0
  70. package/latticeai/core/agent/deps.py +125 -0
  71. package/latticeai/core/agent/execution.py +622 -0
  72. package/latticeai/core/agent/planning.py +145 -0
  73. package/latticeai/core/agent/recovery.py +157 -0
  74. package/latticeai/core/agent/runtime.py +210 -0
  75. package/latticeai/core/agent/verification.py +231 -0
  76. package/latticeai/core/embedding_providers/__init__.py +151 -0
  77. package/latticeai/core/embedding_providers/base.py +199 -0
  78. package/latticeai/core/embedding_providers/captions.py +162 -0
  79. package/latticeai/core/embedding_providers/profiles.py +126 -0
  80. package/latticeai/core/embedding_providers/text.py +350 -0
  81. package/latticeai/core/embedding_providers/vision.py +352 -0
  82. package/latticeai/core/file_generation/__init__.py +115 -0
  83. package/latticeai/core/file_generation/bundles.py +76 -0
  84. package/latticeai/core/file_generation/extraction.py +154 -0
  85. package/latticeai/core/file_generation/inference.py +235 -0
  86. package/latticeai/core/file_generation/orchestration.py +152 -0
  87. package/latticeai/core/file_generation/prompting.py +117 -0
  88. package/latticeai/core/file_generation/repair.py +114 -0
  89. package/latticeai/core/file_generation/sanitize.py +61 -0
  90. package/latticeai/core/file_generation/validation.py +201 -0
  91. package/latticeai/core/legacy_compatibility.py +1 -1
  92. package/latticeai/core/marketplace.py +1 -1
  93. package/latticeai/core/messages.py +9 -0
  94. package/latticeai/core/workspace_os_constants.py +1 -1
  95. package/latticeai/integrations/telegram_bot/__init__.py +123 -0
  96. package/latticeai/integrations/telegram_bot/__main__.py +17 -0
  97. package/latticeai/integrations/telegram_bot/config.py +86 -0
  98. package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
  99. package/latticeai/integrations/telegram_bot/flows.py +478 -0
  100. package/latticeai/integrations/telegram_bot/helpers.py +322 -0
  101. package/latticeai/integrations/telegram_bot/screens.py +394 -0
  102. package/latticeai/models/router/__init__.py +88 -0
  103. package/latticeai/models/router/_contract.py +66 -0
  104. package/latticeai/models/router/branding.py +56 -0
  105. package/latticeai/models/router/catalog.py +69 -0
  106. package/latticeai/models/router/documents.py +199 -0
  107. package/latticeai/models/router/errors.py +37 -0
  108. package/latticeai/models/router/generation.py +258 -0
  109. package/latticeai/models/router/loading.py +291 -0
  110. package/latticeai/models/router/local_models.py +85 -0
  111. package/latticeai/models/router/registry.py +147 -0
  112. package/latticeai/runtime/build_phases/__init__.py +82 -0
  113. package/latticeai/runtime/build_phases/features.py +407 -0
  114. package/latticeai/runtime/build_phases/foundation.py +555 -0
  115. package/latticeai/runtime/build_phases/web.py +492 -0
  116. package/latticeai/runtime/runtime_context.py +1 -0
  117. package/latticeai/services/architecture_readiness.py +48 -19
  118. package/latticeai/services/brain_intelligence/__init__.py +58 -0
  119. package/latticeai/services/brain_intelligence/_contract.py +71 -0
  120. package/latticeai/services/brain_intelligence/consistency.py +193 -0
  121. package/latticeai/services/brain_intelligence/constants.py +47 -0
  122. package/latticeai/services/brain_intelligence/digest.py +258 -0
  123. package/latticeai/services/brain_intelligence/health.py +331 -0
  124. package/latticeai/services/brain_intelligence/proposals.py +264 -0
  125. package/latticeai/services/brain_intelligence/sampling.py +84 -0
  126. package/latticeai/services/brain_intelligence/service.py +48 -0
  127. package/latticeai/services/chronicle.py +557 -0
  128. package/latticeai/services/memory_service/__init__.py +52 -0
  129. package/latticeai/services/memory_service/_contract.py +100 -0
  130. package/latticeai/services/memory_service/brief.py +431 -0
  131. package/latticeai/services/memory_service/constants.py +57 -0
  132. package/latticeai/services/memory_service/maintenance.py +138 -0
  133. package/latticeai/services/memory_service/manager.py +186 -0
  134. package/latticeai/services/memory_service/proof.py +136 -0
  135. package/latticeai/services/memory_service/recall.py +225 -0
  136. package/latticeai/services/memory_service/service.py +48 -0
  137. package/latticeai/services/memory_service/stores.py +110 -0
  138. package/latticeai/services/model_runtime/__init__.py +322 -0
  139. package/latticeai/services/model_runtime/cloud.py +87 -0
  140. package/latticeai/services/model_runtime/download.py +282 -0
  141. package/latticeai/services/model_runtime/engines.py +341 -0
  142. package/latticeai/services/model_runtime/loading.py +178 -0
  143. package/latticeai/services/model_runtime/service.py +129 -0
  144. package/latticeai/services/model_runtime/state.py +131 -0
  145. package/latticeai/services/model_runtime/status.py +255 -0
  146. package/latticeai/services/product_readiness.py +15 -7
  147. package/latticeai/setup/wizard/__init__.py +126 -0
  148. package/latticeai/setup/wizard/catalog.py +172 -0
  149. package/latticeai/setup/wizard/detect.py +323 -0
  150. package/latticeai/setup/wizard/install.py +348 -0
  151. package/latticeai/setup/wizard/paths.py +168 -0
  152. package/latticeai/setup/wizard/plans.py +74 -0
  153. package/latticeai/setup/wizard/recommend.py +320 -0
  154. package/package.json +6 -2
  155. package/scripts/bump_version.py +14 -0
  156. package/scripts/capture_release_evidence.mjs +33 -21
  157. package/scripts/check_current_release_docs.mjs +1 -1
  158. package/scripts/check_i18n_namespace_coverage.mjs +41 -4
  159. package/scripts/check_max_file_lines.mjs +102 -0
  160. package/scripts/check_release_evidence_bound.mjs +30 -15
  161. package/scripts/check_screenshot_pixel_delta.py +34 -4
  162. package/scripts/check_server_i18n.mjs +1 -0
  163. package/scripts/generate_rust_parity_fixtures.py +562 -0
  164. package/scripts/lib/mock_server_fingerprint.mjs +94 -0
  165. package/scripts/release_screen_claims.json +31 -2
  166. package/src-tauri/Cargo.lock +361 -3
  167. package/src-tauri/Cargo.toml +6 -1
  168. package/src-tauri/src/backend.rs +349 -0
  169. package/src-tauri/src/folder.rs +33 -0
  170. package/src-tauri/src/main.rs +97 -399
  171. package/src-tauri/tauri.conf.json +1 -1
  172. package/static/app/asset-manifest.json +41 -37
  173. package/static/app/assets/Act-yYpYnn0v.js +1 -0
  174. package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
  175. package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
  176. package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
  177. package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
  178. package/static/app/assets/Capture-CFIRsFNE.js +1 -0
  179. package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
  180. package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
  181. package/static/app/assets/Library-DwO3yZST.js +1 -0
  182. package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
  183. package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
  184. package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
  185. package/static/app/assets/System-DW8F-2xL.js +1 -0
  186. package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
  187. package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
  188. package/static/app/assets/brain-Ci1CkWjM.js +1 -0
  189. package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
  190. package/static/app/assets/circle-check-DfInj-qD.js +1 -0
  191. package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
  192. package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
  193. package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
  194. package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
  195. package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
  196. package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
  197. package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
  198. package/static/app/assets/index-_u5iUHDr.js +10 -0
  199. package/static/app/assets/input-B0lPdRQZ.js +1 -0
  200. package/static/app/assets/link-2-CoFbooHS.js +1 -0
  201. package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
  202. package/static/app/assets/primitives-DEbN-d6p.js +1 -0
  203. package/static/app/assets/search-BybIWPNd.js +1 -0
  204. package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
  205. package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
  206. package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
  207. package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
  208. package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
  209. package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
  210. package/static/app/assets/utils-BlZr7Pd4.js +4 -0
  211. package/static/app/assets/workspace-jJY4RuAV.js +1 -0
  212. package/static/app/index.html +4 -4
  213. package/static/sw.js +1 -1
  214. package/lattice_brain/graph/_kg_common.py +0 -1331
  215. package/lattice_brain/graph/discovery_index.py +0 -1141
  216. package/lattice_brain/graph/retrieval.py +0 -1120
  217. package/lattice_brain/graph/retrieval_vector.py +0 -1293
  218. package/lattice_brain/ingestion.py +0 -1525
  219. package/lattice_brain/multimodal.py +0 -1258
  220. package/latticeai/core/agent.py +0 -1465
  221. package/latticeai/core/embedding_providers.py +0 -1196
  222. package/latticeai/core/file_generation.py +0 -1047
  223. package/latticeai/integrations/telegram_bot.py +0 -1390
  224. package/latticeai/models/router.py +0 -1007
  225. package/latticeai/runtime/build_phases.py +0 -1450
  226. package/latticeai/services/brain_intelligence.py +0 -1083
  227. package/latticeai/services/memory_service.py +0 -1177
  228. package/latticeai/services/model_runtime.py +0 -1281
  229. package/latticeai/setup/wizard.py +0 -1310
  230. package/static/app/assets/Act-AWf0SAKp.js +0 -1
  231. package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
  232. package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
  233. package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
  234. package/static/app/assets/Capture-CqOSzyPr.js +0 -1
  235. package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
  236. package/static/app/assets/Library-CX-bbhmK.js +0 -1
  237. package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
  238. package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
  239. package/static/app/assets/System-Bu2t5hn1.js +0 -1
  240. package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
  241. package/static/app/assets/brain-DJMoqrwx.js +0 -1
  242. package/static/app/assets/index-BpYkzcVm.js +0 -10
  243. package/static/app/assets/input-DSlJJxRs.js +0 -1
  244. package/static/app/assets/primitives-BCx6TvfG.js +0 -1
  245. package/static/app/assets/search-Cgy8cCFJ.js +0 -1
  246. package/static/app/assets/utils-zqPZJxdx.js +0 -4
  247. package/static/app/assets/workspace-DXTihhfU.js +0 -1
@@ -0,0 +1,127 @@
1
+ """What routes where, what a folder scan admits, and which gates decide.
2
+
3
+ Source-type sets, folder-scan filters, size budgets and the four feature gates,
4
+ with no logic beyond them. Every other submodule imports this one; this one
5
+ imports nothing from the package, which is what keeps the layering acyclic.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from ..gates import FeatureGate
11
+ from ..multimodal import AUDIO_EXTENSIONS, IMAGE_EXTENSIONS, VIDEO_EXTENSIONS
12
+
13
+ # Source types that arrive as a file on disk (read via ingest_document).
14
+ FILE_SOURCE_TYPES = frozenset({"file", "local_file", "upload", "pdf"})
15
+ # Source types that arrive as extracted text (read via ingest_source).
16
+ TEXT_SOURCE_TYPES = frozenset(
17
+ {"web_url", "browser_tab", "text", "markdown", "note", "code", "clipboard"}
18
+ )
19
+ # Conversational exchanges (read via ingest_message — role/content semantics,
20
+ # conversation chaining). v4: chat and MCP messages stop bypassing the
21
+ # pipeline, so they carry provenance and fire the hook lifecycle like every
22
+ # other source.
23
+ CHAT_SOURCE_TYPES = frozenset({"chat_message", "mcp_message"})
24
+ # Typed memory records (read via ingest_event → Decision/Experience/Event
25
+ # nodes). The Memory System writes through the same door as everything else.
26
+ MEMORY_SOURCE_TYPES = frozenset({"decision", "experience", "workspace_event"})
27
+ _MEMORY_NODE_TYPES = {"decision": "Decision", "experience": "Experience", "workspace_event": "Event"}
28
+
29
+ DEFAULT_MAX_TEXT_BYTES = 5 * 1024 * 1024 # 5 MB of extracted text per item
30
+
31
+
32
+ # ── Folder ingestion (ingest_folder) filters ─────────────────────────────────
33
+ # Directories that are always pruned regardless of .latticeignore.
34
+ FOLDER_DEFAULT_SKIP_DIRS = frozenset(
35
+ {
36
+ ".git",
37
+ "node_modules",
38
+ "__pycache__",
39
+ ".venv",
40
+ "venv",
41
+ "env",
42
+ ".pytest_cache",
43
+ ".mypy_cache",
44
+ ".ruff_cache",
45
+ "dist",
46
+ "build",
47
+ ".next",
48
+ "target",
49
+ ".cache",
50
+ ".idea",
51
+ ".vscode",
52
+ }
53
+ )
54
+ # Extension filter matching FILE_SOURCE_TYPES conventions: text/markdown/code
55
+ # are read inline (extracted content → chunks); .pdf routes as source_type
56
+ # "pdf" through ingest_document (content extraction is upstream's concern).
57
+ FOLDER_TEXT_EXTENSIONS = frozenset(
58
+ {".txt", ".md", ".markdown", ".rst", ".csv", ".json", ".yaml", ".yml", ".toml", ".ini"}
59
+ )
60
+ FOLDER_CODE_EXTENSIONS = frozenset(
61
+ {
62
+ ".py", ".js", ".ts", ".tsx", ".jsx", ".html", ".css", ".go", ".rs",
63
+ ".java", ".c", ".h", ".cpp", ".hpp", ".rb", ".php", ".swift", ".kt",
64
+ ".sh", ".sql",
65
+ }
66
+ )
67
+ FOLDER_DOCUMENT_EXTENSIONS = frozenset({".pdf"})
68
+ DEFAULT_FOLDER_EXTENSIONS = (
69
+ FOLDER_TEXT_EXTENSIONS | FOLDER_CODE_EXTENSIONS | FOLDER_DOCUMENT_EXTENSIONS
70
+ )
71
+ DEFAULT_MAX_FILE_BYTES = 4_000_000 # matches the local-index text/code budget
72
+ LATTICEIGNORE_FILENAME = ".latticeignore"
73
+ # Opt-out escape hatch for the post-ingest incremental vector sync.
74
+ AUTO_VECTOR_INDEX_ENV = "LATTICEAI_AUTO_VECTOR_INDEX"
75
+ #: Default *on*, unlike every other gate here: new material has always been made
76
+ #: searchable straight away, and this exists so a settings surface can turn that
77
+ #: off (batch reindex later) without a restart. ``FeatureGate`` parses the env
78
+ #: var with the same words the hand-written opt-out check used, so an untouched
79
+ #: install — including one with a nonsense value — answers exactly as before.
80
+ AUTO_VECTOR_INDEX_GATE = FeatureGate(
81
+ AUTO_VECTOR_INDEX_ENV,
82
+ default=True,
83
+ name="auto_vector_index",
84
+ detail="New material is prepared for semantic search as soon as it lands.",
85
+ )
86
+
87
+ # ── Multi-modal ingestion (v11.1.0 Track 3) ──────────────────────────────────
88
+ # Opt-in, default off, on purpose. Turning it on changes what a folder scan
89
+ # *stores* (pictures and recordings, with OCR and — if a model is loaded —
90
+ # captions and vectors), and that is the user's call, not a default. With the
91
+ # flag off every routing decision below is skipped and behaviour is byte-for-
92
+ # byte what it was before this release.
93
+ ALLOW_MULTIMODAL_ENV = "LATTICEAI_ALLOW_MULTIMODAL"
94
+ #: The multi-modal switch, resolved when it is asked rather than frozen into
95
+ #: ``self`` at construction (v11.2.0). The environment variable is still the
96
+ #: answer for an untouched install — same var, same words, same default off —
97
+ #: but a settings surface can bind a resolver and move it without a restart.
98
+ MULTIMODAL_GATE = FeatureGate(
99
+ ALLOW_MULTIMODAL_ENV,
100
+ default=False,
101
+ name="allow_multimodal",
102
+ detail="Pictures and recordings are only ingested when this is turned on.",
103
+ )
104
+ #: Video is a *sub-switch* of the one above: with multi-modal off nothing about
105
+ #: video happens at all, and with it on video is included unless this is
106
+ #: explicitly turned off. The effective default is therefore still "no video",
107
+ #: and the seam exists so a settings screen can offer pictures without films.
108
+ ALLOW_VIDEO_ENV = "LATTICEAI_ALLOW_VIDEO"
109
+ VIDEO_GATE = FeatureGate(
110
+ ALLOW_VIDEO_ENV,
111
+ default=True,
112
+ name="allow_video",
113
+ detail="Videos are ingested as keyframes plus subtitles when multi-modal is on.",
114
+ )
115
+ #: Source types that name a modality outright (a caller who already knows).
116
+ IMAGE_SOURCE_TYPES = frozenset({"image", "screenshot", "photo"})
117
+ AUDIO_SOURCE_TYPES = frozenset({"audio", "voice_memo", "recording"})
118
+ VIDEO_SOURCE_TYPES = frozenset({"video", "screen_recording", "movie"})
119
+ #: Added to the folder-scan allow-list only while multimodal is enabled.
120
+ FOLDER_MULTIMODAL_EXTENSIONS = IMAGE_EXTENSIONS | AUDIO_EXTENSIONS
121
+ #: Videos join the folder allow-list only when this machine can decode one —
122
+ #: scanning a folder into a pile of refusals is not a feature.
123
+ FOLDER_VIDEO_EXTENSIONS = VIDEO_EXTENSIONS
124
+ #: Graph node type for a recording. ``NodeType.AUDIO`` normalizes this on the
125
+ #: KG v2 write side; the legacy tables keep the label verbatim, which is what
126
+ #: every type-aware read (graph view, context sections, doc-gen) matches on.
127
+ AUDIO_NODE_TYPE = "Audio"
@@ -0,0 +1,57 @@
1
+ """``.latticeignore`` parsing and matching for the folder walk.
2
+
3
+ A gitignore-like subset: blank lines and ``#`` comments are dropped, patterns
4
+ are ``fnmatch`` globs, and a trailing ``/`` restricts a pattern to directories.
5
+ Patterns match against both the root-relative posix path and the basename, so
6
+ ``*.log`` and ``docs/draft.md`` both behave the way a reader expects.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import fnmatch
12
+ from pathlib import Path
13
+ from typing import Iterable, List
14
+
15
+ from .constants import LATTICEIGNORE_FILENAME
16
+
17
+
18
+ def _load_latticeignore(root: Path) -> List[str]:
19
+ """Parse ``root/.latticeignore`` → glob patterns (gitignore-like subset)."""
20
+ ignore_file = root / LATTICEIGNORE_FILENAME
21
+ patterns: List[str] = []
22
+ if not ignore_file.is_file():
23
+ return patterns
24
+ try:
25
+ lines = ignore_file.read_text(encoding="utf-8", errors="ignore").splitlines()
26
+ except OSError:
27
+ return patterns
28
+ for raw in lines:
29
+ line = raw.strip()
30
+ if not line or line.startswith("#"):
31
+ continue
32
+ patterns.append(line)
33
+ return patterns
34
+
35
+
36
+ def _matches_ignore(
37
+ rel_posix: str, name: str, *, is_dir: bool, patterns: Iterable[str]
38
+ ) -> bool:
39
+ """fnmatch-based .latticeignore matching.
40
+
41
+ - ``pattern/`` matches directories only (files under it never appear
42
+ because ignored directories are pruned during the walk).
43
+ - Patterns match against both the root-relative posix path and the
44
+ basename, so ``*.log`` and ``docs/draft.md`` both behave as expected.
45
+ """
46
+ for raw in patterns:
47
+ pattern = raw
48
+ if pattern.endswith("/"):
49
+ if not is_dir:
50
+ continue
51
+ pattern = pattern.rstrip("/")
52
+ pattern = pattern.lstrip("/")
53
+ if not pattern:
54
+ continue
55
+ if fnmatch.fnmatch(rel_posix, pattern) or fnmatch.fnmatch(name, pattern):
56
+ return True
57
+ return False
@@ -0,0 +1,258 @@
1
+ """Walking a folder, and the one-page web hand-off, into the standard door.
2
+
3
+ Neither method is a second ingest path: both build ordinary ``IngestionItem``
4
+ values and hand them to ``IngestionPipeline.ingest``. The folder walk owns the
5
+ filtering order (hard skip-list → hidden → ``.latticeignore`` → extension →
6
+ size) and the choice between ingesting inline and scheduling in the background;
7
+ the web hand-off owns the refusal to fetch or parse anything itself.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import os
13
+ from pathlib import Path
14
+ from typing import Any, Dict, Iterable, List, Optional
15
+
16
+ from ._contract import IngestionCore as _Core
17
+ from .constants import (
18
+ DEFAULT_FOLDER_EXTENSIONS,
19
+ DEFAULT_MAX_FILE_BYTES,
20
+ FOLDER_DEFAULT_SKIP_DIRS,
21
+ FOLDER_DOCUMENT_EXTENSIONS,
22
+ FOLDER_MULTIMODAL_EXTENSIONS,
23
+ FOLDER_VIDEO_EXTENSIONS,
24
+ LATTICEIGNORE_FILENAME,
25
+ )
26
+ from .folder_scan import _load_latticeignore, _matches_ignore
27
+ from .models import IngestionItem, IngestionResult
28
+
29
+
30
+ class IngestionFolderMixin(_Core):
31
+ """Folder walk + web hand-off. Mixed into ``IngestionPipeline``."""
32
+
33
+ def ingest_web_page(
34
+ self,
35
+ url: str,
36
+ extracted_text: str,
37
+ *,
38
+ title: Optional[str] = None,
39
+ metadata: Optional[Dict[str, Any]] = None,
40
+ owner: Optional[str] = None,
41
+ workspace_id: Optional[str] = None,
42
+ captured_at: Optional[str] = None,
43
+ user_email: Optional[str] = None,
44
+ ) -> IngestionResult:
45
+ """Ingest an *already-extracted* web page (see module docstring seam).
46
+
47
+ Fetching/parsing is upstream's responsibility (browser extension /
48
+ tools layer); this wrapper only normalizes ``(url, extracted_text)``
49
+ into an ``IngestionItem(source_type="web_url")`` and routes it through
50
+ the standard :meth:`ingest` door.
51
+ """
52
+ url = str(url or "").strip()
53
+ if not url:
54
+ return IngestionResult(
55
+ status="failed", source_type="web_url",
56
+ indexing_status="skipped", detail="url required",
57
+ )
58
+ text = str(extracted_text or "")
59
+ if not text.strip():
60
+ return IngestionResult(
61
+ status="failed", source_type="web_url",
62
+ indexing_status="skipped",
63
+ detail=(
64
+ "extracted_text required — the graph layer does not fetch or "
65
+ "parse the web; extraction happens upstream."
66
+ ),
67
+ )
68
+ item = IngestionItem(
69
+ source_type="web_url",
70
+ title=title or url,
71
+ text=text,
72
+ source_uri=url,
73
+ owner=owner,
74
+ workspace_id=workspace_id,
75
+ captured_at=captured_at,
76
+ metadata=dict(metadata or {}),
77
+ )
78
+ return self.ingest(item, user_email=user_email or owner)
79
+
80
+ def ingest_folder(
81
+ self,
82
+ root_path: Any,
83
+ *,
84
+ recursive: bool = True,
85
+ background: bool = False,
86
+ extensions: Optional[Iterable[str]] = None,
87
+ max_file_bytes: int = DEFAULT_MAX_FILE_BYTES,
88
+ include_hidden: bool = False,
89
+ max_files: int = 1000,
90
+ max_errors: int = 25,
91
+ owner: Optional[str] = None,
92
+ workspace_id: Optional[str] = None,
93
+ user_email: Optional[str] = None,
94
+ ) -> Dict[str, Any]:
95
+ """Walk ``root_path`` and ingest every eligible file through the pipeline.
96
+
97
+ Filtering, in order: hard skip-list directories (``.git`` …), hidden
98
+ entries (unless ``include_hidden``), root ``.latticeignore`` patterns
99
+ (fnmatch globs; ``dir/`` suffix prunes directories), extension
100
+ allow-list, then ``max_file_bytes``. Text/code files are read inline so
101
+ their content is chunked; ``.pdf`` routes through the file door without
102
+ inline extraction.
103
+
104
+ ``background=True`` schedules the built items on the existing
105
+ :class:`BackgroundIngestionQueue` instead of ingesting inline.
106
+ Returns a summary dict with counts and per-file errors (capped at
107
+ ``max_errors``).
108
+ """
109
+ summary: Dict[str, Any] = {
110
+ "root": str(root_path),
111
+ "recursive": bool(recursive),
112
+ "background": bool(background),
113
+ "scanned": 0,
114
+ "matched": 0,
115
+ "ingested": 0,
116
+ "duplicate": 0,
117
+ "failed": 0,
118
+ "skipped": {"ignored": 0, "extension": 0, "too_large": 0, "hidden": 0},
119
+ "truncated": False,
120
+ "errors": [],
121
+ }
122
+ try:
123
+ root = Path(root_path).expanduser()
124
+ except TypeError:
125
+ summary.update(status="failed", detail=f"invalid root path: {root_path!r}")
126
+ return summary
127
+ if not root.is_dir():
128
+ summary.update(status="failed", detail=f"not a directory: {root}")
129
+ return summary
130
+ if not self.available():
131
+ summary.update(
132
+ status="unavailable",
133
+ detail="Knowledge Graph is disabled (LATTICEAI_ENABLE_GRAPH).",
134
+ )
135
+ return summary
136
+ summary["root"] = str(root)
137
+ max_files = max(1, int(max_files))
138
+ max_errors = max(0, int(max_errors))
139
+ max_file_bytes = max(1, int(max_file_bytes))
140
+ allowed_exts = (
141
+ frozenset(str(e).lower() if str(e).startswith(".") else f".{str(e).lower()}" for e in extensions)
142
+ if extensions
143
+ else self._folder_extensions()
144
+ )
145
+ patterns = _load_latticeignore(root)
146
+ errors: List[Dict[str, Any]] = summary["errors"]
147
+ skipped = summary["skipped"]
148
+ items: List[IngestionItem] = []
149
+
150
+ def _record_error(path: Path, detail: str, status: str = "failed") -> None:
151
+ summary["failed"] += 1
152
+ if len(errors) < max_errors:
153
+ errors.append({"path": str(path), "status": status, "detail": detail})
154
+
155
+ for dirpath, dirnames, filenames in os.walk(root):
156
+ current = Path(dirpath)
157
+ rel_dir = current.relative_to(root)
158
+ kept_dirs: List[str] = []
159
+ for name in sorted(dirnames):
160
+ if name in FOLDER_DEFAULT_SKIP_DIRS:
161
+ continue
162
+ if name.startswith(".") and not include_hidden:
163
+ continue
164
+ rel = name if str(rel_dir) == "." else (rel_dir / name).as_posix()
165
+ if _matches_ignore(rel, name, is_dir=True, patterns=patterns):
166
+ skipped["ignored"] += 1
167
+ continue
168
+ kept_dirs.append(name)
169
+ dirnames[:] = kept_dirs if recursive else []
170
+
171
+ for name in sorted(filenames):
172
+ if name == LATTICEIGNORE_FILENAME:
173
+ continue
174
+ summary["scanned"] += 1
175
+ path = current / name
176
+ rel = name if str(rel_dir) == "." else (rel_dir / name).as_posix()
177
+ if name.startswith(".") and not include_hidden:
178
+ skipped["hidden"] += 1
179
+ continue
180
+ if _matches_ignore(rel, name, is_dir=False, patterns=patterns):
181
+ skipped["ignored"] += 1
182
+ continue
183
+ ext = path.suffix.lower()
184
+ if ext not in allowed_exts:
185
+ skipped["extension"] += 1
186
+ continue
187
+ try:
188
+ size = path.stat().st_size
189
+ except OSError as exc:
190
+ _record_error(path, f"stat failed: {exc}")
191
+ continue
192
+ if size > max_file_bytes:
193
+ skipped["too_large"] += 1
194
+ continue
195
+ if len(items) >= max_files:
196
+ summary["truncated"] = True
197
+ break
198
+ item_metadata: Dict[str, Any] = {"relative_path": rel}
199
+ if ext in (FOLDER_MULTIMODAL_EXTENSIONS | FOLDER_VIDEO_EXTENSIONS) and self._allow_multimodal:
200
+ # Routed by modality inside ``ingest``; reading the bytes as
201
+ # UTF-8 here would only produce mojibake.
202
+ source_type = "file"
203
+ elif ext in FOLDER_DOCUMENT_EXTENSIONS:
204
+ source_type = "pdf"
205
+ else:
206
+ source_type = "file"
207
+ try:
208
+ content = path.read_text(encoding="utf-8", errors="ignore")
209
+ except OSError as exc:
210
+ _record_error(path, f"read failed: {exc}")
211
+ continue
212
+ item_metadata["extracted"] = {"content": content, "chars": len(content)}
213
+ items.append(
214
+ IngestionItem(
215
+ source_type=source_type,
216
+ title=name,
217
+ path=str(path),
218
+ source_uri=str(path),
219
+ owner=owner,
220
+ workspace_id=workspace_id,
221
+ metadata=item_metadata,
222
+ )
223
+ )
224
+ if summary["truncated"]:
225
+ break
226
+
227
+ summary["matched"] = len(items)
228
+ if background:
229
+ job = self.schedule_background(
230
+ items, incremental=True, user_email=user_email or owner,
231
+ )
232
+ summary.update(status="scheduled", job_id=job.job_id, scheduled=len(items))
233
+ return summary
234
+
235
+ for item in items:
236
+ result = self.ingest(item, user_email=user_email or owner)
237
+ if result.status == "ok":
238
+ if result.duplicate:
239
+ summary["duplicate"] += 1
240
+ else:
241
+ summary["ingested"] += 1
242
+ else:
243
+ _record_error(Path(item.path or ""), result.detail or result.status, result.status)
244
+ summary["status"] = "ok" if summary["failed"] == 0 else "partial"
245
+ return summary
246
+
247
+ def _folder_extensions(self) -> frozenset:
248
+ """Folder-scan allow-list — pictures, recordings and films when enabled.
249
+
250
+ Video joins only when this machine can actually decode one, so a scan
251
+ never fills the error list with files it was always going to refuse.
252
+ """
253
+ if not self._allow_multimodal:
254
+ return DEFAULT_FOLDER_EXTENSIONS
255
+ allowed = DEFAULT_FOLDER_EXTENSIONS | FOLDER_MULTIMODAL_EXTENSIONS
256
+ if self._allow_video:
257
+ return allowed | FOLDER_VIDEO_EXTENSIONS
258
+ return allowed
@@ -0,0 +1,26 @@
1
+ """The two content hashes the pipeline names things by.
2
+
3
+ :func:`content_hash_text` matches the store's own hashing scheme, so a text
4
+ payload hashed here and a text payload hashed there dedupe against each other.
5
+ :func:`_file_digest` streams a file instead of reading it whole — it is the key
6
+ a video's keyframe folder is named by, and videos are large.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import hashlib
12
+ from pathlib import Path
13
+
14
+
15
+ def content_hash_text(text: str) -> str:
16
+ """Canonical content hash for a text payload (matches store hashing scheme)."""
17
+ return hashlib.sha256((text or "").encode("utf-8", "ignore")).hexdigest()
18
+
19
+
20
+ def _file_digest(path: Path) -> str:
21
+ """Streaming sha256 of a file — the key a video's frame folder is named by."""
22
+ digest = hashlib.sha256()
23
+ with path.open("rb") as handle:
24
+ for block in iter(lambda: handle.read(1024 * 1024), b""):
25
+ digest.update(block)
26
+ return digest.hexdigest()
@@ -0,0 +1,107 @@
1
+ """Scheduling many items, and running (or resuming) what was scheduled.
2
+
3
+ The pipeline owns "ingest one item"; ``BackgroundIngestionQueue`` owns
4
+ "schedule many and report progress". This mixin is the seam between them:
5
+ per-item errors are recorded and never abort a job, progress is checkpointed
6
+ after every item, and the same method powers both the first run and a resume
7
+ because already-completed items are simply skipped.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from typing import Any, Dict, List, Optional
13
+
14
+ from ..ingestion_jobs import BackgroundIngestionJob
15
+ from ._contract import IngestionCore as _Core
16
+ from .models import IngestionItem
17
+
18
+
19
+ class IngestionJobsMixin(_Core):
20
+ """Background job scheduling. Mixed into ``IngestionPipeline``."""
21
+
22
+ # --- Large candidate #1: background / incremental scheduling (slice) ---
23
+ def schedule_background(
24
+ self,
25
+ items: List[IngestionItem],
26
+ *,
27
+ incremental: bool = True,
28
+ user_email: Optional[str] = None,
29
+ ) -> BackgroundIngestionJob:
30
+ """Schedule items for background incremental indexing.
31
+
32
+ Returns a job handle. Actual execution can be driven by caller
33
+ (or future worker) calling pipeline.ingest on each — or through
34
+ :meth:`run_background_job`. This seam enables large-corpus scale
35
+ without blocking user requests.
36
+ """
37
+ job = self._bg_queue.schedule(items, incremental=incremental, user_email=user_email)
38
+ # mark initial status on results concept (jobs track)
39
+ return job
40
+
41
+ def get_background_job(self, job_id: str) -> Optional[BackgroundIngestionJob]:
42
+ return self._bg_queue.get(job_id)
43
+
44
+ def list_background_jobs(self, limit: int = 20) -> List[Dict[str, Any]]:
45
+ """Recent jobs (newest first) in the frozen ``/api/ingestion`` schema."""
46
+ return [job.as_dict() for job in self._bg_queue.list_recent(limit=limit)]
47
+
48
+ def run_background_job(
49
+ self, job_id: str, *, user_email: Optional[str] = None
50
+ ) -> Dict[str, Any]:
51
+ """Execute a queued/interrupted job's remaining items.
52
+
53
+ Per-item errors are recorded (capped) and never abort the job. The
54
+ final status is ``completed`` (all done), ``partial`` (some done),
55
+ or ``failed`` (nothing done). Already-completed items are skipped, so
56
+ the same method safely powers both first-run and resume.
57
+ """
58
+ job = self._bg_queue.get(job_id)
59
+ if job is None:
60
+ return {"status": "not_found", "job_id": job_id}
61
+ if job.status == "running":
62
+ return job.as_dict()
63
+ return self._execute_background_job(job, user_email=user_email)
64
+
65
+ def resume_background_job(
66
+ self, job_id: str, *, user_email: Optional[str] = None
67
+ ) -> Dict[str, Any]:
68
+ """Resume an interrupted/partial/failed job from its remaining items."""
69
+ return self.run_background_job(job_id, user_email=user_email)
70
+
71
+ def _execute_background_job(
72
+ self, job: BackgroundIngestionJob, *, user_email: Optional[str] = None
73
+ ) -> Dict[str, Any]:
74
+ job.status = "running"
75
+ # Retried items get a fresh verdict: reset failure state for this run.
76
+ job.failed = 0
77
+ job.errors = []
78
+ job.touch()
79
+ self._bg_queue.save(job)
80
+ runner_email = user_email or job.user_email
81
+ for index in job.remaining_indices():
82
+ item = job.items[index]
83
+ try:
84
+ result = self.ingest(item, user_email=runner_email or item.owner)
85
+ status, detail = result.status, result.detail
86
+ except Exception as exc: # noqa: BLE001 — per-item isolation: keep going
87
+ status, detail = "failed", str(exc)
88
+ if status == "ok":
89
+ job.done_indices.add(index)
90
+ else:
91
+ job.record_error(index, item, detail or status)
92
+ job.processed = len(job.done_indices)
93
+ job.touch()
94
+ # Checkpoint per item: a crash here must cost at most the item in
95
+ # flight, never the whole job's progress. One small UPDATE against
96
+ # an ingest (parse + chunk + embed) is noise.
97
+ self._bg_queue.save(job)
98
+ job.processed = len(job.done_indices)
99
+ if job.total == 0 or job.processed >= job.total:
100
+ job.status = "completed"
101
+ elif job.processed > 0:
102
+ job.status = "partial"
103
+ else:
104
+ job.status = "failed"
105
+ job.touch()
106
+ self._bg_queue.save(job)
107
+ return job.as_dict()
@@ -0,0 +1,80 @@
1
+ """The two records the pipeline speaks in: one item in, one result out.
2
+
3
+ :class:`IngestionItem` is what every source normalizes to before the pipeline
4
+ sees it; :class:`IngestionResult` is what every source normalizes to after. The
5
+ result's ``as_dict`` is the frozen ``/api/ingestion`` payload shape — additive
6
+ keys appear only when populated, so pre-v9.8 consumers see what they always did.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from dataclasses import dataclass, field
12
+ from typing import Any, Dict, List, Optional
13
+
14
+
15
+ @dataclass
16
+ class IngestionItem:
17
+ """A single thing to ingest, normalized across every source type."""
18
+
19
+ source_type: str
20
+ title: Optional[str] = None
21
+ text: Optional[str] = None # text/web sources
22
+ path: Optional[str] = None # file sources
23
+ source_uri: Optional[str] = None
24
+ mime_type: Optional[str] = None
25
+ owner: Optional[str] = None
26
+ workspace_id: Optional[str] = None
27
+ permissions: Optional[Dict[str, Any]] = None
28
+ captured_at: Optional[str] = None
29
+ modified_at: Optional[str] = None
30
+ conversation_id: Optional[str] = None
31
+ agent_used: Optional[str] = None
32
+ metadata: Dict[str, Any] = field(default_factory=dict)
33
+
34
+
35
+ @dataclass
36
+ class IngestionResult:
37
+ """The outcome of one ingestion, including provenance and idempotency."""
38
+
39
+ status: str # ok | unavailable | blocked | failed
40
+ source_type: str
41
+ node_id: Optional[str] = None
42
+ source_node_id: Optional[str] = None
43
+ content_hash: Optional[str] = None
44
+ title: Optional[str] = None
45
+ chunk_ids: List[str] = field(default_factory=list)
46
+ chunk_count: int = 0
47
+ duplicate: bool = False
48
+ embedded: bool = False
49
+ indexing_status: str = "pending" # indexed | skipped | failed | pending
50
+ provenance_id: Optional[str] = None
51
+ detail: Optional[str] = None
52
+ # v9.8.0 additive quality fields — advisory only, never gate behavior.
53
+ extraction_quality: Optional[Dict[str, Any]] = None
54
+ warnings: List[str] = field(default_factory=list)
55
+ quality_gate: Optional[Dict[str, Any]] = None
56
+
57
+ def as_dict(self) -> Dict[str, Any]:
58
+ payload: Dict[str, Any] = {
59
+ "status": self.status,
60
+ "source_type": self.source_type,
61
+ "node_id": self.node_id,
62
+ "source_node_id": self.source_node_id,
63
+ "content_hash": self.content_hash,
64
+ "title": self.title,
65
+ "chunk_ids": self.chunk_ids,
66
+ "chunk_count": self.chunk_count,
67
+ "duplicate": self.duplicate,
68
+ "embedded": self.embedded,
69
+ "indexing_status": self.indexing_status,
70
+ "provenance_id": self.provenance_id,
71
+ "detail": self.detail,
72
+ }
73
+ # Additive keys only when populated so pre-v9.8 payloads are unchanged.
74
+ if self.extraction_quality is not None:
75
+ payload["extraction_quality"] = self.extraction_quality
76
+ if self.warnings:
77
+ payload["warnings"] = list(self.warnings)
78
+ if self.quality_gate is not None:
79
+ payload["quality_gate"] = self.quality_gate
80
+ return payload