ltcai 11.2.0 → 11.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (253) hide show
  1. package/README.md +50 -53
  2. package/docs/CHANGELOG.md +87 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/MULTI_AGENT_RUNTIME.md +1 -1
  6. package/docs/ONBOARDING.md +1 -1
  7. package/docs/OPERATIONS.md +6 -2
  8. package/docs/PERMISSION_MODE.md +1 -1
  9. package/docs/TRUST_MODEL.md +1 -1
  10. package/docs/WHY_LATTICE.md +1 -1
  11. package/docs/kg-schema.md +2 -2
  12. package/docs/v11.3.0_PLAN.md +202 -0
  13. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +181 -0
  14. package/docs/v11.5.0_RUST_COMPLETE_PLAN.md +145 -0
  15. package/lattice_brain/__init__.py +1 -1
  16. package/lattice_brain/graph/_kg_common/__init__.py +287 -0
  17. package/lattice_brain/graph/_kg_common/extraction.py +516 -0
  18. package/lattice_brain/graph/_kg_common/relations.py +161 -0
  19. package/lattice_brain/graph/_kg_common/text.py +479 -0
  20. package/lattice_brain/graph/discovery_index/__init__.py +35 -0
  21. package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
  22. package/lattice_brain/graph/discovery_index/extract.py +137 -0
  23. package/lattice_brain/graph/discovery_index/scan.py +411 -0
  24. package/lattice_brain/graph/discovery_index/upsert.py +495 -0
  25. package/lattice_brain/graph/projection/__init__.py +42 -0
  26. package/lattice_brain/graph/projection/curation.py +500 -0
  27. package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
  28. package/lattice_brain/graph/retrieval/__init__.py +54 -0
  29. package/lattice_brain/graph/retrieval/context.py +197 -0
  30. package/lattice_brain/graph/retrieval/graph_view.py +319 -0
  31. package/lattice_brain/graph/retrieval/hybrid.py +488 -0
  32. package/lattice_brain/graph/retrieval/maintenance.py +121 -0
  33. package/lattice_brain/graph/retrieval/signals.py +95 -0
  34. package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
  35. package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
  36. package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
  37. package/lattice_brain/graph/retrieval_vector/search.py +560 -0
  38. package/lattice_brain/graph/retrieval_vector/status.py +374 -0
  39. package/lattice_brain/ingestion/__init__.py +130 -0
  40. package/lattice_brain/ingestion/_contract.py +90 -0
  41. package/lattice_brain/ingestion/constants.py +127 -0
  42. package/lattice_brain/ingestion/folder_scan.py +57 -0
  43. package/lattice_brain/ingestion/folders.py +258 -0
  44. package/lattice_brain/ingestion/hashing.py +26 -0
  45. package/lattice_brain/ingestion/jobs_api.py +107 -0
  46. package/lattice_brain/ingestion/models.py +80 -0
  47. package/lattice_brain/ingestion/pipeline.py +486 -0
  48. package/lattice_brain/ingestion/quality.py +209 -0
  49. package/lattice_brain/ingestion/routing.py +295 -0
  50. package/lattice_brain/multimodal/__init__.py +164 -0
  51. package/lattice_brain/multimodal/audio.py +77 -0
  52. package/lattice_brain/multimodal/common.py +118 -0
  53. package/lattice_brain/multimodal/images.py +498 -0
  54. package/lattice_brain/multimodal/ports.py +169 -0
  55. package/lattice_brain/multimodal/video.py +410 -0
  56. package/lattice_brain/portability/__init__.py +90 -0
  57. package/lattice_brain/portability/_contract.py +42 -0
  58. package/lattice_brain/portability/backups.py +338 -0
  59. package/lattice_brain/portability/bundles.py +136 -0
  60. package/lattice_brain/portability/constants.py +93 -0
  61. package/lattice_brain/portability/fsops.py +138 -0
  62. package/lattice_brain/portability/service.py +41 -0
  63. package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
  64. package/lattice_brain/runtime/__init__.py +1 -1
  65. package/lattice_brain/runtime/multi_agent.py +1 -1
  66. package/latticeai/__init__.py +1 -1
  67. package/latticeai/api/chronicle.py +63 -0
  68. package/latticeai/api/index_jobs.py +145 -0
  69. package/latticeai/core/agent/__init__.py +93 -0
  70. package/latticeai/core/agent/_contract.py +79 -0
  71. package/latticeai/core/agent/context.py +57 -0
  72. package/latticeai/core/agent/deps.py +125 -0
  73. package/latticeai/core/agent/execution.py +622 -0
  74. package/latticeai/core/agent/planning.py +145 -0
  75. package/latticeai/core/agent/recovery.py +157 -0
  76. package/latticeai/core/agent/runtime.py +210 -0
  77. package/latticeai/core/agent/verification.py +231 -0
  78. package/latticeai/core/embedding_providers/__init__.py +151 -0
  79. package/latticeai/core/embedding_providers/base.py +199 -0
  80. package/latticeai/core/embedding_providers/captions.py +162 -0
  81. package/latticeai/core/embedding_providers/profiles.py +126 -0
  82. package/latticeai/core/embedding_providers/text.py +350 -0
  83. package/latticeai/core/embedding_providers/vision.py +352 -0
  84. package/latticeai/core/file_generation/__init__.py +115 -0
  85. package/latticeai/core/file_generation/bundles.py +76 -0
  86. package/latticeai/core/file_generation/extraction.py +154 -0
  87. package/latticeai/core/file_generation/inference.py +235 -0
  88. package/latticeai/core/file_generation/orchestration.py +152 -0
  89. package/latticeai/core/file_generation/prompting.py +117 -0
  90. package/latticeai/core/file_generation/repair.py +114 -0
  91. package/latticeai/core/file_generation/sanitize.py +61 -0
  92. package/latticeai/core/file_generation/validation.py +201 -0
  93. package/latticeai/core/legacy_compatibility.py +1 -1
  94. package/latticeai/core/marketplace.py +1 -1
  95. package/latticeai/core/messages.py +14 -0
  96. package/latticeai/core/workspace_os_constants.py +1 -1
  97. package/latticeai/integrations/telegram_bot/__init__.py +123 -0
  98. package/latticeai/integrations/telegram_bot/__main__.py +17 -0
  99. package/latticeai/integrations/telegram_bot/config.py +86 -0
  100. package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
  101. package/latticeai/integrations/telegram_bot/flows.py +478 -0
  102. package/latticeai/integrations/telegram_bot/helpers.py +322 -0
  103. package/latticeai/integrations/telegram_bot/screens.py +394 -0
  104. package/latticeai/models/router/__init__.py +88 -0
  105. package/latticeai/models/router/_contract.py +66 -0
  106. package/latticeai/models/router/branding.py +56 -0
  107. package/latticeai/models/router/catalog.py +69 -0
  108. package/latticeai/models/router/documents.py +199 -0
  109. package/latticeai/models/router/errors.py +37 -0
  110. package/latticeai/models/router/generation.py +258 -0
  111. package/latticeai/models/router/loading.py +291 -0
  112. package/latticeai/models/router/local_models.py +85 -0
  113. package/latticeai/models/router/registry.py +147 -0
  114. package/latticeai/runtime/build_phases/__init__.py +82 -0
  115. package/latticeai/runtime/build_phases/features.py +421 -0
  116. package/latticeai/runtime/build_phases/foundation.py +555 -0
  117. package/latticeai/runtime/build_phases/web.py +492 -0
  118. package/latticeai/runtime/runtime_context.py +1 -0
  119. package/latticeai/services/architecture_readiness.py +48 -19
  120. package/latticeai/services/brain_intelligence/__init__.py +58 -0
  121. package/latticeai/services/brain_intelligence/_contract.py +71 -0
  122. package/latticeai/services/brain_intelligence/consistency.py +193 -0
  123. package/latticeai/services/brain_intelligence/constants.py +47 -0
  124. package/latticeai/services/brain_intelligence/digest.py +258 -0
  125. package/latticeai/services/brain_intelligence/health.py +331 -0
  126. package/latticeai/services/brain_intelligence/proposals.py +264 -0
  127. package/latticeai/services/brain_intelligence/sampling.py +84 -0
  128. package/latticeai/services/brain_intelligence/service.py +48 -0
  129. package/latticeai/services/chronicle.py +557 -0
  130. package/latticeai/services/memory_service/__init__.py +52 -0
  131. package/latticeai/services/memory_service/_contract.py +100 -0
  132. package/latticeai/services/memory_service/brief.py +431 -0
  133. package/latticeai/services/memory_service/constants.py +57 -0
  134. package/latticeai/services/memory_service/maintenance.py +138 -0
  135. package/latticeai/services/memory_service/manager.py +186 -0
  136. package/latticeai/services/memory_service/proof.py +136 -0
  137. package/latticeai/services/memory_service/recall.py +225 -0
  138. package/latticeai/services/memory_service/service.py +48 -0
  139. package/latticeai/services/memory_service/stores.py +110 -0
  140. package/latticeai/services/model_runtime/__init__.py +322 -0
  141. package/latticeai/services/model_runtime/cloud.py +87 -0
  142. package/latticeai/services/model_runtime/download.py +282 -0
  143. package/latticeai/services/model_runtime/engines.py +341 -0
  144. package/latticeai/services/model_runtime/loading.py +178 -0
  145. package/latticeai/services/model_runtime/service.py +129 -0
  146. package/latticeai/services/model_runtime/state.py +131 -0
  147. package/latticeai/services/model_runtime/status.py +255 -0
  148. package/latticeai/services/product_readiness.py +15 -7
  149. package/latticeai/setup/wizard/__init__.py +126 -0
  150. package/latticeai/setup/wizard/catalog.py +172 -0
  151. package/latticeai/setup/wizard/detect.py +323 -0
  152. package/latticeai/setup/wizard/install.py +348 -0
  153. package/latticeai/setup/wizard/paths.py +168 -0
  154. package/latticeai/setup/wizard/plans.py +74 -0
  155. package/latticeai/setup/wizard/recommend.py +320 -0
  156. package/package.json +6 -2
  157. package/scripts/bump_version.py +14 -0
  158. package/scripts/capture_release_evidence.mjs +33 -21
  159. package/scripts/check_current_release_docs.mjs +1 -1
  160. package/scripts/check_i18n_namespace_coverage.mjs +41 -4
  161. package/scripts/check_max_file_lines.mjs +102 -0
  162. package/scripts/check_release_evidence_bound.mjs +30 -15
  163. package/scripts/check_screenshot_pixel_delta.py +34 -4
  164. package/scripts/check_server_i18n.mjs +2 -0
  165. package/scripts/chunking_parity_corpus.py +449 -0
  166. package/scripts/generate_agent_parity_fixtures.py +752 -0
  167. package/scripts/generate_chunking_parity_fixtures.py +259 -0
  168. package/scripts/generate_rust_parity_fixtures.py +997 -0
  169. package/scripts/lib/mock_server_fingerprint.mjs +94 -0
  170. package/scripts/release_screen_claims.json +42 -2
  171. package/src-tauri/Cargo.lock +404 -3
  172. package/src-tauri/Cargo.toml +13 -1
  173. package/src-tauri/src/backend.rs +460 -0
  174. package/src-tauri/src/folder.rs +33 -0
  175. package/src-tauri/src/main.rs +109 -399
  176. package/src-tauri/src/topology.rs +356 -0
  177. package/src-tauri/tauri.conf.json +1 -1
  178. package/static/app/asset-manifest.json +41 -37
  179. package/static/app/assets/Act-CWnxSCgN.js +1 -0
  180. package/static/app/assets/AdminConsole-BEQYU6kF.js +1 -0
  181. package/static/app/assets/{Brain-tuhI4sOC.js → Brain-DWu1BhFg.js} +2 -2
  182. package/static/app/assets/BrainHome-95Hilr9R.js +2 -0
  183. package/static/app/assets/BrainSignals-QdeqCpAF.js +1 -0
  184. package/static/app/assets/Capture-BHpCxnzb.js +1 -0
  185. package/static/app/assets/Chronicle-B4xYKoed.js +1 -0
  186. package/static/app/assets/CommandPalette-BVXnttSz.js +1 -0
  187. package/static/app/assets/Library-DgYcHome.js +1 -0
  188. package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-CrJLDbf7.js} +1 -1
  189. package/static/app/assets/ProductFlow-DFlScKoJ.js +1 -0
  190. package/static/app/assets/ReviewCard-Cy5f48Pj.js +3 -0
  191. package/static/app/assets/System-NF8IfhTa.js +1 -0
  192. package/static/app/assets/arrow-left-DwkSYrjR.js +1 -0
  193. package/static/app/assets/{bot-Cia42c2h.js → bot-CucuhLhm.js} +1 -1
  194. package/static/app/assets/brain-BBnSryW_.js +1 -0
  195. package/static/app/assets/{button-2j2Ijzgq.js → button-C2GUj2Ai.js} +1 -1
  196. package/static/app/assets/circle-check-CxOVPwYq.js +1 -0
  197. package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-CbkWzBmG.js} +1 -1
  198. package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-7lEaqHdJ.js} +1 -1
  199. package/static/app/assets/{cpu-k4awryFq.js → cpu-DAlCXlIy.js} +1 -1
  200. package/static/app/assets/{download-DFbLJ_ig.js → download-RNhuuJwh.js} +1 -1
  201. package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CLW4odzM.js} +1 -1
  202. package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-NKEiDIAJ.js} +1 -1
  203. package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
  204. package/static/app/assets/index-DMurvUuR.js +10 -0
  205. package/static/app/assets/input-D2UhPC1X.js +1 -0
  206. package/static/app/assets/link-2-6amKbP_P.js +1 -0
  207. package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-Cu9TZtdR.js} +1 -1
  208. package/static/app/assets/primitives-gPsccucr.js +1 -0
  209. package/static/app/assets/search-Cj_TKk_2.js +1 -0
  210. package/static/app/assets/{share-2-BH1M-WNi.js → share-2-Bau7KkPq.js} +1 -1
  211. package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-BufNYypi.js} +1 -1
  212. package/static/app/assets/{textarea-CCWbUfFB.js → textarea-BQnVWhYs.js} +1 -1
  213. package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-B3_w60si.js} +1 -1
  214. package/static/app/assets/useMutation-BHhCflT6.js +1 -0
  215. package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-rBWfI-5t.js} +1 -1
  216. package/static/app/assets/utils-V_5-wxr5.js +4 -0
  217. package/static/app/assets/workspace-K1zjYUHj.js +1 -0
  218. package/static/app/index.html +4 -4
  219. package/static/sw.js +1 -1
  220. package/lattice_brain/graph/_kg_common.py +0 -1331
  221. package/lattice_brain/graph/discovery_index.py +0 -1141
  222. package/lattice_brain/graph/retrieval.py +0 -1120
  223. package/lattice_brain/graph/retrieval_vector.py +0 -1293
  224. package/lattice_brain/ingestion.py +0 -1525
  225. package/lattice_brain/multimodal.py +0 -1258
  226. package/latticeai/core/agent.py +0 -1465
  227. package/latticeai/core/embedding_providers.py +0 -1196
  228. package/latticeai/core/file_generation.py +0 -1047
  229. package/latticeai/integrations/telegram_bot.py +0 -1390
  230. package/latticeai/models/router.py +0 -1007
  231. package/latticeai/runtime/build_phases.py +0 -1450
  232. package/latticeai/services/brain_intelligence.py +0 -1083
  233. package/latticeai/services/memory_service.py +0 -1177
  234. package/latticeai/services/model_runtime.py +0 -1281
  235. package/latticeai/setup/wizard.py +0 -1310
  236. package/static/app/assets/Act-AWf0SAKp.js +0 -1
  237. package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
  238. package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
  239. package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
  240. package/static/app/assets/Capture-CqOSzyPr.js +0 -1
  241. package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
  242. package/static/app/assets/Library-CX-bbhmK.js +0 -1
  243. package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
  244. package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
  245. package/static/app/assets/System-Bu2t5hn1.js +0 -1
  246. package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
  247. package/static/app/assets/brain-DJMoqrwx.js +0 -1
  248. package/static/app/assets/index-BpYkzcVm.js +0 -10
  249. package/static/app/assets/input-DSlJJxRs.js +0 -1
  250. package/static/app/assets/primitives-BCx6TvfG.js +0 -1
  251. package/static/app/assets/search-Cgy8cCFJ.js +0 -1
  252. package/static/app/assets/utils-zqPZJxdx.js +0 -4
  253. package/static/app/assets/workspace-DXTihhfU.js +0 -1
@@ -0,0 +1,137 @@
1
+ """Local file → text: the parsers behind local indexing.
2
+
3
+ PDF/Word/Excel/PowerPoint text plus the image signal path (dimensions, OCR,
4
+ and — only when a VLM exists — a caption). Moved verbatim out of
5
+ ``discovery_index.py`` (v11.3.0 decomposition).
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from typing import TYPE_CHECKING
11
+
12
+ # ruff: noqa: F403,F405
13
+ from .._kg_common import * # noqa: F403,F401
14
+
15
+ # The cross-mixin surface (`_connect`, `_upsert_node`, …) is declared in
16
+ # `_kg_contract.KnowledgeGraphCore`. It is a typing-only base: at runtime this
17
+ # is `object`, so the MRO of `KnowledgeGraphStore` is unchanged.
18
+ if TYPE_CHECKING:
19
+ from .._kg_contract import KnowledgeGraphCore as _Core
20
+ else:
21
+ _Core = object
22
+
23
+
24
+ class _LocalExtractMixin(_Core):
25
+ """File-text and image-signal extraction. Composed into the public mixin."""
26
+
27
+ def _extract_local_file_text(
28
+ self, path: Path, category: str, *, include_ocr: bool
29
+ ) -> Tuple[str, Dict[str, Any]]:
30
+ ext = path.suffix.lower()
31
+ meta: Dict[str, Any] = {"parser": _parser_type_for_category(category, ext)}
32
+ text = ""
33
+ if category in {"text", "code"} or ext == ".csv":
34
+ text = path.read_text(encoding="utf-8", errors="replace")
35
+ elif ext == ".pdf":
36
+ import pdfplumber
37
+
38
+ with pdfplumber.open(str(path)) as pdf:
39
+ meta["pages"] = len(pdf.pages)
40
+ text = "\n\n".join((page.extract_text() or "") for page in pdf.pages)
41
+ elif ext == ".docx":
42
+ from docx import Document
43
+
44
+ doc = Document(str(path))
45
+ paragraphs = [p.text for p in doc.paragraphs if p.text.strip()]
46
+ table_lines = []
47
+ for table in doc.tables:
48
+ for row in table.rows:
49
+ cells = [_clean_text(cell.text) for cell in row.cells]
50
+ if any(cells):
51
+ table_lines.append("\t".join(cells))
52
+ meta["paragraphs"] = len(paragraphs)
53
+ meta["tables"] = len(doc.tables)
54
+ meta["table_rows"] = len(table_lines)
55
+ text = "\n\n".join([*paragraphs, *table_lines])
56
+ elif ext == ".xlsx":
57
+ from openpyxl import load_workbook
58
+
59
+ wb = load_workbook(str(path), read_only=True, data_only=True)
60
+ rows_all = []
61
+ non_empty_rows = 0
62
+ non_empty_cells = 0
63
+ char_count = 0
64
+ for ws in wb.worksheets:
65
+ sheet_rows = []
66
+ for row in ws.iter_rows(values_only=True):
67
+ cells = [
68
+ str(cell).strip() if cell is not None else "" for cell in row
69
+ ]
70
+ if not any(cells):
71
+ continue
72
+ line = "\t".join(cells)
73
+ non_empty_rows += 1
74
+ non_empty_cells += sum(1 for cell in cells if cell)
75
+ sheet_rows.append(line)
76
+ char_count += len(line) + 1
77
+ if char_count > 200_000:
78
+ break
79
+ if sheet_rows:
80
+ rows_all.append(f"[Sheet: {ws.title}]")
81
+ rows_all.extend(sheet_rows)
82
+ if char_count > 200_000:
83
+ break
84
+ meta["sheets"] = len(wb.worksheets)
85
+ meta["rows"] = non_empty_rows
86
+ meta["cells"] = non_empty_cells
87
+ text = "\n".join(rows_all)
88
+ elif ext == ".pptx":
89
+ from pptx import Presentation
90
+
91
+ prs = Presentation(str(path))
92
+ slides_text = []
93
+ for index, slide in enumerate(prs.slides, 1):
94
+ parts = []
95
+ for shape in slide.shapes:
96
+ if getattr(shape, "has_text_frame", False):
97
+ slide_text = shape.text_frame.text.strip()
98
+ if slide_text:
99
+ parts.append(slide_text)
100
+ if parts:
101
+ slides_text.append(f"[Slide {index}]\n" + "\n".join(parts))
102
+ meta["slides"] = len(prs.slides)
103
+ meta["text_slides"] = len(slides_text)
104
+ text = "\n\n".join(slides_text)
105
+ elif category == "image":
106
+ text = self._extract_image_signals(path, meta, include_ocr=include_ocr)
107
+ return text[:200_000], meta
108
+
109
+ def _extract_image_signals(
110
+ self, path: Path, meta: Dict[str, Any], *, include_ocr: bool
111
+ ) -> str:
112
+ """Dimensions, OCR text, and — only if a VLM exists — a caption.
113
+
114
+ Until v11.1.0 this path always attached a ``vision_caption`` built out
115
+ of the filename and the pixel dimensions (``Image pic.png (PNG 12x8)``)
116
+ and used it as the retrieval text. Nothing downstream could tell that
117
+ string apart from something a vision model had actually said about the
118
+ picture, so every screenshot in the graph carried a fake description.
119
+
120
+ Now the caption comes from the injected port and from nowhere else. A
121
+ picture with no OCR text and no model still gets indexed — under its
122
+ filename, which is a fact — and ``caption_status`` says why there is no
123
+ caption.
124
+ """
125
+ from ...multimodal import MultimodalPorts, extract_image_facts
126
+
127
+ ports = getattr(self, "multimodal_ports", None) or MultimodalPorts()
128
+ facts = extract_image_facts(str(path), ports=ports, ocr=include_ocr)
129
+ meta.update(facts.as_metadata())
130
+ meta["ocr_enabled"] = bool(include_ocr)
131
+ if facts.ocr_text:
132
+ meta["ocr_chars"] = len(facts.ocr_text)
133
+ if facts.ocr_status == "failed":
134
+ meta["ocr_error"] = facts.ocr_detail
135
+ if not facts.readable:
136
+ return ""
137
+ return facts.index_text() or path.name
@@ -0,0 +1,411 @@
1
+ """``index_local_folder``: the driver that walks a folder into the graph.
2
+
3
+ Decides per file whether to skip, refresh, or fully re-index, and reports
4
+ counts, errors, and the honest notice about what was converted. Moved
5
+ verbatim out of ``discovery_index.py`` (v11.3.0 decomposition).
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from typing import TYPE_CHECKING
11
+
12
+ # ruff: noqa: F403,F405
13
+ from .._kg_common import * # noqa: F403,F401
14
+
15
+ # Typing-only base (runtime value is `object`, so the store's MRO is
16
+ # unchanged). The driver calls all three other halves through `self` — text
17
+ # extraction, the row/node upserts, and the skip checks — so they are named as
18
+ # bases rather than re-declared here, where their signatures could drift. The
19
+ # store contract (`_connect`, `_upsert_node`, …) arrives with them.
20
+ if TYPE_CHECKING:
21
+ from .cleanup import _LocalCleanupMixin
22
+ from .extract import _LocalExtractMixin
23
+ from .upsert import _LocalUpsertMixin
24
+
25
+ class _Core(_LocalExtractMixin, _LocalUpsertMixin, _LocalCleanupMixin):
26
+ """The sibling halves this one reaches through ``self``."""
27
+ else:
28
+ _Core = object
29
+
30
+
31
+ class _LocalScanMixin(_Core):
32
+ """The folder-indexing driver. Composed into the public mixin."""
33
+
34
+ def index_local_folder(
35
+ self,
36
+ path: Path,
37
+ *,
38
+ include_ocr: bool = False,
39
+ watch_enabled: bool = False,
40
+ user_email: Optional[str] = None,
41
+ workspace_id: Optional[str] = None,
42
+ consent: Optional[Dict[str, Any]] = None,
43
+ max_files: int = 5_000,
44
+ source_id_override: Optional[str] = None,
45
+ ) -> Dict[str, Any]:
46
+ """Read approved files from a local folder and connect them to Graph RAG."""
47
+ root = Path(path).expanduser().resolve()
48
+ if not root.exists():
49
+ raise ValueError(f"경로가 존재하지 않습니다: {path}")
50
+ if not root.is_dir():
51
+ raise ValueError(f"폴더가 아닙니다: {path}")
52
+
53
+ os_type = _current_os_type()
54
+ drive_id = _drive_id_for_path(root)
55
+ path_fingerprint = _path_fingerprint(root)
56
+ source_id = str(source_id_override or "").strip()
57
+ if not source_id:
58
+ source_id = (
59
+ f"source:{_sha256_text(f'{workspace_id}|{path_fingerprint}')[:24]}"
60
+ if workspace_id
61
+ else f"source:{path_fingerprint}"
62
+ )
63
+ now = _now()
64
+ max_files = max(1, min(int(max_files or 5_000), 50_000))
65
+ consent_payload = {
66
+ "approved_at": now,
67
+ "knowledge_source": True,
68
+ "include_ocr": bool(include_ocr),
69
+ "watch_enabled": bool(watch_enabled),
70
+ "sensitive_files_default_excluded": True,
71
+ **(consent or {}),
72
+ "approved_by": user_email or (consent or {}).get("approved_by"),
73
+ "workspace_id": workspace_id or (consent or {}).get("workspace_id"),
74
+ }
75
+ counts: Counter = Counter()
76
+ seen_relative_paths: set = set()
77
+ indexed_nodes: List[str] = []
78
+ errors: List[Dict[str, str]] = []
79
+ limit_reached = False
80
+
81
+ with self._connect() as conn:
82
+ existing_source = conn.execute(
83
+ "SELECT id, consent_json FROM knowledge_sources WHERE root_path=?",
84
+ (str(root),),
85
+ ).fetchone()
86
+ if existing_source is not None:
87
+ existing_consent = _safe_loads(existing_source["consent_json"])
88
+ existing_scope = existing_consent.get("workspace_id") or "personal"
89
+ requested_scope = workspace_id or consent_payload.get("workspace_id") or "personal"
90
+ if existing_scope != requested_scope:
91
+ raise ValueError(
92
+ "This folder is already connected to another workspace. "
93
+ "Disconnect it there before assigning it to a different Brain."
94
+ )
95
+ if existing_source["id"] != source_id:
96
+ # Reuse the legacy source identity so a personal source is
97
+ # reprojected in place instead of duplicated during upgrade.
98
+ source_id = existing_source["id"]
99
+ conn.execute(
100
+ """
101
+ INSERT INTO knowledge_sources(
102
+ id, root_path, os_type, drive_id, label, status, include_ocr,
103
+ watch_enabled, consent_json, created_at, updated_at, last_scanned_at
104
+ )
105
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
106
+ ON CONFLICT(id) DO UPDATE SET
107
+ root_path=excluded.root_path,
108
+ os_type=excluded.os_type,
109
+ drive_id=excluded.drive_id,
110
+ label=excluded.label,
111
+ status=excluded.status,
112
+ include_ocr=excluded.include_ocr,
113
+ watch_enabled=excluded.watch_enabled,
114
+ consent_json=excluded.consent_json,
115
+ updated_at=excluded.updated_at,
116
+ last_scanned_at=excluded.last_scanned_at
117
+ """,
118
+ (
119
+ source_id,
120
+ str(root),
121
+ os_type,
122
+ drive_id,
123
+ root.name or str(root),
124
+ "scanning",
125
+ 1 if include_ocr else 0,
126
+ 1 if watch_enabled else 0,
127
+ _json(consent_payload),
128
+ now,
129
+ now,
130
+ now,
131
+ ),
132
+ )
133
+
134
+ for entry in self._iter_local_scan_entries(root, max_files=max_files):
135
+ kind = entry["kind"]
136
+ file_path = entry["path"]
137
+ if kind == "limit_reached":
138
+ counts["limit_reached"] += 1
139
+ limit_reached = True
140
+ break
141
+ if kind in {"excluded_dir", "excluded"}:
142
+ counts["excluded"] += 1
143
+ continue
144
+ if kind in {"inaccessible_dir", "inaccessible_file"}:
145
+ counts["failed"] += 1
146
+ errors.append(
147
+ {
148
+ "path": str(file_path),
149
+ "error": entry.get("reason", "inaccessible"),
150
+ }
151
+ )
152
+ continue
153
+ if kind != "file":
154
+ continue
155
+
156
+ stat = entry["stat"]
157
+ try:
158
+ relative_path = file_path.relative_to(root).as_posix()
159
+ except ValueError:
160
+ relative_path = file_path.name
161
+ seen_relative_paths.add(relative_path)
162
+ modified_at = _safe_iso_from_stat_mtime(stat.st_mtime)
163
+ existing = conn.execute(
164
+ """
165
+ SELECT size_bytes, modified_at, sha256, graph_node_id, status, metadata_json
166
+ FROM local_file_index
167
+ WHERE source_id=? AND relative_path=?
168
+ """,
169
+ (source_id, relative_path),
170
+ ).fetchone()
171
+ decision = self._local_file_decision(file_path, root, stat)
172
+ parser_type = decision["parser_type"]
173
+ if not decision["indexable"]:
174
+ counts[decision["status"]] += 1
175
+ if existing and existing["graph_node_id"]:
176
+ self._delete_local_file_graph(conn, existing["graph_node_id"])
177
+ self._upsert_local_file_index(
178
+ conn,
179
+ source_id=source_id,
180
+ root=root,
181
+ file_path=file_path,
182
+ stat=stat,
183
+ os_type=os_type,
184
+ drive_id=drive_id,
185
+ status=decision["status"],
186
+ parser_type=parser_type,
187
+ metadata={
188
+ "reason": decision["reason"],
189
+ "category": decision["category"],
190
+ },
191
+ )
192
+ continue
193
+
194
+ if (
195
+ existing
196
+ and existing["status"] == "indexed"
197
+ and existing["graph_node_id"]
198
+ and self._local_file_index_has_extracted_text(existing)
199
+ and self._node_matches_workspace(
200
+ conn, existing["graph_node_id"], workspace_id
201
+ )
202
+ and existing["size_bytes"] == stat.st_size
203
+ and existing["modified_at"] == modified_at
204
+ ):
205
+ counts["skipped_unchanged"] += 1
206
+ self._upsert_local_file_index(
207
+ conn,
208
+ source_id=source_id,
209
+ root=root,
210
+ file_path=file_path,
211
+ stat=stat,
212
+ os_type=os_type,
213
+ drive_id=drive_id,
214
+ status="indexed",
215
+ parser_type=parser_type,
216
+ sha256=existing["sha256"],
217
+ graph_node_id=existing["graph_node_id"],
218
+ metadata={
219
+ **_safe_loads(existing["metadata_json"]),
220
+ "category": decision["category"],
221
+ "unchanged": True,
222
+ },
223
+ )
224
+ continue
225
+
226
+ try:
227
+ data = file_path.read_bytes()
228
+ digest = _sha256_bytes(data)
229
+ except Exception as exc:
230
+ counts["failed"] += 1
231
+ errors.append({"path": str(file_path), "error": str(exc)})
232
+ if existing and existing["graph_node_id"]:
233
+ self._delete_local_file_graph(conn, existing["graph_node_id"])
234
+ self._upsert_local_file_index(
235
+ conn,
236
+ source_id=source_id,
237
+ root=root,
238
+ file_path=file_path,
239
+ stat=stat,
240
+ os_type=os_type,
241
+ drive_id=drive_id,
242
+ status="failed",
243
+ parser_type=parser_type,
244
+ error_message=str(exc),
245
+ metadata={"category": decision["category"]},
246
+ )
247
+ continue
248
+
249
+ if (
250
+ existing
251
+ and existing["sha256"] == digest
252
+ and existing["graph_node_id"]
253
+ and self._local_file_index_has_extracted_text(existing)
254
+ and self._node_matches_workspace(
255
+ conn, existing["graph_node_id"], workspace_id
256
+ )
257
+ ):
258
+ counts["skipped_unchanged"] += 1
259
+ self._upsert_local_file_index(
260
+ conn,
261
+ source_id=source_id,
262
+ root=root,
263
+ file_path=file_path,
264
+ stat=stat,
265
+ os_type=os_type,
266
+ drive_id=drive_id,
267
+ status="indexed",
268
+ parser_type=parser_type,
269
+ sha256=digest,
270
+ graph_node_id=existing["graph_node_id"],
271
+ metadata={
272
+ **_safe_loads(existing["metadata_json"]),
273
+ "category": decision["category"],
274
+ "sha256_unchanged": True,
275
+ },
276
+ )
277
+ continue
278
+
279
+ try:
280
+ text, parser_meta = self._extract_local_file_text(
281
+ file_path,
282
+ decision["category"],
283
+ include_ocr=include_ocr,
284
+ )
285
+ text = _clean_text(text)
286
+ parser_meta = {**parser_meta, "extracted_chars": len(text)}
287
+ if not text:
288
+ counts["skipped_empty_text"] += 1
289
+ if existing and existing["graph_node_id"]:
290
+ self._delete_local_file_graph(
291
+ conn, existing["graph_node_id"]
292
+ )
293
+ self._upsert_local_file_index(
294
+ conn,
295
+ source_id=source_id,
296
+ root=root,
297
+ file_path=file_path,
298
+ stat=stat,
299
+ os_type=os_type,
300
+ drive_id=drive_id,
301
+ status="skipped_empty_text",
302
+ parser_type=parser_type,
303
+ sha256=digest,
304
+ error_message="텍스트 추출 결과가 비어 있습니다.",
305
+ metadata={
306
+ "category": decision["category"],
307
+ "parser": parser_meta,
308
+ },
309
+ )
310
+ continue
311
+ graph_node_id = self._upsert_local_file_node(
312
+ conn,
313
+ source_id=source_id,
314
+ root=root,
315
+ file_path=file_path,
316
+ stat=stat,
317
+ os_type=os_type,
318
+ drive_id=drive_id,
319
+ sha256=digest,
320
+ category=decision["category"],
321
+ parser_type=parser_type,
322
+ text=text,
323
+ parser_meta=parser_meta,
324
+ user_email=user_email,
325
+ workspace_id=workspace_id,
326
+ )
327
+ self._upsert_local_file_index(
328
+ conn,
329
+ source_id=source_id,
330
+ root=root,
331
+ file_path=file_path,
332
+ stat=stat,
333
+ os_type=os_type,
334
+ drive_id=drive_id,
335
+ status="indexed",
336
+ parser_type=parser_type,
337
+ sha256=digest,
338
+ graph_node_id=graph_node_id,
339
+ metadata={
340
+ "category": decision["category"],
341
+ "parser": parser_meta,
342
+ },
343
+ )
344
+ counts["indexed"] += 1
345
+ indexed_nodes.append(graph_node_id)
346
+ except Exception as exc:
347
+ counts["failed"] += 1
348
+ errors.append({"path": str(file_path), "error": str(exc)})
349
+ if existing and existing["graph_node_id"]:
350
+ self._delete_local_file_graph(conn, existing["graph_node_id"])
351
+ self._upsert_local_file_index(
352
+ conn,
353
+ source_id=source_id,
354
+ root=root,
355
+ file_path=file_path,
356
+ stat=stat,
357
+ os_type=os_type,
358
+ drive_id=drive_id,
359
+ status="failed",
360
+ parser_type=parser_type,
361
+ sha256=digest,
362
+ error_message=str(exc),
363
+ metadata={"category": decision["category"]},
364
+ )
365
+
366
+ if not limit_reached:
367
+ existing_rows = {
368
+ row["relative_path"]: row["graph_node_id"]
369
+ for row in conn.execute(
370
+ "SELECT relative_path, graph_node_id FROM local_file_index WHERE source_id=?",
371
+ (source_id,),
372
+ )
373
+ }
374
+ deleted_paths = set(existing_rows) - seen_relative_paths
375
+ for relative_path in deleted_paths:
376
+ self._delete_local_file_graph(
377
+ conn, existing_rows.get(relative_path)
378
+ )
379
+ conn.execute(
380
+ """
381
+ UPDATE local_file_index
382
+ SET status='deleted', deleted=1, last_scanned_at=?, error_message=NULL, graph_node_id=NULL
383
+ WHERE source_id=? AND relative_path=?
384
+ """,
385
+ (_now(), source_id, relative_path),
386
+ )
387
+ counts["deleted"] = len(deleted_paths)
388
+ conn.execute(
389
+ """
390
+ UPDATE knowledge_sources
391
+ SET status='active', updated_at=?, last_scanned_at=?
392
+ WHERE id=?
393
+ """,
394
+ (_now(), _now(), source_id),
395
+ )
396
+
397
+ return {
398
+ "status": "ok",
399
+ "source": {
400
+ "id": source_id,
401
+ "root_path": str(root),
402
+ "os_type": os_type,
403
+ "drive_id": drive_id,
404
+ "include_ocr": bool(include_ocr),
405
+ "watch_enabled": bool(watch_enabled),
406
+ },
407
+ "counts": dict(counts),
408
+ "indexed_nodes": indexed_nodes[:100],
409
+ "errors": errors[:50],
410
+ "notice": "Lattice AI는 사용자가 선택한 폴더만 AI 지식으로 변환합니다.",
411
+ }