ltcai 11.2.0 → 11.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (253) hide show
  1. package/README.md +50 -53
  2. package/docs/CHANGELOG.md +87 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/MULTI_AGENT_RUNTIME.md +1 -1
  6. package/docs/ONBOARDING.md +1 -1
  7. package/docs/OPERATIONS.md +6 -2
  8. package/docs/PERMISSION_MODE.md +1 -1
  9. package/docs/TRUST_MODEL.md +1 -1
  10. package/docs/WHY_LATTICE.md +1 -1
  11. package/docs/kg-schema.md +2 -2
  12. package/docs/v11.3.0_PLAN.md +202 -0
  13. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +181 -0
  14. package/docs/v11.5.0_RUST_COMPLETE_PLAN.md +145 -0
  15. package/lattice_brain/__init__.py +1 -1
  16. package/lattice_brain/graph/_kg_common/__init__.py +287 -0
  17. package/lattice_brain/graph/_kg_common/extraction.py +516 -0
  18. package/lattice_brain/graph/_kg_common/relations.py +161 -0
  19. package/lattice_brain/graph/_kg_common/text.py +479 -0
  20. package/lattice_brain/graph/discovery_index/__init__.py +35 -0
  21. package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
  22. package/lattice_brain/graph/discovery_index/extract.py +137 -0
  23. package/lattice_brain/graph/discovery_index/scan.py +411 -0
  24. package/lattice_brain/graph/discovery_index/upsert.py +495 -0
  25. package/lattice_brain/graph/projection/__init__.py +42 -0
  26. package/lattice_brain/graph/projection/curation.py +500 -0
  27. package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
  28. package/lattice_brain/graph/retrieval/__init__.py +54 -0
  29. package/lattice_brain/graph/retrieval/context.py +197 -0
  30. package/lattice_brain/graph/retrieval/graph_view.py +319 -0
  31. package/lattice_brain/graph/retrieval/hybrid.py +488 -0
  32. package/lattice_brain/graph/retrieval/maintenance.py +121 -0
  33. package/lattice_brain/graph/retrieval/signals.py +95 -0
  34. package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
  35. package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
  36. package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
  37. package/lattice_brain/graph/retrieval_vector/search.py +560 -0
  38. package/lattice_brain/graph/retrieval_vector/status.py +374 -0
  39. package/lattice_brain/ingestion/__init__.py +130 -0
  40. package/lattice_brain/ingestion/_contract.py +90 -0
  41. package/lattice_brain/ingestion/constants.py +127 -0
  42. package/lattice_brain/ingestion/folder_scan.py +57 -0
  43. package/lattice_brain/ingestion/folders.py +258 -0
  44. package/lattice_brain/ingestion/hashing.py +26 -0
  45. package/lattice_brain/ingestion/jobs_api.py +107 -0
  46. package/lattice_brain/ingestion/models.py +80 -0
  47. package/lattice_brain/ingestion/pipeline.py +486 -0
  48. package/lattice_brain/ingestion/quality.py +209 -0
  49. package/lattice_brain/ingestion/routing.py +295 -0
  50. package/lattice_brain/multimodal/__init__.py +164 -0
  51. package/lattice_brain/multimodal/audio.py +77 -0
  52. package/lattice_brain/multimodal/common.py +118 -0
  53. package/lattice_brain/multimodal/images.py +498 -0
  54. package/lattice_brain/multimodal/ports.py +169 -0
  55. package/lattice_brain/multimodal/video.py +410 -0
  56. package/lattice_brain/portability/__init__.py +90 -0
  57. package/lattice_brain/portability/_contract.py +42 -0
  58. package/lattice_brain/portability/backups.py +338 -0
  59. package/lattice_brain/portability/bundles.py +136 -0
  60. package/lattice_brain/portability/constants.py +93 -0
  61. package/lattice_brain/portability/fsops.py +138 -0
  62. package/lattice_brain/portability/service.py +41 -0
  63. package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
  64. package/lattice_brain/runtime/__init__.py +1 -1
  65. package/lattice_brain/runtime/multi_agent.py +1 -1
  66. package/latticeai/__init__.py +1 -1
  67. package/latticeai/api/chronicle.py +63 -0
  68. package/latticeai/api/index_jobs.py +145 -0
  69. package/latticeai/core/agent/__init__.py +93 -0
  70. package/latticeai/core/agent/_contract.py +79 -0
  71. package/latticeai/core/agent/context.py +57 -0
  72. package/latticeai/core/agent/deps.py +125 -0
  73. package/latticeai/core/agent/execution.py +622 -0
  74. package/latticeai/core/agent/planning.py +145 -0
  75. package/latticeai/core/agent/recovery.py +157 -0
  76. package/latticeai/core/agent/runtime.py +210 -0
  77. package/latticeai/core/agent/verification.py +231 -0
  78. package/latticeai/core/embedding_providers/__init__.py +151 -0
  79. package/latticeai/core/embedding_providers/base.py +199 -0
  80. package/latticeai/core/embedding_providers/captions.py +162 -0
  81. package/latticeai/core/embedding_providers/profiles.py +126 -0
  82. package/latticeai/core/embedding_providers/text.py +350 -0
  83. package/latticeai/core/embedding_providers/vision.py +352 -0
  84. package/latticeai/core/file_generation/__init__.py +115 -0
  85. package/latticeai/core/file_generation/bundles.py +76 -0
  86. package/latticeai/core/file_generation/extraction.py +154 -0
  87. package/latticeai/core/file_generation/inference.py +235 -0
  88. package/latticeai/core/file_generation/orchestration.py +152 -0
  89. package/latticeai/core/file_generation/prompting.py +117 -0
  90. package/latticeai/core/file_generation/repair.py +114 -0
  91. package/latticeai/core/file_generation/sanitize.py +61 -0
  92. package/latticeai/core/file_generation/validation.py +201 -0
  93. package/latticeai/core/legacy_compatibility.py +1 -1
  94. package/latticeai/core/marketplace.py +1 -1
  95. package/latticeai/core/messages.py +14 -0
  96. package/latticeai/core/workspace_os_constants.py +1 -1
  97. package/latticeai/integrations/telegram_bot/__init__.py +123 -0
  98. package/latticeai/integrations/telegram_bot/__main__.py +17 -0
  99. package/latticeai/integrations/telegram_bot/config.py +86 -0
  100. package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
  101. package/latticeai/integrations/telegram_bot/flows.py +478 -0
  102. package/latticeai/integrations/telegram_bot/helpers.py +322 -0
  103. package/latticeai/integrations/telegram_bot/screens.py +394 -0
  104. package/latticeai/models/router/__init__.py +88 -0
  105. package/latticeai/models/router/_contract.py +66 -0
  106. package/latticeai/models/router/branding.py +56 -0
  107. package/latticeai/models/router/catalog.py +69 -0
  108. package/latticeai/models/router/documents.py +199 -0
  109. package/latticeai/models/router/errors.py +37 -0
  110. package/latticeai/models/router/generation.py +258 -0
  111. package/latticeai/models/router/loading.py +291 -0
  112. package/latticeai/models/router/local_models.py +85 -0
  113. package/latticeai/models/router/registry.py +147 -0
  114. package/latticeai/runtime/build_phases/__init__.py +82 -0
  115. package/latticeai/runtime/build_phases/features.py +421 -0
  116. package/latticeai/runtime/build_phases/foundation.py +555 -0
  117. package/latticeai/runtime/build_phases/web.py +492 -0
  118. package/latticeai/runtime/runtime_context.py +1 -0
  119. package/latticeai/services/architecture_readiness.py +48 -19
  120. package/latticeai/services/brain_intelligence/__init__.py +58 -0
  121. package/latticeai/services/brain_intelligence/_contract.py +71 -0
  122. package/latticeai/services/brain_intelligence/consistency.py +193 -0
  123. package/latticeai/services/brain_intelligence/constants.py +47 -0
  124. package/latticeai/services/brain_intelligence/digest.py +258 -0
  125. package/latticeai/services/brain_intelligence/health.py +331 -0
  126. package/latticeai/services/brain_intelligence/proposals.py +264 -0
  127. package/latticeai/services/brain_intelligence/sampling.py +84 -0
  128. package/latticeai/services/brain_intelligence/service.py +48 -0
  129. package/latticeai/services/chronicle.py +557 -0
  130. package/latticeai/services/memory_service/__init__.py +52 -0
  131. package/latticeai/services/memory_service/_contract.py +100 -0
  132. package/latticeai/services/memory_service/brief.py +431 -0
  133. package/latticeai/services/memory_service/constants.py +57 -0
  134. package/latticeai/services/memory_service/maintenance.py +138 -0
  135. package/latticeai/services/memory_service/manager.py +186 -0
  136. package/latticeai/services/memory_service/proof.py +136 -0
  137. package/latticeai/services/memory_service/recall.py +225 -0
  138. package/latticeai/services/memory_service/service.py +48 -0
  139. package/latticeai/services/memory_service/stores.py +110 -0
  140. package/latticeai/services/model_runtime/__init__.py +322 -0
  141. package/latticeai/services/model_runtime/cloud.py +87 -0
  142. package/latticeai/services/model_runtime/download.py +282 -0
  143. package/latticeai/services/model_runtime/engines.py +341 -0
  144. package/latticeai/services/model_runtime/loading.py +178 -0
  145. package/latticeai/services/model_runtime/service.py +129 -0
  146. package/latticeai/services/model_runtime/state.py +131 -0
  147. package/latticeai/services/model_runtime/status.py +255 -0
  148. package/latticeai/services/product_readiness.py +15 -7
  149. package/latticeai/setup/wizard/__init__.py +126 -0
  150. package/latticeai/setup/wizard/catalog.py +172 -0
  151. package/latticeai/setup/wizard/detect.py +323 -0
  152. package/latticeai/setup/wizard/install.py +348 -0
  153. package/latticeai/setup/wizard/paths.py +168 -0
  154. package/latticeai/setup/wizard/plans.py +74 -0
  155. package/latticeai/setup/wizard/recommend.py +320 -0
  156. package/package.json +6 -2
  157. package/scripts/bump_version.py +14 -0
  158. package/scripts/capture_release_evidence.mjs +33 -21
  159. package/scripts/check_current_release_docs.mjs +1 -1
  160. package/scripts/check_i18n_namespace_coverage.mjs +41 -4
  161. package/scripts/check_max_file_lines.mjs +102 -0
  162. package/scripts/check_release_evidence_bound.mjs +30 -15
  163. package/scripts/check_screenshot_pixel_delta.py +34 -4
  164. package/scripts/check_server_i18n.mjs +2 -0
  165. package/scripts/chunking_parity_corpus.py +449 -0
  166. package/scripts/generate_agent_parity_fixtures.py +752 -0
  167. package/scripts/generate_chunking_parity_fixtures.py +259 -0
  168. package/scripts/generate_rust_parity_fixtures.py +997 -0
  169. package/scripts/lib/mock_server_fingerprint.mjs +94 -0
  170. package/scripts/release_screen_claims.json +42 -2
  171. package/src-tauri/Cargo.lock +404 -3
  172. package/src-tauri/Cargo.toml +13 -1
  173. package/src-tauri/src/backend.rs +460 -0
  174. package/src-tauri/src/folder.rs +33 -0
  175. package/src-tauri/src/main.rs +109 -399
  176. package/src-tauri/src/topology.rs +356 -0
  177. package/src-tauri/tauri.conf.json +1 -1
  178. package/static/app/asset-manifest.json +41 -37
  179. package/static/app/assets/Act-CWnxSCgN.js +1 -0
  180. package/static/app/assets/AdminConsole-BEQYU6kF.js +1 -0
  181. package/static/app/assets/{Brain-tuhI4sOC.js → Brain-DWu1BhFg.js} +2 -2
  182. package/static/app/assets/BrainHome-95Hilr9R.js +2 -0
  183. package/static/app/assets/BrainSignals-QdeqCpAF.js +1 -0
  184. package/static/app/assets/Capture-BHpCxnzb.js +1 -0
  185. package/static/app/assets/Chronicle-B4xYKoed.js +1 -0
  186. package/static/app/assets/CommandPalette-BVXnttSz.js +1 -0
  187. package/static/app/assets/Library-DgYcHome.js +1 -0
  188. package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-CrJLDbf7.js} +1 -1
  189. package/static/app/assets/ProductFlow-DFlScKoJ.js +1 -0
  190. package/static/app/assets/ReviewCard-Cy5f48Pj.js +3 -0
  191. package/static/app/assets/System-NF8IfhTa.js +1 -0
  192. package/static/app/assets/arrow-left-DwkSYrjR.js +1 -0
  193. package/static/app/assets/{bot-Cia42c2h.js → bot-CucuhLhm.js} +1 -1
  194. package/static/app/assets/brain-BBnSryW_.js +1 -0
  195. package/static/app/assets/{button-2j2Ijzgq.js → button-C2GUj2Ai.js} +1 -1
  196. package/static/app/assets/circle-check-CxOVPwYq.js +1 -0
  197. package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-CbkWzBmG.js} +1 -1
  198. package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-7lEaqHdJ.js} +1 -1
  199. package/static/app/assets/{cpu-k4awryFq.js → cpu-DAlCXlIy.js} +1 -1
  200. package/static/app/assets/{download-DFbLJ_ig.js → download-RNhuuJwh.js} +1 -1
  201. package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CLW4odzM.js} +1 -1
  202. package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-NKEiDIAJ.js} +1 -1
  203. package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
  204. package/static/app/assets/index-DMurvUuR.js +10 -0
  205. package/static/app/assets/input-D2UhPC1X.js +1 -0
  206. package/static/app/assets/link-2-6amKbP_P.js +1 -0
  207. package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-Cu9TZtdR.js} +1 -1
  208. package/static/app/assets/primitives-gPsccucr.js +1 -0
  209. package/static/app/assets/search-Cj_TKk_2.js +1 -0
  210. package/static/app/assets/{share-2-BH1M-WNi.js → share-2-Bau7KkPq.js} +1 -1
  211. package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-BufNYypi.js} +1 -1
  212. package/static/app/assets/{textarea-CCWbUfFB.js → textarea-BQnVWhYs.js} +1 -1
  213. package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-B3_w60si.js} +1 -1
  214. package/static/app/assets/useMutation-BHhCflT6.js +1 -0
  215. package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-rBWfI-5t.js} +1 -1
  216. package/static/app/assets/utils-V_5-wxr5.js +4 -0
  217. package/static/app/assets/workspace-K1zjYUHj.js +1 -0
  218. package/static/app/index.html +4 -4
  219. package/static/sw.js +1 -1
  220. package/lattice_brain/graph/_kg_common.py +0 -1331
  221. package/lattice_brain/graph/discovery_index.py +0 -1141
  222. package/lattice_brain/graph/retrieval.py +0 -1120
  223. package/lattice_brain/graph/retrieval_vector.py +0 -1293
  224. package/lattice_brain/ingestion.py +0 -1525
  225. package/lattice_brain/multimodal.py +0 -1258
  226. package/latticeai/core/agent.py +0 -1465
  227. package/latticeai/core/embedding_providers.py +0 -1196
  228. package/latticeai/core/file_generation.py +0 -1047
  229. package/latticeai/integrations/telegram_bot.py +0 -1390
  230. package/latticeai/models/router.py +0 -1007
  231. package/latticeai/runtime/build_phases.py +0 -1450
  232. package/latticeai/services/brain_intelligence.py +0 -1083
  233. package/latticeai/services/memory_service.py +0 -1177
  234. package/latticeai/services/model_runtime.py +0 -1281
  235. package/latticeai/setup/wizard.py +0 -1310
  236. package/static/app/assets/Act-AWf0SAKp.js +0 -1
  237. package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
  238. package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
  239. package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
  240. package/static/app/assets/Capture-CqOSzyPr.js +0 -1
  241. package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
  242. package/static/app/assets/Library-CX-bbhmK.js +0 -1
  243. package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
  244. package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
  245. package/static/app/assets/System-Bu2t5hn1.js +0 -1
  246. package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
  247. package/static/app/assets/brain-DJMoqrwx.js +0 -1
  248. package/static/app/assets/index-BpYkzcVm.js +0 -10
  249. package/static/app/assets/input-DSlJJxRs.js +0 -1
  250. package/static/app/assets/primitives-BCx6TvfG.js +0 -1
  251. package/static/app/assets/search-Cgy8cCFJ.js +0 -1
  252. package/static/app/assets/utils-zqPZJxdx.js +0 -4
  253. package/static/app/assets/workspace-DXTihhfU.js +0 -1
@@ -1,1141 +0,0 @@
1
- from __future__ import annotations
2
-
3
- from typing import TYPE_CHECKING
4
-
5
- # ruff: noqa: F403,F405
6
- from ._kg_common import * # noqa: F403,F401
7
-
8
- # The cross-mixin surface (`_connect`, `_upsert_node`, …) is declared in
9
- # `_kg_contract.KnowledgeGraphCore`. It is a typing-only base: at runtime this
10
- # is `object`, so the MRO of `KnowledgeGraphStore` is unchanged.
11
- if TYPE_CHECKING:
12
- from ._kg_contract import KnowledgeGraphCore as _Core
13
- else:
14
- _Core = object
15
-
16
-
17
-
18
- def _local_scoped_slug(prefix: str, value: str, workspace_id: Optional[str]) -> str:
19
- """Preserve legacy IDs while isolating newly workspace-scoped nodes."""
20
- slug = _slug(value)
21
- if not workspace_id:
22
- return f"{prefix}:{slug}"
23
- scope = _sha256_text(str(workspace_id))[:12]
24
- return f"{prefix}:{scope}:{slug}"
25
-
26
-
27
- class KnowledgeGraphLocalIndexMixin(_Core):
28
- """Local file → graph indexing (text extraction, node/index upserts,
29
- graph-node deletion, orphan cleanup, and the index_local_folder driver),
30
- split out of discovery. Composed into KnowledgeGraphStore alongside
31
- KnowledgeGraphDiscoveryMixin; both share the instance so these methods
32
- still reach sibling discovery/write helpers through the class MRO.
33
- """
34
-
35
- def _extract_local_file_text(
36
- self, path: Path, category: str, *, include_ocr: bool
37
- ) -> Tuple[str, Dict[str, Any]]:
38
- ext = path.suffix.lower()
39
- meta: Dict[str, Any] = {"parser": _parser_type_for_category(category, ext)}
40
- text = ""
41
- if category in {"text", "code"} or ext == ".csv":
42
- text = path.read_text(encoding="utf-8", errors="replace")
43
- elif ext == ".pdf":
44
- import pdfplumber
45
-
46
- with pdfplumber.open(str(path)) as pdf:
47
- meta["pages"] = len(pdf.pages)
48
- text = "\n\n".join((page.extract_text() or "") for page in pdf.pages)
49
- elif ext == ".docx":
50
- from docx import Document
51
-
52
- doc = Document(str(path))
53
- paragraphs = [p.text for p in doc.paragraphs if p.text.strip()]
54
- table_lines = []
55
- for table in doc.tables:
56
- for row in table.rows:
57
- cells = [_clean_text(cell.text) for cell in row.cells]
58
- if any(cells):
59
- table_lines.append("\t".join(cells))
60
- meta["paragraphs"] = len(paragraphs)
61
- meta["tables"] = len(doc.tables)
62
- meta["table_rows"] = len(table_lines)
63
- text = "\n\n".join([*paragraphs, *table_lines])
64
- elif ext == ".xlsx":
65
- from openpyxl import load_workbook
66
-
67
- wb = load_workbook(str(path), read_only=True, data_only=True)
68
- rows_all = []
69
- non_empty_rows = 0
70
- non_empty_cells = 0
71
- char_count = 0
72
- for ws in wb.worksheets:
73
- sheet_rows = []
74
- for row in ws.iter_rows(values_only=True):
75
- cells = [
76
- str(cell).strip() if cell is not None else "" for cell in row
77
- ]
78
- if not any(cells):
79
- continue
80
- line = "\t".join(cells)
81
- non_empty_rows += 1
82
- non_empty_cells += sum(1 for cell in cells if cell)
83
- sheet_rows.append(line)
84
- char_count += len(line) + 1
85
- if char_count > 200_000:
86
- break
87
- if sheet_rows:
88
- rows_all.append(f"[Sheet: {ws.title}]")
89
- rows_all.extend(sheet_rows)
90
- if char_count > 200_000:
91
- break
92
- meta["sheets"] = len(wb.worksheets)
93
- meta["rows"] = non_empty_rows
94
- meta["cells"] = non_empty_cells
95
- text = "\n".join(rows_all)
96
- elif ext == ".pptx":
97
- from pptx import Presentation
98
-
99
- prs = Presentation(str(path))
100
- slides_text = []
101
- for index, slide in enumerate(prs.slides, 1):
102
- parts = []
103
- for shape in slide.shapes:
104
- if getattr(shape, "has_text_frame", False):
105
- slide_text = shape.text_frame.text.strip()
106
- if slide_text:
107
- parts.append(slide_text)
108
- if parts:
109
- slides_text.append(f"[Slide {index}]\n" + "\n".join(parts))
110
- meta["slides"] = len(prs.slides)
111
- meta["text_slides"] = len(slides_text)
112
- text = "\n\n".join(slides_text)
113
- elif category == "image":
114
- text = self._extract_image_signals(path, meta, include_ocr=include_ocr)
115
- return text[:200_000], meta
116
-
117
- def _extract_image_signals(
118
- self, path: Path, meta: Dict[str, Any], *, include_ocr: bool
119
- ) -> str:
120
- """Dimensions, OCR text, and — only if a VLM exists — a caption.
121
-
122
- Until v11.1.0 this path always attached a ``vision_caption`` built out
123
- of the filename and the pixel dimensions (``Image pic.png (PNG 12x8)``)
124
- and used it as the retrieval text. Nothing downstream could tell that
125
- string apart from something a vision model had actually said about the
126
- picture, so every screenshot in the graph carried a fake description.
127
-
128
- Now the caption comes from the injected port and from nowhere else. A
129
- picture with no OCR text and no model still gets indexed — under its
130
- filename, which is a fact — and ``caption_status`` says why there is no
131
- caption.
132
- """
133
- from ..multimodal import MultimodalPorts, extract_image_facts
134
-
135
- ports = getattr(self, "multimodal_ports", None) or MultimodalPorts()
136
- facts = extract_image_facts(str(path), ports=ports, ocr=include_ocr)
137
- meta.update(facts.as_metadata())
138
- meta["ocr_enabled"] = bool(include_ocr)
139
- if facts.ocr_text:
140
- meta["ocr_chars"] = len(facts.ocr_text)
141
- if facts.ocr_status == "failed":
142
- meta["ocr_error"] = facts.ocr_detail
143
- if not facts.readable:
144
- return ""
145
- return facts.index_text() or path.name
146
-
147
- def _ensure_local_hierarchy(
148
- self,
149
- conn: sqlite3.Connection,
150
- *,
151
- source_id: str,
152
- root: Path,
153
- file_path: Path,
154
- os_type: str,
155
- drive_id: str,
156
- user_email: Optional[str] = None,
157
- workspace_id: Optional[str] = None,
158
- ) -> str:
159
- computer_label = platform.node() or "내 컴퓨터"
160
- computer_id = _local_scoped_slug("computer", computer_label, workspace_id)
161
- drive_identity = f"{workspace_id}|{os_type}:{drive_id}" if workspace_id else f"{os_type}:{drive_id}"
162
- drive_node_id = f"drive:{_sha256_text(drive_identity)[:24]}"
163
- root_folder_id = f"folder:{_sha256_text(f'{source_id}:root')[:24]}"
164
- self._upsert_node(
165
- conn,
166
- computer_id,
167
- "Computer",
168
- computer_label,
169
- metadata={"os_type": os_type, "workspace_id": workspace_id},
170
- owner=user_email,
171
- workspace_id=workspace_id,
172
- )
173
- self._upsert_node(
174
- conn,
175
- drive_node_id,
176
- "Drive",
177
- drive_id,
178
- metadata={"os_type": os_type, "drive_id": drive_id, "workspace_id": workspace_id},
179
- owner=user_email,
180
- workspace_id=workspace_id,
181
- )
182
- stale_parents = conn.execute(
183
- """
184
- SELECT e.from_node
185
- FROM edges e
186
- JOIN nodes n ON n.id=e.from_node
187
- WHERE e.to_node=? AND n.type='Drive' AND e.from_node<>?
188
- """,
189
- (root_folder_id, drive_node_id),
190
- ).fetchall()
191
- for row in stale_parents:
192
- conn.execute(
193
- "DELETE FROM edges WHERE from_node=? AND to_node=?",
194
- (row["from_node"], root_folder_id),
195
- )
196
- conn.execute(
197
- "DELETE FROM edges_v2 WHERE source=? AND target=?",
198
- (row["from_node"], root_folder_id),
199
- )
200
- self._upsert_edge(
201
- conn,
202
- computer_id,
203
- drive_node_id,
204
- "포함함",
205
- metadata={"source": "local_scan", "workspace_id": workspace_id},
206
- )
207
- self._upsert_node(
208
- conn,
209
- root_folder_id,
210
- "Folder",
211
- root.name or str(root),
212
- summary=str(root),
213
- metadata={"source_id": source_id, "path": str(root), "root": True, "workspace_id": workspace_id},
214
- owner=user_email,
215
- workspace_id=workspace_id,
216
- )
217
- self._upsert_edge(
218
- conn,
219
- drive_node_id,
220
- root_folder_id,
221
- "포함함",
222
- metadata={"source": "local_scan", "workspace_id": workspace_id},
223
- )
224
-
225
- try:
226
- relative_parent = file_path.parent.relative_to(root)
227
- except ValueError:
228
- relative_parent = Path()
229
- parent_id = root_folder_id
230
- current_path = root
231
- for part in relative_parent.parts:
232
- current_path = current_path / part
233
- folder_id = (
234
- f"folder:{_sha256_text(f'{source_id}:{current_path.as_posix()}')[:24]}"
235
- )
236
- self._upsert_node(
237
- conn,
238
- folder_id,
239
- "Folder",
240
- part,
241
- summary=str(current_path),
242
- metadata={
243
- "source_id": source_id,
244
- "path": str(current_path),
245
- "root": False,
246
- "workspace_id": workspace_id,
247
- },
248
- owner=user_email,
249
- workspace_id=workspace_id,
250
- )
251
- self._upsert_edge(
252
- conn,
253
- parent_id,
254
- folder_id,
255
- "포함함",
256
- metadata={"source": "local_scan", "workspace_id": workspace_id},
257
- )
258
- parent_id = folder_id
259
- return parent_id
260
-
261
- def _upsert_local_file_index(
262
- self,
263
- conn: sqlite3.Connection,
264
- *,
265
- source_id: str,
266
- root: Path,
267
- file_path: Path,
268
- stat: Optional[os.stat_result],
269
- os_type: str,
270
- drive_id: str,
271
- status: str,
272
- parser_type: str,
273
- sha256: Optional[str] = None,
274
- graph_node_id: Optional[str] = None,
275
- error_message: Optional[str] = None,
276
- metadata: Optional[Dict[str, Any]] = None,
277
- ) -> str:
278
- try:
279
- relative_path = file_path.relative_to(root).as_posix()
280
- except ValueError:
281
- relative_path = file_path.name
282
- index_id = f"local-index:{_sha256_text(f'{source_id}:{relative_path}')[:24]}"
283
- now = _now()
284
- size = stat.st_size if stat else None
285
- modified_at = _safe_iso_from_stat_mtime(stat.st_mtime) if stat else ""
286
- conn.execute(
287
- """
288
- INSERT INTO local_file_index(
289
- id, source_id, os_type, drive_id, root_path, file_path, relative_path,
290
- file_name, extension, size_bytes, modified_at, sha256, last_scanned_at,
291
- last_indexed_at, parser_type, status, error_message, graph_node_id,
292
- deleted, metadata_json
293
- )
294
- VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
295
- ON CONFLICT(source_id, relative_path) DO UPDATE SET
296
- os_type=excluded.os_type,
297
- drive_id=excluded.drive_id,
298
- root_path=excluded.root_path,
299
- file_path=excluded.file_path,
300
- file_name=excluded.file_name,
301
- extension=excluded.extension,
302
- size_bytes=excluded.size_bytes,
303
- modified_at=excluded.modified_at,
304
- sha256=excluded.sha256,
305
- last_scanned_at=excluded.last_scanned_at,
306
- last_indexed_at=excluded.last_indexed_at,
307
- parser_type=excluded.parser_type,
308
- status=excluded.status,
309
- error_message=excluded.error_message,
310
- graph_node_id=excluded.graph_node_id,
311
- deleted=excluded.deleted,
312
- metadata_json=excluded.metadata_json
313
- """,
314
- (
315
- index_id,
316
- source_id,
317
- os_type,
318
- drive_id,
319
- str(root),
320
- str(file_path),
321
- relative_path,
322
- file_path.name,
323
- file_path.suffix.lower(),
324
- size,
325
- modified_at,
326
- sha256,
327
- now,
328
- now if status == "indexed" else None,
329
- parser_type,
330
- status,
331
- error_message,
332
- graph_node_id,
333
- 0 if status != "deleted" else 1,
334
- _json(metadata),
335
- ),
336
- )
337
- return index_id
338
-
339
- def _upsert_local_file_node(
340
- self,
341
- conn: sqlite3.Connection,
342
- *,
343
- source_id: str,
344
- root: Path,
345
- file_path: Path,
346
- stat: os.stat_result,
347
- os_type: str,
348
- drive_id: str,
349
- sha256: str,
350
- category: str,
351
- parser_type: str,
352
- text: str,
353
- parser_meta: Dict[str, Any],
354
- user_email: Optional[str] = None,
355
- workspace_id: Optional[str] = None,
356
- ) -> str:
357
- text = _clean_text(text)
358
- if not text:
359
- raise ValueError("텍스트 추출 결과가 비어 있습니다.")
360
- try:
361
- relative_path = file_path.relative_to(root).as_posix()
362
- except ValueError:
363
- relative_path = file_path.name
364
- file_node_id = f"local-file:{_sha256_text(f'{source_id}:{relative_path}')[:24]}"
365
- parent_folder_id = self._ensure_local_hierarchy(
366
- conn,
367
- source_id=source_id,
368
- root=root,
369
- file_path=file_path,
370
- os_type=os_type,
371
- drive_id=drive_id,
372
- user_email=user_email,
373
- workspace_id=workspace_id,
374
- )
375
- linked_rows = conn.execute(
376
- """
377
- SELECT e.to_node AS id, n.type, n.metadata_json
378
- FROM edges e
379
- JOIN nodes n ON n.id=e.to_node
380
- WHERE e.from_node=?
381
- """,
382
- (file_node_id,),
383
- ).fetchall()
384
- child_ids = []
385
- auto_candidate_ids = set()
386
- for row in linked_rows:
387
- linked_metadata = _safe_loads(row["metadata_json"])
388
- if row["type"] in {"Chunk", "ImageText", "Section"} or linked_metadata.get("source_node") == file_node_id:
389
- child_ids.append(row["id"])
390
- elif linked_metadata.get("auto_extracted") and linked_metadata.get("source") == "local_folder":
391
- auto_candidate_ids.add(row["id"])
392
- conn.execute("DELETE FROM chunks WHERE source_node=?", (file_node_id,))
393
- if child_ids:
394
- placeholders = ",".join("?" * len(child_ids))
395
- conn.execute(f"DELETE FROM nodes WHERE id IN ({placeholders})", child_ids)
396
- self._v2_delete_nodes(conn, child_ids)
397
- conn.execute("DELETE FROM edges WHERE from_node=?", (file_node_id,))
398
- self._v2_delete_edges_from(conn, file_node_id)
399
- removable_auto_ids = set()
400
- for node_id in auto_candidate_ids:
401
- remaining_edges = conn.execute(
402
- "SELECT from_node, to_node FROM edges WHERE from_node=? OR to_node=?",
403
- (node_id, node_id),
404
- ).fetchall()
405
- if all(
406
- row["from_node"] in auto_candidate_ids
407
- and row["to_node"] in auto_candidate_ids
408
- for row in remaining_edges
409
- ):
410
- removable_auto_ids.add(node_id)
411
- if removable_auto_ids:
412
- placeholders = ",".join("?" * len(removable_auto_ids))
413
- params = list(removable_auto_ids)
414
- conn.execute(
415
- f"DELETE FROM edges WHERE from_node IN ({placeholders}) OR to_node IN ({placeholders})",
416
- params * 2,
417
- )
418
- conn.execute(f"DELETE FROM nodes WHERE id IN ({placeholders})", params)
419
- self._v2_delete_nodes(conn, params)
420
-
421
- metadata = {
422
- "source": "local_folder",
423
- "source_id": source_id,
424
- "root_path": str(root),
425
- "file_path": str(file_path),
426
- "relative_path": relative_path,
427
- "filename": file_path.name,
428
- "ext": file_path.suffix.lower(),
429
- "category": category,
430
- "parser_type": parser_type,
431
- "bytes": stat.st_size,
432
- "modified_at": _safe_iso_from_stat_mtime(stat.st_mtime),
433
- "sha256": sha256,
434
- "parser": parser_meta,
435
- "workspace_id": workspace_id,
436
- }
437
- self._upsert_node(
438
- conn,
439
- file_node_id,
440
- _node_type_for_category(category),
441
- file_path.name,
442
- summary=text[:700],
443
- metadata=metadata,
444
- raw=metadata,
445
- owner=user_email,
446
- workspace_id=workspace_id,
447
- )
448
- self._upsert_edge(
449
- conn,
450
- parent_folder_id,
451
- file_node_id,
452
- "포함함",
453
- weight=1.0,
454
- metadata={"source": "local_scan", "workspace_id": workspace_id},
455
- )
456
- self._cleanup_local_graph_orphans(conn, source_id)
457
-
458
- target_for_concepts = text
459
- if category == "image" and text:
460
- image_text_id = f"imagetext:{_sha256_text(f'{file_node_id}:ocr')[:24]}"
461
- self._upsert_node(
462
- conn,
463
- image_text_id,
464
- "ImageText",
465
- f"{file_path.name} OCR",
466
- summary=_clean_text(text)[:700],
467
- metadata={
468
- "source_node": file_node_id,
469
- "source_id": source_id,
470
- "chars": len(text),
471
- "workspace_id": workspace_id,
472
- },
473
- owner=user_email,
474
- workspace_id=workspace_id,
475
- )
476
- self._upsert_edge(
477
- conn,
478
- file_node_id,
479
- image_text_id,
480
- "포함함",
481
- weight=0.8,
482
- metadata={"source": "ocr", "workspace_id": workspace_id},
483
- )
484
-
485
- # Typed chunking by file extension (markdown/code/plain); plain files
486
- # keep legacy _chunks boundaries and therefore identical chunk ids.
487
- chunk_strategy = chunk_strategy_for(file_path.name)
488
- for index, piece in enumerate(typed_chunks(text, strategy=chunk_strategy)):
489
- chunk = piece["text"]
490
- chunk_fields = typed_chunk_meta_fields(piece)
491
- chunk_id = f"chunk:{_sha256_text(f'{file_node_id}:{index}:{chunk}')[:24]}"
492
- self._upsert_node(
493
- conn,
494
- chunk_id,
495
- "Chunk",
496
- f"{file_path.name} chunk {index + 1}",
497
- summary=chunk[:500],
498
- metadata={
499
- "index": index,
500
- "source_node": file_node_id,
501
- "source_id": source_id,
502
- "workspace_id": workspace_id,
503
- **chunk_fields,
504
- },
505
- owner=user_email,
506
- workspace_id=workspace_id,
507
- )
508
- self._upsert_chunk(
509
- conn,
510
- chunk_id=chunk_id,
511
- source_node=file_node_id,
512
- text=chunk,
513
- metadata={
514
- "index": index,
515
- "source_node": file_node_id,
516
- "source_id": source_id,
517
- "workspace_id": workspace_id,
518
- **chunk_fields,
519
- },
520
- )
521
- self._upsert_edge(
522
- conn,
523
- file_node_id,
524
- chunk_id,
525
- "포함함",
526
- weight=0.7,
527
- metadata={"source": "local_scan", "workspace_id": workspace_id},
528
- )
529
-
530
- concepts = _extract_concepts(target_for_concepts, limit=18)
531
- concept_ids: Dict[str, str] = {}
532
- for concept in concepts:
533
- node_t = _classify_node_type(concept, target_for_concepts)
534
- concept_id = _local_scoped_slug(node_t.lower(), concept, workspace_id)
535
- concept_ids[concept.lower()] = concept_id
536
- self._upsert_node(
537
- conn,
538
- concept_id,
539
- node_t,
540
- concept,
541
- metadata={
542
- "auto_extracted": True,
543
- "source": "local_folder",
544
- "source_id": source_id,
545
- "workspace_id": workspace_id,
546
- },
547
- owner=user_email,
548
- workspace_id=workspace_id,
549
- )
550
- self._upsert_edge(
551
- conn,
552
- file_node_id,
553
- concept_id,
554
- "언급함",
555
- weight=0.75,
556
- metadata={"source": "local_scan", "workspace_id": workspace_id},
557
- )
558
-
559
- for triple in _extract_triples(target_for_concepts, concepts, limit=20):
560
- subj_id = concept_ids.get(triple["subject"].lower())
561
- obj_id = concept_ids.get(triple["object"].lower())
562
- if subj_id and obj_id and subj_id != obj_id:
563
- self._upsert_edge(
564
- conn,
565
- subj_id,
566
- obj_id,
567
- triple["relation"],
568
- weight=0.9,
569
- metadata={
570
- "context": triple.get("context", "")[:240],
571
- "source_id": source_id,
572
- "workspace_id": workspace_id,
573
- },
574
- )
575
-
576
- for item in _semantic_items(target_for_concepts):
577
- sem_type = item["type"]
578
- sem_title = item["title"]
579
- sem_id = f"{sem_type.lower()}:{_sha256_text(f'{file_node_id}:{sem_type}:{sem_title}')[:24]}"
580
- self._upsert_node(
581
- conn,
582
- sem_id,
583
- sem_type,
584
- sem_title,
585
- summary=item["summary"],
586
- metadata={
587
- "auto_extracted": True,
588
- "source_node": file_node_id,
589
- "filename": file_path.name,
590
- "workspace_id": workspace_id,
591
- },
592
- raw=item,
593
- owner=user_email,
594
- workspace_id=workspace_id,
595
- )
596
- self._upsert_edge(
597
- conn,
598
- file_node_id,
599
- sem_id,
600
- "포함함",
601
- weight=0.9,
602
- metadata={"source": "local_scan", "workspace_id": workspace_id},
603
- )
604
-
605
- return file_node_id
606
-
607
- def _delete_local_file_graph(
608
- self, conn: sqlite3.Connection, file_node_id: Optional[str]
609
- ) -> None:
610
- if not file_node_id:
611
- return
612
-
613
- file_row = conn.execute(
614
- "SELECT metadata_json FROM nodes WHERE id=?",
615
- (file_node_id,),
616
- ).fetchone()
617
- source_id = None
618
- if file_row:
619
- source_id = _safe_loads(file_row["metadata_json"]).get("source_id")
620
-
621
- linked_rows = conn.execute(
622
- """
623
- SELECT n.id, n.type, n.metadata_json
624
- FROM edges e
625
- JOIN nodes n ON n.id=e.to_node
626
- WHERE e.from_node=?
627
- """,
628
- (file_node_id,),
629
- ).fetchall()
630
- owned_ids: set = set()
631
- auto_candidate_ids: set = set()
632
- for row in linked_rows:
633
- metadata = _safe_loads(row["metadata_json"])
634
- if (
635
- row["type"] in {"Chunk", "ImageText", "Section"}
636
- or metadata.get("source_node") == file_node_id
637
- ):
638
- owned_ids.add(row["id"])
639
- elif (
640
- metadata.get("auto_extracted")
641
- and metadata.get("source") == "local_folder"
642
- ):
643
- auto_candidate_ids.add(row["id"])
644
-
645
- conn.execute("DELETE FROM chunks WHERE source_node=?", (file_node_id,))
646
- conn.execute(
647
- "DELETE FROM edges WHERE from_node=? OR to_node=?",
648
- (file_node_id, file_node_id),
649
- )
650
- conn.execute("DELETE FROM nodes WHERE id=?", (file_node_id,))
651
- self._v2_delete_nodes(conn, [file_node_id])
652
-
653
- def delete_nodes(node_ids: set) -> None:
654
- if not node_ids:
655
- return
656
- placeholders = ",".join("?" * len(node_ids))
657
- params = list(node_ids)
658
- conn.execute(
659
- f"DELETE FROM chunks WHERE source_node IN ({placeholders})", params
660
- )
661
- conn.execute(
662
- f"DELETE FROM edges WHERE from_node IN ({placeholders}) OR to_node IN ({placeholders})",
663
- params * 2,
664
- )
665
- conn.execute(f"DELETE FROM nodes WHERE id IN ({placeholders})", params)
666
- self._v2_delete_nodes(conn, params)
667
-
668
- delete_nodes(owned_ids)
669
-
670
- removable_auto_ids: set = set()
671
- for node_id in auto_candidate_ids:
672
- remaining_edges = conn.execute(
673
- "SELECT from_node, to_node FROM edges WHERE from_node=? OR to_node=?",
674
- (node_id, node_id),
675
- ).fetchall()
676
- if all(
677
- (
678
- row["from_node"] in auto_candidate_ids
679
- and row["to_node"] in auto_candidate_ids
680
- )
681
- for row in remaining_edges
682
- ):
683
- removable_auto_ids.add(node_id)
684
- delete_nodes(removable_auto_ids)
685
- if source_id:
686
- self._cleanup_local_graph_orphans(conn, str(source_id))
687
-
688
- def _cleanup_local_graph_orphans(
689
- self, conn: sqlite3.Connection, source_id: str
690
- ) -> None:
691
- while True:
692
- folder_rows = conn.execute(
693
- "SELECT id, metadata_json FROM nodes WHERE type='Folder'"
694
- ).fetchall()
695
- leaf_ids = []
696
- for row in folder_rows:
697
- metadata = _safe_loads(row["metadata_json"])
698
- if metadata.get("source_id") != source_id:
699
- continue
700
- has_children = conn.execute(
701
- "SELECT 1 FROM edges WHERE from_node=? LIMIT 1",
702
- (row["id"],),
703
- ).fetchone()
704
- if not has_children:
705
- leaf_ids.append(row["id"])
706
- if not leaf_ids:
707
- break
708
- placeholders = ",".join("?" * len(leaf_ids))
709
- conn.execute(
710
- f"DELETE FROM edges WHERE from_node IN ({placeholders}) OR to_node IN ({placeholders})",
711
- leaf_ids * 2,
712
- )
713
- conn.execute(f"DELETE FROM nodes WHERE id IN ({placeholders})", leaf_ids)
714
- self._v2_delete_nodes(conn, leaf_ids)
715
-
716
- for node_type in ("Drive", "Computer"):
717
- rows = conn.execute(
718
- "SELECT id FROM nodes WHERE type=?", (node_type,)
719
- ).fetchall()
720
- removable = []
721
- for row in rows:
722
- has_children = conn.execute(
723
- "SELECT 1 FROM edges WHERE from_node=? LIMIT 1",
724
- (row["id"],),
725
- ).fetchone()
726
- if not has_children:
727
- removable.append(row["id"])
728
- if removable:
729
- placeholders = ",".join("?" * len(removable))
730
- conn.execute(
731
- f"DELETE FROM edges WHERE from_node IN ({placeholders}) OR to_node IN ({placeholders})",
732
- removable * 2,
733
- )
734
- conn.execute(
735
- f"DELETE FROM nodes WHERE id IN ({placeholders})", removable
736
- )
737
- self._v2_delete_nodes(conn, removable)
738
-
739
- def _local_file_index_has_extracted_text(self, row: sqlite3.Row) -> bool:
740
- metadata = _safe_loads(row["metadata_json"])
741
- parser = metadata.get("parser") if isinstance(metadata, dict) else {}
742
- if not isinstance(parser, dict):
743
- return False
744
- try:
745
- return int(parser.get("extracted_chars") or 0) > 0
746
- except (TypeError, ValueError):
747
- return False
748
-
749
- @staticmethod
750
- def _node_matches_workspace(
751
- conn: sqlite3.Connection,
752
- node_id: Optional[str],
753
- workspace_id: Optional[str],
754
- ) -> bool:
755
- """Return true only when the projected node has the expected scope."""
756
- if not node_id:
757
- return False
758
- row = conn.execute(
759
- "SELECT workspace_id FROM nodes_v2 WHERE id=?",
760
- (node_id,),
761
- ).fetchone()
762
- return bool(row is not None and row["workspace_id"] == workspace_id)
763
-
764
- def index_local_folder(
765
- self,
766
- path: Path,
767
- *,
768
- include_ocr: bool = False,
769
- watch_enabled: bool = False,
770
- user_email: Optional[str] = None,
771
- workspace_id: Optional[str] = None,
772
- consent: Optional[Dict[str, Any]] = None,
773
- max_files: int = 5_000,
774
- source_id_override: Optional[str] = None,
775
- ) -> Dict[str, Any]:
776
- """Read approved files from a local folder and connect them to Graph RAG."""
777
- root = Path(path).expanduser().resolve()
778
- if not root.exists():
779
- raise ValueError(f"경로가 존재하지 않습니다: {path}")
780
- if not root.is_dir():
781
- raise ValueError(f"폴더가 아닙니다: {path}")
782
-
783
- os_type = _current_os_type()
784
- drive_id = _drive_id_for_path(root)
785
- path_fingerprint = _path_fingerprint(root)
786
- source_id = str(source_id_override or "").strip()
787
- if not source_id:
788
- source_id = (
789
- f"source:{_sha256_text(f'{workspace_id}|{path_fingerprint}')[:24]}"
790
- if workspace_id
791
- else f"source:{path_fingerprint}"
792
- )
793
- now = _now()
794
- max_files = max(1, min(int(max_files or 5_000), 50_000))
795
- consent_payload = {
796
- "approved_at": now,
797
- "knowledge_source": True,
798
- "include_ocr": bool(include_ocr),
799
- "watch_enabled": bool(watch_enabled),
800
- "sensitive_files_default_excluded": True,
801
- **(consent or {}),
802
- "approved_by": user_email or (consent or {}).get("approved_by"),
803
- "workspace_id": workspace_id or (consent or {}).get("workspace_id"),
804
- }
805
- counts: Counter = Counter()
806
- seen_relative_paths: set = set()
807
- indexed_nodes: List[str] = []
808
- errors: List[Dict[str, str]] = []
809
- limit_reached = False
810
-
811
- with self._connect() as conn:
812
- existing_source = conn.execute(
813
- "SELECT id, consent_json FROM knowledge_sources WHERE root_path=?",
814
- (str(root),),
815
- ).fetchone()
816
- if existing_source is not None:
817
- existing_consent = _safe_loads(existing_source["consent_json"])
818
- existing_scope = existing_consent.get("workspace_id") or "personal"
819
- requested_scope = workspace_id or consent_payload.get("workspace_id") or "personal"
820
- if existing_scope != requested_scope:
821
- raise ValueError(
822
- "This folder is already connected to another workspace. "
823
- "Disconnect it there before assigning it to a different Brain."
824
- )
825
- if existing_source["id"] != source_id:
826
- # Reuse the legacy source identity so a personal source is
827
- # reprojected in place instead of duplicated during upgrade.
828
- source_id = existing_source["id"]
829
- conn.execute(
830
- """
831
- INSERT INTO knowledge_sources(
832
- id, root_path, os_type, drive_id, label, status, include_ocr,
833
- watch_enabled, consent_json, created_at, updated_at, last_scanned_at
834
- )
835
- VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
836
- ON CONFLICT(id) DO UPDATE SET
837
- root_path=excluded.root_path,
838
- os_type=excluded.os_type,
839
- drive_id=excluded.drive_id,
840
- label=excluded.label,
841
- status=excluded.status,
842
- include_ocr=excluded.include_ocr,
843
- watch_enabled=excluded.watch_enabled,
844
- consent_json=excluded.consent_json,
845
- updated_at=excluded.updated_at,
846
- last_scanned_at=excluded.last_scanned_at
847
- """,
848
- (
849
- source_id,
850
- str(root),
851
- os_type,
852
- drive_id,
853
- root.name or str(root),
854
- "scanning",
855
- 1 if include_ocr else 0,
856
- 1 if watch_enabled else 0,
857
- _json(consent_payload),
858
- now,
859
- now,
860
- now,
861
- ),
862
- )
863
-
864
- for entry in self._iter_local_scan_entries(root, max_files=max_files):
865
- kind = entry["kind"]
866
- file_path = entry["path"]
867
- if kind == "limit_reached":
868
- counts["limit_reached"] += 1
869
- limit_reached = True
870
- break
871
- if kind in {"excluded_dir", "excluded"}:
872
- counts["excluded"] += 1
873
- continue
874
- if kind in {"inaccessible_dir", "inaccessible_file"}:
875
- counts["failed"] += 1
876
- errors.append(
877
- {
878
- "path": str(file_path),
879
- "error": entry.get("reason", "inaccessible"),
880
- }
881
- )
882
- continue
883
- if kind != "file":
884
- continue
885
-
886
- stat = entry["stat"]
887
- try:
888
- relative_path = file_path.relative_to(root).as_posix()
889
- except ValueError:
890
- relative_path = file_path.name
891
- seen_relative_paths.add(relative_path)
892
- modified_at = _safe_iso_from_stat_mtime(stat.st_mtime)
893
- existing = conn.execute(
894
- """
895
- SELECT size_bytes, modified_at, sha256, graph_node_id, status, metadata_json
896
- FROM local_file_index
897
- WHERE source_id=? AND relative_path=?
898
- """,
899
- (source_id, relative_path),
900
- ).fetchone()
901
- decision = self._local_file_decision(file_path, root, stat)
902
- parser_type = decision["parser_type"]
903
- if not decision["indexable"]:
904
- counts[decision["status"]] += 1
905
- if existing and existing["graph_node_id"]:
906
- self._delete_local_file_graph(conn, existing["graph_node_id"])
907
- self._upsert_local_file_index(
908
- conn,
909
- source_id=source_id,
910
- root=root,
911
- file_path=file_path,
912
- stat=stat,
913
- os_type=os_type,
914
- drive_id=drive_id,
915
- status=decision["status"],
916
- parser_type=parser_type,
917
- metadata={
918
- "reason": decision["reason"],
919
- "category": decision["category"],
920
- },
921
- )
922
- continue
923
-
924
- if (
925
- existing
926
- and existing["status"] == "indexed"
927
- and existing["graph_node_id"]
928
- and self._local_file_index_has_extracted_text(existing)
929
- and self._node_matches_workspace(
930
- conn, existing["graph_node_id"], workspace_id
931
- )
932
- and existing["size_bytes"] == stat.st_size
933
- and existing["modified_at"] == modified_at
934
- ):
935
- counts["skipped_unchanged"] += 1
936
- self._upsert_local_file_index(
937
- conn,
938
- source_id=source_id,
939
- root=root,
940
- file_path=file_path,
941
- stat=stat,
942
- os_type=os_type,
943
- drive_id=drive_id,
944
- status="indexed",
945
- parser_type=parser_type,
946
- sha256=existing["sha256"],
947
- graph_node_id=existing["graph_node_id"],
948
- metadata={
949
- **_safe_loads(existing["metadata_json"]),
950
- "category": decision["category"],
951
- "unchanged": True,
952
- },
953
- )
954
- continue
955
-
956
- try:
957
- data = file_path.read_bytes()
958
- digest = _sha256_bytes(data)
959
- except Exception as exc:
960
- counts["failed"] += 1
961
- errors.append({"path": str(file_path), "error": str(exc)})
962
- if existing and existing["graph_node_id"]:
963
- self._delete_local_file_graph(conn, existing["graph_node_id"])
964
- self._upsert_local_file_index(
965
- conn,
966
- source_id=source_id,
967
- root=root,
968
- file_path=file_path,
969
- stat=stat,
970
- os_type=os_type,
971
- drive_id=drive_id,
972
- status="failed",
973
- parser_type=parser_type,
974
- error_message=str(exc),
975
- metadata={"category": decision["category"]},
976
- )
977
- continue
978
-
979
- if (
980
- existing
981
- and existing["sha256"] == digest
982
- and existing["graph_node_id"]
983
- and self._local_file_index_has_extracted_text(existing)
984
- and self._node_matches_workspace(
985
- conn, existing["graph_node_id"], workspace_id
986
- )
987
- ):
988
- counts["skipped_unchanged"] += 1
989
- self._upsert_local_file_index(
990
- conn,
991
- source_id=source_id,
992
- root=root,
993
- file_path=file_path,
994
- stat=stat,
995
- os_type=os_type,
996
- drive_id=drive_id,
997
- status="indexed",
998
- parser_type=parser_type,
999
- sha256=digest,
1000
- graph_node_id=existing["graph_node_id"],
1001
- metadata={
1002
- **_safe_loads(existing["metadata_json"]),
1003
- "category": decision["category"],
1004
- "sha256_unchanged": True,
1005
- },
1006
- )
1007
- continue
1008
-
1009
- try:
1010
- text, parser_meta = self._extract_local_file_text(
1011
- file_path,
1012
- decision["category"],
1013
- include_ocr=include_ocr,
1014
- )
1015
- text = _clean_text(text)
1016
- parser_meta = {**parser_meta, "extracted_chars": len(text)}
1017
- if not text:
1018
- counts["skipped_empty_text"] += 1
1019
- if existing and existing["graph_node_id"]:
1020
- self._delete_local_file_graph(
1021
- conn, existing["graph_node_id"]
1022
- )
1023
- self._upsert_local_file_index(
1024
- conn,
1025
- source_id=source_id,
1026
- root=root,
1027
- file_path=file_path,
1028
- stat=stat,
1029
- os_type=os_type,
1030
- drive_id=drive_id,
1031
- status="skipped_empty_text",
1032
- parser_type=parser_type,
1033
- sha256=digest,
1034
- error_message="텍스트 추출 결과가 비어 있습니다.",
1035
- metadata={
1036
- "category": decision["category"],
1037
- "parser": parser_meta,
1038
- },
1039
- )
1040
- continue
1041
- graph_node_id = self._upsert_local_file_node(
1042
- conn,
1043
- source_id=source_id,
1044
- root=root,
1045
- file_path=file_path,
1046
- stat=stat,
1047
- os_type=os_type,
1048
- drive_id=drive_id,
1049
- sha256=digest,
1050
- category=decision["category"],
1051
- parser_type=parser_type,
1052
- text=text,
1053
- parser_meta=parser_meta,
1054
- user_email=user_email,
1055
- workspace_id=workspace_id,
1056
- )
1057
- self._upsert_local_file_index(
1058
- conn,
1059
- source_id=source_id,
1060
- root=root,
1061
- file_path=file_path,
1062
- stat=stat,
1063
- os_type=os_type,
1064
- drive_id=drive_id,
1065
- status="indexed",
1066
- parser_type=parser_type,
1067
- sha256=digest,
1068
- graph_node_id=graph_node_id,
1069
- metadata={
1070
- "category": decision["category"],
1071
- "parser": parser_meta,
1072
- },
1073
- )
1074
- counts["indexed"] += 1
1075
- indexed_nodes.append(graph_node_id)
1076
- except Exception as exc:
1077
- counts["failed"] += 1
1078
- errors.append({"path": str(file_path), "error": str(exc)})
1079
- if existing and existing["graph_node_id"]:
1080
- self._delete_local_file_graph(conn, existing["graph_node_id"])
1081
- self._upsert_local_file_index(
1082
- conn,
1083
- source_id=source_id,
1084
- root=root,
1085
- file_path=file_path,
1086
- stat=stat,
1087
- os_type=os_type,
1088
- drive_id=drive_id,
1089
- status="failed",
1090
- parser_type=parser_type,
1091
- sha256=digest,
1092
- error_message=str(exc),
1093
- metadata={"category": decision["category"]},
1094
- )
1095
-
1096
- if not limit_reached:
1097
- existing_rows = {
1098
- row["relative_path"]: row["graph_node_id"]
1099
- for row in conn.execute(
1100
- "SELECT relative_path, graph_node_id FROM local_file_index WHERE source_id=?",
1101
- (source_id,),
1102
- )
1103
- }
1104
- deleted_paths = set(existing_rows) - seen_relative_paths
1105
- for relative_path in deleted_paths:
1106
- self._delete_local_file_graph(
1107
- conn, existing_rows.get(relative_path)
1108
- )
1109
- conn.execute(
1110
- """
1111
- UPDATE local_file_index
1112
- SET status='deleted', deleted=1, last_scanned_at=?, error_message=NULL, graph_node_id=NULL
1113
- WHERE source_id=? AND relative_path=?
1114
- """,
1115
- (_now(), source_id, relative_path),
1116
- )
1117
- counts["deleted"] = len(deleted_paths)
1118
- conn.execute(
1119
- """
1120
- UPDATE knowledge_sources
1121
- SET status='active', updated_at=?, last_scanned_at=?
1122
- WHERE id=?
1123
- """,
1124
- (_now(), _now(), source_id),
1125
- )
1126
-
1127
- return {
1128
- "status": "ok",
1129
- "source": {
1130
- "id": source_id,
1131
- "root_path": str(root),
1132
- "os_type": os_type,
1133
- "drive_id": drive_id,
1134
- "include_ocr": bool(include_ocr),
1135
- "watch_enabled": bool(watch_enabled),
1136
- },
1137
- "counts": dict(counts),
1138
- "indexed_nodes": indexed_nodes[:100],
1139
- "errors": errors[:50],
1140
- "notice": "Lattice AI는 사용자가 선택한 폴더만 AI 지식으로 변환합니다.",
1141
- }