ltcai 11.2.0 → 11.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (247) hide show
  1. package/README.md +46 -53
  2. package/docs/CHANGELOG.md +61 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/MULTI_AGENT_RUNTIME.md +1 -1
  6. package/docs/ONBOARDING.md +1 -1
  7. package/docs/OPERATIONS.md +6 -2
  8. package/docs/PERMISSION_MODE.md +1 -1
  9. package/docs/TRUST_MODEL.md +1 -1
  10. package/docs/WHY_LATTICE.md +1 -1
  11. package/docs/kg-schema.md +2 -2
  12. package/docs/v11.3.0_PLAN.md +202 -0
  13. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
  14. package/lattice_brain/__init__.py +1 -1
  15. package/lattice_brain/graph/_kg_common/__init__.py +287 -0
  16. package/lattice_brain/graph/_kg_common/extraction.py +516 -0
  17. package/lattice_brain/graph/_kg_common/relations.py +161 -0
  18. package/lattice_brain/graph/_kg_common/text.py +479 -0
  19. package/lattice_brain/graph/discovery_index/__init__.py +35 -0
  20. package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
  21. package/lattice_brain/graph/discovery_index/extract.py +137 -0
  22. package/lattice_brain/graph/discovery_index/scan.py +411 -0
  23. package/lattice_brain/graph/discovery_index/upsert.py +495 -0
  24. package/lattice_brain/graph/projection/__init__.py +42 -0
  25. package/lattice_brain/graph/projection/curation.py +500 -0
  26. package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
  27. package/lattice_brain/graph/retrieval/__init__.py +54 -0
  28. package/lattice_brain/graph/retrieval/context.py +197 -0
  29. package/lattice_brain/graph/retrieval/graph_view.py +319 -0
  30. package/lattice_brain/graph/retrieval/hybrid.py +488 -0
  31. package/lattice_brain/graph/retrieval/maintenance.py +121 -0
  32. package/lattice_brain/graph/retrieval/signals.py +95 -0
  33. package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
  34. package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
  35. package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
  36. package/lattice_brain/graph/retrieval_vector/search.py +560 -0
  37. package/lattice_brain/graph/retrieval_vector/status.py +374 -0
  38. package/lattice_brain/ingestion/__init__.py +130 -0
  39. package/lattice_brain/ingestion/_contract.py +90 -0
  40. package/lattice_brain/ingestion/constants.py +127 -0
  41. package/lattice_brain/ingestion/folder_scan.py +57 -0
  42. package/lattice_brain/ingestion/folders.py +258 -0
  43. package/lattice_brain/ingestion/hashing.py +26 -0
  44. package/lattice_brain/ingestion/jobs_api.py +107 -0
  45. package/lattice_brain/ingestion/models.py +80 -0
  46. package/lattice_brain/ingestion/pipeline.py +486 -0
  47. package/lattice_brain/ingestion/quality.py +209 -0
  48. package/lattice_brain/ingestion/routing.py +295 -0
  49. package/lattice_brain/multimodal/__init__.py +164 -0
  50. package/lattice_brain/multimodal/audio.py +77 -0
  51. package/lattice_brain/multimodal/common.py +118 -0
  52. package/lattice_brain/multimodal/images.py +498 -0
  53. package/lattice_brain/multimodal/ports.py +169 -0
  54. package/lattice_brain/multimodal/video.py +410 -0
  55. package/lattice_brain/portability/__init__.py +90 -0
  56. package/lattice_brain/portability/_contract.py +42 -0
  57. package/lattice_brain/portability/backups.py +338 -0
  58. package/lattice_brain/portability/bundles.py +136 -0
  59. package/lattice_brain/portability/constants.py +93 -0
  60. package/lattice_brain/portability/fsops.py +138 -0
  61. package/lattice_brain/portability/service.py +41 -0
  62. package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
  63. package/lattice_brain/runtime/__init__.py +1 -1
  64. package/lattice_brain/runtime/multi_agent.py +1 -1
  65. package/latticeai/__init__.py +1 -1
  66. package/latticeai/api/chronicle.py +63 -0
  67. package/latticeai/core/agent/__init__.py +93 -0
  68. package/latticeai/core/agent/_contract.py +79 -0
  69. package/latticeai/core/agent/context.py +57 -0
  70. package/latticeai/core/agent/deps.py +125 -0
  71. package/latticeai/core/agent/execution.py +622 -0
  72. package/latticeai/core/agent/planning.py +145 -0
  73. package/latticeai/core/agent/recovery.py +157 -0
  74. package/latticeai/core/agent/runtime.py +210 -0
  75. package/latticeai/core/agent/verification.py +231 -0
  76. package/latticeai/core/embedding_providers/__init__.py +151 -0
  77. package/latticeai/core/embedding_providers/base.py +199 -0
  78. package/latticeai/core/embedding_providers/captions.py +162 -0
  79. package/latticeai/core/embedding_providers/profiles.py +126 -0
  80. package/latticeai/core/embedding_providers/text.py +350 -0
  81. package/latticeai/core/embedding_providers/vision.py +352 -0
  82. package/latticeai/core/file_generation/__init__.py +115 -0
  83. package/latticeai/core/file_generation/bundles.py +76 -0
  84. package/latticeai/core/file_generation/extraction.py +154 -0
  85. package/latticeai/core/file_generation/inference.py +235 -0
  86. package/latticeai/core/file_generation/orchestration.py +152 -0
  87. package/latticeai/core/file_generation/prompting.py +117 -0
  88. package/latticeai/core/file_generation/repair.py +114 -0
  89. package/latticeai/core/file_generation/sanitize.py +61 -0
  90. package/latticeai/core/file_generation/validation.py +201 -0
  91. package/latticeai/core/legacy_compatibility.py +1 -1
  92. package/latticeai/core/marketplace.py +1 -1
  93. package/latticeai/core/messages.py +9 -0
  94. package/latticeai/core/workspace_os_constants.py +1 -1
  95. package/latticeai/integrations/telegram_bot/__init__.py +123 -0
  96. package/latticeai/integrations/telegram_bot/__main__.py +17 -0
  97. package/latticeai/integrations/telegram_bot/config.py +86 -0
  98. package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
  99. package/latticeai/integrations/telegram_bot/flows.py +478 -0
  100. package/latticeai/integrations/telegram_bot/helpers.py +322 -0
  101. package/latticeai/integrations/telegram_bot/screens.py +394 -0
  102. package/latticeai/models/router/__init__.py +88 -0
  103. package/latticeai/models/router/_contract.py +66 -0
  104. package/latticeai/models/router/branding.py +56 -0
  105. package/latticeai/models/router/catalog.py +69 -0
  106. package/latticeai/models/router/documents.py +199 -0
  107. package/latticeai/models/router/errors.py +37 -0
  108. package/latticeai/models/router/generation.py +258 -0
  109. package/latticeai/models/router/loading.py +291 -0
  110. package/latticeai/models/router/local_models.py +85 -0
  111. package/latticeai/models/router/registry.py +147 -0
  112. package/latticeai/runtime/build_phases/__init__.py +82 -0
  113. package/latticeai/runtime/build_phases/features.py +407 -0
  114. package/latticeai/runtime/build_phases/foundation.py +555 -0
  115. package/latticeai/runtime/build_phases/web.py +492 -0
  116. package/latticeai/runtime/runtime_context.py +1 -0
  117. package/latticeai/services/architecture_readiness.py +48 -19
  118. package/latticeai/services/brain_intelligence/__init__.py +58 -0
  119. package/latticeai/services/brain_intelligence/_contract.py +71 -0
  120. package/latticeai/services/brain_intelligence/consistency.py +193 -0
  121. package/latticeai/services/brain_intelligence/constants.py +47 -0
  122. package/latticeai/services/brain_intelligence/digest.py +258 -0
  123. package/latticeai/services/brain_intelligence/health.py +331 -0
  124. package/latticeai/services/brain_intelligence/proposals.py +264 -0
  125. package/latticeai/services/brain_intelligence/sampling.py +84 -0
  126. package/latticeai/services/brain_intelligence/service.py +48 -0
  127. package/latticeai/services/chronicle.py +557 -0
  128. package/latticeai/services/memory_service/__init__.py +52 -0
  129. package/latticeai/services/memory_service/_contract.py +100 -0
  130. package/latticeai/services/memory_service/brief.py +431 -0
  131. package/latticeai/services/memory_service/constants.py +57 -0
  132. package/latticeai/services/memory_service/maintenance.py +138 -0
  133. package/latticeai/services/memory_service/manager.py +186 -0
  134. package/latticeai/services/memory_service/proof.py +136 -0
  135. package/latticeai/services/memory_service/recall.py +225 -0
  136. package/latticeai/services/memory_service/service.py +48 -0
  137. package/latticeai/services/memory_service/stores.py +110 -0
  138. package/latticeai/services/model_runtime/__init__.py +322 -0
  139. package/latticeai/services/model_runtime/cloud.py +87 -0
  140. package/latticeai/services/model_runtime/download.py +282 -0
  141. package/latticeai/services/model_runtime/engines.py +341 -0
  142. package/latticeai/services/model_runtime/loading.py +178 -0
  143. package/latticeai/services/model_runtime/service.py +129 -0
  144. package/latticeai/services/model_runtime/state.py +131 -0
  145. package/latticeai/services/model_runtime/status.py +255 -0
  146. package/latticeai/services/product_readiness.py +15 -7
  147. package/latticeai/setup/wizard/__init__.py +126 -0
  148. package/latticeai/setup/wizard/catalog.py +172 -0
  149. package/latticeai/setup/wizard/detect.py +323 -0
  150. package/latticeai/setup/wizard/install.py +348 -0
  151. package/latticeai/setup/wizard/paths.py +168 -0
  152. package/latticeai/setup/wizard/plans.py +74 -0
  153. package/latticeai/setup/wizard/recommend.py +320 -0
  154. package/package.json +6 -2
  155. package/scripts/bump_version.py +14 -0
  156. package/scripts/capture_release_evidence.mjs +33 -21
  157. package/scripts/check_current_release_docs.mjs +1 -1
  158. package/scripts/check_i18n_namespace_coverage.mjs +41 -4
  159. package/scripts/check_max_file_lines.mjs +102 -0
  160. package/scripts/check_release_evidence_bound.mjs +30 -15
  161. package/scripts/check_screenshot_pixel_delta.py +34 -4
  162. package/scripts/check_server_i18n.mjs +1 -0
  163. package/scripts/generate_rust_parity_fixtures.py +562 -0
  164. package/scripts/lib/mock_server_fingerprint.mjs +94 -0
  165. package/scripts/release_screen_claims.json +31 -2
  166. package/src-tauri/Cargo.lock +361 -3
  167. package/src-tauri/Cargo.toml +6 -1
  168. package/src-tauri/src/backend.rs +349 -0
  169. package/src-tauri/src/folder.rs +33 -0
  170. package/src-tauri/src/main.rs +97 -399
  171. package/src-tauri/tauri.conf.json +1 -1
  172. package/static/app/asset-manifest.json +41 -37
  173. package/static/app/assets/Act-yYpYnn0v.js +1 -0
  174. package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
  175. package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
  176. package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
  177. package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
  178. package/static/app/assets/Capture-CFIRsFNE.js +1 -0
  179. package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
  180. package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
  181. package/static/app/assets/Library-DwO3yZST.js +1 -0
  182. package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
  183. package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
  184. package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
  185. package/static/app/assets/System-DW8F-2xL.js +1 -0
  186. package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
  187. package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
  188. package/static/app/assets/brain-Ci1CkWjM.js +1 -0
  189. package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
  190. package/static/app/assets/circle-check-DfInj-qD.js +1 -0
  191. package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
  192. package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
  193. package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
  194. package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
  195. package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
  196. package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
  197. package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
  198. package/static/app/assets/index-_u5iUHDr.js +10 -0
  199. package/static/app/assets/input-B0lPdRQZ.js +1 -0
  200. package/static/app/assets/link-2-CoFbooHS.js +1 -0
  201. package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
  202. package/static/app/assets/primitives-DEbN-d6p.js +1 -0
  203. package/static/app/assets/search-BybIWPNd.js +1 -0
  204. package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
  205. package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
  206. package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
  207. package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
  208. package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
  209. package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
  210. package/static/app/assets/utils-BlZr7Pd4.js +4 -0
  211. package/static/app/assets/workspace-jJY4RuAV.js +1 -0
  212. package/static/app/index.html +4 -4
  213. package/static/sw.js +1 -1
  214. package/lattice_brain/graph/_kg_common.py +0 -1331
  215. package/lattice_brain/graph/discovery_index.py +0 -1141
  216. package/lattice_brain/graph/retrieval.py +0 -1120
  217. package/lattice_brain/graph/retrieval_vector.py +0 -1293
  218. package/lattice_brain/ingestion.py +0 -1525
  219. package/lattice_brain/multimodal.py +0 -1258
  220. package/latticeai/core/agent.py +0 -1465
  221. package/latticeai/core/embedding_providers.py +0 -1196
  222. package/latticeai/core/file_generation.py +0 -1047
  223. package/latticeai/integrations/telegram_bot.py +0 -1390
  224. package/latticeai/models/router.py +0 -1007
  225. package/latticeai/runtime/build_phases.py +0 -1450
  226. package/latticeai/services/brain_intelligence.py +0 -1083
  227. package/latticeai/services/memory_service.py +0 -1177
  228. package/latticeai/services/model_runtime.py +0 -1281
  229. package/latticeai/setup/wizard.py +0 -1310
  230. package/static/app/assets/Act-AWf0SAKp.js +0 -1
  231. package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
  232. package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
  233. package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
  234. package/static/app/assets/Capture-CqOSzyPr.js +0 -1
  235. package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
  236. package/static/app/assets/Library-CX-bbhmK.js +0 -1
  237. package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
  238. package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
  239. package/static/app/assets/System-Bu2t5hn1.js +0 -1
  240. package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
  241. package/static/app/assets/brain-DJMoqrwx.js +0 -1
  242. package/static/app/assets/index-BpYkzcVm.js +0 -10
  243. package/static/app/assets/input-DSlJJxRs.js +0 -1
  244. package/static/app/assets/primitives-BCx6TvfG.js +0 -1
  245. package/static/app/assets/search-Cgy8cCFJ.js +0 -1
  246. package/static/app/assets/utils-zqPZJxdx.js +0 -4
  247. package/static/app/assets/workspace-DXTihhfU.js +0 -1
@@ -0,0 +1,479 @@
1
+ """Text cleaning, chunking, and citation-locator maths.
2
+
3
+ Moved verbatim out of the ``_kg_common`` grab-bag (v11.3.0 decomposition).
4
+ Nothing here reaches back into the rest of the package — the import graph is
5
+ ``text ← relations ← extraction ← __init__`` — so this is the layer every
6
+ other one may build on.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import re
12
+ from typing import Any, Dict, List, Optional, Tuple
13
+
14
+ from ...quiet import quiet
15
+
16
+
17
+ def _clean_text(text: str) -> str:
18
+ return re.sub(r"\s+", " ", str(text or "")).strip()
19
+
20
+
21
+ def _chunks(text: str, size: int = 1200, overlap: int = 160) -> List[str]:
22
+ cleaned = str(text or "").strip()
23
+ if not cleaned:
24
+ return []
25
+ chunks: List[str] = []
26
+ start = 0
27
+ while start < len(cleaned):
28
+ end = min(len(cleaned), start + size)
29
+ chunks.append(cleaned[start:end])
30
+ if end >= len(cleaned):
31
+ break
32
+ start = max(0, end - overlap)
33
+ return chunks
34
+
35
+
36
+ # ── Typed chunking (review 2026-07-25 §5.2 S2 — Wave 2.1 + 2.4) ──────────────
37
+ # ``_chunks`` above is a compatibility contract (chunk ids hash over the chunk
38
+ # text) and stays byte-for-byte untouched. ``typed_chunks`` layers strategy-
39
+ # aware boundaries plus per-chunk provenance (start_char / heading_path) on
40
+ # top; ``strategy="plain"`` reproduces the exact ``_chunks`` boundaries so
41
+ # unchanged plain content keeps identical chunk ids.
42
+
43
+ _MARKDOWN_CHUNK_EXTENSIONS = {".md", ".markdown"}
44
+ _CODE_CHUNK_EXTENSIONS = {
45
+ ".py", ".js", ".jsx", ".ts", ".tsx", ".go", ".rs", ".java", ".rb",
46
+ ".c", ".h", ".cpp", ".css", ".sh", ".sql", ".vue", ".svelte",
47
+ ".json", ".yaml", ".yml", ".toml",
48
+ }
49
+ _PROSE_CHUNK_EXTENSIONS = {
50
+ ".txt", ".pdf", ".docx", ".doc", ".rtf", ".odt", ".epub", ".html", ".htm",
51
+ }
52
+ _CHUNK_STRATEGIES = {"plain", "markdown", "code", "prose"}
53
+ # Markdown sections smaller than this merge forward into the next section so
54
+ # heading-dense documents don't shatter into confetti chunks.
55
+ _MARKDOWN_MIN_SECTION_CHARS = 200
56
+ _MARKDOWN_HEADING_RE = re.compile(r"^(#{1,6}) (.*)$", re.MULTILINE)
57
+ _CODE_BOUNDARY_LINE_RE = re.compile(
58
+ r"^(?:def |class |function |export |const |public |private )", re.MULTILINE
59
+ )
60
+ _CODE_BLANK_RUN_RE = re.compile(r"\n\s*\n")
61
+
62
+
63
+ def chunk_strategy_for(filename: Any, *, content_type: str = "") -> str:
64
+ """Route a filename / path / URI (plus optional MIME hint) to a strategy.
65
+
66
+ Returns ``"markdown"`` for .md/.markdown, ``"code"`` for known source-code
67
+ extensions, ``"prose"`` for document formats whose text is running prose
68
+ (.txt/.pdf/.docx/.html/…), ``"plain"`` otherwise. Case-insensitive,
69
+ tolerant of URLs (query/fragment stripped) and ``Path`` objects; never
70
+ raises — any malformed input falls back to ``"plain"``.
71
+
72
+ Unknown/extension-less input stays ``"plain"`` on purpose: the plain
73
+ strategy is the byte-compatible legacy walk, and guessing prose for
74
+ something that might be a data dump would move chunk boundaries for no
75
+ retrieval gain.
76
+ """
77
+ try:
78
+ name = str(filename or "").strip().lower()
79
+ for sep in ("?", "#"):
80
+ name = name.split(sep, 1)[0]
81
+ name = name.replace("\\", "/").rstrip("/").rsplit("/", 1)[-1]
82
+ dot = name.rfind(".")
83
+ ext = name[dot:] if dot > 0 else ""
84
+ if ext in _MARKDOWN_CHUNK_EXTENSIONS:
85
+ return "markdown"
86
+ if ext in _CODE_CHUNK_EXTENSIONS:
87
+ return "code"
88
+ if ext in _PROSE_CHUNK_EXTENSIONS:
89
+ return "prose"
90
+ mime = str(content_type or "").strip().lower()
91
+ if "markdown" in mime:
92
+ return "markdown"
93
+ if mime.startswith("text/html") or mime.startswith("text/plain"):
94
+ return "prose"
95
+ except Exception:
96
+ quiet()
97
+ return "plain"
98
+
99
+
100
+ def _plain_windows(
101
+ cleaned: str,
102
+ size: int,
103
+ overlap: int,
104
+ *,
105
+ base_offset: int = 0,
106
+ strategy: str = "plain",
107
+ heading_path: Optional[str] = None,
108
+ ) -> List[Dict[str, Any]]:
109
+ """The exact ``_chunks`` walk with ``start_char`` tracked.
110
+
111
+ Boundaries and chunk texts are byte-identical to ``_chunks`` over the same
112
+ string — this is the plain-strategy compatibility guarantee.
113
+ """
114
+ out: List[Dict[str, Any]] = []
115
+ start = 0
116
+ total = len(cleaned)
117
+ while start < total:
118
+ end = min(total, start + size)
119
+ out.append(
120
+ {
121
+ "text": cleaned[start:end],
122
+ "meta": {
123
+ "strategy": strategy,
124
+ "start_char": base_offset + start,
125
+ "heading_path": heading_path,
126
+ },
127
+ }
128
+ )
129
+ if end >= total:
130
+ break
131
+ start = max(0, end - overlap)
132
+ return out
133
+
134
+
135
+ def _markdown_section_spans(cleaned: str) -> List[Tuple[int, int, Optional[str]]]:
136
+ """``(start, end, heading_path)`` spans split at ``^#{1,6} `` heading lines.
137
+
138
+ ``heading_path`` is the " > "-joined path of the enclosing headings
139
+ including the section's own heading (e.g. ``"Guide > Setup"``); the
140
+ preamble before the first heading carries ``None``. Spans are contiguous
141
+ raw slices of ``cleaned`` so every chunk text round-trips via start_char.
142
+ """
143
+ spans: List[Tuple[int, int, Optional[str]]] = []
144
+ stack: List[Tuple[int, str]] = []
145
+ prev_start = 0
146
+ prev_path: Optional[str] = None
147
+ for match in _MARKDOWN_HEADING_RE.finditer(cleaned):
148
+ offset = match.start()
149
+ if offset > prev_start:
150
+ spans.append((prev_start, offset, prev_path))
151
+ level = len(match.group(1))
152
+ while stack and stack[-1][0] >= level:
153
+ stack.pop()
154
+ stack.append((level, match.group(2).strip()))
155
+ prev_start = offset
156
+ prev_path = " > ".join(title for _, title in stack) or None
157
+ if len(cleaned) > prev_start:
158
+ spans.append((prev_start, len(cleaned), prev_path))
159
+ return spans
160
+
161
+
162
+ def _merge_small_sections(
163
+ spans: List[Tuple[int, int, Optional[str]]], min_chars: int
164
+ ) -> List[Tuple[int, int, Optional[str]]]:
165
+ """Merge sections under ``min_chars`` forward into the next section.
166
+
167
+ A merged section keeps the heading_path of its first constituent (the
168
+ path in effect at the chunk start). A trailing undersized section merges
169
+ backward into the previous emitted section when one exists.
170
+ """
171
+ merged: List[Tuple[int, int, Optional[str]]] = []
172
+ pending: Optional[Tuple[int, int, Optional[str]]] = None
173
+ for start, end, path in spans:
174
+ if pending is None:
175
+ pending = (start, end, path)
176
+ else:
177
+ pending = (pending[0], end, pending[2])
178
+ if pending[1] - pending[0] >= min_chars:
179
+ merged.append(pending)
180
+ pending = None
181
+ if pending is not None:
182
+ if merged and pending[1] - pending[0] < min_chars:
183
+ last = merged.pop()
184
+ merged.append((last[0], pending[1], last[2]))
185
+ else:
186
+ merged.append(pending)
187
+ return merged
188
+
189
+
190
+ def _markdown_chunks(cleaned: str, size: int, overlap: int) -> List[Dict[str, Any]]:
191
+ sections = _merge_small_sections(
192
+ _markdown_section_spans(cleaned), _MARKDOWN_MIN_SECTION_CHARS
193
+ )
194
+ out: List[Dict[str, Any]] = []
195
+ for start, end, path in sections:
196
+ body = cleaned[start:end]
197
+ if len(body) <= size:
198
+ out.append(
199
+ {
200
+ "text": body,
201
+ "meta": {
202
+ "strategy": "markdown",
203
+ "start_char": start,
204
+ "heading_path": path,
205
+ },
206
+ }
207
+ )
208
+ else:
209
+ out.extend(
210
+ _plain_windows(
211
+ body,
212
+ size,
213
+ overlap,
214
+ base_offset=start,
215
+ strategy="markdown",
216
+ heading_path=path,
217
+ )
218
+ )
219
+ return out
220
+
221
+
222
+ def _code_segment_spans(cleaned: str) -> List[Tuple[int, int]]:
223
+ """Contiguous top-level segments split at blank-line runs and decl lines."""
224
+ boundaries = {0, len(cleaned)}
225
+ for match in _CODE_BLANK_RUN_RE.finditer(cleaned):
226
+ boundaries.add(match.end())
227
+ for match in _CODE_BOUNDARY_LINE_RE.finditer(cleaned):
228
+ boundaries.add(match.start())
229
+ ordered = sorted(boundaries)
230
+ return [
231
+ (ordered[i], ordered[i + 1])
232
+ for i in range(len(ordered) - 1)
233
+ if ordered[i + 1] > ordered[i]
234
+ ]
235
+
236
+
237
+ def _code_chunks(cleaned: str, size: int, overlap: int) -> List[Dict[str, Any]]:
238
+ hard_limit = int(size * 1.5)
239
+ out: List[Dict[str, Any]] = []
240
+ pack: Optional[Tuple[int, int]] = None
241
+
242
+ def _emit(span: Tuple[int, int]) -> None:
243
+ out.append(
244
+ {
245
+ "text": cleaned[span[0] : span[1]],
246
+ "meta": {
247
+ "strategy": "code",
248
+ "start_char": span[0],
249
+ "heading_path": None,
250
+ },
251
+ }
252
+ )
253
+
254
+ for start, end in _code_segment_spans(cleaned):
255
+ if end - start > hard_limit:
256
+ # Monster segment: flush the pack, then window it like plain text.
257
+ if pack is not None:
258
+ _emit(pack)
259
+ pack = None
260
+ out.extend(
261
+ _plain_windows(
262
+ cleaned[start:end],
263
+ size,
264
+ overlap,
265
+ base_offset=start,
266
+ strategy="code",
267
+ )
268
+ )
269
+ continue
270
+ if pack is None:
271
+ pack = (start, end)
272
+ elif end - pack[0] <= size:
273
+ pack = (pack[0], end)
274
+ else:
275
+ _emit(pack)
276
+ pack = (start, end)
277
+ if pack is not None:
278
+ _emit(pack)
279
+ return out
280
+
281
+
282
+ # ── Prose chunking (review 2026-07-27 P1 #4) ────────────────────────────────
283
+ # The plain walk cuts every ``size`` characters, which lands mid-sentence and
284
+ # — for Korean, where the verb carrying the meaning sits at the end — routinely
285
+ # splits a claim from its predicate. Retrieval then matches half a statement
286
+ # and the citation shows a fragment. The prose strategy keeps the same window
287
+ # budget but ends each chunk at the last sentence/paragraph boundary inside it.
288
+
289
+ # Strong: sentence-final punctuation (ASCII + CJK) with optional closing
290
+ # quotes/brackets, followed by whitespace; or a blank-line paragraph break.
291
+ _PROSE_STRONG_BOUNDARY_RE = re.compile(
292
+ r"(?:[.!?。!?…]+[\"'”’」』\)\]]*\s+|\n[ \t]*\n)"
293
+ )
294
+ # Weak: a single line break. Korean notes and bullet lists often carry no
295
+ # sentence punctuation at all; a line end is still a real boundary there.
296
+ _PROSE_WEAK_BOUNDARY_RE = re.compile(r"\n")
297
+ # Never emit a chunk shorter than this fraction of ``size`` just to hit a
298
+ # boundary — tiny chunks hurt recall more than a mid-sentence cut.
299
+ _PROSE_MIN_SPAN_RATIO = 0.5
300
+
301
+
302
+ def _last_boundary(cleaned: str, lo: int, hi: int) -> Optional[int]:
303
+ """End offset of the last sentence/paragraph boundary in ``cleaned[lo:hi]``.
304
+
305
+ Strong boundaries win; a single line break is the fallback. Returns None
306
+ when the span holds neither, so the caller keeps the hard window cut.
307
+ """
308
+ window = cleaned[lo:hi]
309
+ for pattern in (_PROSE_STRONG_BOUNDARY_RE, _PROSE_WEAK_BOUNDARY_RE):
310
+ last = None
311
+ for match in pattern.finditer(window):
312
+ last = match.end()
313
+ if last:
314
+ return lo + last
315
+ return None
316
+
317
+
318
+ def _prose_chunks(cleaned: str, size: int, overlap: int) -> List[Dict[str, Any]]:
319
+ out: List[Dict[str, Any]] = []
320
+ total = len(cleaned)
321
+ min_span = max(1, int(size * _PROSE_MIN_SPAN_RATIO))
322
+ start = 0
323
+ while start < total:
324
+ hard_end = min(total, start + size)
325
+ end = hard_end
326
+ if hard_end < total:
327
+ boundary = _last_boundary(cleaned, start + min_span, hard_end)
328
+ if boundary is not None and boundary > start:
329
+ end = boundary
330
+ out.append(
331
+ {
332
+ "text": cleaned[start:end],
333
+ "meta": {
334
+ "strategy": "prose",
335
+ "start_char": start,
336
+ "heading_path": None,
337
+ },
338
+ }
339
+ )
340
+ if end >= total:
341
+ break
342
+ # Overlap carries the tail of the previous chunk into the next one so
343
+ # a claim split across a boundary is still retrievable from both.
344
+ start = max(start + 1, end - overlap)
345
+ return out
346
+
347
+
348
+ def typed_chunks(
349
+ text: str,
350
+ *,
351
+ strategy: str = "plain",
352
+ size: int = 1200,
353
+ overlap: int = 160,
354
+ ) -> List[Dict[str, Any]]:
355
+ """Strategy-aware chunking with per-chunk provenance metadata.
356
+
357
+ Returns ``[{"text": str, "meta": {"strategy", "start_char", "heading_path"}}]``
358
+ where ``start_char`` is the offset in ``str(text or "").strip()`` (every
359
+ chunk text is an exact substring at that offset).
360
+
361
+ Contract: ``[c["text"] for c in typed_chunks(t)] == _chunks(t)`` for the
362
+ default plain strategy — unknown strategies also fall back to plain.
363
+ """
364
+ cleaned = str(text or "").strip()
365
+ if not cleaned:
366
+ return []
367
+ try:
368
+ size = max(1, int(size))
369
+ except Exception:
370
+ size = 1200
371
+ try:
372
+ overlap = min(max(0, int(overlap)), size - 1)
373
+ except Exception:
374
+ overlap = min(160, size - 1)
375
+ label = strategy if strategy in _CHUNK_STRATEGIES else "plain"
376
+ if label == "markdown":
377
+ return _markdown_chunks(cleaned, size, overlap)
378
+ if label == "code":
379
+ return _code_chunks(cleaned, size, overlap)
380
+ if label == "prose":
381
+ return _prose_chunks(cleaned, size, overlap)
382
+ return _plain_windows(cleaned, size, overlap)
383
+
384
+
385
+ def typed_chunk_meta_fields(piece: Dict[str, Any]) -> Dict[str, Any]:
386
+ """Additive chunk-metadata fields for one ``typed_chunks`` piece.
387
+
388
+ Ingest call sites merge this into the existing ``{"index", "source_node"}``
389
+ chunk metadata; ``heading_path`` is only present when known — honest
390
+ absence over empty labels.
391
+ """
392
+ meta = piece.get("meta") or {}
393
+ fields: Dict[str, Any] = {
394
+ "strategy": str(meta.get("strategy") or "plain"),
395
+ "start_char": int(meta.get("start_char") or 0),
396
+ }
397
+ heading_path = meta.get("heading_path")
398
+ if heading_path:
399
+ fields["heading_path"] = str(heading_path)
400
+ return fields
401
+
402
+
403
+ def citation_locator(chunk_metadata: Any) -> str:
404
+ """Human "where in the document" label for one chunk, or "".
405
+
406
+ Built only from provenance the chunk actually carries — a section heading
407
+ path and/or a page number. When neither is known the answer is the empty
408
+ string, so a citation never claims a location it cannot prove.
409
+ """
410
+ if not isinstance(chunk_metadata, dict):
411
+ return ""
412
+ parts: List[str] = []
413
+ heading = str(chunk_metadata.get("heading_path") or "").strip()
414
+ if heading:
415
+ parts.append(heading)
416
+ def _page(key: str) -> int:
417
+ value = chunk_metadata.get(key)
418
+ try:
419
+ return int(value) if value is not None else 0
420
+ except (TypeError, ValueError):
421
+ return 0
422
+
423
+ page_number = _page("page")
424
+ if page_number > 0:
425
+ page_end = _page("page_end")
426
+ parts.append(
427
+ f"p.{page_number}–{page_end}" if page_end > page_number else f"p.{page_number}"
428
+ )
429
+ return " · ".join(parts)
430
+
431
+
432
+ def pdf_page_offsets(structure: Any) -> List[int]:
433
+ """Start offset of each PDF page in the "\\n\\n"-joined page text.
434
+
435
+ ``structure`` is the ``metadata["structure"]`` dict produced by
436
+ ``_pdf_structure`` (``pages`` = ``[{"chars": int, ...}, ...]``); pages were
437
+ joined with ``"\\n\\n"`` (see ``read_document``), so page k starts at
438
+ ``sum(chars[j] + 2 for j < k)``. Empty or malformed input returns ``[]``.
439
+ """
440
+ if not isinstance(structure, dict):
441
+ return []
442
+ pages = structure.get("pages")
443
+ if not isinstance(pages, list) or not pages:
444
+ return []
445
+ offsets: List[int] = []
446
+ cursor = 0
447
+ for page in pages:
448
+ if not isinstance(page, dict):
449
+ return []
450
+ chars = page.get("chars")
451
+ if isinstance(chars, bool) or not isinstance(chars, (int, float)) or chars < 0:
452
+ return []
453
+ offsets.append(cursor)
454
+ cursor += int(chars) + 2 # +2 for the "\n\n" page joiner
455
+ return offsets
456
+
457
+
458
+ def page_for_offset(page_offsets: List[int], offset: int) -> Optional[int]:
459
+ """1-based page number containing ``offset`` given page start offsets.
460
+
461
+ Returns ``None`` when ``page_offsets`` is empty or the offset precedes the
462
+ first page start (honest absence over a wrong label).
463
+ """
464
+ if not page_offsets:
465
+ return None
466
+ try:
467
+ target = int(offset)
468
+ except Exception:
469
+ return None
470
+ page = 0
471
+ for index, start in enumerate(page_offsets):
472
+ try:
473
+ if target >= int(start):
474
+ page = index + 1
475
+ else:
476
+ break
477
+ except Exception:
478
+ return None
479
+ return page if page >= 1 else None
@@ -0,0 +1,35 @@
1
+ """Local filesystem indexing: a chosen folder becomes graph knowledge.
2
+
3
+ v11.3.0 turned this module into a package. ``KnowledgeGraphLocalIndexMixin``
4
+ is now composed from four cohesive sub-mixins — text extraction, node/index
5
+ upserts, graph cleanup, and the folder-scan driver — each moved here
6
+ verbatim. Every name this module exported before still resolves from
7
+ ``lattice_brain.graph.discovery_index``.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ # ruff: noqa: F403,F405
13
+ from .._kg_common import * # noqa: F403,F401
14
+ from .cleanup import _LocalCleanupMixin
15
+ from .extract import _LocalExtractMixin
16
+ from .scan import _LocalScanMixin
17
+ from .upsert import _local_scoped_slug, _LocalUpsertMixin # noqa: F401
18
+
19
+
20
+ # Base order is most-composed-first (the scan driver, then the three halves it
21
+ # calls): each half is named as the driver's typing-only base, and C3 needs a
22
+ # subclass ahead of the class it extends. The method sets are disjoint, so at
23
+ # runtime the order changes nothing.
24
+ class KnowledgeGraphLocalIndexMixin(
25
+ _LocalScanMixin,
26
+ _LocalExtractMixin,
27
+ _LocalUpsertMixin,
28
+ _LocalCleanupMixin,
29
+ ):
30
+ """Local file → graph indexing (text extraction, node/index upserts,
31
+ graph-node deletion, orphan cleanup, and the index_local_folder driver),
32
+ split out of discovery. Composed into KnowledgeGraphStore alongside
33
+ KnowledgeGraphDiscoveryMixin; both share the instance so these methods
34
+ still reach sibling discovery/write helpers through the class MRO.
35
+ """
@@ -0,0 +1,182 @@
1
+ """Removing a local file from the graph, and the scope checks around it.
2
+
3
+ Deletes a file's graph node, sweeps the concepts left orphaned by it, and
4
+ answers the two "is this row still good?" questions the scanner asks before
5
+ skipping work. Moved verbatim out of ``discovery_index.py`` (v11.3.0).
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from typing import TYPE_CHECKING
11
+
12
+ # ruff: noqa: F403,F405
13
+ from .._kg_common import * # noqa: F403,F401
14
+
15
+ # The cross-mixin surface (`_connect`, `_upsert_node`, …) is declared in
16
+ # `_kg_contract.KnowledgeGraphCore`. It is a typing-only base: at runtime this
17
+ # is `object`, so the MRO of `KnowledgeGraphStore` is unchanged.
18
+ if TYPE_CHECKING:
19
+ from .._kg_contract import KnowledgeGraphCore as _Core
20
+ else:
21
+ _Core = object
22
+
23
+
24
+ class _LocalCleanupMixin(_Core):
25
+ """Graph deletion + orphan sweep. Composed into the public mixin."""
26
+
27
+ def _delete_local_file_graph(
28
+ self, conn: sqlite3.Connection, file_node_id: Optional[str]
29
+ ) -> None:
30
+ if not file_node_id:
31
+ return
32
+
33
+ file_row = conn.execute(
34
+ "SELECT metadata_json FROM nodes WHERE id=?",
35
+ (file_node_id,),
36
+ ).fetchone()
37
+ source_id = None
38
+ if file_row:
39
+ source_id = _safe_loads(file_row["metadata_json"]).get("source_id")
40
+
41
+ linked_rows = conn.execute(
42
+ """
43
+ SELECT n.id, n.type, n.metadata_json
44
+ FROM edges e
45
+ JOIN nodes n ON n.id=e.to_node
46
+ WHERE e.from_node=?
47
+ """,
48
+ (file_node_id,),
49
+ ).fetchall()
50
+ owned_ids: set = set()
51
+ auto_candidate_ids: set = set()
52
+ for row in linked_rows:
53
+ metadata = _safe_loads(row["metadata_json"])
54
+ if (
55
+ row["type"] in {"Chunk", "ImageText", "Section"}
56
+ or metadata.get("source_node") == file_node_id
57
+ ):
58
+ owned_ids.add(row["id"])
59
+ elif (
60
+ metadata.get("auto_extracted")
61
+ and metadata.get("source") == "local_folder"
62
+ ):
63
+ auto_candidate_ids.add(row["id"])
64
+
65
+ conn.execute("DELETE FROM chunks WHERE source_node=?", (file_node_id,))
66
+ conn.execute(
67
+ "DELETE FROM edges WHERE from_node=? OR to_node=?",
68
+ (file_node_id, file_node_id),
69
+ )
70
+ conn.execute("DELETE FROM nodes WHERE id=?", (file_node_id,))
71
+ self._v2_delete_nodes(conn, [file_node_id])
72
+
73
+ def delete_nodes(node_ids: set) -> None:
74
+ if not node_ids:
75
+ return
76
+ placeholders = ",".join("?" * len(node_ids))
77
+ params = list(node_ids)
78
+ conn.execute(
79
+ f"DELETE FROM chunks WHERE source_node IN ({placeholders})", params
80
+ )
81
+ conn.execute(
82
+ f"DELETE FROM edges WHERE from_node IN ({placeholders}) OR to_node IN ({placeholders})",
83
+ params * 2,
84
+ )
85
+ conn.execute(f"DELETE FROM nodes WHERE id IN ({placeholders})", params)
86
+ self._v2_delete_nodes(conn, params)
87
+
88
+ delete_nodes(owned_ids)
89
+
90
+ removable_auto_ids: set = set()
91
+ for node_id in auto_candidate_ids:
92
+ remaining_edges = conn.execute(
93
+ "SELECT from_node, to_node FROM edges WHERE from_node=? OR to_node=?",
94
+ (node_id, node_id),
95
+ ).fetchall()
96
+ if all(
97
+ (
98
+ row["from_node"] in auto_candidate_ids
99
+ and row["to_node"] in auto_candidate_ids
100
+ )
101
+ for row in remaining_edges
102
+ ):
103
+ removable_auto_ids.add(node_id)
104
+ delete_nodes(removable_auto_ids)
105
+ if source_id:
106
+ self._cleanup_local_graph_orphans(conn, str(source_id))
107
+
108
+ def _cleanup_local_graph_orphans(
109
+ self, conn: sqlite3.Connection, source_id: str
110
+ ) -> None:
111
+ while True:
112
+ folder_rows = conn.execute(
113
+ "SELECT id, metadata_json FROM nodes WHERE type='Folder'"
114
+ ).fetchall()
115
+ leaf_ids = []
116
+ for row in folder_rows:
117
+ metadata = _safe_loads(row["metadata_json"])
118
+ if metadata.get("source_id") != source_id:
119
+ continue
120
+ has_children = conn.execute(
121
+ "SELECT 1 FROM edges WHERE from_node=? LIMIT 1",
122
+ (row["id"],),
123
+ ).fetchone()
124
+ if not has_children:
125
+ leaf_ids.append(row["id"])
126
+ if not leaf_ids:
127
+ break
128
+ placeholders = ",".join("?" * len(leaf_ids))
129
+ conn.execute(
130
+ f"DELETE FROM edges WHERE from_node IN ({placeholders}) OR to_node IN ({placeholders})",
131
+ leaf_ids * 2,
132
+ )
133
+ conn.execute(f"DELETE FROM nodes WHERE id IN ({placeholders})", leaf_ids)
134
+ self._v2_delete_nodes(conn, leaf_ids)
135
+
136
+ for node_type in ("Drive", "Computer"):
137
+ rows = conn.execute(
138
+ "SELECT id FROM nodes WHERE type=?", (node_type,)
139
+ ).fetchall()
140
+ removable = []
141
+ for row in rows:
142
+ has_children = conn.execute(
143
+ "SELECT 1 FROM edges WHERE from_node=? LIMIT 1",
144
+ (row["id"],),
145
+ ).fetchone()
146
+ if not has_children:
147
+ removable.append(row["id"])
148
+ if removable:
149
+ placeholders = ",".join("?" * len(removable))
150
+ conn.execute(
151
+ f"DELETE FROM edges WHERE from_node IN ({placeholders}) OR to_node IN ({placeholders})",
152
+ removable * 2,
153
+ )
154
+ conn.execute(
155
+ f"DELETE FROM nodes WHERE id IN ({placeholders})", removable
156
+ )
157
+ self._v2_delete_nodes(conn, removable)
158
+
159
+ def _local_file_index_has_extracted_text(self, row: sqlite3.Row) -> bool:
160
+ metadata = _safe_loads(row["metadata_json"])
161
+ parser = metadata.get("parser") if isinstance(metadata, dict) else {}
162
+ if not isinstance(parser, dict):
163
+ return False
164
+ try:
165
+ return int(parser.get("extracted_chars") or 0) > 0
166
+ except (TypeError, ValueError):
167
+ return False
168
+
169
+ @staticmethod
170
+ def _node_matches_workspace(
171
+ conn: sqlite3.Connection,
172
+ node_id: Optional[str],
173
+ workspace_id: Optional[str],
174
+ ) -> bool:
175
+ """Return true only when the projected node has the expected scope."""
176
+ if not node_id:
177
+ return False
178
+ row = conn.execute(
179
+ "SELECT workspace_id FROM nodes_v2 WHERE id=?",
180
+ (node_id,),
181
+ ).fetchone()
182
+ return bool(row is not None and row["workspace_id"] == workspace_id)