ltcai 11.2.0 → 11.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (247) hide show
  1. package/README.md +46 -53
  2. package/docs/CHANGELOG.md +61 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/MULTI_AGENT_RUNTIME.md +1 -1
  6. package/docs/ONBOARDING.md +1 -1
  7. package/docs/OPERATIONS.md +6 -2
  8. package/docs/PERMISSION_MODE.md +1 -1
  9. package/docs/TRUST_MODEL.md +1 -1
  10. package/docs/WHY_LATTICE.md +1 -1
  11. package/docs/kg-schema.md +2 -2
  12. package/docs/v11.3.0_PLAN.md +202 -0
  13. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
  14. package/lattice_brain/__init__.py +1 -1
  15. package/lattice_brain/graph/_kg_common/__init__.py +287 -0
  16. package/lattice_brain/graph/_kg_common/extraction.py +516 -0
  17. package/lattice_brain/graph/_kg_common/relations.py +161 -0
  18. package/lattice_brain/graph/_kg_common/text.py +479 -0
  19. package/lattice_brain/graph/discovery_index/__init__.py +35 -0
  20. package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
  21. package/lattice_brain/graph/discovery_index/extract.py +137 -0
  22. package/lattice_brain/graph/discovery_index/scan.py +411 -0
  23. package/lattice_brain/graph/discovery_index/upsert.py +495 -0
  24. package/lattice_brain/graph/projection/__init__.py +42 -0
  25. package/lattice_brain/graph/projection/curation.py +500 -0
  26. package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
  27. package/lattice_brain/graph/retrieval/__init__.py +54 -0
  28. package/lattice_brain/graph/retrieval/context.py +197 -0
  29. package/lattice_brain/graph/retrieval/graph_view.py +319 -0
  30. package/lattice_brain/graph/retrieval/hybrid.py +488 -0
  31. package/lattice_brain/graph/retrieval/maintenance.py +121 -0
  32. package/lattice_brain/graph/retrieval/signals.py +95 -0
  33. package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
  34. package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
  35. package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
  36. package/lattice_brain/graph/retrieval_vector/search.py +560 -0
  37. package/lattice_brain/graph/retrieval_vector/status.py +374 -0
  38. package/lattice_brain/ingestion/__init__.py +130 -0
  39. package/lattice_brain/ingestion/_contract.py +90 -0
  40. package/lattice_brain/ingestion/constants.py +127 -0
  41. package/lattice_brain/ingestion/folder_scan.py +57 -0
  42. package/lattice_brain/ingestion/folders.py +258 -0
  43. package/lattice_brain/ingestion/hashing.py +26 -0
  44. package/lattice_brain/ingestion/jobs_api.py +107 -0
  45. package/lattice_brain/ingestion/models.py +80 -0
  46. package/lattice_brain/ingestion/pipeline.py +486 -0
  47. package/lattice_brain/ingestion/quality.py +209 -0
  48. package/lattice_brain/ingestion/routing.py +295 -0
  49. package/lattice_brain/multimodal/__init__.py +164 -0
  50. package/lattice_brain/multimodal/audio.py +77 -0
  51. package/lattice_brain/multimodal/common.py +118 -0
  52. package/lattice_brain/multimodal/images.py +498 -0
  53. package/lattice_brain/multimodal/ports.py +169 -0
  54. package/lattice_brain/multimodal/video.py +410 -0
  55. package/lattice_brain/portability/__init__.py +90 -0
  56. package/lattice_brain/portability/_contract.py +42 -0
  57. package/lattice_brain/portability/backups.py +338 -0
  58. package/lattice_brain/portability/bundles.py +136 -0
  59. package/lattice_brain/portability/constants.py +93 -0
  60. package/lattice_brain/portability/fsops.py +138 -0
  61. package/lattice_brain/portability/service.py +41 -0
  62. package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
  63. package/lattice_brain/runtime/__init__.py +1 -1
  64. package/lattice_brain/runtime/multi_agent.py +1 -1
  65. package/latticeai/__init__.py +1 -1
  66. package/latticeai/api/chronicle.py +63 -0
  67. package/latticeai/core/agent/__init__.py +93 -0
  68. package/latticeai/core/agent/_contract.py +79 -0
  69. package/latticeai/core/agent/context.py +57 -0
  70. package/latticeai/core/agent/deps.py +125 -0
  71. package/latticeai/core/agent/execution.py +622 -0
  72. package/latticeai/core/agent/planning.py +145 -0
  73. package/latticeai/core/agent/recovery.py +157 -0
  74. package/latticeai/core/agent/runtime.py +210 -0
  75. package/latticeai/core/agent/verification.py +231 -0
  76. package/latticeai/core/embedding_providers/__init__.py +151 -0
  77. package/latticeai/core/embedding_providers/base.py +199 -0
  78. package/latticeai/core/embedding_providers/captions.py +162 -0
  79. package/latticeai/core/embedding_providers/profiles.py +126 -0
  80. package/latticeai/core/embedding_providers/text.py +350 -0
  81. package/latticeai/core/embedding_providers/vision.py +352 -0
  82. package/latticeai/core/file_generation/__init__.py +115 -0
  83. package/latticeai/core/file_generation/bundles.py +76 -0
  84. package/latticeai/core/file_generation/extraction.py +154 -0
  85. package/latticeai/core/file_generation/inference.py +235 -0
  86. package/latticeai/core/file_generation/orchestration.py +152 -0
  87. package/latticeai/core/file_generation/prompting.py +117 -0
  88. package/latticeai/core/file_generation/repair.py +114 -0
  89. package/latticeai/core/file_generation/sanitize.py +61 -0
  90. package/latticeai/core/file_generation/validation.py +201 -0
  91. package/latticeai/core/legacy_compatibility.py +1 -1
  92. package/latticeai/core/marketplace.py +1 -1
  93. package/latticeai/core/messages.py +9 -0
  94. package/latticeai/core/workspace_os_constants.py +1 -1
  95. package/latticeai/integrations/telegram_bot/__init__.py +123 -0
  96. package/latticeai/integrations/telegram_bot/__main__.py +17 -0
  97. package/latticeai/integrations/telegram_bot/config.py +86 -0
  98. package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
  99. package/latticeai/integrations/telegram_bot/flows.py +478 -0
  100. package/latticeai/integrations/telegram_bot/helpers.py +322 -0
  101. package/latticeai/integrations/telegram_bot/screens.py +394 -0
  102. package/latticeai/models/router/__init__.py +88 -0
  103. package/latticeai/models/router/_contract.py +66 -0
  104. package/latticeai/models/router/branding.py +56 -0
  105. package/latticeai/models/router/catalog.py +69 -0
  106. package/latticeai/models/router/documents.py +199 -0
  107. package/latticeai/models/router/errors.py +37 -0
  108. package/latticeai/models/router/generation.py +258 -0
  109. package/latticeai/models/router/loading.py +291 -0
  110. package/latticeai/models/router/local_models.py +85 -0
  111. package/latticeai/models/router/registry.py +147 -0
  112. package/latticeai/runtime/build_phases/__init__.py +82 -0
  113. package/latticeai/runtime/build_phases/features.py +407 -0
  114. package/latticeai/runtime/build_phases/foundation.py +555 -0
  115. package/latticeai/runtime/build_phases/web.py +492 -0
  116. package/latticeai/runtime/runtime_context.py +1 -0
  117. package/latticeai/services/architecture_readiness.py +48 -19
  118. package/latticeai/services/brain_intelligence/__init__.py +58 -0
  119. package/latticeai/services/brain_intelligence/_contract.py +71 -0
  120. package/latticeai/services/brain_intelligence/consistency.py +193 -0
  121. package/latticeai/services/brain_intelligence/constants.py +47 -0
  122. package/latticeai/services/brain_intelligence/digest.py +258 -0
  123. package/latticeai/services/brain_intelligence/health.py +331 -0
  124. package/latticeai/services/brain_intelligence/proposals.py +264 -0
  125. package/latticeai/services/brain_intelligence/sampling.py +84 -0
  126. package/latticeai/services/brain_intelligence/service.py +48 -0
  127. package/latticeai/services/chronicle.py +557 -0
  128. package/latticeai/services/memory_service/__init__.py +52 -0
  129. package/latticeai/services/memory_service/_contract.py +100 -0
  130. package/latticeai/services/memory_service/brief.py +431 -0
  131. package/latticeai/services/memory_service/constants.py +57 -0
  132. package/latticeai/services/memory_service/maintenance.py +138 -0
  133. package/latticeai/services/memory_service/manager.py +186 -0
  134. package/latticeai/services/memory_service/proof.py +136 -0
  135. package/latticeai/services/memory_service/recall.py +225 -0
  136. package/latticeai/services/memory_service/service.py +48 -0
  137. package/latticeai/services/memory_service/stores.py +110 -0
  138. package/latticeai/services/model_runtime/__init__.py +322 -0
  139. package/latticeai/services/model_runtime/cloud.py +87 -0
  140. package/latticeai/services/model_runtime/download.py +282 -0
  141. package/latticeai/services/model_runtime/engines.py +341 -0
  142. package/latticeai/services/model_runtime/loading.py +178 -0
  143. package/latticeai/services/model_runtime/service.py +129 -0
  144. package/latticeai/services/model_runtime/state.py +131 -0
  145. package/latticeai/services/model_runtime/status.py +255 -0
  146. package/latticeai/services/product_readiness.py +15 -7
  147. package/latticeai/setup/wizard/__init__.py +126 -0
  148. package/latticeai/setup/wizard/catalog.py +172 -0
  149. package/latticeai/setup/wizard/detect.py +323 -0
  150. package/latticeai/setup/wizard/install.py +348 -0
  151. package/latticeai/setup/wizard/paths.py +168 -0
  152. package/latticeai/setup/wizard/plans.py +74 -0
  153. package/latticeai/setup/wizard/recommend.py +320 -0
  154. package/package.json +6 -2
  155. package/scripts/bump_version.py +14 -0
  156. package/scripts/capture_release_evidence.mjs +33 -21
  157. package/scripts/check_current_release_docs.mjs +1 -1
  158. package/scripts/check_i18n_namespace_coverage.mjs +41 -4
  159. package/scripts/check_max_file_lines.mjs +102 -0
  160. package/scripts/check_release_evidence_bound.mjs +30 -15
  161. package/scripts/check_screenshot_pixel_delta.py +34 -4
  162. package/scripts/check_server_i18n.mjs +1 -0
  163. package/scripts/generate_rust_parity_fixtures.py +562 -0
  164. package/scripts/lib/mock_server_fingerprint.mjs +94 -0
  165. package/scripts/release_screen_claims.json +31 -2
  166. package/src-tauri/Cargo.lock +361 -3
  167. package/src-tauri/Cargo.toml +6 -1
  168. package/src-tauri/src/backend.rs +349 -0
  169. package/src-tauri/src/folder.rs +33 -0
  170. package/src-tauri/src/main.rs +97 -399
  171. package/src-tauri/tauri.conf.json +1 -1
  172. package/static/app/asset-manifest.json +41 -37
  173. package/static/app/assets/Act-yYpYnn0v.js +1 -0
  174. package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
  175. package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
  176. package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
  177. package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
  178. package/static/app/assets/Capture-CFIRsFNE.js +1 -0
  179. package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
  180. package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
  181. package/static/app/assets/Library-DwO3yZST.js +1 -0
  182. package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
  183. package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
  184. package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
  185. package/static/app/assets/System-DW8F-2xL.js +1 -0
  186. package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
  187. package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
  188. package/static/app/assets/brain-Ci1CkWjM.js +1 -0
  189. package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
  190. package/static/app/assets/circle-check-DfInj-qD.js +1 -0
  191. package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
  192. package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
  193. package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
  194. package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
  195. package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
  196. package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
  197. package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
  198. package/static/app/assets/index-_u5iUHDr.js +10 -0
  199. package/static/app/assets/input-B0lPdRQZ.js +1 -0
  200. package/static/app/assets/link-2-CoFbooHS.js +1 -0
  201. package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
  202. package/static/app/assets/primitives-DEbN-d6p.js +1 -0
  203. package/static/app/assets/search-BybIWPNd.js +1 -0
  204. package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
  205. package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
  206. package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
  207. package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
  208. package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
  209. package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
  210. package/static/app/assets/utils-BlZr7Pd4.js +4 -0
  211. package/static/app/assets/workspace-jJY4RuAV.js +1 -0
  212. package/static/app/index.html +4 -4
  213. package/static/sw.js +1 -1
  214. package/lattice_brain/graph/_kg_common.py +0 -1331
  215. package/lattice_brain/graph/discovery_index.py +0 -1141
  216. package/lattice_brain/graph/retrieval.py +0 -1120
  217. package/lattice_brain/graph/retrieval_vector.py +0 -1293
  218. package/lattice_brain/ingestion.py +0 -1525
  219. package/lattice_brain/multimodal.py +0 -1258
  220. package/latticeai/core/agent.py +0 -1465
  221. package/latticeai/core/embedding_providers.py +0 -1196
  222. package/latticeai/core/file_generation.py +0 -1047
  223. package/latticeai/integrations/telegram_bot.py +0 -1390
  224. package/latticeai/models/router.py +0 -1007
  225. package/latticeai/runtime/build_phases.py +0 -1450
  226. package/latticeai/services/brain_intelligence.py +0 -1083
  227. package/latticeai/services/memory_service.py +0 -1177
  228. package/latticeai/services/model_runtime.py +0 -1281
  229. package/latticeai/setup/wizard.py +0 -1310
  230. package/static/app/assets/Act-AWf0SAKp.js +0 -1
  231. package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
  232. package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
  233. package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
  234. package/static/app/assets/Capture-CqOSzyPr.js +0 -1
  235. package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
  236. package/static/app/assets/Library-CX-bbhmK.js +0 -1
  237. package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
  238. package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
  239. package/static/app/assets/System-Bu2t5hn1.js +0 -1
  240. package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
  241. package/static/app/assets/brain-DJMoqrwx.js +0 -1
  242. package/static/app/assets/index-BpYkzcVm.js +0 -10
  243. package/static/app/assets/input-DSlJJxRs.js +0 -1
  244. package/static/app/assets/primitives-BCx6TvfG.js +0 -1
  245. package/static/app/assets/search-Cgy8cCFJ.js +0 -1
  246. package/static/app/assets/utils-zqPZJxdx.js +0 -4
  247. package/static/app/assets/workspace-DXTihhfU.js +0 -1
@@ -1,1331 +0,0 @@
1
- """
2
- SQLite knowledge graph for Lattice AI workspace memory.
3
-
4
- The graph keeps raw event JSON, normalized node metadata, and edges in one
5
- portable database so it can later migrate to Neo4j/Postgres without changing
6
- the ingestion contract.
7
- """
8
-
9
- # ruff: noqa: F401,F841
10
-
11
- import asyncio
12
- import hashlib
13
- import json
14
- import logging
15
- import math
16
- import os
17
- import platform
18
- import re
19
- import shutil
20
- import sqlite3
21
- import time
22
- import zipfile
23
- from collections import Counter
24
- from contextlib import contextmanager
25
- from datetime import datetime
26
- from pathlib import Path
27
- from typing import Any, Dict, Iterable, Iterator, List, Optional, Tuple
28
-
29
- try:
30
- from .schema import EdgeType, KGStoreV2, NodeType, _exec_script
31
- except Exception: # pragma: no cover - v2 schema is optional at import time
32
- KGStoreV2 = None # type: ignore[assignment,misc]
33
- NodeType = None # type: ignore[assignment,misc]
34
- EdgeType = None # type: ignore[assignment,misc]
35
- _exec_script = None # type: ignore[assignment]
36
-
37
- from ..embeddings import LocalEmbeddingModel
38
- from .json_utils import _json, _safe_loads
39
- from .runtime import get_llm_router, set_llm_router
40
-
41
- # Default read source for the graph queries: v2 reconstruction views.
42
- # Override with LATTICEAI_KG_READ_V2=0 to fall back to the legacy tables.
43
- _READ_FROM_V2_DEFAULT = os.getenv("LATTICEAI_KG_READ_V2", "1") != "0"
44
-
45
- # Static constants (projection/format versions, local-ingestion classification
46
- # tables, OS exclusion lists) live in ._kg_constants; re-exported here so every
47
- # existing ``from ._kg_common import <CONST>`` site is unaffected.
48
- from ..quiet import quiet
49
- from ._kg_constants import ( # noqa: E402
50
- _KG_DB_FORMAT_KEY,
51
- _KG_DB_FORMAT_VERSION,
52
- _PROJECTION_VERSION,
53
- _V2_WRITE_MASTER_KEY,
54
- COMMON_EXCLUDED_DIRS,
55
- COMMON_EXCLUDED_FILE_NAMES,
56
- COMMON_EXCLUDED_FILE_SUFFIXES,
57
- GRAPH_SCHEMA_VERSION,
58
- LINUX_EXCLUDED_PREFIXES,
59
- LOCAL_CODE_EXTENSIONS,
60
- LOCAL_DOCUMENT_EXTENSIONS,
61
- LOCAL_IMAGE_EXTENSIONS,
62
- LOCAL_SIZE_LIMITS,
63
- LOCAL_SLIDE_EXTENSIONS,
64
- LOCAL_SPREADSHEET_EXTENSIONS,
65
- LOCAL_SUPPORTED_EXTENSIONS,
66
- LOCAL_TEXT_EXTENSIONS,
67
- MACOS_EXCLUDED_PREFIXES,
68
- SENSITIVE_PATH_KEYWORDS,
69
- WINDOWS_EXCLUDED_NAMES,
70
- )
71
-
72
- # Pure fs/path/hash/classification helpers → ._kg_fsutil, re-exported so the
73
- # static __all__ below forwards them to the graph mixins. Listed explicitly
74
- # rather than star-imported: a star import here made every name in this
75
- # module unverifiable to both ruff and mypy.
76
- from ._kg_fsutil import ( # noqa: E402,F401
77
- _current_os_type,
78
- _drive_id_for_path,
79
- _excluded_directory_reason,
80
- _file_category,
81
- _is_hidden_path,
82
- _is_relative_to,
83
- _node_type_for_category,
84
- _now,
85
- _parse_iso,
86
- _parser_type_for_category,
87
- _path_fingerprint,
88
- _path_parts_lower,
89
- _recency_score,
90
- _root_warning,
91
- _safe_iso_from_stat_mtime,
92
- _sample_file,
93
- _sensitive_file_reason,
94
- _sha256_bytes,
95
- _sha256_text,
96
- _size_limit_for_category,
97
- _slug,
98
- )
99
-
100
-
101
- def _clean_text(text: str) -> str:
102
- return re.sub(r"\s+", " ", str(text or "")).strip()
103
-
104
-
105
- def _chunks(text: str, size: int = 1200, overlap: int = 160) -> List[str]:
106
- cleaned = str(text or "").strip()
107
- if not cleaned:
108
- return []
109
- chunks: List[str] = []
110
- start = 0
111
- while start < len(cleaned):
112
- end = min(len(cleaned), start + size)
113
- chunks.append(cleaned[start:end])
114
- if end >= len(cleaned):
115
- break
116
- start = max(0, end - overlap)
117
- return chunks
118
-
119
-
120
- # ── Typed chunking (review 2026-07-25 §5.2 S2 — Wave 2.1 + 2.4) ──────────────
121
- # ``_chunks`` above is a compatibility contract (chunk ids hash over the chunk
122
- # text) and stays byte-for-byte untouched. ``typed_chunks`` layers strategy-
123
- # aware boundaries plus per-chunk provenance (start_char / heading_path) on
124
- # top; ``strategy="plain"`` reproduces the exact ``_chunks`` boundaries so
125
- # unchanged plain content keeps identical chunk ids.
126
-
127
- _MARKDOWN_CHUNK_EXTENSIONS = {".md", ".markdown"}
128
- _CODE_CHUNK_EXTENSIONS = {
129
- ".py", ".js", ".jsx", ".ts", ".tsx", ".go", ".rs", ".java", ".rb",
130
- ".c", ".h", ".cpp", ".css", ".sh", ".sql", ".vue", ".svelte",
131
- ".json", ".yaml", ".yml", ".toml",
132
- }
133
- _PROSE_CHUNK_EXTENSIONS = {
134
- ".txt", ".pdf", ".docx", ".doc", ".rtf", ".odt", ".epub", ".html", ".htm",
135
- }
136
- _CHUNK_STRATEGIES = {"plain", "markdown", "code", "prose"}
137
- # Markdown sections smaller than this merge forward into the next section so
138
- # heading-dense documents don't shatter into confetti chunks.
139
- _MARKDOWN_MIN_SECTION_CHARS = 200
140
- _MARKDOWN_HEADING_RE = re.compile(r"^(#{1,6}) (.*)$", re.MULTILINE)
141
- _CODE_BOUNDARY_LINE_RE = re.compile(
142
- r"^(?:def |class |function |export |const |public |private )", re.MULTILINE
143
- )
144
- _CODE_BLANK_RUN_RE = re.compile(r"\n\s*\n")
145
-
146
-
147
- def chunk_strategy_for(filename: Any, *, content_type: str = "") -> str:
148
- """Route a filename / path / URI (plus optional MIME hint) to a strategy.
149
-
150
- Returns ``"markdown"`` for .md/.markdown, ``"code"`` for known source-code
151
- extensions, ``"prose"`` for document formats whose text is running prose
152
- (.txt/.pdf/.docx/.html/…), ``"plain"`` otherwise. Case-insensitive,
153
- tolerant of URLs (query/fragment stripped) and ``Path`` objects; never
154
- raises — any malformed input falls back to ``"plain"``.
155
-
156
- Unknown/extension-less input stays ``"plain"`` on purpose: the plain
157
- strategy is the byte-compatible legacy walk, and guessing prose for
158
- something that might be a data dump would move chunk boundaries for no
159
- retrieval gain.
160
- """
161
- try:
162
- name = str(filename or "").strip().lower()
163
- for sep in ("?", "#"):
164
- name = name.split(sep, 1)[0]
165
- name = name.replace("\\", "/").rstrip("/").rsplit("/", 1)[-1]
166
- dot = name.rfind(".")
167
- ext = name[dot:] if dot > 0 else ""
168
- if ext in _MARKDOWN_CHUNK_EXTENSIONS:
169
- return "markdown"
170
- if ext in _CODE_CHUNK_EXTENSIONS:
171
- return "code"
172
- if ext in _PROSE_CHUNK_EXTENSIONS:
173
- return "prose"
174
- mime = str(content_type or "").strip().lower()
175
- if "markdown" in mime:
176
- return "markdown"
177
- if mime.startswith("text/html") or mime.startswith("text/plain"):
178
- return "prose"
179
- except Exception:
180
- quiet()
181
- return "plain"
182
-
183
-
184
- def _plain_windows(
185
- cleaned: str,
186
- size: int,
187
- overlap: int,
188
- *,
189
- base_offset: int = 0,
190
- strategy: str = "plain",
191
- heading_path: Optional[str] = None,
192
- ) -> List[Dict[str, Any]]:
193
- """The exact ``_chunks`` walk with ``start_char`` tracked.
194
-
195
- Boundaries and chunk texts are byte-identical to ``_chunks`` over the same
196
- string — this is the plain-strategy compatibility guarantee.
197
- """
198
- out: List[Dict[str, Any]] = []
199
- start = 0
200
- total = len(cleaned)
201
- while start < total:
202
- end = min(total, start + size)
203
- out.append(
204
- {
205
- "text": cleaned[start:end],
206
- "meta": {
207
- "strategy": strategy,
208
- "start_char": base_offset + start,
209
- "heading_path": heading_path,
210
- },
211
- }
212
- )
213
- if end >= total:
214
- break
215
- start = max(0, end - overlap)
216
- return out
217
-
218
-
219
- def _markdown_section_spans(cleaned: str) -> List[Tuple[int, int, Optional[str]]]:
220
- """``(start, end, heading_path)`` spans split at ``^#{1,6} `` heading lines.
221
-
222
- ``heading_path`` is the " > "-joined path of the enclosing headings
223
- including the section's own heading (e.g. ``"Guide > Setup"``); the
224
- preamble before the first heading carries ``None``. Spans are contiguous
225
- raw slices of ``cleaned`` so every chunk text round-trips via start_char.
226
- """
227
- spans: List[Tuple[int, int, Optional[str]]] = []
228
- stack: List[Tuple[int, str]] = []
229
- prev_start = 0
230
- prev_path: Optional[str] = None
231
- for match in _MARKDOWN_HEADING_RE.finditer(cleaned):
232
- offset = match.start()
233
- if offset > prev_start:
234
- spans.append((prev_start, offset, prev_path))
235
- level = len(match.group(1))
236
- while stack and stack[-1][0] >= level:
237
- stack.pop()
238
- stack.append((level, match.group(2).strip()))
239
- prev_start = offset
240
- prev_path = " > ".join(title for _, title in stack) or None
241
- if len(cleaned) > prev_start:
242
- spans.append((prev_start, len(cleaned), prev_path))
243
- return spans
244
-
245
-
246
- def _merge_small_sections(
247
- spans: List[Tuple[int, int, Optional[str]]], min_chars: int
248
- ) -> List[Tuple[int, int, Optional[str]]]:
249
- """Merge sections under ``min_chars`` forward into the next section.
250
-
251
- A merged section keeps the heading_path of its first constituent (the
252
- path in effect at the chunk start). A trailing undersized section merges
253
- backward into the previous emitted section when one exists.
254
- """
255
- merged: List[Tuple[int, int, Optional[str]]] = []
256
- pending: Optional[Tuple[int, int, Optional[str]]] = None
257
- for start, end, path in spans:
258
- if pending is None:
259
- pending = (start, end, path)
260
- else:
261
- pending = (pending[0], end, pending[2])
262
- if pending[1] - pending[0] >= min_chars:
263
- merged.append(pending)
264
- pending = None
265
- if pending is not None:
266
- if merged and pending[1] - pending[0] < min_chars:
267
- last = merged.pop()
268
- merged.append((last[0], pending[1], last[2]))
269
- else:
270
- merged.append(pending)
271
- return merged
272
-
273
-
274
- def _markdown_chunks(cleaned: str, size: int, overlap: int) -> List[Dict[str, Any]]:
275
- sections = _merge_small_sections(
276
- _markdown_section_spans(cleaned), _MARKDOWN_MIN_SECTION_CHARS
277
- )
278
- out: List[Dict[str, Any]] = []
279
- for start, end, path in sections:
280
- body = cleaned[start:end]
281
- if len(body) <= size:
282
- out.append(
283
- {
284
- "text": body,
285
- "meta": {
286
- "strategy": "markdown",
287
- "start_char": start,
288
- "heading_path": path,
289
- },
290
- }
291
- )
292
- else:
293
- out.extend(
294
- _plain_windows(
295
- body,
296
- size,
297
- overlap,
298
- base_offset=start,
299
- strategy="markdown",
300
- heading_path=path,
301
- )
302
- )
303
- return out
304
-
305
-
306
- def _code_segment_spans(cleaned: str) -> List[Tuple[int, int]]:
307
- """Contiguous top-level segments split at blank-line runs and decl lines."""
308
- boundaries = {0, len(cleaned)}
309
- for match in _CODE_BLANK_RUN_RE.finditer(cleaned):
310
- boundaries.add(match.end())
311
- for match in _CODE_BOUNDARY_LINE_RE.finditer(cleaned):
312
- boundaries.add(match.start())
313
- ordered = sorted(boundaries)
314
- return [
315
- (ordered[i], ordered[i + 1])
316
- for i in range(len(ordered) - 1)
317
- if ordered[i + 1] > ordered[i]
318
- ]
319
-
320
-
321
- def _code_chunks(cleaned: str, size: int, overlap: int) -> List[Dict[str, Any]]:
322
- hard_limit = int(size * 1.5)
323
- out: List[Dict[str, Any]] = []
324
- pack: Optional[Tuple[int, int]] = None
325
-
326
- def _emit(span: Tuple[int, int]) -> None:
327
- out.append(
328
- {
329
- "text": cleaned[span[0] : span[1]],
330
- "meta": {
331
- "strategy": "code",
332
- "start_char": span[0],
333
- "heading_path": None,
334
- },
335
- }
336
- )
337
-
338
- for start, end in _code_segment_spans(cleaned):
339
- if end - start > hard_limit:
340
- # Monster segment: flush the pack, then window it like plain text.
341
- if pack is not None:
342
- _emit(pack)
343
- pack = None
344
- out.extend(
345
- _plain_windows(
346
- cleaned[start:end],
347
- size,
348
- overlap,
349
- base_offset=start,
350
- strategy="code",
351
- )
352
- )
353
- continue
354
- if pack is None:
355
- pack = (start, end)
356
- elif end - pack[0] <= size:
357
- pack = (pack[0], end)
358
- else:
359
- _emit(pack)
360
- pack = (start, end)
361
- if pack is not None:
362
- _emit(pack)
363
- return out
364
-
365
-
366
- # ── Prose chunking (review 2026-07-27 P1 #4) ────────────────────────────────
367
- # The plain walk cuts every ``size`` characters, which lands mid-sentence and
368
- # — for Korean, where the verb carrying the meaning sits at the end — routinely
369
- # splits a claim from its predicate. Retrieval then matches half a statement
370
- # and the citation shows a fragment. The prose strategy keeps the same window
371
- # budget but ends each chunk at the last sentence/paragraph boundary inside it.
372
-
373
- # Strong: sentence-final punctuation (ASCII + CJK) with optional closing
374
- # quotes/brackets, followed by whitespace; or a blank-line paragraph break.
375
- _PROSE_STRONG_BOUNDARY_RE = re.compile(
376
- r"(?:[.!?。!?…]+[\"'”’」』\)\]]*\s+|\n[ \t]*\n)"
377
- )
378
- # Weak: a single line break. Korean notes and bullet lists often carry no
379
- # sentence punctuation at all; a line end is still a real boundary there.
380
- _PROSE_WEAK_BOUNDARY_RE = re.compile(r"\n")
381
- # Never emit a chunk shorter than this fraction of ``size`` just to hit a
382
- # boundary — tiny chunks hurt recall more than a mid-sentence cut.
383
- _PROSE_MIN_SPAN_RATIO = 0.5
384
-
385
-
386
- def _last_boundary(cleaned: str, lo: int, hi: int) -> Optional[int]:
387
- """End offset of the last sentence/paragraph boundary in ``cleaned[lo:hi]``.
388
-
389
- Strong boundaries win; a single line break is the fallback. Returns None
390
- when the span holds neither, so the caller keeps the hard window cut.
391
- """
392
- window = cleaned[lo:hi]
393
- for pattern in (_PROSE_STRONG_BOUNDARY_RE, _PROSE_WEAK_BOUNDARY_RE):
394
- last = None
395
- for match in pattern.finditer(window):
396
- last = match.end()
397
- if last:
398
- return lo + last
399
- return None
400
-
401
-
402
- def _prose_chunks(cleaned: str, size: int, overlap: int) -> List[Dict[str, Any]]:
403
- out: List[Dict[str, Any]] = []
404
- total = len(cleaned)
405
- min_span = max(1, int(size * _PROSE_MIN_SPAN_RATIO))
406
- start = 0
407
- while start < total:
408
- hard_end = min(total, start + size)
409
- end = hard_end
410
- if hard_end < total:
411
- boundary = _last_boundary(cleaned, start + min_span, hard_end)
412
- if boundary is not None and boundary > start:
413
- end = boundary
414
- out.append(
415
- {
416
- "text": cleaned[start:end],
417
- "meta": {
418
- "strategy": "prose",
419
- "start_char": start,
420
- "heading_path": None,
421
- },
422
- }
423
- )
424
- if end >= total:
425
- break
426
- # Overlap carries the tail of the previous chunk into the next one so
427
- # a claim split across a boundary is still retrievable from both.
428
- start = max(start + 1, end - overlap)
429
- return out
430
-
431
-
432
- def typed_chunks(
433
- text: str,
434
- *,
435
- strategy: str = "plain",
436
- size: int = 1200,
437
- overlap: int = 160,
438
- ) -> List[Dict[str, Any]]:
439
- """Strategy-aware chunking with per-chunk provenance metadata.
440
-
441
- Returns ``[{"text": str, "meta": {"strategy", "start_char", "heading_path"}}]``
442
- where ``start_char`` is the offset in ``str(text or "").strip()`` (every
443
- chunk text is an exact substring at that offset).
444
-
445
- Contract: ``[c["text"] for c in typed_chunks(t)] == _chunks(t)`` for the
446
- default plain strategy — unknown strategies also fall back to plain.
447
- """
448
- cleaned = str(text or "").strip()
449
- if not cleaned:
450
- return []
451
- try:
452
- size = max(1, int(size))
453
- except Exception:
454
- size = 1200
455
- try:
456
- overlap = min(max(0, int(overlap)), size - 1)
457
- except Exception:
458
- overlap = min(160, size - 1)
459
- label = strategy if strategy in _CHUNK_STRATEGIES else "plain"
460
- if label == "markdown":
461
- return _markdown_chunks(cleaned, size, overlap)
462
- if label == "code":
463
- return _code_chunks(cleaned, size, overlap)
464
- if label == "prose":
465
- return _prose_chunks(cleaned, size, overlap)
466
- return _plain_windows(cleaned, size, overlap)
467
-
468
-
469
- def typed_chunk_meta_fields(piece: Dict[str, Any]) -> Dict[str, Any]:
470
- """Additive chunk-metadata fields for one ``typed_chunks`` piece.
471
-
472
- Ingest call sites merge this into the existing ``{"index", "source_node"}``
473
- chunk metadata; ``heading_path`` is only present when known — honest
474
- absence over empty labels.
475
- """
476
- meta = piece.get("meta") or {}
477
- fields: Dict[str, Any] = {
478
- "strategy": str(meta.get("strategy") or "plain"),
479
- "start_char": int(meta.get("start_char") or 0),
480
- }
481
- heading_path = meta.get("heading_path")
482
- if heading_path:
483
- fields["heading_path"] = str(heading_path)
484
- return fields
485
-
486
-
487
- def citation_locator(chunk_metadata: Any) -> str:
488
- """Human "where in the document" label for one chunk, or "".
489
-
490
- Built only from provenance the chunk actually carries — a section heading
491
- path and/or a page number. When neither is known the answer is the empty
492
- string, so a citation never claims a location it cannot prove.
493
- """
494
- if not isinstance(chunk_metadata, dict):
495
- return ""
496
- parts: List[str] = []
497
- heading = str(chunk_metadata.get("heading_path") or "").strip()
498
- if heading:
499
- parts.append(heading)
500
- def _page(key: str) -> int:
501
- value = chunk_metadata.get(key)
502
- try:
503
- return int(value) if value is not None else 0
504
- except (TypeError, ValueError):
505
- return 0
506
-
507
- page_number = _page("page")
508
- if page_number > 0:
509
- page_end = _page("page_end")
510
- parts.append(
511
- f"p.{page_number}–{page_end}" if page_end > page_number else f"p.{page_number}"
512
- )
513
- return " · ".join(parts)
514
-
515
-
516
- def pdf_page_offsets(structure: Any) -> List[int]:
517
- """Start offset of each PDF page in the "\\n\\n"-joined page text.
518
-
519
- ``structure`` is the ``metadata["structure"]`` dict produced by
520
- ``_pdf_structure`` (``pages`` = ``[{"chars": int, ...}, ...]``); pages were
521
- joined with ``"\\n\\n"`` (see ``read_document``), so page k starts at
522
- ``sum(chars[j] + 2 for j < k)``. Empty or malformed input returns ``[]``.
523
- """
524
- if not isinstance(structure, dict):
525
- return []
526
- pages = structure.get("pages")
527
- if not isinstance(pages, list) or not pages:
528
- return []
529
- offsets: List[int] = []
530
- cursor = 0
531
- for page in pages:
532
- if not isinstance(page, dict):
533
- return []
534
- chars = page.get("chars")
535
- if isinstance(chars, bool) or not isinstance(chars, (int, float)) or chars < 0:
536
- return []
537
- offsets.append(cursor)
538
- cursor += int(chars) + 2 # +2 for the "\n\n" page joiner
539
- return offsets
540
-
541
-
542
- def page_for_offset(page_offsets: List[int], offset: int) -> Optional[int]:
543
- """1-based page number containing ``offset`` given page start offsets.
544
-
545
- Returns ``None`` when ``page_offsets`` is empty or the offset precedes the
546
- first page start (honest absence over a wrong label).
547
- """
548
- if not page_offsets:
549
- return None
550
- try:
551
- target = int(offset)
552
- except Exception:
553
- return None
554
- page = 0
555
- for index, start in enumerate(page_offsets):
556
- try:
557
- if target >= int(start):
558
- page = index + 1
559
- else:
560
- break
561
- except Exception:
562
- return None
563
- return page if page >= 1 else None
564
-
565
-
566
- _LLM_EXTRACT_CONCEPT_PROMPT = """Extract the key concepts from the following text.
567
- Return ONLY a JSON array of objects, each with "concept" (string) and "importance" (float 0-1).
568
- Extract up to {limit} concepts. Focus on named entities, technical terms, and domain-specific nouns.
569
- Do NOT include common words, stop words, or generic terms.
570
-
571
- Text:
572
- {text}
573
-
574
- JSON:"""
575
-
576
- _LLM_EXTRACT_TRIPLE_PROMPT = """Extract relationship triples from the following text.
577
- Return ONLY a JSON array of objects, each with:
578
- - "subject": source concept (string)
579
- - "relation": relationship verb (string, Korean or English)
580
- - "object": target concept (string)
581
- - "evidence": the sentence supporting this triple (string, max 240 chars)
582
- - "confidence": how confident you are (float 0-1)
583
-
584
- Extract up to {limit} triples. Focus on meaningful semantic relationships.
585
-
586
- Text:
587
- {text}
588
-
589
- Concepts already identified: {concepts}
590
-
591
- JSON:"""
592
-
593
- ENABLE_LLM_EXTRACTION = os.getenv("LATTICEAI_LLM_EXTRACTION", "true").lower() in (
594
- "1",
595
- "true",
596
- "yes",
597
- )
598
-
599
-
600
- def _llm_extract_concepts(text: str, limit: int = 12) -> Optional[List[str]]:
601
- router = get_llm_router()
602
- if not ENABLE_LLM_EXTRACTION or not router:
603
- return None
604
- if not router.current_model_id:
605
- return None
606
- prompt = _LLM_EXTRACT_CONCEPT_PROMPT.format(text=text[:3000], limit=limit)
607
- try:
608
- loop = asyncio.get_event_loop()
609
- if loop.is_running():
610
- import concurrent.futures
611
-
612
- with concurrent.futures.ThreadPoolExecutor(max_workers=1) as pool:
613
- future = pool.submit(
614
- asyncio.run,
615
- router.generate(prompt, max_tokens=1024, temperature=0.1),
616
- )
617
- raw = future.result(timeout=30)
618
- else:
619
- raw = asyncio.run(
620
- router.generate(prompt, max_tokens=1024, temperature=0.1)
621
- )
622
- raw = raw.strip()
623
- if raw.startswith("```"):
624
- raw = re.sub(r"^```(?:json)?\s*", "", raw)
625
- raw = re.sub(r"\s*```$", "", raw)
626
- parsed = json.loads(raw)
627
- if isinstance(parsed, list):
628
- concepts = []
629
- for item in parsed[:limit]:
630
- if isinstance(item, dict) and "concept" in item:
631
- concepts.append(item["concept"])
632
- elif isinstance(item, str):
633
- concepts.append(item)
634
- return concepts if concepts else None
635
- except Exception as e:
636
- logging.debug("LLM concept extraction failed (falling back to rules): %s", e)
637
- return None
638
-
639
-
640
- # Triples carry a numeric ``weight``/``confidence`` alongside string fields,
641
- # so the value type is Any rather than str.
642
- def _llm_extract_triples(
643
- text: str, concepts: List[str], limit: int = 20
644
- ) -> Optional[List[Dict[str, Any]]]:
645
- router = get_llm_router()
646
- if not ENABLE_LLM_EXTRACTION or not router:
647
- return None
648
- if not router.current_model_id:
649
- return None
650
- prompt = _LLM_EXTRACT_TRIPLE_PROMPT.format(
651
- text=text[:3000],
652
- limit=limit,
653
- concepts=", ".join(concepts[:15]),
654
- )
655
- try:
656
- loop = asyncio.get_event_loop()
657
- if loop.is_running():
658
- import concurrent.futures
659
-
660
- with concurrent.futures.ThreadPoolExecutor(max_workers=1) as pool:
661
- future = pool.submit(
662
- asyncio.run,
663
- router.generate(prompt, max_tokens=2048, temperature=0.1),
664
- )
665
- raw = future.result(timeout=30)
666
- else:
667
- raw = asyncio.run(
668
- router.generate(prompt, max_tokens=2048, temperature=0.1)
669
- )
670
- raw = raw.strip()
671
- if raw.startswith("```"):
672
- raw = re.sub(r"^```(?:json)?\s*", "", raw)
673
- raw = re.sub(r"\s*```$", "", raw)
674
- parsed = json.loads(raw)
675
- if isinstance(parsed, list):
676
- triples: List[Dict[str, Any]] = []
677
- for item in parsed[:limit]:
678
- if isinstance(item, dict) and "subject" in item and "object" in item:
679
- relation = str(item.get("relation", "관련됨"))
680
- evidence_text = str(item.get("evidence", ""))[:240]
681
- confidence = float(item.get("confidence", 0.8))
682
- # An LLM triple that names a real verb and cites the
683
- # sentence it came from is semantic evidence; a bare
684
- # "관련됨" with no quoted evidence is the model restating
685
- # co-occurrence, and is weighted (and labelled) as such.
686
- is_semantic = bool(evidence_text) and relation != "관련됨"
687
- triples.append(
688
- {
689
- "subject": str(item["subject"]),
690
- "relation": relation,
691
- "object": str(item["object"]),
692
- "context": evidence_text,
693
- "confidence": confidence,
694
- "evidence": "verb" if is_semantic else "cooccurrence",
695
- "weight": round(
696
- (VERB_EDGE_WEIGHT if is_semantic else COOCCURRENCE_EDGE_WEIGHT)
697
- * max(0.1, min(confidence, 1.0)),
698
- 4,
699
- ),
700
- }
701
- )
702
- return triples if triples else None
703
- except Exception as e:
704
- logging.debug("LLM triple extraction failed (falling back to rules): %s", e)
705
- return None
706
-
707
-
708
- _CONCEPT_STOP: set = {
709
- # English stop words
710
- "the",
711
- "and",
712
- "for",
713
- "with",
714
- "this",
715
- "that",
716
- "from",
717
- "into",
718
- "which",
719
- "are",
720
- "was",
721
- "were",
722
- "has",
723
- "have",
724
- "had",
725
- "can",
726
- "will",
727
- "would",
728
- "could",
729
- "should",
730
- "may",
731
- "might",
732
- "must",
733
- "shall",
734
- "being",
735
- "been",
736
- "also",
737
- "just",
738
- "then",
739
- "than",
740
- "when",
741
- "where",
742
- "what",
743
- "how",
744
- "why",
745
- "its",
746
- "their",
747
- "your",
748
- "our",
749
- "you",
750
- "they",
751
- "them",
752
- "these",
753
- "those",
754
- "use",
755
- "used",
756
- "using",
757
- "based",
758
- "like",
759
- "such",
760
- "via",
761
- "per",
762
- "let",
763
- "yes",
764
- "not",
765
- "but",
766
- "all",
767
- "any",
768
- "out",
769
- "new",
770
- "get",
771
- "set",
772
- # Korean stop words
773
- "사용자",
774
- "내용",
775
- "파일",
776
- "채팅",
777
- "답변",
778
- "입니다",
779
- "그리고",
780
- "처럼",
781
- "있어",
782
- "없어",
783
- "이야",
784
- "이다",
785
- "한다",
786
- "하다",
787
- "되다",
788
- "됩니다",
789
- "경우",
790
- "방법",
791
- "부분",
792
- "상태",
793
- "정도",
794
- "결과",
795
- "이후",
796
- "이전",
797
- "그것",
798
- "이것",
799
- "저것",
800
- "여기",
801
- "거기",
802
- "저기",
803
- "우리",
804
- "저희",
805
- "기능",
806
- "서버",
807
- "모델",
808
- "설정",
809
- "설명",
810
- "버전",
811
- "지원",
812
- "사용",
813
- "실행",
814
- "todo",
815
- "fixme",
816
- "note",
817
- "참고",
818
- "주의",
819
- "warning",
820
- }
821
-
822
-
823
- def _extract_concepts(text: str, limit: int = 12) -> List[str]:
824
- """LLM-first concept extraction with rule-based fallback."""
825
- llm_result = _llm_extract_concepts(text, limit)
826
- if llm_result:
827
- return llm_result
828
- return _extract_concepts_rules(text, limit)
829
-
830
-
831
- def _extract_concepts_rules(text: str, limit: int = 12) -> List[str]:
832
- """Extract meaningful named concepts from text (rule-based).
833
-
834
- Priority order:
835
- 1. Backtick / quoted terms (explicitly technical)
836
- 2. Multi-word proper nouns (Lattice AI, GPT-4o, Claude Sonnet)
837
- 3. Single capitalized proper nouns not at sentence start (Claude, Python, FastAPI)
838
- 4. Korean compound technical terms (멀티모달, 에이전트, 그래프RAG)
839
- 5. Hyphenated / versioned identifiers (gpt-4o, mlx-vlm, gemma-4)
840
- """
841
- text = str(text or "")
842
- seen: dict = {} # concept_lower → original form
843
-
844
- def _add(term: str) -> None:
845
- key = term.strip().lower()
846
- if key and key not in _CONCEPT_STOP and not key.isdigit() and len(key) >= 2:
847
- seen.setdefault(key, term.strip())
848
-
849
- # 1. Backtick-quoted code/term (highest confidence)
850
- for m in re.findall(r"`([^`]{2,40})`", text):
851
- if not re.search(r"[\(\)\[\]{}]", m): # skip code expressions
852
- _add(m)
853
-
854
- # 2. Double/single quoted terms
855
- for m in re.findall(r'"([^"]{2,40})"', text):
856
- _add(m)
857
-
858
- # 3. Multi-word English proper nouns (Title Case or ALL-CAPS first word, 2–4 words).
859
- # Pattern A: Mixed-case first word — "Lattice AI", "Tool Use", "Graph RAG"
860
- for m in re.findall(
861
- r"([A-Z][a-z]{1,20}(?:\s+(?:[A-Z]{2,10}|[A-Z][a-z0-9]{1,20}|\d[\w.]{0,6})){1,3})",
862
- text,
863
- ):
864
- _add(m)
865
- # Pattern B: ALL-CAPS first word — "VS Code", "MCP Server", "GPT-4o Mini"
866
- for m in re.findall(
867
- r"([A-Z]{2,6}(?:\s+(?:[A-Z]{2,10}|[A-Z][a-z0-9]{1,20})){1,2})",
868
- text,
869
- ):
870
- _add(m)
871
-
872
- # 4. Single capitalized proper noun.
873
- # Use ASCII-boundary lookaround instead of \b so Korean particles
874
- # (와, 의, 는 …) after an English word don't block the match.
875
- all_caps_words = re.findall(
876
- r"(?<![A-Za-z0-9])([A-Z][A-Za-z0-9]{2,24})(?![A-Za-z0-9])", text
877
- )
878
- freq: Dict[str, int] = {}
879
- for w in all_caps_words:
880
- freq[w] = freq.get(w, 0) + 1
881
- sentence_starts = set(re.findall(r"(?:^|(?<=[.!?])\s+)([A-Z][a-z]+)", text))
882
- for m, cnt in freq.items():
883
- if m.lower() in _CONCEPT_STOP:
884
- continue
885
- if cnt >= 2 or m not in sentence_starts:
886
- _add(m)
887
-
888
- # 5. Korean technical compound nouns (3–12 chars, no common particles)
889
- for m in re.findall(
890
- r"[가-힣]{2,12}(?:AI|LLM|API|UI|RAG|bot|Bot|기능|모델|서버|에이전트|파이프라인|워크플로)",
891
- text,
892
- ):
893
- _add(m)
894
- # Korean standalone terms that appear after topic markers (은/는/이/가 앞)
895
- for m in re.findall(
896
- r"([가-힣]{2,12})(?:은|는|이|가|을|를|의|에서|으로|와|과)", text
897
- ):
898
- if m.lower() not in _CONCEPT_STOP and len(m) >= 2:
899
- # Only add if it's non-trivial (has 3+ chars or appears multiple times)
900
- cnt = text.count(m)
901
- if len(m) >= 3 or cnt >= 2:
902
- _add(m)
903
-
904
- # 6. Hyphenated / versioned identifiers (gpt-4o, gemma-4, mlx-vlm)
905
- for m in re.findall(r"\b([a-zA-Z][a-zA-Z0-9]*(?:-[a-zA-Z0-9.]+)+)\b", text):
906
- if len(m) >= 4:
907
- _add(m)
908
-
909
- # De-duplicate: remove shorter if ALL its occurrences in the source text
910
- # are followed immediately by the suffix that forms the longer concept.
911
- # "Lattice" → dropped when every occurrence is "Lattice AI"
912
- # "Claude" → kept because it appears as just "Claude" too.
913
- values = list(seen.values())
914
- values_lower = [v.lower() for v in values]
915
- keep = set(range(len(values)))
916
- for i, v in enumerate(values):
917
- vl = v.lower()
918
- for j, wl in enumerate(values_lower):
919
- if i == j or j not in keep:
920
- continue
921
- # Check if vl is a word-prefix of wl
922
- suffix = wl[len(vl) :]
923
- if not (wl.startswith(vl) and re.match(r"^[\s\-]", suffix)):
924
- continue
925
- # Count occurrences of v NOT followed by the suffix
926
- suffix_stripped = suffix.lstrip(" -")
927
- # Escape for regex
928
- pattern_with_suffix = re.escape(v) + r"[\s\-]+" + re.escape(suffix_stripped)
929
- pattern_alone = (
930
- re.escape(v) + r"(?![\s\-]*" + re.escape(suffix_stripped) + r")"
931
- )
932
- alone_count = len(re.findall(pattern_alone, text, re.IGNORECASE))
933
- if alone_count == 0:
934
- # Shorter term never appears alone → safe to remove
935
- keep.discard(i)
936
- break
937
-
938
- final = [values[i] for i in range(len(values)) if i in keep]
939
- return final[:limit]
940
-
941
-
942
- # ──────────────────────────────────────────────────────────────────────────────
943
- # Node type taxonomy (점 = 명사)
944
- # ──────────────────────────────────────────────────────────────────────────────
945
- # Chat — 대화 세션
946
- # Document — 파일 (PDF·PPT·Word·Excel·이미지 등)
947
- # Concept — 개념·아이디어·기술 용어
948
- # Person — 사람 (사용자, 언급된 인물)
949
- # Error — 오류·버그·예외
950
- # Code — 코드 스니펫·함수·클래스
951
- # Feature — 소프트웨어 기능
952
- # Task — 할 일·액션 아이템
953
- # Decision — 결정 사항
954
-
955
- # Edge type vocabulary (선 = 동사 — 과거형 서술어)
956
- EDGE_VERB = {
957
- "언급함": r"언급|mention|refer|cited",
958
- "포함함": r"포함|include|consist|구성|탑재|contains",
959
- "해결함": r"해결|resolv|fix|수정|고쳤|closed",
960
- "의존함": r"의존|depend|require|필요|based on",
961
- "설명함": r"설명|explain|describe|정의|란|이란|means",
962
- "비교함": r"비교|versus|vs\.?|차이|다르|compare",
963
- "사용함": r"사용|use|활용|이용|apply",
964
- "연결함": r"연결|connect|통합|integrate|연동|link",
965
- "확장함": r"확장|extend|플러그인|plugin|addon",
966
- "생성함": r"생성|만들|create|generate|build|produced",
967
- "대체함": r"대체|replace|instead|alternative",
968
- "지원함": r"지원|support|제공|provide|offer",
969
- "발생함": r"발생|occur|throw|raise|triggered",
970
- "관련됨": r"관련|related|associated|연관",
971
- }
972
-
973
-
974
- # Concepts in a list-like sentence ("A, B, C, D를 사용한다") sit together by
975
- # enumeration, not by relation. Beyond this many concepts in one sentence, a
976
- # verb-less pairing is enumeration noise and is dropped outright.
977
- COOCCURRENCE_CONCEPT_LIMIT = 4
978
- # Verb-backed relations carry the sentence's own evidence; co-occurrence
979
- # relations carry only adjacency, so they enter the graph at a lower weight
980
- # and are labelled as such.
981
- VERB_EDGE_WEIGHT = 1.0
982
- COOCCURRENCE_EDGE_WEIGHT = 0.35
983
-
984
-
985
- def infer_edge_relation(sentence: str) -> Dict[str, Any]:
986
- """Classify the relation between two concepts in one sentence.
987
-
988
- Review 2026-07-27 P1 #6: the graph drifted toward co-occurrence because a
989
- verb-less sentence still produced a "관련됨" edge indistinguishable from a
990
- real semantic relation. The label alone cannot carry that difference, so
991
- the evidence class rides with it::
992
-
993
- {"relation": "사용함", "evidence": "verb", "weight": 1.0}
994
- {"relation": "관련됨", "evidence": "cooccurrence", "weight": 0.35}
995
-
996
- ``evidence`` is what the graph, the curator, and the UI use to tell a
997
- meaning edge from an adjacency edge — the honest distinction the previous
998
- label-only output erased.
999
- """
1000
- s = str(sentence or "").lower()
1001
- for label, pattern in EDGE_VERB.items():
1002
- if re.search(pattern, s):
1003
- # "관련됨" is itself a weak, generic label: matching it by keyword
1004
- # ("관련", "related") is still verb evidence, but nothing stronger.
1005
- return {
1006
- "relation": label,
1007
- "evidence": "verb",
1008
- "weight": VERB_EDGE_WEIGHT,
1009
- }
1010
- return {
1011
- "relation": "관련됨",
1012
- "evidence": "cooccurrence",
1013
- "weight": COOCCURRENCE_EDGE_WEIGHT,
1014
- }
1015
-
1016
-
1017
- def _infer_edge(sentence: str) -> str:
1018
- """Back-compat wrapper: the verb label only (see :func:`infer_edge_relation`)."""
1019
- return infer_edge_relation(sentence)["relation"]
1020
-
1021
-
1022
- # Technical words that cannot be person names
1023
- _NOT_PERSON_WORDS: set = {
1024
- "use",
1025
- "api",
1026
- "rag",
1027
- "sdk",
1028
- "ide",
1029
- "cli",
1030
- "llm",
1031
- "mcp",
1032
- "ui",
1033
- "ux",
1034
- "new",
1035
- "old",
1036
- "get",
1037
- "set",
1038
- "run",
1039
- "add",
1040
- "fix",
1041
- "tool",
1042
- "code",
1043
- "base",
1044
- "core",
1045
- "data",
1046
- "file",
1047
- "test",
1048
- "type",
1049
- "mode",
1050
- "view",
1051
- }
1052
-
1053
-
1054
- def _classify_node_type(concept: str, text: str) -> str:
1055
- """Classify a concept into the node taxonomy.
1056
-
1057
- Term-level signals take priority; then a tight ±60-char window is used
1058
- so distant keywords don't cause mis-classification.
1059
- """
1060
- term = concept.lower()
1061
-
1062
- # ── Term-level signals (highest confidence) ───────────────────────────
1063
- if re.search(r"(?:error|exception|traceback|오류|에러|버그)$", term, re.I):
1064
- return "Error"
1065
- if re.search(r"error|exception|err\b", term, re.I) and len(concept) < 30:
1066
- return "Error"
1067
- if re.search(r"\(\)|\.py$|\.js$|\.ts$|\.go$|::\w", term):
1068
- return "Code"
1069
-
1070
- # Person: "First Last" pattern, neither word is a known technical term
1071
- if re.match(r"^[A-Z][a-z]{1,15} [A-Z][a-z]{1,15}$", concept):
1072
- words = term.split()
1073
- if not any(w in _NOT_PERSON_WORDS for w in words):
1074
- return "Person"
1075
-
1076
- # ── Windowed context (±60 chars) — NOT used for Error to avoid false positives
1077
- idx = text.lower().find(term)
1078
- if idx >= 0:
1079
- win = text[max(0, idx - 60) : idx + len(concept) + 60].lower()
1080
- if re.search(r"def |class |function|함수|클래스|메서드|import", win):
1081
- return "Code"
1082
- # Feature: concept appears DIRECTLY adjacent to 기능/feature keyword
1083
- if len(concept) <= 12 and re.search(
1084
- rf"{re.escape(term)}.{{0,8}}(?:기능|feature)|(?:기능|feature).{{0,8}}{re.escape(term)}",
1085
- win,
1086
- ):
1087
- return "Feature"
1088
-
1089
- return "Concept"
1090
-
1091
-
1092
- def _extract_triples(
1093
- text: str,
1094
- concepts: List[str],
1095
- limit: int = 20,
1096
- ) -> List[Dict[str, str]]:
1097
- """LLM-first triple extraction with rule-based fallback."""
1098
- llm_result = _llm_extract_triples(text, concepts, limit)
1099
- if llm_result:
1100
- return llm_result
1101
- return _extract_triples_rules(text, concepts, limit)
1102
-
1103
-
1104
- def _extract_triples_rules(
1105
- text: str,
1106
- concepts: List[str],
1107
- limit: int = 20,
1108
- ) -> List[Dict[str, str]]:
1109
- """Extract (subject, verb-edge, object, context) triples from text (rule-based).
1110
-
1111
- For each sentence containing ≥2 concepts, infer the verb-form edge label
1112
- from surrounding context and create a directed triple.
1113
- """
1114
- if len(concepts) < 2:
1115
- return []
1116
-
1117
- concept_lower = {c.lower(): c for c in concepts}
1118
- triples: List[Dict[str, str]] = []
1119
- seen_pairs: set = set()
1120
-
1121
- # Split on sentence boundaries
1122
- sentences = re.split(r"(?<=[.!?\n])\s+|\n{2,}", text)
1123
- for sent in sentences:
1124
- sent = sent.strip()
1125
- if len(sent) < 8:
1126
- continue
1127
- sent_lower = sent.lower()
1128
-
1129
- present = [concept_lower[k] for k in concept_lower if k in sent_lower]
1130
- if len(present) < 2:
1131
- continue
1132
-
1133
- relation = infer_edge_relation(sent)
1134
- edge = relation["relation"]
1135
- # Enumeration guard (review 2026-07-27 P1 #6): a verb-less sentence
1136
- # listing many concepts is a list, not a set of relations. Verb-backed
1137
- # sentences keep every pair — the verb is the evidence.
1138
- if (
1139
- relation["evidence"] == "cooccurrence"
1140
- and len(present) > COOCCURRENCE_CONCEPT_LIMIT
1141
- ):
1142
- continue
1143
-
1144
- for i in range(len(present) - 1):
1145
- subj, obj = present[i], present[i + 1]
1146
- # Deduplicate by (subj, obj) regardless of direction for same edge
1147
- pair_key = tuple(sorted([subj.lower(), obj.lower()])) + (edge,)
1148
- if pair_key in seen_pairs:
1149
- continue
1150
- seen_pairs.add(pair_key)
1151
- triples.append(
1152
- {
1153
- "subject": subj,
1154
- "relation": edge, # verb form (동사)
1155
- "object": obj,
1156
- "context": sent[:240],
1157
- "evidence": relation["evidence"],
1158
- "weight": relation["weight"],
1159
- }
1160
- )
1161
- if len(triples) >= limit:
1162
- return triples
1163
-
1164
- return triples
1165
-
1166
-
1167
- def _semantic_items(text: str) -> List[Dict[str, str]]:
1168
- """Extract explicit decision / task items from text."""
1169
- items: List[Dict[str, str]] = []
1170
- for raw_line in str(text or "").splitlines():
1171
- line = _clean_text(raw_line)
1172
- if len(line) < 6:
1173
- continue
1174
- lowered = line.lower()
1175
- if re.search(r"(결정|확정|하기로|decided|decision)", lowered):
1176
- items.append(
1177
- {"type": "Decision", "title": line[:120], "summary": line[:500]}
1178
- )
1179
- if re.search(r"(todo|해야|하자|진행|구현|수정|확인|next|task|\[ \])", lowered):
1180
- items.append({"type": "Task", "title": line[:120], "summary": line[:500]})
1181
- return items[:8]
1182
-
1183
-
1184
- def _topic_candidates(text: str, limit: int = 8) -> List[str]:
1185
- """Return compact keyword candidates for fallback graph search."""
1186
- candidates = _extract_concepts(text, limit=limit)
1187
- if candidates:
1188
- return candidates[:limit]
1189
- seen: Dict[str, str] = {}
1190
- for token in re.findall(
1191
- r"[A-Za-z][A-Za-z0-9_.:-]{2,}|[가-힣]{2,12}", str(text or "")
1192
- ):
1193
- key = token.lower()
1194
- if key in _CONCEPT_STOP or key.isdigit():
1195
- continue
1196
- seen.setdefault(key, token)
1197
- if len(seen) >= limit:
1198
- break
1199
- return list(seen.values())[:limit]
1200
-
1201
-
1202
- # Static export list. This used to be `[name for name in globals() if not
1203
- # name.startswith("__")]`, which is invisible to a type checker: every
1204
- # `from ._kg_common import *` consumer then had *no* resolvable names, and
1205
- # mypy reported ~750 spurious `name-defined` errors across the graph
1206
- # package. `tests/unit/test_kg_common_exports.py` asserts this list still
1207
- # equals what the computed expression would produce, so it cannot drift.
1208
- __all__ = [
1209
- "Any",
1210
- "COMMON_EXCLUDED_DIRS",
1211
- "COMMON_EXCLUDED_FILE_NAMES",
1212
- "COMMON_EXCLUDED_FILE_SUFFIXES",
1213
- "COOCCURRENCE_CONCEPT_LIMIT",
1214
- "COOCCURRENCE_EDGE_WEIGHT",
1215
- "Counter",
1216
- "Dict",
1217
- "EDGE_VERB",
1218
- "ENABLE_LLM_EXTRACTION",
1219
- "EdgeType",
1220
- "GRAPH_SCHEMA_VERSION",
1221
- "Iterable",
1222
- "Iterator",
1223
- "KGStoreV2",
1224
- "LINUX_EXCLUDED_PREFIXES",
1225
- "LOCAL_CODE_EXTENSIONS",
1226
- "LOCAL_DOCUMENT_EXTENSIONS",
1227
- "LOCAL_IMAGE_EXTENSIONS",
1228
- "LOCAL_SIZE_LIMITS",
1229
- "LOCAL_SLIDE_EXTENSIONS",
1230
- "LOCAL_SPREADSHEET_EXTENSIONS",
1231
- "LOCAL_SUPPORTED_EXTENSIONS",
1232
- "LOCAL_TEXT_EXTENSIONS",
1233
- "List",
1234
- "LocalEmbeddingModel",
1235
- "MACOS_EXCLUDED_PREFIXES",
1236
- "NodeType",
1237
- "Optional",
1238
- "Path",
1239
- "SENSITIVE_PATH_KEYWORDS",
1240
- "Tuple",
1241
- "VERB_EDGE_WEIGHT",
1242
- "WINDOWS_EXCLUDED_NAMES",
1243
- "_CHUNK_STRATEGIES",
1244
- "_CODE_BLANK_RUN_RE",
1245
- "_CODE_BOUNDARY_LINE_RE",
1246
- "_CODE_CHUNK_EXTENSIONS",
1247
- "_CONCEPT_STOP",
1248
- "_KG_DB_FORMAT_KEY",
1249
- "_KG_DB_FORMAT_VERSION",
1250
- "_LLM_EXTRACT_CONCEPT_PROMPT",
1251
- "_LLM_EXTRACT_TRIPLE_PROMPT",
1252
- "_MARKDOWN_CHUNK_EXTENSIONS",
1253
- "_MARKDOWN_HEADING_RE",
1254
- "_MARKDOWN_MIN_SECTION_CHARS",
1255
- "_NOT_PERSON_WORDS",
1256
- "_PROJECTION_VERSION",
1257
- "_PROSE_CHUNK_EXTENSIONS",
1258
- "_PROSE_MIN_SPAN_RATIO",
1259
- "_PROSE_STRONG_BOUNDARY_RE",
1260
- "_PROSE_WEAK_BOUNDARY_RE",
1261
- "_READ_FROM_V2_DEFAULT",
1262
- "_V2_WRITE_MASTER_KEY",
1263
- "_chunks",
1264
- "_classify_node_type",
1265
- "_clean_text",
1266
- "_code_chunks",
1267
- "_code_segment_spans",
1268
- "_current_os_type",
1269
- "_drive_id_for_path",
1270
- "_excluded_directory_reason",
1271
- "_exec_script",
1272
- "_extract_concepts",
1273
- "_extract_concepts_rules",
1274
- "_extract_triples",
1275
- "_extract_triples_rules",
1276
- "_file_category",
1277
- "_infer_edge",
1278
- "_is_hidden_path",
1279
- "_is_relative_to",
1280
- "_json",
1281
- "_last_boundary",
1282
- "_llm_extract_concepts",
1283
- "_llm_extract_triples",
1284
- "_markdown_chunks",
1285
- "_markdown_section_spans",
1286
- "_merge_small_sections",
1287
- "_node_type_for_category",
1288
- "_now",
1289
- "_parse_iso",
1290
- "_parser_type_for_category",
1291
- "_path_fingerprint",
1292
- "_path_parts_lower",
1293
- "_plain_windows",
1294
- "_prose_chunks",
1295
- "_recency_score",
1296
- "_root_warning",
1297
- "_safe_iso_from_stat_mtime",
1298
- "_safe_loads",
1299
- "_sample_file",
1300
- "_semantic_items",
1301
- "_sensitive_file_reason",
1302
- "_sha256_bytes",
1303
- "_sha256_text",
1304
- "_size_limit_for_category",
1305
- "_slug",
1306
- "_topic_candidates",
1307
- "asyncio",
1308
- "chunk_strategy_for",
1309
- "citation_locator",
1310
- "contextmanager",
1311
- "datetime",
1312
- "get_llm_router",
1313
- "hashlib",
1314
- "infer_edge_relation",
1315
- "json",
1316
- "logging",
1317
- "math",
1318
- "os",
1319
- "page_for_offset",
1320
- "pdf_page_offsets",
1321
- "platform",
1322
- "quiet",
1323
- "re",
1324
- "set_llm_router",
1325
- "shutil",
1326
- "sqlite3",
1327
- "time",
1328
- "typed_chunk_meta_fields",
1329
- "typed_chunks",
1330
- "zipfile",
1331
- ]