ltcai 11.2.0 → 11.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (247) hide show
  1. package/README.md +46 -53
  2. package/docs/CHANGELOG.md +61 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/MULTI_AGENT_RUNTIME.md +1 -1
  6. package/docs/ONBOARDING.md +1 -1
  7. package/docs/OPERATIONS.md +6 -2
  8. package/docs/PERMISSION_MODE.md +1 -1
  9. package/docs/TRUST_MODEL.md +1 -1
  10. package/docs/WHY_LATTICE.md +1 -1
  11. package/docs/kg-schema.md +2 -2
  12. package/docs/v11.3.0_PLAN.md +202 -0
  13. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
  14. package/lattice_brain/__init__.py +1 -1
  15. package/lattice_brain/graph/_kg_common/__init__.py +287 -0
  16. package/lattice_brain/graph/_kg_common/extraction.py +516 -0
  17. package/lattice_brain/graph/_kg_common/relations.py +161 -0
  18. package/lattice_brain/graph/_kg_common/text.py +479 -0
  19. package/lattice_brain/graph/discovery_index/__init__.py +35 -0
  20. package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
  21. package/lattice_brain/graph/discovery_index/extract.py +137 -0
  22. package/lattice_brain/graph/discovery_index/scan.py +411 -0
  23. package/lattice_brain/graph/discovery_index/upsert.py +495 -0
  24. package/lattice_brain/graph/projection/__init__.py +42 -0
  25. package/lattice_brain/graph/projection/curation.py +500 -0
  26. package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
  27. package/lattice_brain/graph/retrieval/__init__.py +54 -0
  28. package/lattice_brain/graph/retrieval/context.py +197 -0
  29. package/lattice_brain/graph/retrieval/graph_view.py +319 -0
  30. package/lattice_brain/graph/retrieval/hybrid.py +488 -0
  31. package/lattice_brain/graph/retrieval/maintenance.py +121 -0
  32. package/lattice_brain/graph/retrieval/signals.py +95 -0
  33. package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
  34. package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
  35. package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
  36. package/lattice_brain/graph/retrieval_vector/search.py +560 -0
  37. package/lattice_brain/graph/retrieval_vector/status.py +374 -0
  38. package/lattice_brain/ingestion/__init__.py +130 -0
  39. package/lattice_brain/ingestion/_contract.py +90 -0
  40. package/lattice_brain/ingestion/constants.py +127 -0
  41. package/lattice_brain/ingestion/folder_scan.py +57 -0
  42. package/lattice_brain/ingestion/folders.py +258 -0
  43. package/lattice_brain/ingestion/hashing.py +26 -0
  44. package/lattice_brain/ingestion/jobs_api.py +107 -0
  45. package/lattice_brain/ingestion/models.py +80 -0
  46. package/lattice_brain/ingestion/pipeline.py +486 -0
  47. package/lattice_brain/ingestion/quality.py +209 -0
  48. package/lattice_brain/ingestion/routing.py +295 -0
  49. package/lattice_brain/multimodal/__init__.py +164 -0
  50. package/lattice_brain/multimodal/audio.py +77 -0
  51. package/lattice_brain/multimodal/common.py +118 -0
  52. package/lattice_brain/multimodal/images.py +498 -0
  53. package/lattice_brain/multimodal/ports.py +169 -0
  54. package/lattice_brain/multimodal/video.py +410 -0
  55. package/lattice_brain/portability/__init__.py +90 -0
  56. package/lattice_brain/portability/_contract.py +42 -0
  57. package/lattice_brain/portability/backups.py +338 -0
  58. package/lattice_brain/portability/bundles.py +136 -0
  59. package/lattice_brain/portability/constants.py +93 -0
  60. package/lattice_brain/portability/fsops.py +138 -0
  61. package/lattice_brain/portability/service.py +41 -0
  62. package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
  63. package/lattice_brain/runtime/__init__.py +1 -1
  64. package/lattice_brain/runtime/multi_agent.py +1 -1
  65. package/latticeai/__init__.py +1 -1
  66. package/latticeai/api/chronicle.py +63 -0
  67. package/latticeai/core/agent/__init__.py +93 -0
  68. package/latticeai/core/agent/_contract.py +79 -0
  69. package/latticeai/core/agent/context.py +57 -0
  70. package/latticeai/core/agent/deps.py +125 -0
  71. package/latticeai/core/agent/execution.py +622 -0
  72. package/latticeai/core/agent/planning.py +145 -0
  73. package/latticeai/core/agent/recovery.py +157 -0
  74. package/latticeai/core/agent/runtime.py +210 -0
  75. package/latticeai/core/agent/verification.py +231 -0
  76. package/latticeai/core/embedding_providers/__init__.py +151 -0
  77. package/latticeai/core/embedding_providers/base.py +199 -0
  78. package/latticeai/core/embedding_providers/captions.py +162 -0
  79. package/latticeai/core/embedding_providers/profiles.py +126 -0
  80. package/latticeai/core/embedding_providers/text.py +350 -0
  81. package/latticeai/core/embedding_providers/vision.py +352 -0
  82. package/latticeai/core/file_generation/__init__.py +115 -0
  83. package/latticeai/core/file_generation/bundles.py +76 -0
  84. package/latticeai/core/file_generation/extraction.py +154 -0
  85. package/latticeai/core/file_generation/inference.py +235 -0
  86. package/latticeai/core/file_generation/orchestration.py +152 -0
  87. package/latticeai/core/file_generation/prompting.py +117 -0
  88. package/latticeai/core/file_generation/repair.py +114 -0
  89. package/latticeai/core/file_generation/sanitize.py +61 -0
  90. package/latticeai/core/file_generation/validation.py +201 -0
  91. package/latticeai/core/legacy_compatibility.py +1 -1
  92. package/latticeai/core/marketplace.py +1 -1
  93. package/latticeai/core/messages.py +9 -0
  94. package/latticeai/core/workspace_os_constants.py +1 -1
  95. package/latticeai/integrations/telegram_bot/__init__.py +123 -0
  96. package/latticeai/integrations/telegram_bot/__main__.py +17 -0
  97. package/latticeai/integrations/telegram_bot/config.py +86 -0
  98. package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
  99. package/latticeai/integrations/telegram_bot/flows.py +478 -0
  100. package/latticeai/integrations/telegram_bot/helpers.py +322 -0
  101. package/latticeai/integrations/telegram_bot/screens.py +394 -0
  102. package/latticeai/models/router/__init__.py +88 -0
  103. package/latticeai/models/router/_contract.py +66 -0
  104. package/latticeai/models/router/branding.py +56 -0
  105. package/latticeai/models/router/catalog.py +69 -0
  106. package/latticeai/models/router/documents.py +199 -0
  107. package/latticeai/models/router/errors.py +37 -0
  108. package/latticeai/models/router/generation.py +258 -0
  109. package/latticeai/models/router/loading.py +291 -0
  110. package/latticeai/models/router/local_models.py +85 -0
  111. package/latticeai/models/router/registry.py +147 -0
  112. package/latticeai/runtime/build_phases/__init__.py +82 -0
  113. package/latticeai/runtime/build_phases/features.py +407 -0
  114. package/latticeai/runtime/build_phases/foundation.py +555 -0
  115. package/latticeai/runtime/build_phases/web.py +492 -0
  116. package/latticeai/runtime/runtime_context.py +1 -0
  117. package/latticeai/services/architecture_readiness.py +48 -19
  118. package/latticeai/services/brain_intelligence/__init__.py +58 -0
  119. package/latticeai/services/brain_intelligence/_contract.py +71 -0
  120. package/latticeai/services/brain_intelligence/consistency.py +193 -0
  121. package/latticeai/services/brain_intelligence/constants.py +47 -0
  122. package/latticeai/services/brain_intelligence/digest.py +258 -0
  123. package/latticeai/services/brain_intelligence/health.py +331 -0
  124. package/latticeai/services/brain_intelligence/proposals.py +264 -0
  125. package/latticeai/services/brain_intelligence/sampling.py +84 -0
  126. package/latticeai/services/brain_intelligence/service.py +48 -0
  127. package/latticeai/services/chronicle.py +557 -0
  128. package/latticeai/services/memory_service/__init__.py +52 -0
  129. package/latticeai/services/memory_service/_contract.py +100 -0
  130. package/latticeai/services/memory_service/brief.py +431 -0
  131. package/latticeai/services/memory_service/constants.py +57 -0
  132. package/latticeai/services/memory_service/maintenance.py +138 -0
  133. package/latticeai/services/memory_service/manager.py +186 -0
  134. package/latticeai/services/memory_service/proof.py +136 -0
  135. package/latticeai/services/memory_service/recall.py +225 -0
  136. package/latticeai/services/memory_service/service.py +48 -0
  137. package/latticeai/services/memory_service/stores.py +110 -0
  138. package/latticeai/services/model_runtime/__init__.py +322 -0
  139. package/latticeai/services/model_runtime/cloud.py +87 -0
  140. package/latticeai/services/model_runtime/download.py +282 -0
  141. package/latticeai/services/model_runtime/engines.py +341 -0
  142. package/latticeai/services/model_runtime/loading.py +178 -0
  143. package/latticeai/services/model_runtime/service.py +129 -0
  144. package/latticeai/services/model_runtime/state.py +131 -0
  145. package/latticeai/services/model_runtime/status.py +255 -0
  146. package/latticeai/services/product_readiness.py +15 -7
  147. package/latticeai/setup/wizard/__init__.py +126 -0
  148. package/latticeai/setup/wizard/catalog.py +172 -0
  149. package/latticeai/setup/wizard/detect.py +323 -0
  150. package/latticeai/setup/wizard/install.py +348 -0
  151. package/latticeai/setup/wizard/paths.py +168 -0
  152. package/latticeai/setup/wizard/plans.py +74 -0
  153. package/latticeai/setup/wizard/recommend.py +320 -0
  154. package/package.json +6 -2
  155. package/scripts/bump_version.py +14 -0
  156. package/scripts/capture_release_evidence.mjs +33 -21
  157. package/scripts/check_current_release_docs.mjs +1 -1
  158. package/scripts/check_i18n_namespace_coverage.mjs +41 -4
  159. package/scripts/check_max_file_lines.mjs +102 -0
  160. package/scripts/check_release_evidence_bound.mjs +30 -15
  161. package/scripts/check_screenshot_pixel_delta.py +34 -4
  162. package/scripts/check_server_i18n.mjs +1 -0
  163. package/scripts/generate_rust_parity_fixtures.py +562 -0
  164. package/scripts/lib/mock_server_fingerprint.mjs +94 -0
  165. package/scripts/release_screen_claims.json +31 -2
  166. package/src-tauri/Cargo.lock +361 -3
  167. package/src-tauri/Cargo.toml +6 -1
  168. package/src-tauri/src/backend.rs +349 -0
  169. package/src-tauri/src/folder.rs +33 -0
  170. package/src-tauri/src/main.rs +97 -399
  171. package/src-tauri/tauri.conf.json +1 -1
  172. package/static/app/asset-manifest.json +41 -37
  173. package/static/app/assets/Act-yYpYnn0v.js +1 -0
  174. package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
  175. package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
  176. package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
  177. package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
  178. package/static/app/assets/Capture-CFIRsFNE.js +1 -0
  179. package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
  180. package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
  181. package/static/app/assets/Library-DwO3yZST.js +1 -0
  182. package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
  183. package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
  184. package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
  185. package/static/app/assets/System-DW8F-2xL.js +1 -0
  186. package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
  187. package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
  188. package/static/app/assets/brain-Ci1CkWjM.js +1 -0
  189. package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
  190. package/static/app/assets/circle-check-DfInj-qD.js +1 -0
  191. package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
  192. package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
  193. package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
  194. package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
  195. package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
  196. package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
  197. package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
  198. package/static/app/assets/index-_u5iUHDr.js +10 -0
  199. package/static/app/assets/input-B0lPdRQZ.js +1 -0
  200. package/static/app/assets/link-2-CoFbooHS.js +1 -0
  201. package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
  202. package/static/app/assets/primitives-DEbN-d6p.js +1 -0
  203. package/static/app/assets/search-BybIWPNd.js +1 -0
  204. package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
  205. package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
  206. package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
  207. package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
  208. package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
  209. package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
  210. package/static/app/assets/utils-BlZr7Pd4.js +4 -0
  211. package/static/app/assets/workspace-jJY4RuAV.js +1 -0
  212. package/static/app/index.html +4 -4
  213. package/static/sw.js +1 -1
  214. package/lattice_brain/graph/_kg_common.py +0 -1331
  215. package/lattice_brain/graph/discovery_index.py +0 -1141
  216. package/lattice_brain/graph/retrieval.py +0 -1120
  217. package/lattice_brain/graph/retrieval_vector.py +0 -1293
  218. package/lattice_brain/ingestion.py +0 -1525
  219. package/lattice_brain/multimodal.py +0 -1258
  220. package/latticeai/core/agent.py +0 -1465
  221. package/latticeai/core/embedding_providers.py +0 -1196
  222. package/latticeai/core/file_generation.py +0 -1047
  223. package/latticeai/integrations/telegram_bot.py +0 -1390
  224. package/latticeai/models/router.py +0 -1007
  225. package/latticeai/runtime/build_phases.py +0 -1450
  226. package/latticeai/services/brain_intelligence.py +0 -1083
  227. package/latticeai/services/memory_service.py +0 -1177
  228. package/latticeai/services/model_runtime.py +0 -1281
  229. package/latticeai/setup/wizard.py +0 -1310
  230. package/static/app/assets/Act-AWf0SAKp.js +0 -1
  231. package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
  232. package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
  233. package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
  234. package/static/app/assets/Capture-CqOSzyPr.js +0 -1
  235. package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
  236. package/static/app/assets/Library-CX-bbhmK.js +0 -1
  237. package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
  238. package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
  239. package/static/app/assets/System-Bu2t5hn1.js +0 -1
  240. package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
  241. package/static/app/assets/brain-DJMoqrwx.js +0 -1
  242. package/static/app/assets/index-BpYkzcVm.js +0 -10
  243. package/static/app/assets/input-DSlJJxRs.js +0 -1
  244. package/static/app/assets/primitives-BCx6TvfG.js +0 -1
  245. package/static/app/assets/search-Cgy8cCFJ.js +0 -1
  246. package/static/app/assets/utils-zqPZJxdx.js +0 -4
  247. package/static/app/assets/workspace-DXTihhfU.js +0 -1
@@ -0,0 +1,209 @@
1
+ """Advisory extraction-quality scoring, and the capture CTA built on it.
2
+
3
+ Pure heuristics over already-extracted text — no model call, no network, and
4
+ deterministic. The score never blocks an ingest; it annotates the result so a
5
+ capture surface can say "this capture is thin" and offer a way to fix it
6
+ instead of silently storing junk.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from typing import Any, Dict, List, Optional
12
+
13
+ # ── Extraction quality heuristics (v9.8.0 A1) ────────────────────────────────
14
+ # Pure heuristics over the extracted text — no model calls, no network. The
15
+ # score is *advisory*: it never blocks an ingest, it only annotates the result
16
+ # so capture surfaces (browser, folder scan) can surface low-quality warnings.
17
+ QUALITY_HIGH_THRESHOLD = 0.7
18
+ QUALITY_LOW_THRESHOLD = 0.4
19
+ QUALITY_LOW_WARNING = "추출 품질이 낮습니다 — 원문 확인을 권장합니다."
20
+ _WEB_SOURCE_TYPES = frozenset({"web_url", "browser_tab"})
21
+ # Standalone short lines that smell like leftover site chrome (nav/menu/footer).
22
+ _BOILERPLATE_LINE_MARKERS = frozenset(
23
+ {
24
+ "home", "menu", "nav", "navigation", "login", "log in", "sign in",
25
+ "sign up", "register", "subscribe", "search", "about", "about us",
26
+ "contact", "contact us", "privacy policy", "terms of service",
27
+ "cookie policy", "accept cookies", "accept all cookies", "share",
28
+ "skip to content", "copyright", "all rights reserved", "sitemap",
29
+ "back to top", "footer", "read more", "next", "previous",
30
+ }
31
+ )
32
+
33
+
34
+ def _quality_level(score: float) -> str:
35
+ if score >= QUALITY_HIGH_THRESHOLD:
36
+ return "high"
37
+ if score >= QUALITY_LOW_THRESHOLD:
38
+ return "medium"
39
+ return "low"
40
+
41
+
42
+ def assess_extraction_quality(
43
+ text: Optional[str],
44
+ *,
45
+ source_type: Optional[str] = None,
46
+ upstream_confidence: Optional[Any] = None,
47
+ ) -> Dict[str, Any]:
48
+ """Score extracted text 0..1 with reasons (pure heuristic, deterministic).
49
+
50
+ Signals: text length, whitespace ratio, character/word diversity
51
+ (repetition), sentence structure, and — for web sources — leftover
52
+ nav/menu boilerplate. When the upstream extractor supplies its own
53
+ confidence (``upstream_confidence``), that value wins verbatim: the
54
+ extractor saw the raw document, this function only sees its output.
55
+ """
56
+ if upstream_confidence is not None:
57
+ try:
58
+ score = max(0.0, min(1.0, float(upstream_confidence)))
59
+ except (TypeError, ValueError):
60
+ score = None
61
+ if score is not None:
62
+ return {
63
+ "score": round(score, 4),
64
+ "level": _quality_level(score),
65
+ "reasons": ["upstream_confidence"],
66
+ }
67
+
68
+ raw = str(text or "")
69
+ stripped = raw.strip()
70
+ if not stripped:
71
+ return {"score": 0.0, "level": "low", "reasons": ["empty_text"]}
72
+
73
+ reasons: List[str] = []
74
+ length = len(stripped)
75
+ sample = stripped[:4000]
76
+ lines = [ln.strip() for ln in stripped.splitlines() if ln.strip()]
77
+ words = stripped.split()
78
+
79
+ # 1) Length — very short extractions rarely carry recall value.
80
+ if length < 40:
81
+ length_factor = 0.35
82
+ reasons.append("very_short_text")
83
+ elif length < 120:
84
+ length_factor = 0.6
85
+ reasons.append("short_text")
86
+ elif length < 300:
87
+ length_factor = 0.85
88
+ else:
89
+ length_factor = 1.0
90
+
91
+ # 2) Sentence structure — prose has sentence-ending punctuation.
92
+ sentence_marks = sum(sample.count(mark) for mark in (".", "!", "?", "…", "。", "!", "?"))
93
+ if sentence_marks > 0:
94
+ structure_factor = 1.0
95
+ elif length < 200:
96
+ structure_factor = 0.75 # titles/snippets legitimately lack periods
97
+ else:
98
+ structure_factor = 0.45
99
+ reasons.append("no_sentence_structure")
100
+
101
+ # 3) Diversity — repeated characters/lines/words indicate extraction junk.
102
+ diversity_factor = 1.0
103
+ distinct_chars = len(set(sample.lower()))
104
+ if distinct_chars < 10:
105
+ diversity_factor *= 0.2
106
+ reasons.append("low_character_diversity")
107
+ elif distinct_chars < 20:
108
+ diversity_factor *= 0.7
109
+ if len(lines) >= 6:
110
+ top_count = max(lines.count(ln) for ln in set(lines))
111
+ if top_count >= max(3, len(lines) // 4):
112
+ diversity_factor *= 0.5
113
+ reasons.append("repetitive_lines")
114
+ if len(words) >= 30 and (len(set(w.lower() for w in words)) / len(words)) < 0.25:
115
+ diversity_factor *= 0.5
116
+ reasons.append("repetitive_words")
117
+
118
+ # 4) Cleanliness — whitespace floods, fragmented lines, site chrome.
119
+ cleanliness_factor = 1.0
120
+ whitespace_ratio = sum(1 for ch in raw if ch.isspace()) / max(1, len(raw))
121
+ if whitespace_ratio > 0.45:
122
+ cleanliness_factor *= 0.6
123
+ reasons.append("high_whitespace_ratio")
124
+ if len(lines) >= 8:
125
+ short_lines = sum(1 for ln in lines if len(ln.split()) <= 3)
126
+ if short_lines / len(lines) > 0.6:
127
+ cleanliness_factor *= 0.6
128
+ reasons.append("fragmented_lines")
129
+ boilerplate_hits = sum(
130
+ 1 for ln in lines if ln.lower().strip(" .:>|•·-–—*") in _BOILERPLATE_LINE_MARKERS
131
+ )
132
+ if lines and boilerplate_hits >= 3 and (boilerplate_hits / len(lines)) > 0.2:
133
+ cleanliness_factor *= 0.35
134
+ if str(source_type or "").lower() in _WEB_SOURCE_TYPES:
135
+ reasons.append("nav_menu_remnants")
136
+ else:
137
+ reasons.append("boilerplate_markers")
138
+
139
+ score = length_factor * structure_factor * diversity_factor * cleanliness_factor
140
+ score = max(0.0, min(1.0, score))
141
+ if not reasons:
142
+ reasons.append("clean_extraction")
143
+ return {"score": round(score, 4), "level": _quality_level(score), "reasons": reasons}
144
+
145
+
146
+ # ── capture quality CTA (backlog #9, review §7.2 C) ──────────────────────────
147
+ # Structured verdict over the same extraction-quality schema the rest of the
148
+ # pipeline uses, so capture surfaces (browser extension, read-url) can render
149
+ # an honest "this capture is thin" CTA instead of silently storing junk.
150
+ CAPTURE_SUGGESTIONS_THIN = ["recapture", "paste_manually", "highlight_source"]
151
+ _CAPTURE_REASON_LABELS = {
152
+ "empty_text": "추출된 본문이 비어 있습니다",
153
+ "very_short_text": "추출된 본문이 매우 짧습니다",
154
+ "short_text": "추출된 본문이 짧습니다",
155
+ "no_sentence_structure": "문장 구조가 거의 없습니다",
156
+ "low_character_diversity": "반복 문자가 대부분입니다",
157
+ "repetitive_lines": "같은 줄이 반복됩니다",
158
+ "repetitive_words": "같은 단어가 반복됩니다",
159
+ "high_whitespace_ratio": "공백이 지나치게 많습니다",
160
+ "fragmented_lines": "줄이 잘게 조각나 있습니다",
161
+ "nav_menu_remnants": "메뉴/내비게이션 잔여물이 많습니다",
162
+ "boilerplate_markers": "상용구 텍스트가 많습니다",
163
+ "no_extracted_text": "추출된 텍스트가 없습니다",
164
+ }
165
+
166
+
167
+ def capture_quality_verdict(
168
+ extraction_quality: Optional[Dict[str, Any]],
169
+ *,
170
+ source_type: Optional[str] = None,
171
+ ) -> Dict[str, Any]:
172
+ """Structured CTA verdict from a pipeline ``extraction_quality`` dict.
173
+
174
+ ``{"status": "thin"|"ok", "reason": str|None, "suggestions": [...],
175
+ "score": float|None, "level": str|None}``. ``thin`` (level == "low", the
176
+ same threshold as the ingest warning) carries actionable suggestions —
177
+ ``recapture`` / ``paste_manually`` / ``highlight_source`` — so the UI can
178
+ offer the user a way to fix the capture instead of hiding the problem.
179
+ Deterministic and never raises; ``None`` input yields an honest ``thin``.
180
+ """
181
+ if not isinstance(extraction_quality, dict):
182
+ return {
183
+ "status": "thin",
184
+ "reason": _CAPTURE_REASON_LABELS["no_extracted_text"],
185
+ "reason_codes": ["no_extracted_text"],
186
+ "suggestions": list(CAPTURE_SUGGESTIONS_THIN),
187
+ "score": None,
188
+ "level": None,
189
+ }
190
+ level = str(extraction_quality.get("level") or "")
191
+ score = extraction_quality.get("score")
192
+ reasons = [str(item) for item in (extraction_quality.get("reasons") or [])]
193
+ thin = level == "low"
194
+ reason = None
195
+ if thin:
196
+ labeled = [
197
+ _CAPTURE_REASON_LABELS[code]
198
+ for code in reasons
199
+ if code in _CAPTURE_REASON_LABELS
200
+ ]
201
+ reason = "; ".join(labeled) if labeled else QUALITY_LOW_WARNING
202
+ return {
203
+ "status": "thin" if thin else "ok",
204
+ "reason": reason,
205
+ "reason_codes": reasons if thin else [],
206
+ "suggestions": list(CAPTURE_SUGGESTIONS_THIN) if thin else [],
207
+ "score": score,
208
+ "level": level or None,
209
+ }
@@ -0,0 +1,295 @@
1
+ """One door per kind of thing: text, chat, memory record, picture, film, file.
2
+
3
+ Every method here returns the raw store payload that ``IngestionPipeline.\
4
+ ingest`` normalizes; none of them decide *whether* to run. Modality routing
5
+ (:meth:`IngestionRoutingMixin._modality_for`) answers ``"text"`` for everything
6
+ while multi-modal is off, which is what makes "off" mean *unchanged* rather than
7
+ *slightly different*.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from pathlib import Path
13
+ from typing import Any, Dict
14
+
15
+ from ..multimodal import (
16
+ MODALITY_AUDIO,
17
+ MODALITY_IMAGE,
18
+ MODALITY_VIDEO,
19
+ ImageFacts,
20
+ audio_quality_score,
21
+ detect_modality,
22
+ extract_image_facts,
23
+ image_quality_score,
24
+ read_video_facts,
25
+ transcribe_audio,
26
+ video_frame_dir,
27
+ video_quality_score,
28
+ write_image_memory,
29
+ write_video_memory,
30
+ )
31
+ from ..utils import utc_now_iso
32
+ from ._contract import IngestionCore as _Core
33
+ from .constants import (
34
+ _MEMORY_NODE_TYPES,
35
+ AUDIO_NODE_TYPE,
36
+ AUDIO_SOURCE_TYPES,
37
+ IMAGE_SOURCE_TYPES,
38
+ VIDEO_SOURCE_TYPES,
39
+ )
40
+ from .hashing import _file_digest
41
+ from .models import IngestionItem
42
+ from .quality import _quality_level
43
+
44
+
45
+ class IngestionRoutingMixin(_Core):
46
+ """The per-source-type ingest doors. Mixed into ``IngestionPipeline``."""
47
+
48
+ # ── routing helpers ──────────────────────────────────────────────────────
49
+ def _ingest_text(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
50
+ text = item.text or ""
51
+ if not text.strip():
52
+ raise ValueError(
53
+ f"Empty content: {source_type} ingestion requires non-empty text."
54
+ )
55
+ if len(text.encode("utf-8", "ignore")) > self._max_text_bytes:
56
+ raise ValueError(
57
+ f"Text payload exceeds the {self._max_text_bytes // (1024 * 1024)}MB ingestion limit."
58
+ )
59
+ title = item.title or item.source_uri or source_type
60
+ return self._kg.ingest_source(
61
+ source_type=source_type,
62
+ title=title,
63
+ text=text,
64
+ source_uri=item.source_uri,
65
+ owner=owner,
66
+ workspace_id=item.workspace_id,
67
+ permissions=item.permissions,
68
+ captured_at=captured_at,
69
+ modified_at=item.modified_at,
70
+ conversation_id=item.conversation_id,
71
+ metadata={"mime_type": item.mime_type, **(item.metadata or {})},
72
+ )
73
+
74
+ def _ingest_chat(self, item, *, source_type, owner) -> Dict[str, Any]:
75
+ text = item.text or ""
76
+ meta = item.metadata or {}
77
+ role = str(meta.get("role") or "user")
78
+ result = self._kg.ingest_message(
79
+ role,
80
+ text,
81
+ user_email=owner,
82
+ user_nickname=meta.get("user_nickname"),
83
+ source=meta.get("source") or source_type,
84
+ conversation_id=item.conversation_id,
85
+ workspace_id=item.workspace_id,
86
+ raw=meta.get("raw"),
87
+ )
88
+ # ingest_message reports message/response node ids; normalize the keys
89
+ # the provenance step expects.
90
+ result.setdefault("node_id", result.get("node_id") or result.get("message_node_id") or result.get("id"))
91
+ result.setdefault("title", item.title or text[:80])
92
+ return result
93
+
94
+ def _ingest_memory_record(self, item, *, source_type, owner) -> Dict[str, Any]:
95
+ node_type = _MEMORY_NODE_TYPES[source_type]
96
+ meta = item.metadata or {}
97
+ result = self._kg.ingest_event(
98
+ node_type,
99
+ item.title or (item.text or node_type)[:120],
100
+ user_email=owner,
101
+ source=meta.get("source") or source_type,
102
+ conversation_id=item.conversation_id,
103
+ workspace_id=item.workspace_id,
104
+ metadata={**meta, "detail": (item.text or "")[:2000]},
105
+ )
106
+ result.setdefault("node_id", result.get("node_id") or result.get("id"))
107
+ result.setdefault("title", item.title)
108
+ return result
109
+
110
+ # ── multi-modal routing (v11.1.0 Track 3) ────────────────────────────────
111
+ def _modality_for(self, item: IngestionItem, source_type: str) -> str:
112
+ """``image`` / ``audio`` / ``video`` / ``text`` for this item.
113
+
114
+ Always ``"text"`` while the flag is off, which is what makes "off" mean
115
+ *unchanged* rather than *slightly different*.
116
+ """
117
+ if not self._allow_multimodal:
118
+ return "text"
119
+ if source_type in IMAGE_SOURCE_TYPES:
120
+ return MODALITY_IMAGE
121
+ if source_type in AUDIO_SOURCE_TYPES:
122
+ return MODALITY_AUDIO
123
+ if source_type in VIDEO_SOURCE_TYPES:
124
+ return MODALITY_VIDEO
125
+ if not item.path:
126
+ return "text"
127
+ return detect_modality(item.path, item.mime_type)
128
+
129
+ def _resolve_file_path(self, item: IngestionItem) -> Path:
130
+ if not item.path:
131
+ raise ValueError("File ingestion requires a path.")
132
+ path = Path(item.path)
133
+ if not path.exists():
134
+ raise FileNotFoundError(f"File not found: {path}")
135
+ if path.is_dir():
136
+ raise ValueError(f"File ingestion requires a file, got a directory: {path}")
137
+ return path
138
+
139
+ def _ingest_image(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
140
+ """Store one picture as an ``Image`` node — OCR, caption, vector.
141
+
142
+ The image vector (when a vision model produced one) goes to its own
143
+ index; the OCR/caption text rides the ordinary text index. That split
144
+ is what lets a typed question find a screenshot without ever comparing
145
+ a text vector to an image vector.
146
+ """
147
+ path = self._resolve_file_path(item)
148
+ facts = extract_image_facts(str(path), ports=self._multimodal)
149
+ result = write_image_memory(
150
+ self._kg,
151
+ path=path,
152
+ facts=facts,
153
+ title=item.title or path.name,
154
+ source_type=source_type if source_type in IMAGE_SOURCE_TYPES else MODALITY_IMAGE,
155
+ source_uri=item.source_uri,
156
+ owner=owner,
157
+ workspace_id=item.workspace_id,
158
+ conversation_id=item.conversation_id,
159
+ captured_at=captured_at,
160
+ modified_at=item.modified_at,
161
+ permissions=item.permissions,
162
+ extra_metadata={"mime_type": item.mime_type, **(item.metadata or {})},
163
+ )
164
+ self._record_image_vector(result["node_id"], facts)
165
+ quality = image_quality_score(facts)
166
+ result["extraction_quality"] = {
167
+ "score": quality["score"],
168
+ "level": _quality_level(quality["score"]),
169
+ "reasons": quality["reasons"],
170
+ }
171
+ return result
172
+
173
+ def _record_image_vector(self, node_id: str, facts: ImageFacts) -> None:
174
+ """File the image-space vector, if a vision model actually made one."""
175
+ if facts.embedding is None:
176
+ return
177
+ from ..graph.image_vectors import record_image_vector
178
+
179
+ record_image_vector(
180
+ self._kg,
181
+ node_id=node_id,
182
+ vector=facts.embedding,
183
+ model_id=self._multimodal.vision_model_id or "vision:unnamed",
184
+ space=self._multimodal.vision_space,
185
+ updated_at=utc_now_iso(),
186
+ )
187
+
188
+ def _ingest_audio(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
189
+ """Store one recording as an ``Audio`` node, transcribed when possible.
190
+
191
+ The transcript is text and rides the ordinary text index — chunks,
192
+ concepts, provenance, dedupe all unchanged — but the node itself is a
193
+ recording, because that is what it is whether or not anyone could hear
194
+ it. The recording's own facts stay in the metadata (``modality``,
195
+ ``audio_path``, ``transcription``, ``searchable``). Without a
196
+ transcriber the memory is still kept, and its body says plainly that
197
+ the words were never recognized instead of leaving a blank note.
198
+ """
199
+ path = self._resolve_file_path(item)
200
+ facts = transcribe_audio(str(path), ports=self._multimodal, transcript=item.text)
201
+ title = item.title or path.stem
202
+ body = facts.transcript or (
203
+ f"[{MODALITY_AUDIO}] {title}\n"
204
+ "이 녹음은 아직 글로 바뀌지 않았습니다 — 음성 인식기가 없어 내용 검색은 되지 않습니다."
205
+ )
206
+ result = self._kg.ingest_source(
207
+ source_type=source_type,
208
+ title=title,
209
+ text=body,
210
+ source_uri=item.source_uri or str(path),
211
+ owner=owner,
212
+ workspace_id=item.workspace_id,
213
+ permissions=item.permissions,
214
+ captured_at=captured_at,
215
+ modified_at=item.modified_at,
216
+ conversation_id=item.conversation_id,
217
+ node_type=AUDIO_NODE_TYPE,
218
+ metadata={
219
+ "mime_type": item.mime_type,
220
+ "modality": MODALITY_AUDIO,
221
+ "audio_path": str(path),
222
+ "audio_bytes": path.stat().st_size,
223
+ "transcription": facts.transcription_status,
224
+ "searchable": facts.searchable,
225
+ **({"transcription_detail": facts.detail} if facts.detail else {}),
226
+ **(item.metadata or {}),
227
+ },
228
+ )
229
+ result.setdefault("title", title)
230
+ quality = audio_quality_score(facts)
231
+ result["extraction_quality"] = {
232
+ "score": quality["score"],
233
+ "level": _quality_level(quality["score"]),
234
+ "reasons": quality["reasons"],
235
+ }
236
+ return result
237
+
238
+ def _ingest_video(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
239
+ """Store one video as keyframes through the image door plus subtitles.
240
+
241
+ Nothing here is a new retrieval path: the stills become ordinary
242
+ ``Image`` nodes (OCR, caption, vector, thumbnail) joined by
243
+ ``CONTAINS_IMAGE``, and the subtitle text becomes ordinary chunks. What
244
+ the ``Video`` node adds is the thing they belong to — and an honest
245
+ body when there were no subtitles to read.
246
+ """
247
+ path = self._resolve_file_path(item)
248
+ facts = read_video_facts(
249
+ str(path),
250
+ video_frame_dir(getattr(self._kg, "blob_dir", path.parent), _file_digest(path)),
251
+ count=self._keyframes,
252
+ ports=self._multimodal,
253
+ subtitle_text=item.text,
254
+ )
255
+ result = write_video_memory(
256
+ self._kg,
257
+ path=path,
258
+ facts=facts,
259
+ title=item.title or path.stem,
260
+ source_type=source_type if source_type in VIDEO_SOURCE_TYPES else MODALITY_VIDEO,
261
+ source_uri=item.source_uri,
262
+ owner=owner,
263
+ workspace_id=item.workspace_id,
264
+ conversation_id=item.conversation_id,
265
+ captured_at=captured_at,
266
+ modified_at=item.modified_at,
267
+ permissions=item.permissions,
268
+ extra_metadata={"mime_type": item.mime_type, **(item.metadata or {})},
269
+ ports=self._multimodal,
270
+ )
271
+ quality = video_quality_score(facts)
272
+ result["extraction_quality"] = {
273
+ "score": quality["score"],
274
+ "level": _quality_level(quality["score"]),
275
+ "reasons": quality["reasons"],
276
+ }
277
+ return result
278
+
279
+ def _ingest_file(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
280
+ path = self._resolve_file_path(item)
281
+ return self._kg.ingest_document(
282
+ path,
283
+ original_filename=item.title or path.name,
284
+ mime_type=item.mime_type,
285
+ uploader=owner,
286
+ conversation_id=item.conversation_id,
287
+ extracted=item.metadata.get("extracted") if item.metadata else None,
288
+ source_type=source_type,
289
+ source_uri=item.source_uri or str(path),
290
+ captured_at=captured_at,
291
+ modified_at=item.modified_at,
292
+ owner=owner,
293
+ workspace_id=item.workspace_id,
294
+ permissions=item.permissions,
295
+ )
@@ -0,0 +1,164 @@
1
+ """Images and audio as first-class memories (v11.1.0, Track 3).
2
+
3
+ Before this module the Brain could only remember things that arrived as text.
4
+ A screenshot of a whiteboard, a photo of a receipt, a voice memo — all of them
5
+ either bounced off the ingestion pipeline or landed as an opaque ``Document``
6
+ node whose only searchable content was its filename.
7
+
8
+ What routes here
9
+ ----------------
10
+ :func:`detect_modality` reads the MIME type first and the extension second, and
11
+ answers with one of ``text`` / ``image`` / ``audio`` / ``video``. ``video`` is
12
+ deliberately a *recognized but unsupported* answer in this release: keyframe
13
+ extraction needs a decoder this project does not ship, and returning "video,
14
+ out of scope" is worth more than pretending a ``.mov`` is a picture.
15
+
16
+ What an image memory contains
17
+ -----------------------------
18
+ :func:`extract_image_facts` gathers only what it can actually observe:
19
+
20
+ * **dimensions/format** from Pillow (a core dependency);
21
+ * **ocr_text** from ``pytesseract`` when it is installed — otherwise
22
+ ``ocr_status="unavailable"`` and no text, never an empty string dressed up as
23
+ a successful read;
24
+ * **caption** from an injected vision-language port, and *only* from there. No
25
+ VLM means ``caption is None``. Composing "Image IMG_2381.png (JPEG 3024x4032)"
26
+ and storing it in the caption field would make metadata indistinguishable
27
+ from a model's description forever after;
28
+ * **embedding** from an injected vision port, which lives in its own vector
29
+ space (see :mod:`latticeai.core.embedding_providers`) and therefore its own
30
+ index — text queries reach images through OCR/caption text, not by scoring a
31
+ BGE vector against CLIP vectors.
32
+
33
+ Brain Core owns none of those models. Every heavy dependency arrives as an
34
+ injected callable (:class:`MultimodalPorts`), which is also why this module
35
+ imports nothing from ``latticeai``.
36
+
37
+ Split into cohesive submodules in v11.3.0 (no behaviour change): ``common``
38
+ (taxonomy + shared helpers), ``ports`` (injected capabilities + the ffmpeg
39
+ fallback), ``images``, ``audio``, ``video``. This module re-exports every name
40
+ the single file exposed, so ``lattice_brain.multimodal.X`` keeps working.
41
+ """
42
+
43
+ from __future__ import annotations
44
+
45
+ # Internals that predate the split. They are not public API, but they were
46
+ # reachable as ``lattice_brain.multimodal.<name>`` before it and callers (and
47
+ # tests) still reach them that way, so they are re-exported explicitly. The
48
+ # redundant-alias form says "this is a re-export", not a leftover import.
49
+ #
50
+ # Stubbing note: replacing one of these *here* rebinds only this module's name.
51
+ # The submodule that calls it holds its own reference, so a test that wants to
52
+ # stand in for ``_which_ffmpeg`` patches ``lattice_brain.multimodal.ports``.
53
+ from ..quiet import quiet as quiet
54
+ from ..utils import utc_now_iso as utc_now_iso
55
+ from .audio import AudioFacts, audio_quality_score, transcribe_audio
56
+ from .common import (
57
+ AUDIO_EXTENSIONS,
58
+ IMAGE_CHUNK_CHARS,
59
+ IMAGE_EXTENSIONS,
60
+ MAX_INDEX_TEXT_CHARS,
61
+ MAX_THUMBNAIL_CHARS,
62
+ MODALITY_AUDIO,
63
+ MODALITY_IMAGE,
64
+ MODALITY_TEXT,
65
+ MODALITY_VIDEO,
66
+ SUBTITLE_EXTENSIONS,
67
+ SUMMARY_CHARS,
68
+ THUMBNAIL_EDGE,
69
+ VIDEO_EXTENSIONS,
70
+ VIDEO_OUT_OF_SCOPE,
71
+ VIDEO_UNAVAILABLE_DETAIL,
72
+ detect_modality,
73
+ )
74
+ from .common import _sha256_file as _sha256_file
75
+ from .common import _sha256_text as _sha256_text
76
+ from .common import _split_index_text as _split_index_text
77
+ from .images import (
78
+ ImageFacts,
79
+ extract_image_facts,
80
+ image_node_id,
81
+ image_quality_score,
82
+ write_image_memory,
83
+ )
84
+ from .images import _apply_vision_embedding as _apply_vision_embedding
85
+ from .images import _attach_concepts as _attach_concepts
86
+ from .images import _attach_source as _attach_source
87
+ from .images import _open_image as _open_image
88
+ from .images import _run_ocr as _run_ocr
89
+ from .images import _safe_caption as _safe_caption
90
+ from .images import _thumbnail_data_uri as _thumbnail_data_uri
91
+ from .ports import (
92
+ DEFAULT_KEYFRAMES,
93
+ FFMPEG_BINARY,
94
+ MultimodalPorts,
95
+ extract_keyframes,
96
+ ffmpeg_available,
97
+ )
98
+ from .ports import KEYFRAME_TIMEOUT_SECONDS as KEYFRAME_TIMEOUT_SECONDS
99
+ from .ports import KEYFRAME_WINDOW as KEYFRAME_WINDOW
100
+ from .ports import _injected_keyframes as _injected_keyframes
101
+ from .ports import _run_ffmpeg as _run_ffmpeg
102
+ from .ports import _which_ffmpeg as _which_ffmpeg
103
+ from .video import _CUE_TAG_RE as _CUE_TAG_RE
104
+ from .video import _SRT_INDEX_RE as _SRT_INDEX_RE
105
+ from .video import _TIMECODE_RE as _TIMECODE_RE
106
+ from .video import (
107
+ MAX_SUBTITLE_CHARS,
108
+ VIDEO_FRAME_RELATION,
109
+ VIDEO_FRAME_SOURCE_TYPE,
110
+ VIDEO_NODE_TYPE,
111
+ VideoFacts,
112
+ find_subtitle,
113
+ parse_subtitles,
114
+ read_video_facts,
115
+ video_frame_dir,
116
+ video_node_id,
117
+ video_quality_score,
118
+ write_video_memory,
119
+ )
120
+ from .video import _write_keyframes as _write_keyframes
121
+
122
+ __all__ = [
123
+ "AUDIO_EXTENSIONS",
124
+ "DEFAULT_KEYFRAMES",
125
+ "FFMPEG_BINARY",
126
+ "IMAGE_CHUNK_CHARS",
127
+ "IMAGE_EXTENSIONS",
128
+ "MAX_INDEX_TEXT_CHARS",
129
+ "MAX_SUBTITLE_CHARS",
130
+ "MAX_THUMBNAIL_CHARS",
131
+ "MODALITY_AUDIO",
132
+ "MODALITY_IMAGE",
133
+ "MODALITY_TEXT",
134
+ "MODALITY_VIDEO",
135
+ "SUBTITLE_EXTENSIONS",
136
+ "SUMMARY_CHARS",
137
+ "THUMBNAIL_EDGE",
138
+ "VIDEO_EXTENSIONS",
139
+ "VIDEO_FRAME_RELATION",
140
+ "VIDEO_FRAME_SOURCE_TYPE",
141
+ "VIDEO_NODE_TYPE",
142
+ "VIDEO_OUT_OF_SCOPE",
143
+ "VIDEO_UNAVAILABLE_DETAIL",
144
+ "AudioFacts",
145
+ "ImageFacts",
146
+ "MultimodalPorts",
147
+ "VideoFacts",
148
+ "audio_quality_score",
149
+ "detect_modality",
150
+ "extract_image_facts",
151
+ "extract_keyframes",
152
+ "ffmpeg_available",
153
+ "find_subtitle",
154
+ "image_node_id",
155
+ "image_quality_score",
156
+ "parse_subtitles",
157
+ "read_video_facts",
158
+ "transcribe_audio",
159
+ "video_frame_dir",
160
+ "video_node_id",
161
+ "video_quality_score",
162
+ "write_image_memory",
163
+ "write_video_memory",
164
+ ]