ltcai 11.2.0 → 11.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (253) hide show
  1. package/README.md +50 -53
  2. package/docs/CHANGELOG.md +87 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/MULTI_AGENT_RUNTIME.md +1 -1
  6. package/docs/ONBOARDING.md +1 -1
  7. package/docs/OPERATIONS.md +6 -2
  8. package/docs/PERMISSION_MODE.md +1 -1
  9. package/docs/TRUST_MODEL.md +1 -1
  10. package/docs/WHY_LATTICE.md +1 -1
  11. package/docs/kg-schema.md +2 -2
  12. package/docs/v11.3.0_PLAN.md +202 -0
  13. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +181 -0
  14. package/docs/v11.5.0_RUST_COMPLETE_PLAN.md +145 -0
  15. package/lattice_brain/__init__.py +1 -1
  16. package/lattice_brain/graph/_kg_common/__init__.py +287 -0
  17. package/lattice_brain/graph/_kg_common/extraction.py +516 -0
  18. package/lattice_brain/graph/_kg_common/relations.py +161 -0
  19. package/lattice_brain/graph/_kg_common/text.py +479 -0
  20. package/lattice_brain/graph/discovery_index/__init__.py +35 -0
  21. package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
  22. package/lattice_brain/graph/discovery_index/extract.py +137 -0
  23. package/lattice_brain/graph/discovery_index/scan.py +411 -0
  24. package/lattice_brain/graph/discovery_index/upsert.py +495 -0
  25. package/lattice_brain/graph/projection/__init__.py +42 -0
  26. package/lattice_brain/graph/projection/curation.py +500 -0
  27. package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
  28. package/lattice_brain/graph/retrieval/__init__.py +54 -0
  29. package/lattice_brain/graph/retrieval/context.py +197 -0
  30. package/lattice_brain/graph/retrieval/graph_view.py +319 -0
  31. package/lattice_brain/graph/retrieval/hybrid.py +488 -0
  32. package/lattice_brain/graph/retrieval/maintenance.py +121 -0
  33. package/lattice_brain/graph/retrieval/signals.py +95 -0
  34. package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
  35. package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
  36. package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
  37. package/lattice_brain/graph/retrieval_vector/search.py +560 -0
  38. package/lattice_brain/graph/retrieval_vector/status.py +374 -0
  39. package/lattice_brain/ingestion/__init__.py +130 -0
  40. package/lattice_brain/ingestion/_contract.py +90 -0
  41. package/lattice_brain/ingestion/constants.py +127 -0
  42. package/lattice_brain/ingestion/folder_scan.py +57 -0
  43. package/lattice_brain/ingestion/folders.py +258 -0
  44. package/lattice_brain/ingestion/hashing.py +26 -0
  45. package/lattice_brain/ingestion/jobs_api.py +107 -0
  46. package/lattice_brain/ingestion/models.py +80 -0
  47. package/lattice_brain/ingestion/pipeline.py +486 -0
  48. package/lattice_brain/ingestion/quality.py +209 -0
  49. package/lattice_brain/ingestion/routing.py +295 -0
  50. package/lattice_brain/multimodal/__init__.py +164 -0
  51. package/lattice_brain/multimodal/audio.py +77 -0
  52. package/lattice_brain/multimodal/common.py +118 -0
  53. package/lattice_brain/multimodal/images.py +498 -0
  54. package/lattice_brain/multimodal/ports.py +169 -0
  55. package/lattice_brain/multimodal/video.py +410 -0
  56. package/lattice_brain/portability/__init__.py +90 -0
  57. package/lattice_brain/portability/_contract.py +42 -0
  58. package/lattice_brain/portability/backups.py +338 -0
  59. package/lattice_brain/portability/bundles.py +136 -0
  60. package/lattice_brain/portability/constants.py +93 -0
  61. package/lattice_brain/portability/fsops.py +138 -0
  62. package/lattice_brain/portability/service.py +41 -0
  63. package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
  64. package/lattice_brain/runtime/__init__.py +1 -1
  65. package/lattice_brain/runtime/multi_agent.py +1 -1
  66. package/latticeai/__init__.py +1 -1
  67. package/latticeai/api/chronicle.py +63 -0
  68. package/latticeai/api/index_jobs.py +145 -0
  69. package/latticeai/core/agent/__init__.py +93 -0
  70. package/latticeai/core/agent/_contract.py +79 -0
  71. package/latticeai/core/agent/context.py +57 -0
  72. package/latticeai/core/agent/deps.py +125 -0
  73. package/latticeai/core/agent/execution.py +622 -0
  74. package/latticeai/core/agent/planning.py +145 -0
  75. package/latticeai/core/agent/recovery.py +157 -0
  76. package/latticeai/core/agent/runtime.py +210 -0
  77. package/latticeai/core/agent/verification.py +231 -0
  78. package/latticeai/core/embedding_providers/__init__.py +151 -0
  79. package/latticeai/core/embedding_providers/base.py +199 -0
  80. package/latticeai/core/embedding_providers/captions.py +162 -0
  81. package/latticeai/core/embedding_providers/profiles.py +126 -0
  82. package/latticeai/core/embedding_providers/text.py +350 -0
  83. package/latticeai/core/embedding_providers/vision.py +352 -0
  84. package/latticeai/core/file_generation/__init__.py +115 -0
  85. package/latticeai/core/file_generation/bundles.py +76 -0
  86. package/latticeai/core/file_generation/extraction.py +154 -0
  87. package/latticeai/core/file_generation/inference.py +235 -0
  88. package/latticeai/core/file_generation/orchestration.py +152 -0
  89. package/latticeai/core/file_generation/prompting.py +117 -0
  90. package/latticeai/core/file_generation/repair.py +114 -0
  91. package/latticeai/core/file_generation/sanitize.py +61 -0
  92. package/latticeai/core/file_generation/validation.py +201 -0
  93. package/latticeai/core/legacy_compatibility.py +1 -1
  94. package/latticeai/core/marketplace.py +1 -1
  95. package/latticeai/core/messages.py +14 -0
  96. package/latticeai/core/workspace_os_constants.py +1 -1
  97. package/latticeai/integrations/telegram_bot/__init__.py +123 -0
  98. package/latticeai/integrations/telegram_bot/__main__.py +17 -0
  99. package/latticeai/integrations/telegram_bot/config.py +86 -0
  100. package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
  101. package/latticeai/integrations/telegram_bot/flows.py +478 -0
  102. package/latticeai/integrations/telegram_bot/helpers.py +322 -0
  103. package/latticeai/integrations/telegram_bot/screens.py +394 -0
  104. package/latticeai/models/router/__init__.py +88 -0
  105. package/latticeai/models/router/_contract.py +66 -0
  106. package/latticeai/models/router/branding.py +56 -0
  107. package/latticeai/models/router/catalog.py +69 -0
  108. package/latticeai/models/router/documents.py +199 -0
  109. package/latticeai/models/router/errors.py +37 -0
  110. package/latticeai/models/router/generation.py +258 -0
  111. package/latticeai/models/router/loading.py +291 -0
  112. package/latticeai/models/router/local_models.py +85 -0
  113. package/latticeai/models/router/registry.py +147 -0
  114. package/latticeai/runtime/build_phases/__init__.py +82 -0
  115. package/latticeai/runtime/build_phases/features.py +421 -0
  116. package/latticeai/runtime/build_phases/foundation.py +555 -0
  117. package/latticeai/runtime/build_phases/web.py +492 -0
  118. package/latticeai/runtime/runtime_context.py +1 -0
  119. package/latticeai/services/architecture_readiness.py +48 -19
  120. package/latticeai/services/brain_intelligence/__init__.py +58 -0
  121. package/latticeai/services/brain_intelligence/_contract.py +71 -0
  122. package/latticeai/services/brain_intelligence/consistency.py +193 -0
  123. package/latticeai/services/brain_intelligence/constants.py +47 -0
  124. package/latticeai/services/brain_intelligence/digest.py +258 -0
  125. package/latticeai/services/brain_intelligence/health.py +331 -0
  126. package/latticeai/services/brain_intelligence/proposals.py +264 -0
  127. package/latticeai/services/brain_intelligence/sampling.py +84 -0
  128. package/latticeai/services/brain_intelligence/service.py +48 -0
  129. package/latticeai/services/chronicle.py +557 -0
  130. package/latticeai/services/memory_service/__init__.py +52 -0
  131. package/latticeai/services/memory_service/_contract.py +100 -0
  132. package/latticeai/services/memory_service/brief.py +431 -0
  133. package/latticeai/services/memory_service/constants.py +57 -0
  134. package/latticeai/services/memory_service/maintenance.py +138 -0
  135. package/latticeai/services/memory_service/manager.py +186 -0
  136. package/latticeai/services/memory_service/proof.py +136 -0
  137. package/latticeai/services/memory_service/recall.py +225 -0
  138. package/latticeai/services/memory_service/service.py +48 -0
  139. package/latticeai/services/memory_service/stores.py +110 -0
  140. package/latticeai/services/model_runtime/__init__.py +322 -0
  141. package/latticeai/services/model_runtime/cloud.py +87 -0
  142. package/latticeai/services/model_runtime/download.py +282 -0
  143. package/latticeai/services/model_runtime/engines.py +341 -0
  144. package/latticeai/services/model_runtime/loading.py +178 -0
  145. package/latticeai/services/model_runtime/service.py +129 -0
  146. package/latticeai/services/model_runtime/state.py +131 -0
  147. package/latticeai/services/model_runtime/status.py +255 -0
  148. package/latticeai/services/product_readiness.py +15 -7
  149. package/latticeai/setup/wizard/__init__.py +126 -0
  150. package/latticeai/setup/wizard/catalog.py +172 -0
  151. package/latticeai/setup/wizard/detect.py +323 -0
  152. package/latticeai/setup/wizard/install.py +348 -0
  153. package/latticeai/setup/wizard/paths.py +168 -0
  154. package/latticeai/setup/wizard/plans.py +74 -0
  155. package/latticeai/setup/wizard/recommend.py +320 -0
  156. package/package.json +6 -2
  157. package/scripts/bump_version.py +14 -0
  158. package/scripts/capture_release_evidence.mjs +33 -21
  159. package/scripts/check_current_release_docs.mjs +1 -1
  160. package/scripts/check_i18n_namespace_coverage.mjs +41 -4
  161. package/scripts/check_max_file_lines.mjs +102 -0
  162. package/scripts/check_release_evidence_bound.mjs +30 -15
  163. package/scripts/check_screenshot_pixel_delta.py +34 -4
  164. package/scripts/check_server_i18n.mjs +2 -0
  165. package/scripts/chunking_parity_corpus.py +449 -0
  166. package/scripts/generate_agent_parity_fixtures.py +752 -0
  167. package/scripts/generate_chunking_parity_fixtures.py +259 -0
  168. package/scripts/generate_rust_parity_fixtures.py +997 -0
  169. package/scripts/lib/mock_server_fingerprint.mjs +94 -0
  170. package/scripts/release_screen_claims.json +42 -2
  171. package/src-tauri/Cargo.lock +404 -3
  172. package/src-tauri/Cargo.toml +13 -1
  173. package/src-tauri/src/backend.rs +460 -0
  174. package/src-tauri/src/folder.rs +33 -0
  175. package/src-tauri/src/main.rs +109 -399
  176. package/src-tauri/src/topology.rs +356 -0
  177. package/src-tauri/tauri.conf.json +1 -1
  178. package/static/app/asset-manifest.json +41 -37
  179. package/static/app/assets/Act-CWnxSCgN.js +1 -0
  180. package/static/app/assets/AdminConsole-BEQYU6kF.js +1 -0
  181. package/static/app/assets/{Brain-tuhI4sOC.js → Brain-DWu1BhFg.js} +2 -2
  182. package/static/app/assets/BrainHome-95Hilr9R.js +2 -0
  183. package/static/app/assets/BrainSignals-QdeqCpAF.js +1 -0
  184. package/static/app/assets/Capture-BHpCxnzb.js +1 -0
  185. package/static/app/assets/Chronicle-B4xYKoed.js +1 -0
  186. package/static/app/assets/CommandPalette-BVXnttSz.js +1 -0
  187. package/static/app/assets/Library-DgYcHome.js +1 -0
  188. package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-CrJLDbf7.js} +1 -1
  189. package/static/app/assets/ProductFlow-DFlScKoJ.js +1 -0
  190. package/static/app/assets/ReviewCard-Cy5f48Pj.js +3 -0
  191. package/static/app/assets/System-NF8IfhTa.js +1 -0
  192. package/static/app/assets/arrow-left-DwkSYrjR.js +1 -0
  193. package/static/app/assets/{bot-Cia42c2h.js → bot-CucuhLhm.js} +1 -1
  194. package/static/app/assets/brain-BBnSryW_.js +1 -0
  195. package/static/app/assets/{button-2j2Ijzgq.js → button-C2GUj2Ai.js} +1 -1
  196. package/static/app/assets/circle-check-CxOVPwYq.js +1 -0
  197. package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-CbkWzBmG.js} +1 -1
  198. package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-7lEaqHdJ.js} +1 -1
  199. package/static/app/assets/{cpu-k4awryFq.js → cpu-DAlCXlIy.js} +1 -1
  200. package/static/app/assets/{download-DFbLJ_ig.js → download-RNhuuJwh.js} +1 -1
  201. package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CLW4odzM.js} +1 -1
  202. package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-NKEiDIAJ.js} +1 -1
  203. package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
  204. package/static/app/assets/index-DMurvUuR.js +10 -0
  205. package/static/app/assets/input-D2UhPC1X.js +1 -0
  206. package/static/app/assets/link-2-6amKbP_P.js +1 -0
  207. package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-Cu9TZtdR.js} +1 -1
  208. package/static/app/assets/primitives-gPsccucr.js +1 -0
  209. package/static/app/assets/search-Cj_TKk_2.js +1 -0
  210. package/static/app/assets/{share-2-BH1M-WNi.js → share-2-Bau7KkPq.js} +1 -1
  211. package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-BufNYypi.js} +1 -1
  212. package/static/app/assets/{textarea-CCWbUfFB.js → textarea-BQnVWhYs.js} +1 -1
  213. package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-B3_w60si.js} +1 -1
  214. package/static/app/assets/useMutation-BHhCflT6.js +1 -0
  215. package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-rBWfI-5t.js} +1 -1
  216. package/static/app/assets/utils-V_5-wxr5.js +4 -0
  217. package/static/app/assets/workspace-K1zjYUHj.js +1 -0
  218. package/static/app/index.html +4 -4
  219. package/static/sw.js +1 -1
  220. package/lattice_brain/graph/_kg_common.py +0 -1331
  221. package/lattice_brain/graph/discovery_index.py +0 -1141
  222. package/lattice_brain/graph/retrieval.py +0 -1120
  223. package/lattice_brain/graph/retrieval_vector.py +0 -1293
  224. package/lattice_brain/ingestion.py +0 -1525
  225. package/lattice_brain/multimodal.py +0 -1258
  226. package/latticeai/core/agent.py +0 -1465
  227. package/latticeai/core/embedding_providers.py +0 -1196
  228. package/latticeai/core/file_generation.py +0 -1047
  229. package/latticeai/integrations/telegram_bot.py +0 -1390
  230. package/latticeai/models/router.py +0 -1007
  231. package/latticeai/runtime/build_phases.py +0 -1450
  232. package/latticeai/services/brain_intelligence.py +0 -1083
  233. package/latticeai/services/memory_service.py +0 -1177
  234. package/latticeai/services/model_runtime.py +0 -1281
  235. package/latticeai/setup/wizard.py +0 -1310
  236. package/static/app/assets/Act-AWf0SAKp.js +0 -1
  237. package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
  238. package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
  239. package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
  240. package/static/app/assets/Capture-CqOSzyPr.js +0 -1
  241. package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
  242. package/static/app/assets/Library-CX-bbhmK.js +0 -1
  243. package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
  244. package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
  245. package/static/app/assets/System-Bu2t5hn1.js +0 -1
  246. package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
  247. package/static/app/assets/brain-DJMoqrwx.js +0 -1
  248. package/static/app/assets/index-BpYkzcVm.js +0 -10
  249. package/static/app/assets/input-DSlJJxRs.js +0 -1
  250. package/static/app/assets/primitives-BCx6TvfG.js +0 -1
  251. package/static/app/assets/search-Cgy8cCFJ.js +0 -1
  252. package/static/app/assets/utils-zqPZJxdx.js +0 -4
  253. package/static/app/assets/workspace-DXTihhfU.js +0 -1
@@ -0,0 +1,449 @@
1
+ """The chunking-parity corpus: every input the goldens are built from.
2
+
3
+ Split out of :mod:`generate_chunking_parity_fixtures` so the generator stays a
4
+ runner and this stays a specification. Nothing here imports the product — these
5
+ are inputs, and the only thing that decides what they chunk into is the real
6
+ ``lattice_brain`` code the generator calls.
7
+
8
+ The corpus is the coverage, so every case is here for a reason a reader can
9
+ check. Broadly:
10
+
11
+ * **shape** — empty, whitespace-only, exactly ``size``, ``size - 1``,
12
+ ``size + 1``, a clamped overlap, a zero size;
13
+ * **markdown** — nested headings, an empty heading title, seven hashes and
14
+ ``#NoSpace`` (neither is a heading), sections at exactly 199 / 200 / 201
15
+ characters so the merge floor is observable, and a section too big for one
16
+ window;
17
+ * **code** — every one of the seven declaration prefixes at a line start that
18
+ is not already a boundary, blank-line runs filled with C0 separators (which
19
+ Python's ``\\s`` accepts and Unicode's ``White_Space`` does not), a segment
20
+ one character past ``int(size * 1.5)``, and greedy packing at a small window;
21
+ * **prose** — every sentence terminator and every closing mark, each alone in
22
+ its window so each is individually load-bearing, plus paragraph breaks,
23
+ line-break-only text and text with no boundary at all;
24
+ * **multibyte** — Korean, emoji with zero-width joiners, a regional-indicator
25
+ flag and a combining mark, straddling boundaries at small windows. Python
26
+ slices by *characters*; a byte-sliced port disagrees here and panics there.
27
+ """
28
+
29
+ from __future__ import annotations
30
+
31
+ from typing import Any, Dict, List, Optional
32
+
33
+
34
+ # ── corpus builders ──────────────────────────────────────────────────────────
35
+ def mixed_paragraphs(count: int) -> str:
36
+ """``count`` numbered ko/en lines, ~100 chars each, newline separated."""
37
+ return "\n".join(
38
+ f"{index:02d}. 회의에서 결정한 사항을 정리합니다. "
39
+ f"The retrieval pipeline chunks text before it is embedded. "
40
+ f"결정 근거는 문서 {index}번에 있습니다."
41
+ for index in range(count)
42
+ )
43
+
44
+
45
+ def markdown_document() -> str:
46
+ """Preamble, nested headings, tiny sections, an empty title, a big section."""
47
+ filler = " ".join(f"근거 문단 {i}: 검색 품질은 청크 경계에 좌우됩니다." for i in range(18))
48
+ return "\n".join(
49
+ [
50
+ "안내서 서문입니다. 이 문단은 첫 제목 앞에 옵니다.",
51
+ "",
52
+ "# 안내서",
53
+ "짧다.",
54
+ "## 설치",
55
+ "설치 방법은 다음과 같습니다. " * 12,
56
+ "### 사전 준비",
57
+ "준비물.",
58
+ "### 실행",
59
+ "실행 방법.",
60
+ "## 사용",
61
+ filler,
62
+ "# ",
63
+ "빈 제목 아래 본문입니다. " * 18,
64
+ "###### 여섯 단계",
65
+ "가장 깊은 제목입니다. " * 18,
66
+ "## 마무리",
67
+ "끝.",
68
+ ]
69
+ )
70
+
71
+
72
+ def markdown_all_tiny() -> str:
73
+ """Every section under the 200-char floor — the merge never completes."""
74
+ return "\n".join(
75
+ ["# 하나", "짧다.", "## 둘", "더 짧다.", "### 셋", "끝."]
76
+ )
77
+
78
+
79
+ def markdown_threshold() -> str:
80
+ """Sections spanning 250 / 199 / 250 / 200 / 250 / 201 / 300 characters.
81
+
82
+ The 200-char merge floor is only observable when a section lands exactly on
83
+ it: a corpus of tiny-and-huge sections passes with any floor between them.
84
+ Here 199 merges forward, 200 stands alone and 201 stands alone, so both a
85
+ floor of 199 and a floor of 201 produce a different chunk list.
86
+ """
87
+ parts = []
88
+ for index, span in enumerate((250, 199, 250, 200, 250, 201, 300)):
89
+ heading = f"# S{index}\n"
90
+ parts.append(heading + "x" * (span - len(heading) - 1) + "\n")
91
+ return "".join(parts)
92
+
93
+
94
+ def markdown_seven_hashes() -> str:
95
+ """Things that look like headings and are not (7 hashes, no space)."""
96
+ return "\n".join(
97
+ [
98
+ "####### 일곱 개는 제목이 아니다",
99
+ "#공백없음",
100
+ "## 진짜 제목",
101
+ "본문입니다. " * 40,
102
+ " ## 들여쓴 제목은 제목이 아니다",
103
+ "마지막 본문. " * 40,
104
+ ]
105
+ )
106
+
107
+
108
+ def python_source() -> str:
109
+ """Declaration lines and blank-line runs, the two code-segment boundaries."""
110
+ return "\n".join(
111
+ [
112
+ "import os",
113
+ "",
114
+ "",
115
+ "def alpha(value):",
116
+ " return value + 1",
117
+ "",
118
+ "",
119
+ "class Beta:",
120
+ ' """도움말 문자열입니다."""',
121
+ "",
122
+ " def gamma(self):",
123
+ " return 2",
124
+ "",
125
+ "",
126
+ "const_value = 3",
127
+ "const = 4",
128
+ "",
129
+ "public = 5",
130
+ "private = 6",
131
+ "",
132
+ "",
133
+ "def omega(value):",
134
+ " # 마지막 함수입니다.",
135
+ " return value * 2",
136
+ ]
137
+ )
138
+
139
+
140
+ def javascript_source() -> str:
141
+ """``export`` / ``const`` / ``function`` declaration starts."""
142
+ return "\n".join(
143
+ [
144
+ "export const alpha = 1;",
145
+ "",
146
+ "function beta(value) {",
147
+ " return value + 1;",
148
+ "}",
149
+ "",
150
+ "",
151
+ "export default function gamma() {",
152
+ " return '결과';",
153
+ "}",
154
+ "",
155
+ "const delta = () => 4;",
156
+ ]
157
+ )
158
+
159
+
160
+ def code_monster_segment() -> str:
161
+ """One segment past ``size * 1.5`` (1.8x at size=100), flanked by small ones."""
162
+ monster = "x = [" + ", ".join(str(n) for n in range(60)) + "]"
163
+ return "\n".join(["def head():", " return 1", "", "", monster, "", "", "def tail():", " return 2"])
164
+
165
+
166
+ def code_hard_limit_boundary() -> str:
167
+ """A segment of exactly 38 characters — one past ``int(25 * 1.5)``.
168
+
169
+ ``int(size * 1.5)`` truncates. A port that rounds instead computes 38 and
170
+ packs this segment where Python windows it, which is a whole different
171
+ chunk list from a single character of arithmetic.
172
+ """
173
+ return "def a():\n\n\n" + "a" * 35 + "\n\n\ndef b():\n 2"
174
+
175
+
176
+ def code_c0_blank_run() -> str:
177
+ """A blank-line run whose filler is a C0 separator, not a space.
178
+
179
+ Python's ``\\s`` accepts ``\\x1c``-``\\x1f``; the Unicode ``White_Space``
180
+ property does not. A port that reached for a stock Unicode ``\\s`` would
181
+ miss these runs and pack the statements around them into one segment.
182
+ """
183
+ return "x = 1\n\ny = 2\n\x1c\nz = 3\n\x1f \x1e\nw = 4\n\nv = 5"
184
+
185
+
186
+ def code_declaration_matrix() -> str:
187
+ """All seven declaration prefixes, each the only boundary on its line.
188
+
189
+ No blank lines anywhere, so every segment boundary comes from a declaration
190
+ start. At a small window each of the seven is individually load-bearing:
191
+ drop one and the two segments around it fuse into a different chunk.
192
+ """
193
+ return "\n".join(
194
+ [
195
+ "def a(): 1",
196
+ "class b: 2",
197
+ "function c() {}",
198
+ "export d = 4",
199
+ "const e = 5",
200
+ "public f = 6",
201
+ "private g = 7",
202
+ "def h(): 8",
203
+ "tail = 9",
204
+ ]
205
+ )
206
+
207
+
208
+ def prose_terminator_matrix() -> str:
209
+ """One sentence per sentence-final mark, each alone in its window."""
210
+ body = "가나다라마바사아자차카타파하"
211
+ return "".join(f"{body}{mark} " for mark in (".", "!", "?", "。", "!", "?", "…"))
212
+
213
+
214
+ def prose_closer_matrix() -> str:
215
+ """One sentence per closing quote/bracket, each alone in its window.
216
+
217
+ Drop any single closer from the port's set and that sentence's boundary
218
+ becomes a hard window cut instead, which moves every chunk after it.
219
+ """
220
+ body = "가나다라마바사아자차카타파하"
221
+ closers = ('"', "'", "\u201d", "\u2019", "\u300d", "\u300f", ")", "]")
222
+ # A plain trailing sentence, so the *last* closer is load-bearing too:
223
+ # a boundary at the very end of the text never changes a chunk list.
224
+ return "".join(f"{body}.{closer} " for closer in closers) + f"{body}. "
225
+
226
+
227
+ def prose_closers() -> str:
228
+ """Every closing quote and bracket the strong boundary allows."""
229
+ return (
230
+ '그는 "끝났다." 라고 했다. '
231
+ "그녀는 '정말?' 이라고 물었다. "
232
+ "안내문은 “확인했다.” 였다. "
233
+ "메모는 ‘완료’ 였다. "
234
+ "인용은 「검토함.」 이었다. "
235
+ "출처는 『보고서.』 이다. "
236
+ "각주는 (참고함.) 이다. "
237
+ "표는 [완료됨.] 이다. "
238
+ "마지막 문장이다."
239
+ )
240
+
241
+
242
+ def prose_english() -> str:
243
+ return (
244
+ "The retrieval pipeline chunks text before it is embedded. "
245
+ "Each chunk carries a start offset so a citation can point at it! "
246
+ "Does the boundary land on a sentence? It does, when one is in range. "
247
+ 'She said "the boundary matters" (twice), and nobody disagreed. '
248
+ "A final sentence closes the paragraph without any surprises."
249
+ )
250
+
251
+
252
+ def prose_korean() -> str:
253
+ return (
254
+ "회의에서 결정한 사항을 정리합니다。 근거는 문서에 남겨 두었습니다! "
255
+ "다음 주에 다시 검토할까요? 검토 결과는 여기에 덧붙입니다… "
256
+ "한국어는 문장 끝에 서술어가 오기 때문에 경계가 특히 중요합니다. "
257
+ "그래서 청크 경계를 문장 끝에 맞춥니다."
258
+ )
259
+
260
+
261
+ def prose_no_boundary() -> str:
262
+ """No punctuation, no line break — the hard window cut is the only answer."""
263
+ return "가나다라마바사아자차카타파하" * 12
264
+
265
+
266
+ def prose_lines_only() -> str:
267
+ """Bullet lines with no sentence punctuation — the weak boundary path."""
268
+ return "\n".join(f"- 항목 {index} 준비 완료" for index in range(30))
269
+
270
+
271
+ def prose_paragraphs() -> str:
272
+ return "\n\n".join(
273
+ f"{index}번 문단입니다 이 문단에는 마침표가 없습니다 그래서 문단 경계만 남습니다"
274
+ for index in range(12)
275
+ )
276
+
277
+
278
+ #: Emoji with zero-width joiners, a regional-indicator flag and a combining
279
+ #: mark: four graphemes, eleven code points, thirty-eight UTF-8 bytes. Slicing
280
+ #: this by bytes is not "slightly different", it is a panic.
281
+ GRAPHEME_SOUP = "a👨‍👩‍👧‍👦b🇰🇷cée"
282
+
283
+
284
+ def unicode_boundary_text() -> str:
285
+ return (GRAPHEME_SOUP + "한글") * 8
286
+
287
+
288
+ def ascii_of_length(length: int) -> str:
289
+ """``length`` printable ASCII chars — one char is one byte, on purpose."""
290
+ alphabet = "abcdefghijklmnopqrstuvwxyz0123456789"
291
+ return "".join(alphabet[index % len(alphabet)] for index in range(length))
292
+
293
+
294
+ def korean_of_length(length: int) -> str:
295
+ """``length`` Hangul syllables — one char is three bytes, on purpose."""
296
+ syllables = "가나다라마바사아자차카타파하"
297
+ return "".join(syllables[index % len(syllables)] for index in range(length))
298
+
299
+
300
+ # ── the case set ─────────────────────────────────────────────────────────────
301
+ # ``strategy`` None means "route it from ``filename`` via chunk_strategy_for",
302
+ # which is how every real call site picks one.
303
+ CASES: List[Dict[str, Any]] = [
304
+ {"key": "empty", "text": "", "filename": "empty.txt"},
305
+ {"key": "whitespace_only", "text": " \n\t\r\n ", "filename": "blank.txt"},
306
+ {"key": "plain_short", "text": "짧은 메모 한 줄.", "filename": "note"},
307
+ {"key": "plain_strip_not_collapse", "text": " \n 두 칸 사이 간격은 유지된다. \n ", "filename": "note"},
308
+ {"key": "plain_exact_minus_one", "text": ascii_of_length(63), "filename": "n", "size": 64, "overlap": 8},
309
+ {"key": "plain_exact", "text": ascii_of_length(64), "filename": "n", "size": 64, "overlap": 8},
310
+ {"key": "plain_exact_plus_one", "text": ascii_of_length(65), "filename": "n", "size": 64, "overlap": 8},
311
+ {"key": "plain_default_long", "text": mixed_paragraphs(30), "filename": "log"},
312
+ {"key": "plain_korean_small_window", "text": korean_of_length(37), "filename": "n", "size": 10, "overlap": 3},
313
+ {"key": "plain_grapheme_soup", "text": unicode_boundary_text(), "filename": "n", "size": 7, "overlap": 2},
314
+ {"key": "plain_overlap_clamped", "text": ascii_of_length(30), "filename": "n", "size": 5, "overlap": 100},
315
+ {"key": "plain_overlap_zero", "text": ascii_of_length(30), "filename": "n", "size": 7, "overlap": 0},
316
+ {"key": "plain_overlap_negative", "text": ascii_of_length(30), "filename": "n", "size": 7, "overlap": -4},
317
+ {"key": "plain_size_one", "text": korean_of_length(6), "filename": "n", "size": 1, "overlap": 160},
318
+ {"key": "plain_size_zero_coerced", "text": ascii_of_length(9), "filename": "n", "size": 0, "overlap": 3},
319
+ {"key": "unknown_strategy_falls_back", "text": mixed_paragraphs(4), "filename": "x.md", "strategy": "sideways"},
320
+ {"key": "markdown_full", "text": markdown_document(), "filename": "guide.md"},
321
+ {"key": "markdown_full_small_window", "text": markdown_document(), "filename": "guide.md", "size": 120, "overlap": 30},
322
+ {"key": "markdown_all_tiny", "text": markdown_all_tiny(), "filename": "tiny.markdown"},
323
+ {"key": "markdown_threshold", "text": markdown_threshold(), "filename": "floor.md"},
324
+ {"key": "markdown_not_headings", "text": markdown_seven_hashes(), "filename": "edge.md"},
325
+ {"key": "markdown_no_heading", "text": mixed_paragraphs(6), "filename": "plainish.md"},
326
+ {"key": "markdown_grapheme_soup", "text": "# 제목\n" + unicode_boundary_text(), "filename": "u.md", "size": 9, "overlap": 3},
327
+ {"key": "code_python", "text": python_source(), "filename": "module.py"},
328
+ {"key": "code_python_packed", "text": python_source(), "filename": "module.py", "size": 60, "overlap": 12},
329
+ {"key": "code_javascript", "text": javascript_source(), "filename": "app.tsx", "size": 80, "overlap": 16},
330
+ {"key": "code_monster_segment", "text": code_monster_segment(), "filename": "big.py", "size": 100, "overlap": 20},
331
+ {"key": "code_korean_small", "text": python_source(), "filename": "module.py", "size": 24, "overlap": 5},
332
+ {"key": "code_hard_limit_boundary", "text": code_hard_limit_boundary(), "filename": "edge.py", "size": 25, "overlap": 5},
333
+ {"key": "code_c0_blank_run", "text": code_c0_blank_run(), "filename": "c0.py", "size": 12, "overlap": 3},
334
+ {"key": "prose_closers", "text": prose_closers(), "filename": "quotes.txt", "size": 45, "overlap": 10},
335
+ {"key": "code_declaration_matrix", "text": code_declaration_matrix(), "filename": "matrix.py", "size": 12, "overlap": 3},
336
+ {"key": "prose_terminator_matrix", "text": prose_terminator_matrix(), "filename": "terms.txt", "size": 20, "overlap": 2},
337
+ {"key": "prose_closer_matrix", "text": prose_closer_matrix(), "filename": "closers.txt", "size": 20, "overlap": 2},
338
+ {"key": "prose_english", "text": prose_english(), "filename": "essay.txt", "size": 90, "overlap": 20},
339
+ {"key": "prose_korean", "text": prose_korean(), "filename": "essay.txt", "size": 60, "overlap": 15},
340
+ {"key": "prose_no_boundary", "text": prose_no_boundary(), "filename": "essay.txt", "size": 40, "overlap": 9},
341
+ {"key": "prose_lines_only", "text": prose_lines_only(), "filename": "list.txt", "size": 70, "overlap": 14},
342
+ {"key": "prose_paragraphs", "text": prose_paragraphs(), "filename": "para.txt", "size": 110, "overlap": 25},
343
+ {"key": "prose_default_window", "text": mixed_paragraphs(40), "filename": "report.pdf"},
344
+ {"key": "prose_grapheme_soup", "text": unicode_boundary_text(), "filename": "u.html", "size": 11, "overlap": 4},
345
+ # overlap == size - 1: whenever a sentence boundary lands closer to the
346
+ # start than the overlap, ``end - overlap`` goes backwards and the
347
+ # ``max(start + 1, …)`` guard is the only thing that ends the loop.
348
+ {"key": "prose_tight_overlap", "text": prose_english()[:110], "filename": "essay.txt", "size": 40, "overlap": 39},
349
+ ]
350
+
351
+ #: Filename/MIME pairs pinning every branch of ``chunk_strategy_for``.
352
+ STRATEGY_CASES: List[Dict[str, str]] = [
353
+ {"filename": "guide.md", "content_type": ""},
354
+ {"filename": "GUIDE.MARKDOWN", "content_type": ""},
355
+ {"filename": "module.py", "content_type": ""},
356
+ {"filename": "app.TSX", "content_type": ""},
357
+ {"filename": "data.json", "content_type": ""},
358
+ {"filename": "conf.toml", "content_type": ""},
359
+ {"filename": "notes.txt", "content_type": ""},
360
+ {"filename": "report.pdf", "content_type": ""},
361
+ {"filename": "page.HTM", "content_type": ""},
362
+ {"filename": "book.epub", "content_type": ""},
363
+ {"filename": "archive.tar.gz", "content_type": ""},
364
+ {"filename": "noextension", "content_type": ""},
365
+ {"filename": ".hidden", "content_type": ""},
366
+ {"filename": "trailing.", "content_type": ""},
367
+ {"filename": "", "content_type": ""},
368
+ {"filename": " ", "content_type": ""},
369
+ {"filename": "https://example.com/docs/guide.md?v=2#top", "content_type": ""},
370
+ {"filename": "https://example.com/docs/guide?v=2#top", "content_type": ""},
371
+ {"filename": "C:\\projects\\lattice\\module.py", "content_type": ""},
372
+ {"filename": "/var/data/notes/", "content_type": ""},
373
+ {"filename": "/var/data/notes//", "content_type": ""},
374
+ {"filename": "noextension", "content_type": "text/markdown"},
375
+ {"filename": "noextension", "content_type": "TEXT/HTML; charset=utf-8"},
376
+ {"filename": "noextension", "content_type": "text/plain"},
377
+ {"filename": "noextension", "content_type": "application/octet-stream"},
378
+ {"filename": "noextension", "content_type": " application/x-markdown "},
379
+ {"filename": "report.pdf", "content_type": "text/markdown"},
380
+ {"filename": "한글 문서.md", "content_type": ""},
381
+ {"filename": "한글 문서", "content_type": "text/plain"},
382
+ ]
383
+
384
+ #: ``metadata["structure"]`` shapes for ``pdf_page_offsets``, well formed and not.
385
+ PDF_STRUCTURES: List[Dict[str, Any]] = [
386
+ {"key": "three_pages", "structure": {"pages": [{"chars": 100}, {"chars": 250}, {"chars": 40}]}},
387
+ {"key": "single_page", "structure": {"pages": [{"chars": 1200}]}},
388
+ {"key": "zero_length_page", "structure": {"pages": [{"chars": 0}, {"chars": 10}, {"chars": 0}]}},
389
+ {"key": "float_chars", "structure": {"pages": [{"chars": 10.9}, {"chars": 5.0}]}},
390
+ {"key": "no_pages_key", "structure": {"meta": 1}},
391
+ {"key": "pages_not_list", "structure": {"pages": {"chars": 5}}},
392
+ {"key": "pages_empty", "structure": {"pages": []}},
393
+ {"key": "page_not_dict", "structure": {"pages": [{"chars": 5}, 7]}},
394
+ {"key": "chars_missing", "structure": {"pages": [{"chars": 5}, {}]}},
395
+ {"key": "chars_negative", "structure": {"pages": [{"chars": 5}, {"chars": -1}]}},
396
+ {"key": "chars_bool", "structure": {"pages": [{"chars": True}]}},
397
+ {"key": "chars_string", "structure": {"pages": [{"chars": "5"}]}},
398
+ {"key": "structure_not_dict", "structure": [1, 2, 3]},
399
+ {"key": "structure_null", "structure": None},
400
+ ]
401
+
402
+ #: Offsets probed against the ``three_pages`` / ``zero_length_page`` offsets.
403
+ PAGE_PROBES: List[int] = [-5, -1, 0, 1, 99, 100, 101, 102, 351, 352, 353, 10_000]
404
+
405
+ #: Chunk-metadata blobs for ``citation_locator``.
406
+ LOCATOR_CASES: List[Dict[str, Any]] = [
407
+ {},
408
+ {"heading_path": "안내서 > 온보딩"},
409
+ {"heading_path": " spaced "},
410
+ {"heading_path": ""},
411
+ {"page": 3},
412
+ {"page": 3, "page_end": 5},
413
+ {"page": 3, "page_end": 3},
414
+ {"page": 3, "page_end": 2},
415
+ {"page": 0},
416
+ {"page": -1},
417
+ {"page": "4"},
418
+ {"page": "nope"},
419
+ {"page": None},
420
+ {"heading_path": "Retrieval > Fusion", "page": 2, "page_end": 4},
421
+ ]
422
+
423
+ #: ``(source_type, source_uri, text, workspace_id)`` for the text/web hash rule.
424
+ TEXT_HASH_CASES: List[Dict[str, Optional[str]]] = [
425
+ {"source_type": "note", "source_uri": None, "text": "회의 결정 사항", "workspace_id": None},
426
+ {"source_type": "note", "source_uri": "", "text": "회의 결정 사항", "workspace_id": "ws-alpha"},
427
+ {"source_type": "web_url", "source_uri": "https://example.com/a", "text": "hello", "workspace_id": "ws-beta"},
428
+ {"source_type": "text", "source_uri": "file:///tmp/x.txt", "text": "", "workspace_id": None},
429
+ {"source_type": "markdown", "source_uri": "s3://b/k", "text": GRAPHEME_SOUP, "workspace_id": "ws-∅"},
430
+ ]
431
+
432
+ #: Byte payloads for the file content-hash rule (files hash their **bytes**).
433
+ FILE_HASH_CASES: List[bytes] = [
434
+ b"",
435
+ b"hello world\n",
436
+ "회의록\n".encode(),
437
+ bytes(range(256)),
438
+ ]
439
+
440
+ #: Texts whose ``_clean_text`` → sha256 pair is the vector-index ``text_hash``.
441
+ VECTOR_TEXT_CASES: List[str] = [
442
+ "",
443
+ " ",
444
+ "a",
445
+ " 회의 결정\t사항 ",
446
+ "line one\nline two\r\nline three",
447
+ "회의\x1c록",
448
+ GRAPHEME_SOUP,
449
+ ]