ltcai 11.4.0 → 11.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. package/README.md +48 -44
  2. package/docs/CHANGELOG.md +26 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/ONBOARDING.md +1 -1
  6. package/docs/OPERATIONS.md +1 -1
  7. package/docs/TRUST_MODEL.md +1 -1
  8. package/docs/WHY_LATTICE.md +1 -1
  9. package/docs/kg-schema.md +1 -1
  10. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +11 -6
  11. package/docs/v11.5.0_RUST_COMPLETE_PLAN.md +145 -0
  12. package/lattice_brain/__init__.py +1 -1
  13. package/lattice_brain/runtime/multi_agent.py +1 -1
  14. package/latticeai/__init__.py +1 -1
  15. package/latticeai/api/index_jobs.py +145 -0
  16. package/latticeai/core/legacy_compatibility.py +1 -1
  17. package/latticeai/core/marketplace.py +1 -1
  18. package/latticeai/core/messages.py +5 -0
  19. package/latticeai/core/workspace_os_constants.py +1 -1
  20. package/latticeai/runtime/build_phases/features.py +14 -0
  21. package/latticeai/services/architecture_readiness.py +1 -1
  22. package/latticeai/services/product_readiness.py +1 -1
  23. package/package.json +1 -1
  24. package/scripts/check_current_release_docs.mjs +1 -1
  25. package/scripts/check_server_i18n.mjs +1 -0
  26. package/scripts/chunking_parity_corpus.py +449 -0
  27. package/scripts/generate_agent_parity_fixtures.py +752 -0
  28. package/scripts/generate_chunking_parity_fixtures.py +259 -0
  29. package/scripts/generate_rust_parity_fixtures.py +525 -90
  30. package/scripts/release_screen_claims.json +11 -0
  31. package/src-tauri/Cargo.lock +47 -4
  32. package/src-tauri/Cargo.toml +11 -4
  33. package/src-tauri/src/backend.rs +251 -140
  34. package/src-tauri/src/main.rs +16 -4
  35. package/src-tauri/src/topology.rs +356 -0
  36. package/src-tauri/tauri.conf.json +1 -1
  37. package/static/app/asset-manifest.json +40 -40
  38. package/static/app/assets/{Act-yYpYnn0v.js → Act-CWnxSCgN.js} +1 -1
  39. package/static/app/assets/{AdminConsole-DL3Cr5pL.js → AdminConsole-BEQYU6kF.js} +1 -1
  40. package/static/app/assets/{Brain-C1HBN0Wf.js → Brain-DWu1BhFg.js} +1 -1
  41. package/static/app/assets/{BrainHome-DoXRhUUC.js → BrainHome-95Hilr9R.js} +1 -1
  42. package/static/app/assets/{BrainSignals-6yR6ir5t.js → BrainSignals-QdeqCpAF.js} +1 -1
  43. package/static/app/assets/{Capture-CFIRsFNE.js → Capture-BHpCxnzb.js} +1 -1
  44. package/static/app/assets/{Chronicle-BZbEgiwN.js → Chronicle-B4xYKoed.js} +1 -1
  45. package/static/app/assets/{CommandPalette-D2pMxC2I.js → CommandPalette-BVXnttSz.js} +1 -1
  46. package/static/app/assets/{Library-DwO3yZST.js → Library-DgYcHome.js} +1 -1
  47. package/static/app/assets/{LivingBrain-Jn1GK0-S.js → LivingBrain-CrJLDbf7.js} +1 -1
  48. package/static/app/assets/{ProductFlow-B-w1R4Oo.js → ProductFlow-DFlScKoJ.js} +1 -1
  49. package/static/app/assets/{ReviewCard-6B27X8Vg.js → ReviewCard-Cy5f48Pj.js} +1 -1
  50. package/static/app/assets/{System-DW8F-2xL.js → System-NF8IfhTa.js} +1 -1
  51. package/static/app/assets/arrow-left-DwkSYrjR.js +1 -0
  52. package/static/app/assets/{bot-IM_E_Y12.js → bot-CucuhLhm.js} +1 -1
  53. package/static/app/assets/{brain-Ci1CkWjM.js → brain-BBnSryW_.js} +1 -1
  54. package/static/app/assets/{button-COwyqfHM.js → button-C2GUj2Ai.js} +1 -1
  55. package/static/app/assets/circle-check-CxOVPwYq.js +1 -0
  56. package/static/app/assets/{circle-pause-DEM4A1Y5.js → circle-pause-CbkWzBmG.js} +1 -1
  57. package/static/app/assets/{circle-play-C9djDuLd.js → circle-play-7lEaqHdJ.js} +1 -1
  58. package/static/app/assets/{cpu-DFdo1gw-.js → cpu-DAlCXlIy.js} +1 -1
  59. package/static/app/assets/{download-SnJL6oqk.js → download-RNhuuJwh.js} +1 -1
  60. package/static/app/assets/{folder-open-CqZeDkjE.js → folder-open-CLW4odzM.js} +1 -1
  61. package/static/app/assets/{hard-drive-j1jJXYYf.js → hard-drive-NKEiDIAJ.js} +1 -1
  62. package/static/app/assets/{index-_u5iUHDr.js → index-DMurvUuR.js} +3 -3
  63. package/static/app/assets/{input-B0lPdRQZ.js → input-D2UhPC1X.js} +1 -1
  64. package/static/app/assets/{link-2-CoFbooHS.js → link-2-6amKbP_P.js} +1 -1
  65. package/static/app/assets/{permissionCopy-BsyLxtao.js → permissionCopy-Cu9TZtdR.js} +1 -1
  66. package/static/app/assets/{primitives-DEbN-d6p.js → primitives-gPsccucr.js} +1 -1
  67. package/static/app/assets/search-Cj_TKk_2.js +1 -0
  68. package/static/app/assets/{share-2-CVtZ_ewX.js → share-2-Bau7KkPq.js} +1 -1
  69. package/static/app/assets/{shield-alert-CBi2GNWM.js → shield-alert-BufNYypi.js} +1 -1
  70. package/static/app/assets/{textarea-DNMpB5ih.js → textarea-BQnVWhYs.js} +1 -1
  71. package/static/app/assets/{useFocusTrap-C83t3GXF.js → useFocusTrap-B3_w60si.js} +1 -1
  72. package/static/app/assets/{useMutation-DtbJDoyz.js → useMutation-BHhCflT6.js} +1 -1
  73. package/static/app/assets/{useQuery-Dcp1OChy.js → useQuery-rBWfI-5t.js} +1 -1
  74. package/static/app/assets/{utils-BlZr7Pd4.js → utils-V_5-wxr5.js} +1 -1
  75. package/static/app/assets/{workspace-jJY4RuAV.js → workspace-K1zjYUHj.js} +1 -1
  76. package/static/app/index.html +3 -3
  77. package/static/sw.js +1 -1
  78. package/static/app/assets/arrow-left-DXvKg9U6.js +0 -1
  79. package/static/app/assets/circle-check-DfInj-qD.js +0 -1
  80. package/static/app/assets/search-BybIWPNd.js +0 -1
@@ -25,6 +25,7 @@ def phase_platform_features(ctx: RuntimeContext) -> None:
25
25
  from latticeai.api.command_center import create_command_center_router
26
26
  from latticeai.api.evidence_actions import create_evidence_actions_router
27
27
  from latticeai.api.funnel_metrics import create_funnel_metrics_router
28
+ from latticeai.api.index_jobs import create_index_jobs_router
28
29
  from latticeai.api.marketplace import create_marketplace_router
29
30
  from latticeai.api.plugins import create_plugins_router
30
31
  from latticeai.api.project_sessions import create_project_sessions_router
@@ -176,6 +177,19 @@ def phase_platform_features(ctx: RuntimeContext) -> None:
176
177
  )
177
178
  )
178
179
 
180
+ # Index jobs (v11.5.0): the embed backlog has been durable since 11.1.0 and
181
+ # had no trigger — the only drain was a pipeline method no scheduler could
182
+ # reach. This is the HTTP surface lattice-jobs ticks.
183
+ ctx.app.include_router(
184
+ create_index_jobs_router(
185
+ pipeline=ctx.INGESTION_PIPELINE if ctx.ENABLE_GRAPH else None,
186
+ knowledge_graph=ctx.KNOWLEDGE_GRAPH if ctx.ENABLE_GRAPH else None,
187
+ require_user=ctx.require_user,
188
+ gate_read=ctx.PLATFORM.gate_read,
189
+ gate_write=ctx.PLATFORM.gate_write,
190
+ )
191
+ )
192
+
179
193
  ctx.set(
180
194
  CHANGE_PROPOSALS=ChangeProposalService(
181
195
  review_queue=ctx.REVIEW_QUEUE,
@@ -16,7 +16,7 @@ from typing import Any, Dict, List
16
16
 
17
17
  from latticeai.core.legacy_compatibility import legacy_shim_report
18
18
 
19
- ARCHITECTURE_VERSION_TARGET = "11.4.0"
19
+ ARCHITECTURE_VERSION_TARGET = "11.5.0"
20
20
 
21
21
  PREFERRED_REFACTORING_ORDER = [
22
22
  "agent-runtime",
@@ -18,7 +18,7 @@ from typing import Any, Dict, List
18
18
 
19
19
  from latticeai.services.architecture_readiness import architecture_readiness
20
20
 
21
- PRODUCT_VERSION_TARGET = "11.4.0"
21
+ PRODUCT_VERSION_TARGET = "11.5.0"
22
22
 
23
23
 
24
24
  @dataclass(frozen=True)
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "ltcai",
3
- "version": "11.4.0",
3
+ "version": "11.5.0",
4
4
  "description": "Lattice AI — local-first Digital Brain that keeps your knowledge durable across any AI model.",
5
5
  "homepage": "https://github.com/TaeSooPark-PTS/LatticeAI#readme",
6
6
  "repository": {
@@ -6,7 +6,7 @@ const root = process.cwd();
6
6
  const pkg = JSON.parse(readFileSync(path.join(root, "package.json"), "utf8"));
7
7
  const version = pkg.version;
8
8
  const releaseDir = `output/release/v${version}`;
9
- const releaseTheme = "Rust Foundation";
9
+ const releaseTheme = "Rust Complete";
10
10
  const title = `${version} — ${releaseTheme}`;
11
11
  const escapedVersion = version.replaceAll(".", "\\.");
12
12
 
@@ -36,6 +36,7 @@ const LOCALIZED = [
36
36
  "chat_intents",
37
37
  "chronicle",
38
38
  "features",
39
+ "index_jobs",
39
40
  "knowledge_graph",
40
41
  "local_files",
41
42
  "mcp",
@@ -0,0 +1,449 @@
1
+ """The chunking-parity corpus: every input the goldens are built from.
2
+
3
+ Split out of :mod:`generate_chunking_parity_fixtures` so the generator stays a
4
+ runner and this stays a specification. Nothing here imports the product — these
5
+ are inputs, and the only thing that decides what they chunk into is the real
6
+ ``lattice_brain`` code the generator calls.
7
+
8
+ The corpus is the coverage, so every case is here for a reason a reader can
9
+ check. Broadly:
10
+
11
+ * **shape** — empty, whitespace-only, exactly ``size``, ``size - 1``,
12
+ ``size + 1``, a clamped overlap, a zero size;
13
+ * **markdown** — nested headings, an empty heading title, seven hashes and
14
+ ``#NoSpace`` (neither is a heading), sections at exactly 199 / 200 / 201
15
+ characters so the merge floor is observable, and a section too big for one
16
+ window;
17
+ * **code** — every one of the seven declaration prefixes at a line start that
18
+ is not already a boundary, blank-line runs filled with C0 separators (which
19
+ Python's ``\\s`` accepts and Unicode's ``White_Space`` does not), a segment
20
+ one character past ``int(size * 1.5)``, and greedy packing at a small window;
21
+ * **prose** — every sentence terminator and every closing mark, each alone in
22
+ its window so each is individually load-bearing, plus paragraph breaks,
23
+ line-break-only text and text with no boundary at all;
24
+ * **multibyte** — Korean, emoji with zero-width joiners, a regional-indicator
25
+ flag and a combining mark, straddling boundaries at small windows. Python
26
+ slices by *characters*; a byte-sliced port disagrees here and panics there.
27
+ """
28
+
29
+ from __future__ import annotations
30
+
31
+ from typing import Any, Dict, List, Optional
32
+
33
+
34
+ # ── corpus builders ──────────────────────────────────────────────────────────
35
+ def mixed_paragraphs(count: int) -> str:
36
+ """``count`` numbered ko/en lines, ~100 chars each, newline separated."""
37
+ return "\n".join(
38
+ f"{index:02d}. 회의에서 결정한 사항을 정리합니다. "
39
+ f"The retrieval pipeline chunks text before it is embedded. "
40
+ f"결정 근거는 문서 {index}번에 있습니다."
41
+ for index in range(count)
42
+ )
43
+
44
+
45
+ def markdown_document() -> str:
46
+ """Preamble, nested headings, tiny sections, an empty title, a big section."""
47
+ filler = " ".join(f"근거 문단 {i}: 검색 품질은 청크 경계에 좌우됩니다." for i in range(18))
48
+ return "\n".join(
49
+ [
50
+ "안내서 서문입니다. 이 문단은 첫 제목 앞에 옵니다.",
51
+ "",
52
+ "# 안내서",
53
+ "짧다.",
54
+ "## 설치",
55
+ "설치 방법은 다음과 같습니다. " * 12,
56
+ "### 사전 준비",
57
+ "준비물.",
58
+ "### 실행",
59
+ "실행 방법.",
60
+ "## 사용",
61
+ filler,
62
+ "# ",
63
+ "빈 제목 아래 본문입니다. " * 18,
64
+ "###### 여섯 단계",
65
+ "가장 깊은 제목입니다. " * 18,
66
+ "## 마무리",
67
+ "끝.",
68
+ ]
69
+ )
70
+
71
+
72
+ def markdown_all_tiny() -> str:
73
+ """Every section under the 200-char floor — the merge never completes."""
74
+ return "\n".join(
75
+ ["# 하나", "짧다.", "## 둘", "더 짧다.", "### 셋", "끝."]
76
+ )
77
+
78
+
79
+ def markdown_threshold() -> str:
80
+ """Sections spanning 250 / 199 / 250 / 200 / 250 / 201 / 300 characters.
81
+
82
+ The 200-char merge floor is only observable when a section lands exactly on
83
+ it: a corpus of tiny-and-huge sections passes with any floor between them.
84
+ Here 199 merges forward, 200 stands alone and 201 stands alone, so both a
85
+ floor of 199 and a floor of 201 produce a different chunk list.
86
+ """
87
+ parts = []
88
+ for index, span in enumerate((250, 199, 250, 200, 250, 201, 300)):
89
+ heading = f"# S{index}\n"
90
+ parts.append(heading + "x" * (span - len(heading) - 1) + "\n")
91
+ return "".join(parts)
92
+
93
+
94
+ def markdown_seven_hashes() -> str:
95
+ """Things that look like headings and are not (7 hashes, no space)."""
96
+ return "\n".join(
97
+ [
98
+ "####### 일곱 개는 제목이 아니다",
99
+ "#공백없음",
100
+ "## 진짜 제목",
101
+ "본문입니다. " * 40,
102
+ " ## 들여쓴 제목은 제목이 아니다",
103
+ "마지막 본문. " * 40,
104
+ ]
105
+ )
106
+
107
+
108
+ def python_source() -> str:
109
+ """Declaration lines and blank-line runs, the two code-segment boundaries."""
110
+ return "\n".join(
111
+ [
112
+ "import os",
113
+ "",
114
+ "",
115
+ "def alpha(value):",
116
+ " return value + 1",
117
+ "",
118
+ "",
119
+ "class Beta:",
120
+ ' """도움말 문자열입니다."""',
121
+ "",
122
+ " def gamma(self):",
123
+ " return 2",
124
+ "",
125
+ "",
126
+ "const_value = 3",
127
+ "const = 4",
128
+ "",
129
+ "public = 5",
130
+ "private = 6",
131
+ "",
132
+ "",
133
+ "def omega(value):",
134
+ " # 마지막 함수입니다.",
135
+ " return value * 2",
136
+ ]
137
+ )
138
+
139
+
140
+ def javascript_source() -> str:
141
+ """``export`` / ``const`` / ``function`` declaration starts."""
142
+ return "\n".join(
143
+ [
144
+ "export const alpha = 1;",
145
+ "",
146
+ "function beta(value) {",
147
+ " return value + 1;",
148
+ "}",
149
+ "",
150
+ "",
151
+ "export default function gamma() {",
152
+ " return '결과';",
153
+ "}",
154
+ "",
155
+ "const delta = () => 4;",
156
+ ]
157
+ )
158
+
159
+
160
+ def code_monster_segment() -> str:
161
+ """One segment past ``size * 1.5`` (1.8x at size=100), flanked by small ones."""
162
+ monster = "x = [" + ", ".join(str(n) for n in range(60)) + "]"
163
+ return "\n".join(["def head():", " return 1", "", "", monster, "", "", "def tail():", " return 2"])
164
+
165
+
166
+ def code_hard_limit_boundary() -> str:
167
+ """A segment of exactly 38 characters — one past ``int(25 * 1.5)``.
168
+
169
+ ``int(size * 1.5)`` truncates. A port that rounds instead computes 38 and
170
+ packs this segment where Python windows it, which is a whole different
171
+ chunk list from a single character of arithmetic.
172
+ """
173
+ return "def a():\n\n\n" + "a" * 35 + "\n\n\ndef b():\n 2"
174
+
175
+
176
+ def code_c0_blank_run() -> str:
177
+ """A blank-line run whose filler is a C0 separator, not a space.
178
+
179
+ Python's ``\\s`` accepts ``\\x1c``-``\\x1f``; the Unicode ``White_Space``
180
+ property does not. A port that reached for a stock Unicode ``\\s`` would
181
+ miss these runs and pack the statements around them into one segment.
182
+ """
183
+ return "x = 1\n\ny = 2\n\x1c\nz = 3\n\x1f \x1e\nw = 4\n\nv = 5"
184
+
185
+
186
+ def code_declaration_matrix() -> str:
187
+ """All seven declaration prefixes, each the only boundary on its line.
188
+
189
+ No blank lines anywhere, so every segment boundary comes from a declaration
190
+ start. At a small window each of the seven is individually load-bearing:
191
+ drop one and the two segments around it fuse into a different chunk.
192
+ """
193
+ return "\n".join(
194
+ [
195
+ "def a(): 1",
196
+ "class b: 2",
197
+ "function c() {}",
198
+ "export d = 4",
199
+ "const e = 5",
200
+ "public f = 6",
201
+ "private g = 7",
202
+ "def h(): 8",
203
+ "tail = 9",
204
+ ]
205
+ )
206
+
207
+
208
+ def prose_terminator_matrix() -> str:
209
+ """One sentence per sentence-final mark, each alone in its window."""
210
+ body = "가나다라마바사아자차카타파하"
211
+ return "".join(f"{body}{mark} " for mark in (".", "!", "?", "。", "!", "?", "…"))
212
+
213
+
214
+ def prose_closer_matrix() -> str:
215
+ """One sentence per closing quote/bracket, each alone in its window.
216
+
217
+ Drop any single closer from the port's set and that sentence's boundary
218
+ becomes a hard window cut instead, which moves every chunk after it.
219
+ """
220
+ body = "가나다라마바사아자차카타파하"
221
+ closers = ('"', "'", "\u201d", "\u2019", "\u300d", "\u300f", ")", "]")
222
+ # A plain trailing sentence, so the *last* closer is load-bearing too:
223
+ # a boundary at the very end of the text never changes a chunk list.
224
+ return "".join(f"{body}.{closer} " for closer in closers) + f"{body}. "
225
+
226
+
227
+ def prose_closers() -> str:
228
+ """Every closing quote and bracket the strong boundary allows."""
229
+ return (
230
+ '그는 "끝났다." 라고 했다. '
231
+ "그녀는 '정말?' 이라고 물었다. "
232
+ "안내문은 “확인했다.” 였다. "
233
+ "메모는 ‘완료’ 였다. "
234
+ "인용은 「검토함.」 이었다. "
235
+ "출처는 『보고서.』 이다. "
236
+ "각주는 (참고함.) 이다. "
237
+ "표는 [완료됨.] 이다. "
238
+ "마지막 문장이다."
239
+ )
240
+
241
+
242
+ def prose_english() -> str:
243
+ return (
244
+ "The retrieval pipeline chunks text before it is embedded. "
245
+ "Each chunk carries a start offset so a citation can point at it! "
246
+ "Does the boundary land on a sentence? It does, when one is in range. "
247
+ 'She said "the boundary matters" (twice), and nobody disagreed. '
248
+ "A final sentence closes the paragraph without any surprises."
249
+ )
250
+
251
+
252
+ def prose_korean() -> str:
253
+ return (
254
+ "회의에서 결정한 사항을 정리합니다。 근거는 문서에 남겨 두었습니다! "
255
+ "다음 주에 다시 검토할까요? 검토 결과는 여기에 덧붙입니다… "
256
+ "한국어는 문장 끝에 서술어가 오기 때문에 경계가 특히 중요합니다. "
257
+ "그래서 청크 경계를 문장 끝에 맞춥니다."
258
+ )
259
+
260
+
261
+ def prose_no_boundary() -> str:
262
+ """No punctuation, no line break — the hard window cut is the only answer."""
263
+ return "가나다라마바사아자차카타파하" * 12
264
+
265
+
266
+ def prose_lines_only() -> str:
267
+ """Bullet lines with no sentence punctuation — the weak boundary path."""
268
+ return "\n".join(f"- 항목 {index} 준비 완료" for index in range(30))
269
+
270
+
271
+ def prose_paragraphs() -> str:
272
+ return "\n\n".join(
273
+ f"{index}번 문단입니다 이 문단에는 마침표가 없습니다 그래서 문단 경계만 남습니다"
274
+ for index in range(12)
275
+ )
276
+
277
+
278
+ #: Emoji with zero-width joiners, a regional-indicator flag and a combining
279
+ #: mark: four graphemes, eleven code points, thirty-eight UTF-8 bytes. Slicing
280
+ #: this by bytes is not "slightly different", it is a panic.
281
+ GRAPHEME_SOUP = "a👨‍👩‍👧‍👦b🇰🇷cée"
282
+
283
+
284
+ def unicode_boundary_text() -> str:
285
+ return (GRAPHEME_SOUP + "한글") * 8
286
+
287
+
288
+ def ascii_of_length(length: int) -> str:
289
+ """``length`` printable ASCII chars — one char is one byte, on purpose."""
290
+ alphabet = "abcdefghijklmnopqrstuvwxyz0123456789"
291
+ return "".join(alphabet[index % len(alphabet)] for index in range(length))
292
+
293
+
294
+ def korean_of_length(length: int) -> str:
295
+ """``length`` Hangul syllables — one char is three bytes, on purpose."""
296
+ syllables = "가나다라마바사아자차카타파하"
297
+ return "".join(syllables[index % len(syllables)] for index in range(length))
298
+
299
+
300
+ # ── the case set ─────────────────────────────────────────────────────────────
301
+ # ``strategy`` None means "route it from ``filename`` via chunk_strategy_for",
302
+ # which is how every real call site picks one.
303
+ CASES: List[Dict[str, Any]] = [
304
+ {"key": "empty", "text": "", "filename": "empty.txt"},
305
+ {"key": "whitespace_only", "text": " \n\t\r\n ", "filename": "blank.txt"},
306
+ {"key": "plain_short", "text": "짧은 메모 한 줄.", "filename": "note"},
307
+ {"key": "plain_strip_not_collapse", "text": " \n 두 칸 사이 간격은 유지된다. \n ", "filename": "note"},
308
+ {"key": "plain_exact_minus_one", "text": ascii_of_length(63), "filename": "n", "size": 64, "overlap": 8},
309
+ {"key": "plain_exact", "text": ascii_of_length(64), "filename": "n", "size": 64, "overlap": 8},
310
+ {"key": "plain_exact_plus_one", "text": ascii_of_length(65), "filename": "n", "size": 64, "overlap": 8},
311
+ {"key": "plain_default_long", "text": mixed_paragraphs(30), "filename": "log"},
312
+ {"key": "plain_korean_small_window", "text": korean_of_length(37), "filename": "n", "size": 10, "overlap": 3},
313
+ {"key": "plain_grapheme_soup", "text": unicode_boundary_text(), "filename": "n", "size": 7, "overlap": 2},
314
+ {"key": "plain_overlap_clamped", "text": ascii_of_length(30), "filename": "n", "size": 5, "overlap": 100},
315
+ {"key": "plain_overlap_zero", "text": ascii_of_length(30), "filename": "n", "size": 7, "overlap": 0},
316
+ {"key": "plain_overlap_negative", "text": ascii_of_length(30), "filename": "n", "size": 7, "overlap": -4},
317
+ {"key": "plain_size_one", "text": korean_of_length(6), "filename": "n", "size": 1, "overlap": 160},
318
+ {"key": "plain_size_zero_coerced", "text": ascii_of_length(9), "filename": "n", "size": 0, "overlap": 3},
319
+ {"key": "unknown_strategy_falls_back", "text": mixed_paragraphs(4), "filename": "x.md", "strategy": "sideways"},
320
+ {"key": "markdown_full", "text": markdown_document(), "filename": "guide.md"},
321
+ {"key": "markdown_full_small_window", "text": markdown_document(), "filename": "guide.md", "size": 120, "overlap": 30},
322
+ {"key": "markdown_all_tiny", "text": markdown_all_tiny(), "filename": "tiny.markdown"},
323
+ {"key": "markdown_threshold", "text": markdown_threshold(), "filename": "floor.md"},
324
+ {"key": "markdown_not_headings", "text": markdown_seven_hashes(), "filename": "edge.md"},
325
+ {"key": "markdown_no_heading", "text": mixed_paragraphs(6), "filename": "plainish.md"},
326
+ {"key": "markdown_grapheme_soup", "text": "# 제목\n" + unicode_boundary_text(), "filename": "u.md", "size": 9, "overlap": 3},
327
+ {"key": "code_python", "text": python_source(), "filename": "module.py"},
328
+ {"key": "code_python_packed", "text": python_source(), "filename": "module.py", "size": 60, "overlap": 12},
329
+ {"key": "code_javascript", "text": javascript_source(), "filename": "app.tsx", "size": 80, "overlap": 16},
330
+ {"key": "code_monster_segment", "text": code_monster_segment(), "filename": "big.py", "size": 100, "overlap": 20},
331
+ {"key": "code_korean_small", "text": python_source(), "filename": "module.py", "size": 24, "overlap": 5},
332
+ {"key": "code_hard_limit_boundary", "text": code_hard_limit_boundary(), "filename": "edge.py", "size": 25, "overlap": 5},
333
+ {"key": "code_c0_blank_run", "text": code_c0_blank_run(), "filename": "c0.py", "size": 12, "overlap": 3},
334
+ {"key": "prose_closers", "text": prose_closers(), "filename": "quotes.txt", "size": 45, "overlap": 10},
335
+ {"key": "code_declaration_matrix", "text": code_declaration_matrix(), "filename": "matrix.py", "size": 12, "overlap": 3},
336
+ {"key": "prose_terminator_matrix", "text": prose_terminator_matrix(), "filename": "terms.txt", "size": 20, "overlap": 2},
337
+ {"key": "prose_closer_matrix", "text": prose_closer_matrix(), "filename": "closers.txt", "size": 20, "overlap": 2},
338
+ {"key": "prose_english", "text": prose_english(), "filename": "essay.txt", "size": 90, "overlap": 20},
339
+ {"key": "prose_korean", "text": prose_korean(), "filename": "essay.txt", "size": 60, "overlap": 15},
340
+ {"key": "prose_no_boundary", "text": prose_no_boundary(), "filename": "essay.txt", "size": 40, "overlap": 9},
341
+ {"key": "prose_lines_only", "text": prose_lines_only(), "filename": "list.txt", "size": 70, "overlap": 14},
342
+ {"key": "prose_paragraphs", "text": prose_paragraphs(), "filename": "para.txt", "size": 110, "overlap": 25},
343
+ {"key": "prose_default_window", "text": mixed_paragraphs(40), "filename": "report.pdf"},
344
+ {"key": "prose_grapheme_soup", "text": unicode_boundary_text(), "filename": "u.html", "size": 11, "overlap": 4},
345
+ # overlap == size - 1: whenever a sentence boundary lands closer to the
346
+ # start than the overlap, ``end - overlap`` goes backwards and the
347
+ # ``max(start + 1, …)`` guard is the only thing that ends the loop.
348
+ {"key": "prose_tight_overlap", "text": prose_english()[:110], "filename": "essay.txt", "size": 40, "overlap": 39},
349
+ ]
350
+
351
+ #: Filename/MIME pairs pinning every branch of ``chunk_strategy_for``.
352
+ STRATEGY_CASES: List[Dict[str, str]] = [
353
+ {"filename": "guide.md", "content_type": ""},
354
+ {"filename": "GUIDE.MARKDOWN", "content_type": ""},
355
+ {"filename": "module.py", "content_type": ""},
356
+ {"filename": "app.TSX", "content_type": ""},
357
+ {"filename": "data.json", "content_type": ""},
358
+ {"filename": "conf.toml", "content_type": ""},
359
+ {"filename": "notes.txt", "content_type": ""},
360
+ {"filename": "report.pdf", "content_type": ""},
361
+ {"filename": "page.HTM", "content_type": ""},
362
+ {"filename": "book.epub", "content_type": ""},
363
+ {"filename": "archive.tar.gz", "content_type": ""},
364
+ {"filename": "noextension", "content_type": ""},
365
+ {"filename": ".hidden", "content_type": ""},
366
+ {"filename": "trailing.", "content_type": ""},
367
+ {"filename": "", "content_type": ""},
368
+ {"filename": " ", "content_type": ""},
369
+ {"filename": "https://example.com/docs/guide.md?v=2#top", "content_type": ""},
370
+ {"filename": "https://example.com/docs/guide?v=2#top", "content_type": ""},
371
+ {"filename": "C:\\projects\\lattice\\module.py", "content_type": ""},
372
+ {"filename": "/var/data/notes/", "content_type": ""},
373
+ {"filename": "/var/data/notes//", "content_type": ""},
374
+ {"filename": "noextension", "content_type": "text/markdown"},
375
+ {"filename": "noextension", "content_type": "TEXT/HTML; charset=utf-8"},
376
+ {"filename": "noextension", "content_type": "text/plain"},
377
+ {"filename": "noextension", "content_type": "application/octet-stream"},
378
+ {"filename": "noextension", "content_type": " application/x-markdown "},
379
+ {"filename": "report.pdf", "content_type": "text/markdown"},
380
+ {"filename": "한글 문서.md", "content_type": ""},
381
+ {"filename": "한글 문서", "content_type": "text/plain"},
382
+ ]
383
+
384
+ #: ``metadata["structure"]`` shapes for ``pdf_page_offsets``, well formed and not.
385
+ PDF_STRUCTURES: List[Dict[str, Any]] = [
386
+ {"key": "three_pages", "structure": {"pages": [{"chars": 100}, {"chars": 250}, {"chars": 40}]}},
387
+ {"key": "single_page", "structure": {"pages": [{"chars": 1200}]}},
388
+ {"key": "zero_length_page", "structure": {"pages": [{"chars": 0}, {"chars": 10}, {"chars": 0}]}},
389
+ {"key": "float_chars", "structure": {"pages": [{"chars": 10.9}, {"chars": 5.0}]}},
390
+ {"key": "no_pages_key", "structure": {"meta": 1}},
391
+ {"key": "pages_not_list", "structure": {"pages": {"chars": 5}}},
392
+ {"key": "pages_empty", "structure": {"pages": []}},
393
+ {"key": "page_not_dict", "structure": {"pages": [{"chars": 5}, 7]}},
394
+ {"key": "chars_missing", "structure": {"pages": [{"chars": 5}, {}]}},
395
+ {"key": "chars_negative", "structure": {"pages": [{"chars": 5}, {"chars": -1}]}},
396
+ {"key": "chars_bool", "structure": {"pages": [{"chars": True}]}},
397
+ {"key": "chars_string", "structure": {"pages": [{"chars": "5"}]}},
398
+ {"key": "structure_not_dict", "structure": [1, 2, 3]},
399
+ {"key": "structure_null", "structure": None},
400
+ ]
401
+
402
+ #: Offsets probed against the ``three_pages`` / ``zero_length_page`` offsets.
403
+ PAGE_PROBES: List[int] = [-5, -1, 0, 1, 99, 100, 101, 102, 351, 352, 353, 10_000]
404
+
405
+ #: Chunk-metadata blobs for ``citation_locator``.
406
+ LOCATOR_CASES: List[Dict[str, Any]] = [
407
+ {},
408
+ {"heading_path": "안내서 > 온보딩"},
409
+ {"heading_path": " spaced "},
410
+ {"heading_path": ""},
411
+ {"page": 3},
412
+ {"page": 3, "page_end": 5},
413
+ {"page": 3, "page_end": 3},
414
+ {"page": 3, "page_end": 2},
415
+ {"page": 0},
416
+ {"page": -1},
417
+ {"page": "4"},
418
+ {"page": "nope"},
419
+ {"page": None},
420
+ {"heading_path": "Retrieval > Fusion", "page": 2, "page_end": 4},
421
+ ]
422
+
423
+ #: ``(source_type, source_uri, text, workspace_id)`` for the text/web hash rule.
424
+ TEXT_HASH_CASES: List[Dict[str, Optional[str]]] = [
425
+ {"source_type": "note", "source_uri": None, "text": "회의 결정 사항", "workspace_id": None},
426
+ {"source_type": "note", "source_uri": "", "text": "회의 결정 사항", "workspace_id": "ws-alpha"},
427
+ {"source_type": "web_url", "source_uri": "https://example.com/a", "text": "hello", "workspace_id": "ws-beta"},
428
+ {"source_type": "text", "source_uri": "file:///tmp/x.txt", "text": "", "workspace_id": None},
429
+ {"source_type": "markdown", "source_uri": "s3://b/k", "text": GRAPHEME_SOUP, "workspace_id": "ws-∅"},
430
+ ]
431
+
432
+ #: Byte payloads for the file content-hash rule (files hash their **bytes**).
433
+ FILE_HASH_CASES: List[bytes] = [
434
+ b"",
435
+ b"hello world\n",
436
+ "회의록\n".encode(),
437
+ bytes(range(256)),
438
+ ]
439
+
440
+ #: Texts whose ``_clean_text`` → sha256 pair is the vector-index ``text_hash``.
441
+ VECTOR_TEXT_CASES: List[str] = [
442
+ "",
443
+ " ",
444
+ "a",
445
+ " 회의 결정\t사항 ",
446
+ "line one\nline two\r\nline three",
447
+ "회의\x1c록",
448
+ GRAPHEME_SOUP,
449
+ ]