ltcai 11.7.0 → 11.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (136) hide show
  1. package/README.md +73 -70
  2. package/docs/BENCHMARKS.md +9 -2
  3. package/docs/CHANGELOG.md +157 -0
  4. package/docs/CI_AND_RELEASE_GATES.md +126 -41
  5. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  6. package/docs/DEVELOPMENT.md +34 -14
  7. package/docs/LEGACY_COMPATIBILITY.md +10 -6
  8. package/docs/ONBOARDING.md +5 -3
  9. package/docs/OPERATIONS.md +5 -1
  10. package/docs/PERMISSION_MODE.md +14 -9
  11. package/docs/TRUST_MODEL.md +13 -6
  12. package/docs/USABILITY_AUDIT.md +5 -0
  13. package/docs/WHY_LATTICE.md +6 -4
  14. package/docs/kg-schema.md +7 -3
  15. package/docs/mcp-tools.md +73 -79
  16. package/docs/security-model.md +6 -3
  17. package/lattice_brain/__init__.py +1 -1
  18. package/lattice_brain/graph/_kg_common/__init__.py +1 -54
  19. package/lattice_brain/graph/_kg_common/text.py +14 -450
  20. package/lattice_brain/ingestion/__init__.py +6 -3
  21. package/lattice_brain/multimodal/__init__.py +9 -3
  22. package/latticeai/__init__.py +1 -1
  23. package/latticeai/api/agent_worker_seam.py +12 -1
  24. package/latticeai/api/models.py +18 -110
  25. package/latticeai/api/search.py +7 -30
  26. package/latticeai/api/worker_compute.py +53 -107
  27. package/latticeai/api/worker_seams.py +17 -2
  28. package/latticeai/core/http_origin.py +3 -3
  29. package/latticeai/core/messages.py +0 -5
  30. package/latticeai/core/policy.py +1 -6
  31. package/latticeai/core/quiet.py +1 -20
  32. package/latticeai/core/security.py +29 -83
  33. package/latticeai/core/sessions.py +95 -4
  34. package/latticeai/core/users.py +0 -38
  35. package/latticeai/models/router/catalog.py +2 -2
  36. package/latticeai/models/router/generation.py +81 -14
  37. package/latticeai/models/router/loading.py +41 -5
  38. package/latticeai/runtime/access_runtime.py +7 -4
  39. package/latticeai/runtime/build_phases/features.py +8 -31
  40. package/latticeai/runtime/build_phases/foundation.py +7 -16
  41. package/latticeai/runtime/build_phases/web.py +3 -3
  42. package/latticeai/runtime/build_phases/worker_profile.py +21 -28
  43. package/latticeai/runtime/runtime_context.py +0 -2
  44. package/latticeai/services/architecture_readiness.py +18 -19
  45. package/latticeai/services/process_audit.py +1 -22
  46. package/latticeai/services/product_readiness.py +33 -11
  47. package/latticeai/services/voice_capture.py +8 -28
  48. package/latticeai/tools/__init__.py +6 -46
  49. package/latticeai/tools/commands.py +9 -15
  50. package/latticeai/tools/knowledge.py +0 -6
  51. package/package.json +3 -5
  52. package/requirements.txt +0 -1
  53. package/scripts/check_current_release_docs.mjs +1 -1
  54. package/scripts/check_openapi_drift.mjs +3 -2
  55. package/scripts/check_server_i18n.mjs +5 -4
  56. package/scripts/compose_openapi.py +2 -1
  57. package/scripts/export_openapi.py +5 -4
  58. package/scripts/gen_worker_allowlist_fixture.py +2 -2
  59. package/scripts/openapi_route_families.json +14 -73
  60. package/scripts/release_screen_claims.json +131 -27
  61. package/src-tauri/Cargo.lock +11 -10
  62. package/src-tauri/Cargo.toml +1 -1
  63. package/src-tauri/tauri.conf.json +1 -1
  64. package/static/app/asset-manifest.json +41 -41
  65. package/static/app/assets/{Act-BPcVAbOL.js → Act-B4WT81kh.js} +1 -1
  66. package/static/app/assets/AdminConsole--Wf71m-o.js +1 -0
  67. package/static/app/assets/{Brain-CT92Kos0.js → Brain-CtYaa26c.js} +2 -2
  68. package/static/app/assets/BrainHome-Be2VPJEc.js +2 -0
  69. package/static/app/assets/BrainSignals-C0__xgpG.js +1 -0
  70. package/static/app/assets/Capture-DdNi5Peb.js +1 -0
  71. package/static/app/assets/Chronicle-B3hNveeI.js +1 -0
  72. package/static/app/assets/CommandPalette-CIsnSsFL.js +1 -0
  73. package/static/app/assets/Library-Bl5XClFV.js +1 -0
  74. package/static/app/assets/LivingBrain-DjkTB_gh.js +1 -0
  75. package/static/app/assets/{ProductFlow-DXBC6brE.js → ProductFlow-DCUNRNHs.js} +1 -1
  76. package/static/app/assets/{ReviewCard-HXRle3qq.js → ReviewCard-CD3yWvUB.js} +2 -2
  77. package/static/app/assets/System-BJ6jQ_SL.js +1 -0
  78. package/static/app/assets/arrow-left-6_28Z0qH.js +1 -0
  79. package/static/app/assets/{bot-Cn8bWRuq.js → bot-Bc3Q27YR.js} +1 -1
  80. package/static/app/assets/brain-B9BDrMTe.js +1 -0
  81. package/static/app/assets/{button-Ct9f2_oT.js → button-D6JcpYcf.js} +1 -1
  82. package/static/app/assets/circle-check-BFu9lD-3.js +1 -0
  83. package/static/app/assets/{circle-pause-CmzC_apg.js → circle-pause-BGMiV8UU.js} +1 -1
  84. package/static/app/assets/{circle-play-D8mW2aQ7.js → circle-play-DoanLHnd.js} +1 -1
  85. package/static/app/assets/{cpu-DZcdd0PZ.js → cpu-DwzNf82m.js} +1 -1
  86. package/static/app/assets/{download-bv1KEPGQ.js → download-Ddw49yCV.js} +1 -1
  87. package/static/app/assets/{folder-open-d-Pip5gr.js → folder-open-Brd6Kvto.js} +1 -1
  88. package/static/app/assets/{hard-drive-D20iavUb.js → hard-drive-Bu-DTJdB.js} +1 -1
  89. package/static/app/assets/index-CGdg_aq9.css +2 -0
  90. package/static/app/assets/{index-Do83hDzJ.js → index-CWKRRsLW.js} +4 -4
  91. package/static/app/assets/input-CEqsxtil.js +1 -0
  92. package/static/app/assets/{link-2-BPJOFlAy.js → link-2-DZ4OA5tJ.js} +1 -1
  93. package/static/app/assets/{permissionCopy-ChdJd493.js → permissionCopy-D9TR0F8b.js} +1 -1
  94. package/static/app/assets/primitives-DORg7Z_7.js +1 -0
  95. package/static/app/assets/search-0NQ21wXe.js +1 -0
  96. package/static/app/assets/{share-2-YNX_NtMU.js → share-2-BC5FirFv.js} +1 -1
  97. package/static/app/assets/{shield-alert-DuQ3zrVL.js → shield-alert-DUbR2W2s.js} +1 -1
  98. package/static/app/assets/textarea-jtQcRSXo.js +1 -0
  99. package/static/app/assets/{useFocusTrap-ZVI98jaW.js → useFocusTrap-HRemcWId.js} +1 -1
  100. package/static/app/assets/{useMutation-CVC4qv_D.js → useMutation-DqlFE-Bw.js} +1 -1
  101. package/static/app/assets/{useQuery-C7BeG4HU.js → useQuery-BizqBNGw.js} +1 -1
  102. package/static/app/assets/utils-WgW4V69R.js +4 -0
  103. package/static/app/assets/workspace-DSek3jCY.js +1 -0
  104. package/static/app/index.html +4 -4
  105. package/static/sw.js +1 -1
  106. package/lattice_brain/ingestion/pipeline.py +0 -108
  107. package/latticeai/api/local_files.py +0 -44
  108. package/latticeai/api/tools.py +0 -126
  109. package/latticeai/api/voice_capture.py +0 -32
  110. package/latticeai/core/agent_permission.py +0 -85
  111. package/scripts/agent_eval.py +0 -34
  112. package/scripts/brain_quality_eval.py +0 -37
  113. package/scripts/check_legacy_debt.mjs +0 -91
  114. package/scripts/check_python.py +0 -100
  115. package/scripts/chunking_parity_corpus.py +0 -449
  116. package/scripts/generate_agent_parity_fixtures.py +0 -771
  117. package/scripts/generate_chunking_parity_fixtures.py +0 -259
  118. package/static/app/assets/AdminConsole-Bw1ATQL0.js +0 -1
  119. package/static/app/assets/BrainHome-CFBkt1K_.js +0 -2
  120. package/static/app/assets/BrainSignals-ReLWF2H8.js +0 -1
  121. package/static/app/assets/Capture-BsTokYkk.js +0 -1
  122. package/static/app/assets/Chronicle-B6f0T9id.js +0 -1
  123. package/static/app/assets/CommandPalette-CuvjTv1u.js +0 -1
  124. package/static/app/assets/Library-BGJbG9Hd.js +0 -1
  125. package/static/app/assets/LivingBrain-DGYK_Jsa.js +0 -1
  126. package/static/app/assets/System-CMHSO9qM.js +0 -1
  127. package/static/app/assets/arrow-left-BfmkskWx.js +0 -1
  128. package/static/app/assets/brain-CQJberbE.js +0 -1
  129. package/static/app/assets/circle-check-DruOxB-4.js +0 -1
  130. package/static/app/assets/index-D9x-kSNy.css +0 -2
  131. package/static/app/assets/input-BLXVNmj1.js +0 -1
  132. package/static/app/assets/primitives-Cv5tbZBY.js +0 -1
  133. package/static/app/assets/search-CT9aho2j.js +0 -1
  134. package/static/app/assets/textarea-DqwLnli4.js +0 -1
  135. package/static/app/assets/utils-CiFtIdZq.js +0 -4
  136. package/static/app/assets/workspace-DQz9vIId.js +0 -1
@@ -1,17 +1,27 @@
1
- """Text cleaning, chunking, and citation-locator maths.
1
+ """Text cleaning and the legacy fixed-width chunk walk.
2
2
 
3
3
  Moved verbatim out of the ``_kg_common`` grab-bag (v11.3.0 decomposition).
4
4
  Nothing here reaches back into the rest of the package — the import graph is
5
5
  ``text ← relations ← extraction ← __init__`` — so this is the layer every
6
6
  other one may build on.
7
+
8
+ The **typed chunker** that used to live here — ``typed_chunks`` and its four
9
+ strategies, ``chunk_strategy_for``, ``pdf_page_offsets``, and the three
10
+ readers that only ever consumed their output (``typed_chunk_meta_fields``,
11
+ ``citation_locator``, ``page_for_offset``) — was removed in 11.8.0. Chunking
12
+ is native: ``lattice-ingest`` owns it, pinned by
13
+ ``rust/lattice-ingest/tests/chunking_parity.rs`` against the committed
14
+ ``rust/fixtures/chunking`` goldens. The worker imports only the extraction
15
+ helpers (``POST /worker/extract`` — see
16
+ ``latticeai/api/worker_compute.py::build_extract_reply``), so the Python copy
17
+ had no shipping call site left; keeping a second boundary algorithm that
18
+ nothing runs is how two chunkers quietly stop agreeing.
7
19
  """
8
20
 
9
21
  from __future__ import annotations
10
22
 
11
23
  import re
12
- from typing import Any, Dict, List, Optional, Tuple
13
-
14
- from ...quiet import quiet
24
+ from typing import List
15
25
 
16
26
 
17
27
  def _clean_text(text: str) -> str:
@@ -31,449 +41,3 @@ def _chunks(text: str, size: int = 1200, overlap: int = 160) -> List[str]:
31
41
  break
32
42
  start = max(0, end - overlap)
33
43
  return chunks
34
-
35
-
36
- # ── Typed chunking (review 2026-07-25 §5.2 S2 — Wave 2.1 + 2.4) ──────────────
37
- # ``_chunks`` above is a compatibility contract (chunk ids hash over the chunk
38
- # text) and stays byte-for-byte untouched. ``typed_chunks`` layers strategy-
39
- # aware boundaries plus per-chunk provenance (start_char / heading_path) on
40
- # top; ``strategy="plain"`` reproduces the exact ``_chunks`` boundaries so
41
- # unchanged plain content keeps identical chunk ids.
42
-
43
- _MARKDOWN_CHUNK_EXTENSIONS = {".md", ".markdown"}
44
- _CODE_CHUNK_EXTENSIONS = {
45
- ".py", ".js", ".jsx", ".ts", ".tsx", ".go", ".rs", ".java", ".rb",
46
- ".c", ".h", ".cpp", ".css", ".sh", ".sql", ".vue", ".svelte",
47
- ".json", ".yaml", ".yml", ".toml",
48
- }
49
- _PROSE_CHUNK_EXTENSIONS = {
50
- ".txt", ".pdf", ".docx", ".doc", ".rtf", ".odt", ".epub", ".html", ".htm",
51
- }
52
- _CHUNK_STRATEGIES = {"plain", "markdown", "code", "prose"}
53
- # Markdown sections smaller than this merge forward into the next section so
54
- # heading-dense documents don't shatter into confetti chunks.
55
- _MARKDOWN_MIN_SECTION_CHARS = 200
56
- _MARKDOWN_HEADING_RE = re.compile(r"^(#{1,6}) (.*)$", re.MULTILINE)
57
- _CODE_BOUNDARY_LINE_RE = re.compile(
58
- r"^(?:def |class |function |export |const |public |private )", re.MULTILINE
59
- )
60
- _CODE_BLANK_RUN_RE = re.compile(r"\n\s*\n")
61
-
62
-
63
- def chunk_strategy_for(filename: Any, *, content_type: str = "") -> str:
64
- """Route a filename / path / URI (plus optional MIME hint) to a strategy.
65
-
66
- Returns ``"markdown"`` for .md/.markdown, ``"code"`` for known source-code
67
- extensions, ``"prose"`` for document formats whose text is running prose
68
- (.txt/.pdf/.docx/.html/…), ``"plain"`` otherwise. Case-insensitive,
69
- tolerant of URLs (query/fragment stripped) and ``Path`` objects; never
70
- raises — any malformed input falls back to ``"plain"``.
71
-
72
- Unknown/extension-less input stays ``"plain"`` on purpose: the plain
73
- strategy is the byte-compatible legacy walk, and guessing prose for
74
- something that might be a data dump would move chunk boundaries for no
75
- retrieval gain.
76
- """
77
- try:
78
- name = str(filename or "").strip().lower()
79
- for sep in ("?", "#"):
80
- name = name.split(sep, 1)[0]
81
- name = name.replace("\\", "/").rstrip("/").rsplit("/", 1)[-1]
82
- dot = name.rfind(".")
83
- ext = name[dot:] if dot > 0 else ""
84
- if ext in _MARKDOWN_CHUNK_EXTENSIONS:
85
- return "markdown"
86
- if ext in _CODE_CHUNK_EXTENSIONS:
87
- return "code"
88
- if ext in _PROSE_CHUNK_EXTENSIONS:
89
- return "prose"
90
- mime = str(content_type or "").strip().lower()
91
- if "markdown" in mime:
92
- return "markdown"
93
- if mime.startswith("text/html") or mime.startswith("text/plain"):
94
- return "prose"
95
- except Exception:
96
- quiet()
97
- return "plain"
98
-
99
-
100
- def _plain_windows(
101
- cleaned: str,
102
- size: int,
103
- overlap: int,
104
- *,
105
- base_offset: int = 0,
106
- strategy: str = "plain",
107
- heading_path: Optional[str] = None,
108
- ) -> List[Dict[str, Any]]:
109
- """The exact ``_chunks`` walk with ``start_char`` tracked.
110
-
111
- Boundaries and chunk texts are byte-identical to ``_chunks`` over the same
112
- string — this is the plain-strategy compatibility guarantee.
113
- """
114
- out: List[Dict[str, Any]] = []
115
- start = 0
116
- total = len(cleaned)
117
- while start < total:
118
- end = min(total, start + size)
119
- out.append(
120
- {
121
- "text": cleaned[start:end],
122
- "meta": {
123
- "strategy": strategy,
124
- "start_char": base_offset + start,
125
- "heading_path": heading_path,
126
- },
127
- }
128
- )
129
- if end >= total:
130
- break
131
- start = max(0, end - overlap)
132
- return out
133
-
134
-
135
- def _markdown_section_spans(cleaned: str) -> List[Tuple[int, int, Optional[str]]]:
136
- """``(start, end, heading_path)`` spans split at ``^#{1,6} `` heading lines.
137
-
138
- ``heading_path`` is the " > "-joined path of the enclosing headings
139
- including the section's own heading (e.g. ``"Guide > Setup"``); the
140
- preamble before the first heading carries ``None``. Spans are contiguous
141
- raw slices of ``cleaned`` so every chunk text round-trips via start_char.
142
- """
143
- spans: List[Tuple[int, int, Optional[str]]] = []
144
- stack: List[Tuple[int, str]] = []
145
- prev_start = 0
146
- prev_path: Optional[str] = None
147
- for match in _MARKDOWN_HEADING_RE.finditer(cleaned):
148
- offset = match.start()
149
- if offset > prev_start:
150
- spans.append((prev_start, offset, prev_path))
151
- level = len(match.group(1))
152
- while stack and stack[-1][0] >= level:
153
- stack.pop()
154
- stack.append((level, match.group(2).strip()))
155
- prev_start = offset
156
- prev_path = " > ".join(title for _, title in stack) or None
157
- if len(cleaned) > prev_start:
158
- spans.append((prev_start, len(cleaned), prev_path))
159
- return spans
160
-
161
-
162
- def _merge_small_sections(
163
- spans: List[Tuple[int, int, Optional[str]]], min_chars: int
164
- ) -> List[Tuple[int, int, Optional[str]]]:
165
- """Merge sections under ``min_chars`` forward into the next section.
166
-
167
- A merged section keeps the heading_path of its first constituent (the
168
- path in effect at the chunk start). A trailing undersized section merges
169
- backward into the previous emitted section when one exists.
170
- """
171
- merged: List[Tuple[int, int, Optional[str]]] = []
172
- pending: Optional[Tuple[int, int, Optional[str]]] = None
173
- for start, end, path in spans:
174
- if pending is None:
175
- pending = (start, end, path)
176
- else:
177
- pending = (pending[0], end, pending[2])
178
- if pending[1] - pending[0] >= min_chars:
179
- merged.append(pending)
180
- pending = None
181
- if pending is not None:
182
- if merged and pending[1] - pending[0] < min_chars:
183
- last = merged.pop()
184
- merged.append((last[0], pending[1], last[2]))
185
- else:
186
- merged.append(pending)
187
- return merged
188
-
189
-
190
- def _markdown_chunks(cleaned: str, size: int, overlap: int) -> List[Dict[str, Any]]:
191
- sections = _merge_small_sections(
192
- _markdown_section_spans(cleaned), _MARKDOWN_MIN_SECTION_CHARS
193
- )
194
- out: List[Dict[str, Any]] = []
195
- for start, end, path in sections:
196
- body = cleaned[start:end]
197
- if len(body) <= size:
198
- out.append(
199
- {
200
- "text": body,
201
- "meta": {
202
- "strategy": "markdown",
203
- "start_char": start,
204
- "heading_path": path,
205
- },
206
- }
207
- )
208
- else:
209
- out.extend(
210
- _plain_windows(
211
- body,
212
- size,
213
- overlap,
214
- base_offset=start,
215
- strategy="markdown",
216
- heading_path=path,
217
- )
218
- )
219
- return out
220
-
221
-
222
- def _code_segment_spans(cleaned: str) -> List[Tuple[int, int]]:
223
- """Contiguous top-level segments split at blank-line runs and decl lines."""
224
- boundaries = {0, len(cleaned)}
225
- for match in _CODE_BLANK_RUN_RE.finditer(cleaned):
226
- boundaries.add(match.end())
227
- for match in _CODE_BOUNDARY_LINE_RE.finditer(cleaned):
228
- boundaries.add(match.start())
229
- ordered = sorted(boundaries)
230
- return [
231
- (ordered[i], ordered[i + 1])
232
- for i in range(len(ordered) - 1)
233
- if ordered[i + 1] > ordered[i]
234
- ]
235
-
236
-
237
- def _code_chunks(cleaned: str, size: int, overlap: int) -> List[Dict[str, Any]]:
238
- hard_limit = int(size * 1.5)
239
- out: List[Dict[str, Any]] = []
240
- pack: Optional[Tuple[int, int]] = None
241
-
242
- def _emit(span: Tuple[int, int]) -> None:
243
- out.append(
244
- {
245
- "text": cleaned[span[0] : span[1]],
246
- "meta": {
247
- "strategy": "code",
248
- "start_char": span[0],
249
- "heading_path": None,
250
- },
251
- }
252
- )
253
-
254
- for start, end in _code_segment_spans(cleaned):
255
- if end - start > hard_limit:
256
- # Monster segment: flush the pack, then window it like plain text.
257
- if pack is not None:
258
- _emit(pack)
259
- pack = None
260
- out.extend(
261
- _plain_windows(
262
- cleaned[start:end],
263
- size,
264
- overlap,
265
- base_offset=start,
266
- strategy="code",
267
- )
268
- )
269
- continue
270
- if pack is None:
271
- pack = (start, end)
272
- elif end - pack[0] <= size:
273
- pack = (pack[0], end)
274
- else:
275
- _emit(pack)
276
- pack = (start, end)
277
- if pack is not None:
278
- _emit(pack)
279
- return out
280
-
281
-
282
- # ── Prose chunking (review 2026-07-27 P1 #4) ────────────────────────────────
283
- # The plain walk cuts every ``size`` characters, which lands mid-sentence and
284
- # — for Korean, where the verb carrying the meaning sits at the end — routinely
285
- # splits a claim from its predicate. Retrieval then matches half a statement
286
- # and the citation shows a fragment. The prose strategy keeps the same window
287
- # budget but ends each chunk at the last sentence/paragraph boundary inside it.
288
-
289
- # Strong: sentence-final punctuation (ASCII + CJK) with optional closing
290
- # quotes/brackets, followed by whitespace; or a blank-line paragraph break.
291
- _PROSE_STRONG_BOUNDARY_RE = re.compile(
292
- r"(?:[.!?。!?…]+[\"'”’」』\)\]]*\s+|\n[ \t]*\n)"
293
- )
294
- # Weak: a single line break. Korean notes and bullet lists often carry no
295
- # sentence punctuation at all; a line end is still a real boundary there.
296
- _PROSE_WEAK_BOUNDARY_RE = re.compile(r"\n")
297
- # Never emit a chunk shorter than this fraction of ``size`` just to hit a
298
- # boundary — tiny chunks hurt recall more than a mid-sentence cut.
299
- _PROSE_MIN_SPAN_RATIO = 0.5
300
-
301
-
302
- def _last_boundary(cleaned: str, lo: int, hi: int) -> Optional[int]:
303
- """End offset of the last sentence/paragraph boundary in ``cleaned[lo:hi]``.
304
-
305
- Strong boundaries win; a single line break is the fallback. Returns None
306
- when the span holds neither, so the caller keeps the hard window cut.
307
- """
308
- window = cleaned[lo:hi]
309
- for pattern in (_PROSE_STRONG_BOUNDARY_RE, _PROSE_WEAK_BOUNDARY_RE):
310
- last = None
311
- for match in pattern.finditer(window):
312
- last = match.end()
313
- if last:
314
- return lo + last
315
- return None
316
-
317
-
318
- def _prose_chunks(cleaned: str, size: int, overlap: int) -> List[Dict[str, Any]]:
319
- out: List[Dict[str, Any]] = []
320
- total = len(cleaned)
321
- min_span = max(1, int(size * _PROSE_MIN_SPAN_RATIO))
322
- start = 0
323
- while start < total:
324
- hard_end = min(total, start + size)
325
- end = hard_end
326
- if hard_end < total:
327
- boundary = _last_boundary(cleaned, start + min_span, hard_end)
328
- if boundary is not None and boundary > start:
329
- end = boundary
330
- out.append(
331
- {
332
- "text": cleaned[start:end],
333
- "meta": {
334
- "strategy": "prose",
335
- "start_char": start,
336
- "heading_path": None,
337
- },
338
- }
339
- )
340
- if end >= total:
341
- break
342
- # Overlap carries the tail of the previous chunk into the next one so
343
- # a claim split across a boundary is still retrievable from both.
344
- start = max(start + 1, end - overlap)
345
- return out
346
-
347
-
348
- def typed_chunks(
349
- text: str,
350
- *,
351
- strategy: str = "plain",
352
- size: int = 1200,
353
- overlap: int = 160,
354
- ) -> List[Dict[str, Any]]:
355
- """Strategy-aware chunking with per-chunk provenance metadata.
356
-
357
- Returns ``[{"text": str, "meta": {"strategy", "start_char", "heading_path"}}]``
358
- where ``start_char`` is the offset in ``str(text or "").strip()`` (every
359
- chunk text is an exact substring at that offset).
360
-
361
- Contract: ``[c["text"] for c in typed_chunks(t)] == _chunks(t)`` for the
362
- default plain strategy — unknown strategies also fall back to plain.
363
- """
364
- cleaned = str(text or "").strip()
365
- if not cleaned:
366
- return []
367
- try:
368
- size = max(1, int(size))
369
- except Exception:
370
- size = 1200
371
- try:
372
- overlap = min(max(0, int(overlap)), size - 1)
373
- except Exception:
374
- overlap = min(160, size - 1)
375
- label = strategy if strategy in _CHUNK_STRATEGIES else "plain"
376
- if label == "markdown":
377
- return _markdown_chunks(cleaned, size, overlap)
378
- if label == "code":
379
- return _code_chunks(cleaned, size, overlap)
380
- if label == "prose":
381
- return _prose_chunks(cleaned, size, overlap)
382
- return _plain_windows(cleaned, size, overlap)
383
-
384
-
385
- def typed_chunk_meta_fields(piece: Dict[str, Any]) -> Dict[str, Any]:
386
- """Additive chunk-metadata fields for one ``typed_chunks`` piece.
387
-
388
- Ingest call sites merge this into the existing ``{"index", "source_node"}``
389
- chunk metadata; ``heading_path`` is only present when known — honest
390
- absence over empty labels.
391
- """
392
- meta = piece.get("meta") or {}
393
- fields: Dict[str, Any] = {
394
- "strategy": str(meta.get("strategy") or "plain"),
395
- "start_char": int(meta.get("start_char") or 0),
396
- }
397
- heading_path = meta.get("heading_path")
398
- if heading_path:
399
- fields["heading_path"] = str(heading_path)
400
- return fields
401
-
402
-
403
- def citation_locator(chunk_metadata: Any) -> str:
404
- """Human "where in the document" label for one chunk, or "".
405
-
406
- Built only from provenance the chunk actually carries — a section heading
407
- path and/or a page number. When neither is known the answer is the empty
408
- string, so a citation never claims a location it cannot prove.
409
- """
410
- if not isinstance(chunk_metadata, dict):
411
- return ""
412
- parts: List[str] = []
413
- heading = str(chunk_metadata.get("heading_path") or "").strip()
414
- if heading:
415
- parts.append(heading)
416
- def _page(key: str) -> int:
417
- value = chunk_metadata.get(key)
418
- try:
419
- return int(value) if value is not None else 0
420
- except (TypeError, ValueError):
421
- return 0
422
-
423
- page_number = _page("page")
424
- if page_number > 0:
425
- page_end = _page("page_end")
426
- parts.append(
427
- f"p.{page_number}–{page_end}" if page_end > page_number else f"p.{page_number}"
428
- )
429
- return " · ".join(parts)
430
-
431
-
432
- def pdf_page_offsets(structure: Any) -> List[int]:
433
- """Start offset of each PDF page in the "\\n\\n"-joined page text.
434
-
435
- ``structure`` is the ``metadata["structure"]`` dict produced by
436
- ``_pdf_structure`` (``pages`` = ``[{"chars": int, ...}, ...]``); pages were
437
- joined with ``"\\n\\n"`` (see ``read_document``), so page k starts at
438
- ``sum(chars[j] + 2 for j < k)``. Empty or malformed input returns ``[]``.
439
- """
440
- if not isinstance(structure, dict):
441
- return []
442
- pages = structure.get("pages")
443
- if not isinstance(pages, list) or not pages:
444
- return []
445
- offsets: List[int] = []
446
- cursor = 0
447
- for page in pages:
448
- if not isinstance(page, dict):
449
- return []
450
- chars = page.get("chars")
451
- if isinstance(chars, bool) or not isinstance(chars, (int, float)) or chars < 0:
452
- return []
453
- offsets.append(cursor)
454
- cursor += int(chars) + 2 # +2 for the "\n\n" page joiner
455
- return offsets
456
-
457
-
458
- def page_for_offset(page_offsets: List[int], offset: int) -> Optional[int]:
459
- """1-based page number containing ``offset`` given page start offsets.
460
-
461
- Returns ``None`` when ``page_offsets`` is empty or the offset precedes the
462
- first page start (honest absence over a wrong label).
463
- """
464
- if not page_offsets:
465
- return None
466
- try:
467
- target = int(offset)
468
- except Exception:
469
- return None
470
- page = 0
471
- for index, start in enumerate(page_offsets):
472
- try:
473
- if target >= int(start):
474
- page = index + 1
475
- else:
476
- break
477
- except Exception:
478
- return None
479
- return page if page >= 1 else None
@@ -16,8 +16,12 @@ asks this worker for:
16
16
  of a parse request rather than of a write;
17
17
  * ``hashing`` — ``content_hash_text`` and the file digest, which decide
18
18
  idempotency and must produce the same bytes on both sides;
19
- * ``quality`` — the advisory extraction score behind ``POST /worker/parse``;
20
- * ``pipeline`` — the multi-modal capability probe.
19
+ * ``quality`` — the advisory extraction score behind ``POST /worker/parse``.
20
+
21
+ ``pipeline`` was a fifth: an ``IngestionPipeline`` reduced to a single
22
+ capability probe, whose one route (``GET /api/ingestion/multimodal``) had no
23
+ caller. v11.8.0 removed the route and the class with it — the gates it read
24
+ still live in ``constants``, where anything that needs them can ask directly.
21
25
  """
22
26
 
23
27
  from __future__ import annotations
@@ -78,7 +82,6 @@ from .hashing import _file_digest as _file_digest
78
82
  from .hashing import content_hash_text as content_hash_text
79
83
  from .models import IngestionItem as IngestionItem
80
84
  from .models import IngestionResult as IngestionResult
81
- from .pipeline import IngestionPipeline as IngestionPipeline
82
85
  from .quality import _BOILERPLATE_LINE_MARKERS as _BOILERPLATE_LINE_MARKERS
83
86
  from .quality import _CAPTURE_REASON_LABELS as _CAPTURE_REASON_LABELS
84
87
  from .quality import _WEB_SOURCE_TYPES as _WEB_SOURCE_TYPES
@@ -36,9 +36,15 @@ imports nothing from ``latticeai``.
36
36
 
37
37
  v11.6.0 removed the *writing* half — ``write_image_memory``,
38
38
  ``write_video_memory``, the keyframe writer and the node-id helpers. Extraction
39
- returns facts; ``lattice-core``'s graph write engine turns them into nodes. What
40
- is left here is exactly what ``POST /worker/multimodal/describe`` and
41
- ``POST /worker/asr`` answer with.
39
+ returns facts; ``lattice-core``'s graph write engine turns them into nodes.
40
+
41
+ The audio half is what ``POST /worker/asr`` answers with. The image and video
42
+ halves currently have **no HTTP door**: ``POST /worker/multimodal/describe``
43
+ wrapped :func:`extract_image_facts` for a native image ingest that was never
44
+ built, and v11.8.0 deleted the seam rather than keep a route nothing called.
45
+ The observation functions stay — they are Brain Core's account of what a
46
+ picture or a recording contains, unit-tested directly, and the seam is a
47
+ handful of lines to restore on the day a native image ingest needs one.
42
48
 
43
49
  Split into cohesive submodules in v11.3.0 (no behaviour change): ``common``
44
50
  (taxonomy + shared helpers), ``ports`` (injected capabilities + the ffmpeg
@@ -1,3 +1,3 @@
1
1
  """Lattice AI - modular server package."""
2
2
 
3
- __version__ = "11.7.0"
3
+ __version__ = "11.9.0"
@@ -53,7 +53,7 @@ from __future__ import annotations
53
53
 
54
54
  import asyncio
55
55
  import os
56
- from typing import Any, Callable, Dict, Optional
56
+ from typing import Any, Callable, Dict, List, Optional
57
57
 
58
58
  from fastapi import APIRouter, HTTPException, Request
59
59
  from pydantic import BaseModel, Field
@@ -78,6 +78,9 @@ MAX_MAX_TOKENS = 8192
78
78
  MIN_TEMPERATURE = 0.0
79
79
  MAX_TEMPERATURE = 2.0
80
80
 
81
+ #: How many stop strings one completion may name. The kernel sends two.
82
+ MAX_STOP_STRINGS = 8
83
+
81
84
  #: Rate-limit bucket. Deliberately *not* the ``"agent"`` bucket ``/agent`` uses:
82
85
  #: that one is sized per *run* (10 burst, one refill per 10s) because one HTTP
83
86
  #: call there is a whole agent run. Here one call is a single loop step, and a
@@ -101,6 +104,13 @@ class AgentLLMRequest(BaseModel):
101
104
  context: Optional[str] = None
102
105
  max_tokens: int = 4096
103
106
  temperature: float = 0.2
107
+ #: Strings that end the reply (v11.9.0). Optional, and absent means what it
108
+ #: always meant: generate to ``max_tokens``. The kernel sends it on exactly
109
+ #: one call — the strict verification re-ask, whose reply is one short
110
+ #: closed object — because a stop string that is safe there ("\n```") would
111
+ #: truncate any reply carrying file content. Bounded so a caller cannot make
112
+ #: the sampler check a thousand needles per token.
113
+ stop: Optional[List[str]] = Field(default=None, max_length=MAX_STOP_STRINGS)
104
114
 
105
115
 
106
116
  class AgentToolRequest(BaseModel):
@@ -247,6 +257,7 @@ def create_agent_worker_seam_router(
247
257
  context=req.context,
248
258
  max_tokens=req.max_tokens,
249
259
  temperature=req.temperature,
260
+ stop=req.stop or None,
250
261
  )
251
262
  return {"text": str(text)}
252
263