ltcai 11.7.0 → 12.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/README.md +100 -76
  2. package/docs/BENCHMARKS.md +9 -2
  3. package/docs/CHANGELOG.md +249 -0
  4. package/docs/CI_AND_RELEASE_GATES.md +126 -41
  5. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  6. package/docs/DEVELOPMENT.md +271 -103
  7. package/docs/ENTERPRISE.md +1 -1
  8. package/docs/LEGACY_COMPATIBILITY.md +10 -6
  9. package/docs/MULTI_AGENT_RUNTIME.md +4 -4
  10. package/docs/ONBOARDING.md +16 -4
  11. package/docs/OPERATIONS.md +14 -1
  12. package/docs/PERMISSION_MODE.md +14 -9
  13. package/docs/REALTIME_COLLABORATION.md +1 -1
  14. package/docs/ROADMAP.md +113 -0
  15. package/docs/TRUST_MODEL.md +28 -7
  16. package/docs/USABILITY_AUDIT.md +5 -0
  17. package/docs/WHY_LATTICE.md +13 -5
  18. package/docs/WORKFLOW_DESIGNER.md +2 -2
  19. package/docs/kg-schema.md +57 -7
  20. package/docs/mcp-tools.md +93 -82
  21. package/docs/security-model.md +6 -3
  22. package/lattice_brain/__init__.py +1 -1
  23. package/lattice_brain/graph/_kg_common/__init__.py +1 -54
  24. package/lattice_brain/graph/_kg_common/extraction.py +459 -105
  25. package/lattice_brain/graph/_kg_common/normalize.py +305 -0
  26. package/lattice_brain/graph/_kg_common/patterns.py +275 -0
  27. package/lattice_brain/graph/_kg_common/relations.py +12 -3
  28. package/lattice_brain/graph/_kg_common/sections.py +107 -0
  29. package/lattice_brain/graph/_kg_common/text.py +14 -450
  30. package/lattice_brain/graph/_kg_constants.py +7 -0
  31. package/lattice_brain/ingestion/__init__.py +6 -3
  32. package/lattice_brain/multimodal/__init__.py +9 -3
  33. package/latticeai/__init__.py +1 -1
  34. package/latticeai/api/agent_worker_seam.py +44 -1
  35. package/latticeai/api/models.py +18 -110
  36. package/latticeai/api/search.py +7 -30
  37. package/latticeai/api/worker_compute.py +127 -106
  38. package/latticeai/api/worker_seams.py +17 -2
  39. package/latticeai/core/embedding_providers/__init__.py +16 -0
  40. package/latticeai/core/embedding_providers/autodetect.py +302 -0
  41. package/latticeai/core/embedding_providers/base.py +25 -0
  42. package/latticeai/core/embedding_providers/profiles.py +44 -0
  43. package/latticeai/core/embedding_providers/text.py +74 -8
  44. package/latticeai/core/http_origin.py +3 -3
  45. package/latticeai/core/messages.py +0 -5
  46. package/latticeai/core/policy.py +1 -6
  47. package/latticeai/core/quiet.py +1 -20
  48. package/latticeai/core/security.py +29 -83
  49. package/latticeai/core/sessions.py +95 -4
  50. package/latticeai/core/users.py +0 -38
  51. package/latticeai/core/vector_index/__init__.py +61 -0
  52. package/latticeai/core/vector_index/hnsw.py +383 -0
  53. package/latticeai/core/vector_index/sidecar.py +329 -0
  54. package/latticeai/models/router/catalog.py +2 -2
  55. package/latticeai/models/router/generation.py +176 -30
  56. package/latticeai/models/router/loading.py +150 -9
  57. package/latticeai/runtime/access_runtime.py +7 -4
  58. package/latticeai/runtime/brain_runtime.py +43 -9
  59. package/latticeai/runtime/build_phases/features.py +8 -31
  60. package/latticeai/runtime/build_phases/foundation.py +7 -16
  61. package/latticeai/runtime/build_phases/web.py +3 -3
  62. package/latticeai/runtime/build_phases/worker_profile.py +29 -27
  63. package/latticeai/runtime/runtime_context.py +0 -2
  64. package/latticeai/services/architecture_readiness.py +18 -19
  65. package/latticeai/services/process_audit.py +1 -22
  66. package/latticeai/services/product_readiness.py +39 -12
  67. package/latticeai/services/search_service.py +7 -0
  68. package/latticeai/services/voice_capture.py +8 -28
  69. package/latticeai/tools/__init__.py +12 -47
  70. package/latticeai/tools/commands.py +9 -15
  71. package/latticeai/tools/documents.py +12 -0
  72. package/latticeai/tools/knowledge.py +0 -6
  73. package/latticeai/tools/markup.py +152 -0
  74. package/package.json +4 -5
  75. package/requirements.txt +0 -1
  76. package/scripts/check_current_release_docs.mjs +1 -1
  77. package/scripts/check_openapi_drift.mjs +3 -2
  78. package/scripts/check_server_i18n.mjs +5 -4
  79. package/scripts/compose_openapi.py +4 -1
  80. package/scripts/export_openapi.py +5 -4
  81. package/scripts/gen_worker_allowlist_fixture.py +2 -2
  82. package/scripts/openapi_route_families.json +19 -74
  83. package/scripts/publish_release.mjs +157 -0
  84. package/scripts/release_screen_claims.json +144 -28
  85. package/src-tauri/Cargo.lock +45 -10
  86. package/src-tauri/Cargo.toml +1 -1
  87. package/src-tauri/tauri.conf.json +1 -1
  88. package/static/app/asset-manifest.json +47 -41
  89. package/static/app/assets/Act-Cf1L2709.js +2 -0
  90. package/static/app/assets/AdminConsole-DPAbLTYV.js +1 -0
  91. package/static/app/assets/Brain-DqamGrj-.js +2 -0
  92. package/static/app/assets/BrainHome-MHe2_RYs.js +2 -0
  93. package/static/app/assets/BrainSignals-CQPPfyyH.js +1 -0
  94. package/static/app/assets/Capture-DGdIH_Zc.js +1 -0
  95. package/static/app/assets/Chronicle-C-UlCJoJ.js +1 -0
  96. package/static/app/assets/CommandPalette-WNT4EqUX.js +1 -0
  97. package/static/app/assets/DigitalBrainExplorer-CEBH5Cwc.js +321 -0
  98. package/static/app/assets/Library-C6xd1dlf.js +1 -0
  99. package/static/app/assets/LivingBrain-BEk-0ohw.js +1 -0
  100. package/static/app/assets/ProductFlow-CZLm5iXh.js +1 -0
  101. package/static/app/assets/QueryClientProvider-B3OjqSyJ.js +1 -0
  102. package/static/app/assets/{ReviewCard-HXRle3qq.js → ReviewCard-CEHG6evf.js} +2 -2
  103. package/static/app/assets/RunsListPanel-CLtEJSRW.js +1 -0
  104. package/static/app/assets/System-CAxwBUXw.js +1 -0
  105. package/static/app/assets/WorkflowGraph-Dj10RuGE.js +1 -0
  106. package/static/app/assets/WorkflowsPanel-Kyeh_LIT.js +2 -0
  107. package/static/app/assets/actHelpers-CtSmK9Dw.js +1 -0
  108. package/static/app/assets/arrow-left-CRl5EO4D.js +1 -0
  109. package/static/app/assets/{bot-Cn8bWRuq.js → bot-DhUGRel2.js} +1 -1
  110. package/static/app/assets/brain-CLkhHsHF.js +1 -0
  111. package/static/app/assets/button-CmaEqG1T.js +1 -0
  112. package/static/app/assets/circle-check-CFgejkOS.js +1 -0
  113. package/static/app/assets/{circle-pause-CmzC_apg.js → circle-pause-l96izbxj.js} +1 -1
  114. package/static/app/assets/{circle-play-D8mW2aQ7.js → circle-play-CrZa25_q.js} +1 -1
  115. package/static/app/assets/{cpu-DZcdd0PZ.js → cpu-BaXudqwl.js} +1 -1
  116. package/static/app/assets/{download-bv1KEPGQ.js → download-hCVFPiyc.js} +1 -1
  117. package/static/app/assets/{folder-open-d-Pip5gr.js → folder-open-CHL82Yp7.js} +1 -1
  118. package/static/app/assets/{hard-drive-D20iavUb.js → hard-drive-DDzET7lk.js} +1 -1
  119. package/static/app/assets/index-CB93CZWW.css +2 -0
  120. package/static/app/assets/index-D2H-wSl6.js +13 -0
  121. package/static/app/assets/input-Df1CAY_I.js +1 -0
  122. package/static/app/assets/jsx-runtime-bzQ4Vb5N.js +1 -0
  123. package/static/app/assets/{link-2-BPJOFlAy.js → link-2-xNnTIX1_.js} +1 -1
  124. package/static/app/assets/{permissionCopy-ChdJd493.js → permissionCopy-D3aWHco-.js} +1 -1
  125. package/static/app/assets/primitives-BioD2slS.js +1 -0
  126. package/static/app/assets/search-BzBw8YcW.js +1 -0
  127. package/static/app/assets/{share-2-YNX_NtMU.js → share-2-FkzGf8Df.js} +1 -1
  128. package/static/app/assets/{shield-alert-DuQ3zrVL.js → shield-alert-B3dwzik4.js} +1 -1
  129. package/static/app/assets/sourceMeta-DQSY_tah.js +1 -0
  130. package/static/app/assets/textarea-P8o6pvOP.js +1 -0
  131. package/static/app/assets/useFocusTrap-hswOIkXE.js +1 -0
  132. package/static/app/assets/useMutation-OJLrYSRA.js +1 -0
  133. package/static/app/assets/workspace-BCuk3Ku9.js +1 -0
  134. package/static/app/index.html +4 -4
  135. package/static/sw.js +1 -1
  136. package/lattice_brain/ingestion/pipeline.py +0 -108
  137. package/latticeai/api/local_files.py +0 -44
  138. package/latticeai/api/tools.py +0 -126
  139. package/latticeai/api/voice_capture.py +0 -32
  140. package/latticeai/core/agent_permission.py +0 -85
  141. package/scripts/agent_eval.py +0 -34
  142. package/scripts/brain_quality_eval.py +0 -37
  143. package/scripts/check_legacy_debt.mjs +0 -91
  144. package/scripts/check_python.py +0 -100
  145. package/scripts/chunking_parity_corpus.py +0 -449
  146. package/scripts/generate_agent_parity_fixtures.py +0 -771
  147. package/scripts/generate_chunking_parity_fixtures.py +0 -259
  148. package/static/app/assets/Act-BPcVAbOL.js +0 -1
  149. package/static/app/assets/AdminConsole-Bw1ATQL0.js +0 -1
  150. package/static/app/assets/Brain-CT92Kos0.js +0 -321
  151. package/static/app/assets/BrainHome-CFBkt1K_.js +0 -2
  152. package/static/app/assets/BrainSignals-ReLWF2H8.js +0 -1
  153. package/static/app/assets/Capture-BsTokYkk.js +0 -1
  154. package/static/app/assets/Chronicle-B6f0T9id.js +0 -1
  155. package/static/app/assets/CommandPalette-CuvjTv1u.js +0 -1
  156. package/static/app/assets/Library-BGJbG9Hd.js +0 -1
  157. package/static/app/assets/LivingBrain-DGYK_Jsa.js +0 -1
  158. package/static/app/assets/ProductFlow-DXBC6brE.js +0 -1
  159. package/static/app/assets/System-CMHSO9qM.js +0 -1
  160. package/static/app/assets/arrow-left-BfmkskWx.js +0 -1
  161. package/static/app/assets/brain-CQJberbE.js +0 -1
  162. package/static/app/assets/button-Ct9f2_oT.js +0 -1
  163. package/static/app/assets/circle-check-DruOxB-4.js +0 -1
  164. package/static/app/assets/index-D9x-kSNy.css +0 -2
  165. package/static/app/assets/index-Do83hDzJ.js +0 -10
  166. package/static/app/assets/input-BLXVNmj1.js +0 -1
  167. package/static/app/assets/primitives-Cv5tbZBY.js +0 -1
  168. package/static/app/assets/search-CT9aho2j.js +0 -1
  169. package/static/app/assets/textarea-DqwLnli4.js +0 -1
  170. package/static/app/assets/useFocusTrap-ZVI98jaW.js +0 -1
  171. package/static/app/assets/useMutation-CVC4qv_D.js +0 -1
  172. package/static/app/assets/useQuery-C7BeG4HU.js +0 -1
  173. package/static/app/assets/utils-CiFtIdZq.js +0 -4
  174. package/static/app/assets/workspace-DQz9vIId.js +0 -1
@@ -0,0 +1,107 @@
1
+ """Heading paths for a document, so a fact can say *where* it came from.
2
+
3
+ The typed chunker already computes a `" > "`-joined heading path per chunk and
4
+ files it on the Chunk node (`heading_path`). Extraction ran on the whole
5
+ document text and had no idea which section a sentence sat in, so an edge could
6
+ say "이 문장이 근거다" but never "그 문장은 「아키텍처 > 저장소」 절에 있다".
7
+
8
+ This module closes that gap with the *same* rule the chunker uses — a line
9
+ matching `^#{1,6} ` opens a section — so the heading a triple names and the
10
+ heading its chunk carries are the same string.
11
+
12
+ Character offsets throughout, because Python slices `str` by code point and the
13
+ rest of the pipeline (chunk `start_char`, the Rust port) does too.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import re
19
+ from typing import List, Optional, Sequence, Tuple
20
+
21
+ #: `^(#{1,6}) (.*)$` under `re.MULTILINE` — the chunker's heading rule.
22
+ _HEADING = re.compile(r"^(#{1,6}) (.*)$", re.MULTILINE)
23
+
24
+ #: `(start, end, heading_path)`; `end` is exclusive.
25
+ Span = Tuple[int, int, str]
26
+
27
+
28
+ def heading_spans(text: str) -> List[Span]:
29
+ """Every heading's span and its `" > "`-joined path, in document order.
30
+
31
+ Text before the first heading belongs to no section and is deliberately
32
+ absent from the result — an honest "no heading" beats inventing one from
33
+ the filename.
34
+
35
+ >>> heading_spans("# A\\nintro\\n## B\\nbody")
36
+ [(0, 12, 'A'), (12, 20, 'A > B')]
37
+ """
38
+ matches = list(_HEADING.finditer(text or ""))
39
+ spans: List[Span] = []
40
+ stack: List[Tuple[int, str]] = []
41
+ for index, match in enumerate(matches):
42
+ level = len(match.group(1))
43
+ title = match.group(2).strip()
44
+ while stack and stack[-1][0] >= level:
45
+ stack.pop()
46
+ stack.append((level, title))
47
+ end = matches[index + 1].start() if index + 1 < len(matches) else len(text)
48
+ spans.append((match.start(), end, " > ".join(part for _, part in stack)))
49
+ return spans
50
+
51
+
52
+ def heading_at(spans: Sequence[Span], offset: int) -> str:
53
+ """The heading path covering ``offset``, or `""` when there is none."""
54
+ for start, end, path in spans:
55
+ if start <= offset < end:
56
+ return path
57
+ return ""
58
+
59
+
60
+ def with_section(context: str, section: str) -> str:
61
+ """``context`` prefixed with the section it came from, when there is one.
62
+
63
+ ``"[아키텍처 > 저장소] 쓰기는 GraphWriter가 담당한다."`` — one string,
64
+ because `TripleSpec.context` is the only free-text channel an extracted
65
+ edge has. Blank sections leave the context untouched rather than adding an
66
+ empty bracket.
67
+ """
68
+ section = (section or "").strip()
69
+ if not section:
70
+ return context
71
+ return f"[{section}] {context}"
72
+
73
+
74
+ def sentence_offsets(text: str, pattern: "re.Pattern[str]") -> List[Tuple[int, str]]:
75
+ """``(offset, sentence)`` for a split that keeps each piece's position.
76
+
77
+ ``re.split`` throws the offsets away, and the offset is exactly what maps a
78
+ sentence back to its heading. Walking the separators keeps both.
79
+ """
80
+ out: List[Tuple[int, str]] = []
81
+ cursor = 0
82
+ for match in pattern.finditer(text or ""):
83
+ out.append((cursor, text[cursor : match.start()]))
84
+ cursor = match.end()
85
+ out.append((cursor, (text or "")[cursor:]))
86
+ return out
87
+
88
+
89
+ def leading_offset(raw: str, stripped: str) -> Optional[int]:
90
+ """How many characters ``str.strip()`` removed from the front of ``raw``.
91
+
92
+ ``None`` when ``stripped`` is empty — there is no position to report for a
93
+ piece that stripped away entirely.
94
+ """
95
+ if not stripped:
96
+ return None
97
+ return raw.index(stripped)
98
+
99
+
100
+ __all__ = [
101
+ "Span",
102
+ "heading_at",
103
+ "heading_spans",
104
+ "leading_offset",
105
+ "sentence_offsets",
106
+ "with_section",
107
+ ]
@@ -1,17 +1,27 @@
1
- """Text cleaning, chunking, and citation-locator maths.
1
+ """Text cleaning and the legacy fixed-width chunk walk.
2
2
 
3
3
  Moved verbatim out of the ``_kg_common`` grab-bag (v11.3.0 decomposition).
4
4
  Nothing here reaches back into the rest of the package — the import graph is
5
5
  ``text ← relations ← extraction ← __init__`` — so this is the layer every
6
6
  other one may build on.
7
+
8
+ The **typed chunker** that used to live here — ``typed_chunks`` and its four
9
+ strategies, ``chunk_strategy_for``, ``pdf_page_offsets``, and the three
10
+ readers that only ever consumed their output (``typed_chunk_meta_fields``,
11
+ ``citation_locator``, ``page_for_offset``) — was removed in 11.8.0. Chunking
12
+ is native: ``lattice-ingest`` owns it, pinned by
13
+ ``rust/lattice-ingest/tests/chunking_parity.rs`` against the committed
14
+ ``rust/fixtures/chunking`` goldens. The worker imports only the extraction
15
+ helpers (``POST /worker/extract`` — see
16
+ ``latticeai/api/worker_compute.py::build_extract_reply``), so the Python copy
17
+ had no shipping call site left; keeping a second boundary algorithm that
18
+ nothing runs is how two chunkers quietly stop agreeing.
7
19
  """
8
20
 
9
21
  from __future__ import annotations
10
22
 
11
23
  import re
12
- from typing import Any, Dict, List, Optional, Tuple
13
-
14
- from ...quiet import quiet
24
+ from typing import List
15
25
 
16
26
 
17
27
  def _clean_text(text: str) -> str:
@@ -31,449 +41,3 @@ def _chunks(text: str, size: int = 1200, overlap: int = 160) -> List[str]:
31
41
  break
32
42
  start = max(0, end - overlap)
33
43
  return chunks
34
-
35
-
36
- # ── Typed chunking (review 2026-07-25 §5.2 S2 — Wave 2.1 + 2.4) ──────────────
37
- # ``_chunks`` above is a compatibility contract (chunk ids hash over the chunk
38
- # text) and stays byte-for-byte untouched. ``typed_chunks`` layers strategy-
39
- # aware boundaries plus per-chunk provenance (start_char / heading_path) on
40
- # top; ``strategy="plain"`` reproduces the exact ``_chunks`` boundaries so
41
- # unchanged plain content keeps identical chunk ids.
42
-
43
- _MARKDOWN_CHUNK_EXTENSIONS = {".md", ".markdown"}
44
- _CODE_CHUNK_EXTENSIONS = {
45
- ".py", ".js", ".jsx", ".ts", ".tsx", ".go", ".rs", ".java", ".rb",
46
- ".c", ".h", ".cpp", ".css", ".sh", ".sql", ".vue", ".svelte",
47
- ".json", ".yaml", ".yml", ".toml",
48
- }
49
- _PROSE_CHUNK_EXTENSIONS = {
50
- ".txt", ".pdf", ".docx", ".doc", ".rtf", ".odt", ".epub", ".html", ".htm",
51
- }
52
- _CHUNK_STRATEGIES = {"plain", "markdown", "code", "prose"}
53
- # Markdown sections smaller than this merge forward into the next section so
54
- # heading-dense documents don't shatter into confetti chunks.
55
- _MARKDOWN_MIN_SECTION_CHARS = 200
56
- _MARKDOWN_HEADING_RE = re.compile(r"^(#{1,6}) (.*)$", re.MULTILINE)
57
- _CODE_BOUNDARY_LINE_RE = re.compile(
58
- r"^(?:def |class |function |export |const |public |private )", re.MULTILINE
59
- )
60
- _CODE_BLANK_RUN_RE = re.compile(r"\n\s*\n")
61
-
62
-
63
- def chunk_strategy_for(filename: Any, *, content_type: str = "") -> str:
64
- """Route a filename / path / URI (plus optional MIME hint) to a strategy.
65
-
66
- Returns ``"markdown"`` for .md/.markdown, ``"code"`` for known source-code
67
- extensions, ``"prose"`` for document formats whose text is running prose
68
- (.txt/.pdf/.docx/.html/…), ``"plain"`` otherwise. Case-insensitive,
69
- tolerant of URLs (query/fragment stripped) and ``Path`` objects; never
70
- raises — any malformed input falls back to ``"plain"``.
71
-
72
- Unknown/extension-less input stays ``"plain"`` on purpose: the plain
73
- strategy is the byte-compatible legacy walk, and guessing prose for
74
- something that might be a data dump would move chunk boundaries for no
75
- retrieval gain.
76
- """
77
- try:
78
- name = str(filename or "").strip().lower()
79
- for sep in ("?", "#"):
80
- name = name.split(sep, 1)[0]
81
- name = name.replace("\\", "/").rstrip("/").rsplit("/", 1)[-1]
82
- dot = name.rfind(".")
83
- ext = name[dot:] if dot > 0 else ""
84
- if ext in _MARKDOWN_CHUNK_EXTENSIONS:
85
- return "markdown"
86
- if ext in _CODE_CHUNK_EXTENSIONS:
87
- return "code"
88
- if ext in _PROSE_CHUNK_EXTENSIONS:
89
- return "prose"
90
- mime = str(content_type or "").strip().lower()
91
- if "markdown" in mime:
92
- return "markdown"
93
- if mime.startswith("text/html") or mime.startswith("text/plain"):
94
- return "prose"
95
- except Exception:
96
- quiet()
97
- return "plain"
98
-
99
-
100
- def _plain_windows(
101
- cleaned: str,
102
- size: int,
103
- overlap: int,
104
- *,
105
- base_offset: int = 0,
106
- strategy: str = "plain",
107
- heading_path: Optional[str] = None,
108
- ) -> List[Dict[str, Any]]:
109
- """The exact ``_chunks`` walk with ``start_char`` tracked.
110
-
111
- Boundaries and chunk texts are byte-identical to ``_chunks`` over the same
112
- string — this is the plain-strategy compatibility guarantee.
113
- """
114
- out: List[Dict[str, Any]] = []
115
- start = 0
116
- total = len(cleaned)
117
- while start < total:
118
- end = min(total, start + size)
119
- out.append(
120
- {
121
- "text": cleaned[start:end],
122
- "meta": {
123
- "strategy": strategy,
124
- "start_char": base_offset + start,
125
- "heading_path": heading_path,
126
- },
127
- }
128
- )
129
- if end >= total:
130
- break
131
- start = max(0, end - overlap)
132
- return out
133
-
134
-
135
- def _markdown_section_spans(cleaned: str) -> List[Tuple[int, int, Optional[str]]]:
136
- """``(start, end, heading_path)`` spans split at ``^#{1,6} `` heading lines.
137
-
138
- ``heading_path`` is the " > "-joined path of the enclosing headings
139
- including the section's own heading (e.g. ``"Guide > Setup"``); the
140
- preamble before the first heading carries ``None``. Spans are contiguous
141
- raw slices of ``cleaned`` so every chunk text round-trips via start_char.
142
- """
143
- spans: List[Tuple[int, int, Optional[str]]] = []
144
- stack: List[Tuple[int, str]] = []
145
- prev_start = 0
146
- prev_path: Optional[str] = None
147
- for match in _MARKDOWN_HEADING_RE.finditer(cleaned):
148
- offset = match.start()
149
- if offset > prev_start:
150
- spans.append((prev_start, offset, prev_path))
151
- level = len(match.group(1))
152
- while stack and stack[-1][0] >= level:
153
- stack.pop()
154
- stack.append((level, match.group(2).strip()))
155
- prev_start = offset
156
- prev_path = " > ".join(title for _, title in stack) or None
157
- if len(cleaned) > prev_start:
158
- spans.append((prev_start, len(cleaned), prev_path))
159
- return spans
160
-
161
-
162
- def _merge_small_sections(
163
- spans: List[Tuple[int, int, Optional[str]]], min_chars: int
164
- ) -> List[Tuple[int, int, Optional[str]]]:
165
- """Merge sections under ``min_chars`` forward into the next section.
166
-
167
- A merged section keeps the heading_path of its first constituent (the
168
- path in effect at the chunk start). A trailing undersized section merges
169
- backward into the previous emitted section when one exists.
170
- """
171
- merged: List[Tuple[int, int, Optional[str]]] = []
172
- pending: Optional[Tuple[int, int, Optional[str]]] = None
173
- for start, end, path in spans:
174
- if pending is None:
175
- pending = (start, end, path)
176
- else:
177
- pending = (pending[0], end, pending[2])
178
- if pending[1] - pending[0] >= min_chars:
179
- merged.append(pending)
180
- pending = None
181
- if pending is not None:
182
- if merged and pending[1] - pending[0] < min_chars:
183
- last = merged.pop()
184
- merged.append((last[0], pending[1], last[2]))
185
- else:
186
- merged.append(pending)
187
- return merged
188
-
189
-
190
- def _markdown_chunks(cleaned: str, size: int, overlap: int) -> List[Dict[str, Any]]:
191
- sections = _merge_small_sections(
192
- _markdown_section_spans(cleaned), _MARKDOWN_MIN_SECTION_CHARS
193
- )
194
- out: List[Dict[str, Any]] = []
195
- for start, end, path in sections:
196
- body = cleaned[start:end]
197
- if len(body) <= size:
198
- out.append(
199
- {
200
- "text": body,
201
- "meta": {
202
- "strategy": "markdown",
203
- "start_char": start,
204
- "heading_path": path,
205
- },
206
- }
207
- )
208
- else:
209
- out.extend(
210
- _plain_windows(
211
- body,
212
- size,
213
- overlap,
214
- base_offset=start,
215
- strategy="markdown",
216
- heading_path=path,
217
- )
218
- )
219
- return out
220
-
221
-
222
- def _code_segment_spans(cleaned: str) -> List[Tuple[int, int]]:
223
- """Contiguous top-level segments split at blank-line runs and decl lines."""
224
- boundaries = {0, len(cleaned)}
225
- for match in _CODE_BLANK_RUN_RE.finditer(cleaned):
226
- boundaries.add(match.end())
227
- for match in _CODE_BOUNDARY_LINE_RE.finditer(cleaned):
228
- boundaries.add(match.start())
229
- ordered = sorted(boundaries)
230
- return [
231
- (ordered[i], ordered[i + 1])
232
- for i in range(len(ordered) - 1)
233
- if ordered[i + 1] > ordered[i]
234
- ]
235
-
236
-
237
- def _code_chunks(cleaned: str, size: int, overlap: int) -> List[Dict[str, Any]]:
238
- hard_limit = int(size * 1.5)
239
- out: List[Dict[str, Any]] = []
240
- pack: Optional[Tuple[int, int]] = None
241
-
242
- def _emit(span: Tuple[int, int]) -> None:
243
- out.append(
244
- {
245
- "text": cleaned[span[0] : span[1]],
246
- "meta": {
247
- "strategy": "code",
248
- "start_char": span[0],
249
- "heading_path": None,
250
- },
251
- }
252
- )
253
-
254
- for start, end in _code_segment_spans(cleaned):
255
- if end - start > hard_limit:
256
- # Monster segment: flush the pack, then window it like plain text.
257
- if pack is not None:
258
- _emit(pack)
259
- pack = None
260
- out.extend(
261
- _plain_windows(
262
- cleaned[start:end],
263
- size,
264
- overlap,
265
- base_offset=start,
266
- strategy="code",
267
- )
268
- )
269
- continue
270
- if pack is None:
271
- pack = (start, end)
272
- elif end - pack[0] <= size:
273
- pack = (pack[0], end)
274
- else:
275
- _emit(pack)
276
- pack = (start, end)
277
- if pack is not None:
278
- _emit(pack)
279
- return out
280
-
281
-
282
- # ── Prose chunking (review 2026-07-27 P1 #4) ────────────────────────────────
283
- # The plain walk cuts every ``size`` characters, which lands mid-sentence and
284
- # — for Korean, where the verb carrying the meaning sits at the end — routinely
285
- # splits a claim from its predicate. Retrieval then matches half a statement
286
- # and the citation shows a fragment. The prose strategy keeps the same window
287
- # budget but ends each chunk at the last sentence/paragraph boundary inside it.
288
-
289
- # Strong: sentence-final punctuation (ASCII + CJK) with optional closing
290
- # quotes/brackets, followed by whitespace; or a blank-line paragraph break.
291
- _PROSE_STRONG_BOUNDARY_RE = re.compile(
292
- r"(?:[.!?。!?…]+[\"'”’」』\)\]]*\s+|\n[ \t]*\n)"
293
- )
294
- # Weak: a single line break. Korean notes and bullet lists often carry no
295
- # sentence punctuation at all; a line end is still a real boundary there.
296
- _PROSE_WEAK_BOUNDARY_RE = re.compile(r"\n")
297
- # Never emit a chunk shorter than this fraction of ``size`` just to hit a
298
- # boundary — tiny chunks hurt recall more than a mid-sentence cut.
299
- _PROSE_MIN_SPAN_RATIO = 0.5
300
-
301
-
302
- def _last_boundary(cleaned: str, lo: int, hi: int) -> Optional[int]:
303
- """End offset of the last sentence/paragraph boundary in ``cleaned[lo:hi]``.
304
-
305
- Strong boundaries win; a single line break is the fallback. Returns None
306
- when the span holds neither, so the caller keeps the hard window cut.
307
- """
308
- window = cleaned[lo:hi]
309
- for pattern in (_PROSE_STRONG_BOUNDARY_RE, _PROSE_WEAK_BOUNDARY_RE):
310
- last = None
311
- for match in pattern.finditer(window):
312
- last = match.end()
313
- if last:
314
- return lo + last
315
- return None
316
-
317
-
318
- def _prose_chunks(cleaned: str, size: int, overlap: int) -> List[Dict[str, Any]]:
319
- out: List[Dict[str, Any]] = []
320
- total = len(cleaned)
321
- min_span = max(1, int(size * _PROSE_MIN_SPAN_RATIO))
322
- start = 0
323
- while start < total:
324
- hard_end = min(total, start + size)
325
- end = hard_end
326
- if hard_end < total:
327
- boundary = _last_boundary(cleaned, start + min_span, hard_end)
328
- if boundary is not None and boundary > start:
329
- end = boundary
330
- out.append(
331
- {
332
- "text": cleaned[start:end],
333
- "meta": {
334
- "strategy": "prose",
335
- "start_char": start,
336
- "heading_path": None,
337
- },
338
- }
339
- )
340
- if end >= total:
341
- break
342
- # Overlap carries the tail of the previous chunk into the next one so
343
- # a claim split across a boundary is still retrievable from both.
344
- start = max(start + 1, end - overlap)
345
- return out
346
-
347
-
348
- def typed_chunks(
349
- text: str,
350
- *,
351
- strategy: str = "plain",
352
- size: int = 1200,
353
- overlap: int = 160,
354
- ) -> List[Dict[str, Any]]:
355
- """Strategy-aware chunking with per-chunk provenance metadata.
356
-
357
- Returns ``[{"text": str, "meta": {"strategy", "start_char", "heading_path"}}]``
358
- where ``start_char`` is the offset in ``str(text or "").strip()`` (every
359
- chunk text is an exact substring at that offset).
360
-
361
- Contract: ``[c["text"] for c in typed_chunks(t)] == _chunks(t)`` for the
362
- default plain strategy — unknown strategies also fall back to plain.
363
- """
364
- cleaned = str(text or "").strip()
365
- if not cleaned:
366
- return []
367
- try:
368
- size = max(1, int(size))
369
- except Exception:
370
- size = 1200
371
- try:
372
- overlap = min(max(0, int(overlap)), size - 1)
373
- except Exception:
374
- overlap = min(160, size - 1)
375
- label = strategy if strategy in _CHUNK_STRATEGIES else "plain"
376
- if label == "markdown":
377
- return _markdown_chunks(cleaned, size, overlap)
378
- if label == "code":
379
- return _code_chunks(cleaned, size, overlap)
380
- if label == "prose":
381
- return _prose_chunks(cleaned, size, overlap)
382
- return _plain_windows(cleaned, size, overlap)
383
-
384
-
385
- def typed_chunk_meta_fields(piece: Dict[str, Any]) -> Dict[str, Any]:
386
- """Additive chunk-metadata fields for one ``typed_chunks`` piece.
387
-
388
- Ingest call sites merge this into the existing ``{"index", "source_node"}``
389
- chunk metadata; ``heading_path`` is only present when known — honest
390
- absence over empty labels.
391
- """
392
- meta = piece.get("meta") or {}
393
- fields: Dict[str, Any] = {
394
- "strategy": str(meta.get("strategy") or "plain"),
395
- "start_char": int(meta.get("start_char") or 0),
396
- }
397
- heading_path = meta.get("heading_path")
398
- if heading_path:
399
- fields["heading_path"] = str(heading_path)
400
- return fields
401
-
402
-
403
- def citation_locator(chunk_metadata: Any) -> str:
404
- """Human "where in the document" label for one chunk, or "".
405
-
406
- Built only from provenance the chunk actually carries — a section heading
407
- path and/or a page number. When neither is known the answer is the empty
408
- string, so a citation never claims a location it cannot prove.
409
- """
410
- if not isinstance(chunk_metadata, dict):
411
- return ""
412
- parts: List[str] = []
413
- heading = str(chunk_metadata.get("heading_path") or "").strip()
414
- if heading:
415
- parts.append(heading)
416
- def _page(key: str) -> int:
417
- value = chunk_metadata.get(key)
418
- try:
419
- return int(value) if value is not None else 0
420
- except (TypeError, ValueError):
421
- return 0
422
-
423
- page_number = _page("page")
424
- if page_number > 0:
425
- page_end = _page("page_end")
426
- parts.append(
427
- f"p.{page_number}–{page_end}" if page_end > page_number else f"p.{page_number}"
428
- )
429
- return " · ".join(parts)
430
-
431
-
432
- def pdf_page_offsets(structure: Any) -> List[int]:
433
- """Start offset of each PDF page in the "\\n\\n"-joined page text.
434
-
435
- ``structure`` is the ``metadata["structure"]`` dict produced by
436
- ``_pdf_structure`` (``pages`` = ``[{"chars": int, ...}, ...]``); pages were
437
- joined with ``"\\n\\n"`` (see ``read_document``), so page k starts at
438
- ``sum(chars[j] + 2 for j < k)``. Empty or malformed input returns ``[]``.
439
- """
440
- if not isinstance(structure, dict):
441
- return []
442
- pages = structure.get("pages")
443
- if not isinstance(pages, list) or not pages:
444
- return []
445
- offsets: List[int] = []
446
- cursor = 0
447
- for page in pages:
448
- if not isinstance(page, dict):
449
- return []
450
- chars = page.get("chars")
451
- if isinstance(chars, bool) or not isinstance(chars, (int, float)) or chars < 0:
452
- return []
453
- offsets.append(cursor)
454
- cursor += int(chars) + 2 # +2 for the "\n\n" page joiner
455
- return offsets
456
-
457
-
458
- def page_for_offset(page_offsets: List[int], offset: int) -> Optional[int]:
459
- """1-based page number containing ``offset`` given page start offsets.
460
-
461
- Returns ``None`` when ``page_offsets`` is empty or the offset precedes the
462
- first page start (honest absence over a wrong label).
463
- """
464
- if not page_offsets:
465
- return None
466
- try:
467
- target = int(offset)
468
- except Exception:
469
- return None
470
- page = 0
471
- for index, start in enumerate(page_offsets):
472
- try:
473
- if target >= int(start):
474
- page = index + 1
475
- else:
476
- break
477
- except Exception:
478
- return None
479
- return page if page >= 1 else None
@@ -28,6 +28,13 @@ LOCAL_CODE_EXTENSIONS = {
28
28
  ".tsx",
29
29
  ".jsx",
30
30
  ".html",
31
+ # v12.0.0: `.htm` sat beside `.html` in every other table (the chunker's
32
+ # prose list, the parser matrix) and was missing only here, so a folder of
33
+ # `.htm` pages was scanned past in silence. `.rs` was missing outright —
34
+ # this repository's own Rust half was invisible to its own folder ingest.
35
+ ".htm",
36
+ ".rs",
37
+ ".go",
31
38
  ".css",
32
39
  ".json",
33
40
  ".yaml",
@@ -16,8 +16,12 @@ asks this worker for:
16
16
  of a parse request rather than of a write;
17
17
  * ``hashing`` — ``content_hash_text`` and the file digest, which decide
18
18
  idempotency and must produce the same bytes on both sides;
19
- * ``quality`` — the advisory extraction score behind ``POST /worker/parse``;
20
- * ``pipeline`` — the multi-modal capability probe.
19
+ * ``quality`` — the advisory extraction score behind ``POST /worker/parse``.
20
+
21
+ ``pipeline`` was a fifth: an ``IngestionPipeline`` reduced to a single
22
+ capability probe, whose one route (``GET /api/ingestion/multimodal``) had no
23
+ caller. v11.8.0 removed the route and the class with it — the gates it read
24
+ still live in ``constants``, where anything that needs them can ask directly.
21
25
  """
22
26
 
23
27
  from __future__ import annotations
@@ -78,7 +82,6 @@ from .hashing import _file_digest as _file_digest
78
82
  from .hashing import content_hash_text as content_hash_text
79
83
  from .models import IngestionItem as IngestionItem
80
84
  from .models import IngestionResult as IngestionResult
81
- from .pipeline import IngestionPipeline as IngestionPipeline
82
85
  from .quality import _BOILERPLATE_LINE_MARKERS as _BOILERPLATE_LINE_MARKERS
83
86
  from .quality import _CAPTURE_REASON_LABELS as _CAPTURE_REASON_LABELS
84
87
  from .quality import _WEB_SOURCE_TYPES as _WEB_SOURCE_TYPES
@@ -36,9 +36,15 @@ imports nothing from ``latticeai``.
36
36
 
37
37
  v11.6.0 removed the *writing* half — ``write_image_memory``,
38
38
  ``write_video_memory``, the keyframe writer and the node-id helpers. Extraction
39
- returns facts; ``lattice-core``'s graph write engine turns them into nodes. What
40
- is left here is exactly what ``POST /worker/multimodal/describe`` and
41
- ``POST /worker/asr`` answer with.
39
+ returns facts; ``lattice-core``'s graph write engine turns them into nodes.
40
+
41
+ The audio half is what ``POST /worker/asr`` answers with. The image and video
42
+ halves currently have **no HTTP door**: ``POST /worker/multimodal/describe``
43
+ wrapped :func:`extract_image_facts` for a native image ingest that was never
44
+ built, and v11.8.0 deleted the seam rather than keep a route nothing called.
45
+ The observation functions stay — they are Brain Core's account of what a
46
+ picture or a recording contains, unit-tested directly, and the seam is a
47
+ handful of lines to restore on the day a native image ingest needs one.
42
48
 
43
49
  Split into cohesive submodules in v11.3.0 (no behaviour change): ``common``
44
50
  (taxonomy + shared helpers), ``ports`` (injected capabilities + the ffmpeg
@@ -1,3 +1,3 @@
1
1
  """Lattice AI - modular server package."""
2
2
 
3
- __version__ = "11.7.0"
3
+ __version__ = "12.0.0"