ltcai 11.7.0 → 12.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +100 -76
- package/docs/BENCHMARKS.md +9 -2
- package/docs/CHANGELOG.md +249 -0
- package/docs/CI_AND_RELEASE_GATES.md +126 -41
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +271 -103
- package/docs/ENTERPRISE.md +1 -1
- package/docs/LEGACY_COMPATIBILITY.md +10 -6
- package/docs/MULTI_AGENT_RUNTIME.md +4 -4
- package/docs/ONBOARDING.md +16 -4
- package/docs/OPERATIONS.md +14 -1
- package/docs/PERMISSION_MODE.md +14 -9
- package/docs/REALTIME_COLLABORATION.md +1 -1
- package/docs/ROADMAP.md +113 -0
- package/docs/TRUST_MODEL.md +28 -7
- package/docs/USABILITY_AUDIT.md +5 -0
- package/docs/WHY_LATTICE.md +13 -5
- package/docs/WORKFLOW_DESIGNER.md +2 -2
- package/docs/kg-schema.md +57 -7
- package/docs/mcp-tools.md +93 -82
- package/docs/security-model.md +6 -3
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +1 -54
- package/lattice_brain/graph/_kg_common/extraction.py +459 -105
- package/lattice_brain/graph/_kg_common/normalize.py +305 -0
- package/lattice_brain/graph/_kg_common/patterns.py +275 -0
- package/lattice_brain/graph/_kg_common/relations.py +12 -3
- package/lattice_brain/graph/_kg_common/sections.py +107 -0
- package/lattice_brain/graph/_kg_common/text.py +14 -450
- package/lattice_brain/graph/_kg_constants.py +7 -0
- package/lattice_brain/ingestion/__init__.py +6 -3
- package/lattice_brain/multimodal/__init__.py +9 -3
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/agent_worker_seam.py +44 -1
- package/latticeai/api/models.py +18 -110
- package/latticeai/api/search.py +7 -30
- package/latticeai/api/worker_compute.py +127 -106
- package/latticeai/api/worker_seams.py +17 -2
- package/latticeai/core/embedding_providers/__init__.py +16 -0
- package/latticeai/core/embedding_providers/autodetect.py +302 -0
- package/latticeai/core/embedding_providers/base.py +25 -0
- package/latticeai/core/embedding_providers/profiles.py +44 -0
- package/latticeai/core/embedding_providers/text.py +74 -8
- package/latticeai/core/http_origin.py +3 -3
- package/latticeai/core/messages.py +0 -5
- package/latticeai/core/policy.py +1 -6
- package/latticeai/core/quiet.py +1 -20
- package/latticeai/core/security.py +29 -83
- package/latticeai/core/sessions.py +95 -4
- package/latticeai/core/users.py +0 -38
- package/latticeai/core/vector_index/__init__.py +61 -0
- package/latticeai/core/vector_index/hnsw.py +383 -0
- package/latticeai/core/vector_index/sidecar.py +329 -0
- package/latticeai/models/router/catalog.py +2 -2
- package/latticeai/models/router/generation.py +176 -30
- package/latticeai/models/router/loading.py +150 -9
- package/latticeai/runtime/access_runtime.py +7 -4
- package/latticeai/runtime/brain_runtime.py +43 -9
- package/latticeai/runtime/build_phases/features.py +8 -31
- package/latticeai/runtime/build_phases/foundation.py +7 -16
- package/latticeai/runtime/build_phases/web.py +3 -3
- package/latticeai/runtime/build_phases/worker_profile.py +29 -27
- package/latticeai/runtime/runtime_context.py +0 -2
- package/latticeai/services/architecture_readiness.py +18 -19
- package/latticeai/services/process_audit.py +1 -22
- package/latticeai/services/product_readiness.py +39 -12
- package/latticeai/services/search_service.py +7 -0
- package/latticeai/services/voice_capture.py +8 -28
- package/latticeai/tools/__init__.py +12 -47
- package/latticeai/tools/commands.py +9 -15
- package/latticeai/tools/documents.py +12 -0
- package/latticeai/tools/knowledge.py +0 -6
- package/latticeai/tools/markup.py +152 -0
- package/package.json +4 -5
- package/requirements.txt +0 -1
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_openapi_drift.mjs +3 -2
- package/scripts/check_server_i18n.mjs +5 -4
- package/scripts/compose_openapi.py +4 -1
- package/scripts/export_openapi.py +5 -4
- package/scripts/gen_worker_allowlist_fixture.py +2 -2
- package/scripts/openapi_route_families.json +19 -74
- package/scripts/publish_release.mjs +157 -0
- package/scripts/release_screen_claims.json +144 -28
- package/src-tauri/Cargo.lock +45 -10
- package/src-tauri/Cargo.toml +1 -1
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +47 -41
- package/static/app/assets/Act-Cf1L2709.js +2 -0
- package/static/app/assets/AdminConsole-DPAbLTYV.js +1 -0
- package/static/app/assets/Brain-DqamGrj-.js +2 -0
- package/static/app/assets/BrainHome-MHe2_RYs.js +2 -0
- package/static/app/assets/BrainSignals-CQPPfyyH.js +1 -0
- package/static/app/assets/Capture-DGdIH_Zc.js +1 -0
- package/static/app/assets/Chronicle-C-UlCJoJ.js +1 -0
- package/static/app/assets/CommandPalette-WNT4EqUX.js +1 -0
- package/static/app/assets/DigitalBrainExplorer-CEBH5Cwc.js +321 -0
- package/static/app/assets/Library-C6xd1dlf.js +1 -0
- package/static/app/assets/LivingBrain-BEk-0ohw.js +1 -0
- package/static/app/assets/ProductFlow-CZLm5iXh.js +1 -0
- package/static/app/assets/QueryClientProvider-B3OjqSyJ.js +1 -0
- package/static/app/assets/{ReviewCard-HXRle3qq.js → ReviewCard-CEHG6evf.js} +2 -2
- package/static/app/assets/RunsListPanel-CLtEJSRW.js +1 -0
- package/static/app/assets/System-CAxwBUXw.js +1 -0
- package/static/app/assets/WorkflowGraph-Dj10RuGE.js +1 -0
- package/static/app/assets/WorkflowsPanel-Kyeh_LIT.js +2 -0
- package/static/app/assets/actHelpers-CtSmK9Dw.js +1 -0
- package/static/app/assets/arrow-left-CRl5EO4D.js +1 -0
- package/static/app/assets/{bot-Cn8bWRuq.js → bot-DhUGRel2.js} +1 -1
- package/static/app/assets/brain-CLkhHsHF.js +1 -0
- package/static/app/assets/button-CmaEqG1T.js +1 -0
- package/static/app/assets/circle-check-CFgejkOS.js +1 -0
- package/static/app/assets/{circle-pause-CmzC_apg.js → circle-pause-l96izbxj.js} +1 -1
- package/static/app/assets/{circle-play-D8mW2aQ7.js → circle-play-CrZa25_q.js} +1 -1
- package/static/app/assets/{cpu-DZcdd0PZ.js → cpu-BaXudqwl.js} +1 -1
- package/static/app/assets/{download-bv1KEPGQ.js → download-hCVFPiyc.js} +1 -1
- package/static/app/assets/{folder-open-d-Pip5gr.js → folder-open-CHL82Yp7.js} +1 -1
- package/static/app/assets/{hard-drive-D20iavUb.js → hard-drive-DDzET7lk.js} +1 -1
- package/static/app/assets/index-CB93CZWW.css +2 -0
- package/static/app/assets/index-D2H-wSl6.js +13 -0
- package/static/app/assets/input-Df1CAY_I.js +1 -0
- package/static/app/assets/jsx-runtime-bzQ4Vb5N.js +1 -0
- package/static/app/assets/{link-2-BPJOFlAy.js → link-2-xNnTIX1_.js} +1 -1
- package/static/app/assets/{permissionCopy-ChdJd493.js → permissionCopy-D3aWHco-.js} +1 -1
- package/static/app/assets/primitives-BioD2slS.js +1 -0
- package/static/app/assets/search-BzBw8YcW.js +1 -0
- package/static/app/assets/{share-2-YNX_NtMU.js → share-2-FkzGf8Df.js} +1 -1
- package/static/app/assets/{shield-alert-DuQ3zrVL.js → shield-alert-B3dwzik4.js} +1 -1
- package/static/app/assets/sourceMeta-DQSY_tah.js +1 -0
- package/static/app/assets/textarea-P8o6pvOP.js +1 -0
- package/static/app/assets/useFocusTrap-hswOIkXE.js +1 -0
- package/static/app/assets/useMutation-OJLrYSRA.js +1 -0
- package/static/app/assets/workspace-BCuk3Ku9.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/ingestion/pipeline.py +0 -108
- package/latticeai/api/local_files.py +0 -44
- package/latticeai/api/tools.py +0 -126
- package/latticeai/api/voice_capture.py +0 -32
- package/latticeai/core/agent_permission.py +0 -85
- package/scripts/agent_eval.py +0 -34
- package/scripts/brain_quality_eval.py +0 -37
- package/scripts/check_legacy_debt.mjs +0 -91
- package/scripts/check_python.py +0 -100
- package/scripts/chunking_parity_corpus.py +0 -449
- package/scripts/generate_agent_parity_fixtures.py +0 -771
- package/scripts/generate_chunking_parity_fixtures.py +0 -259
- package/static/app/assets/Act-BPcVAbOL.js +0 -1
- package/static/app/assets/AdminConsole-Bw1ATQL0.js +0 -1
- package/static/app/assets/Brain-CT92Kos0.js +0 -321
- package/static/app/assets/BrainHome-CFBkt1K_.js +0 -2
- package/static/app/assets/BrainSignals-ReLWF2H8.js +0 -1
- package/static/app/assets/Capture-BsTokYkk.js +0 -1
- package/static/app/assets/Chronicle-B6f0T9id.js +0 -1
- package/static/app/assets/CommandPalette-CuvjTv1u.js +0 -1
- package/static/app/assets/Library-BGJbG9Hd.js +0 -1
- package/static/app/assets/LivingBrain-DGYK_Jsa.js +0 -1
- package/static/app/assets/ProductFlow-DXBC6brE.js +0 -1
- package/static/app/assets/System-CMHSO9qM.js +0 -1
- package/static/app/assets/arrow-left-BfmkskWx.js +0 -1
- package/static/app/assets/brain-CQJberbE.js +0 -1
- package/static/app/assets/button-Ct9f2_oT.js +0 -1
- package/static/app/assets/circle-check-DruOxB-4.js +0 -1
- package/static/app/assets/index-D9x-kSNy.css +0 -2
- package/static/app/assets/index-Do83hDzJ.js +0 -10
- package/static/app/assets/input-BLXVNmj1.js +0 -1
- package/static/app/assets/primitives-Cv5tbZBY.js +0 -1
- package/static/app/assets/search-CT9aho2j.js +0 -1
- package/static/app/assets/textarea-DqwLnli4.js +0 -1
- package/static/app/assets/useFocusTrap-ZVI98jaW.js +0 -1
- package/static/app/assets/useMutation-CVC4qv_D.js +0 -1
- package/static/app/assets/useQuery-C7BeG4HU.js +0 -1
- package/static/app/assets/utils-CiFtIdZq.js +0 -4
- package/static/app/assets/workspace-DQz9vIId.js +0 -1
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
"""Heading paths for a document, so a fact can say *where* it came from.
|
|
2
|
+
|
|
3
|
+
The typed chunker already computes a `" > "`-joined heading path per chunk and
|
|
4
|
+
files it on the Chunk node (`heading_path`). Extraction ran on the whole
|
|
5
|
+
document text and had no idea which section a sentence sat in, so an edge could
|
|
6
|
+
say "이 문장이 근거다" but never "그 문장은 「아키텍처 > 저장소」 절에 있다".
|
|
7
|
+
|
|
8
|
+
This module closes that gap with the *same* rule the chunker uses — a line
|
|
9
|
+
matching `^#{1,6} ` opens a section — so the heading a triple names and the
|
|
10
|
+
heading its chunk carries are the same string.
|
|
11
|
+
|
|
12
|
+
Character offsets throughout, because Python slices `str` by code point and the
|
|
13
|
+
rest of the pipeline (chunk `start_char`, the Rust port) does too.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import re
|
|
19
|
+
from typing import List, Optional, Sequence, Tuple
|
|
20
|
+
|
|
21
|
+
#: `^(#{1,6}) (.*)$` under `re.MULTILINE` — the chunker's heading rule.
|
|
22
|
+
_HEADING = re.compile(r"^(#{1,6}) (.*)$", re.MULTILINE)
|
|
23
|
+
|
|
24
|
+
#: `(start, end, heading_path)`; `end` is exclusive.
|
|
25
|
+
Span = Tuple[int, int, str]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def heading_spans(text: str) -> List[Span]:
|
|
29
|
+
"""Every heading's span and its `" > "`-joined path, in document order.
|
|
30
|
+
|
|
31
|
+
Text before the first heading belongs to no section and is deliberately
|
|
32
|
+
absent from the result — an honest "no heading" beats inventing one from
|
|
33
|
+
the filename.
|
|
34
|
+
|
|
35
|
+
>>> heading_spans("# A\\nintro\\n## B\\nbody")
|
|
36
|
+
[(0, 12, 'A'), (12, 20, 'A > B')]
|
|
37
|
+
"""
|
|
38
|
+
matches = list(_HEADING.finditer(text or ""))
|
|
39
|
+
spans: List[Span] = []
|
|
40
|
+
stack: List[Tuple[int, str]] = []
|
|
41
|
+
for index, match in enumerate(matches):
|
|
42
|
+
level = len(match.group(1))
|
|
43
|
+
title = match.group(2).strip()
|
|
44
|
+
while stack and stack[-1][0] >= level:
|
|
45
|
+
stack.pop()
|
|
46
|
+
stack.append((level, title))
|
|
47
|
+
end = matches[index + 1].start() if index + 1 < len(matches) else len(text)
|
|
48
|
+
spans.append((match.start(), end, " > ".join(part for _, part in stack)))
|
|
49
|
+
return spans
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def heading_at(spans: Sequence[Span], offset: int) -> str:
|
|
53
|
+
"""The heading path covering ``offset``, or `""` when there is none."""
|
|
54
|
+
for start, end, path in spans:
|
|
55
|
+
if start <= offset < end:
|
|
56
|
+
return path
|
|
57
|
+
return ""
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def with_section(context: str, section: str) -> str:
|
|
61
|
+
"""``context`` prefixed with the section it came from, when there is one.
|
|
62
|
+
|
|
63
|
+
``"[아키텍처 > 저장소] 쓰기는 GraphWriter가 담당한다."`` — one string,
|
|
64
|
+
because `TripleSpec.context` is the only free-text channel an extracted
|
|
65
|
+
edge has. Blank sections leave the context untouched rather than adding an
|
|
66
|
+
empty bracket.
|
|
67
|
+
"""
|
|
68
|
+
section = (section or "").strip()
|
|
69
|
+
if not section:
|
|
70
|
+
return context
|
|
71
|
+
return f"[{section}] {context}"
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def sentence_offsets(text: str, pattern: "re.Pattern[str]") -> List[Tuple[int, str]]:
|
|
75
|
+
"""``(offset, sentence)`` for a split that keeps each piece's position.
|
|
76
|
+
|
|
77
|
+
``re.split`` throws the offsets away, and the offset is exactly what maps a
|
|
78
|
+
sentence back to its heading. Walking the separators keeps both.
|
|
79
|
+
"""
|
|
80
|
+
out: List[Tuple[int, str]] = []
|
|
81
|
+
cursor = 0
|
|
82
|
+
for match in pattern.finditer(text or ""):
|
|
83
|
+
out.append((cursor, text[cursor : match.start()]))
|
|
84
|
+
cursor = match.end()
|
|
85
|
+
out.append((cursor, (text or "")[cursor:]))
|
|
86
|
+
return out
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def leading_offset(raw: str, stripped: str) -> Optional[int]:
|
|
90
|
+
"""How many characters ``str.strip()`` removed from the front of ``raw``.
|
|
91
|
+
|
|
92
|
+
``None`` when ``stripped`` is empty — there is no position to report for a
|
|
93
|
+
piece that stripped away entirely.
|
|
94
|
+
"""
|
|
95
|
+
if not stripped:
|
|
96
|
+
return None
|
|
97
|
+
return raw.index(stripped)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
__all__ = [
|
|
101
|
+
"Span",
|
|
102
|
+
"heading_at",
|
|
103
|
+
"heading_spans",
|
|
104
|
+
"leading_offset",
|
|
105
|
+
"sentence_offsets",
|
|
106
|
+
"with_section",
|
|
107
|
+
]
|
|
@@ -1,17 +1,27 @@
|
|
|
1
|
-
"""Text cleaning
|
|
1
|
+
"""Text cleaning and the legacy fixed-width chunk walk.
|
|
2
2
|
|
|
3
3
|
Moved verbatim out of the ``_kg_common`` grab-bag (v11.3.0 decomposition).
|
|
4
4
|
Nothing here reaches back into the rest of the package — the import graph is
|
|
5
5
|
``text ← relations ← extraction ← __init__`` — so this is the layer every
|
|
6
6
|
other one may build on.
|
|
7
|
+
|
|
8
|
+
The **typed chunker** that used to live here — ``typed_chunks`` and its four
|
|
9
|
+
strategies, ``chunk_strategy_for``, ``pdf_page_offsets``, and the three
|
|
10
|
+
readers that only ever consumed their output (``typed_chunk_meta_fields``,
|
|
11
|
+
``citation_locator``, ``page_for_offset``) — was removed in 11.8.0. Chunking
|
|
12
|
+
is native: ``lattice-ingest`` owns it, pinned by
|
|
13
|
+
``rust/lattice-ingest/tests/chunking_parity.rs`` against the committed
|
|
14
|
+
``rust/fixtures/chunking`` goldens. The worker imports only the extraction
|
|
15
|
+
helpers (``POST /worker/extract`` — see
|
|
16
|
+
``latticeai/api/worker_compute.py::build_extract_reply``), so the Python copy
|
|
17
|
+
had no shipping call site left; keeping a second boundary algorithm that
|
|
18
|
+
nothing runs is how two chunkers quietly stop agreeing.
|
|
7
19
|
"""
|
|
8
20
|
|
|
9
21
|
from __future__ import annotations
|
|
10
22
|
|
|
11
23
|
import re
|
|
12
|
-
from typing import
|
|
13
|
-
|
|
14
|
-
from ...quiet import quiet
|
|
24
|
+
from typing import List
|
|
15
25
|
|
|
16
26
|
|
|
17
27
|
def _clean_text(text: str) -> str:
|
|
@@ -31,449 +41,3 @@ def _chunks(text: str, size: int = 1200, overlap: int = 160) -> List[str]:
|
|
|
31
41
|
break
|
|
32
42
|
start = max(0, end - overlap)
|
|
33
43
|
return chunks
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
# ── Typed chunking (review 2026-07-25 §5.2 S2 — Wave 2.1 + 2.4) ──────────────
|
|
37
|
-
# ``_chunks`` above is a compatibility contract (chunk ids hash over the chunk
|
|
38
|
-
# text) and stays byte-for-byte untouched. ``typed_chunks`` layers strategy-
|
|
39
|
-
# aware boundaries plus per-chunk provenance (start_char / heading_path) on
|
|
40
|
-
# top; ``strategy="plain"`` reproduces the exact ``_chunks`` boundaries so
|
|
41
|
-
# unchanged plain content keeps identical chunk ids.
|
|
42
|
-
|
|
43
|
-
_MARKDOWN_CHUNK_EXTENSIONS = {".md", ".markdown"}
|
|
44
|
-
_CODE_CHUNK_EXTENSIONS = {
|
|
45
|
-
".py", ".js", ".jsx", ".ts", ".tsx", ".go", ".rs", ".java", ".rb",
|
|
46
|
-
".c", ".h", ".cpp", ".css", ".sh", ".sql", ".vue", ".svelte",
|
|
47
|
-
".json", ".yaml", ".yml", ".toml",
|
|
48
|
-
}
|
|
49
|
-
_PROSE_CHUNK_EXTENSIONS = {
|
|
50
|
-
".txt", ".pdf", ".docx", ".doc", ".rtf", ".odt", ".epub", ".html", ".htm",
|
|
51
|
-
}
|
|
52
|
-
_CHUNK_STRATEGIES = {"plain", "markdown", "code", "prose"}
|
|
53
|
-
# Markdown sections smaller than this merge forward into the next section so
|
|
54
|
-
# heading-dense documents don't shatter into confetti chunks.
|
|
55
|
-
_MARKDOWN_MIN_SECTION_CHARS = 200
|
|
56
|
-
_MARKDOWN_HEADING_RE = re.compile(r"^(#{1,6}) (.*)$", re.MULTILINE)
|
|
57
|
-
_CODE_BOUNDARY_LINE_RE = re.compile(
|
|
58
|
-
r"^(?:def |class |function |export |const |public |private )", re.MULTILINE
|
|
59
|
-
)
|
|
60
|
-
_CODE_BLANK_RUN_RE = re.compile(r"\n\s*\n")
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
def chunk_strategy_for(filename: Any, *, content_type: str = "") -> str:
|
|
64
|
-
"""Route a filename / path / URI (plus optional MIME hint) to a strategy.
|
|
65
|
-
|
|
66
|
-
Returns ``"markdown"`` for .md/.markdown, ``"code"`` for known source-code
|
|
67
|
-
extensions, ``"prose"`` for document formats whose text is running prose
|
|
68
|
-
(.txt/.pdf/.docx/.html/…), ``"plain"`` otherwise. Case-insensitive,
|
|
69
|
-
tolerant of URLs (query/fragment stripped) and ``Path`` objects; never
|
|
70
|
-
raises — any malformed input falls back to ``"plain"``.
|
|
71
|
-
|
|
72
|
-
Unknown/extension-less input stays ``"plain"`` on purpose: the plain
|
|
73
|
-
strategy is the byte-compatible legacy walk, and guessing prose for
|
|
74
|
-
something that might be a data dump would move chunk boundaries for no
|
|
75
|
-
retrieval gain.
|
|
76
|
-
"""
|
|
77
|
-
try:
|
|
78
|
-
name = str(filename or "").strip().lower()
|
|
79
|
-
for sep in ("?", "#"):
|
|
80
|
-
name = name.split(sep, 1)[0]
|
|
81
|
-
name = name.replace("\\", "/").rstrip("/").rsplit("/", 1)[-1]
|
|
82
|
-
dot = name.rfind(".")
|
|
83
|
-
ext = name[dot:] if dot > 0 else ""
|
|
84
|
-
if ext in _MARKDOWN_CHUNK_EXTENSIONS:
|
|
85
|
-
return "markdown"
|
|
86
|
-
if ext in _CODE_CHUNK_EXTENSIONS:
|
|
87
|
-
return "code"
|
|
88
|
-
if ext in _PROSE_CHUNK_EXTENSIONS:
|
|
89
|
-
return "prose"
|
|
90
|
-
mime = str(content_type or "").strip().lower()
|
|
91
|
-
if "markdown" in mime:
|
|
92
|
-
return "markdown"
|
|
93
|
-
if mime.startswith("text/html") or mime.startswith("text/plain"):
|
|
94
|
-
return "prose"
|
|
95
|
-
except Exception:
|
|
96
|
-
quiet()
|
|
97
|
-
return "plain"
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
def _plain_windows(
|
|
101
|
-
cleaned: str,
|
|
102
|
-
size: int,
|
|
103
|
-
overlap: int,
|
|
104
|
-
*,
|
|
105
|
-
base_offset: int = 0,
|
|
106
|
-
strategy: str = "plain",
|
|
107
|
-
heading_path: Optional[str] = None,
|
|
108
|
-
) -> List[Dict[str, Any]]:
|
|
109
|
-
"""The exact ``_chunks`` walk with ``start_char`` tracked.
|
|
110
|
-
|
|
111
|
-
Boundaries and chunk texts are byte-identical to ``_chunks`` over the same
|
|
112
|
-
string — this is the plain-strategy compatibility guarantee.
|
|
113
|
-
"""
|
|
114
|
-
out: List[Dict[str, Any]] = []
|
|
115
|
-
start = 0
|
|
116
|
-
total = len(cleaned)
|
|
117
|
-
while start < total:
|
|
118
|
-
end = min(total, start + size)
|
|
119
|
-
out.append(
|
|
120
|
-
{
|
|
121
|
-
"text": cleaned[start:end],
|
|
122
|
-
"meta": {
|
|
123
|
-
"strategy": strategy,
|
|
124
|
-
"start_char": base_offset + start,
|
|
125
|
-
"heading_path": heading_path,
|
|
126
|
-
},
|
|
127
|
-
}
|
|
128
|
-
)
|
|
129
|
-
if end >= total:
|
|
130
|
-
break
|
|
131
|
-
start = max(0, end - overlap)
|
|
132
|
-
return out
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
def _markdown_section_spans(cleaned: str) -> List[Tuple[int, int, Optional[str]]]:
|
|
136
|
-
"""``(start, end, heading_path)`` spans split at ``^#{1,6} `` heading lines.
|
|
137
|
-
|
|
138
|
-
``heading_path`` is the " > "-joined path of the enclosing headings
|
|
139
|
-
including the section's own heading (e.g. ``"Guide > Setup"``); the
|
|
140
|
-
preamble before the first heading carries ``None``. Spans are contiguous
|
|
141
|
-
raw slices of ``cleaned`` so every chunk text round-trips via start_char.
|
|
142
|
-
"""
|
|
143
|
-
spans: List[Tuple[int, int, Optional[str]]] = []
|
|
144
|
-
stack: List[Tuple[int, str]] = []
|
|
145
|
-
prev_start = 0
|
|
146
|
-
prev_path: Optional[str] = None
|
|
147
|
-
for match in _MARKDOWN_HEADING_RE.finditer(cleaned):
|
|
148
|
-
offset = match.start()
|
|
149
|
-
if offset > prev_start:
|
|
150
|
-
spans.append((prev_start, offset, prev_path))
|
|
151
|
-
level = len(match.group(1))
|
|
152
|
-
while stack and stack[-1][0] >= level:
|
|
153
|
-
stack.pop()
|
|
154
|
-
stack.append((level, match.group(2).strip()))
|
|
155
|
-
prev_start = offset
|
|
156
|
-
prev_path = " > ".join(title for _, title in stack) or None
|
|
157
|
-
if len(cleaned) > prev_start:
|
|
158
|
-
spans.append((prev_start, len(cleaned), prev_path))
|
|
159
|
-
return spans
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
def _merge_small_sections(
|
|
163
|
-
spans: List[Tuple[int, int, Optional[str]]], min_chars: int
|
|
164
|
-
) -> List[Tuple[int, int, Optional[str]]]:
|
|
165
|
-
"""Merge sections under ``min_chars`` forward into the next section.
|
|
166
|
-
|
|
167
|
-
A merged section keeps the heading_path of its first constituent (the
|
|
168
|
-
path in effect at the chunk start). A trailing undersized section merges
|
|
169
|
-
backward into the previous emitted section when one exists.
|
|
170
|
-
"""
|
|
171
|
-
merged: List[Tuple[int, int, Optional[str]]] = []
|
|
172
|
-
pending: Optional[Tuple[int, int, Optional[str]]] = None
|
|
173
|
-
for start, end, path in spans:
|
|
174
|
-
if pending is None:
|
|
175
|
-
pending = (start, end, path)
|
|
176
|
-
else:
|
|
177
|
-
pending = (pending[0], end, pending[2])
|
|
178
|
-
if pending[1] - pending[0] >= min_chars:
|
|
179
|
-
merged.append(pending)
|
|
180
|
-
pending = None
|
|
181
|
-
if pending is not None:
|
|
182
|
-
if merged and pending[1] - pending[0] < min_chars:
|
|
183
|
-
last = merged.pop()
|
|
184
|
-
merged.append((last[0], pending[1], last[2]))
|
|
185
|
-
else:
|
|
186
|
-
merged.append(pending)
|
|
187
|
-
return merged
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
def _markdown_chunks(cleaned: str, size: int, overlap: int) -> List[Dict[str, Any]]:
|
|
191
|
-
sections = _merge_small_sections(
|
|
192
|
-
_markdown_section_spans(cleaned), _MARKDOWN_MIN_SECTION_CHARS
|
|
193
|
-
)
|
|
194
|
-
out: List[Dict[str, Any]] = []
|
|
195
|
-
for start, end, path in sections:
|
|
196
|
-
body = cleaned[start:end]
|
|
197
|
-
if len(body) <= size:
|
|
198
|
-
out.append(
|
|
199
|
-
{
|
|
200
|
-
"text": body,
|
|
201
|
-
"meta": {
|
|
202
|
-
"strategy": "markdown",
|
|
203
|
-
"start_char": start,
|
|
204
|
-
"heading_path": path,
|
|
205
|
-
},
|
|
206
|
-
}
|
|
207
|
-
)
|
|
208
|
-
else:
|
|
209
|
-
out.extend(
|
|
210
|
-
_plain_windows(
|
|
211
|
-
body,
|
|
212
|
-
size,
|
|
213
|
-
overlap,
|
|
214
|
-
base_offset=start,
|
|
215
|
-
strategy="markdown",
|
|
216
|
-
heading_path=path,
|
|
217
|
-
)
|
|
218
|
-
)
|
|
219
|
-
return out
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
def _code_segment_spans(cleaned: str) -> List[Tuple[int, int]]:
|
|
223
|
-
"""Contiguous top-level segments split at blank-line runs and decl lines."""
|
|
224
|
-
boundaries = {0, len(cleaned)}
|
|
225
|
-
for match in _CODE_BLANK_RUN_RE.finditer(cleaned):
|
|
226
|
-
boundaries.add(match.end())
|
|
227
|
-
for match in _CODE_BOUNDARY_LINE_RE.finditer(cleaned):
|
|
228
|
-
boundaries.add(match.start())
|
|
229
|
-
ordered = sorted(boundaries)
|
|
230
|
-
return [
|
|
231
|
-
(ordered[i], ordered[i + 1])
|
|
232
|
-
for i in range(len(ordered) - 1)
|
|
233
|
-
if ordered[i + 1] > ordered[i]
|
|
234
|
-
]
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
def _code_chunks(cleaned: str, size: int, overlap: int) -> List[Dict[str, Any]]:
|
|
238
|
-
hard_limit = int(size * 1.5)
|
|
239
|
-
out: List[Dict[str, Any]] = []
|
|
240
|
-
pack: Optional[Tuple[int, int]] = None
|
|
241
|
-
|
|
242
|
-
def _emit(span: Tuple[int, int]) -> None:
|
|
243
|
-
out.append(
|
|
244
|
-
{
|
|
245
|
-
"text": cleaned[span[0] : span[1]],
|
|
246
|
-
"meta": {
|
|
247
|
-
"strategy": "code",
|
|
248
|
-
"start_char": span[0],
|
|
249
|
-
"heading_path": None,
|
|
250
|
-
},
|
|
251
|
-
}
|
|
252
|
-
)
|
|
253
|
-
|
|
254
|
-
for start, end in _code_segment_spans(cleaned):
|
|
255
|
-
if end - start > hard_limit:
|
|
256
|
-
# Monster segment: flush the pack, then window it like plain text.
|
|
257
|
-
if pack is not None:
|
|
258
|
-
_emit(pack)
|
|
259
|
-
pack = None
|
|
260
|
-
out.extend(
|
|
261
|
-
_plain_windows(
|
|
262
|
-
cleaned[start:end],
|
|
263
|
-
size,
|
|
264
|
-
overlap,
|
|
265
|
-
base_offset=start,
|
|
266
|
-
strategy="code",
|
|
267
|
-
)
|
|
268
|
-
)
|
|
269
|
-
continue
|
|
270
|
-
if pack is None:
|
|
271
|
-
pack = (start, end)
|
|
272
|
-
elif end - pack[0] <= size:
|
|
273
|
-
pack = (pack[0], end)
|
|
274
|
-
else:
|
|
275
|
-
_emit(pack)
|
|
276
|
-
pack = (start, end)
|
|
277
|
-
if pack is not None:
|
|
278
|
-
_emit(pack)
|
|
279
|
-
return out
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
# ── Prose chunking (review 2026-07-27 P1 #4) ────────────────────────────────
|
|
283
|
-
# The plain walk cuts every ``size`` characters, which lands mid-sentence and
|
|
284
|
-
# — for Korean, where the verb carrying the meaning sits at the end — routinely
|
|
285
|
-
# splits a claim from its predicate. Retrieval then matches half a statement
|
|
286
|
-
# and the citation shows a fragment. The prose strategy keeps the same window
|
|
287
|
-
# budget but ends each chunk at the last sentence/paragraph boundary inside it.
|
|
288
|
-
|
|
289
|
-
# Strong: sentence-final punctuation (ASCII + CJK) with optional closing
|
|
290
|
-
# quotes/brackets, followed by whitespace; or a blank-line paragraph break.
|
|
291
|
-
_PROSE_STRONG_BOUNDARY_RE = re.compile(
|
|
292
|
-
r"(?:[.!?。!?…]+[\"'”’」』\)\]]*\s+|\n[ \t]*\n)"
|
|
293
|
-
)
|
|
294
|
-
# Weak: a single line break. Korean notes and bullet lists often carry no
|
|
295
|
-
# sentence punctuation at all; a line end is still a real boundary there.
|
|
296
|
-
_PROSE_WEAK_BOUNDARY_RE = re.compile(r"\n")
|
|
297
|
-
# Never emit a chunk shorter than this fraction of ``size`` just to hit a
|
|
298
|
-
# boundary — tiny chunks hurt recall more than a mid-sentence cut.
|
|
299
|
-
_PROSE_MIN_SPAN_RATIO = 0.5
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
def _last_boundary(cleaned: str, lo: int, hi: int) -> Optional[int]:
|
|
303
|
-
"""End offset of the last sentence/paragraph boundary in ``cleaned[lo:hi]``.
|
|
304
|
-
|
|
305
|
-
Strong boundaries win; a single line break is the fallback. Returns None
|
|
306
|
-
when the span holds neither, so the caller keeps the hard window cut.
|
|
307
|
-
"""
|
|
308
|
-
window = cleaned[lo:hi]
|
|
309
|
-
for pattern in (_PROSE_STRONG_BOUNDARY_RE, _PROSE_WEAK_BOUNDARY_RE):
|
|
310
|
-
last = None
|
|
311
|
-
for match in pattern.finditer(window):
|
|
312
|
-
last = match.end()
|
|
313
|
-
if last:
|
|
314
|
-
return lo + last
|
|
315
|
-
return None
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
def _prose_chunks(cleaned: str, size: int, overlap: int) -> List[Dict[str, Any]]:
|
|
319
|
-
out: List[Dict[str, Any]] = []
|
|
320
|
-
total = len(cleaned)
|
|
321
|
-
min_span = max(1, int(size * _PROSE_MIN_SPAN_RATIO))
|
|
322
|
-
start = 0
|
|
323
|
-
while start < total:
|
|
324
|
-
hard_end = min(total, start + size)
|
|
325
|
-
end = hard_end
|
|
326
|
-
if hard_end < total:
|
|
327
|
-
boundary = _last_boundary(cleaned, start + min_span, hard_end)
|
|
328
|
-
if boundary is not None and boundary > start:
|
|
329
|
-
end = boundary
|
|
330
|
-
out.append(
|
|
331
|
-
{
|
|
332
|
-
"text": cleaned[start:end],
|
|
333
|
-
"meta": {
|
|
334
|
-
"strategy": "prose",
|
|
335
|
-
"start_char": start,
|
|
336
|
-
"heading_path": None,
|
|
337
|
-
},
|
|
338
|
-
}
|
|
339
|
-
)
|
|
340
|
-
if end >= total:
|
|
341
|
-
break
|
|
342
|
-
# Overlap carries the tail of the previous chunk into the next one so
|
|
343
|
-
# a claim split across a boundary is still retrievable from both.
|
|
344
|
-
start = max(start + 1, end - overlap)
|
|
345
|
-
return out
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
def typed_chunks(
|
|
349
|
-
text: str,
|
|
350
|
-
*,
|
|
351
|
-
strategy: str = "plain",
|
|
352
|
-
size: int = 1200,
|
|
353
|
-
overlap: int = 160,
|
|
354
|
-
) -> List[Dict[str, Any]]:
|
|
355
|
-
"""Strategy-aware chunking with per-chunk provenance metadata.
|
|
356
|
-
|
|
357
|
-
Returns ``[{"text": str, "meta": {"strategy", "start_char", "heading_path"}}]``
|
|
358
|
-
where ``start_char`` is the offset in ``str(text or "").strip()`` (every
|
|
359
|
-
chunk text is an exact substring at that offset).
|
|
360
|
-
|
|
361
|
-
Contract: ``[c["text"] for c in typed_chunks(t)] == _chunks(t)`` for the
|
|
362
|
-
default plain strategy — unknown strategies also fall back to plain.
|
|
363
|
-
"""
|
|
364
|
-
cleaned = str(text or "").strip()
|
|
365
|
-
if not cleaned:
|
|
366
|
-
return []
|
|
367
|
-
try:
|
|
368
|
-
size = max(1, int(size))
|
|
369
|
-
except Exception:
|
|
370
|
-
size = 1200
|
|
371
|
-
try:
|
|
372
|
-
overlap = min(max(0, int(overlap)), size - 1)
|
|
373
|
-
except Exception:
|
|
374
|
-
overlap = min(160, size - 1)
|
|
375
|
-
label = strategy if strategy in _CHUNK_STRATEGIES else "plain"
|
|
376
|
-
if label == "markdown":
|
|
377
|
-
return _markdown_chunks(cleaned, size, overlap)
|
|
378
|
-
if label == "code":
|
|
379
|
-
return _code_chunks(cleaned, size, overlap)
|
|
380
|
-
if label == "prose":
|
|
381
|
-
return _prose_chunks(cleaned, size, overlap)
|
|
382
|
-
return _plain_windows(cleaned, size, overlap)
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
def typed_chunk_meta_fields(piece: Dict[str, Any]) -> Dict[str, Any]:
|
|
386
|
-
"""Additive chunk-metadata fields for one ``typed_chunks`` piece.
|
|
387
|
-
|
|
388
|
-
Ingest call sites merge this into the existing ``{"index", "source_node"}``
|
|
389
|
-
chunk metadata; ``heading_path`` is only present when known — honest
|
|
390
|
-
absence over empty labels.
|
|
391
|
-
"""
|
|
392
|
-
meta = piece.get("meta") or {}
|
|
393
|
-
fields: Dict[str, Any] = {
|
|
394
|
-
"strategy": str(meta.get("strategy") or "plain"),
|
|
395
|
-
"start_char": int(meta.get("start_char") or 0),
|
|
396
|
-
}
|
|
397
|
-
heading_path = meta.get("heading_path")
|
|
398
|
-
if heading_path:
|
|
399
|
-
fields["heading_path"] = str(heading_path)
|
|
400
|
-
return fields
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
def citation_locator(chunk_metadata: Any) -> str:
|
|
404
|
-
"""Human "where in the document" label for one chunk, or "".
|
|
405
|
-
|
|
406
|
-
Built only from provenance the chunk actually carries — a section heading
|
|
407
|
-
path and/or a page number. When neither is known the answer is the empty
|
|
408
|
-
string, so a citation never claims a location it cannot prove.
|
|
409
|
-
"""
|
|
410
|
-
if not isinstance(chunk_metadata, dict):
|
|
411
|
-
return ""
|
|
412
|
-
parts: List[str] = []
|
|
413
|
-
heading = str(chunk_metadata.get("heading_path") or "").strip()
|
|
414
|
-
if heading:
|
|
415
|
-
parts.append(heading)
|
|
416
|
-
def _page(key: str) -> int:
|
|
417
|
-
value = chunk_metadata.get(key)
|
|
418
|
-
try:
|
|
419
|
-
return int(value) if value is not None else 0
|
|
420
|
-
except (TypeError, ValueError):
|
|
421
|
-
return 0
|
|
422
|
-
|
|
423
|
-
page_number = _page("page")
|
|
424
|
-
if page_number > 0:
|
|
425
|
-
page_end = _page("page_end")
|
|
426
|
-
parts.append(
|
|
427
|
-
f"p.{page_number}–{page_end}" if page_end > page_number else f"p.{page_number}"
|
|
428
|
-
)
|
|
429
|
-
return " · ".join(parts)
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
def pdf_page_offsets(structure: Any) -> List[int]:
|
|
433
|
-
"""Start offset of each PDF page in the "\\n\\n"-joined page text.
|
|
434
|
-
|
|
435
|
-
``structure`` is the ``metadata["structure"]`` dict produced by
|
|
436
|
-
``_pdf_structure`` (``pages`` = ``[{"chars": int, ...}, ...]``); pages were
|
|
437
|
-
joined with ``"\\n\\n"`` (see ``read_document``), so page k starts at
|
|
438
|
-
``sum(chars[j] + 2 for j < k)``. Empty or malformed input returns ``[]``.
|
|
439
|
-
"""
|
|
440
|
-
if not isinstance(structure, dict):
|
|
441
|
-
return []
|
|
442
|
-
pages = structure.get("pages")
|
|
443
|
-
if not isinstance(pages, list) or not pages:
|
|
444
|
-
return []
|
|
445
|
-
offsets: List[int] = []
|
|
446
|
-
cursor = 0
|
|
447
|
-
for page in pages:
|
|
448
|
-
if not isinstance(page, dict):
|
|
449
|
-
return []
|
|
450
|
-
chars = page.get("chars")
|
|
451
|
-
if isinstance(chars, bool) or not isinstance(chars, (int, float)) or chars < 0:
|
|
452
|
-
return []
|
|
453
|
-
offsets.append(cursor)
|
|
454
|
-
cursor += int(chars) + 2 # +2 for the "\n\n" page joiner
|
|
455
|
-
return offsets
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
def page_for_offset(page_offsets: List[int], offset: int) -> Optional[int]:
|
|
459
|
-
"""1-based page number containing ``offset`` given page start offsets.
|
|
460
|
-
|
|
461
|
-
Returns ``None`` when ``page_offsets`` is empty or the offset precedes the
|
|
462
|
-
first page start (honest absence over a wrong label).
|
|
463
|
-
"""
|
|
464
|
-
if not page_offsets:
|
|
465
|
-
return None
|
|
466
|
-
try:
|
|
467
|
-
target = int(offset)
|
|
468
|
-
except Exception:
|
|
469
|
-
return None
|
|
470
|
-
page = 0
|
|
471
|
-
for index, start in enumerate(page_offsets):
|
|
472
|
-
try:
|
|
473
|
-
if target >= int(start):
|
|
474
|
-
page = index + 1
|
|
475
|
-
else:
|
|
476
|
-
break
|
|
477
|
-
except Exception:
|
|
478
|
-
return None
|
|
479
|
-
return page if page >= 1 else None
|
|
@@ -28,6 +28,13 @@ LOCAL_CODE_EXTENSIONS = {
|
|
|
28
28
|
".tsx",
|
|
29
29
|
".jsx",
|
|
30
30
|
".html",
|
|
31
|
+
# v12.0.0: `.htm` sat beside `.html` in every other table (the chunker's
|
|
32
|
+
# prose list, the parser matrix) and was missing only here, so a folder of
|
|
33
|
+
# `.htm` pages was scanned past in silence. `.rs` was missing outright —
|
|
34
|
+
# this repository's own Rust half was invisible to its own folder ingest.
|
|
35
|
+
".htm",
|
|
36
|
+
".rs",
|
|
37
|
+
".go",
|
|
31
38
|
".css",
|
|
32
39
|
".json",
|
|
33
40
|
".yaml",
|
|
@@ -16,8 +16,12 @@ asks this worker for:
|
|
|
16
16
|
of a parse request rather than of a write;
|
|
17
17
|
* ``hashing`` — ``content_hash_text`` and the file digest, which decide
|
|
18
18
|
idempotency and must produce the same bytes on both sides;
|
|
19
|
-
* ``quality`` — the advisory extraction score behind ``POST /worker/parse
|
|
20
|
-
|
|
19
|
+
* ``quality`` — the advisory extraction score behind ``POST /worker/parse``.
|
|
20
|
+
|
|
21
|
+
``pipeline`` was a fifth: an ``IngestionPipeline`` reduced to a single
|
|
22
|
+
capability probe, whose one route (``GET /api/ingestion/multimodal``) had no
|
|
23
|
+
caller. v11.8.0 removed the route and the class with it — the gates it read
|
|
24
|
+
still live in ``constants``, where anything that needs them can ask directly.
|
|
21
25
|
"""
|
|
22
26
|
|
|
23
27
|
from __future__ import annotations
|
|
@@ -78,7 +82,6 @@ from .hashing import _file_digest as _file_digest
|
|
|
78
82
|
from .hashing import content_hash_text as content_hash_text
|
|
79
83
|
from .models import IngestionItem as IngestionItem
|
|
80
84
|
from .models import IngestionResult as IngestionResult
|
|
81
|
-
from .pipeline import IngestionPipeline as IngestionPipeline
|
|
82
85
|
from .quality import _BOILERPLATE_LINE_MARKERS as _BOILERPLATE_LINE_MARKERS
|
|
83
86
|
from .quality import _CAPTURE_REASON_LABELS as _CAPTURE_REASON_LABELS
|
|
84
87
|
from .quality import _WEB_SOURCE_TYPES as _WEB_SOURCE_TYPES
|
|
@@ -36,9 +36,15 @@ imports nothing from ``latticeai``.
|
|
|
36
36
|
|
|
37
37
|
v11.6.0 removed the *writing* half — ``write_image_memory``,
|
|
38
38
|
``write_video_memory``, the keyframe writer and the node-id helpers. Extraction
|
|
39
|
-
returns facts; ``lattice-core``'s graph write engine turns them into nodes.
|
|
40
|
-
|
|
41
|
-
``POST /worker/asr``
|
|
39
|
+
returns facts; ``lattice-core``'s graph write engine turns them into nodes.
|
|
40
|
+
|
|
41
|
+
The audio half is what ``POST /worker/asr`` answers with. The image and video
|
|
42
|
+
halves currently have **no HTTP door**: ``POST /worker/multimodal/describe``
|
|
43
|
+
wrapped :func:`extract_image_facts` for a native image ingest that was never
|
|
44
|
+
built, and v11.8.0 deleted the seam rather than keep a route nothing called.
|
|
45
|
+
The observation functions stay — they are Brain Core's account of what a
|
|
46
|
+
picture or a recording contains, unit-tested directly, and the seam is a
|
|
47
|
+
handful of lines to restore on the day a native image ingest needs one.
|
|
42
48
|
|
|
43
49
|
Split into cohesive submodules in v11.3.0 (no behaviour change): ``common``
|
|
44
50
|
(taxonomy + shared helpers), ``ports`` (injected capabilities + the ffmpeg
|
package/latticeai/__init__.py
CHANGED