ltcai 11.4.0 → 11.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +48 -44
- package/docs/CHANGELOG.md +26 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +1 -1
- package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +11 -6
- package/docs/v11.5.0_RUST_COMPLETE_PLAN.md +145 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/index_jobs.py +145 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +5 -0
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/runtime/build_phases/features.py +14 -0
- package/latticeai/services/architecture_readiness.py +1 -1
- package/latticeai/services/product_readiness.py +1 -1
- package/package.json +1 -1
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_server_i18n.mjs +1 -0
- package/scripts/chunking_parity_corpus.py +449 -0
- package/scripts/generate_agent_parity_fixtures.py +752 -0
- package/scripts/generate_chunking_parity_fixtures.py +259 -0
- package/scripts/generate_rust_parity_fixtures.py +525 -90
- package/scripts/release_screen_claims.json +11 -0
- package/src-tauri/Cargo.lock +47 -4
- package/src-tauri/Cargo.toml +11 -4
- package/src-tauri/src/backend.rs +251 -140
- package/src-tauri/src/main.rs +16 -4
- package/src-tauri/src/topology.rs +356 -0
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +40 -40
- package/static/app/assets/{Act-yYpYnn0v.js → Act-CWnxSCgN.js} +1 -1
- package/static/app/assets/{AdminConsole-DL3Cr5pL.js → AdminConsole-BEQYU6kF.js} +1 -1
- package/static/app/assets/{Brain-C1HBN0Wf.js → Brain-DWu1BhFg.js} +1 -1
- package/static/app/assets/{BrainHome-DoXRhUUC.js → BrainHome-95Hilr9R.js} +1 -1
- package/static/app/assets/{BrainSignals-6yR6ir5t.js → BrainSignals-QdeqCpAF.js} +1 -1
- package/static/app/assets/{Capture-CFIRsFNE.js → Capture-BHpCxnzb.js} +1 -1
- package/static/app/assets/{Chronicle-BZbEgiwN.js → Chronicle-B4xYKoed.js} +1 -1
- package/static/app/assets/{CommandPalette-D2pMxC2I.js → CommandPalette-BVXnttSz.js} +1 -1
- package/static/app/assets/{Library-DwO3yZST.js → Library-DgYcHome.js} +1 -1
- package/static/app/assets/{LivingBrain-Jn1GK0-S.js → LivingBrain-CrJLDbf7.js} +1 -1
- package/static/app/assets/{ProductFlow-B-w1R4Oo.js → ProductFlow-DFlScKoJ.js} +1 -1
- package/static/app/assets/{ReviewCard-6B27X8Vg.js → ReviewCard-Cy5f48Pj.js} +1 -1
- package/static/app/assets/{System-DW8F-2xL.js → System-NF8IfhTa.js} +1 -1
- package/static/app/assets/arrow-left-DwkSYrjR.js +1 -0
- package/static/app/assets/{bot-IM_E_Y12.js → bot-CucuhLhm.js} +1 -1
- package/static/app/assets/{brain-Ci1CkWjM.js → brain-BBnSryW_.js} +1 -1
- package/static/app/assets/{button-COwyqfHM.js → button-C2GUj2Ai.js} +1 -1
- package/static/app/assets/circle-check-CxOVPwYq.js +1 -0
- package/static/app/assets/{circle-pause-DEM4A1Y5.js → circle-pause-CbkWzBmG.js} +1 -1
- package/static/app/assets/{circle-play-C9djDuLd.js → circle-play-7lEaqHdJ.js} +1 -1
- package/static/app/assets/{cpu-DFdo1gw-.js → cpu-DAlCXlIy.js} +1 -1
- package/static/app/assets/{download-SnJL6oqk.js → download-RNhuuJwh.js} +1 -1
- package/static/app/assets/{folder-open-CqZeDkjE.js → folder-open-CLW4odzM.js} +1 -1
- package/static/app/assets/{hard-drive-j1jJXYYf.js → hard-drive-NKEiDIAJ.js} +1 -1
- package/static/app/assets/{index-_u5iUHDr.js → index-DMurvUuR.js} +3 -3
- package/static/app/assets/{input-B0lPdRQZ.js → input-D2UhPC1X.js} +1 -1
- package/static/app/assets/{link-2-CoFbooHS.js → link-2-6amKbP_P.js} +1 -1
- package/static/app/assets/{permissionCopy-BsyLxtao.js → permissionCopy-Cu9TZtdR.js} +1 -1
- package/static/app/assets/{primitives-DEbN-d6p.js → primitives-gPsccucr.js} +1 -1
- package/static/app/assets/search-Cj_TKk_2.js +1 -0
- package/static/app/assets/{share-2-CVtZ_ewX.js → share-2-Bau7KkPq.js} +1 -1
- package/static/app/assets/{shield-alert-CBi2GNWM.js → shield-alert-BufNYypi.js} +1 -1
- package/static/app/assets/{textarea-DNMpB5ih.js → textarea-BQnVWhYs.js} +1 -1
- package/static/app/assets/{useFocusTrap-C83t3GXF.js → useFocusTrap-B3_w60si.js} +1 -1
- package/static/app/assets/{useMutation-DtbJDoyz.js → useMutation-BHhCflT6.js} +1 -1
- package/static/app/assets/{useQuery-Dcp1OChy.js → useQuery-rBWfI-5t.js} +1 -1
- package/static/app/assets/{utils-BlZr7Pd4.js → utils-V_5-wxr5.js} +1 -1
- package/static/app/assets/{workspace-jJY4RuAV.js → workspace-K1zjYUHj.js} +1 -1
- package/static/app/index.html +3 -3
- package/static/sw.js +1 -1
- package/static/app/assets/arrow-left-DXvKg9U6.js +0 -1
- package/static/app/assets/circle-check-DfInj-qD.js +0 -1
- package/static/app/assets/search-BybIWPNd.js +0 -1
|
@@ -0,0 +1,259 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Build the committed Python↔Rust *chunking* parity fixture (v11.5.0 §2c).
|
|
3
|
+
|
|
4
|
+
``rust/lattice-ingest`` ports ``lattice_brain/graph/_kg_common/text.py`` — the
|
|
5
|
+
four chunking strategies, the strategy router, the chunk-id and content-hash
|
|
6
|
+
conventions, and the PDF page arithmetic. A port is only worth having if
|
|
7
|
+
something keeps proving it is still one, so this script is the Python half of
|
|
8
|
+
that proof: it runs the **real** ``typed_chunks`` / ``chunk_strategy_for`` /
|
|
9
|
+
``typed_chunk_meta_fields`` / ``pdf_page_offsets`` / ``page_for_offset`` /
|
|
10
|
+
``citation_locator`` over a deliberately awkward corpus and writes what they
|
|
11
|
+
answer to ``rust/fixtures/chunking/golden/``.
|
|
12
|
+
|
|
13
|
+
Two consumers read what it writes:
|
|
14
|
+
|
|
15
|
+
* ``tests/unit/test_chunking_parity_contract.py`` re-runs the Python functions
|
|
16
|
+
against the committed goldens, so a change to Python chunking semantics fails
|
|
17
|
+
loudly instead of silently invalidating the contract the Rust side is pinned
|
|
18
|
+
to;
|
|
19
|
+
* ``rust/lattice-ingest/tests/chunking_parity.rs`` runs the Rust port against
|
|
20
|
+
the same goldens, comparing exactly.
|
|
21
|
+
|
|
22
|
+
Determinism is free here — every function in the port is pure — so the only
|
|
23
|
+
design constraint is *coverage*. The inputs live next door in
|
|
24
|
+
``scripts/chunking_parity_corpus.py`` (this file is the runner; that one is the
|
|
25
|
+
specification), and they are shaped to reach every branch:
|
|
26
|
+
|
|
27
|
+
* all four strategies plus an unknown label (which must fall back to plain);
|
|
28
|
+
* empty, whitespace-only, exactly ``size``, ``size-1`` and ``size+1`` inputs;
|
|
29
|
+
* markdown with nested headings, an empty heading title, sub-200-char sections
|
|
30
|
+
that merge forward, a trailing undersized section that merges backward, and a
|
|
31
|
+
section too big for one window;
|
|
32
|
+
* code with declaration lines, blank-line runs, greedy packing, and a segment
|
|
33
|
+
past the ``size * 1.5`` hard limit;
|
|
34
|
+
* prose with strong (sentence) boundaries, weak (line-break) boundaries and no
|
|
35
|
+
boundary at all;
|
|
36
|
+
* **multibyte text straddling every boundary**, because Python slices strings by
|
|
37
|
+
*characters* and a Rust port that slices by bytes would silently disagree (or
|
|
38
|
+
panic). Every chunk records ``len_chars`` and ``len_bytes`` so the difference
|
|
39
|
+
is visible in the artefact rather than asserted from memory.
|
|
40
|
+
|
|
41
|
+
Usage::
|
|
42
|
+
|
|
43
|
+
.venv/bin/python scripts/generate_chunking_parity_fixtures.py
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
from __future__ import annotations
|
|
47
|
+
|
|
48
|
+
import importlib.util
|
|
49
|
+
import json
|
|
50
|
+
import shutil
|
|
51
|
+
import sys
|
|
52
|
+
from pathlib import Path
|
|
53
|
+
from typing import Any, Dict, List
|
|
54
|
+
|
|
55
|
+
REPO_ROOT = Path(__file__).resolve().parents[1]
|
|
56
|
+
if str(REPO_ROOT) not in sys.path:
|
|
57
|
+
sys.path.insert(0, str(REPO_ROOT))
|
|
58
|
+
|
|
59
|
+
FIXTURE_DIR = REPO_ROOT / "rust" / "fixtures" / "chunking"
|
|
60
|
+
GOLDEN_DIR = FIXTURE_DIR / "golden"
|
|
61
|
+
|
|
62
|
+
# The corpus lives beside this file. ``scripts`` is not a package, so it is
|
|
63
|
+
# loaded by path — the same trick tests/unit/test_chunking_parity_contract.py
|
|
64
|
+
# uses on this module, for the same reason.
|
|
65
|
+
_CORPUS_SPEC = importlib.util.spec_from_file_location(
|
|
66
|
+
"chunking_parity_corpus",
|
|
67
|
+
Path(__file__).resolve().parent / "chunking_parity_corpus.py",
|
|
68
|
+
)
|
|
69
|
+
corpus = importlib.util.module_from_spec(_CORPUS_SPEC)
|
|
70
|
+
_CORPUS_SPEC.loader.exec_module(corpus)
|
|
71
|
+
|
|
72
|
+
#: Re-exported so a consumer of this module (the contract test) sees one
|
|
73
|
+
#: surface rather than having to know the corpus moved.
|
|
74
|
+
CASES = corpus.CASES
|
|
75
|
+
STRATEGY_CASES = corpus.STRATEGY_CASES
|
|
76
|
+
PDF_STRUCTURES = corpus.PDF_STRUCTURES
|
|
77
|
+
PAGE_PROBES = corpus.PAGE_PROBES
|
|
78
|
+
LOCATOR_CASES = corpus.LOCATOR_CASES
|
|
79
|
+
TEXT_HASH_CASES = corpus.TEXT_HASH_CASES
|
|
80
|
+
FILE_HASH_CASES = corpus.FILE_HASH_CASES
|
|
81
|
+
VECTOR_TEXT_CASES = corpus.VECTOR_TEXT_CASES
|
|
82
|
+
GRAPHEME_SOUP = corpus.GRAPHEME_SOUP
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _dump(path: Path, payload: Any) -> None:
|
|
86
|
+
path.write_text(
|
|
87
|
+
json.dumps(payload, ensure_ascii=False, sort_keys=True, indent=2) + "\n",
|
|
88
|
+
encoding="utf-8",
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def case_defaults(case: Dict[str, Any]) -> Dict[str, Any]:
|
|
93
|
+
"""One case with every optional field resolved — the shared shape."""
|
|
94
|
+
from lattice_brain.graph._kg_common.text import chunk_strategy_for
|
|
95
|
+
|
|
96
|
+
filename = case.get("filename", "")
|
|
97
|
+
content_type = case.get("content_type", "")
|
|
98
|
+
strategy = case.get("strategy") or chunk_strategy_for(filename, content_type=content_type)
|
|
99
|
+
return {
|
|
100
|
+
"key": case["key"],
|
|
101
|
+
"filename": filename,
|
|
102
|
+
"content_type": content_type,
|
|
103
|
+
"requested_strategy": case.get("strategy"),
|
|
104
|
+
"strategy": strategy,
|
|
105
|
+
"size": case.get("size", 1200),
|
|
106
|
+
"overlap": case.get("overlap", 160),
|
|
107
|
+
"source_node_id": case.get("source_node_id", f"file:{case['key']}"),
|
|
108
|
+
"text": case["text"],
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def chunk_golden(case: Dict[str, Any]) -> Dict[str, Any]:
|
|
113
|
+
"""Run the real chunker over one case and record everything it produced."""
|
|
114
|
+
from lattice_brain.graph._kg_common.text import (
|
|
115
|
+
typed_chunk_meta_fields,
|
|
116
|
+
typed_chunks,
|
|
117
|
+
)
|
|
118
|
+
from lattice_brain.graph._kg_fsutil import _sha256_text
|
|
119
|
+
|
|
120
|
+
resolved = case_defaults(case)
|
|
121
|
+
pieces = typed_chunks(
|
|
122
|
+
resolved["text"],
|
|
123
|
+
strategy=resolved["strategy"],
|
|
124
|
+
size=resolved["size"],
|
|
125
|
+
overlap=resolved["overlap"],
|
|
126
|
+
)
|
|
127
|
+
node = resolved["source_node_id"]
|
|
128
|
+
chunks = []
|
|
129
|
+
for index, piece in enumerate(pieces):
|
|
130
|
+
text = piece["text"]
|
|
131
|
+
chunks.append(
|
|
132
|
+
{
|
|
133
|
+
"index": index,
|
|
134
|
+
"text": text,
|
|
135
|
+
"meta": piece["meta"],
|
|
136
|
+
"meta_fields": typed_chunk_meta_fields(piece),
|
|
137
|
+
"chunk_id": f"chunk:{_sha256_text(f'{node}:{index}:{text}')[:24]}",
|
|
138
|
+
"len_chars": len(text),
|
|
139
|
+
"len_bytes": len(text.encode("utf-8")),
|
|
140
|
+
}
|
|
141
|
+
)
|
|
142
|
+
cleaned = str(resolved["text"] or "").strip()
|
|
143
|
+
return {
|
|
144
|
+
**resolved,
|
|
145
|
+
"cleaned_len_chars": len(cleaned),
|
|
146
|
+
"cleaned_len_bytes": len(cleaned.encode("utf-8")),
|
|
147
|
+
"chunk_count": len(chunks),
|
|
148
|
+
"chunks": chunks,
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def strategy_golden() -> List[Dict[str, str]]:
|
|
153
|
+
from lattice_brain.graph._kg_common.text import chunk_strategy_for
|
|
154
|
+
|
|
155
|
+
return [
|
|
156
|
+
{**case, "expected": chunk_strategy_for(case["filename"], content_type=case["content_type"])}
|
|
157
|
+
for case in STRATEGY_CASES
|
|
158
|
+
]
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def pdf_golden() -> Dict[str, Any]:
|
|
162
|
+
from lattice_brain.graph._kg_common.text import (
|
|
163
|
+
citation_locator,
|
|
164
|
+
page_for_offset,
|
|
165
|
+
pdf_page_offsets,
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
structures = [
|
|
169
|
+
{"key": case["key"], "structure": case["structure"], "offsets": pdf_page_offsets(case["structure"])}
|
|
170
|
+
for case in PDF_STRUCTURES
|
|
171
|
+
]
|
|
172
|
+
by_key = {entry["key"]: entry["offsets"] for entry in structures}
|
|
173
|
+
probes = [
|
|
174
|
+
{
|
|
175
|
+
"offsets_key": key,
|
|
176
|
+
"offsets": by_key[key],
|
|
177
|
+
"probes": [{"offset": offset, "page": page_for_offset(by_key[key], offset)} for offset in PAGE_PROBES],
|
|
178
|
+
}
|
|
179
|
+
for key in ("three_pages", "single_page", "zero_length_page", "pages_empty")
|
|
180
|
+
]
|
|
181
|
+
return {
|
|
182
|
+
"structures": structures,
|
|
183
|
+
"page_for_offset": probes,
|
|
184
|
+
"citation_locator": [
|
|
185
|
+
{"metadata": metadata, "expected": citation_locator(metadata)} for metadata in LOCATOR_CASES
|
|
186
|
+
],
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def hash_golden() -> Dict[str, Any]:
|
|
191
|
+
from lattice_brain.graph._kg_common.text import _clean_text
|
|
192
|
+
from lattice_brain.graph._kg_fsutil import _sha256_bytes, _sha256_text
|
|
193
|
+
|
|
194
|
+
text_cases = []
|
|
195
|
+
for case in TEXT_HASH_CASES:
|
|
196
|
+
content_hash = _sha256_text(f"{case['source_type']}|{case['source_uri'] or ''}|{case['text']}")
|
|
197
|
+
identity = _sha256_text(f"{case['workspace_id'] or 'legacy-global'}|{content_hash}")
|
|
198
|
+
text_cases.append(
|
|
199
|
+
{
|
|
200
|
+
**case,
|
|
201
|
+
"content_hash": content_hash,
|
|
202
|
+
"identity_hash": identity,
|
|
203
|
+
"content_id": f"webdoc:{identity[:24]}",
|
|
204
|
+
}
|
|
205
|
+
)
|
|
206
|
+
return {
|
|
207
|
+
"sha256_text": [
|
|
208
|
+
{"text": text, "sha256": _sha256_text(text)}
|
|
209
|
+
for text in ["", "a", "회의 결정 사항", GRAPHEME_SOUP, "x" * 1000]
|
|
210
|
+
],
|
|
211
|
+
"file_content_hash": [
|
|
212
|
+
{"bytes_hex": payload.hex(), "sha256": _sha256_bytes(payload)} for payload in FILE_HASH_CASES
|
|
213
|
+
],
|
|
214
|
+
"text_content_hash": text_cases,
|
|
215
|
+
"vector_text_hash": [
|
|
216
|
+
{"text": text, "cleaned": _clean_text(text), "text_hash": _sha256_text(_clean_text(text))}
|
|
217
|
+
for text in VECTOR_TEXT_CASES
|
|
218
|
+
],
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def manifest() -> Dict[str, Any]:
|
|
223
|
+
cases = [case_defaults(case) for case in CASES]
|
|
224
|
+
strategies = sorted({case["strategy"] for case in cases})
|
|
225
|
+
return {
|
|
226
|
+
"source": "lattice_brain/graph/_kg_common/text.py",
|
|
227
|
+
"defaults": {"size": 1200, "overlap": 160, "markdown_min_section_chars": 200},
|
|
228
|
+
"strategies": strategies,
|
|
229
|
+
"cases": [{key: case[key] for key in ("key", "filename", "content_type", "strategy", "size", "overlap", "source_node_id")} for case in cases],
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def main() -> int:
|
|
234
|
+
if GOLDEN_DIR.exists():
|
|
235
|
+
shutil.rmtree(GOLDEN_DIR)
|
|
236
|
+
GOLDEN_DIR.mkdir(parents=True, exist_ok=True)
|
|
237
|
+
keys = [case["key"] for case in CASES]
|
|
238
|
+
if len(set(keys)) != len(keys):
|
|
239
|
+
raise SystemExit("duplicate case keys")
|
|
240
|
+
total_chunks = 0
|
|
241
|
+
multibyte_cases = 0
|
|
242
|
+
for case in CASES:
|
|
243
|
+
golden = chunk_golden(case)
|
|
244
|
+
total_chunks += golden["chunk_count"]
|
|
245
|
+
if any(chunk["len_bytes"] != chunk["len_chars"] for chunk in golden["chunks"]):
|
|
246
|
+
multibyte_cases += 1
|
|
247
|
+
_dump(GOLDEN_DIR / f"chunks__{case['key']}.json", golden)
|
|
248
|
+
_dump(GOLDEN_DIR / "strategy_golden.json", strategy_golden())
|
|
249
|
+
_dump(GOLDEN_DIR / "pdf_golden.json", pdf_golden())
|
|
250
|
+
_dump(GOLDEN_DIR / "hash_golden.json", hash_golden())
|
|
251
|
+
_dump(GOLDEN_DIR / "manifest.json", manifest())
|
|
252
|
+
print(f"golden: {len(CASES)} chunking cases ({total_chunks} chunks), {multibyte_cases} with multibyte chunks")
|
|
253
|
+
print(f" {len(STRATEGY_CASES)} strategy cases, {len(PDF_STRUCTURES)} pdf structures")
|
|
254
|
+
print(f" written to {GOLDEN_DIR.relative_to(REPO_ROOT)}")
|
|
255
|
+
return 0
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
if __name__ == "__main__":
|
|
259
|
+
raise SystemExit(main())
|