ltcai 11.2.0 → 11.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -53
- package/docs/CHANGELOG.md +87 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/MULTI_AGENT_RUNTIME.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +6 -2
- package/docs/PERMISSION_MODE.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +2 -2
- package/docs/v11.3.0_PLAN.md +202 -0
- package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +181 -0
- package/docs/v11.5.0_RUST_COMPLETE_PLAN.md +145 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +287 -0
- package/lattice_brain/graph/_kg_common/extraction.py +516 -0
- package/lattice_brain/graph/_kg_common/relations.py +161 -0
- package/lattice_brain/graph/_kg_common/text.py +479 -0
- package/lattice_brain/graph/discovery_index/__init__.py +35 -0
- package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
- package/lattice_brain/graph/discovery_index/extract.py +137 -0
- package/lattice_brain/graph/discovery_index/scan.py +411 -0
- package/lattice_brain/graph/discovery_index/upsert.py +495 -0
- package/lattice_brain/graph/projection/__init__.py +42 -0
- package/lattice_brain/graph/projection/curation.py +500 -0
- package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
- package/lattice_brain/graph/retrieval/__init__.py +54 -0
- package/lattice_brain/graph/retrieval/context.py +197 -0
- package/lattice_brain/graph/retrieval/graph_view.py +319 -0
- package/lattice_brain/graph/retrieval/hybrid.py +488 -0
- package/lattice_brain/graph/retrieval/maintenance.py +121 -0
- package/lattice_brain/graph/retrieval/signals.py +95 -0
- package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
- package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
- package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
- package/lattice_brain/graph/retrieval_vector/search.py +560 -0
- package/lattice_brain/graph/retrieval_vector/status.py +374 -0
- package/lattice_brain/ingestion/__init__.py +130 -0
- package/lattice_brain/ingestion/_contract.py +90 -0
- package/lattice_brain/ingestion/constants.py +127 -0
- package/lattice_brain/ingestion/folder_scan.py +57 -0
- package/lattice_brain/ingestion/folders.py +258 -0
- package/lattice_brain/ingestion/hashing.py +26 -0
- package/lattice_brain/ingestion/jobs_api.py +107 -0
- package/lattice_brain/ingestion/models.py +80 -0
- package/lattice_brain/ingestion/pipeline.py +486 -0
- package/lattice_brain/ingestion/quality.py +209 -0
- package/lattice_brain/ingestion/routing.py +295 -0
- package/lattice_brain/multimodal/__init__.py +164 -0
- package/lattice_brain/multimodal/audio.py +77 -0
- package/lattice_brain/multimodal/common.py +118 -0
- package/lattice_brain/multimodal/images.py +498 -0
- package/lattice_brain/multimodal/ports.py +169 -0
- package/lattice_brain/multimodal/video.py +410 -0
- package/lattice_brain/portability/__init__.py +90 -0
- package/lattice_brain/portability/_contract.py +42 -0
- package/lattice_brain/portability/backups.py +338 -0
- package/lattice_brain/portability/bundles.py +136 -0
- package/lattice_brain/portability/constants.py +93 -0
- package/lattice_brain/portability/fsops.py +138 -0
- package/lattice_brain/portability/service.py +41 -0
- package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
- package/lattice_brain/runtime/__init__.py +1 -1
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/chronicle.py +63 -0
- package/latticeai/api/index_jobs.py +145 -0
- package/latticeai/core/agent/__init__.py +93 -0
- package/latticeai/core/agent/_contract.py +79 -0
- package/latticeai/core/agent/context.py +57 -0
- package/latticeai/core/agent/deps.py +125 -0
- package/latticeai/core/agent/execution.py +622 -0
- package/latticeai/core/agent/planning.py +145 -0
- package/latticeai/core/agent/recovery.py +157 -0
- package/latticeai/core/agent/runtime.py +210 -0
- package/latticeai/core/agent/verification.py +231 -0
- package/latticeai/core/embedding_providers/__init__.py +151 -0
- package/latticeai/core/embedding_providers/base.py +199 -0
- package/latticeai/core/embedding_providers/captions.py +162 -0
- package/latticeai/core/embedding_providers/profiles.py +126 -0
- package/latticeai/core/embedding_providers/text.py +350 -0
- package/latticeai/core/embedding_providers/vision.py +352 -0
- package/latticeai/core/file_generation/__init__.py +115 -0
- package/latticeai/core/file_generation/bundles.py +76 -0
- package/latticeai/core/file_generation/extraction.py +154 -0
- package/latticeai/core/file_generation/inference.py +235 -0
- package/latticeai/core/file_generation/orchestration.py +152 -0
- package/latticeai/core/file_generation/prompting.py +117 -0
- package/latticeai/core/file_generation/repair.py +114 -0
- package/latticeai/core/file_generation/sanitize.py +61 -0
- package/latticeai/core/file_generation/validation.py +201 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +14 -0
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/integrations/telegram_bot/__init__.py +123 -0
- package/latticeai/integrations/telegram_bot/__main__.py +17 -0
- package/latticeai/integrations/telegram_bot/config.py +86 -0
- package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
- package/latticeai/integrations/telegram_bot/flows.py +478 -0
- package/latticeai/integrations/telegram_bot/helpers.py +322 -0
- package/latticeai/integrations/telegram_bot/screens.py +394 -0
- package/latticeai/models/router/__init__.py +88 -0
- package/latticeai/models/router/_contract.py +66 -0
- package/latticeai/models/router/branding.py +56 -0
- package/latticeai/models/router/catalog.py +69 -0
- package/latticeai/models/router/documents.py +199 -0
- package/latticeai/models/router/errors.py +37 -0
- package/latticeai/models/router/generation.py +258 -0
- package/latticeai/models/router/loading.py +291 -0
- package/latticeai/models/router/local_models.py +85 -0
- package/latticeai/models/router/registry.py +147 -0
- package/latticeai/runtime/build_phases/__init__.py +82 -0
- package/latticeai/runtime/build_phases/features.py +421 -0
- package/latticeai/runtime/build_phases/foundation.py +555 -0
- package/latticeai/runtime/build_phases/web.py +492 -0
- package/latticeai/runtime/runtime_context.py +1 -0
- package/latticeai/services/architecture_readiness.py +48 -19
- package/latticeai/services/brain_intelligence/__init__.py +58 -0
- package/latticeai/services/brain_intelligence/_contract.py +71 -0
- package/latticeai/services/brain_intelligence/consistency.py +193 -0
- package/latticeai/services/brain_intelligence/constants.py +47 -0
- package/latticeai/services/brain_intelligence/digest.py +258 -0
- package/latticeai/services/brain_intelligence/health.py +331 -0
- package/latticeai/services/brain_intelligence/proposals.py +264 -0
- package/latticeai/services/brain_intelligence/sampling.py +84 -0
- package/latticeai/services/brain_intelligence/service.py +48 -0
- package/latticeai/services/chronicle.py +557 -0
- package/latticeai/services/memory_service/__init__.py +52 -0
- package/latticeai/services/memory_service/_contract.py +100 -0
- package/latticeai/services/memory_service/brief.py +431 -0
- package/latticeai/services/memory_service/constants.py +57 -0
- package/latticeai/services/memory_service/maintenance.py +138 -0
- package/latticeai/services/memory_service/manager.py +186 -0
- package/latticeai/services/memory_service/proof.py +136 -0
- package/latticeai/services/memory_service/recall.py +225 -0
- package/latticeai/services/memory_service/service.py +48 -0
- package/latticeai/services/memory_service/stores.py +110 -0
- package/latticeai/services/model_runtime/__init__.py +322 -0
- package/latticeai/services/model_runtime/cloud.py +87 -0
- package/latticeai/services/model_runtime/download.py +282 -0
- package/latticeai/services/model_runtime/engines.py +341 -0
- package/latticeai/services/model_runtime/loading.py +178 -0
- package/latticeai/services/model_runtime/service.py +129 -0
- package/latticeai/services/model_runtime/state.py +131 -0
- package/latticeai/services/model_runtime/status.py +255 -0
- package/latticeai/services/product_readiness.py +15 -7
- package/latticeai/setup/wizard/__init__.py +126 -0
- package/latticeai/setup/wizard/catalog.py +172 -0
- package/latticeai/setup/wizard/detect.py +323 -0
- package/latticeai/setup/wizard/install.py +348 -0
- package/latticeai/setup/wizard/paths.py +168 -0
- package/latticeai/setup/wizard/plans.py +74 -0
- package/latticeai/setup/wizard/recommend.py +320 -0
- package/package.json +6 -2
- package/scripts/bump_version.py +14 -0
- package/scripts/capture_release_evidence.mjs +33 -21
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_i18n_namespace_coverage.mjs +41 -4
- package/scripts/check_max_file_lines.mjs +102 -0
- package/scripts/check_release_evidence_bound.mjs +30 -15
- package/scripts/check_screenshot_pixel_delta.py +34 -4
- package/scripts/check_server_i18n.mjs +2 -0
- package/scripts/chunking_parity_corpus.py +449 -0
- package/scripts/generate_agent_parity_fixtures.py +752 -0
- package/scripts/generate_chunking_parity_fixtures.py +259 -0
- package/scripts/generate_rust_parity_fixtures.py +997 -0
- package/scripts/lib/mock_server_fingerprint.mjs +94 -0
- package/scripts/release_screen_claims.json +42 -2
- package/src-tauri/Cargo.lock +404 -3
- package/src-tauri/Cargo.toml +13 -1
- package/src-tauri/src/backend.rs +460 -0
- package/src-tauri/src/folder.rs +33 -0
- package/src-tauri/src/main.rs +109 -399
- package/src-tauri/src/topology.rs +356 -0
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +41 -37
- package/static/app/assets/Act-CWnxSCgN.js +1 -0
- package/static/app/assets/AdminConsole-BEQYU6kF.js +1 -0
- package/static/app/assets/{Brain-tuhI4sOC.js → Brain-DWu1BhFg.js} +2 -2
- package/static/app/assets/BrainHome-95Hilr9R.js +2 -0
- package/static/app/assets/BrainSignals-QdeqCpAF.js +1 -0
- package/static/app/assets/Capture-BHpCxnzb.js +1 -0
- package/static/app/assets/Chronicle-B4xYKoed.js +1 -0
- package/static/app/assets/CommandPalette-BVXnttSz.js +1 -0
- package/static/app/assets/Library-DgYcHome.js +1 -0
- package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-CrJLDbf7.js} +1 -1
- package/static/app/assets/ProductFlow-DFlScKoJ.js +1 -0
- package/static/app/assets/ReviewCard-Cy5f48Pj.js +3 -0
- package/static/app/assets/System-NF8IfhTa.js +1 -0
- package/static/app/assets/arrow-left-DwkSYrjR.js +1 -0
- package/static/app/assets/{bot-Cia42c2h.js → bot-CucuhLhm.js} +1 -1
- package/static/app/assets/brain-BBnSryW_.js +1 -0
- package/static/app/assets/{button-2j2Ijzgq.js → button-C2GUj2Ai.js} +1 -1
- package/static/app/assets/circle-check-CxOVPwYq.js +1 -0
- package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-CbkWzBmG.js} +1 -1
- package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-7lEaqHdJ.js} +1 -1
- package/static/app/assets/{cpu-k4awryFq.js → cpu-DAlCXlIy.js} +1 -1
- package/static/app/assets/{download-DFbLJ_ig.js → download-RNhuuJwh.js} +1 -1
- package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CLW4odzM.js} +1 -1
- package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-NKEiDIAJ.js} +1 -1
- package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
- package/static/app/assets/index-DMurvUuR.js +10 -0
- package/static/app/assets/input-D2UhPC1X.js +1 -0
- package/static/app/assets/link-2-6amKbP_P.js +1 -0
- package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-Cu9TZtdR.js} +1 -1
- package/static/app/assets/primitives-gPsccucr.js +1 -0
- package/static/app/assets/search-Cj_TKk_2.js +1 -0
- package/static/app/assets/{share-2-BH1M-WNi.js → share-2-Bau7KkPq.js} +1 -1
- package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-BufNYypi.js} +1 -1
- package/static/app/assets/{textarea-CCWbUfFB.js → textarea-BQnVWhYs.js} +1 -1
- package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-B3_w60si.js} +1 -1
- package/static/app/assets/useMutation-BHhCflT6.js +1 -0
- package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-rBWfI-5t.js} +1 -1
- package/static/app/assets/utils-V_5-wxr5.js +4 -0
- package/static/app/assets/workspace-K1zjYUHj.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/graph/_kg_common.py +0 -1331
- package/lattice_brain/graph/discovery_index.py +0 -1141
- package/lattice_brain/graph/retrieval.py +0 -1120
- package/lattice_brain/graph/retrieval_vector.py +0 -1293
- package/lattice_brain/ingestion.py +0 -1525
- package/lattice_brain/multimodal.py +0 -1258
- package/latticeai/core/agent.py +0 -1465
- package/latticeai/core/embedding_providers.py +0 -1196
- package/latticeai/core/file_generation.py +0 -1047
- package/latticeai/integrations/telegram_bot.py +0 -1390
- package/latticeai/models/router.py +0 -1007
- package/latticeai/runtime/build_phases.py +0 -1450
- package/latticeai/services/brain_intelligence.py +0 -1083
- package/latticeai/services/memory_service.py +0 -1177
- package/latticeai/services/model_runtime.py +0 -1281
- package/latticeai/setup/wizard.py +0 -1310
- package/static/app/assets/Act-AWf0SAKp.js +0 -1
- package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
- package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
- package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
- package/static/app/assets/Capture-CqOSzyPr.js +0 -1
- package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
- package/static/app/assets/Library-CX-bbhmK.js +0 -1
- package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
- package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
- package/static/app/assets/System-Bu2t5hn1.js +0 -1
- package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
- package/static/app/assets/brain-DJMoqrwx.js +0 -1
- package/static/app/assets/index-BpYkzcVm.js +0 -10
- package/static/app/assets/input-DSlJJxRs.js +0 -1
- package/static/app/assets/primitives-BCx6TvfG.js +0 -1
- package/static/app/assets/search-Cgy8cCFJ.js +0 -1
- package/static/app/assets/utils-zqPZJxdx.js +0 -4
- package/static/app/assets/workspace-DXTihhfU.js +0 -1
|
@@ -0,0 +1,479 @@
|
|
|
1
|
+
"""Text cleaning, chunking, and citation-locator maths.
|
|
2
|
+
|
|
3
|
+
Moved verbatim out of the ``_kg_common`` grab-bag (v11.3.0 decomposition).
|
|
4
|
+
Nothing here reaches back into the rest of the package — the import graph is
|
|
5
|
+
``text ← relations ← extraction ← __init__`` — so this is the layer every
|
|
6
|
+
other one may build on.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import re
|
|
12
|
+
from typing import Any, Dict, List, Optional, Tuple
|
|
13
|
+
|
|
14
|
+
from ...quiet import quiet
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _clean_text(text: str) -> str:
|
|
18
|
+
return re.sub(r"\s+", " ", str(text or "")).strip()
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _chunks(text: str, size: int = 1200, overlap: int = 160) -> List[str]:
|
|
22
|
+
cleaned = str(text or "").strip()
|
|
23
|
+
if not cleaned:
|
|
24
|
+
return []
|
|
25
|
+
chunks: List[str] = []
|
|
26
|
+
start = 0
|
|
27
|
+
while start < len(cleaned):
|
|
28
|
+
end = min(len(cleaned), start + size)
|
|
29
|
+
chunks.append(cleaned[start:end])
|
|
30
|
+
if end >= len(cleaned):
|
|
31
|
+
break
|
|
32
|
+
start = max(0, end - overlap)
|
|
33
|
+
return chunks
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
# ── Typed chunking (review 2026-07-25 §5.2 S2 — Wave 2.1 + 2.4) ──────────────
|
|
37
|
+
# ``_chunks`` above is a compatibility contract (chunk ids hash over the chunk
|
|
38
|
+
# text) and stays byte-for-byte untouched. ``typed_chunks`` layers strategy-
|
|
39
|
+
# aware boundaries plus per-chunk provenance (start_char / heading_path) on
|
|
40
|
+
# top; ``strategy="plain"`` reproduces the exact ``_chunks`` boundaries so
|
|
41
|
+
# unchanged plain content keeps identical chunk ids.
|
|
42
|
+
|
|
43
|
+
_MARKDOWN_CHUNK_EXTENSIONS = {".md", ".markdown"}
|
|
44
|
+
_CODE_CHUNK_EXTENSIONS = {
|
|
45
|
+
".py", ".js", ".jsx", ".ts", ".tsx", ".go", ".rs", ".java", ".rb",
|
|
46
|
+
".c", ".h", ".cpp", ".css", ".sh", ".sql", ".vue", ".svelte",
|
|
47
|
+
".json", ".yaml", ".yml", ".toml",
|
|
48
|
+
}
|
|
49
|
+
_PROSE_CHUNK_EXTENSIONS = {
|
|
50
|
+
".txt", ".pdf", ".docx", ".doc", ".rtf", ".odt", ".epub", ".html", ".htm",
|
|
51
|
+
}
|
|
52
|
+
_CHUNK_STRATEGIES = {"plain", "markdown", "code", "prose"}
|
|
53
|
+
# Markdown sections smaller than this merge forward into the next section so
|
|
54
|
+
# heading-dense documents don't shatter into confetti chunks.
|
|
55
|
+
_MARKDOWN_MIN_SECTION_CHARS = 200
|
|
56
|
+
_MARKDOWN_HEADING_RE = re.compile(r"^(#{1,6}) (.*)$", re.MULTILINE)
|
|
57
|
+
_CODE_BOUNDARY_LINE_RE = re.compile(
|
|
58
|
+
r"^(?:def |class |function |export |const |public |private )", re.MULTILINE
|
|
59
|
+
)
|
|
60
|
+
_CODE_BLANK_RUN_RE = re.compile(r"\n\s*\n")
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def chunk_strategy_for(filename: Any, *, content_type: str = "") -> str:
|
|
64
|
+
"""Route a filename / path / URI (plus optional MIME hint) to a strategy.
|
|
65
|
+
|
|
66
|
+
Returns ``"markdown"`` for .md/.markdown, ``"code"`` for known source-code
|
|
67
|
+
extensions, ``"prose"`` for document formats whose text is running prose
|
|
68
|
+
(.txt/.pdf/.docx/.html/…), ``"plain"`` otherwise. Case-insensitive,
|
|
69
|
+
tolerant of URLs (query/fragment stripped) and ``Path`` objects; never
|
|
70
|
+
raises — any malformed input falls back to ``"plain"``.
|
|
71
|
+
|
|
72
|
+
Unknown/extension-less input stays ``"plain"`` on purpose: the plain
|
|
73
|
+
strategy is the byte-compatible legacy walk, and guessing prose for
|
|
74
|
+
something that might be a data dump would move chunk boundaries for no
|
|
75
|
+
retrieval gain.
|
|
76
|
+
"""
|
|
77
|
+
try:
|
|
78
|
+
name = str(filename or "").strip().lower()
|
|
79
|
+
for sep in ("?", "#"):
|
|
80
|
+
name = name.split(sep, 1)[0]
|
|
81
|
+
name = name.replace("\\", "/").rstrip("/").rsplit("/", 1)[-1]
|
|
82
|
+
dot = name.rfind(".")
|
|
83
|
+
ext = name[dot:] if dot > 0 else ""
|
|
84
|
+
if ext in _MARKDOWN_CHUNK_EXTENSIONS:
|
|
85
|
+
return "markdown"
|
|
86
|
+
if ext in _CODE_CHUNK_EXTENSIONS:
|
|
87
|
+
return "code"
|
|
88
|
+
if ext in _PROSE_CHUNK_EXTENSIONS:
|
|
89
|
+
return "prose"
|
|
90
|
+
mime = str(content_type or "").strip().lower()
|
|
91
|
+
if "markdown" in mime:
|
|
92
|
+
return "markdown"
|
|
93
|
+
if mime.startswith("text/html") or mime.startswith("text/plain"):
|
|
94
|
+
return "prose"
|
|
95
|
+
except Exception:
|
|
96
|
+
quiet()
|
|
97
|
+
return "plain"
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _plain_windows(
|
|
101
|
+
cleaned: str,
|
|
102
|
+
size: int,
|
|
103
|
+
overlap: int,
|
|
104
|
+
*,
|
|
105
|
+
base_offset: int = 0,
|
|
106
|
+
strategy: str = "plain",
|
|
107
|
+
heading_path: Optional[str] = None,
|
|
108
|
+
) -> List[Dict[str, Any]]:
|
|
109
|
+
"""The exact ``_chunks`` walk with ``start_char`` tracked.
|
|
110
|
+
|
|
111
|
+
Boundaries and chunk texts are byte-identical to ``_chunks`` over the same
|
|
112
|
+
string — this is the plain-strategy compatibility guarantee.
|
|
113
|
+
"""
|
|
114
|
+
out: List[Dict[str, Any]] = []
|
|
115
|
+
start = 0
|
|
116
|
+
total = len(cleaned)
|
|
117
|
+
while start < total:
|
|
118
|
+
end = min(total, start + size)
|
|
119
|
+
out.append(
|
|
120
|
+
{
|
|
121
|
+
"text": cleaned[start:end],
|
|
122
|
+
"meta": {
|
|
123
|
+
"strategy": strategy,
|
|
124
|
+
"start_char": base_offset + start,
|
|
125
|
+
"heading_path": heading_path,
|
|
126
|
+
},
|
|
127
|
+
}
|
|
128
|
+
)
|
|
129
|
+
if end >= total:
|
|
130
|
+
break
|
|
131
|
+
start = max(0, end - overlap)
|
|
132
|
+
return out
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _markdown_section_spans(cleaned: str) -> List[Tuple[int, int, Optional[str]]]:
|
|
136
|
+
"""``(start, end, heading_path)`` spans split at ``^#{1,6} `` heading lines.
|
|
137
|
+
|
|
138
|
+
``heading_path`` is the " > "-joined path of the enclosing headings
|
|
139
|
+
including the section's own heading (e.g. ``"Guide > Setup"``); the
|
|
140
|
+
preamble before the first heading carries ``None``. Spans are contiguous
|
|
141
|
+
raw slices of ``cleaned`` so every chunk text round-trips via start_char.
|
|
142
|
+
"""
|
|
143
|
+
spans: List[Tuple[int, int, Optional[str]]] = []
|
|
144
|
+
stack: List[Tuple[int, str]] = []
|
|
145
|
+
prev_start = 0
|
|
146
|
+
prev_path: Optional[str] = None
|
|
147
|
+
for match in _MARKDOWN_HEADING_RE.finditer(cleaned):
|
|
148
|
+
offset = match.start()
|
|
149
|
+
if offset > prev_start:
|
|
150
|
+
spans.append((prev_start, offset, prev_path))
|
|
151
|
+
level = len(match.group(1))
|
|
152
|
+
while stack and stack[-1][0] >= level:
|
|
153
|
+
stack.pop()
|
|
154
|
+
stack.append((level, match.group(2).strip()))
|
|
155
|
+
prev_start = offset
|
|
156
|
+
prev_path = " > ".join(title for _, title in stack) or None
|
|
157
|
+
if len(cleaned) > prev_start:
|
|
158
|
+
spans.append((prev_start, len(cleaned), prev_path))
|
|
159
|
+
return spans
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def _merge_small_sections(
|
|
163
|
+
spans: List[Tuple[int, int, Optional[str]]], min_chars: int
|
|
164
|
+
) -> List[Tuple[int, int, Optional[str]]]:
|
|
165
|
+
"""Merge sections under ``min_chars`` forward into the next section.
|
|
166
|
+
|
|
167
|
+
A merged section keeps the heading_path of its first constituent (the
|
|
168
|
+
path in effect at the chunk start). A trailing undersized section merges
|
|
169
|
+
backward into the previous emitted section when one exists.
|
|
170
|
+
"""
|
|
171
|
+
merged: List[Tuple[int, int, Optional[str]]] = []
|
|
172
|
+
pending: Optional[Tuple[int, int, Optional[str]]] = None
|
|
173
|
+
for start, end, path in spans:
|
|
174
|
+
if pending is None:
|
|
175
|
+
pending = (start, end, path)
|
|
176
|
+
else:
|
|
177
|
+
pending = (pending[0], end, pending[2])
|
|
178
|
+
if pending[1] - pending[0] >= min_chars:
|
|
179
|
+
merged.append(pending)
|
|
180
|
+
pending = None
|
|
181
|
+
if pending is not None:
|
|
182
|
+
if merged and pending[1] - pending[0] < min_chars:
|
|
183
|
+
last = merged.pop()
|
|
184
|
+
merged.append((last[0], pending[1], last[2]))
|
|
185
|
+
else:
|
|
186
|
+
merged.append(pending)
|
|
187
|
+
return merged
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def _markdown_chunks(cleaned: str, size: int, overlap: int) -> List[Dict[str, Any]]:
|
|
191
|
+
sections = _merge_small_sections(
|
|
192
|
+
_markdown_section_spans(cleaned), _MARKDOWN_MIN_SECTION_CHARS
|
|
193
|
+
)
|
|
194
|
+
out: List[Dict[str, Any]] = []
|
|
195
|
+
for start, end, path in sections:
|
|
196
|
+
body = cleaned[start:end]
|
|
197
|
+
if len(body) <= size:
|
|
198
|
+
out.append(
|
|
199
|
+
{
|
|
200
|
+
"text": body,
|
|
201
|
+
"meta": {
|
|
202
|
+
"strategy": "markdown",
|
|
203
|
+
"start_char": start,
|
|
204
|
+
"heading_path": path,
|
|
205
|
+
},
|
|
206
|
+
}
|
|
207
|
+
)
|
|
208
|
+
else:
|
|
209
|
+
out.extend(
|
|
210
|
+
_plain_windows(
|
|
211
|
+
body,
|
|
212
|
+
size,
|
|
213
|
+
overlap,
|
|
214
|
+
base_offset=start,
|
|
215
|
+
strategy="markdown",
|
|
216
|
+
heading_path=path,
|
|
217
|
+
)
|
|
218
|
+
)
|
|
219
|
+
return out
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _code_segment_spans(cleaned: str) -> List[Tuple[int, int]]:
|
|
223
|
+
"""Contiguous top-level segments split at blank-line runs and decl lines."""
|
|
224
|
+
boundaries = {0, len(cleaned)}
|
|
225
|
+
for match in _CODE_BLANK_RUN_RE.finditer(cleaned):
|
|
226
|
+
boundaries.add(match.end())
|
|
227
|
+
for match in _CODE_BOUNDARY_LINE_RE.finditer(cleaned):
|
|
228
|
+
boundaries.add(match.start())
|
|
229
|
+
ordered = sorted(boundaries)
|
|
230
|
+
return [
|
|
231
|
+
(ordered[i], ordered[i + 1])
|
|
232
|
+
for i in range(len(ordered) - 1)
|
|
233
|
+
if ordered[i + 1] > ordered[i]
|
|
234
|
+
]
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def _code_chunks(cleaned: str, size: int, overlap: int) -> List[Dict[str, Any]]:
|
|
238
|
+
hard_limit = int(size * 1.5)
|
|
239
|
+
out: List[Dict[str, Any]] = []
|
|
240
|
+
pack: Optional[Tuple[int, int]] = None
|
|
241
|
+
|
|
242
|
+
def _emit(span: Tuple[int, int]) -> None:
|
|
243
|
+
out.append(
|
|
244
|
+
{
|
|
245
|
+
"text": cleaned[span[0] : span[1]],
|
|
246
|
+
"meta": {
|
|
247
|
+
"strategy": "code",
|
|
248
|
+
"start_char": span[0],
|
|
249
|
+
"heading_path": None,
|
|
250
|
+
},
|
|
251
|
+
}
|
|
252
|
+
)
|
|
253
|
+
|
|
254
|
+
for start, end in _code_segment_spans(cleaned):
|
|
255
|
+
if end - start > hard_limit:
|
|
256
|
+
# Monster segment: flush the pack, then window it like plain text.
|
|
257
|
+
if pack is not None:
|
|
258
|
+
_emit(pack)
|
|
259
|
+
pack = None
|
|
260
|
+
out.extend(
|
|
261
|
+
_plain_windows(
|
|
262
|
+
cleaned[start:end],
|
|
263
|
+
size,
|
|
264
|
+
overlap,
|
|
265
|
+
base_offset=start,
|
|
266
|
+
strategy="code",
|
|
267
|
+
)
|
|
268
|
+
)
|
|
269
|
+
continue
|
|
270
|
+
if pack is None:
|
|
271
|
+
pack = (start, end)
|
|
272
|
+
elif end - pack[0] <= size:
|
|
273
|
+
pack = (pack[0], end)
|
|
274
|
+
else:
|
|
275
|
+
_emit(pack)
|
|
276
|
+
pack = (start, end)
|
|
277
|
+
if pack is not None:
|
|
278
|
+
_emit(pack)
|
|
279
|
+
return out
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
# ── Prose chunking (review 2026-07-27 P1 #4) ────────────────────────────────
|
|
283
|
+
# The plain walk cuts every ``size`` characters, which lands mid-sentence and
|
|
284
|
+
# — for Korean, where the verb carrying the meaning sits at the end — routinely
|
|
285
|
+
# splits a claim from its predicate. Retrieval then matches half a statement
|
|
286
|
+
# and the citation shows a fragment. The prose strategy keeps the same window
|
|
287
|
+
# budget but ends each chunk at the last sentence/paragraph boundary inside it.
|
|
288
|
+
|
|
289
|
+
# Strong: sentence-final punctuation (ASCII + CJK) with optional closing
|
|
290
|
+
# quotes/brackets, followed by whitespace; or a blank-line paragraph break.
|
|
291
|
+
_PROSE_STRONG_BOUNDARY_RE = re.compile(
|
|
292
|
+
r"(?:[.!?。!?…]+[\"'”’」』\)\]]*\s+|\n[ \t]*\n)"
|
|
293
|
+
)
|
|
294
|
+
# Weak: a single line break. Korean notes and bullet lists often carry no
|
|
295
|
+
# sentence punctuation at all; a line end is still a real boundary there.
|
|
296
|
+
_PROSE_WEAK_BOUNDARY_RE = re.compile(r"\n")
|
|
297
|
+
# Never emit a chunk shorter than this fraction of ``size`` just to hit a
|
|
298
|
+
# boundary — tiny chunks hurt recall more than a mid-sentence cut.
|
|
299
|
+
_PROSE_MIN_SPAN_RATIO = 0.5
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def _last_boundary(cleaned: str, lo: int, hi: int) -> Optional[int]:
|
|
303
|
+
"""End offset of the last sentence/paragraph boundary in ``cleaned[lo:hi]``.
|
|
304
|
+
|
|
305
|
+
Strong boundaries win; a single line break is the fallback. Returns None
|
|
306
|
+
when the span holds neither, so the caller keeps the hard window cut.
|
|
307
|
+
"""
|
|
308
|
+
window = cleaned[lo:hi]
|
|
309
|
+
for pattern in (_PROSE_STRONG_BOUNDARY_RE, _PROSE_WEAK_BOUNDARY_RE):
|
|
310
|
+
last = None
|
|
311
|
+
for match in pattern.finditer(window):
|
|
312
|
+
last = match.end()
|
|
313
|
+
if last:
|
|
314
|
+
return lo + last
|
|
315
|
+
return None
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def _prose_chunks(cleaned: str, size: int, overlap: int) -> List[Dict[str, Any]]:
|
|
319
|
+
out: List[Dict[str, Any]] = []
|
|
320
|
+
total = len(cleaned)
|
|
321
|
+
min_span = max(1, int(size * _PROSE_MIN_SPAN_RATIO))
|
|
322
|
+
start = 0
|
|
323
|
+
while start < total:
|
|
324
|
+
hard_end = min(total, start + size)
|
|
325
|
+
end = hard_end
|
|
326
|
+
if hard_end < total:
|
|
327
|
+
boundary = _last_boundary(cleaned, start + min_span, hard_end)
|
|
328
|
+
if boundary is not None and boundary > start:
|
|
329
|
+
end = boundary
|
|
330
|
+
out.append(
|
|
331
|
+
{
|
|
332
|
+
"text": cleaned[start:end],
|
|
333
|
+
"meta": {
|
|
334
|
+
"strategy": "prose",
|
|
335
|
+
"start_char": start,
|
|
336
|
+
"heading_path": None,
|
|
337
|
+
},
|
|
338
|
+
}
|
|
339
|
+
)
|
|
340
|
+
if end >= total:
|
|
341
|
+
break
|
|
342
|
+
# Overlap carries the tail of the previous chunk into the next one so
|
|
343
|
+
# a claim split across a boundary is still retrievable from both.
|
|
344
|
+
start = max(start + 1, end - overlap)
|
|
345
|
+
return out
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
def typed_chunks(
|
|
349
|
+
text: str,
|
|
350
|
+
*,
|
|
351
|
+
strategy: str = "plain",
|
|
352
|
+
size: int = 1200,
|
|
353
|
+
overlap: int = 160,
|
|
354
|
+
) -> List[Dict[str, Any]]:
|
|
355
|
+
"""Strategy-aware chunking with per-chunk provenance metadata.
|
|
356
|
+
|
|
357
|
+
Returns ``[{"text": str, "meta": {"strategy", "start_char", "heading_path"}}]``
|
|
358
|
+
where ``start_char`` is the offset in ``str(text or "").strip()`` (every
|
|
359
|
+
chunk text is an exact substring at that offset).
|
|
360
|
+
|
|
361
|
+
Contract: ``[c["text"] for c in typed_chunks(t)] == _chunks(t)`` for the
|
|
362
|
+
default plain strategy — unknown strategies also fall back to plain.
|
|
363
|
+
"""
|
|
364
|
+
cleaned = str(text or "").strip()
|
|
365
|
+
if not cleaned:
|
|
366
|
+
return []
|
|
367
|
+
try:
|
|
368
|
+
size = max(1, int(size))
|
|
369
|
+
except Exception:
|
|
370
|
+
size = 1200
|
|
371
|
+
try:
|
|
372
|
+
overlap = min(max(0, int(overlap)), size - 1)
|
|
373
|
+
except Exception:
|
|
374
|
+
overlap = min(160, size - 1)
|
|
375
|
+
label = strategy if strategy in _CHUNK_STRATEGIES else "plain"
|
|
376
|
+
if label == "markdown":
|
|
377
|
+
return _markdown_chunks(cleaned, size, overlap)
|
|
378
|
+
if label == "code":
|
|
379
|
+
return _code_chunks(cleaned, size, overlap)
|
|
380
|
+
if label == "prose":
|
|
381
|
+
return _prose_chunks(cleaned, size, overlap)
|
|
382
|
+
return _plain_windows(cleaned, size, overlap)
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
def typed_chunk_meta_fields(piece: Dict[str, Any]) -> Dict[str, Any]:
|
|
386
|
+
"""Additive chunk-metadata fields for one ``typed_chunks`` piece.
|
|
387
|
+
|
|
388
|
+
Ingest call sites merge this into the existing ``{"index", "source_node"}``
|
|
389
|
+
chunk metadata; ``heading_path`` is only present when known — honest
|
|
390
|
+
absence over empty labels.
|
|
391
|
+
"""
|
|
392
|
+
meta = piece.get("meta") or {}
|
|
393
|
+
fields: Dict[str, Any] = {
|
|
394
|
+
"strategy": str(meta.get("strategy") or "plain"),
|
|
395
|
+
"start_char": int(meta.get("start_char") or 0),
|
|
396
|
+
}
|
|
397
|
+
heading_path = meta.get("heading_path")
|
|
398
|
+
if heading_path:
|
|
399
|
+
fields["heading_path"] = str(heading_path)
|
|
400
|
+
return fields
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
def citation_locator(chunk_metadata: Any) -> str:
|
|
404
|
+
"""Human "where in the document" label for one chunk, or "".
|
|
405
|
+
|
|
406
|
+
Built only from provenance the chunk actually carries — a section heading
|
|
407
|
+
path and/or a page number. When neither is known the answer is the empty
|
|
408
|
+
string, so a citation never claims a location it cannot prove.
|
|
409
|
+
"""
|
|
410
|
+
if not isinstance(chunk_metadata, dict):
|
|
411
|
+
return ""
|
|
412
|
+
parts: List[str] = []
|
|
413
|
+
heading = str(chunk_metadata.get("heading_path") or "").strip()
|
|
414
|
+
if heading:
|
|
415
|
+
parts.append(heading)
|
|
416
|
+
def _page(key: str) -> int:
|
|
417
|
+
value = chunk_metadata.get(key)
|
|
418
|
+
try:
|
|
419
|
+
return int(value) if value is not None else 0
|
|
420
|
+
except (TypeError, ValueError):
|
|
421
|
+
return 0
|
|
422
|
+
|
|
423
|
+
page_number = _page("page")
|
|
424
|
+
if page_number > 0:
|
|
425
|
+
page_end = _page("page_end")
|
|
426
|
+
parts.append(
|
|
427
|
+
f"p.{page_number}–{page_end}" if page_end > page_number else f"p.{page_number}"
|
|
428
|
+
)
|
|
429
|
+
return " · ".join(parts)
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
def pdf_page_offsets(structure: Any) -> List[int]:
|
|
433
|
+
"""Start offset of each PDF page in the "\\n\\n"-joined page text.
|
|
434
|
+
|
|
435
|
+
``structure`` is the ``metadata["structure"]`` dict produced by
|
|
436
|
+
``_pdf_structure`` (``pages`` = ``[{"chars": int, ...}, ...]``); pages were
|
|
437
|
+
joined with ``"\\n\\n"`` (see ``read_document``), so page k starts at
|
|
438
|
+
``sum(chars[j] + 2 for j < k)``. Empty or malformed input returns ``[]``.
|
|
439
|
+
"""
|
|
440
|
+
if not isinstance(structure, dict):
|
|
441
|
+
return []
|
|
442
|
+
pages = structure.get("pages")
|
|
443
|
+
if not isinstance(pages, list) or not pages:
|
|
444
|
+
return []
|
|
445
|
+
offsets: List[int] = []
|
|
446
|
+
cursor = 0
|
|
447
|
+
for page in pages:
|
|
448
|
+
if not isinstance(page, dict):
|
|
449
|
+
return []
|
|
450
|
+
chars = page.get("chars")
|
|
451
|
+
if isinstance(chars, bool) or not isinstance(chars, (int, float)) or chars < 0:
|
|
452
|
+
return []
|
|
453
|
+
offsets.append(cursor)
|
|
454
|
+
cursor += int(chars) + 2 # +2 for the "\n\n" page joiner
|
|
455
|
+
return offsets
|
|
456
|
+
|
|
457
|
+
|
|
458
|
+
def page_for_offset(page_offsets: List[int], offset: int) -> Optional[int]:
|
|
459
|
+
"""1-based page number containing ``offset`` given page start offsets.
|
|
460
|
+
|
|
461
|
+
Returns ``None`` when ``page_offsets`` is empty or the offset precedes the
|
|
462
|
+
first page start (honest absence over a wrong label).
|
|
463
|
+
"""
|
|
464
|
+
if not page_offsets:
|
|
465
|
+
return None
|
|
466
|
+
try:
|
|
467
|
+
target = int(offset)
|
|
468
|
+
except Exception:
|
|
469
|
+
return None
|
|
470
|
+
page = 0
|
|
471
|
+
for index, start in enumerate(page_offsets):
|
|
472
|
+
try:
|
|
473
|
+
if target >= int(start):
|
|
474
|
+
page = index + 1
|
|
475
|
+
else:
|
|
476
|
+
break
|
|
477
|
+
except Exception:
|
|
478
|
+
return None
|
|
479
|
+
return page if page >= 1 else None
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""Local filesystem indexing: a chosen folder becomes graph knowledge.
|
|
2
|
+
|
|
3
|
+
v11.3.0 turned this module into a package. ``KnowledgeGraphLocalIndexMixin``
|
|
4
|
+
is now composed from four cohesive sub-mixins — text extraction, node/index
|
|
5
|
+
upserts, graph cleanup, and the folder-scan driver — each moved here
|
|
6
|
+
verbatim. Every name this module exported before still resolves from
|
|
7
|
+
``lattice_brain.graph.discovery_index``.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
# ruff: noqa: F403,F405
|
|
13
|
+
from .._kg_common import * # noqa: F403,F401
|
|
14
|
+
from .cleanup import _LocalCleanupMixin
|
|
15
|
+
from .extract import _LocalExtractMixin
|
|
16
|
+
from .scan import _LocalScanMixin
|
|
17
|
+
from .upsert import _local_scoped_slug, _LocalUpsertMixin # noqa: F401
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
# Base order is most-composed-first (the scan driver, then the three halves it
|
|
21
|
+
# calls): each half is named as the driver's typing-only base, and C3 needs a
|
|
22
|
+
# subclass ahead of the class it extends. The method sets are disjoint, so at
|
|
23
|
+
# runtime the order changes nothing.
|
|
24
|
+
class KnowledgeGraphLocalIndexMixin(
|
|
25
|
+
_LocalScanMixin,
|
|
26
|
+
_LocalExtractMixin,
|
|
27
|
+
_LocalUpsertMixin,
|
|
28
|
+
_LocalCleanupMixin,
|
|
29
|
+
):
|
|
30
|
+
"""Local file → graph indexing (text extraction, node/index upserts,
|
|
31
|
+
graph-node deletion, orphan cleanup, and the index_local_folder driver),
|
|
32
|
+
split out of discovery. Composed into KnowledgeGraphStore alongside
|
|
33
|
+
KnowledgeGraphDiscoveryMixin; both share the instance so these methods
|
|
34
|
+
still reach sibling discovery/write helpers through the class MRO.
|
|
35
|
+
"""
|
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
"""Removing a local file from the graph, and the scope checks around it.
|
|
2
|
+
|
|
3
|
+
Deletes a file's graph node, sweeps the concepts left orphaned by it, and
|
|
4
|
+
answers the two "is this row still good?" questions the scanner asks before
|
|
5
|
+
skipping work. Moved verbatim out of ``discovery_index.py`` (v11.3.0).
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import TYPE_CHECKING
|
|
11
|
+
|
|
12
|
+
# ruff: noqa: F403,F405
|
|
13
|
+
from .._kg_common import * # noqa: F403,F401
|
|
14
|
+
|
|
15
|
+
# The cross-mixin surface (`_connect`, `_upsert_node`, …) is declared in
|
|
16
|
+
# `_kg_contract.KnowledgeGraphCore`. It is a typing-only base: at runtime this
|
|
17
|
+
# is `object`, so the MRO of `KnowledgeGraphStore` is unchanged.
|
|
18
|
+
if TYPE_CHECKING:
|
|
19
|
+
from .._kg_contract import KnowledgeGraphCore as _Core
|
|
20
|
+
else:
|
|
21
|
+
_Core = object
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class _LocalCleanupMixin(_Core):
|
|
25
|
+
"""Graph deletion + orphan sweep. Composed into the public mixin."""
|
|
26
|
+
|
|
27
|
+
def _delete_local_file_graph(
|
|
28
|
+
self, conn: sqlite3.Connection, file_node_id: Optional[str]
|
|
29
|
+
) -> None:
|
|
30
|
+
if not file_node_id:
|
|
31
|
+
return
|
|
32
|
+
|
|
33
|
+
file_row = conn.execute(
|
|
34
|
+
"SELECT metadata_json FROM nodes WHERE id=?",
|
|
35
|
+
(file_node_id,),
|
|
36
|
+
).fetchone()
|
|
37
|
+
source_id = None
|
|
38
|
+
if file_row:
|
|
39
|
+
source_id = _safe_loads(file_row["metadata_json"]).get("source_id")
|
|
40
|
+
|
|
41
|
+
linked_rows = conn.execute(
|
|
42
|
+
"""
|
|
43
|
+
SELECT n.id, n.type, n.metadata_json
|
|
44
|
+
FROM edges e
|
|
45
|
+
JOIN nodes n ON n.id=e.to_node
|
|
46
|
+
WHERE e.from_node=?
|
|
47
|
+
""",
|
|
48
|
+
(file_node_id,),
|
|
49
|
+
).fetchall()
|
|
50
|
+
owned_ids: set = set()
|
|
51
|
+
auto_candidate_ids: set = set()
|
|
52
|
+
for row in linked_rows:
|
|
53
|
+
metadata = _safe_loads(row["metadata_json"])
|
|
54
|
+
if (
|
|
55
|
+
row["type"] in {"Chunk", "ImageText", "Section"}
|
|
56
|
+
or metadata.get("source_node") == file_node_id
|
|
57
|
+
):
|
|
58
|
+
owned_ids.add(row["id"])
|
|
59
|
+
elif (
|
|
60
|
+
metadata.get("auto_extracted")
|
|
61
|
+
and metadata.get("source") == "local_folder"
|
|
62
|
+
):
|
|
63
|
+
auto_candidate_ids.add(row["id"])
|
|
64
|
+
|
|
65
|
+
conn.execute("DELETE FROM chunks WHERE source_node=?", (file_node_id,))
|
|
66
|
+
conn.execute(
|
|
67
|
+
"DELETE FROM edges WHERE from_node=? OR to_node=?",
|
|
68
|
+
(file_node_id, file_node_id),
|
|
69
|
+
)
|
|
70
|
+
conn.execute("DELETE FROM nodes WHERE id=?", (file_node_id,))
|
|
71
|
+
self._v2_delete_nodes(conn, [file_node_id])
|
|
72
|
+
|
|
73
|
+
def delete_nodes(node_ids: set) -> None:
|
|
74
|
+
if not node_ids:
|
|
75
|
+
return
|
|
76
|
+
placeholders = ",".join("?" * len(node_ids))
|
|
77
|
+
params = list(node_ids)
|
|
78
|
+
conn.execute(
|
|
79
|
+
f"DELETE FROM chunks WHERE source_node IN ({placeholders})", params
|
|
80
|
+
)
|
|
81
|
+
conn.execute(
|
|
82
|
+
f"DELETE FROM edges WHERE from_node IN ({placeholders}) OR to_node IN ({placeholders})",
|
|
83
|
+
params * 2,
|
|
84
|
+
)
|
|
85
|
+
conn.execute(f"DELETE FROM nodes WHERE id IN ({placeholders})", params)
|
|
86
|
+
self._v2_delete_nodes(conn, params)
|
|
87
|
+
|
|
88
|
+
delete_nodes(owned_ids)
|
|
89
|
+
|
|
90
|
+
removable_auto_ids: set = set()
|
|
91
|
+
for node_id in auto_candidate_ids:
|
|
92
|
+
remaining_edges = conn.execute(
|
|
93
|
+
"SELECT from_node, to_node FROM edges WHERE from_node=? OR to_node=?",
|
|
94
|
+
(node_id, node_id),
|
|
95
|
+
).fetchall()
|
|
96
|
+
if all(
|
|
97
|
+
(
|
|
98
|
+
row["from_node"] in auto_candidate_ids
|
|
99
|
+
and row["to_node"] in auto_candidate_ids
|
|
100
|
+
)
|
|
101
|
+
for row in remaining_edges
|
|
102
|
+
):
|
|
103
|
+
removable_auto_ids.add(node_id)
|
|
104
|
+
delete_nodes(removable_auto_ids)
|
|
105
|
+
if source_id:
|
|
106
|
+
self._cleanup_local_graph_orphans(conn, str(source_id))
|
|
107
|
+
|
|
108
|
+
def _cleanup_local_graph_orphans(
|
|
109
|
+
self, conn: sqlite3.Connection, source_id: str
|
|
110
|
+
) -> None:
|
|
111
|
+
while True:
|
|
112
|
+
folder_rows = conn.execute(
|
|
113
|
+
"SELECT id, metadata_json FROM nodes WHERE type='Folder'"
|
|
114
|
+
).fetchall()
|
|
115
|
+
leaf_ids = []
|
|
116
|
+
for row in folder_rows:
|
|
117
|
+
metadata = _safe_loads(row["metadata_json"])
|
|
118
|
+
if metadata.get("source_id") != source_id:
|
|
119
|
+
continue
|
|
120
|
+
has_children = conn.execute(
|
|
121
|
+
"SELECT 1 FROM edges WHERE from_node=? LIMIT 1",
|
|
122
|
+
(row["id"],),
|
|
123
|
+
).fetchone()
|
|
124
|
+
if not has_children:
|
|
125
|
+
leaf_ids.append(row["id"])
|
|
126
|
+
if not leaf_ids:
|
|
127
|
+
break
|
|
128
|
+
placeholders = ",".join("?" * len(leaf_ids))
|
|
129
|
+
conn.execute(
|
|
130
|
+
f"DELETE FROM edges WHERE from_node IN ({placeholders}) OR to_node IN ({placeholders})",
|
|
131
|
+
leaf_ids * 2,
|
|
132
|
+
)
|
|
133
|
+
conn.execute(f"DELETE FROM nodes WHERE id IN ({placeholders})", leaf_ids)
|
|
134
|
+
self._v2_delete_nodes(conn, leaf_ids)
|
|
135
|
+
|
|
136
|
+
for node_type in ("Drive", "Computer"):
|
|
137
|
+
rows = conn.execute(
|
|
138
|
+
"SELECT id FROM nodes WHERE type=?", (node_type,)
|
|
139
|
+
).fetchall()
|
|
140
|
+
removable = []
|
|
141
|
+
for row in rows:
|
|
142
|
+
has_children = conn.execute(
|
|
143
|
+
"SELECT 1 FROM edges WHERE from_node=? LIMIT 1",
|
|
144
|
+
(row["id"],),
|
|
145
|
+
).fetchone()
|
|
146
|
+
if not has_children:
|
|
147
|
+
removable.append(row["id"])
|
|
148
|
+
if removable:
|
|
149
|
+
placeholders = ",".join("?" * len(removable))
|
|
150
|
+
conn.execute(
|
|
151
|
+
f"DELETE FROM edges WHERE from_node IN ({placeholders}) OR to_node IN ({placeholders})",
|
|
152
|
+
removable * 2,
|
|
153
|
+
)
|
|
154
|
+
conn.execute(
|
|
155
|
+
f"DELETE FROM nodes WHERE id IN ({placeholders})", removable
|
|
156
|
+
)
|
|
157
|
+
self._v2_delete_nodes(conn, removable)
|
|
158
|
+
|
|
159
|
+
def _local_file_index_has_extracted_text(self, row: sqlite3.Row) -> bool:
|
|
160
|
+
metadata = _safe_loads(row["metadata_json"])
|
|
161
|
+
parser = metadata.get("parser") if isinstance(metadata, dict) else {}
|
|
162
|
+
if not isinstance(parser, dict):
|
|
163
|
+
return False
|
|
164
|
+
try:
|
|
165
|
+
return int(parser.get("extracted_chars") or 0) > 0
|
|
166
|
+
except (TypeError, ValueError):
|
|
167
|
+
return False
|
|
168
|
+
|
|
169
|
+
@staticmethod
|
|
170
|
+
def _node_matches_workspace(
|
|
171
|
+
conn: sqlite3.Connection,
|
|
172
|
+
node_id: Optional[str],
|
|
173
|
+
workspace_id: Optional[str],
|
|
174
|
+
) -> bool:
|
|
175
|
+
"""Return true only when the projected node has the expected scope."""
|
|
176
|
+
if not node_id:
|
|
177
|
+
return False
|
|
178
|
+
row = conn.execute(
|
|
179
|
+
"SELECT workspace_id FROM nodes_v2 WHERE id=?",
|
|
180
|
+
(node_id,),
|
|
181
|
+
).fetchone()
|
|
182
|
+
return bool(row is not None and row["workspace_id"] == workspace_id)
|