ltcai 11.2.0 → 11.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -53
- package/docs/CHANGELOG.md +87 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/MULTI_AGENT_RUNTIME.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +6 -2
- package/docs/PERMISSION_MODE.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +2 -2
- package/docs/v11.3.0_PLAN.md +202 -0
- package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +181 -0
- package/docs/v11.5.0_RUST_COMPLETE_PLAN.md +145 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +287 -0
- package/lattice_brain/graph/_kg_common/extraction.py +516 -0
- package/lattice_brain/graph/_kg_common/relations.py +161 -0
- package/lattice_brain/graph/_kg_common/text.py +479 -0
- package/lattice_brain/graph/discovery_index/__init__.py +35 -0
- package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
- package/lattice_brain/graph/discovery_index/extract.py +137 -0
- package/lattice_brain/graph/discovery_index/scan.py +411 -0
- package/lattice_brain/graph/discovery_index/upsert.py +495 -0
- package/lattice_brain/graph/projection/__init__.py +42 -0
- package/lattice_brain/graph/projection/curation.py +500 -0
- package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
- package/lattice_brain/graph/retrieval/__init__.py +54 -0
- package/lattice_brain/graph/retrieval/context.py +197 -0
- package/lattice_brain/graph/retrieval/graph_view.py +319 -0
- package/lattice_brain/graph/retrieval/hybrid.py +488 -0
- package/lattice_brain/graph/retrieval/maintenance.py +121 -0
- package/lattice_brain/graph/retrieval/signals.py +95 -0
- package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
- package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
- package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
- package/lattice_brain/graph/retrieval_vector/search.py +560 -0
- package/lattice_brain/graph/retrieval_vector/status.py +374 -0
- package/lattice_brain/ingestion/__init__.py +130 -0
- package/lattice_brain/ingestion/_contract.py +90 -0
- package/lattice_brain/ingestion/constants.py +127 -0
- package/lattice_brain/ingestion/folder_scan.py +57 -0
- package/lattice_brain/ingestion/folders.py +258 -0
- package/lattice_brain/ingestion/hashing.py +26 -0
- package/lattice_brain/ingestion/jobs_api.py +107 -0
- package/lattice_brain/ingestion/models.py +80 -0
- package/lattice_brain/ingestion/pipeline.py +486 -0
- package/lattice_brain/ingestion/quality.py +209 -0
- package/lattice_brain/ingestion/routing.py +295 -0
- package/lattice_brain/multimodal/__init__.py +164 -0
- package/lattice_brain/multimodal/audio.py +77 -0
- package/lattice_brain/multimodal/common.py +118 -0
- package/lattice_brain/multimodal/images.py +498 -0
- package/lattice_brain/multimodal/ports.py +169 -0
- package/lattice_brain/multimodal/video.py +410 -0
- package/lattice_brain/portability/__init__.py +90 -0
- package/lattice_brain/portability/_contract.py +42 -0
- package/lattice_brain/portability/backups.py +338 -0
- package/lattice_brain/portability/bundles.py +136 -0
- package/lattice_brain/portability/constants.py +93 -0
- package/lattice_brain/portability/fsops.py +138 -0
- package/lattice_brain/portability/service.py +41 -0
- package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
- package/lattice_brain/runtime/__init__.py +1 -1
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/chronicle.py +63 -0
- package/latticeai/api/index_jobs.py +145 -0
- package/latticeai/core/agent/__init__.py +93 -0
- package/latticeai/core/agent/_contract.py +79 -0
- package/latticeai/core/agent/context.py +57 -0
- package/latticeai/core/agent/deps.py +125 -0
- package/latticeai/core/agent/execution.py +622 -0
- package/latticeai/core/agent/planning.py +145 -0
- package/latticeai/core/agent/recovery.py +157 -0
- package/latticeai/core/agent/runtime.py +210 -0
- package/latticeai/core/agent/verification.py +231 -0
- package/latticeai/core/embedding_providers/__init__.py +151 -0
- package/latticeai/core/embedding_providers/base.py +199 -0
- package/latticeai/core/embedding_providers/captions.py +162 -0
- package/latticeai/core/embedding_providers/profiles.py +126 -0
- package/latticeai/core/embedding_providers/text.py +350 -0
- package/latticeai/core/embedding_providers/vision.py +352 -0
- package/latticeai/core/file_generation/__init__.py +115 -0
- package/latticeai/core/file_generation/bundles.py +76 -0
- package/latticeai/core/file_generation/extraction.py +154 -0
- package/latticeai/core/file_generation/inference.py +235 -0
- package/latticeai/core/file_generation/orchestration.py +152 -0
- package/latticeai/core/file_generation/prompting.py +117 -0
- package/latticeai/core/file_generation/repair.py +114 -0
- package/latticeai/core/file_generation/sanitize.py +61 -0
- package/latticeai/core/file_generation/validation.py +201 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +14 -0
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/integrations/telegram_bot/__init__.py +123 -0
- package/latticeai/integrations/telegram_bot/__main__.py +17 -0
- package/latticeai/integrations/telegram_bot/config.py +86 -0
- package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
- package/latticeai/integrations/telegram_bot/flows.py +478 -0
- package/latticeai/integrations/telegram_bot/helpers.py +322 -0
- package/latticeai/integrations/telegram_bot/screens.py +394 -0
- package/latticeai/models/router/__init__.py +88 -0
- package/latticeai/models/router/_contract.py +66 -0
- package/latticeai/models/router/branding.py +56 -0
- package/latticeai/models/router/catalog.py +69 -0
- package/latticeai/models/router/documents.py +199 -0
- package/latticeai/models/router/errors.py +37 -0
- package/latticeai/models/router/generation.py +258 -0
- package/latticeai/models/router/loading.py +291 -0
- package/latticeai/models/router/local_models.py +85 -0
- package/latticeai/models/router/registry.py +147 -0
- package/latticeai/runtime/build_phases/__init__.py +82 -0
- package/latticeai/runtime/build_phases/features.py +421 -0
- package/latticeai/runtime/build_phases/foundation.py +555 -0
- package/latticeai/runtime/build_phases/web.py +492 -0
- package/latticeai/runtime/runtime_context.py +1 -0
- package/latticeai/services/architecture_readiness.py +48 -19
- package/latticeai/services/brain_intelligence/__init__.py +58 -0
- package/latticeai/services/brain_intelligence/_contract.py +71 -0
- package/latticeai/services/brain_intelligence/consistency.py +193 -0
- package/latticeai/services/brain_intelligence/constants.py +47 -0
- package/latticeai/services/brain_intelligence/digest.py +258 -0
- package/latticeai/services/brain_intelligence/health.py +331 -0
- package/latticeai/services/brain_intelligence/proposals.py +264 -0
- package/latticeai/services/brain_intelligence/sampling.py +84 -0
- package/latticeai/services/brain_intelligence/service.py +48 -0
- package/latticeai/services/chronicle.py +557 -0
- package/latticeai/services/memory_service/__init__.py +52 -0
- package/latticeai/services/memory_service/_contract.py +100 -0
- package/latticeai/services/memory_service/brief.py +431 -0
- package/latticeai/services/memory_service/constants.py +57 -0
- package/latticeai/services/memory_service/maintenance.py +138 -0
- package/latticeai/services/memory_service/manager.py +186 -0
- package/latticeai/services/memory_service/proof.py +136 -0
- package/latticeai/services/memory_service/recall.py +225 -0
- package/latticeai/services/memory_service/service.py +48 -0
- package/latticeai/services/memory_service/stores.py +110 -0
- package/latticeai/services/model_runtime/__init__.py +322 -0
- package/latticeai/services/model_runtime/cloud.py +87 -0
- package/latticeai/services/model_runtime/download.py +282 -0
- package/latticeai/services/model_runtime/engines.py +341 -0
- package/latticeai/services/model_runtime/loading.py +178 -0
- package/latticeai/services/model_runtime/service.py +129 -0
- package/latticeai/services/model_runtime/state.py +131 -0
- package/latticeai/services/model_runtime/status.py +255 -0
- package/latticeai/services/product_readiness.py +15 -7
- package/latticeai/setup/wizard/__init__.py +126 -0
- package/latticeai/setup/wizard/catalog.py +172 -0
- package/latticeai/setup/wizard/detect.py +323 -0
- package/latticeai/setup/wizard/install.py +348 -0
- package/latticeai/setup/wizard/paths.py +168 -0
- package/latticeai/setup/wizard/plans.py +74 -0
- package/latticeai/setup/wizard/recommend.py +320 -0
- package/package.json +6 -2
- package/scripts/bump_version.py +14 -0
- package/scripts/capture_release_evidence.mjs +33 -21
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_i18n_namespace_coverage.mjs +41 -4
- package/scripts/check_max_file_lines.mjs +102 -0
- package/scripts/check_release_evidence_bound.mjs +30 -15
- package/scripts/check_screenshot_pixel_delta.py +34 -4
- package/scripts/check_server_i18n.mjs +2 -0
- package/scripts/chunking_parity_corpus.py +449 -0
- package/scripts/generate_agent_parity_fixtures.py +752 -0
- package/scripts/generate_chunking_parity_fixtures.py +259 -0
- package/scripts/generate_rust_parity_fixtures.py +997 -0
- package/scripts/lib/mock_server_fingerprint.mjs +94 -0
- package/scripts/release_screen_claims.json +42 -2
- package/src-tauri/Cargo.lock +404 -3
- package/src-tauri/Cargo.toml +13 -1
- package/src-tauri/src/backend.rs +460 -0
- package/src-tauri/src/folder.rs +33 -0
- package/src-tauri/src/main.rs +109 -399
- package/src-tauri/src/topology.rs +356 -0
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +41 -37
- package/static/app/assets/Act-CWnxSCgN.js +1 -0
- package/static/app/assets/AdminConsole-BEQYU6kF.js +1 -0
- package/static/app/assets/{Brain-tuhI4sOC.js → Brain-DWu1BhFg.js} +2 -2
- package/static/app/assets/BrainHome-95Hilr9R.js +2 -0
- package/static/app/assets/BrainSignals-QdeqCpAF.js +1 -0
- package/static/app/assets/Capture-BHpCxnzb.js +1 -0
- package/static/app/assets/Chronicle-B4xYKoed.js +1 -0
- package/static/app/assets/CommandPalette-BVXnttSz.js +1 -0
- package/static/app/assets/Library-DgYcHome.js +1 -0
- package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-CrJLDbf7.js} +1 -1
- package/static/app/assets/ProductFlow-DFlScKoJ.js +1 -0
- package/static/app/assets/ReviewCard-Cy5f48Pj.js +3 -0
- package/static/app/assets/System-NF8IfhTa.js +1 -0
- package/static/app/assets/arrow-left-DwkSYrjR.js +1 -0
- package/static/app/assets/{bot-Cia42c2h.js → bot-CucuhLhm.js} +1 -1
- package/static/app/assets/brain-BBnSryW_.js +1 -0
- package/static/app/assets/{button-2j2Ijzgq.js → button-C2GUj2Ai.js} +1 -1
- package/static/app/assets/circle-check-CxOVPwYq.js +1 -0
- package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-CbkWzBmG.js} +1 -1
- package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-7lEaqHdJ.js} +1 -1
- package/static/app/assets/{cpu-k4awryFq.js → cpu-DAlCXlIy.js} +1 -1
- package/static/app/assets/{download-DFbLJ_ig.js → download-RNhuuJwh.js} +1 -1
- package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CLW4odzM.js} +1 -1
- package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-NKEiDIAJ.js} +1 -1
- package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
- package/static/app/assets/index-DMurvUuR.js +10 -0
- package/static/app/assets/input-D2UhPC1X.js +1 -0
- package/static/app/assets/link-2-6amKbP_P.js +1 -0
- package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-Cu9TZtdR.js} +1 -1
- package/static/app/assets/primitives-gPsccucr.js +1 -0
- package/static/app/assets/search-Cj_TKk_2.js +1 -0
- package/static/app/assets/{share-2-BH1M-WNi.js → share-2-Bau7KkPq.js} +1 -1
- package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-BufNYypi.js} +1 -1
- package/static/app/assets/{textarea-CCWbUfFB.js → textarea-BQnVWhYs.js} +1 -1
- package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-B3_w60si.js} +1 -1
- package/static/app/assets/useMutation-BHhCflT6.js +1 -0
- package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-rBWfI-5t.js} +1 -1
- package/static/app/assets/utils-V_5-wxr5.js +4 -0
- package/static/app/assets/workspace-K1zjYUHj.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/graph/_kg_common.py +0 -1331
- package/lattice_brain/graph/discovery_index.py +0 -1141
- package/lattice_brain/graph/retrieval.py +0 -1120
- package/lattice_brain/graph/retrieval_vector.py +0 -1293
- package/lattice_brain/ingestion.py +0 -1525
- package/lattice_brain/multimodal.py +0 -1258
- package/latticeai/core/agent.py +0 -1465
- package/latticeai/core/embedding_providers.py +0 -1196
- package/latticeai/core/file_generation.py +0 -1047
- package/latticeai/integrations/telegram_bot.py +0 -1390
- package/latticeai/models/router.py +0 -1007
- package/latticeai/runtime/build_phases.py +0 -1450
- package/latticeai/services/brain_intelligence.py +0 -1083
- package/latticeai/services/memory_service.py +0 -1177
- package/latticeai/services/model_runtime.py +0 -1281
- package/latticeai/setup/wizard.py +0 -1310
- package/static/app/assets/Act-AWf0SAKp.js +0 -1
- package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
- package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
- package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
- package/static/app/assets/Capture-CqOSzyPr.js +0 -1
- package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
- package/static/app/assets/Library-CX-bbhmK.js +0 -1
- package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
- package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
- package/static/app/assets/System-Bu2t5hn1.js +0 -1
- package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
- package/static/app/assets/brain-DJMoqrwx.js +0 -1
- package/static/app/assets/index-BpYkzcVm.js +0 -10
- package/static/app/assets/input-DSlJJxRs.js +0 -1
- package/static/app/assets/primitives-BCx6TvfG.js +0 -1
- package/static/app/assets/search-Cgy8cCFJ.js +0 -1
- package/static/app/assets/utils-zqPZJxdx.js +0 -4
- package/static/app/assets/workspace-DXTihhfU.js +0 -1
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
"""Local file → text: the parsers behind local indexing.
|
|
2
|
+
|
|
3
|
+
PDF/Word/Excel/PowerPoint text plus the image signal path (dimensions, OCR,
|
|
4
|
+
and — only when a VLM exists — a caption). Moved verbatim out of
|
|
5
|
+
``discovery_index.py`` (v11.3.0 decomposition).
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import TYPE_CHECKING
|
|
11
|
+
|
|
12
|
+
# ruff: noqa: F403,F405
|
|
13
|
+
from .._kg_common import * # noqa: F403,F401
|
|
14
|
+
|
|
15
|
+
# The cross-mixin surface (`_connect`, `_upsert_node`, …) is declared in
|
|
16
|
+
# `_kg_contract.KnowledgeGraphCore`. It is a typing-only base: at runtime this
|
|
17
|
+
# is `object`, so the MRO of `KnowledgeGraphStore` is unchanged.
|
|
18
|
+
if TYPE_CHECKING:
|
|
19
|
+
from .._kg_contract import KnowledgeGraphCore as _Core
|
|
20
|
+
else:
|
|
21
|
+
_Core = object
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class _LocalExtractMixin(_Core):
|
|
25
|
+
"""File-text and image-signal extraction. Composed into the public mixin."""
|
|
26
|
+
|
|
27
|
+
def _extract_local_file_text(
|
|
28
|
+
self, path: Path, category: str, *, include_ocr: bool
|
|
29
|
+
) -> Tuple[str, Dict[str, Any]]:
|
|
30
|
+
ext = path.suffix.lower()
|
|
31
|
+
meta: Dict[str, Any] = {"parser": _parser_type_for_category(category, ext)}
|
|
32
|
+
text = ""
|
|
33
|
+
if category in {"text", "code"} or ext == ".csv":
|
|
34
|
+
text = path.read_text(encoding="utf-8", errors="replace")
|
|
35
|
+
elif ext == ".pdf":
|
|
36
|
+
import pdfplumber
|
|
37
|
+
|
|
38
|
+
with pdfplumber.open(str(path)) as pdf:
|
|
39
|
+
meta["pages"] = len(pdf.pages)
|
|
40
|
+
text = "\n\n".join((page.extract_text() or "") for page in pdf.pages)
|
|
41
|
+
elif ext == ".docx":
|
|
42
|
+
from docx import Document
|
|
43
|
+
|
|
44
|
+
doc = Document(str(path))
|
|
45
|
+
paragraphs = [p.text for p in doc.paragraphs if p.text.strip()]
|
|
46
|
+
table_lines = []
|
|
47
|
+
for table in doc.tables:
|
|
48
|
+
for row in table.rows:
|
|
49
|
+
cells = [_clean_text(cell.text) for cell in row.cells]
|
|
50
|
+
if any(cells):
|
|
51
|
+
table_lines.append("\t".join(cells))
|
|
52
|
+
meta["paragraphs"] = len(paragraphs)
|
|
53
|
+
meta["tables"] = len(doc.tables)
|
|
54
|
+
meta["table_rows"] = len(table_lines)
|
|
55
|
+
text = "\n\n".join([*paragraphs, *table_lines])
|
|
56
|
+
elif ext == ".xlsx":
|
|
57
|
+
from openpyxl import load_workbook
|
|
58
|
+
|
|
59
|
+
wb = load_workbook(str(path), read_only=True, data_only=True)
|
|
60
|
+
rows_all = []
|
|
61
|
+
non_empty_rows = 0
|
|
62
|
+
non_empty_cells = 0
|
|
63
|
+
char_count = 0
|
|
64
|
+
for ws in wb.worksheets:
|
|
65
|
+
sheet_rows = []
|
|
66
|
+
for row in ws.iter_rows(values_only=True):
|
|
67
|
+
cells = [
|
|
68
|
+
str(cell).strip() if cell is not None else "" for cell in row
|
|
69
|
+
]
|
|
70
|
+
if not any(cells):
|
|
71
|
+
continue
|
|
72
|
+
line = "\t".join(cells)
|
|
73
|
+
non_empty_rows += 1
|
|
74
|
+
non_empty_cells += sum(1 for cell in cells if cell)
|
|
75
|
+
sheet_rows.append(line)
|
|
76
|
+
char_count += len(line) + 1
|
|
77
|
+
if char_count > 200_000:
|
|
78
|
+
break
|
|
79
|
+
if sheet_rows:
|
|
80
|
+
rows_all.append(f"[Sheet: {ws.title}]")
|
|
81
|
+
rows_all.extend(sheet_rows)
|
|
82
|
+
if char_count > 200_000:
|
|
83
|
+
break
|
|
84
|
+
meta["sheets"] = len(wb.worksheets)
|
|
85
|
+
meta["rows"] = non_empty_rows
|
|
86
|
+
meta["cells"] = non_empty_cells
|
|
87
|
+
text = "\n".join(rows_all)
|
|
88
|
+
elif ext == ".pptx":
|
|
89
|
+
from pptx import Presentation
|
|
90
|
+
|
|
91
|
+
prs = Presentation(str(path))
|
|
92
|
+
slides_text = []
|
|
93
|
+
for index, slide in enumerate(prs.slides, 1):
|
|
94
|
+
parts = []
|
|
95
|
+
for shape in slide.shapes:
|
|
96
|
+
if getattr(shape, "has_text_frame", False):
|
|
97
|
+
slide_text = shape.text_frame.text.strip()
|
|
98
|
+
if slide_text:
|
|
99
|
+
parts.append(slide_text)
|
|
100
|
+
if parts:
|
|
101
|
+
slides_text.append(f"[Slide {index}]\n" + "\n".join(parts))
|
|
102
|
+
meta["slides"] = len(prs.slides)
|
|
103
|
+
meta["text_slides"] = len(slides_text)
|
|
104
|
+
text = "\n\n".join(slides_text)
|
|
105
|
+
elif category == "image":
|
|
106
|
+
text = self._extract_image_signals(path, meta, include_ocr=include_ocr)
|
|
107
|
+
return text[:200_000], meta
|
|
108
|
+
|
|
109
|
+
def _extract_image_signals(
|
|
110
|
+
self, path: Path, meta: Dict[str, Any], *, include_ocr: bool
|
|
111
|
+
) -> str:
|
|
112
|
+
"""Dimensions, OCR text, and — only if a VLM exists — a caption.
|
|
113
|
+
|
|
114
|
+
Until v11.1.0 this path always attached a ``vision_caption`` built out
|
|
115
|
+
of the filename and the pixel dimensions (``Image pic.png (PNG 12x8)``)
|
|
116
|
+
and used it as the retrieval text. Nothing downstream could tell that
|
|
117
|
+
string apart from something a vision model had actually said about the
|
|
118
|
+
picture, so every screenshot in the graph carried a fake description.
|
|
119
|
+
|
|
120
|
+
Now the caption comes from the injected port and from nowhere else. A
|
|
121
|
+
picture with no OCR text and no model still gets indexed — under its
|
|
122
|
+
filename, which is a fact — and ``caption_status`` says why there is no
|
|
123
|
+
caption.
|
|
124
|
+
"""
|
|
125
|
+
from ...multimodal import MultimodalPorts, extract_image_facts
|
|
126
|
+
|
|
127
|
+
ports = getattr(self, "multimodal_ports", None) or MultimodalPorts()
|
|
128
|
+
facts = extract_image_facts(str(path), ports=ports, ocr=include_ocr)
|
|
129
|
+
meta.update(facts.as_metadata())
|
|
130
|
+
meta["ocr_enabled"] = bool(include_ocr)
|
|
131
|
+
if facts.ocr_text:
|
|
132
|
+
meta["ocr_chars"] = len(facts.ocr_text)
|
|
133
|
+
if facts.ocr_status == "failed":
|
|
134
|
+
meta["ocr_error"] = facts.ocr_detail
|
|
135
|
+
if not facts.readable:
|
|
136
|
+
return ""
|
|
137
|
+
return facts.index_text() or path.name
|
|
@@ -0,0 +1,411 @@
|
|
|
1
|
+
"""``index_local_folder``: the driver that walks a folder into the graph.
|
|
2
|
+
|
|
3
|
+
Decides per file whether to skip, refresh, or fully re-index, and reports
|
|
4
|
+
counts, errors, and the honest notice about what was converted. Moved
|
|
5
|
+
verbatim out of ``discovery_index.py`` (v11.3.0 decomposition).
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import TYPE_CHECKING
|
|
11
|
+
|
|
12
|
+
# ruff: noqa: F403,F405
|
|
13
|
+
from .._kg_common import * # noqa: F403,F401
|
|
14
|
+
|
|
15
|
+
# Typing-only base (runtime value is `object`, so the store's MRO is
|
|
16
|
+
# unchanged). The driver calls all three other halves through `self` — text
|
|
17
|
+
# extraction, the row/node upserts, and the skip checks — so they are named as
|
|
18
|
+
# bases rather than re-declared here, where their signatures could drift. The
|
|
19
|
+
# store contract (`_connect`, `_upsert_node`, …) arrives with them.
|
|
20
|
+
if TYPE_CHECKING:
|
|
21
|
+
from .cleanup import _LocalCleanupMixin
|
|
22
|
+
from .extract import _LocalExtractMixin
|
|
23
|
+
from .upsert import _LocalUpsertMixin
|
|
24
|
+
|
|
25
|
+
class _Core(_LocalExtractMixin, _LocalUpsertMixin, _LocalCleanupMixin):
|
|
26
|
+
"""The sibling halves this one reaches through ``self``."""
|
|
27
|
+
else:
|
|
28
|
+
_Core = object
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class _LocalScanMixin(_Core):
|
|
32
|
+
"""The folder-indexing driver. Composed into the public mixin."""
|
|
33
|
+
|
|
34
|
+
def index_local_folder(
|
|
35
|
+
self,
|
|
36
|
+
path: Path,
|
|
37
|
+
*,
|
|
38
|
+
include_ocr: bool = False,
|
|
39
|
+
watch_enabled: bool = False,
|
|
40
|
+
user_email: Optional[str] = None,
|
|
41
|
+
workspace_id: Optional[str] = None,
|
|
42
|
+
consent: Optional[Dict[str, Any]] = None,
|
|
43
|
+
max_files: int = 5_000,
|
|
44
|
+
source_id_override: Optional[str] = None,
|
|
45
|
+
) -> Dict[str, Any]:
|
|
46
|
+
"""Read approved files from a local folder and connect them to Graph RAG."""
|
|
47
|
+
root = Path(path).expanduser().resolve()
|
|
48
|
+
if not root.exists():
|
|
49
|
+
raise ValueError(f"경로가 존재하지 않습니다: {path}")
|
|
50
|
+
if not root.is_dir():
|
|
51
|
+
raise ValueError(f"폴더가 아닙니다: {path}")
|
|
52
|
+
|
|
53
|
+
os_type = _current_os_type()
|
|
54
|
+
drive_id = _drive_id_for_path(root)
|
|
55
|
+
path_fingerprint = _path_fingerprint(root)
|
|
56
|
+
source_id = str(source_id_override or "").strip()
|
|
57
|
+
if not source_id:
|
|
58
|
+
source_id = (
|
|
59
|
+
f"source:{_sha256_text(f'{workspace_id}|{path_fingerprint}')[:24]}"
|
|
60
|
+
if workspace_id
|
|
61
|
+
else f"source:{path_fingerprint}"
|
|
62
|
+
)
|
|
63
|
+
now = _now()
|
|
64
|
+
max_files = max(1, min(int(max_files or 5_000), 50_000))
|
|
65
|
+
consent_payload = {
|
|
66
|
+
"approved_at": now,
|
|
67
|
+
"knowledge_source": True,
|
|
68
|
+
"include_ocr": bool(include_ocr),
|
|
69
|
+
"watch_enabled": bool(watch_enabled),
|
|
70
|
+
"sensitive_files_default_excluded": True,
|
|
71
|
+
**(consent or {}),
|
|
72
|
+
"approved_by": user_email or (consent or {}).get("approved_by"),
|
|
73
|
+
"workspace_id": workspace_id or (consent or {}).get("workspace_id"),
|
|
74
|
+
}
|
|
75
|
+
counts: Counter = Counter()
|
|
76
|
+
seen_relative_paths: set = set()
|
|
77
|
+
indexed_nodes: List[str] = []
|
|
78
|
+
errors: List[Dict[str, str]] = []
|
|
79
|
+
limit_reached = False
|
|
80
|
+
|
|
81
|
+
with self._connect() as conn:
|
|
82
|
+
existing_source = conn.execute(
|
|
83
|
+
"SELECT id, consent_json FROM knowledge_sources WHERE root_path=?",
|
|
84
|
+
(str(root),),
|
|
85
|
+
).fetchone()
|
|
86
|
+
if existing_source is not None:
|
|
87
|
+
existing_consent = _safe_loads(existing_source["consent_json"])
|
|
88
|
+
existing_scope = existing_consent.get("workspace_id") or "personal"
|
|
89
|
+
requested_scope = workspace_id or consent_payload.get("workspace_id") or "personal"
|
|
90
|
+
if existing_scope != requested_scope:
|
|
91
|
+
raise ValueError(
|
|
92
|
+
"This folder is already connected to another workspace. "
|
|
93
|
+
"Disconnect it there before assigning it to a different Brain."
|
|
94
|
+
)
|
|
95
|
+
if existing_source["id"] != source_id:
|
|
96
|
+
# Reuse the legacy source identity so a personal source is
|
|
97
|
+
# reprojected in place instead of duplicated during upgrade.
|
|
98
|
+
source_id = existing_source["id"]
|
|
99
|
+
conn.execute(
|
|
100
|
+
"""
|
|
101
|
+
INSERT INTO knowledge_sources(
|
|
102
|
+
id, root_path, os_type, drive_id, label, status, include_ocr,
|
|
103
|
+
watch_enabled, consent_json, created_at, updated_at, last_scanned_at
|
|
104
|
+
)
|
|
105
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
106
|
+
ON CONFLICT(id) DO UPDATE SET
|
|
107
|
+
root_path=excluded.root_path,
|
|
108
|
+
os_type=excluded.os_type,
|
|
109
|
+
drive_id=excluded.drive_id,
|
|
110
|
+
label=excluded.label,
|
|
111
|
+
status=excluded.status,
|
|
112
|
+
include_ocr=excluded.include_ocr,
|
|
113
|
+
watch_enabled=excluded.watch_enabled,
|
|
114
|
+
consent_json=excluded.consent_json,
|
|
115
|
+
updated_at=excluded.updated_at,
|
|
116
|
+
last_scanned_at=excluded.last_scanned_at
|
|
117
|
+
""",
|
|
118
|
+
(
|
|
119
|
+
source_id,
|
|
120
|
+
str(root),
|
|
121
|
+
os_type,
|
|
122
|
+
drive_id,
|
|
123
|
+
root.name or str(root),
|
|
124
|
+
"scanning",
|
|
125
|
+
1 if include_ocr else 0,
|
|
126
|
+
1 if watch_enabled else 0,
|
|
127
|
+
_json(consent_payload),
|
|
128
|
+
now,
|
|
129
|
+
now,
|
|
130
|
+
now,
|
|
131
|
+
),
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
for entry in self._iter_local_scan_entries(root, max_files=max_files):
|
|
135
|
+
kind = entry["kind"]
|
|
136
|
+
file_path = entry["path"]
|
|
137
|
+
if kind == "limit_reached":
|
|
138
|
+
counts["limit_reached"] += 1
|
|
139
|
+
limit_reached = True
|
|
140
|
+
break
|
|
141
|
+
if kind in {"excluded_dir", "excluded"}:
|
|
142
|
+
counts["excluded"] += 1
|
|
143
|
+
continue
|
|
144
|
+
if kind in {"inaccessible_dir", "inaccessible_file"}:
|
|
145
|
+
counts["failed"] += 1
|
|
146
|
+
errors.append(
|
|
147
|
+
{
|
|
148
|
+
"path": str(file_path),
|
|
149
|
+
"error": entry.get("reason", "inaccessible"),
|
|
150
|
+
}
|
|
151
|
+
)
|
|
152
|
+
continue
|
|
153
|
+
if kind != "file":
|
|
154
|
+
continue
|
|
155
|
+
|
|
156
|
+
stat = entry["stat"]
|
|
157
|
+
try:
|
|
158
|
+
relative_path = file_path.relative_to(root).as_posix()
|
|
159
|
+
except ValueError:
|
|
160
|
+
relative_path = file_path.name
|
|
161
|
+
seen_relative_paths.add(relative_path)
|
|
162
|
+
modified_at = _safe_iso_from_stat_mtime(stat.st_mtime)
|
|
163
|
+
existing = conn.execute(
|
|
164
|
+
"""
|
|
165
|
+
SELECT size_bytes, modified_at, sha256, graph_node_id, status, metadata_json
|
|
166
|
+
FROM local_file_index
|
|
167
|
+
WHERE source_id=? AND relative_path=?
|
|
168
|
+
""",
|
|
169
|
+
(source_id, relative_path),
|
|
170
|
+
).fetchone()
|
|
171
|
+
decision = self._local_file_decision(file_path, root, stat)
|
|
172
|
+
parser_type = decision["parser_type"]
|
|
173
|
+
if not decision["indexable"]:
|
|
174
|
+
counts[decision["status"]] += 1
|
|
175
|
+
if existing and existing["graph_node_id"]:
|
|
176
|
+
self._delete_local_file_graph(conn, existing["graph_node_id"])
|
|
177
|
+
self._upsert_local_file_index(
|
|
178
|
+
conn,
|
|
179
|
+
source_id=source_id,
|
|
180
|
+
root=root,
|
|
181
|
+
file_path=file_path,
|
|
182
|
+
stat=stat,
|
|
183
|
+
os_type=os_type,
|
|
184
|
+
drive_id=drive_id,
|
|
185
|
+
status=decision["status"],
|
|
186
|
+
parser_type=parser_type,
|
|
187
|
+
metadata={
|
|
188
|
+
"reason": decision["reason"],
|
|
189
|
+
"category": decision["category"],
|
|
190
|
+
},
|
|
191
|
+
)
|
|
192
|
+
continue
|
|
193
|
+
|
|
194
|
+
if (
|
|
195
|
+
existing
|
|
196
|
+
and existing["status"] == "indexed"
|
|
197
|
+
and existing["graph_node_id"]
|
|
198
|
+
and self._local_file_index_has_extracted_text(existing)
|
|
199
|
+
and self._node_matches_workspace(
|
|
200
|
+
conn, existing["graph_node_id"], workspace_id
|
|
201
|
+
)
|
|
202
|
+
and existing["size_bytes"] == stat.st_size
|
|
203
|
+
and existing["modified_at"] == modified_at
|
|
204
|
+
):
|
|
205
|
+
counts["skipped_unchanged"] += 1
|
|
206
|
+
self._upsert_local_file_index(
|
|
207
|
+
conn,
|
|
208
|
+
source_id=source_id,
|
|
209
|
+
root=root,
|
|
210
|
+
file_path=file_path,
|
|
211
|
+
stat=stat,
|
|
212
|
+
os_type=os_type,
|
|
213
|
+
drive_id=drive_id,
|
|
214
|
+
status="indexed",
|
|
215
|
+
parser_type=parser_type,
|
|
216
|
+
sha256=existing["sha256"],
|
|
217
|
+
graph_node_id=existing["graph_node_id"],
|
|
218
|
+
metadata={
|
|
219
|
+
**_safe_loads(existing["metadata_json"]),
|
|
220
|
+
"category": decision["category"],
|
|
221
|
+
"unchanged": True,
|
|
222
|
+
},
|
|
223
|
+
)
|
|
224
|
+
continue
|
|
225
|
+
|
|
226
|
+
try:
|
|
227
|
+
data = file_path.read_bytes()
|
|
228
|
+
digest = _sha256_bytes(data)
|
|
229
|
+
except Exception as exc:
|
|
230
|
+
counts["failed"] += 1
|
|
231
|
+
errors.append({"path": str(file_path), "error": str(exc)})
|
|
232
|
+
if existing and existing["graph_node_id"]:
|
|
233
|
+
self._delete_local_file_graph(conn, existing["graph_node_id"])
|
|
234
|
+
self._upsert_local_file_index(
|
|
235
|
+
conn,
|
|
236
|
+
source_id=source_id,
|
|
237
|
+
root=root,
|
|
238
|
+
file_path=file_path,
|
|
239
|
+
stat=stat,
|
|
240
|
+
os_type=os_type,
|
|
241
|
+
drive_id=drive_id,
|
|
242
|
+
status="failed",
|
|
243
|
+
parser_type=parser_type,
|
|
244
|
+
error_message=str(exc),
|
|
245
|
+
metadata={"category": decision["category"]},
|
|
246
|
+
)
|
|
247
|
+
continue
|
|
248
|
+
|
|
249
|
+
if (
|
|
250
|
+
existing
|
|
251
|
+
and existing["sha256"] == digest
|
|
252
|
+
and existing["graph_node_id"]
|
|
253
|
+
and self._local_file_index_has_extracted_text(existing)
|
|
254
|
+
and self._node_matches_workspace(
|
|
255
|
+
conn, existing["graph_node_id"], workspace_id
|
|
256
|
+
)
|
|
257
|
+
):
|
|
258
|
+
counts["skipped_unchanged"] += 1
|
|
259
|
+
self._upsert_local_file_index(
|
|
260
|
+
conn,
|
|
261
|
+
source_id=source_id,
|
|
262
|
+
root=root,
|
|
263
|
+
file_path=file_path,
|
|
264
|
+
stat=stat,
|
|
265
|
+
os_type=os_type,
|
|
266
|
+
drive_id=drive_id,
|
|
267
|
+
status="indexed",
|
|
268
|
+
parser_type=parser_type,
|
|
269
|
+
sha256=digest,
|
|
270
|
+
graph_node_id=existing["graph_node_id"],
|
|
271
|
+
metadata={
|
|
272
|
+
**_safe_loads(existing["metadata_json"]),
|
|
273
|
+
"category": decision["category"],
|
|
274
|
+
"sha256_unchanged": True,
|
|
275
|
+
},
|
|
276
|
+
)
|
|
277
|
+
continue
|
|
278
|
+
|
|
279
|
+
try:
|
|
280
|
+
text, parser_meta = self._extract_local_file_text(
|
|
281
|
+
file_path,
|
|
282
|
+
decision["category"],
|
|
283
|
+
include_ocr=include_ocr,
|
|
284
|
+
)
|
|
285
|
+
text = _clean_text(text)
|
|
286
|
+
parser_meta = {**parser_meta, "extracted_chars": len(text)}
|
|
287
|
+
if not text:
|
|
288
|
+
counts["skipped_empty_text"] += 1
|
|
289
|
+
if existing and existing["graph_node_id"]:
|
|
290
|
+
self._delete_local_file_graph(
|
|
291
|
+
conn, existing["graph_node_id"]
|
|
292
|
+
)
|
|
293
|
+
self._upsert_local_file_index(
|
|
294
|
+
conn,
|
|
295
|
+
source_id=source_id,
|
|
296
|
+
root=root,
|
|
297
|
+
file_path=file_path,
|
|
298
|
+
stat=stat,
|
|
299
|
+
os_type=os_type,
|
|
300
|
+
drive_id=drive_id,
|
|
301
|
+
status="skipped_empty_text",
|
|
302
|
+
parser_type=parser_type,
|
|
303
|
+
sha256=digest,
|
|
304
|
+
error_message="텍스트 추출 결과가 비어 있습니다.",
|
|
305
|
+
metadata={
|
|
306
|
+
"category": decision["category"],
|
|
307
|
+
"parser": parser_meta,
|
|
308
|
+
},
|
|
309
|
+
)
|
|
310
|
+
continue
|
|
311
|
+
graph_node_id = self._upsert_local_file_node(
|
|
312
|
+
conn,
|
|
313
|
+
source_id=source_id,
|
|
314
|
+
root=root,
|
|
315
|
+
file_path=file_path,
|
|
316
|
+
stat=stat,
|
|
317
|
+
os_type=os_type,
|
|
318
|
+
drive_id=drive_id,
|
|
319
|
+
sha256=digest,
|
|
320
|
+
category=decision["category"],
|
|
321
|
+
parser_type=parser_type,
|
|
322
|
+
text=text,
|
|
323
|
+
parser_meta=parser_meta,
|
|
324
|
+
user_email=user_email,
|
|
325
|
+
workspace_id=workspace_id,
|
|
326
|
+
)
|
|
327
|
+
self._upsert_local_file_index(
|
|
328
|
+
conn,
|
|
329
|
+
source_id=source_id,
|
|
330
|
+
root=root,
|
|
331
|
+
file_path=file_path,
|
|
332
|
+
stat=stat,
|
|
333
|
+
os_type=os_type,
|
|
334
|
+
drive_id=drive_id,
|
|
335
|
+
status="indexed",
|
|
336
|
+
parser_type=parser_type,
|
|
337
|
+
sha256=digest,
|
|
338
|
+
graph_node_id=graph_node_id,
|
|
339
|
+
metadata={
|
|
340
|
+
"category": decision["category"],
|
|
341
|
+
"parser": parser_meta,
|
|
342
|
+
},
|
|
343
|
+
)
|
|
344
|
+
counts["indexed"] += 1
|
|
345
|
+
indexed_nodes.append(graph_node_id)
|
|
346
|
+
except Exception as exc:
|
|
347
|
+
counts["failed"] += 1
|
|
348
|
+
errors.append({"path": str(file_path), "error": str(exc)})
|
|
349
|
+
if existing and existing["graph_node_id"]:
|
|
350
|
+
self._delete_local_file_graph(conn, existing["graph_node_id"])
|
|
351
|
+
self._upsert_local_file_index(
|
|
352
|
+
conn,
|
|
353
|
+
source_id=source_id,
|
|
354
|
+
root=root,
|
|
355
|
+
file_path=file_path,
|
|
356
|
+
stat=stat,
|
|
357
|
+
os_type=os_type,
|
|
358
|
+
drive_id=drive_id,
|
|
359
|
+
status="failed",
|
|
360
|
+
parser_type=parser_type,
|
|
361
|
+
sha256=digest,
|
|
362
|
+
error_message=str(exc),
|
|
363
|
+
metadata={"category": decision["category"]},
|
|
364
|
+
)
|
|
365
|
+
|
|
366
|
+
if not limit_reached:
|
|
367
|
+
existing_rows = {
|
|
368
|
+
row["relative_path"]: row["graph_node_id"]
|
|
369
|
+
for row in conn.execute(
|
|
370
|
+
"SELECT relative_path, graph_node_id FROM local_file_index WHERE source_id=?",
|
|
371
|
+
(source_id,),
|
|
372
|
+
)
|
|
373
|
+
}
|
|
374
|
+
deleted_paths = set(existing_rows) - seen_relative_paths
|
|
375
|
+
for relative_path in deleted_paths:
|
|
376
|
+
self._delete_local_file_graph(
|
|
377
|
+
conn, existing_rows.get(relative_path)
|
|
378
|
+
)
|
|
379
|
+
conn.execute(
|
|
380
|
+
"""
|
|
381
|
+
UPDATE local_file_index
|
|
382
|
+
SET status='deleted', deleted=1, last_scanned_at=?, error_message=NULL, graph_node_id=NULL
|
|
383
|
+
WHERE source_id=? AND relative_path=?
|
|
384
|
+
""",
|
|
385
|
+
(_now(), source_id, relative_path),
|
|
386
|
+
)
|
|
387
|
+
counts["deleted"] = len(deleted_paths)
|
|
388
|
+
conn.execute(
|
|
389
|
+
"""
|
|
390
|
+
UPDATE knowledge_sources
|
|
391
|
+
SET status='active', updated_at=?, last_scanned_at=?
|
|
392
|
+
WHERE id=?
|
|
393
|
+
""",
|
|
394
|
+
(_now(), _now(), source_id),
|
|
395
|
+
)
|
|
396
|
+
|
|
397
|
+
return {
|
|
398
|
+
"status": "ok",
|
|
399
|
+
"source": {
|
|
400
|
+
"id": source_id,
|
|
401
|
+
"root_path": str(root),
|
|
402
|
+
"os_type": os_type,
|
|
403
|
+
"drive_id": drive_id,
|
|
404
|
+
"include_ocr": bool(include_ocr),
|
|
405
|
+
"watch_enabled": bool(watch_enabled),
|
|
406
|
+
},
|
|
407
|
+
"counts": dict(counts),
|
|
408
|
+
"indexed_nodes": indexed_nodes[:100],
|
|
409
|
+
"errors": errors[:50],
|
|
410
|
+
"notice": "Lattice AI는 사용자가 선택한 폴더만 AI 지식으로 변환합니다.",
|
|
411
|
+
}
|