ltcai 11.2.0 → 11.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -53
- package/docs/CHANGELOG.md +61 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/MULTI_AGENT_RUNTIME.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +6 -2
- package/docs/PERMISSION_MODE.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +2 -2
- package/docs/v11.3.0_PLAN.md +202 -0
- package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +287 -0
- package/lattice_brain/graph/_kg_common/extraction.py +516 -0
- package/lattice_brain/graph/_kg_common/relations.py +161 -0
- package/lattice_brain/graph/_kg_common/text.py +479 -0
- package/lattice_brain/graph/discovery_index/__init__.py +35 -0
- package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
- package/lattice_brain/graph/discovery_index/extract.py +137 -0
- package/lattice_brain/graph/discovery_index/scan.py +411 -0
- package/lattice_brain/graph/discovery_index/upsert.py +495 -0
- package/lattice_brain/graph/projection/__init__.py +42 -0
- package/lattice_brain/graph/projection/curation.py +500 -0
- package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
- package/lattice_brain/graph/retrieval/__init__.py +54 -0
- package/lattice_brain/graph/retrieval/context.py +197 -0
- package/lattice_brain/graph/retrieval/graph_view.py +319 -0
- package/lattice_brain/graph/retrieval/hybrid.py +488 -0
- package/lattice_brain/graph/retrieval/maintenance.py +121 -0
- package/lattice_brain/graph/retrieval/signals.py +95 -0
- package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
- package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
- package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
- package/lattice_brain/graph/retrieval_vector/search.py +560 -0
- package/lattice_brain/graph/retrieval_vector/status.py +374 -0
- package/lattice_brain/ingestion/__init__.py +130 -0
- package/lattice_brain/ingestion/_contract.py +90 -0
- package/lattice_brain/ingestion/constants.py +127 -0
- package/lattice_brain/ingestion/folder_scan.py +57 -0
- package/lattice_brain/ingestion/folders.py +258 -0
- package/lattice_brain/ingestion/hashing.py +26 -0
- package/lattice_brain/ingestion/jobs_api.py +107 -0
- package/lattice_brain/ingestion/models.py +80 -0
- package/lattice_brain/ingestion/pipeline.py +486 -0
- package/lattice_brain/ingestion/quality.py +209 -0
- package/lattice_brain/ingestion/routing.py +295 -0
- package/lattice_brain/multimodal/__init__.py +164 -0
- package/lattice_brain/multimodal/audio.py +77 -0
- package/lattice_brain/multimodal/common.py +118 -0
- package/lattice_brain/multimodal/images.py +498 -0
- package/lattice_brain/multimodal/ports.py +169 -0
- package/lattice_brain/multimodal/video.py +410 -0
- package/lattice_brain/portability/__init__.py +90 -0
- package/lattice_brain/portability/_contract.py +42 -0
- package/lattice_brain/portability/backups.py +338 -0
- package/lattice_brain/portability/bundles.py +136 -0
- package/lattice_brain/portability/constants.py +93 -0
- package/lattice_brain/portability/fsops.py +138 -0
- package/lattice_brain/portability/service.py +41 -0
- package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
- package/lattice_brain/runtime/__init__.py +1 -1
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/chronicle.py +63 -0
- package/latticeai/core/agent/__init__.py +93 -0
- package/latticeai/core/agent/_contract.py +79 -0
- package/latticeai/core/agent/context.py +57 -0
- package/latticeai/core/agent/deps.py +125 -0
- package/latticeai/core/agent/execution.py +622 -0
- package/latticeai/core/agent/planning.py +145 -0
- package/latticeai/core/agent/recovery.py +157 -0
- package/latticeai/core/agent/runtime.py +210 -0
- package/latticeai/core/agent/verification.py +231 -0
- package/latticeai/core/embedding_providers/__init__.py +151 -0
- package/latticeai/core/embedding_providers/base.py +199 -0
- package/latticeai/core/embedding_providers/captions.py +162 -0
- package/latticeai/core/embedding_providers/profiles.py +126 -0
- package/latticeai/core/embedding_providers/text.py +350 -0
- package/latticeai/core/embedding_providers/vision.py +352 -0
- package/latticeai/core/file_generation/__init__.py +115 -0
- package/latticeai/core/file_generation/bundles.py +76 -0
- package/latticeai/core/file_generation/extraction.py +154 -0
- package/latticeai/core/file_generation/inference.py +235 -0
- package/latticeai/core/file_generation/orchestration.py +152 -0
- package/latticeai/core/file_generation/prompting.py +117 -0
- package/latticeai/core/file_generation/repair.py +114 -0
- package/latticeai/core/file_generation/sanitize.py +61 -0
- package/latticeai/core/file_generation/validation.py +201 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +9 -0
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/integrations/telegram_bot/__init__.py +123 -0
- package/latticeai/integrations/telegram_bot/__main__.py +17 -0
- package/latticeai/integrations/telegram_bot/config.py +86 -0
- package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
- package/latticeai/integrations/telegram_bot/flows.py +478 -0
- package/latticeai/integrations/telegram_bot/helpers.py +322 -0
- package/latticeai/integrations/telegram_bot/screens.py +394 -0
- package/latticeai/models/router/__init__.py +88 -0
- package/latticeai/models/router/_contract.py +66 -0
- package/latticeai/models/router/branding.py +56 -0
- package/latticeai/models/router/catalog.py +69 -0
- package/latticeai/models/router/documents.py +199 -0
- package/latticeai/models/router/errors.py +37 -0
- package/latticeai/models/router/generation.py +258 -0
- package/latticeai/models/router/loading.py +291 -0
- package/latticeai/models/router/local_models.py +85 -0
- package/latticeai/models/router/registry.py +147 -0
- package/latticeai/runtime/build_phases/__init__.py +82 -0
- package/latticeai/runtime/build_phases/features.py +407 -0
- package/latticeai/runtime/build_phases/foundation.py +555 -0
- package/latticeai/runtime/build_phases/web.py +492 -0
- package/latticeai/runtime/runtime_context.py +1 -0
- package/latticeai/services/architecture_readiness.py +48 -19
- package/latticeai/services/brain_intelligence/__init__.py +58 -0
- package/latticeai/services/brain_intelligence/_contract.py +71 -0
- package/latticeai/services/brain_intelligence/consistency.py +193 -0
- package/latticeai/services/brain_intelligence/constants.py +47 -0
- package/latticeai/services/brain_intelligence/digest.py +258 -0
- package/latticeai/services/brain_intelligence/health.py +331 -0
- package/latticeai/services/brain_intelligence/proposals.py +264 -0
- package/latticeai/services/brain_intelligence/sampling.py +84 -0
- package/latticeai/services/brain_intelligence/service.py +48 -0
- package/latticeai/services/chronicle.py +557 -0
- package/latticeai/services/memory_service/__init__.py +52 -0
- package/latticeai/services/memory_service/_contract.py +100 -0
- package/latticeai/services/memory_service/brief.py +431 -0
- package/latticeai/services/memory_service/constants.py +57 -0
- package/latticeai/services/memory_service/maintenance.py +138 -0
- package/latticeai/services/memory_service/manager.py +186 -0
- package/latticeai/services/memory_service/proof.py +136 -0
- package/latticeai/services/memory_service/recall.py +225 -0
- package/latticeai/services/memory_service/service.py +48 -0
- package/latticeai/services/memory_service/stores.py +110 -0
- package/latticeai/services/model_runtime/__init__.py +322 -0
- package/latticeai/services/model_runtime/cloud.py +87 -0
- package/latticeai/services/model_runtime/download.py +282 -0
- package/latticeai/services/model_runtime/engines.py +341 -0
- package/latticeai/services/model_runtime/loading.py +178 -0
- package/latticeai/services/model_runtime/service.py +129 -0
- package/latticeai/services/model_runtime/state.py +131 -0
- package/latticeai/services/model_runtime/status.py +255 -0
- package/latticeai/services/product_readiness.py +15 -7
- package/latticeai/setup/wizard/__init__.py +126 -0
- package/latticeai/setup/wizard/catalog.py +172 -0
- package/latticeai/setup/wizard/detect.py +323 -0
- package/latticeai/setup/wizard/install.py +348 -0
- package/latticeai/setup/wizard/paths.py +168 -0
- package/latticeai/setup/wizard/plans.py +74 -0
- package/latticeai/setup/wizard/recommend.py +320 -0
- package/package.json +6 -2
- package/scripts/bump_version.py +14 -0
- package/scripts/capture_release_evidence.mjs +33 -21
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_i18n_namespace_coverage.mjs +41 -4
- package/scripts/check_max_file_lines.mjs +102 -0
- package/scripts/check_release_evidence_bound.mjs +30 -15
- package/scripts/check_screenshot_pixel_delta.py +34 -4
- package/scripts/check_server_i18n.mjs +1 -0
- package/scripts/generate_rust_parity_fixtures.py +562 -0
- package/scripts/lib/mock_server_fingerprint.mjs +94 -0
- package/scripts/release_screen_claims.json +31 -2
- package/src-tauri/Cargo.lock +361 -3
- package/src-tauri/Cargo.toml +6 -1
- package/src-tauri/src/backend.rs +349 -0
- package/src-tauri/src/folder.rs +33 -0
- package/src-tauri/src/main.rs +97 -399
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +41 -37
- package/static/app/assets/Act-yYpYnn0v.js +1 -0
- package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
- package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
- package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
- package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
- package/static/app/assets/Capture-CFIRsFNE.js +1 -0
- package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
- package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
- package/static/app/assets/Library-DwO3yZST.js +1 -0
- package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
- package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
- package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
- package/static/app/assets/System-DW8F-2xL.js +1 -0
- package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
- package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
- package/static/app/assets/brain-Ci1CkWjM.js +1 -0
- package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
- package/static/app/assets/circle-check-DfInj-qD.js +1 -0
- package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
- package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
- package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
- package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
- package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
- package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
- package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
- package/static/app/assets/index-_u5iUHDr.js +10 -0
- package/static/app/assets/input-B0lPdRQZ.js +1 -0
- package/static/app/assets/link-2-CoFbooHS.js +1 -0
- package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
- package/static/app/assets/primitives-DEbN-d6p.js +1 -0
- package/static/app/assets/search-BybIWPNd.js +1 -0
- package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
- package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
- package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
- package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
- package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
- package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
- package/static/app/assets/utils-BlZr7Pd4.js +4 -0
- package/static/app/assets/workspace-jJY4RuAV.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/graph/_kg_common.py +0 -1331
- package/lattice_brain/graph/discovery_index.py +0 -1141
- package/lattice_brain/graph/retrieval.py +0 -1120
- package/lattice_brain/graph/retrieval_vector.py +0 -1293
- package/lattice_brain/ingestion.py +0 -1525
- package/lattice_brain/multimodal.py +0 -1258
- package/latticeai/core/agent.py +0 -1465
- package/latticeai/core/embedding_providers.py +0 -1196
- package/latticeai/core/file_generation.py +0 -1047
- package/latticeai/integrations/telegram_bot.py +0 -1390
- package/latticeai/models/router.py +0 -1007
- package/latticeai/runtime/build_phases.py +0 -1450
- package/latticeai/services/brain_intelligence.py +0 -1083
- package/latticeai/services/memory_service.py +0 -1177
- package/latticeai/services/model_runtime.py +0 -1281
- package/latticeai/setup/wizard.py +0 -1310
- package/static/app/assets/Act-AWf0SAKp.js +0 -1
- package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
- package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
- package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
- package/static/app/assets/Capture-CqOSzyPr.js +0 -1
- package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
- package/static/app/assets/Library-CX-bbhmK.js +0 -1
- package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
- package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
- package/static/app/assets/System-Bu2t5hn1.js +0 -1
- package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
- package/static/app/assets/brain-DJMoqrwx.js +0 -1
- package/static/app/assets/index-BpYkzcVm.js +0 -10
- package/static/app/assets/input-DSlJJxRs.js +0 -1
- package/static/app/assets/primitives-BCx6TvfG.js +0 -1
- package/static/app/assets/search-Cgy8cCFJ.js +0 -1
- package/static/app/assets/utils-zqPZJxdx.js +0 -4
- package/static/app/assets/workspace-DXTihhfU.js +0 -1
|
@@ -1,1525 +0,0 @@
|
|
|
1
|
-
"""Unified ingestion pipeline — the single write-side seam into the Knowledge Graph.
|
|
2
|
-
|
|
3
|
-
v3.6.0 Knowledge Graph First principle: *no data source bypasses the Knowledge
|
|
4
|
-
Graph and no source creates an isolated silo*. Every source — local files,
|
|
5
|
-
connected folders, PDFs/Markdown/text/code, web URLs, browser tabs — is
|
|
6
|
-
normalized into one :class:`IngestionItem` and pushed through one
|
|
7
|
-
:meth:`IngestionPipeline.ingest` entrypoint:
|
|
8
|
-
|
|
9
|
-
Source → normalize → content hash → (file | text) ingest → provenance
|
|
10
|
-
|
|
11
|
-
The pipeline is deliberately thin. It owns normalization, idempotency reporting,
|
|
12
|
-
provenance capture, and — crucially — routing every ingest through the shared
|
|
13
|
-
``dispatch_tool`` lifecycle so ``pre_tool``/``post_tool`` hooks fire on data
|
|
14
|
-
ingestion exactly as they do on tool calls. The heavy graph construction lives in
|
|
15
|
-
:class:`knowledge_graph.KnowledgeGraphStore` (``ingest_document`` for files,
|
|
16
|
-
``ingest_source`` for text/web), which this module composes rather than
|
|
17
|
-
re-implements.
|
|
18
|
-
|
|
19
|
-
Web ingestion seam
|
|
20
|
-
------------------
|
|
21
|
-
The graph layer never fetches or parses the web. Fetching, rendering,
|
|
22
|
-
readability extraction, and parse quality are the responsibility of the
|
|
23
|
-
*upstream* capture surfaces (browser extension, tools layer, MCP servers):
|
|
24
|
-
they hand this module already-extracted text. :meth:`IngestionPipeline.
|
|
25
|
-
ingest_web_page` is the convenience wrapper for that hand-off — it normalizes
|
|
26
|
-
``(url, extracted_text)`` into an ``IngestionItem(source_type="web_url")`` and
|
|
27
|
-
routes it through the exact same :meth:`IngestionPipeline.ingest` door as every
|
|
28
|
-
other source. If the extracted text is bad, fix the extractor upstream; the
|
|
29
|
-
pipeline will not attempt network access or HTML parsing.
|
|
30
|
-
|
|
31
|
-
Folder ingestion (:meth:`IngestionPipeline.ingest_folder`) walks a local
|
|
32
|
-
directory, honors a gitignore-like ``.latticeignore`` file at the root
|
|
33
|
-
(blank lines, ``#`` comments, ``fnmatch`` glob patterns, ``dir/`` suffix for
|
|
34
|
-
directories), always skips common noise (``.git``, ``node_modules``,
|
|
35
|
-
``__pycache__``, virtualenvs, ``dist``, hidden entries by default), applies
|
|
36
|
-
size/extension filters, and either ingests inline or schedules through the
|
|
37
|
-
existing :class:`BackgroundIngestionQueue`.
|
|
38
|
-
"""
|
|
39
|
-
|
|
40
|
-
from __future__ import annotations
|
|
41
|
-
|
|
42
|
-
import fnmatch
|
|
43
|
-
import hashlib
|
|
44
|
-
import os
|
|
45
|
-
from dataclasses import dataclass, field
|
|
46
|
-
from pathlib import Path
|
|
47
|
-
from typing import Any, Dict, Iterable, List, Optional, Tuple
|
|
48
|
-
|
|
49
|
-
from .gates import FeatureGate
|
|
50
|
-
from .graph.vector_index import DEFAULT_TICK_LIMIT as VECTOR_TICK_LIMIT
|
|
51
|
-
from .multimodal import (
|
|
52
|
-
AUDIO_EXTENSIONS,
|
|
53
|
-
DEFAULT_KEYFRAMES,
|
|
54
|
-
IMAGE_EXTENSIONS,
|
|
55
|
-
MODALITY_AUDIO,
|
|
56
|
-
MODALITY_IMAGE,
|
|
57
|
-
MODALITY_VIDEO,
|
|
58
|
-
VIDEO_EXTENSIONS,
|
|
59
|
-
VIDEO_UNAVAILABLE_DETAIL,
|
|
60
|
-
ImageFacts,
|
|
61
|
-
MultimodalPorts,
|
|
62
|
-
audio_quality_score,
|
|
63
|
-
detect_modality,
|
|
64
|
-
extract_image_facts,
|
|
65
|
-
ffmpeg_available,
|
|
66
|
-
image_quality_score,
|
|
67
|
-
read_video_facts,
|
|
68
|
-
transcribe_audio,
|
|
69
|
-
video_frame_dir,
|
|
70
|
-
video_quality_score,
|
|
71
|
-
write_image_memory,
|
|
72
|
-
write_video_memory,
|
|
73
|
-
)
|
|
74
|
-
from .runtime.hooks import dispatch_tool
|
|
75
|
-
from .utils import utc_now_iso
|
|
76
|
-
|
|
77
|
-
# Source types that arrive as a file on disk (read via ingest_document).
|
|
78
|
-
FILE_SOURCE_TYPES = frozenset({"file", "local_file", "upload", "pdf"})
|
|
79
|
-
# Source types that arrive as extracted text (read via ingest_source).
|
|
80
|
-
TEXT_SOURCE_TYPES = frozenset(
|
|
81
|
-
{"web_url", "browser_tab", "text", "markdown", "note", "code", "clipboard"}
|
|
82
|
-
)
|
|
83
|
-
# Conversational exchanges (read via ingest_message — role/content semantics,
|
|
84
|
-
# conversation chaining). v4: chat and MCP messages stop bypassing the
|
|
85
|
-
# pipeline, so they carry provenance and fire the hook lifecycle like every
|
|
86
|
-
# other source.
|
|
87
|
-
CHAT_SOURCE_TYPES = frozenset({"chat_message", "mcp_message"})
|
|
88
|
-
# Typed memory records (read via ingest_event → Decision/Experience/Event
|
|
89
|
-
# nodes). The Memory System writes through the same door as everything else.
|
|
90
|
-
MEMORY_SOURCE_TYPES = frozenset({"decision", "experience", "workspace_event"})
|
|
91
|
-
_MEMORY_NODE_TYPES = {"decision": "Decision", "experience": "Experience", "workspace_event": "Event"}
|
|
92
|
-
|
|
93
|
-
DEFAULT_MAX_TEXT_BYTES = 5 * 1024 * 1024 # 5 MB of extracted text per item
|
|
94
|
-
|
|
95
|
-
# ── Folder ingestion (ingest_folder) filters ─────────────────────────────────
|
|
96
|
-
# Directories that are always pruned regardless of .latticeignore.
|
|
97
|
-
FOLDER_DEFAULT_SKIP_DIRS = frozenset(
|
|
98
|
-
{
|
|
99
|
-
".git",
|
|
100
|
-
"node_modules",
|
|
101
|
-
"__pycache__",
|
|
102
|
-
".venv",
|
|
103
|
-
"venv",
|
|
104
|
-
"env",
|
|
105
|
-
".pytest_cache",
|
|
106
|
-
".mypy_cache",
|
|
107
|
-
".ruff_cache",
|
|
108
|
-
"dist",
|
|
109
|
-
"build",
|
|
110
|
-
".next",
|
|
111
|
-
"target",
|
|
112
|
-
".cache",
|
|
113
|
-
".idea",
|
|
114
|
-
".vscode",
|
|
115
|
-
}
|
|
116
|
-
)
|
|
117
|
-
# Extension filter matching FILE_SOURCE_TYPES conventions: text/markdown/code
|
|
118
|
-
# are read inline (extracted content → chunks); .pdf routes as source_type
|
|
119
|
-
# "pdf" through ingest_document (content extraction is upstream's concern).
|
|
120
|
-
FOLDER_TEXT_EXTENSIONS = frozenset(
|
|
121
|
-
{".txt", ".md", ".markdown", ".rst", ".csv", ".json", ".yaml", ".yml", ".toml", ".ini"}
|
|
122
|
-
)
|
|
123
|
-
FOLDER_CODE_EXTENSIONS = frozenset(
|
|
124
|
-
{
|
|
125
|
-
".py", ".js", ".ts", ".tsx", ".jsx", ".html", ".css", ".go", ".rs",
|
|
126
|
-
".java", ".c", ".h", ".cpp", ".hpp", ".rb", ".php", ".swift", ".kt",
|
|
127
|
-
".sh", ".sql",
|
|
128
|
-
}
|
|
129
|
-
)
|
|
130
|
-
FOLDER_DOCUMENT_EXTENSIONS = frozenset({".pdf"})
|
|
131
|
-
DEFAULT_FOLDER_EXTENSIONS = (
|
|
132
|
-
FOLDER_TEXT_EXTENSIONS | FOLDER_CODE_EXTENSIONS | FOLDER_DOCUMENT_EXTENSIONS
|
|
133
|
-
)
|
|
134
|
-
DEFAULT_MAX_FILE_BYTES = 4_000_000 # matches the local-index text/code budget
|
|
135
|
-
LATTICEIGNORE_FILENAME = ".latticeignore"
|
|
136
|
-
# Opt-out escape hatch for the post-ingest incremental vector sync.
|
|
137
|
-
AUTO_VECTOR_INDEX_ENV = "LATTICEAI_AUTO_VECTOR_INDEX"
|
|
138
|
-
#: Default *on*, unlike every other gate here: new material has always been made
|
|
139
|
-
#: searchable straight away, and this exists so a settings surface can turn that
|
|
140
|
-
#: off (batch reindex later) without a restart. ``FeatureGate`` parses the env
|
|
141
|
-
#: var with the same words the hand-written opt-out check used, so an untouched
|
|
142
|
-
#: install — including one with a nonsense value — answers exactly as before.
|
|
143
|
-
AUTO_VECTOR_INDEX_GATE = FeatureGate(
|
|
144
|
-
AUTO_VECTOR_INDEX_ENV,
|
|
145
|
-
default=True,
|
|
146
|
-
name="auto_vector_index",
|
|
147
|
-
detail="New material is prepared for semantic search as soon as it lands.",
|
|
148
|
-
)
|
|
149
|
-
|
|
150
|
-
# ── Multi-modal ingestion (v11.1.0 Track 3) ──────────────────────────────────
|
|
151
|
-
# Opt-in, default off, on purpose. Turning it on changes what a folder scan
|
|
152
|
-
# *stores* (pictures and recordings, with OCR and — if a model is loaded —
|
|
153
|
-
# captions and vectors), and that is the user's call, not a default. With the
|
|
154
|
-
# flag off every routing decision below is skipped and behaviour is byte-for-
|
|
155
|
-
# byte what it was before this release.
|
|
156
|
-
ALLOW_MULTIMODAL_ENV = "LATTICEAI_ALLOW_MULTIMODAL"
|
|
157
|
-
#: The multi-modal switch, resolved when it is asked rather than frozen into
|
|
158
|
-
#: ``self`` at construction (v11.2.0). The environment variable is still the
|
|
159
|
-
#: answer for an untouched install — same var, same words, same default off —
|
|
160
|
-
#: but a settings surface can bind a resolver and move it without a restart.
|
|
161
|
-
MULTIMODAL_GATE = FeatureGate(
|
|
162
|
-
ALLOW_MULTIMODAL_ENV,
|
|
163
|
-
default=False,
|
|
164
|
-
name="allow_multimodal",
|
|
165
|
-
detail="Pictures and recordings are only ingested when this is turned on.",
|
|
166
|
-
)
|
|
167
|
-
#: Video is a *sub-switch* of the one above: with multi-modal off nothing about
|
|
168
|
-
#: video happens at all, and with it on video is included unless this is
|
|
169
|
-
#: explicitly turned off. The effective default is therefore still "no video",
|
|
170
|
-
#: and the seam exists so a settings screen can offer pictures without films.
|
|
171
|
-
ALLOW_VIDEO_ENV = "LATTICEAI_ALLOW_VIDEO"
|
|
172
|
-
VIDEO_GATE = FeatureGate(
|
|
173
|
-
ALLOW_VIDEO_ENV,
|
|
174
|
-
default=True,
|
|
175
|
-
name="allow_video",
|
|
176
|
-
detail="Videos are ingested as keyframes plus subtitles when multi-modal is on.",
|
|
177
|
-
)
|
|
178
|
-
#: Source types that name a modality outright (a caller who already knows).
|
|
179
|
-
IMAGE_SOURCE_TYPES = frozenset({"image", "screenshot", "photo"})
|
|
180
|
-
AUDIO_SOURCE_TYPES = frozenset({"audio", "voice_memo", "recording"})
|
|
181
|
-
VIDEO_SOURCE_TYPES = frozenset({"video", "screen_recording", "movie"})
|
|
182
|
-
#: Added to the folder-scan allow-list only while multimodal is enabled.
|
|
183
|
-
FOLDER_MULTIMODAL_EXTENSIONS = IMAGE_EXTENSIONS | AUDIO_EXTENSIONS
|
|
184
|
-
#: Videos join the folder allow-list only when this machine can decode one —
|
|
185
|
-
#: scanning a folder into a pile of refusals is not a feature.
|
|
186
|
-
FOLDER_VIDEO_EXTENSIONS = VIDEO_EXTENSIONS
|
|
187
|
-
#: Graph node type for a recording. ``NodeType.AUDIO`` normalizes this on the
|
|
188
|
-
#: KG v2 write side; the legacy tables keep the label verbatim, which is what
|
|
189
|
-
#: every type-aware read (graph view, context sections, doc-gen) matches on.
|
|
190
|
-
AUDIO_NODE_TYPE = "Audio"
|
|
191
|
-
|
|
192
|
-
# ── Extraction quality heuristics (v9.8.0 A1) ────────────────────────────────
|
|
193
|
-
# Pure heuristics over the extracted text — no model calls, no network. The
|
|
194
|
-
# score is *advisory*: it never blocks an ingest, it only annotates the result
|
|
195
|
-
# so capture surfaces (browser, folder scan) can surface low-quality warnings.
|
|
196
|
-
QUALITY_HIGH_THRESHOLD = 0.7
|
|
197
|
-
QUALITY_LOW_THRESHOLD = 0.4
|
|
198
|
-
QUALITY_LOW_WARNING = "추출 품질이 낮습니다 — 원문 확인을 권장합니다."
|
|
199
|
-
_WEB_SOURCE_TYPES = frozenset({"web_url", "browser_tab"})
|
|
200
|
-
# Standalone short lines that smell like leftover site chrome (nav/menu/footer).
|
|
201
|
-
_BOILERPLATE_LINE_MARKERS = frozenset(
|
|
202
|
-
{
|
|
203
|
-
"home", "menu", "nav", "navigation", "login", "log in", "sign in",
|
|
204
|
-
"sign up", "register", "subscribe", "search", "about", "about us",
|
|
205
|
-
"contact", "contact us", "privacy policy", "terms of service",
|
|
206
|
-
"cookie policy", "accept cookies", "accept all cookies", "share",
|
|
207
|
-
"skip to content", "copyright", "all rights reserved", "sitemap",
|
|
208
|
-
"back to top", "footer", "read more", "next", "previous",
|
|
209
|
-
}
|
|
210
|
-
)
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
def _quality_level(score: float) -> str:
|
|
214
|
-
if score >= QUALITY_HIGH_THRESHOLD:
|
|
215
|
-
return "high"
|
|
216
|
-
if score >= QUALITY_LOW_THRESHOLD:
|
|
217
|
-
return "medium"
|
|
218
|
-
return "low"
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
def assess_extraction_quality(
|
|
222
|
-
text: Optional[str],
|
|
223
|
-
*,
|
|
224
|
-
source_type: Optional[str] = None,
|
|
225
|
-
upstream_confidence: Optional[Any] = None,
|
|
226
|
-
) -> Dict[str, Any]:
|
|
227
|
-
"""Score extracted text 0..1 with reasons (pure heuristic, deterministic).
|
|
228
|
-
|
|
229
|
-
Signals: text length, whitespace ratio, character/word diversity
|
|
230
|
-
(repetition), sentence structure, and — for web sources — leftover
|
|
231
|
-
nav/menu boilerplate. When the upstream extractor supplies its own
|
|
232
|
-
confidence (``upstream_confidence``), that value wins verbatim: the
|
|
233
|
-
extractor saw the raw document, this function only sees its output.
|
|
234
|
-
"""
|
|
235
|
-
if upstream_confidence is not None:
|
|
236
|
-
try:
|
|
237
|
-
score = max(0.0, min(1.0, float(upstream_confidence)))
|
|
238
|
-
except (TypeError, ValueError):
|
|
239
|
-
score = None
|
|
240
|
-
if score is not None:
|
|
241
|
-
return {
|
|
242
|
-
"score": round(score, 4),
|
|
243
|
-
"level": _quality_level(score),
|
|
244
|
-
"reasons": ["upstream_confidence"],
|
|
245
|
-
}
|
|
246
|
-
|
|
247
|
-
raw = str(text or "")
|
|
248
|
-
stripped = raw.strip()
|
|
249
|
-
if not stripped:
|
|
250
|
-
return {"score": 0.0, "level": "low", "reasons": ["empty_text"]}
|
|
251
|
-
|
|
252
|
-
reasons: List[str] = []
|
|
253
|
-
length = len(stripped)
|
|
254
|
-
sample = stripped[:4000]
|
|
255
|
-
lines = [ln.strip() for ln in stripped.splitlines() if ln.strip()]
|
|
256
|
-
words = stripped.split()
|
|
257
|
-
|
|
258
|
-
# 1) Length — very short extractions rarely carry recall value.
|
|
259
|
-
if length < 40:
|
|
260
|
-
length_factor = 0.35
|
|
261
|
-
reasons.append("very_short_text")
|
|
262
|
-
elif length < 120:
|
|
263
|
-
length_factor = 0.6
|
|
264
|
-
reasons.append("short_text")
|
|
265
|
-
elif length < 300:
|
|
266
|
-
length_factor = 0.85
|
|
267
|
-
else:
|
|
268
|
-
length_factor = 1.0
|
|
269
|
-
|
|
270
|
-
# 2) Sentence structure — prose has sentence-ending punctuation.
|
|
271
|
-
sentence_marks = sum(sample.count(mark) for mark in (".", "!", "?", "…", "。", "!", "?"))
|
|
272
|
-
if sentence_marks > 0:
|
|
273
|
-
structure_factor = 1.0
|
|
274
|
-
elif length < 200:
|
|
275
|
-
structure_factor = 0.75 # titles/snippets legitimately lack periods
|
|
276
|
-
else:
|
|
277
|
-
structure_factor = 0.45
|
|
278
|
-
reasons.append("no_sentence_structure")
|
|
279
|
-
|
|
280
|
-
# 3) Diversity — repeated characters/lines/words indicate extraction junk.
|
|
281
|
-
diversity_factor = 1.0
|
|
282
|
-
distinct_chars = len(set(sample.lower()))
|
|
283
|
-
if distinct_chars < 10:
|
|
284
|
-
diversity_factor *= 0.2
|
|
285
|
-
reasons.append("low_character_diversity")
|
|
286
|
-
elif distinct_chars < 20:
|
|
287
|
-
diversity_factor *= 0.7
|
|
288
|
-
if len(lines) >= 6:
|
|
289
|
-
top_count = max(lines.count(ln) for ln in set(lines))
|
|
290
|
-
if top_count >= max(3, len(lines) // 4):
|
|
291
|
-
diversity_factor *= 0.5
|
|
292
|
-
reasons.append("repetitive_lines")
|
|
293
|
-
if len(words) >= 30 and (len(set(w.lower() for w in words)) / len(words)) < 0.25:
|
|
294
|
-
diversity_factor *= 0.5
|
|
295
|
-
reasons.append("repetitive_words")
|
|
296
|
-
|
|
297
|
-
# 4) Cleanliness — whitespace floods, fragmented lines, site chrome.
|
|
298
|
-
cleanliness_factor = 1.0
|
|
299
|
-
whitespace_ratio = sum(1 for ch in raw if ch.isspace()) / max(1, len(raw))
|
|
300
|
-
if whitespace_ratio > 0.45:
|
|
301
|
-
cleanliness_factor *= 0.6
|
|
302
|
-
reasons.append("high_whitespace_ratio")
|
|
303
|
-
if len(lines) >= 8:
|
|
304
|
-
short_lines = sum(1 for ln in lines if len(ln.split()) <= 3)
|
|
305
|
-
if short_lines / len(lines) > 0.6:
|
|
306
|
-
cleanliness_factor *= 0.6
|
|
307
|
-
reasons.append("fragmented_lines")
|
|
308
|
-
boilerplate_hits = sum(
|
|
309
|
-
1 for ln in lines if ln.lower().strip(" .:>|•·-–—*") in _BOILERPLATE_LINE_MARKERS
|
|
310
|
-
)
|
|
311
|
-
if lines and boilerplate_hits >= 3 and (boilerplate_hits / len(lines)) > 0.2:
|
|
312
|
-
cleanliness_factor *= 0.35
|
|
313
|
-
if str(source_type or "").lower() in _WEB_SOURCE_TYPES:
|
|
314
|
-
reasons.append("nav_menu_remnants")
|
|
315
|
-
else:
|
|
316
|
-
reasons.append("boilerplate_markers")
|
|
317
|
-
|
|
318
|
-
score = length_factor * structure_factor * diversity_factor * cleanliness_factor
|
|
319
|
-
score = max(0.0, min(1.0, score))
|
|
320
|
-
if not reasons:
|
|
321
|
-
reasons.append("clean_extraction")
|
|
322
|
-
return {"score": round(score, 4), "level": _quality_level(score), "reasons": reasons}
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
# ── capture quality CTA (backlog #9, review §7.2 C) ──────────────────────────
|
|
326
|
-
# Structured verdict over the same extraction-quality schema the rest of the
|
|
327
|
-
# pipeline uses, so capture surfaces (browser extension, read-url) can render
|
|
328
|
-
# an honest "this capture is thin" CTA instead of silently storing junk.
|
|
329
|
-
CAPTURE_SUGGESTIONS_THIN = ["recapture", "paste_manually", "highlight_source"]
|
|
330
|
-
_CAPTURE_REASON_LABELS = {
|
|
331
|
-
"empty_text": "추출된 본문이 비어 있습니다",
|
|
332
|
-
"very_short_text": "추출된 본문이 매우 짧습니다",
|
|
333
|
-
"short_text": "추출된 본문이 짧습니다",
|
|
334
|
-
"no_sentence_structure": "문장 구조가 거의 없습니다",
|
|
335
|
-
"low_character_diversity": "반복 문자가 대부분입니다",
|
|
336
|
-
"repetitive_lines": "같은 줄이 반복됩니다",
|
|
337
|
-
"repetitive_words": "같은 단어가 반복됩니다",
|
|
338
|
-
"high_whitespace_ratio": "공백이 지나치게 많습니다",
|
|
339
|
-
"fragmented_lines": "줄이 잘게 조각나 있습니다",
|
|
340
|
-
"nav_menu_remnants": "메뉴/내비게이션 잔여물이 많습니다",
|
|
341
|
-
"boilerplate_markers": "상용구 텍스트가 많습니다",
|
|
342
|
-
"no_extracted_text": "추출된 텍스트가 없습니다",
|
|
343
|
-
}
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
def capture_quality_verdict(
|
|
347
|
-
extraction_quality: Optional[Dict[str, Any]],
|
|
348
|
-
*,
|
|
349
|
-
source_type: Optional[str] = None,
|
|
350
|
-
) -> Dict[str, Any]:
|
|
351
|
-
"""Structured CTA verdict from a pipeline ``extraction_quality`` dict.
|
|
352
|
-
|
|
353
|
-
``{"status": "thin"|"ok", "reason": str|None, "suggestions": [...],
|
|
354
|
-
"score": float|None, "level": str|None}``. ``thin`` (level == "low", the
|
|
355
|
-
same threshold as the ingest warning) carries actionable suggestions —
|
|
356
|
-
``recapture`` / ``paste_manually`` / ``highlight_source`` — so the UI can
|
|
357
|
-
offer the user a way to fix the capture instead of hiding the problem.
|
|
358
|
-
Deterministic and never raises; ``None`` input yields an honest ``thin``.
|
|
359
|
-
"""
|
|
360
|
-
if not isinstance(extraction_quality, dict):
|
|
361
|
-
return {
|
|
362
|
-
"status": "thin",
|
|
363
|
-
"reason": _CAPTURE_REASON_LABELS["no_extracted_text"],
|
|
364
|
-
"reason_codes": ["no_extracted_text"],
|
|
365
|
-
"suggestions": list(CAPTURE_SUGGESTIONS_THIN),
|
|
366
|
-
"score": None,
|
|
367
|
-
"level": None,
|
|
368
|
-
}
|
|
369
|
-
level = str(extraction_quality.get("level") or "")
|
|
370
|
-
score = extraction_quality.get("score")
|
|
371
|
-
reasons = [str(item) for item in (extraction_quality.get("reasons") or [])]
|
|
372
|
-
thin = level == "low"
|
|
373
|
-
reason = None
|
|
374
|
-
if thin:
|
|
375
|
-
labeled = [
|
|
376
|
-
_CAPTURE_REASON_LABELS[code]
|
|
377
|
-
for code in reasons
|
|
378
|
-
if code in _CAPTURE_REASON_LABELS
|
|
379
|
-
]
|
|
380
|
-
reason = "; ".join(labeled) if labeled else QUALITY_LOW_WARNING
|
|
381
|
-
return {
|
|
382
|
-
"status": "thin" if thin else "ok",
|
|
383
|
-
"reason": reason,
|
|
384
|
-
"reason_codes": reasons if thin else [],
|
|
385
|
-
"suggestions": list(CAPTURE_SUGGESTIONS_THIN) if thin else [],
|
|
386
|
-
"score": score,
|
|
387
|
-
"level": level or None,
|
|
388
|
-
}
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
def _load_latticeignore(root: Path) -> List[str]:
|
|
392
|
-
"""Parse ``root/.latticeignore`` → glob patterns (gitignore-like subset)."""
|
|
393
|
-
ignore_file = root / LATTICEIGNORE_FILENAME
|
|
394
|
-
patterns: List[str] = []
|
|
395
|
-
if not ignore_file.is_file():
|
|
396
|
-
return patterns
|
|
397
|
-
try:
|
|
398
|
-
lines = ignore_file.read_text(encoding="utf-8", errors="ignore").splitlines()
|
|
399
|
-
except OSError:
|
|
400
|
-
return patterns
|
|
401
|
-
for raw in lines:
|
|
402
|
-
line = raw.strip()
|
|
403
|
-
if not line or line.startswith("#"):
|
|
404
|
-
continue
|
|
405
|
-
patterns.append(line)
|
|
406
|
-
return patterns
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
def _matches_ignore(
|
|
410
|
-
rel_posix: str, name: str, *, is_dir: bool, patterns: Iterable[str]
|
|
411
|
-
) -> bool:
|
|
412
|
-
"""fnmatch-based .latticeignore matching.
|
|
413
|
-
|
|
414
|
-
- ``pattern/`` matches directories only (files under it never appear
|
|
415
|
-
because ignored directories are pruned during the walk).
|
|
416
|
-
- Patterns match against both the root-relative posix path and the
|
|
417
|
-
basename, so ``*.log`` and ``docs/draft.md`` both behave as expected.
|
|
418
|
-
"""
|
|
419
|
-
for raw in patterns:
|
|
420
|
-
pattern = raw
|
|
421
|
-
if pattern.endswith("/"):
|
|
422
|
-
if not is_dir:
|
|
423
|
-
continue
|
|
424
|
-
pattern = pattern.rstrip("/")
|
|
425
|
-
pattern = pattern.lstrip("/")
|
|
426
|
-
if not pattern:
|
|
427
|
-
continue
|
|
428
|
-
if fnmatch.fnmatch(rel_posix, pattern) or fnmatch.fnmatch(name, pattern):
|
|
429
|
-
return True
|
|
430
|
-
return False
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
# Background job scheduling + progress lives in its own module (v9.9.6):
|
|
434
|
-
# the pipeline owns "ingest one item", the queue owns "schedule many and
|
|
435
|
-
# report progress". Re-exported so every existing import keeps working.
|
|
436
|
-
from .ingestion_jobs import ( # noqa: E402,F401 — re-export for existing importers
|
|
437
|
-
JOB_ERRORS_CAP,
|
|
438
|
-
BackgroundIngestionJob,
|
|
439
|
-
BackgroundIngestionQueue,
|
|
440
|
-
)
|
|
441
|
-
from .quiet import ( # noqa: E402 — imported after the module constants it depends on
|
|
442
|
-
quiet, # noqa: E402 — imported after the module constants it depends on
|
|
443
|
-
)
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
@dataclass
|
|
447
|
-
class IngestionItem:
|
|
448
|
-
"""A single thing to ingest, normalized across every source type."""
|
|
449
|
-
|
|
450
|
-
source_type: str
|
|
451
|
-
title: Optional[str] = None
|
|
452
|
-
text: Optional[str] = None # text/web sources
|
|
453
|
-
path: Optional[str] = None # file sources
|
|
454
|
-
source_uri: Optional[str] = None
|
|
455
|
-
mime_type: Optional[str] = None
|
|
456
|
-
owner: Optional[str] = None
|
|
457
|
-
workspace_id: Optional[str] = None
|
|
458
|
-
permissions: Optional[Dict[str, Any]] = None
|
|
459
|
-
captured_at: Optional[str] = None
|
|
460
|
-
modified_at: Optional[str] = None
|
|
461
|
-
conversation_id: Optional[str] = None
|
|
462
|
-
agent_used: Optional[str] = None
|
|
463
|
-
metadata: Dict[str, Any] = field(default_factory=dict)
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
@dataclass
|
|
467
|
-
class IngestionResult:
|
|
468
|
-
"""The outcome of one ingestion, including provenance and idempotency."""
|
|
469
|
-
|
|
470
|
-
status: str # ok | unavailable | blocked | failed
|
|
471
|
-
source_type: str
|
|
472
|
-
node_id: Optional[str] = None
|
|
473
|
-
source_node_id: Optional[str] = None
|
|
474
|
-
content_hash: Optional[str] = None
|
|
475
|
-
title: Optional[str] = None
|
|
476
|
-
chunk_ids: List[str] = field(default_factory=list)
|
|
477
|
-
chunk_count: int = 0
|
|
478
|
-
duplicate: bool = False
|
|
479
|
-
embedded: bool = False
|
|
480
|
-
indexing_status: str = "pending" # indexed | skipped | failed | pending
|
|
481
|
-
provenance_id: Optional[str] = None
|
|
482
|
-
detail: Optional[str] = None
|
|
483
|
-
# v9.8.0 additive quality fields — advisory only, never gate behavior.
|
|
484
|
-
extraction_quality: Optional[Dict[str, Any]] = None
|
|
485
|
-
warnings: List[str] = field(default_factory=list)
|
|
486
|
-
quality_gate: Optional[Dict[str, Any]] = None
|
|
487
|
-
|
|
488
|
-
def as_dict(self) -> Dict[str, Any]:
|
|
489
|
-
payload: Dict[str, Any] = {
|
|
490
|
-
"status": self.status,
|
|
491
|
-
"source_type": self.source_type,
|
|
492
|
-
"node_id": self.node_id,
|
|
493
|
-
"source_node_id": self.source_node_id,
|
|
494
|
-
"content_hash": self.content_hash,
|
|
495
|
-
"title": self.title,
|
|
496
|
-
"chunk_ids": self.chunk_ids,
|
|
497
|
-
"chunk_count": self.chunk_count,
|
|
498
|
-
"duplicate": self.duplicate,
|
|
499
|
-
"embedded": self.embedded,
|
|
500
|
-
"indexing_status": self.indexing_status,
|
|
501
|
-
"provenance_id": self.provenance_id,
|
|
502
|
-
"detail": self.detail,
|
|
503
|
-
}
|
|
504
|
-
# Additive keys only when populated so pre-v9.8 payloads are unchanged.
|
|
505
|
-
if self.extraction_quality is not None:
|
|
506
|
-
payload["extraction_quality"] = self.extraction_quality
|
|
507
|
-
if self.warnings:
|
|
508
|
-
payload["warnings"] = list(self.warnings)
|
|
509
|
-
if self.quality_gate is not None:
|
|
510
|
-
payload["quality_gate"] = self.quality_gate
|
|
511
|
-
return payload
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
class IngestionPipeline:
|
|
515
|
-
"""Single normalized entrypoint that feeds every source into the graph."""
|
|
516
|
-
|
|
517
|
-
def __init__(
|
|
518
|
-
self,
|
|
519
|
-
knowledge_graph: Any,
|
|
520
|
-
*,
|
|
521
|
-
hooks: Any = None,
|
|
522
|
-
enable_graph: bool = True,
|
|
523
|
-
audit: Optional[Any] = None,
|
|
524
|
-
max_text_bytes: int = DEFAULT_MAX_TEXT_BYTES,
|
|
525
|
-
pipeline_name: str = "unified-ingestion",
|
|
526
|
-
bg_queue: Optional[BackgroundIngestionQueue] = None,
|
|
527
|
-
auto_vector_index: bool = True,
|
|
528
|
-
allow_multimodal: bool = False,
|
|
529
|
-
multimodal: Optional[MultimodalPorts] = None,
|
|
530
|
-
) -> None:
|
|
531
|
-
self._kg = knowledge_graph
|
|
532
|
-
self._hooks = hooks
|
|
533
|
-
self._enable = bool(enable_graph)
|
|
534
|
-
self._audit = audit
|
|
535
|
-
self._max_text_bytes = int(max_text_bytes)
|
|
536
|
-
self._pipeline_name = pipeline_name
|
|
537
|
-
# Background job state lives in the graph database by default, so a
|
|
538
|
-
# restart resumes from the last completed item instead of replaying the
|
|
539
|
-
# whole corpus. A store without a usable ``db_path`` (mocks, disabled
|
|
540
|
-
# graph) degrades to the historical in-memory queue, which reports
|
|
541
|
-
# itself as non-durable through ``BackgroundIngestionQueue.describe()``.
|
|
542
|
-
self._bg_queue = bg_queue or BackgroundIngestionQueue(
|
|
543
|
-
db_path=getattr(knowledge_graph, "db_path", None)
|
|
544
|
-
)
|
|
545
|
-
# Incremental vector sync after each successful non-duplicate ingest.
|
|
546
|
-
# Constructor opt-out AND gate opt-out (LATTICEAI_AUTO_VECTOR_INDEX=0,
|
|
547
|
-
# or the settings toggle bound to it) both disable it; a vector failure
|
|
548
|
-
# never fails the ingest. The gate half is asked per ingest rather than
|
|
549
|
-
# frozen here, so turning it off takes effect on the next item.
|
|
550
|
-
self._auto_vector_index_opt_in = bool(auto_vector_index)
|
|
551
|
-
# Multi-modal routing. Off unless the caller asks for it *or* the gate
|
|
552
|
-
# says yes — the env behind that gate is the escape hatch for an
|
|
553
|
-
# install with no code path to the constructor (CLI, background
|
|
554
|
-
# worker), and the gate is now asked per call so a runtime toggle can
|
|
555
|
-
# reach it. A constructor ``True`` is still a permanent yes.
|
|
556
|
-
self._multimodal_opt_in = bool(allow_multimodal)
|
|
557
|
-
self._multimodal = multimodal or MultimodalPorts()
|
|
558
|
-
self._keyframes = DEFAULT_KEYFRAMES
|
|
559
|
-
|
|
560
|
-
@property
|
|
561
|
-
def _auto_vector_index(self) -> bool:
|
|
562
|
-
"""Whether a landed ingest also syncs its vector, asked *now*."""
|
|
563
|
-
return self._auto_vector_index_opt_in and AUTO_VECTOR_INDEX_GATE.enabled()
|
|
564
|
-
|
|
565
|
-
@property
|
|
566
|
-
def _allow_multimodal(self) -> bool:
|
|
567
|
-
"""Whether pictures and recordings route by modality, asked *now*."""
|
|
568
|
-
return self._multimodal_opt_in or MULTIMODAL_GATE.enabled()
|
|
569
|
-
|
|
570
|
-
@property
|
|
571
|
-
def _allow_video(self) -> bool:
|
|
572
|
-
"""Video needs multi-modal on, its own sub-switch on, and a decoder."""
|
|
573
|
-
return self._allow_multimodal and VIDEO_GATE.enabled() and self._can_decode_video()
|
|
574
|
-
|
|
575
|
-
def _can_decode_video(self) -> bool:
|
|
576
|
-
"""An injected keyframe port counts as a decoder; otherwise, ffmpeg."""
|
|
577
|
-
return self._multimodal.keyframe_extractor is not None or ffmpeg_available()
|
|
578
|
-
|
|
579
|
-
def available(self) -> bool:
|
|
580
|
-
return self._enable and self._kg is not None
|
|
581
|
-
|
|
582
|
-
def multimodal_status(self) -> Dict[str, Any]:
|
|
583
|
-
"""What this pipeline will do with a picture or a recording, honestly.
|
|
584
|
-
|
|
585
|
-
``enabled`` is the flag; the rest is which model-backed capabilities
|
|
586
|
-
were actually injected. Video reports whether it can really run — the
|
|
587
|
-
answer is no on a machine with no ffmpeg, and it says which of the two
|
|
588
|
-
reasons applies rather than leaving the surface to guess.
|
|
589
|
-
"""
|
|
590
|
-
allowed = self._allow_multimodal
|
|
591
|
-
video = self._allow_video
|
|
592
|
-
return {
|
|
593
|
-
"enabled": allowed,
|
|
594
|
-
"image": allowed,
|
|
595
|
-
"audio": allowed,
|
|
596
|
-
"video": video,
|
|
597
|
-
"video_detail": None if video else self._video_refusal(),
|
|
598
|
-
"gates": {
|
|
599
|
-
"multimodal": MULTIMODAL_GATE.describe(),
|
|
600
|
-
"video": VIDEO_GATE.describe(),
|
|
601
|
-
},
|
|
602
|
-
**self._multimodal.describe(),
|
|
603
|
-
}
|
|
604
|
-
|
|
605
|
-
def _video_refusal(self) -> str:
|
|
606
|
-
"""Why a video would be refused right now — never a stale reason."""
|
|
607
|
-
if not self._allow_multimodal:
|
|
608
|
-
return (
|
|
609
|
-
"multi-modal ingestion is off; pictures, recordings and videos "
|
|
610
|
-
f"are only stored when {ALLOW_MULTIMODAL_ENV} is on"
|
|
611
|
-
)
|
|
612
|
-
if not VIDEO_GATE.enabled():
|
|
613
|
-
return (
|
|
614
|
-
"video ingestion is turned off for this install "
|
|
615
|
-
f"({ALLOW_VIDEO_ENV}); pictures and recordings are unaffected"
|
|
616
|
-
)
|
|
617
|
-
return VIDEO_UNAVAILABLE_DETAIL
|
|
618
|
-
|
|
619
|
-
# ── public API ───────────────────────────────────────────────────────────
|
|
620
|
-
def ingest(self, item: IngestionItem, *, user_email: Optional[str] = None) -> IngestionResult:
|
|
621
|
-
"""Normalize, hash, route through dispatch_tool, and record provenance."""
|
|
622
|
-
source_type = str(item.source_type or "text").strip().lower()
|
|
623
|
-
if not self.available():
|
|
624
|
-
return IngestionResult(
|
|
625
|
-
status="unavailable", source_type=source_type,
|
|
626
|
-
indexing_status="skipped",
|
|
627
|
-
detail="Knowledge Graph is disabled (LATTICEAI_ENABLE_GRAPH).",
|
|
628
|
-
)
|
|
629
|
-
|
|
630
|
-
# Modality routing is a no-op while the flag is off: ``modality`` stays
|
|
631
|
-
# "text" and every branch below behaves exactly as it did before.
|
|
632
|
-
modality = self._modality_for(item, source_type)
|
|
633
|
-
if modality == MODALITY_VIDEO and not self._allow_video:
|
|
634
|
-
# Recognized and refused, with the reason that actually applies
|
|
635
|
-
# right now — a missing decoder is not the same answer as a
|
|
636
|
-
# switched-off feature, and the caller can act on the difference.
|
|
637
|
-
return IngestionResult(
|
|
638
|
-
status="unavailable", source_type=source_type,
|
|
639
|
-
indexing_status="skipped", detail=self._video_refusal(),
|
|
640
|
-
)
|
|
641
|
-
|
|
642
|
-
captured_at = item.captured_at or utc_now_iso()
|
|
643
|
-
owner = item.owner or user_email
|
|
644
|
-
tool_name = f"kg_ingest.{source_type}"
|
|
645
|
-
# Only the keys are read by the hook payload, so this dict is safe/cheap.
|
|
646
|
-
args = {
|
|
647
|
-
"source_type": source_type,
|
|
648
|
-
"source_uri": item.source_uri,
|
|
649
|
-
"owner": owner,
|
|
650
|
-
"workspace_id": item.workspace_id,
|
|
651
|
-
}
|
|
652
|
-
|
|
653
|
-
def _run() -> Dict[str, Any]:
|
|
654
|
-
if source_type in CHAT_SOURCE_TYPES:
|
|
655
|
-
return self._ingest_chat(item, source_type=source_type, owner=owner)
|
|
656
|
-
if source_type in MEMORY_SOURCE_TYPES:
|
|
657
|
-
return self._ingest_memory_record(item, source_type=source_type, owner=owner)
|
|
658
|
-
if modality == MODALITY_IMAGE:
|
|
659
|
-
return self._ingest_image(item, source_type=source_type, owner=owner, captured_at=captured_at)
|
|
660
|
-
if modality == MODALITY_AUDIO:
|
|
661
|
-
return self._ingest_audio(item, source_type=source_type, owner=owner, captured_at=captured_at)
|
|
662
|
-
if modality == MODALITY_VIDEO:
|
|
663
|
-
return self._ingest_video(item, source_type=source_type, owner=owner, captured_at=captured_at)
|
|
664
|
-
if source_type in FILE_SOURCE_TYPES or (item.path and not item.text):
|
|
665
|
-
return self._ingest_file(item, source_type=source_type, owner=owner, captured_at=captured_at)
|
|
666
|
-
return self._ingest_text(item, source_type=source_type, owner=owner, captured_at=captured_at)
|
|
667
|
-
|
|
668
|
-
# v9.8.0 observation-only quality gate: computed *before* the write so
|
|
669
|
-
# the search never matches the node we are about to create. It is
|
|
670
|
-
# recorded on the result and never skips an ingest (behavior unchanged).
|
|
671
|
-
quality_text = self._extractable_text(item)
|
|
672
|
-
quality_gate = self._observe_quality_gate(
|
|
673
|
-
item, source_type=source_type, text=quality_text,
|
|
674
|
-
)
|
|
675
|
-
|
|
676
|
-
try:
|
|
677
|
-
raw = dispatch_tool(
|
|
678
|
-
self._hooks, tool_name, args, _run,
|
|
679
|
-
user_email=user_email, workspace_id=item.workspace_id, source="ingestion",
|
|
680
|
-
)
|
|
681
|
-
except PermissionError as exc:
|
|
682
|
-
return IngestionResult(
|
|
683
|
-
status="blocked", source_type=source_type,
|
|
684
|
-
indexing_status="skipped", detail=str(exc),
|
|
685
|
-
)
|
|
686
|
-
except FileNotFoundError as exc:
|
|
687
|
-
return IngestionResult(
|
|
688
|
-
status="failed", source_type=source_type,
|
|
689
|
-
indexing_status="failed", detail=str(exc),
|
|
690
|
-
)
|
|
691
|
-
except Exception as exc: # noqa: BLE001 — surface as a failed result, never crash the caller
|
|
692
|
-
return IngestionResult(
|
|
693
|
-
status="failed", source_type=source_type,
|
|
694
|
-
indexing_status="failed", detail=str(exc),
|
|
695
|
-
)
|
|
696
|
-
|
|
697
|
-
node_id = raw.get("node_id")
|
|
698
|
-
content_hash = raw.get("content_hash") or raw.get("sha256")
|
|
699
|
-
chunk_ids = list(raw.get("chunk_ids") or [])
|
|
700
|
-
title = raw.get("title") or item.title
|
|
701
|
-
|
|
702
|
-
# Incremental vector-index sync (opt-in via auto_vector_index +
|
|
703
|
-
# LATTICEAI_AUTO_VECTOR_INDEX). Exception-safe by contract: the graph
|
|
704
|
-
# write above already landed, so a vector failure only downgrades
|
|
705
|
-
# indexing_status to "pending" — index_status()/rebuild_vector_index()
|
|
706
|
-
# discover the same node as backlog and pick it up later.
|
|
707
|
-
indexing_status = "indexed"
|
|
708
|
-
vector_detail: Optional[str] = None
|
|
709
|
-
if node_id and self._auto_vector_index and not bool(raw.get("duplicate")):
|
|
710
|
-
indexing_status, vector_detail = self._sync_vector_index(node_id)
|
|
711
|
-
embedded = bool(self._kg.node_is_embedded(node_id)) if node_id else False
|
|
712
|
-
|
|
713
|
-
# Provenance capture must never turn an already-persisted ingest into a
|
|
714
|
-
# caller-visible failure: the graph write above succeeded, so a broken
|
|
715
|
-
# provenance table degrades the result instead of raising.
|
|
716
|
-
provenance_detail: Optional[str] = None
|
|
717
|
-
try:
|
|
718
|
-
prov = self._kg.record_provenance(
|
|
719
|
-
node_id=node_id,
|
|
720
|
-
source_type=source_type,
|
|
721
|
-
pipeline=self._pipeline_name,
|
|
722
|
-
source_uri=item.source_uri,
|
|
723
|
-
content_hash=content_hash,
|
|
724
|
-
title=title,
|
|
725
|
-
owner=owner,
|
|
726
|
-
workspace_id=item.workspace_id,
|
|
727
|
-
captured_at=captured_at,
|
|
728
|
-
modified_at=item.modified_at,
|
|
729
|
-
embedded=embedded,
|
|
730
|
-
linked=bool(raw.get("source_node_id")),
|
|
731
|
-
duplicate=bool(raw.get("duplicate")),
|
|
732
|
-
agent_used=item.agent_used,
|
|
733
|
-
chunk_count=len(chunk_ids),
|
|
734
|
-
permissions=item.permissions,
|
|
735
|
-
metadata=item.metadata,
|
|
736
|
-
)
|
|
737
|
-
except Exception as exc: # noqa: BLE001 — the ingest itself already landed
|
|
738
|
-
prov = {}
|
|
739
|
-
provenance_detail = f"provenance capture failed: {exc}"
|
|
740
|
-
if self._audit is not None:
|
|
741
|
-
try:
|
|
742
|
-
self._audit(
|
|
743
|
-
"kg_ingest",
|
|
744
|
-
{
|
|
745
|
-
"source_type": source_type, "node_id": node_id,
|
|
746
|
-
"content_hash": content_hash, "duplicate": bool(raw.get("duplicate")),
|
|
747
|
-
},
|
|
748
|
-
user_email,
|
|
749
|
-
)
|
|
750
|
-
except Exception: # noqa: BLE001 — audit must never break ingestion
|
|
751
|
-
quiet()
|
|
752
|
-
|
|
753
|
-
# A modality-aware door scores its own extraction (a picture's quality
|
|
754
|
-
# is "how much of it can be retrieved", not "does the text read well"),
|
|
755
|
-
# so its verdict wins. Text/file doors never set the key and keep the
|
|
756
|
-
# historical scoring untouched.
|
|
757
|
-
extraction_quality = raw.get("extraction_quality") or self._assess_item_quality(
|
|
758
|
-
item, source_type=source_type, text=quality_text, chunk_ids=chunk_ids,
|
|
759
|
-
)
|
|
760
|
-
warnings: List[str] = []
|
|
761
|
-
if extraction_quality is not None and extraction_quality.get("level") == "low":
|
|
762
|
-
warnings.append(QUALITY_LOW_WARNING)
|
|
763
|
-
|
|
764
|
-
details = [d for d in (provenance_detail, vector_detail) if d]
|
|
765
|
-
return IngestionResult(
|
|
766
|
-
status="ok",
|
|
767
|
-
source_type=source_type,
|
|
768
|
-
node_id=node_id,
|
|
769
|
-
source_node_id=raw.get("source_node_id"),
|
|
770
|
-
content_hash=content_hash,
|
|
771
|
-
title=title,
|
|
772
|
-
chunk_ids=chunk_ids,
|
|
773
|
-
chunk_count=len(chunk_ids),
|
|
774
|
-
duplicate=bool(raw.get("duplicate")),
|
|
775
|
-
embedded=embedded,
|
|
776
|
-
indexing_status=indexing_status,
|
|
777
|
-
provenance_id=prov.get("id"),
|
|
778
|
-
detail="; ".join(details) if details else None,
|
|
779
|
-
extraction_quality=extraction_quality,
|
|
780
|
-
warnings=warnings,
|
|
781
|
-
quality_gate=quality_gate,
|
|
782
|
-
)
|
|
783
|
-
|
|
784
|
-
# ── extraction quality (v9.8.0 A1 — advisory, never gates) ───────────────
|
|
785
|
-
@staticmethod
|
|
786
|
-
def _extractable_text(item: IngestionItem) -> Optional[str]:
|
|
787
|
-
"""Best available extracted text for quality scoring/gating."""
|
|
788
|
-
if item.text is not None:
|
|
789
|
-
return item.text
|
|
790
|
-
extracted = (item.metadata or {}).get("extracted")
|
|
791
|
-
if isinstance(extracted, dict):
|
|
792
|
-
content = extracted.get("content") or extracted.get("text")
|
|
793
|
-
if content is not None:
|
|
794
|
-
return str(content)
|
|
795
|
-
return None
|
|
796
|
-
|
|
797
|
-
@staticmethod
|
|
798
|
-
def _upstream_confidence(item: IngestionItem) -> Optional[Any]:
|
|
799
|
-
"""Upstream extractor confidence, if the capture surface supplied one."""
|
|
800
|
-
meta = item.metadata or {}
|
|
801
|
-
extracted = meta.get("extracted")
|
|
802
|
-
if isinstance(extracted, dict) and extracted.get("confidence") is not None:
|
|
803
|
-
return extracted.get("confidence")
|
|
804
|
-
if meta.get("extraction_confidence") is not None:
|
|
805
|
-
return meta.get("extraction_confidence")
|
|
806
|
-
return None
|
|
807
|
-
|
|
808
|
-
def _assess_item_quality(
|
|
809
|
-
self,
|
|
810
|
-
item: IngestionItem,
|
|
811
|
-
*,
|
|
812
|
-
source_type: str,
|
|
813
|
-
text: Optional[str],
|
|
814
|
-
chunk_ids: List[str],
|
|
815
|
-
) -> Optional[Dict[str, Any]]:
|
|
816
|
-
"""Quality annotation for document-like sources (not chat/memory)."""
|
|
817
|
-
if source_type in CHAT_SOURCE_TYPES or source_type in MEMORY_SOURCE_TYPES:
|
|
818
|
-
return None
|
|
819
|
-
confidence = self._upstream_confidence(item)
|
|
820
|
-
if text is not None or confidence is not None:
|
|
821
|
-
return assess_extraction_quality(
|
|
822
|
-
text, source_type=source_type, upstream_confidence=confidence,
|
|
823
|
-
)
|
|
824
|
-
# File door without inline extraction (e.g. PDF): the pipeline never saw
|
|
825
|
-
# the text, so score honestly from the chunk output instead of guessing.
|
|
826
|
-
if chunk_ids:
|
|
827
|
-
return {
|
|
828
|
-
"score": 0.5,
|
|
829
|
-
"level": "medium",
|
|
830
|
-
"reasons": ["content_extracted_upstream_not_scored"],
|
|
831
|
-
}
|
|
832
|
-
return {"score": 0.0, "level": "low", "reasons": ["no_extracted_text"]}
|
|
833
|
-
|
|
834
|
-
def _observe_quality_gate(
|
|
835
|
-
self,
|
|
836
|
-
item: IngestionItem,
|
|
837
|
-
*,
|
|
838
|
-
source_type: str,
|
|
839
|
-
text: Optional[str],
|
|
840
|
-
) -> Optional[Dict[str, Any]]:
|
|
841
|
-
"""Observation-mode ``gate_ingest_candidate`` wiring.
|
|
842
|
-
|
|
843
|
-
Records what the proactive gate *would* decide (ingest /
|
|
844
|
-
skip_duplicate / review) without ever acting on it. Any failure —
|
|
845
|
-
import, search, gate — yields ``None``; the ingest proceeds untouched.
|
|
846
|
-
"""
|
|
847
|
-
if source_type in CHAT_SOURCE_TYPES or source_type in MEMORY_SOURCE_TYPES:
|
|
848
|
-
return None
|
|
849
|
-
body = str(text or "").strip()
|
|
850
|
-
if not body:
|
|
851
|
-
return None
|
|
852
|
-
try:
|
|
853
|
-
from .graph.proactive import gate_ingest_candidate
|
|
854
|
-
except Exception: # noqa: BLE001 — optional observation, never required
|
|
855
|
-
return None
|
|
856
|
-
|
|
857
|
-
def _search(query: str) -> Any:
|
|
858
|
-
snippet = str(query or "")[:400]
|
|
859
|
-
try:
|
|
860
|
-
if item.workspace_id:
|
|
861
|
-
return self._kg.search(
|
|
862
|
-
snippet, 20, allowed_workspaces={item.workspace_id},
|
|
863
|
-
)
|
|
864
|
-
return self._kg.search(snippet, 20)
|
|
865
|
-
except TypeError:
|
|
866
|
-
# Older store without workspace-scoped search.
|
|
867
|
-
return self._kg.search(snippet, 20)
|
|
868
|
-
|
|
869
|
-
try:
|
|
870
|
-
gate = gate_ingest_candidate(body, _search)
|
|
871
|
-
except Exception: # noqa: BLE001 — observation must never fail the ingest
|
|
872
|
-
return None
|
|
873
|
-
parts = [str(gate.get("reason") or "")]
|
|
874
|
-
if gate.get("similarity") is not None:
|
|
875
|
-
parts.append(f"similarity={gate.get('similarity')}")
|
|
876
|
-
if gate.get("match_id"):
|
|
877
|
-
parts.append(f"match={gate.get('match_id')}")
|
|
878
|
-
return {
|
|
879
|
-
"action": str(gate.get("action") or "review"),
|
|
880
|
-
"detail": "; ".join(p for p in parts if p),
|
|
881
|
-
}
|
|
882
|
-
|
|
883
|
-
def _queue_pending_embed(self, node_id: str, detail: str) -> bool:
|
|
884
|
-
"""Hand a node the inline sync could not embed to the background queue.
|
|
885
|
-
|
|
886
|
-
Before v11.1.0 ``indexing_status="pending"`` was the end of the story:
|
|
887
|
-
honest, but nobody was coming back for it, so the node stayed
|
|
888
|
-
unsearchable until a human ran a rebuild. The durable queue is who
|
|
889
|
-
comes back. A store without one (older stores, mocks) just keeps the
|
|
890
|
-
old behaviour — the node is still visible as ``index_status`` backlog.
|
|
891
|
-
"""
|
|
892
|
-
queue = getattr(self._kg, "vector_queue", None)
|
|
893
|
-
if queue is None:
|
|
894
|
-
return False
|
|
895
|
-
try:
|
|
896
|
-
return bool(queue.schedule(node_id, detail=detail))
|
|
897
|
-
except Exception: # noqa: BLE001 — queueing must never fail an ingest
|
|
898
|
-
quiet()
|
|
899
|
-
return False
|
|
900
|
-
|
|
901
|
-
def _sync_vector_index(self, node_id: str) -> Tuple[str, Optional[str]]:
|
|
902
|
-
"""Best-effort incremental vector sync → (indexing_status, detail).
|
|
903
|
-
|
|
904
|
-
Any failure — missing method on older stores, embedding provider down,
|
|
905
|
-
storage error — yields ``("pending", detail)`` so a later
|
|
906
|
-
``rebuild_vector_index`` run picks the node up from the backlog, and
|
|
907
|
-
the node is queued for background embedding so that pickup happens on
|
|
908
|
-
its own.
|
|
909
|
-
"""
|
|
910
|
-
sync = getattr(self._kg, "index_node_incremental", None)
|
|
911
|
-
if not callable(sync):
|
|
912
|
-
# Older store without the incremental path: the write-side already
|
|
913
|
-
# embeds inline, so nothing extra to do.
|
|
914
|
-
return "indexed", None
|
|
915
|
-
try:
|
|
916
|
-
outcome = sync(node_id) or {}
|
|
917
|
-
except Exception as exc: # noqa: BLE001 — vector sync must never fail the ingest
|
|
918
|
-
return "pending", self._pending_detail(node_id, f"vector index sync failed: {exc}")
|
|
919
|
-
if str(outcome.get("status") or "") == "failed":
|
|
920
|
-
reason = outcome.get("detail") or "unknown error"
|
|
921
|
-
return "pending", self._pending_detail(
|
|
922
|
-
node_id, f"vector index sync failed: {reason}"
|
|
923
|
-
)
|
|
924
|
-
return "indexed", None
|
|
925
|
-
|
|
926
|
-
def _pending_detail(self, node_id: str, reason: str) -> str:
|
|
927
|
-
"""``reason``, plus whether a background retry was actually scheduled."""
|
|
928
|
-
if self._queue_pending_embed(node_id, reason):
|
|
929
|
-
return f"{reason}; queued for background embedding"
|
|
930
|
-
return reason
|
|
931
|
-
|
|
932
|
-
def drain_vector_queue(self, limit: int = VECTOR_TICK_LIMIT) -> Dict[str, Any]:
|
|
933
|
-
"""Run one background-embedding tick over the store's pending backlog.
|
|
934
|
-
|
|
935
|
-
Deliberately caller-driven (a scheduler, a CLI, a test) rather than a
|
|
936
|
-
thread this pipeline owns: the queue is durable, so "who runs it" is a
|
|
937
|
-
deployment decision, not a property of having ingested something.
|
|
938
|
-
"""
|
|
939
|
-
queue = getattr(self._kg, "vector_queue", None)
|
|
940
|
-
if queue is None:
|
|
941
|
-
return {
|
|
942
|
-
"claimed": 0,
|
|
943
|
-
"indexed": 0,
|
|
944
|
-
"retried": 0,
|
|
945
|
-
"failed": 0,
|
|
946
|
-
"detail": "this store has no background vector queue",
|
|
947
|
-
}
|
|
948
|
-
return dict(queue.tick(limit))
|
|
949
|
-
|
|
950
|
-
# --- Large candidate #1: background / incremental scheduling (slice) ---
|
|
951
|
-
def schedule_background(
|
|
952
|
-
self,
|
|
953
|
-
items: List[IngestionItem],
|
|
954
|
-
*,
|
|
955
|
-
incremental: bool = True,
|
|
956
|
-
user_email: Optional[str] = None,
|
|
957
|
-
) -> BackgroundIngestionJob:
|
|
958
|
-
"""Schedule items for background incremental indexing.
|
|
959
|
-
|
|
960
|
-
Returns a job handle. Actual execution can be driven by caller
|
|
961
|
-
(or future worker) calling pipeline.ingest on each — or through
|
|
962
|
-
:meth:`run_background_job`. This seam enables large-corpus scale
|
|
963
|
-
without blocking user requests.
|
|
964
|
-
"""
|
|
965
|
-
job = self._bg_queue.schedule(items, incremental=incremental, user_email=user_email)
|
|
966
|
-
# mark initial status on results concept (jobs track)
|
|
967
|
-
return job
|
|
968
|
-
|
|
969
|
-
def get_background_job(self, job_id: str) -> Optional[BackgroundIngestionJob]:
|
|
970
|
-
return self._bg_queue.get(job_id)
|
|
971
|
-
|
|
972
|
-
def list_background_jobs(self, limit: int = 20) -> List[Dict[str, Any]]:
|
|
973
|
-
"""Recent jobs (newest first) in the frozen ``/api/ingestion`` schema."""
|
|
974
|
-
return [job.as_dict() for job in self._bg_queue.list_recent(limit=limit)]
|
|
975
|
-
|
|
976
|
-
def run_background_job(
|
|
977
|
-
self, job_id: str, *, user_email: Optional[str] = None
|
|
978
|
-
) -> Dict[str, Any]:
|
|
979
|
-
"""Execute a queued/interrupted job's remaining items.
|
|
980
|
-
|
|
981
|
-
Per-item errors are recorded (capped) and never abort the job. The
|
|
982
|
-
final status is ``completed`` (all done), ``partial`` (some done),
|
|
983
|
-
or ``failed`` (nothing done). Already-completed items are skipped, so
|
|
984
|
-
the same method safely powers both first-run and resume.
|
|
985
|
-
"""
|
|
986
|
-
job = self._bg_queue.get(job_id)
|
|
987
|
-
if job is None:
|
|
988
|
-
return {"status": "not_found", "job_id": job_id}
|
|
989
|
-
if job.status == "running":
|
|
990
|
-
return job.as_dict()
|
|
991
|
-
return self._execute_background_job(job, user_email=user_email)
|
|
992
|
-
|
|
993
|
-
def resume_background_job(
|
|
994
|
-
self, job_id: str, *, user_email: Optional[str] = None
|
|
995
|
-
) -> Dict[str, Any]:
|
|
996
|
-
"""Resume an interrupted/partial/failed job from its remaining items."""
|
|
997
|
-
return self.run_background_job(job_id, user_email=user_email)
|
|
998
|
-
|
|
999
|
-
def _execute_background_job(
|
|
1000
|
-
self, job: BackgroundIngestionJob, *, user_email: Optional[str] = None
|
|
1001
|
-
) -> Dict[str, Any]:
|
|
1002
|
-
job.status = "running"
|
|
1003
|
-
# Retried items get a fresh verdict: reset failure state for this run.
|
|
1004
|
-
job.failed = 0
|
|
1005
|
-
job.errors = []
|
|
1006
|
-
job.touch()
|
|
1007
|
-
self._bg_queue.save(job)
|
|
1008
|
-
runner_email = user_email or job.user_email
|
|
1009
|
-
for index in job.remaining_indices():
|
|
1010
|
-
item = job.items[index]
|
|
1011
|
-
try:
|
|
1012
|
-
result = self.ingest(item, user_email=runner_email or item.owner)
|
|
1013
|
-
status, detail = result.status, result.detail
|
|
1014
|
-
except Exception as exc: # noqa: BLE001 — per-item isolation: keep going
|
|
1015
|
-
status, detail = "failed", str(exc)
|
|
1016
|
-
if status == "ok":
|
|
1017
|
-
job.done_indices.add(index)
|
|
1018
|
-
else:
|
|
1019
|
-
job.record_error(index, item, detail or status)
|
|
1020
|
-
job.processed = len(job.done_indices)
|
|
1021
|
-
job.touch()
|
|
1022
|
-
# Checkpoint per item: a crash here must cost at most the item in
|
|
1023
|
-
# flight, never the whole job's progress. One small UPDATE against
|
|
1024
|
-
# an ingest (parse + chunk + embed) is noise.
|
|
1025
|
-
self._bg_queue.save(job)
|
|
1026
|
-
job.processed = len(job.done_indices)
|
|
1027
|
-
if job.total == 0 or job.processed >= job.total:
|
|
1028
|
-
job.status = "completed"
|
|
1029
|
-
elif job.processed > 0:
|
|
1030
|
-
job.status = "partial"
|
|
1031
|
-
else:
|
|
1032
|
-
job.status = "failed"
|
|
1033
|
-
job.touch()
|
|
1034
|
-
self._bg_queue.save(job)
|
|
1035
|
-
return job.as_dict()
|
|
1036
|
-
|
|
1037
|
-
def ingest_web_page(
|
|
1038
|
-
self,
|
|
1039
|
-
url: str,
|
|
1040
|
-
extracted_text: str,
|
|
1041
|
-
*,
|
|
1042
|
-
title: Optional[str] = None,
|
|
1043
|
-
metadata: Optional[Dict[str, Any]] = None,
|
|
1044
|
-
owner: Optional[str] = None,
|
|
1045
|
-
workspace_id: Optional[str] = None,
|
|
1046
|
-
captured_at: Optional[str] = None,
|
|
1047
|
-
user_email: Optional[str] = None,
|
|
1048
|
-
) -> IngestionResult:
|
|
1049
|
-
"""Ingest an *already-extracted* web page (see module docstring seam).
|
|
1050
|
-
|
|
1051
|
-
Fetching/parsing is upstream's responsibility (browser extension /
|
|
1052
|
-
tools layer); this wrapper only normalizes ``(url, extracted_text)``
|
|
1053
|
-
into an ``IngestionItem(source_type="web_url")`` and routes it through
|
|
1054
|
-
the standard :meth:`ingest` door.
|
|
1055
|
-
"""
|
|
1056
|
-
url = str(url or "").strip()
|
|
1057
|
-
if not url:
|
|
1058
|
-
return IngestionResult(
|
|
1059
|
-
status="failed", source_type="web_url",
|
|
1060
|
-
indexing_status="skipped", detail="url required",
|
|
1061
|
-
)
|
|
1062
|
-
text = str(extracted_text or "")
|
|
1063
|
-
if not text.strip():
|
|
1064
|
-
return IngestionResult(
|
|
1065
|
-
status="failed", source_type="web_url",
|
|
1066
|
-
indexing_status="skipped",
|
|
1067
|
-
detail=(
|
|
1068
|
-
"extracted_text required — the graph layer does not fetch or "
|
|
1069
|
-
"parse the web; extraction happens upstream."
|
|
1070
|
-
),
|
|
1071
|
-
)
|
|
1072
|
-
item = IngestionItem(
|
|
1073
|
-
source_type="web_url",
|
|
1074
|
-
title=title or url,
|
|
1075
|
-
text=text,
|
|
1076
|
-
source_uri=url,
|
|
1077
|
-
owner=owner,
|
|
1078
|
-
workspace_id=workspace_id,
|
|
1079
|
-
captured_at=captured_at,
|
|
1080
|
-
metadata=dict(metadata or {}),
|
|
1081
|
-
)
|
|
1082
|
-
return self.ingest(item, user_email=user_email or owner)
|
|
1083
|
-
|
|
1084
|
-
def ingest_folder(
|
|
1085
|
-
self,
|
|
1086
|
-
root_path: Any,
|
|
1087
|
-
*,
|
|
1088
|
-
recursive: bool = True,
|
|
1089
|
-
background: bool = False,
|
|
1090
|
-
extensions: Optional[Iterable[str]] = None,
|
|
1091
|
-
max_file_bytes: int = DEFAULT_MAX_FILE_BYTES,
|
|
1092
|
-
include_hidden: bool = False,
|
|
1093
|
-
max_files: int = 1000,
|
|
1094
|
-
max_errors: int = 25,
|
|
1095
|
-
owner: Optional[str] = None,
|
|
1096
|
-
workspace_id: Optional[str] = None,
|
|
1097
|
-
user_email: Optional[str] = None,
|
|
1098
|
-
) -> Dict[str, Any]:
|
|
1099
|
-
"""Walk ``root_path`` and ingest every eligible file through the pipeline.
|
|
1100
|
-
|
|
1101
|
-
Filtering, in order: hard skip-list directories (``.git`` …), hidden
|
|
1102
|
-
entries (unless ``include_hidden``), root ``.latticeignore`` patterns
|
|
1103
|
-
(fnmatch globs; ``dir/`` suffix prunes directories), extension
|
|
1104
|
-
allow-list, then ``max_file_bytes``. Text/code files are read inline so
|
|
1105
|
-
their content is chunked; ``.pdf`` routes through the file door without
|
|
1106
|
-
inline extraction.
|
|
1107
|
-
|
|
1108
|
-
``background=True`` schedules the built items on the existing
|
|
1109
|
-
:class:`BackgroundIngestionQueue` instead of ingesting inline.
|
|
1110
|
-
Returns a summary dict with counts and per-file errors (capped at
|
|
1111
|
-
``max_errors``).
|
|
1112
|
-
"""
|
|
1113
|
-
summary: Dict[str, Any] = {
|
|
1114
|
-
"root": str(root_path),
|
|
1115
|
-
"recursive": bool(recursive),
|
|
1116
|
-
"background": bool(background),
|
|
1117
|
-
"scanned": 0,
|
|
1118
|
-
"matched": 0,
|
|
1119
|
-
"ingested": 0,
|
|
1120
|
-
"duplicate": 0,
|
|
1121
|
-
"failed": 0,
|
|
1122
|
-
"skipped": {"ignored": 0, "extension": 0, "too_large": 0, "hidden": 0},
|
|
1123
|
-
"truncated": False,
|
|
1124
|
-
"errors": [],
|
|
1125
|
-
}
|
|
1126
|
-
try:
|
|
1127
|
-
root = Path(root_path).expanduser()
|
|
1128
|
-
except TypeError:
|
|
1129
|
-
summary.update(status="failed", detail=f"invalid root path: {root_path!r}")
|
|
1130
|
-
return summary
|
|
1131
|
-
if not root.is_dir():
|
|
1132
|
-
summary.update(status="failed", detail=f"not a directory: {root}")
|
|
1133
|
-
return summary
|
|
1134
|
-
if not self.available():
|
|
1135
|
-
summary.update(
|
|
1136
|
-
status="unavailable",
|
|
1137
|
-
detail="Knowledge Graph is disabled (LATTICEAI_ENABLE_GRAPH).",
|
|
1138
|
-
)
|
|
1139
|
-
return summary
|
|
1140
|
-
summary["root"] = str(root)
|
|
1141
|
-
max_files = max(1, int(max_files))
|
|
1142
|
-
max_errors = max(0, int(max_errors))
|
|
1143
|
-
max_file_bytes = max(1, int(max_file_bytes))
|
|
1144
|
-
allowed_exts = (
|
|
1145
|
-
frozenset(str(e).lower() if str(e).startswith(".") else f".{str(e).lower()}" for e in extensions)
|
|
1146
|
-
if extensions
|
|
1147
|
-
else self._folder_extensions()
|
|
1148
|
-
)
|
|
1149
|
-
patterns = _load_latticeignore(root)
|
|
1150
|
-
errors: List[Dict[str, Any]] = summary["errors"]
|
|
1151
|
-
skipped = summary["skipped"]
|
|
1152
|
-
items: List[IngestionItem] = []
|
|
1153
|
-
|
|
1154
|
-
def _record_error(path: Path, detail: str, status: str = "failed") -> None:
|
|
1155
|
-
summary["failed"] += 1
|
|
1156
|
-
if len(errors) < max_errors:
|
|
1157
|
-
errors.append({"path": str(path), "status": status, "detail": detail})
|
|
1158
|
-
|
|
1159
|
-
for dirpath, dirnames, filenames in os.walk(root):
|
|
1160
|
-
current = Path(dirpath)
|
|
1161
|
-
rel_dir = current.relative_to(root)
|
|
1162
|
-
kept_dirs: List[str] = []
|
|
1163
|
-
for name in sorted(dirnames):
|
|
1164
|
-
if name in FOLDER_DEFAULT_SKIP_DIRS:
|
|
1165
|
-
continue
|
|
1166
|
-
if name.startswith(".") and not include_hidden:
|
|
1167
|
-
continue
|
|
1168
|
-
rel = name if str(rel_dir) == "." else (rel_dir / name).as_posix()
|
|
1169
|
-
if _matches_ignore(rel, name, is_dir=True, patterns=patterns):
|
|
1170
|
-
skipped["ignored"] += 1
|
|
1171
|
-
continue
|
|
1172
|
-
kept_dirs.append(name)
|
|
1173
|
-
dirnames[:] = kept_dirs if recursive else []
|
|
1174
|
-
|
|
1175
|
-
for name in sorted(filenames):
|
|
1176
|
-
if name == LATTICEIGNORE_FILENAME:
|
|
1177
|
-
continue
|
|
1178
|
-
summary["scanned"] += 1
|
|
1179
|
-
path = current / name
|
|
1180
|
-
rel = name if str(rel_dir) == "." else (rel_dir / name).as_posix()
|
|
1181
|
-
if name.startswith(".") and not include_hidden:
|
|
1182
|
-
skipped["hidden"] += 1
|
|
1183
|
-
continue
|
|
1184
|
-
if _matches_ignore(rel, name, is_dir=False, patterns=patterns):
|
|
1185
|
-
skipped["ignored"] += 1
|
|
1186
|
-
continue
|
|
1187
|
-
ext = path.suffix.lower()
|
|
1188
|
-
if ext not in allowed_exts:
|
|
1189
|
-
skipped["extension"] += 1
|
|
1190
|
-
continue
|
|
1191
|
-
try:
|
|
1192
|
-
size = path.stat().st_size
|
|
1193
|
-
except OSError as exc:
|
|
1194
|
-
_record_error(path, f"stat failed: {exc}")
|
|
1195
|
-
continue
|
|
1196
|
-
if size > max_file_bytes:
|
|
1197
|
-
skipped["too_large"] += 1
|
|
1198
|
-
continue
|
|
1199
|
-
if len(items) >= max_files:
|
|
1200
|
-
summary["truncated"] = True
|
|
1201
|
-
break
|
|
1202
|
-
item_metadata: Dict[str, Any] = {"relative_path": rel}
|
|
1203
|
-
if ext in (FOLDER_MULTIMODAL_EXTENSIONS | FOLDER_VIDEO_EXTENSIONS) and self._allow_multimodal:
|
|
1204
|
-
# Routed by modality inside ``ingest``; reading the bytes as
|
|
1205
|
-
# UTF-8 here would only produce mojibake.
|
|
1206
|
-
source_type = "file"
|
|
1207
|
-
elif ext in FOLDER_DOCUMENT_EXTENSIONS:
|
|
1208
|
-
source_type = "pdf"
|
|
1209
|
-
else:
|
|
1210
|
-
source_type = "file"
|
|
1211
|
-
try:
|
|
1212
|
-
content = path.read_text(encoding="utf-8", errors="ignore")
|
|
1213
|
-
except OSError as exc:
|
|
1214
|
-
_record_error(path, f"read failed: {exc}")
|
|
1215
|
-
continue
|
|
1216
|
-
item_metadata["extracted"] = {"content": content, "chars": len(content)}
|
|
1217
|
-
items.append(
|
|
1218
|
-
IngestionItem(
|
|
1219
|
-
source_type=source_type,
|
|
1220
|
-
title=name,
|
|
1221
|
-
path=str(path),
|
|
1222
|
-
source_uri=str(path),
|
|
1223
|
-
owner=owner,
|
|
1224
|
-
workspace_id=workspace_id,
|
|
1225
|
-
metadata=item_metadata,
|
|
1226
|
-
)
|
|
1227
|
-
)
|
|
1228
|
-
if summary["truncated"]:
|
|
1229
|
-
break
|
|
1230
|
-
|
|
1231
|
-
summary["matched"] = len(items)
|
|
1232
|
-
if background:
|
|
1233
|
-
job = self.schedule_background(
|
|
1234
|
-
items, incremental=True, user_email=user_email or owner,
|
|
1235
|
-
)
|
|
1236
|
-
summary.update(status="scheduled", job_id=job.job_id, scheduled=len(items))
|
|
1237
|
-
return summary
|
|
1238
|
-
|
|
1239
|
-
for item in items:
|
|
1240
|
-
result = self.ingest(item, user_email=user_email or owner)
|
|
1241
|
-
if result.status == "ok":
|
|
1242
|
-
if result.duplicate:
|
|
1243
|
-
summary["duplicate"] += 1
|
|
1244
|
-
else:
|
|
1245
|
-
summary["ingested"] += 1
|
|
1246
|
-
else:
|
|
1247
|
-
_record_error(Path(item.path or ""), result.detail or result.status, result.status)
|
|
1248
|
-
summary["status"] = "ok" if summary["failed"] == 0 else "partial"
|
|
1249
|
-
return summary
|
|
1250
|
-
|
|
1251
|
-
def _folder_extensions(self) -> frozenset:
|
|
1252
|
-
"""Folder-scan allow-list — pictures, recordings and films when enabled.
|
|
1253
|
-
|
|
1254
|
-
Video joins only when this machine can actually decode one, so a scan
|
|
1255
|
-
never fills the error list with files it was always going to refuse.
|
|
1256
|
-
"""
|
|
1257
|
-
if not self._allow_multimodal:
|
|
1258
|
-
return DEFAULT_FOLDER_EXTENSIONS
|
|
1259
|
-
allowed = DEFAULT_FOLDER_EXTENSIONS | FOLDER_MULTIMODAL_EXTENSIONS
|
|
1260
|
-
if self._allow_video:
|
|
1261
|
-
return allowed | FOLDER_VIDEO_EXTENSIONS
|
|
1262
|
-
return allowed
|
|
1263
|
-
|
|
1264
|
-
# ── routing helpers ──────────────────────────────────────────────────────
|
|
1265
|
-
def _ingest_text(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
|
|
1266
|
-
text = item.text or ""
|
|
1267
|
-
if not text.strip():
|
|
1268
|
-
raise ValueError(
|
|
1269
|
-
f"Empty content: {source_type} ingestion requires non-empty text."
|
|
1270
|
-
)
|
|
1271
|
-
if len(text.encode("utf-8", "ignore")) > self._max_text_bytes:
|
|
1272
|
-
raise ValueError(
|
|
1273
|
-
f"Text payload exceeds the {self._max_text_bytes // (1024 * 1024)}MB ingestion limit."
|
|
1274
|
-
)
|
|
1275
|
-
title = item.title or item.source_uri or source_type
|
|
1276
|
-
return self._kg.ingest_source(
|
|
1277
|
-
source_type=source_type,
|
|
1278
|
-
title=title,
|
|
1279
|
-
text=text,
|
|
1280
|
-
source_uri=item.source_uri,
|
|
1281
|
-
owner=owner,
|
|
1282
|
-
workspace_id=item.workspace_id,
|
|
1283
|
-
permissions=item.permissions,
|
|
1284
|
-
captured_at=captured_at,
|
|
1285
|
-
modified_at=item.modified_at,
|
|
1286
|
-
conversation_id=item.conversation_id,
|
|
1287
|
-
metadata={"mime_type": item.mime_type, **(item.metadata or {})},
|
|
1288
|
-
)
|
|
1289
|
-
|
|
1290
|
-
def _ingest_chat(self, item, *, source_type, owner) -> Dict[str, Any]:
|
|
1291
|
-
text = item.text or ""
|
|
1292
|
-
meta = item.metadata or {}
|
|
1293
|
-
role = str(meta.get("role") or "user")
|
|
1294
|
-
result = self._kg.ingest_message(
|
|
1295
|
-
role,
|
|
1296
|
-
text,
|
|
1297
|
-
user_email=owner,
|
|
1298
|
-
user_nickname=meta.get("user_nickname"),
|
|
1299
|
-
source=meta.get("source") or source_type,
|
|
1300
|
-
conversation_id=item.conversation_id,
|
|
1301
|
-
workspace_id=item.workspace_id,
|
|
1302
|
-
raw=meta.get("raw"),
|
|
1303
|
-
)
|
|
1304
|
-
# ingest_message reports message/response node ids; normalize the keys
|
|
1305
|
-
# the provenance step expects.
|
|
1306
|
-
result.setdefault("node_id", result.get("node_id") or result.get("message_node_id") or result.get("id"))
|
|
1307
|
-
result.setdefault("title", item.title or text[:80])
|
|
1308
|
-
return result
|
|
1309
|
-
|
|
1310
|
-
def _ingest_memory_record(self, item, *, source_type, owner) -> Dict[str, Any]:
|
|
1311
|
-
node_type = _MEMORY_NODE_TYPES[source_type]
|
|
1312
|
-
meta = item.metadata or {}
|
|
1313
|
-
result = self._kg.ingest_event(
|
|
1314
|
-
node_type,
|
|
1315
|
-
item.title or (item.text or node_type)[:120],
|
|
1316
|
-
user_email=owner,
|
|
1317
|
-
source=meta.get("source") or source_type,
|
|
1318
|
-
conversation_id=item.conversation_id,
|
|
1319
|
-
workspace_id=item.workspace_id,
|
|
1320
|
-
metadata={**meta, "detail": (item.text or "")[:2000]},
|
|
1321
|
-
)
|
|
1322
|
-
result.setdefault("node_id", result.get("node_id") or result.get("id"))
|
|
1323
|
-
result.setdefault("title", item.title)
|
|
1324
|
-
return result
|
|
1325
|
-
|
|
1326
|
-
# ── multi-modal routing (v11.1.0 Track 3) ────────────────────────────────
|
|
1327
|
-
def _modality_for(self, item: IngestionItem, source_type: str) -> str:
|
|
1328
|
-
"""``image`` / ``audio`` / ``video`` / ``text`` for this item.
|
|
1329
|
-
|
|
1330
|
-
Always ``"text"`` while the flag is off, which is what makes "off" mean
|
|
1331
|
-
*unchanged* rather than *slightly different*.
|
|
1332
|
-
"""
|
|
1333
|
-
if not self._allow_multimodal:
|
|
1334
|
-
return "text"
|
|
1335
|
-
if source_type in IMAGE_SOURCE_TYPES:
|
|
1336
|
-
return MODALITY_IMAGE
|
|
1337
|
-
if source_type in AUDIO_SOURCE_TYPES:
|
|
1338
|
-
return MODALITY_AUDIO
|
|
1339
|
-
if source_type in VIDEO_SOURCE_TYPES:
|
|
1340
|
-
return MODALITY_VIDEO
|
|
1341
|
-
if not item.path:
|
|
1342
|
-
return "text"
|
|
1343
|
-
return detect_modality(item.path, item.mime_type)
|
|
1344
|
-
|
|
1345
|
-
def _resolve_file_path(self, item: IngestionItem) -> Path:
|
|
1346
|
-
if not item.path:
|
|
1347
|
-
raise ValueError("File ingestion requires a path.")
|
|
1348
|
-
path = Path(item.path)
|
|
1349
|
-
if not path.exists():
|
|
1350
|
-
raise FileNotFoundError(f"File not found: {path}")
|
|
1351
|
-
if path.is_dir():
|
|
1352
|
-
raise ValueError(f"File ingestion requires a file, got a directory: {path}")
|
|
1353
|
-
return path
|
|
1354
|
-
|
|
1355
|
-
def _ingest_image(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
|
|
1356
|
-
"""Store one picture as an ``Image`` node — OCR, caption, vector.
|
|
1357
|
-
|
|
1358
|
-
The image vector (when a vision model produced one) goes to its own
|
|
1359
|
-
index; the OCR/caption text rides the ordinary text index. That split
|
|
1360
|
-
is what lets a typed question find a screenshot without ever comparing
|
|
1361
|
-
a text vector to an image vector.
|
|
1362
|
-
"""
|
|
1363
|
-
path = self._resolve_file_path(item)
|
|
1364
|
-
facts = extract_image_facts(str(path), ports=self._multimodal)
|
|
1365
|
-
result = write_image_memory(
|
|
1366
|
-
self._kg,
|
|
1367
|
-
path=path,
|
|
1368
|
-
facts=facts,
|
|
1369
|
-
title=item.title or path.name,
|
|
1370
|
-
source_type=source_type if source_type in IMAGE_SOURCE_TYPES else MODALITY_IMAGE,
|
|
1371
|
-
source_uri=item.source_uri,
|
|
1372
|
-
owner=owner,
|
|
1373
|
-
workspace_id=item.workspace_id,
|
|
1374
|
-
conversation_id=item.conversation_id,
|
|
1375
|
-
captured_at=captured_at,
|
|
1376
|
-
modified_at=item.modified_at,
|
|
1377
|
-
permissions=item.permissions,
|
|
1378
|
-
extra_metadata={"mime_type": item.mime_type, **(item.metadata or {})},
|
|
1379
|
-
)
|
|
1380
|
-
self._record_image_vector(result["node_id"], facts)
|
|
1381
|
-
quality = image_quality_score(facts)
|
|
1382
|
-
result["extraction_quality"] = {
|
|
1383
|
-
"score": quality["score"],
|
|
1384
|
-
"level": _quality_level(quality["score"]),
|
|
1385
|
-
"reasons": quality["reasons"],
|
|
1386
|
-
}
|
|
1387
|
-
return result
|
|
1388
|
-
|
|
1389
|
-
def _record_image_vector(self, node_id: str, facts: ImageFacts) -> None:
|
|
1390
|
-
"""File the image-space vector, if a vision model actually made one."""
|
|
1391
|
-
if facts.embedding is None:
|
|
1392
|
-
return
|
|
1393
|
-
from .graph.image_vectors import record_image_vector
|
|
1394
|
-
|
|
1395
|
-
record_image_vector(
|
|
1396
|
-
self._kg,
|
|
1397
|
-
node_id=node_id,
|
|
1398
|
-
vector=facts.embedding,
|
|
1399
|
-
model_id=self._multimodal.vision_model_id or "vision:unnamed",
|
|
1400
|
-
space=self._multimodal.vision_space,
|
|
1401
|
-
updated_at=utc_now_iso(),
|
|
1402
|
-
)
|
|
1403
|
-
|
|
1404
|
-
def _ingest_audio(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
|
|
1405
|
-
"""Store one recording as an ``Audio`` node, transcribed when possible.
|
|
1406
|
-
|
|
1407
|
-
The transcript is text and rides the ordinary text index — chunks,
|
|
1408
|
-
concepts, provenance, dedupe all unchanged — but the node itself is a
|
|
1409
|
-
recording, because that is what it is whether or not anyone could hear
|
|
1410
|
-
it. The recording's own facts stay in the metadata (``modality``,
|
|
1411
|
-
``audio_path``, ``transcription``, ``searchable``). Without a
|
|
1412
|
-
transcriber the memory is still kept, and its body says plainly that
|
|
1413
|
-
the words were never recognized instead of leaving a blank note.
|
|
1414
|
-
"""
|
|
1415
|
-
path = self._resolve_file_path(item)
|
|
1416
|
-
facts = transcribe_audio(str(path), ports=self._multimodal, transcript=item.text)
|
|
1417
|
-
title = item.title or path.stem
|
|
1418
|
-
body = facts.transcript or (
|
|
1419
|
-
f"[{MODALITY_AUDIO}] {title}\n"
|
|
1420
|
-
"이 녹음은 아직 글로 바뀌지 않았습니다 — 음성 인식기가 없어 내용 검색은 되지 않습니다."
|
|
1421
|
-
)
|
|
1422
|
-
result = self._kg.ingest_source(
|
|
1423
|
-
source_type=source_type,
|
|
1424
|
-
title=title,
|
|
1425
|
-
text=body,
|
|
1426
|
-
source_uri=item.source_uri or str(path),
|
|
1427
|
-
owner=owner,
|
|
1428
|
-
workspace_id=item.workspace_id,
|
|
1429
|
-
permissions=item.permissions,
|
|
1430
|
-
captured_at=captured_at,
|
|
1431
|
-
modified_at=item.modified_at,
|
|
1432
|
-
conversation_id=item.conversation_id,
|
|
1433
|
-
node_type=AUDIO_NODE_TYPE,
|
|
1434
|
-
metadata={
|
|
1435
|
-
"mime_type": item.mime_type,
|
|
1436
|
-
"modality": MODALITY_AUDIO,
|
|
1437
|
-
"audio_path": str(path),
|
|
1438
|
-
"audio_bytes": path.stat().st_size,
|
|
1439
|
-
"transcription": facts.transcription_status,
|
|
1440
|
-
"searchable": facts.searchable,
|
|
1441
|
-
**({"transcription_detail": facts.detail} if facts.detail else {}),
|
|
1442
|
-
**(item.metadata or {}),
|
|
1443
|
-
},
|
|
1444
|
-
)
|
|
1445
|
-
result.setdefault("title", title)
|
|
1446
|
-
quality = audio_quality_score(facts)
|
|
1447
|
-
result["extraction_quality"] = {
|
|
1448
|
-
"score": quality["score"],
|
|
1449
|
-
"level": _quality_level(quality["score"]),
|
|
1450
|
-
"reasons": quality["reasons"],
|
|
1451
|
-
}
|
|
1452
|
-
return result
|
|
1453
|
-
|
|
1454
|
-
def _ingest_video(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
|
|
1455
|
-
"""Store one video as keyframes through the image door plus subtitles.
|
|
1456
|
-
|
|
1457
|
-
Nothing here is a new retrieval path: the stills become ordinary
|
|
1458
|
-
``Image`` nodes (OCR, caption, vector, thumbnail) joined by
|
|
1459
|
-
``CONTAINS_IMAGE``, and the subtitle text becomes ordinary chunks. What
|
|
1460
|
-
the ``Video`` node adds is the thing they belong to — and an honest
|
|
1461
|
-
body when there were no subtitles to read.
|
|
1462
|
-
"""
|
|
1463
|
-
path = self._resolve_file_path(item)
|
|
1464
|
-
facts = read_video_facts(
|
|
1465
|
-
str(path),
|
|
1466
|
-
video_frame_dir(getattr(self._kg, "blob_dir", path.parent), _file_digest(path)),
|
|
1467
|
-
count=self._keyframes,
|
|
1468
|
-
ports=self._multimodal,
|
|
1469
|
-
subtitle_text=item.text,
|
|
1470
|
-
)
|
|
1471
|
-
result = write_video_memory(
|
|
1472
|
-
self._kg,
|
|
1473
|
-
path=path,
|
|
1474
|
-
facts=facts,
|
|
1475
|
-
title=item.title or path.stem,
|
|
1476
|
-
source_type=source_type if source_type in VIDEO_SOURCE_TYPES else MODALITY_VIDEO,
|
|
1477
|
-
source_uri=item.source_uri,
|
|
1478
|
-
owner=owner,
|
|
1479
|
-
workspace_id=item.workspace_id,
|
|
1480
|
-
conversation_id=item.conversation_id,
|
|
1481
|
-
captured_at=captured_at,
|
|
1482
|
-
modified_at=item.modified_at,
|
|
1483
|
-
permissions=item.permissions,
|
|
1484
|
-
extra_metadata={"mime_type": item.mime_type, **(item.metadata or {})},
|
|
1485
|
-
ports=self._multimodal,
|
|
1486
|
-
)
|
|
1487
|
-
quality = video_quality_score(facts)
|
|
1488
|
-
result["extraction_quality"] = {
|
|
1489
|
-
"score": quality["score"],
|
|
1490
|
-
"level": _quality_level(quality["score"]),
|
|
1491
|
-
"reasons": quality["reasons"],
|
|
1492
|
-
}
|
|
1493
|
-
return result
|
|
1494
|
-
|
|
1495
|
-
def _ingest_file(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
|
|
1496
|
-
path = self._resolve_file_path(item)
|
|
1497
|
-
return self._kg.ingest_document(
|
|
1498
|
-
path,
|
|
1499
|
-
original_filename=item.title or path.name,
|
|
1500
|
-
mime_type=item.mime_type,
|
|
1501
|
-
uploader=owner,
|
|
1502
|
-
conversation_id=item.conversation_id,
|
|
1503
|
-
extracted=item.metadata.get("extracted") if item.metadata else None,
|
|
1504
|
-
source_type=source_type,
|
|
1505
|
-
source_uri=item.source_uri or str(path),
|
|
1506
|
-
captured_at=captured_at,
|
|
1507
|
-
modified_at=item.modified_at,
|
|
1508
|
-
owner=owner,
|
|
1509
|
-
workspace_id=item.workspace_id,
|
|
1510
|
-
permissions=item.permissions,
|
|
1511
|
-
)
|
|
1512
|
-
|
|
1513
|
-
|
|
1514
|
-
def content_hash_text(text: str) -> str:
|
|
1515
|
-
"""Canonical content hash for a text payload (matches store hashing scheme)."""
|
|
1516
|
-
return hashlib.sha256((text or "").encode("utf-8", "ignore")).hexdigest()
|
|
1517
|
-
|
|
1518
|
-
|
|
1519
|
-
def _file_digest(path: Path) -> str:
|
|
1520
|
-
"""Streaming sha256 of a file — the key a video's frame folder is named by."""
|
|
1521
|
-
digest = hashlib.sha256()
|
|
1522
|
-
with path.open("rb") as handle:
|
|
1523
|
-
for block in iter(lambda: handle.read(1024 * 1024), b""):
|
|
1524
|
-
digest.update(block)
|
|
1525
|
-
return digest.hexdigest()
|