ltcai 11.2.0 → 11.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -53
- package/docs/CHANGELOG.md +61 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/MULTI_AGENT_RUNTIME.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +6 -2
- package/docs/PERMISSION_MODE.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +2 -2
- package/docs/v11.3.0_PLAN.md +202 -0
- package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +287 -0
- package/lattice_brain/graph/_kg_common/extraction.py +516 -0
- package/lattice_brain/graph/_kg_common/relations.py +161 -0
- package/lattice_brain/graph/_kg_common/text.py +479 -0
- package/lattice_brain/graph/discovery_index/__init__.py +35 -0
- package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
- package/lattice_brain/graph/discovery_index/extract.py +137 -0
- package/lattice_brain/graph/discovery_index/scan.py +411 -0
- package/lattice_brain/graph/discovery_index/upsert.py +495 -0
- package/lattice_brain/graph/projection/__init__.py +42 -0
- package/lattice_brain/graph/projection/curation.py +500 -0
- package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
- package/lattice_brain/graph/retrieval/__init__.py +54 -0
- package/lattice_brain/graph/retrieval/context.py +197 -0
- package/lattice_brain/graph/retrieval/graph_view.py +319 -0
- package/lattice_brain/graph/retrieval/hybrid.py +488 -0
- package/lattice_brain/graph/retrieval/maintenance.py +121 -0
- package/lattice_brain/graph/retrieval/signals.py +95 -0
- package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
- package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
- package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
- package/lattice_brain/graph/retrieval_vector/search.py +560 -0
- package/lattice_brain/graph/retrieval_vector/status.py +374 -0
- package/lattice_brain/ingestion/__init__.py +130 -0
- package/lattice_brain/ingestion/_contract.py +90 -0
- package/lattice_brain/ingestion/constants.py +127 -0
- package/lattice_brain/ingestion/folder_scan.py +57 -0
- package/lattice_brain/ingestion/folders.py +258 -0
- package/lattice_brain/ingestion/hashing.py +26 -0
- package/lattice_brain/ingestion/jobs_api.py +107 -0
- package/lattice_brain/ingestion/models.py +80 -0
- package/lattice_brain/ingestion/pipeline.py +486 -0
- package/lattice_brain/ingestion/quality.py +209 -0
- package/lattice_brain/ingestion/routing.py +295 -0
- package/lattice_brain/multimodal/__init__.py +164 -0
- package/lattice_brain/multimodal/audio.py +77 -0
- package/lattice_brain/multimodal/common.py +118 -0
- package/lattice_brain/multimodal/images.py +498 -0
- package/lattice_brain/multimodal/ports.py +169 -0
- package/lattice_brain/multimodal/video.py +410 -0
- package/lattice_brain/portability/__init__.py +90 -0
- package/lattice_brain/portability/_contract.py +42 -0
- package/lattice_brain/portability/backups.py +338 -0
- package/lattice_brain/portability/bundles.py +136 -0
- package/lattice_brain/portability/constants.py +93 -0
- package/lattice_brain/portability/fsops.py +138 -0
- package/lattice_brain/portability/service.py +41 -0
- package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
- package/lattice_brain/runtime/__init__.py +1 -1
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/chronicle.py +63 -0
- package/latticeai/core/agent/__init__.py +93 -0
- package/latticeai/core/agent/_contract.py +79 -0
- package/latticeai/core/agent/context.py +57 -0
- package/latticeai/core/agent/deps.py +125 -0
- package/latticeai/core/agent/execution.py +622 -0
- package/latticeai/core/agent/planning.py +145 -0
- package/latticeai/core/agent/recovery.py +157 -0
- package/latticeai/core/agent/runtime.py +210 -0
- package/latticeai/core/agent/verification.py +231 -0
- package/latticeai/core/embedding_providers/__init__.py +151 -0
- package/latticeai/core/embedding_providers/base.py +199 -0
- package/latticeai/core/embedding_providers/captions.py +162 -0
- package/latticeai/core/embedding_providers/profiles.py +126 -0
- package/latticeai/core/embedding_providers/text.py +350 -0
- package/latticeai/core/embedding_providers/vision.py +352 -0
- package/latticeai/core/file_generation/__init__.py +115 -0
- package/latticeai/core/file_generation/bundles.py +76 -0
- package/latticeai/core/file_generation/extraction.py +154 -0
- package/latticeai/core/file_generation/inference.py +235 -0
- package/latticeai/core/file_generation/orchestration.py +152 -0
- package/latticeai/core/file_generation/prompting.py +117 -0
- package/latticeai/core/file_generation/repair.py +114 -0
- package/latticeai/core/file_generation/sanitize.py +61 -0
- package/latticeai/core/file_generation/validation.py +201 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +9 -0
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/integrations/telegram_bot/__init__.py +123 -0
- package/latticeai/integrations/telegram_bot/__main__.py +17 -0
- package/latticeai/integrations/telegram_bot/config.py +86 -0
- package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
- package/latticeai/integrations/telegram_bot/flows.py +478 -0
- package/latticeai/integrations/telegram_bot/helpers.py +322 -0
- package/latticeai/integrations/telegram_bot/screens.py +394 -0
- package/latticeai/models/router/__init__.py +88 -0
- package/latticeai/models/router/_contract.py +66 -0
- package/latticeai/models/router/branding.py +56 -0
- package/latticeai/models/router/catalog.py +69 -0
- package/latticeai/models/router/documents.py +199 -0
- package/latticeai/models/router/errors.py +37 -0
- package/latticeai/models/router/generation.py +258 -0
- package/latticeai/models/router/loading.py +291 -0
- package/latticeai/models/router/local_models.py +85 -0
- package/latticeai/models/router/registry.py +147 -0
- package/latticeai/runtime/build_phases/__init__.py +82 -0
- package/latticeai/runtime/build_phases/features.py +407 -0
- package/latticeai/runtime/build_phases/foundation.py +555 -0
- package/latticeai/runtime/build_phases/web.py +492 -0
- package/latticeai/runtime/runtime_context.py +1 -0
- package/latticeai/services/architecture_readiness.py +48 -19
- package/latticeai/services/brain_intelligence/__init__.py +58 -0
- package/latticeai/services/brain_intelligence/_contract.py +71 -0
- package/latticeai/services/brain_intelligence/consistency.py +193 -0
- package/latticeai/services/brain_intelligence/constants.py +47 -0
- package/latticeai/services/brain_intelligence/digest.py +258 -0
- package/latticeai/services/brain_intelligence/health.py +331 -0
- package/latticeai/services/brain_intelligence/proposals.py +264 -0
- package/latticeai/services/brain_intelligence/sampling.py +84 -0
- package/latticeai/services/brain_intelligence/service.py +48 -0
- package/latticeai/services/chronicle.py +557 -0
- package/latticeai/services/memory_service/__init__.py +52 -0
- package/latticeai/services/memory_service/_contract.py +100 -0
- package/latticeai/services/memory_service/brief.py +431 -0
- package/latticeai/services/memory_service/constants.py +57 -0
- package/latticeai/services/memory_service/maintenance.py +138 -0
- package/latticeai/services/memory_service/manager.py +186 -0
- package/latticeai/services/memory_service/proof.py +136 -0
- package/latticeai/services/memory_service/recall.py +225 -0
- package/latticeai/services/memory_service/service.py +48 -0
- package/latticeai/services/memory_service/stores.py +110 -0
- package/latticeai/services/model_runtime/__init__.py +322 -0
- package/latticeai/services/model_runtime/cloud.py +87 -0
- package/latticeai/services/model_runtime/download.py +282 -0
- package/latticeai/services/model_runtime/engines.py +341 -0
- package/latticeai/services/model_runtime/loading.py +178 -0
- package/latticeai/services/model_runtime/service.py +129 -0
- package/latticeai/services/model_runtime/state.py +131 -0
- package/latticeai/services/model_runtime/status.py +255 -0
- package/latticeai/services/product_readiness.py +15 -7
- package/latticeai/setup/wizard/__init__.py +126 -0
- package/latticeai/setup/wizard/catalog.py +172 -0
- package/latticeai/setup/wizard/detect.py +323 -0
- package/latticeai/setup/wizard/install.py +348 -0
- package/latticeai/setup/wizard/paths.py +168 -0
- package/latticeai/setup/wizard/plans.py +74 -0
- package/latticeai/setup/wizard/recommend.py +320 -0
- package/package.json +6 -2
- package/scripts/bump_version.py +14 -0
- package/scripts/capture_release_evidence.mjs +33 -21
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_i18n_namespace_coverage.mjs +41 -4
- package/scripts/check_max_file_lines.mjs +102 -0
- package/scripts/check_release_evidence_bound.mjs +30 -15
- package/scripts/check_screenshot_pixel_delta.py +34 -4
- package/scripts/check_server_i18n.mjs +1 -0
- package/scripts/generate_rust_parity_fixtures.py +562 -0
- package/scripts/lib/mock_server_fingerprint.mjs +94 -0
- package/scripts/release_screen_claims.json +31 -2
- package/src-tauri/Cargo.lock +361 -3
- package/src-tauri/Cargo.toml +6 -1
- package/src-tauri/src/backend.rs +349 -0
- package/src-tauri/src/folder.rs +33 -0
- package/src-tauri/src/main.rs +97 -399
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +41 -37
- package/static/app/assets/Act-yYpYnn0v.js +1 -0
- package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
- package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
- package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
- package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
- package/static/app/assets/Capture-CFIRsFNE.js +1 -0
- package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
- package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
- package/static/app/assets/Library-DwO3yZST.js +1 -0
- package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
- package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
- package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
- package/static/app/assets/System-DW8F-2xL.js +1 -0
- package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
- package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
- package/static/app/assets/brain-Ci1CkWjM.js +1 -0
- package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
- package/static/app/assets/circle-check-DfInj-qD.js +1 -0
- package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
- package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
- package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
- package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
- package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
- package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
- package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
- package/static/app/assets/index-_u5iUHDr.js +10 -0
- package/static/app/assets/input-B0lPdRQZ.js +1 -0
- package/static/app/assets/link-2-CoFbooHS.js +1 -0
- package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
- package/static/app/assets/primitives-DEbN-d6p.js +1 -0
- package/static/app/assets/search-BybIWPNd.js +1 -0
- package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
- package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
- package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
- package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
- package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
- package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
- package/static/app/assets/utils-BlZr7Pd4.js +4 -0
- package/static/app/assets/workspace-jJY4RuAV.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/graph/_kg_common.py +0 -1331
- package/lattice_brain/graph/discovery_index.py +0 -1141
- package/lattice_brain/graph/retrieval.py +0 -1120
- package/lattice_brain/graph/retrieval_vector.py +0 -1293
- package/lattice_brain/ingestion.py +0 -1525
- package/lattice_brain/multimodal.py +0 -1258
- package/latticeai/core/agent.py +0 -1465
- package/latticeai/core/embedding_providers.py +0 -1196
- package/latticeai/core/file_generation.py +0 -1047
- package/latticeai/integrations/telegram_bot.py +0 -1390
- package/latticeai/models/router.py +0 -1007
- package/latticeai/runtime/build_phases.py +0 -1450
- package/latticeai/services/brain_intelligence.py +0 -1083
- package/latticeai/services/memory_service.py +0 -1177
- package/latticeai/services/model_runtime.py +0 -1281
- package/latticeai/setup/wizard.py +0 -1310
- package/static/app/assets/Act-AWf0SAKp.js +0 -1
- package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
- package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
- package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
- package/static/app/assets/Capture-CqOSzyPr.js +0 -1
- package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
- package/static/app/assets/Library-CX-bbhmK.js +0 -1
- package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
- package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
- package/static/app/assets/System-Bu2t5hn1.js +0 -1
- package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
- package/static/app/assets/brain-DJMoqrwx.js +0 -1
- package/static/app/assets/index-BpYkzcVm.js +0 -10
- package/static/app/assets/input-DSlJJxRs.js +0 -1
- package/static/app/assets/primitives-BCx6TvfG.js +0 -1
- package/static/app/assets/search-Cgy8cCFJ.js +0 -1
- package/static/app/assets/utils-zqPZJxdx.js +0 -4
- package/static/app/assets/workspace-DXTihhfU.js +0 -1
|
@@ -0,0 +1,374 @@
|
|
|
1
|
+
"""What the vector index currently knows, and what it still owes.
|
|
2
|
+
|
|
3
|
+
``index_status`` / ``vector_freshness`` / ``vector_queue`` — the honesty
|
|
4
|
+
surface that reports coverage, staleness, and pending work rather than
|
|
5
|
+
letting a half-built index look complete. Moved verbatim out of
|
|
6
|
+
``retrieval_vector.py`` (v11.3.0 decomposition).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from typing import TYPE_CHECKING
|
|
12
|
+
|
|
13
|
+
# ruff: noqa: F403,F405
|
|
14
|
+
from .._kg_common import * # noqa: F403,F401
|
|
15
|
+
from ..vector_index import VectorEmbedQueue
|
|
16
|
+
|
|
17
|
+
# Typing-only base (runtime value is `object`, so the store's MRO is
|
|
18
|
+
# unchanged). Reporting on the index means asking the halves that build and
|
|
19
|
+
# query it — source items and incremental indexing from .indexing, the
|
|
20
|
+
# fingerprint it reaches through, and the backend selection from .search. They
|
|
21
|
+
# are named as bases rather than re-declared, so the signatures cannot drift.
|
|
22
|
+
if TYPE_CHECKING:
|
|
23
|
+
from .indexing import _VectorIndexingMixin
|
|
24
|
+
from .search import _VectorSearchMixin
|
|
25
|
+
|
|
26
|
+
class _Core(_VectorIndexingMixin, _VectorSearchMixin):
|
|
27
|
+
"""The sibling halves this one reaches through ``self``."""
|
|
28
|
+
else:
|
|
29
|
+
_Core = object
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class _VectorStatusMixin(_Core):
|
|
33
|
+
"""Vector index status/freshness. Composed into the public mixin."""
|
|
34
|
+
|
|
35
|
+
def index_status(self) -> Dict[str, Any]:
|
|
36
|
+
storage_capabilities = None
|
|
37
|
+
try:
|
|
38
|
+
storage_capabilities = self.storage_engine.capabilities().as_dict()
|
|
39
|
+
except Exception as exc:
|
|
40
|
+
storage_capabilities = {
|
|
41
|
+
"engine": "sqlite",
|
|
42
|
+
"available": False,
|
|
43
|
+
"reason": str(exc),
|
|
44
|
+
}
|
|
45
|
+
with self._connect() as conn:
|
|
46
|
+
vector_counts = {
|
|
47
|
+
row["item_type"]: row["count"]
|
|
48
|
+
for row in conn.execute(
|
|
49
|
+
"SELECT item_type, COUNT(*) AS count FROM vector_embeddings GROUP BY item_type"
|
|
50
|
+
)
|
|
51
|
+
}
|
|
52
|
+
# Materialised on purpose, unlike the rebuild path: status walks
|
|
53
|
+
# this set twice (once for ids, once to classify each item) and
|
|
54
|
+
# reports a count, none of which a one-shot iterator can serve.
|
|
55
|
+
source_items = list(self._iter_vector_source_items(conn))
|
|
56
|
+
vector_rows = {
|
|
57
|
+
row["item_id"]: row
|
|
58
|
+
for row in conn.execute(
|
|
59
|
+
"""
|
|
60
|
+
SELECT item_id, text_hash, embedding_dim, embedding_model, indexed_at
|
|
61
|
+
FROM vector_embeddings
|
|
62
|
+
"""
|
|
63
|
+
).fetchall()
|
|
64
|
+
}
|
|
65
|
+
latest_rows = conn.execute(
|
|
66
|
+
"""
|
|
67
|
+
SELECT id, operation, status, requested_at, started_at, completed_at,
|
|
68
|
+
items_total, items_indexed, items_skipped, error_message, metadata_json
|
|
69
|
+
FROM vector_index_operations
|
|
70
|
+
ORDER BY requested_at DESC, id DESC
|
|
71
|
+
LIMIT 5
|
|
72
|
+
"""
|
|
73
|
+
).fetchall()
|
|
74
|
+
missing = stale = ready = 0
|
|
75
|
+
source_item_ids = {str(item["item_id"]) for item in source_items}
|
|
76
|
+
backlog_by_type: Dict[str, int] = {}
|
|
77
|
+
backlog_reasons: Dict[str, int] = {}
|
|
78
|
+
backlog_samples: List[Dict[str, Any]] = []
|
|
79
|
+
|
|
80
|
+
def add_backlog(item: Dict[str, Any], reason: str) -> None:
|
|
81
|
+
item_type = str(item.get("item_type") or "unknown")
|
|
82
|
+
backlog_by_type[item_type] = backlog_by_type.get(item_type, 0) + 1
|
|
83
|
+
backlog_reasons[reason] = backlog_reasons.get(reason, 0) + 1
|
|
84
|
+
if len(backlog_samples) >= 20:
|
|
85
|
+
return
|
|
86
|
+
backlog_samples.append(
|
|
87
|
+
{
|
|
88
|
+
"item_id": item.get("item_id"),
|
|
89
|
+
"item_type": item_type,
|
|
90
|
+
"source_node": item.get("source_node"),
|
|
91
|
+
"reason": reason,
|
|
92
|
+
"metadata": {
|
|
93
|
+
key: value
|
|
94
|
+
for key, value in dict(item.get("metadata") or {}).items()
|
|
95
|
+
if key in {"node_type", "source", "conversation_id", "parent_source_node"}
|
|
96
|
+
},
|
|
97
|
+
}
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
for item in source_items:
|
|
101
|
+
vector_row = vector_rows.get(item["item_id"])
|
|
102
|
+
expected_hash = _sha256_text(_clean_text(item["text"]))
|
|
103
|
+
if not vector_row:
|
|
104
|
+
missing += 1
|
|
105
|
+
add_backlog(item, "missing_vector")
|
|
106
|
+
elif (
|
|
107
|
+
vector_row["text_hash"] != expected_hash
|
|
108
|
+
or vector_row["embedding_dim"] != self._embedding_model.dim
|
|
109
|
+
or vector_row["embedding_model"] != self._embedding_model.model_id
|
|
110
|
+
):
|
|
111
|
+
stale += 1
|
|
112
|
+
reason = "text_changed"
|
|
113
|
+
if vector_row["embedding_model"] != self._embedding_model.model_id:
|
|
114
|
+
reason = "model_changed"
|
|
115
|
+
elif vector_row["embedding_dim"] != self._embedding_model.dim:
|
|
116
|
+
reason = "dimension_changed"
|
|
117
|
+
add_backlog(item, reason)
|
|
118
|
+
else:
|
|
119
|
+
ready += 1
|
|
120
|
+
pending = missing + stale
|
|
121
|
+
orphaned_items = max(0, len(set(vector_rows) - source_item_ids))
|
|
122
|
+
coverage_ratio = round(ready / len(source_items), 6) if source_items else 1.0
|
|
123
|
+
latest_completed = None
|
|
124
|
+
for row in latest_rows:
|
|
125
|
+
if row["status"] == "completed":
|
|
126
|
+
latest_completed = row
|
|
127
|
+
break
|
|
128
|
+
latency_budget: Dict[str, Any] = {
|
|
129
|
+
"target_rebuild_ms": 10_000,
|
|
130
|
+
"last_rebuild_duration_ms": None,
|
|
131
|
+
"last_items_per_second": None,
|
|
132
|
+
"within_target": None,
|
|
133
|
+
}
|
|
134
|
+
if latest_completed is not None:
|
|
135
|
+
metadata = _safe_loads(latest_completed["metadata_json"])
|
|
136
|
+
duration_ms = metadata.get("duration_ms")
|
|
137
|
+
items_total = int(latest_completed["items_total"] or 0)
|
|
138
|
+
if isinstance(duration_ms, (int, float)) and duration_ms > 0:
|
|
139
|
+
latency_budget.update(
|
|
140
|
+
{
|
|
141
|
+
"last_rebuild_duration_ms": round(float(duration_ms), 2),
|
|
142
|
+
"last_items_per_second": round(items_total / (float(duration_ms) / 1000.0), 2),
|
|
143
|
+
"within_target": float(duration_ms) <= 10_000,
|
|
144
|
+
}
|
|
145
|
+
)
|
|
146
|
+
embedder_status = self.embedder_fingerprint_status()
|
|
147
|
+
return {
|
|
148
|
+
"status": "ready" if pending == 0 else "needs_reindex",
|
|
149
|
+
"embedder": embedder_status,
|
|
150
|
+
"storage": {
|
|
151
|
+
"db_path": str(self.db_path),
|
|
152
|
+
"backend": "sqlite",
|
|
153
|
+
"embedding_model": self._embedding_model.model_id,
|
|
154
|
+
"embedding_dim": self._embedding_model.dim,
|
|
155
|
+
# Honest capability report: trigram FTS5 keyword index, or
|
|
156
|
+
# LIKE-scan fallback when this SQLite build lacks it.
|
|
157
|
+
"fts_enabled": bool(getattr(self, "_fts_enabled", False)),
|
|
158
|
+
"engine": storage_capabilities,
|
|
159
|
+
"vector_search_backend": (
|
|
160
|
+
storage_capabilities.get("vector_backend")
|
|
161
|
+
if isinstance(storage_capabilities, dict)
|
|
162
|
+
else "bruteforce-cosine"
|
|
163
|
+
),
|
|
164
|
+
"vector_search_mode": (
|
|
165
|
+
(storage_capabilities.get("metadata") or {}).get("vector_mode")
|
|
166
|
+
if isinstance(storage_capabilities, dict)
|
|
167
|
+
else "fallback"
|
|
168
|
+
),
|
|
169
|
+
"sqlite_vec_ann_available": (
|
|
170
|
+
bool((storage_capabilities.get("metadata") or {}).get("sqlite_vec_ann_available"))
|
|
171
|
+
if isinstance(storage_capabilities, dict)
|
|
172
|
+
else False
|
|
173
|
+
),
|
|
174
|
+
# v11.1.0: which in-process index scores a search, and — when
|
|
175
|
+
# the configured one could not be used — the reason it was
|
|
176
|
+
# substituted, so an unavailable optional extra is visible
|
|
177
|
+
# here instead of only showing up as "search feels slow".
|
|
178
|
+
"vector_index": self._vector_index_selection().as_dict(),
|
|
179
|
+
},
|
|
180
|
+
"source_items": len(source_items),
|
|
181
|
+
"indexed_items": sum(vector_counts.values()),
|
|
182
|
+
"ready_items": ready,
|
|
183
|
+
"missing_items": missing,
|
|
184
|
+
"stale_items": stale,
|
|
185
|
+
"pending_items": pending,
|
|
186
|
+
"by_item_type": vector_counts,
|
|
187
|
+
"scale": {
|
|
188
|
+
"version": 1,
|
|
189
|
+
"coverage_ratio": coverage_ratio,
|
|
190
|
+
"coverage_percent": round(coverage_ratio * 100.0, 2),
|
|
191
|
+
"source_items": len(source_items),
|
|
192
|
+
"ready_items": ready,
|
|
193
|
+
"pending_items": pending,
|
|
194
|
+
"missing_items": missing,
|
|
195
|
+
"stale_items": stale,
|
|
196
|
+
"orphaned_items": orphaned_items,
|
|
197
|
+
"backlog_by_item_type": backlog_by_type,
|
|
198
|
+
"backlog_reasons": backlog_reasons,
|
|
199
|
+
"backlog_samples": backlog_samples,
|
|
200
|
+
"incremental_reindex_recommended": pending > 0,
|
|
201
|
+
# A stale embedder means every old-model row must be re-embedded;
|
|
202
|
+
# only a full rebuild (which re-records the fingerprint) heals it.
|
|
203
|
+
"full_rebuild_recommended": bool(
|
|
204
|
+
orphaned_items > 0 or embedder_status["stale_embedder"]
|
|
205
|
+
),
|
|
206
|
+
"latency_budget": latency_budget,
|
|
207
|
+
},
|
|
208
|
+
"operations": [
|
|
209
|
+
{
|
|
210
|
+
"id": row["id"],
|
|
211
|
+
"operation": row["operation"],
|
|
212
|
+
"status": row["status"],
|
|
213
|
+
"requested_at": row["requested_at"],
|
|
214
|
+
"started_at": row["started_at"],
|
|
215
|
+
"completed_at": row["completed_at"],
|
|
216
|
+
"items_total": row["items_total"],
|
|
217
|
+
"items_indexed": row["items_indexed"],
|
|
218
|
+
"items_skipped": row["items_skipped"],
|
|
219
|
+
"error_message": row["error_message"],
|
|
220
|
+
"metadata": _safe_loads(row["metadata_json"]),
|
|
221
|
+
}
|
|
222
|
+
for row in latest_rows
|
|
223
|
+
],
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
def vector_freshness(self) -> Dict[str, Any]:
|
|
227
|
+
"""Compact vector-index freshness summary for API surfaces (v9.8.0).
|
|
228
|
+
|
|
229
|
+
Reduces :meth:`index_status` (``pending = missing + stale``) to the
|
|
230
|
+
fixed contract ``{"status", "pending_items", "total_items", "detail"}``
|
|
231
|
+
with ``status`` in ``ready`` / ``pending`` / ``stale_embedder`` /
|
|
232
|
+
``unavailable``. ``stale_embedder`` (review Wave 2.2) is reported only
|
|
233
|
+
when the recorded embedder fingerprint differs from the current
|
|
234
|
+
embedder AND rows indexed under the old model still exist — the index
|
|
235
|
+
needs a full rebuild, not an incremental sync.
|
|
236
|
+
|
|
237
|
+
Never raises: environments where the embedding provider or index
|
|
238
|
+
storage cannot be used report ``"unavailable"`` with the cause in
|
|
239
|
+
``detail`` instead of surfacing an exception to the API layer.
|
|
240
|
+
"""
|
|
241
|
+
try:
|
|
242
|
+
status = self.index_status()
|
|
243
|
+
except Exception as exc: # noqa: BLE001 — freshness must degrade, not fail
|
|
244
|
+
return {
|
|
245
|
+
"status": "unavailable",
|
|
246
|
+
"pending_items": 0,
|
|
247
|
+
"total_items": 0,
|
|
248
|
+
"detail": f"vector index status unavailable: {exc}",
|
|
249
|
+
}
|
|
250
|
+
return self._vector_freshness_summary(status)
|
|
251
|
+
|
|
252
|
+
def _vector_freshness_summary(self, status: Dict[str, Any]) -> Dict[str, Any]:
|
|
253
|
+
"""The freshness reduction of an already-read :meth:`index_status`.
|
|
254
|
+
|
|
255
|
+
Split out so :meth:`vector_freshness_breakdown` can report both shapes
|
|
256
|
+
from one index scan; ``index_status`` walks every source item, and
|
|
257
|
+
calling it twice to answer one question about freshness would double
|
|
258
|
+
the most expensive read in this module.
|
|
259
|
+
"""
|
|
260
|
+
pending = int(status.get("pending_items") or 0)
|
|
261
|
+
total = int(status.get("source_items") or 0)
|
|
262
|
+
embedder = status.get("embedder") or {}
|
|
263
|
+
if embedder.get("stale_embedder"):
|
|
264
|
+
old_model_rows = 0
|
|
265
|
+
try:
|
|
266
|
+
with self._connect() as conn:
|
|
267
|
+
old_model_rows = int(
|
|
268
|
+
conn.execute(
|
|
269
|
+
"SELECT COUNT(*) AS c FROM vector_embeddings "
|
|
270
|
+
"WHERE embedding_model<>? OR embedding_dim<>?",
|
|
271
|
+
(
|
|
272
|
+
self._embedding_model.model_id,
|
|
273
|
+
int(self._embedding_model.dim),
|
|
274
|
+
),
|
|
275
|
+
).fetchone()["c"]
|
|
276
|
+
)
|
|
277
|
+
except Exception: # noqa: BLE001 — keep the existing statuses on failure
|
|
278
|
+
old_model_rows = 0
|
|
279
|
+
if old_model_rows > 0:
|
|
280
|
+
recorded = embedder.get("recorded") or {}
|
|
281
|
+
return {
|
|
282
|
+
"status": "stale_embedder",
|
|
283
|
+
"pending_items": pending,
|
|
284
|
+
"total_items": total,
|
|
285
|
+
"detail": (
|
|
286
|
+
f"embedding model changed ({recorded.get('model_id')} → "
|
|
287
|
+
f"{self._embedding_model.model_id}); {old_model_rows} indexed "
|
|
288
|
+
"rows still use the previous model — run a full vector index rebuild"
|
|
289
|
+
),
|
|
290
|
+
}
|
|
291
|
+
if pending > 0:
|
|
292
|
+
return {
|
|
293
|
+
"status": "pending",
|
|
294
|
+
"pending_items": pending,
|
|
295
|
+
"total_items": total,
|
|
296
|
+
"detail": (
|
|
297
|
+
f"{pending} of {total} items are missing or stale in the vector index"
|
|
298
|
+
),
|
|
299
|
+
}
|
|
300
|
+
detail = (
|
|
301
|
+
"vector index is up to date"
|
|
302
|
+
if total
|
|
303
|
+
else "vector index is empty (no indexable items yet)"
|
|
304
|
+
)
|
|
305
|
+
return {
|
|
306
|
+
"status": "ready",
|
|
307
|
+
"pending_items": 0,
|
|
308
|
+
"total_items": total,
|
|
309
|
+
"detail": detail,
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
@property
|
|
313
|
+
def vector_queue(self) -> VectorEmbedQueue:
|
|
314
|
+
"""This store's durable pending-embed backlog (created on demand).
|
|
315
|
+
|
|
316
|
+
Built lazily rather than in ``__init__`` so opening a graph never
|
|
317
|
+
creates a table nobody asked for, and hung off the store so the
|
|
318
|
+
ingestion pipeline and the freshness report share one backlog instead
|
|
319
|
+
of each keeping a private view of it.
|
|
320
|
+
"""
|
|
321
|
+
queue = getattr(self, "_vector_queue", None)
|
|
322
|
+
if queue is None:
|
|
323
|
+
queue = VectorEmbedQueue(
|
|
324
|
+
db_path=self.db_path, indexer=self.index_node_incremental
|
|
325
|
+
)
|
|
326
|
+
self._vector_queue = queue
|
|
327
|
+
return queue
|
|
328
|
+
|
|
329
|
+
def vector_freshness_breakdown(self) -> Dict[str, Any]:
|
|
330
|
+
"""The four numbers behind :meth:`vector_freshness` (v11.1.0).
|
|
331
|
+
|
|
332
|
+
``vector_freshness()`` answers one question — *is the index behind?* —
|
|
333
|
+
and its four keys are a frozen wire contract that surfaces already
|
|
334
|
+
read, so this is a sibling rather than an extension of it. The split
|
|
335
|
+
matters because "12 pending" hides two different situations: twelve
|
|
336
|
+
items never embedded (a new import) and twelve items whose text
|
|
337
|
+
changed under an existing embedding (edits). Only the second means
|
|
338
|
+
current answers are quietly wrong.
|
|
339
|
+
|
|
340
|
+
``queued`` counts the durable background backlog
|
|
341
|
+
(:class:`~lattice_brain.graph.vector_index.VectorEmbedQueue`), and is
|
|
342
|
+
``None`` when that queue has no database to persist to — never ``0``,
|
|
343
|
+
which would claim an empty backlog nobody measured.
|
|
344
|
+
|
|
345
|
+
Never raises: an unreadable index reports ``status="unavailable"``
|
|
346
|
+
with the cause in ``detail`` and zeroed counts.
|
|
347
|
+
"""
|
|
348
|
+
status: Dict[str, Any] = {}
|
|
349
|
+
summary: Dict[str, Any]
|
|
350
|
+
try:
|
|
351
|
+
status = self.index_status()
|
|
352
|
+
except Exception as exc: # noqa: BLE001 — freshness must degrade, not fail
|
|
353
|
+
summary = {
|
|
354
|
+
"status": "unavailable",
|
|
355
|
+
"pending_items": 0,
|
|
356
|
+
"total_items": 0,
|
|
357
|
+
"detail": f"vector index status unavailable: {exc}",
|
|
358
|
+
}
|
|
359
|
+
else:
|
|
360
|
+
summary = self._vector_freshness_summary(status)
|
|
361
|
+
breakdown: Dict[str, Any] = {
|
|
362
|
+
"status": summary["status"],
|
|
363
|
+
"detail": summary["detail"],
|
|
364
|
+
"embedded": int(status.get("ready_items") or 0),
|
|
365
|
+
"pending": int(summary["pending_items"]),
|
|
366
|
+
"missing": int(status.get("missing_items") or 0),
|
|
367
|
+
"stale": int(status.get("stale_items") or 0),
|
|
368
|
+
"total": int(summary["total_items"]),
|
|
369
|
+
"queued": None,
|
|
370
|
+
}
|
|
371
|
+
queue = self.vector_queue
|
|
372
|
+
if queue.available:
|
|
373
|
+
breakdown["queued"] = int(queue.pending_count())
|
|
374
|
+
return breakdown
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
"""Unified ingestion pipeline — the single write-side seam into the Knowledge Graph.
|
|
2
|
+
|
|
3
|
+
v3.6.0 Knowledge Graph First principle: *no data source bypasses the Knowledge
|
|
4
|
+
Graph and no source creates an isolated silo*. Every source — local files,
|
|
5
|
+
connected folders, PDFs/Markdown/text/code, web URLs, browser tabs — is
|
|
6
|
+
normalized into one :class:`IngestionItem` and pushed through one
|
|
7
|
+
:meth:`IngestionPipeline.ingest` entrypoint:
|
|
8
|
+
|
|
9
|
+
Source → normalize → content hash → (file | text) ingest → provenance
|
|
10
|
+
|
|
11
|
+
The pipeline is deliberately thin. It owns normalization, idempotency reporting,
|
|
12
|
+
provenance capture, and — crucially — routing every ingest through the shared
|
|
13
|
+
``dispatch_tool`` lifecycle so ``pre_tool``/``post_tool`` hooks fire on data
|
|
14
|
+
ingestion exactly as they do on tool calls. The heavy graph construction lives in
|
|
15
|
+
:class:`knowledge_graph.KnowledgeGraphStore` (``ingest_document`` for files,
|
|
16
|
+
``ingest_source`` for text/web), which this module composes rather than
|
|
17
|
+
re-implements.
|
|
18
|
+
|
|
19
|
+
Web ingestion seam
|
|
20
|
+
------------------
|
|
21
|
+
The graph layer never fetches or parses the web. Fetching, rendering,
|
|
22
|
+
readability extraction, and parse quality are the responsibility of the
|
|
23
|
+
*upstream* capture surfaces (browser extension, tools layer, MCP servers):
|
|
24
|
+
they hand this module already-extracted text. :meth:`IngestionPipeline.
|
|
25
|
+
ingest_web_page` is the convenience wrapper for that hand-off — it normalizes
|
|
26
|
+
``(url, extracted_text)`` into an ``IngestionItem(source_type="web_url")`` and
|
|
27
|
+
routes it through the exact same :meth:`IngestionPipeline.ingest` door as every
|
|
28
|
+
other source. If the extracted text is bad, fix the extractor upstream; the
|
|
29
|
+
pipeline will not attempt network access or HTML parsing.
|
|
30
|
+
|
|
31
|
+
Folder ingestion (:meth:`IngestionPipeline.ingest_folder`) walks a local
|
|
32
|
+
directory, honors a gitignore-like ``.latticeignore`` file at the root
|
|
33
|
+
(blank lines, ``#`` comments, ``fnmatch`` glob patterns, ``dir/`` suffix for
|
|
34
|
+
directories), always skips common noise (``.git``, ``node_modules``,
|
|
35
|
+
``__pycache__``, virtualenvs, ``dist``, hidden entries by default), applies
|
|
36
|
+
size/extension filters, and either ingests inline or schedules through the
|
|
37
|
+
existing :class:`BackgroundIngestionQueue`.
|
|
38
|
+
|
|
39
|
+
Split into cohesive submodules in v11.3.0 (no behaviour change): ``constants``
|
|
40
|
+
(routing tables + gates), ``quality`` (advisory scoring), ``models`` (item and
|
|
41
|
+
result), ``hashing``, ``folder_scan`` (``.latticeignore``), and the three mixins
|
|
42
|
+
``routing`` / ``folders`` / ``jobs_api`` that ``pipeline`` composes into
|
|
43
|
+
``IngestionPipeline``. This module re-exports every name the single file
|
|
44
|
+
exposed, so ``lattice_brain.ingestion.X`` keeps working.
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
from __future__ import annotations
|
|
48
|
+
|
|
49
|
+
# The single file had no ``__all__``, so its public surface was "every module
|
|
50
|
+
# global" — including the names it imported for its own use. Every re-export
|
|
51
|
+
# below therefore uses the redundant-alias form: it reproduces exactly that
|
|
52
|
+
# surface, and it marks each name as deliberate rather than a leftover import.
|
|
53
|
+
#
|
|
54
|
+
# Stubbing note: rebinding one of these *here* changes only this module's name.
|
|
55
|
+
# The submodule that calls it holds its own reference, so a test standing in for
|
|
56
|
+
# a helper patches the submodule that uses it.
|
|
57
|
+
from ..gates import FeatureGate as FeatureGate
|
|
58
|
+
from ..ingestion_jobs import JOB_ERRORS_CAP as JOB_ERRORS_CAP
|
|
59
|
+
from ..ingestion_jobs import BackgroundIngestionJob as BackgroundIngestionJob
|
|
60
|
+
from ..ingestion_jobs import BackgroundIngestionQueue as BackgroundIngestionQueue
|
|
61
|
+
from ..multimodal import AUDIO_EXTENSIONS as AUDIO_EXTENSIONS
|
|
62
|
+
from ..multimodal import DEFAULT_KEYFRAMES as DEFAULT_KEYFRAMES
|
|
63
|
+
from ..multimodal import IMAGE_EXTENSIONS as IMAGE_EXTENSIONS
|
|
64
|
+
from ..multimodal import MODALITY_AUDIO as MODALITY_AUDIO
|
|
65
|
+
from ..multimodal import MODALITY_IMAGE as MODALITY_IMAGE
|
|
66
|
+
from ..multimodal import MODALITY_VIDEO as MODALITY_VIDEO
|
|
67
|
+
from ..multimodal import VIDEO_EXTENSIONS as VIDEO_EXTENSIONS
|
|
68
|
+
from ..multimodal import VIDEO_UNAVAILABLE_DETAIL as VIDEO_UNAVAILABLE_DETAIL
|
|
69
|
+
from ..multimodal import ImageFacts as ImageFacts
|
|
70
|
+
from ..multimodal import MultimodalPorts as MultimodalPorts
|
|
71
|
+
from ..multimodal import audio_quality_score as audio_quality_score
|
|
72
|
+
from ..multimodal import detect_modality as detect_modality
|
|
73
|
+
from ..multimodal import extract_image_facts as extract_image_facts
|
|
74
|
+
from ..multimodal import ffmpeg_available as ffmpeg_available
|
|
75
|
+
from ..multimodal import image_quality_score as image_quality_score
|
|
76
|
+
from ..multimodal import read_video_facts as read_video_facts
|
|
77
|
+
from ..multimodal import transcribe_audio as transcribe_audio
|
|
78
|
+
from ..multimodal import video_frame_dir as video_frame_dir
|
|
79
|
+
from ..multimodal import video_quality_score as video_quality_score
|
|
80
|
+
from ..multimodal import write_image_memory as write_image_memory
|
|
81
|
+
from ..multimodal import write_video_memory as write_video_memory
|
|
82
|
+
from ..quiet import quiet as quiet
|
|
83
|
+
from ..runtime.hooks import dispatch_tool as dispatch_tool
|
|
84
|
+
from ..utils import utc_now_iso as utc_now_iso
|
|
85
|
+
from .constants import _MEMORY_NODE_TYPES as _MEMORY_NODE_TYPES
|
|
86
|
+
from .constants import ALLOW_MULTIMODAL_ENV as ALLOW_MULTIMODAL_ENV
|
|
87
|
+
from .constants import ALLOW_VIDEO_ENV as ALLOW_VIDEO_ENV
|
|
88
|
+
from .constants import AUDIO_NODE_TYPE as AUDIO_NODE_TYPE
|
|
89
|
+
from .constants import AUDIO_SOURCE_TYPES as AUDIO_SOURCE_TYPES
|
|
90
|
+
from .constants import AUTO_VECTOR_INDEX_ENV as AUTO_VECTOR_INDEX_ENV
|
|
91
|
+
from .constants import AUTO_VECTOR_INDEX_GATE as AUTO_VECTOR_INDEX_GATE
|
|
92
|
+
from .constants import CHAT_SOURCE_TYPES as CHAT_SOURCE_TYPES
|
|
93
|
+
from .constants import DEFAULT_FOLDER_EXTENSIONS as DEFAULT_FOLDER_EXTENSIONS
|
|
94
|
+
from .constants import DEFAULT_MAX_FILE_BYTES as DEFAULT_MAX_FILE_BYTES
|
|
95
|
+
from .constants import DEFAULT_MAX_TEXT_BYTES as DEFAULT_MAX_TEXT_BYTES
|
|
96
|
+
from .constants import FILE_SOURCE_TYPES as FILE_SOURCE_TYPES
|
|
97
|
+
from .constants import FOLDER_CODE_EXTENSIONS as FOLDER_CODE_EXTENSIONS
|
|
98
|
+
from .constants import FOLDER_DEFAULT_SKIP_DIRS as FOLDER_DEFAULT_SKIP_DIRS
|
|
99
|
+
from .constants import FOLDER_DOCUMENT_EXTENSIONS as FOLDER_DOCUMENT_EXTENSIONS
|
|
100
|
+
from .constants import FOLDER_MULTIMODAL_EXTENSIONS as FOLDER_MULTIMODAL_EXTENSIONS
|
|
101
|
+
from .constants import FOLDER_TEXT_EXTENSIONS as FOLDER_TEXT_EXTENSIONS
|
|
102
|
+
from .constants import FOLDER_VIDEO_EXTENSIONS as FOLDER_VIDEO_EXTENSIONS
|
|
103
|
+
from .constants import IMAGE_SOURCE_TYPES as IMAGE_SOURCE_TYPES
|
|
104
|
+
from .constants import LATTICEIGNORE_FILENAME as LATTICEIGNORE_FILENAME
|
|
105
|
+
from .constants import MEMORY_SOURCE_TYPES as MEMORY_SOURCE_TYPES
|
|
106
|
+
from .constants import MULTIMODAL_GATE as MULTIMODAL_GATE
|
|
107
|
+
from .constants import TEXT_SOURCE_TYPES as TEXT_SOURCE_TYPES
|
|
108
|
+
from .constants import VIDEO_GATE as VIDEO_GATE
|
|
109
|
+
from .constants import VIDEO_SOURCE_TYPES as VIDEO_SOURCE_TYPES
|
|
110
|
+
from .folder_scan import _load_latticeignore as _load_latticeignore
|
|
111
|
+
from .folder_scan import _matches_ignore as _matches_ignore
|
|
112
|
+
from .folders import IngestionFolderMixin as IngestionFolderMixin
|
|
113
|
+
from .hashing import _file_digest as _file_digest
|
|
114
|
+
from .hashing import content_hash_text as content_hash_text
|
|
115
|
+
from .jobs_api import IngestionJobsMixin as IngestionJobsMixin
|
|
116
|
+
from .models import IngestionItem as IngestionItem
|
|
117
|
+
from .models import IngestionResult as IngestionResult
|
|
118
|
+
from .pipeline import VECTOR_TICK_LIMIT as VECTOR_TICK_LIMIT
|
|
119
|
+
from .pipeline import IngestionPipeline as IngestionPipeline
|
|
120
|
+
from .quality import _BOILERPLATE_LINE_MARKERS as _BOILERPLATE_LINE_MARKERS
|
|
121
|
+
from .quality import _CAPTURE_REASON_LABELS as _CAPTURE_REASON_LABELS
|
|
122
|
+
from .quality import _WEB_SOURCE_TYPES as _WEB_SOURCE_TYPES
|
|
123
|
+
from .quality import CAPTURE_SUGGESTIONS_THIN as CAPTURE_SUGGESTIONS_THIN
|
|
124
|
+
from .quality import QUALITY_HIGH_THRESHOLD as QUALITY_HIGH_THRESHOLD
|
|
125
|
+
from .quality import QUALITY_LOW_THRESHOLD as QUALITY_LOW_THRESHOLD
|
|
126
|
+
from .quality import QUALITY_LOW_WARNING as QUALITY_LOW_WARNING
|
|
127
|
+
from .quality import _quality_level as _quality_level
|
|
128
|
+
from .quality import assess_extraction_quality as assess_extraction_quality
|
|
129
|
+
from .quality import capture_quality_verdict as capture_quality_verdict
|
|
130
|
+
from .routing import IngestionRoutingMixin as IngestionRoutingMixin
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
"""The seam the three ingestion mixins share.
|
|
2
|
+
|
|
3
|
+
``IngestionPipeline`` is assembled from the routing, folder and jobs mixins.
|
|
4
|
+
Each of them reads constructor state it does not own and calls back into
|
|
5
|
+
``ingest`` — the point of the split is that "how one picture is stored" and
|
|
6
|
+
"how a folder is walked" stop sharing a thousand-line file, not that they stop
|
|
7
|
+
sharing ``self``.
|
|
8
|
+
|
|
9
|
+
Typing-only, exactly like :mod:`lattice_brain.graph._kg_contract`: at runtime
|
|
10
|
+
the mixins alias it to ``object``, so the MRO and every method resolution stay
|
|
11
|
+
byte-for-byte what the single-file class had. Adding a cross-mixin call without
|
|
12
|
+
declaring it here is a type error.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Any, Dict, List, Optional
|
|
19
|
+
|
|
20
|
+
from ..ingestion_jobs import BackgroundIngestionJob, BackgroundIngestionQueue
|
|
21
|
+
from ..multimodal import MultimodalPorts
|
|
22
|
+
from .models import IngestionItem, IngestionResult
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class IngestionCore:
|
|
26
|
+
"""What any ingestion mixin may assume about ``self``.
|
|
27
|
+
|
|
28
|
+
Never instantiated and never inherited at runtime — see the module
|
|
29
|
+
docstring. Members are declared, not implemented: the implementation lives
|
|
30
|
+
in ``pipeline.py`` or in whichever mixin owns it.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
# ── State owned by IngestionPipeline.__init__ ────────────────────────────
|
|
34
|
+
_kg: Any
|
|
35
|
+
_hooks: Any
|
|
36
|
+
_enable: bool
|
|
37
|
+
_audit: Optional[Any]
|
|
38
|
+
_max_text_bytes: int
|
|
39
|
+
_pipeline_name: str
|
|
40
|
+
_bg_queue: BackgroundIngestionQueue
|
|
41
|
+
_auto_vector_index_opt_in: bool
|
|
42
|
+
_multimodal_opt_in: bool
|
|
43
|
+
_multimodal: MultimodalPorts
|
|
44
|
+
_keyframes: int
|
|
45
|
+
|
|
46
|
+
# ── pipeline.py: the gates, resolved when they are asked ─────────────────
|
|
47
|
+
@property
|
|
48
|
+
def _allow_multimodal(self) -> bool:
|
|
49
|
+
raise NotImplementedError
|
|
50
|
+
|
|
51
|
+
@property
|
|
52
|
+
def _allow_video(self) -> bool:
|
|
53
|
+
raise NotImplementedError
|
|
54
|
+
|
|
55
|
+
# ── pipeline.py: the one door every mixin routes back through ────────────
|
|
56
|
+
def available(self) -> bool:
|
|
57
|
+
raise NotImplementedError
|
|
58
|
+
|
|
59
|
+
def ingest(
|
|
60
|
+
self, item: IngestionItem, *, user_email: Optional[str] = None
|
|
61
|
+
) -> IngestionResult:
|
|
62
|
+
raise NotImplementedError
|
|
63
|
+
|
|
64
|
+
# ── routing.py: reached from the folder walk and the modality doors ──────
|
|
65
|
+
def _resolve_file_path(self, item: IngestionItem) -> Path:
|
|
66
|
+
raise NotImplementedError
|
|
67
|
+
|
|
68
|
+
# ── folders.py: the folder-scan allow-list the multimodal gates widen ────
|
|
69
|
+
def _folder_extensions(self) -> frozenset:
|
|
70
|
+
raise NotImplementedError
|
|
71
|
+
|
|
72
|
+
# ── jobs_api.py: scheduling, reached from the folder walk ────────────────
|
|
73
|
+
def schedule_background(
|
|
74
|
+
self,
|
|
75
|
+
items: List[IngestionItem],
|
|
76
|
+
*,
|
|
77
|
+
incremental: bool = True,
|
|
78
|
+
user_email: Optional[str] = None,
|
|
79
|
+
) -> BackgroundIngestionJob:
|
|
80
|
+
raise NotImplementedError
|
|
81
|
+
|
|
82
|
+
def run_background_job(
|
|
83
|
+
self, job_id: str, *, user_email: Optional[str] = None
|
|
84
|
+
) -> Dict[str, Any]:
|
|
85
|
+
raise NotImplementedError
|
|
86
|
+
|
|
87
|
+
def _execute_background_job(
|
|
88
|
+
self, job: BackgroundIngestionJob, *, user_email: Optional[str] = None
|
|
89
|
+
) -> Dict[str, Any]:
|
|
90
|
+
raise NotImplementedError
|