ltcai 11.2.0 → 11.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -53
- package/docs/CHANGELOG.md +61 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/MULTI_AGENT_RUNTIME.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +6 -2
- package/docs/PERMISSION_MODE.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +2 -2
- package/docs/v11.3.0_PLAN.md +202 -0
- package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +287 -0
- package/lattice_brain/graph/_kg_common/extraction.py +516 -0
- package/lattice_brain/graph/_kg_common/relations.py +161 -0
- package/lattice_brain/graph/_kg_common/text.py +479 -0
- package/lattice_brain/graph/discovery_index/__init__.py +35 -0
- package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
- package/lattice_brain/graph/discovery_index/extract.py +137 -0
- package/lattice_brain/graph/discovery_index/scan.py +411 -0
- package/lattice_brain/graph/discovery_index/upsert.py +495 -0
- package/lattice_brain/graph/projection/__init__.py +42 -0
- package/lattice_brain/graph/projection/curation.py +500 -0
- package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
- package/lattice_brain/graph/retrieval/__init__.py +54 -0
- package/lattice_brain/graph/retrieval/context.py +197 -0
- package/lattice_brain/graph/retrieval/graph_view.py +319 -0
- package/lattice_brain/graph/retrieval/hybrid.py +488 -0
- package/lattice_brain/graph/retrieval/maintenance.py +121 -0
- package/lattice_brain/graph/retrieval/signals.py +95 -0
- package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
- package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
- package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
- package/lattice_brain/graph/retrieval_vector/search.py +560 -0
- package/lattice_brain/graph/retrieval_vector/status.py +374 -0
- package/lattice_brain/ingestion/__init__.py +130 -0
- package/lattice_brain/ingestion/_contract.py +90 -0
- package/lattice_brain/ingestion/constants.py +127 -0
- package/lattice_brain/ingestion/folder_scan.py +57 -0
- package/lattice_brain/ingestion/folders.py +258 -0
- package/lattice_brain/ingestion/hashing.py +26 -0
- package/lattice_brain/ingestion/jobs_api.py +107 -0
- package/lattice_brain/ingestion/models.py +80 -0
- package/lattice_brain/ingestion/pipeline.py +486 -0
- package/lattice_brain/ingestion/quality.py +209 -0
- package/lattice_brain/ingestion/routing.py +295 -0
- package/lattice_brain/multimodal/__init__.py +164 -0
- package/lattice_brain/multimodal/audio.py +77 -0
- package/lattice_brain/multimodal/common.py +118 -0
- package/lattice_brain/multimodal/images.py +498 -0
- package/lattice_brain/multimodal/ports.py +169 -0
- package/lattice_brain/multimodal/video.py +410 -0
- package/lattice_brain/portability/__init__.py +90 -0
- package/lattice_brain/portability/_contract.py +42 -0
- package/lattice_brain/portability/backups.py +338 -0
- package/lattice_brain/portability/bundles.py +136 -0
- package/lattice_brain/portability/constants.py +93 -0
- package/lattice_brain/portability/fsops.py +138 -0
- package/lattice_brain/portability/service.py +41 -0
- package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
- package/lattice_brain/runtime/__init__.py +1 -1
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/chronicle.py +63 -0
- package/latticeai/core/agent/__init__.py +93 -0
- package/latticeai/core/agent/_contract.py +79 -0
- package/latticeai/core/agent/context.py +57 -0
- package/latticeai/core/agent/deps.py +125 -0
- package/latticeai/core/agent/execution.py +622 -0
- package/latticeai/core/agent/planning.py +145 -0
- package/latticeai/core/agent/recovery.py +157 -0
- package/latticeai/core/agent/runtime.py +210 -0
- package/latticeai/core/agent/verification.py +231 -0
- package/latticeai/core/embedding_providers/__init__.py +151 -0
- package/latticeai/core/embedding_providers/base.py +199 -0
- package/latticeai/core/embedding_providers/captions.py +162 -0
- package/latticeai/core/embedding_providers/profiles.py +126 -0
- package/latticeai/core/embedding_providers/text.py +350 -0
- package/latticeai/core/embedding_providers/vision.py +352 -0
- package/latticeai/core/file_generation/__init__.py +115 -0
- package/latticeai/core/file_generation/bundles.py +76 -0
- package/latticeai/core/file_generation/extraction.py +154 -0
- package/latticeai/core/file_generation/inference.py +235 -0
- package/latticeai/core/file_generation/orchestration.py +152 -0
- package/latticeai/core/file_generation/prompting.py +117 -0
- package/latticeai/core/file_generation/repair.py +114 -0
- package/latticeai/core/file_generation/sanitize.py +61 -0
- package/latticeai/core/file_generation/validation.py +201 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +9 -0
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/integrations/telegram_bot/__init__.py +123 -0
- package/latticeai/integrations/telegram_bot/__main__.py +17 -0
- package/latticeai/integrations/telegram_bot/config.py +86 -0
- package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
- package/latticeai/integrations/telegram_bot/flows.py +478 -0
- package/latticeai/integrations/telegram_bot/helpers.py +322 -0
- package/latticeai/integrations/telegram_bot/screens.py +394 -0
- package/latticeai/models/router/__init__.py +88 -0
- package/latticeai/models/router/_contract.py +66 -0
- package/latticeai/models/router/branding.py +56 -0
- package/latticeai/models/router/catalog.py +69 -0
- package/latticeai/models/router/documents.py +199 -0
- package/latticeai/models/router/errors.py +37 -0
- package/latticeai/models/router/generation.py +258 -0
- package/latticeai/models/router/loading.py +291 -0
- package/latticeai/models/router/local_models.py +85 -0
- package/latticeai/models/router/registry.py +147 -0
- package/latticeai/runtime/build_phases/__init__.py +82 -0
- package/latticeai/runtime/build_phases/features.py +407 -0
- package/latticeai/runtime/build_phases/foundation.py +555 -0
- package/latticeai/runtime/build_phases/web.py +492 -0
- package/latticeai/runtime/runtime_context.py +1 -0
- package/latticeai/services/architecture_readiness.py +48 -19
- package/latticeai/services/brain_intelligence/__init__.py +58 -0
- package/latticeai/services/brain_intelligence/_contract.py +71 -0
- package/latticeai/services/brain_intelligence/consistency.py +193 -0
- package/latticeai/services/brain_intelligence/constants.py +47 -0
- package/latticeai/services/brain_intelligence/digest.py +258 -0
- package/latticeai/services/brain_intelligence/health.py +331 -0
- package/latticeai/services/brain_intelligence/proposals.py +264 -0
- package/latticeai/services/brain_intelligence/sampling.py +84 -0
- package/latticeai/services/brain_intelligence/service.py +48 -0
- package/latticeai/services/chronicle.py +557 -0
- package/latticeai/services/memory_service/__init__.py +52 -0
- package/latticeai/services/memory_service/_contract.py +100 -0
- package/latticeai/services/memory_service/brief.py +431 -0
- package/latticeai/services/memory_service/constants.py +57 -0
- package/latticeai/services/memory_service/maintenance.py +138 -0
- package/latticeai/services/memory_service/manager.py +186 -0
- package/latticeai/services/memory_service/proof.py +136 -0
- package/latticeai/services/memory_service/recall.py +225 -0
- package/latticeai/services/memory_service/service.py +48 -0
- package/latticeai/services/memory_service/stores.py +110 -0
- package/latticeai/services/model_runtime/__init__.py +322 -0
- package/latticeai/services/model_runtime/cloud.py +87 -0
- package/latticeai/services/model_runtime/download.py +282 -0
- package/latticeai/services/model_runtime/engines.py +341 -0
- package/latticeai/services/model_runtime/loading.py +178 -0
- package/latticeai/services/model_runtime/service.py +129 -0
- package/latticeai/services/model_runtime/state.py +131 -0
- package/latticeai/services/model_runtime/status.py +255 -0
- package/latticeai/services/product_readiness.py +15 -7
- package/latticeai/setup/wizard/__init__.py +126 -0
- package/latticeai/setup/wizard/catalog.py +172 -0
- package/latticeai/setup/wizard/detect.py +323 -0
- package/latticeai/setup/wizard/install.py +348 -0
- package/latticeai/setup/wizard/paths.py +168 -0
- package/latticeai/setup/wizard/plans.py +74 -0
- package/latticeai/setup/wizard/recommend.py +320 -0
- package/package.json +6 -2
- package/scripts/bump_version.py +14 -0
- package/scripts/capture_release_evidence.mjs +33 -21
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_i18n_namespace_coverage.mjs +41 -4
- package/scripts/check_max_file_lines.mjs +102 -0
- package/scripts/check_release_evidence_bound.mjs +30 -15
- package/scripts/check_screenshot_pixel_delta.py +34 -4
- package/scripts/check_server_i18n.mjs +1 -0
- package/scripts/generate_rust_parity_fixtures.py +562 -0
- package/scripts/lib/mock_server_fingerprint.mjs +94 -0
- package/scripts/release_screen_claims.json +31 -2
- package/src-tauri/Cargo.lock +361 -3
- package/src-tauri/Cargo.toml +6 -1
- package/src-tauri/src/backend.rs +349 -0
- package/src-tauri/src/folder.rs +33 -0
- package/src-tauri/src/main.rs +97 -399
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +41 -37
- package/static/app/assets/Act-yYpYnn0v.js +1 -0
- package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
- package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
- package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
- package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
- package/static/app/assets/Capture-CFIRsFNE.js +1 -0
- package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
- package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
- package/static/app/assets/Library-DwO3yZST.js +1 -0
- package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
- package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
- package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
- package/static/app/assets/System-DW8F-2xL.js +1 -0
- package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
- package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
- package/static/app/assets/brain-Ci1CkWjM.js +1 -0
- package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
- package/static/app/assets/circle-check-DfInj-qD.js +1 -0
- package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
- package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
- package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
- package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
- package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
- package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
- package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
- package/static/app/assets/index-_u5iUHDr.js +10 -0
- package/static/app/assets/input-B0lPdRQZ.js +1 -0
- package/static/app/assets/link-2-CoFbooHS.js +1 -0
- package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
- package/static/app/assets/primitives-DEbN-d6p.js +1 -0
- package/static/app/assets/search-BybIWPNd.js +1 -0
- package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
- package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
- package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
- package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
- package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
- package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
- package/static/app/assets/utils-BlZr7Pd4.js +4 -0
- package/static/app/assets/workspace-jJY4RuAV.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/graph/_kg_common.py +0 -1331
- package/lattice_brain/graph/discovery_index.py +0 -1141
- package/lattice_brain/graph/retrieval.py +0 -1120
- package/lattice_brain/graph/retrieval_vector.py +0 -1293
- package/lattice_brain/ingestion.py +0 -1525
- package/lattice_brain/multimodal.py +0 -1258
- package/latticeai/core/agent.py +0 -1465
- package/latticeai/core/embedding_providers.py +0 -1196
- package/latticeai/core/file_generation.py +0 -1047
- package/latticeai/integrations/telegram_bot.py +0 -1390
- package/latticeai/models/router.py +0 -1007
- package/latticeai/runtime/build_phases.py +0 -1450
- package/latticeai/services/brain_intelligence.py +0 -1083
- package/latticeai/services/memory_service.py +0 -1177
- package/latticeai/services/model_runtime.py +0 -1281
- package/latticeai/setup/wizard.py +0 -1310
- package/static/app/assets/Act-AWf0SAKp.js +0 -1
- package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
- package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
- package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
- package/static/app/assets/Capture-CqOSzyPr.js +0 -1
- package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
- package/static/app/assets/Library-CX-bbhmK.js +0 -1
- package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
- package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
- package/static/app/assets/System-Bu2t5hn1.js +0 -1
- package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
- package/static/app/assets/brain-DJMoqrwx.js +0 -1
- package/static/app/assets/index-BpYkzcVm.js +0 -10
- package/static/app/assets/input-DSlJJxRs.js +0 -1
- package/static/app/assets/primitives-BCx6TvfG.js +0 -1
- package/static/app/assets/search-Cgy8cCFJ.js +0 -1
- package/static/app/assets/utils-zqPZJxdx.js +0 -4
- package/static/app/assets/workspace-DXTihhfU.js +0 -1
|
@@ -1,1293 +0,0 @@
|
|
|
1
|
-
from __future__ import annotations
|
|
2
|
-
|
|
3
|
-
import dataclasses
|
|
4
|
-
from typing import TYPE_CHECKING
|
|
5
|
-
|
|
6
|
-
# ruff: noqa: F403,F405
|
|
7
|
-
from ._kg_common import * # noqa: F403,F401
|
|
8
|
-
from .vector_index import (
|
|
9
|
-
DEFAULT_VECTOR_INDEX,
|
|
10
|
-
HNSW_BACKEND,
|
|
11
|
-
VECTOR_INDEX_ENV,
|
|
12
|
-
BackendSelection,
|
|
13
|
-
HnswIndex,
|
|
14
|
-
IndexItem,
|
|
15
|
-
VectorEmbedQueue,
|
|
16
|
-
VectorIndex,
|
|
17
|
-
build_index,
|
|
18
|
-
resolve_vector_index,
|
|
19
|
-
)
|
|
20
|
-
|
|
21
|
-
# The cross-mixin surface (`_connect`, `_upsert_node`, …) is declared in
|
|
22
|
-
# `_kg_contract.KnowledgeGraphCore`. It is a typing-only base: at runtime this
|
|
23
|
-
# is `object`, so the MRO of `KnowledgeGraphStore` is unchanged.
|
|
24
|
-
if TYPE_CHECKING:
|
|
25
|
-
from ._kg_contract import KnowledgeGraphCore as _Core
|
|
26
|
-
else:
|
|
27
|
-
_Core = object
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
# ── brute-force recall cap (review 2026-08 P1 #2) ────────────────────────────
|
|
31
|
-
# There is no ANN index in the default build: sqlite-vec is an optional
|
|
32
|
-
# dependency and, when it is absent, `index_status()["storage"]` honestly
|
|
33
|
-
# reports ``vector_search_backend: "bruteforce-cosine"``. Brute force scores
|
|
34
|
-
# every candidate in Python, so *some* cap is unavoidable on a large graph.
|
|
35
|
-
#
|
|
36
|
-
# What is not acceptable is a SILENT cap. The pre-10.7 code took the 10 000
|
|
37
|
-
# most recently indexed rows — ordered by ``indexed_at``, i.e. by recency, not
|
|
38
|
-
# by similarity — and returned them as if they were the whole index, so recall
|
|
39
|
-
# on a 200 000-row brain quietly became "the newest 5%". The cap is now
|
|
40
|
-
# explicit, configurable, and reported back to the caller in ``recall``.
|
|
41
|
-
#
|
|
42
|
-
# ``LATTICEAI_VECTOR_MAX_CANDIDATES`` overrides the default; ``0`` means "no
|
|
43
|
-
# cap — scan the whole index" (exact recall, paid for in latency).
|
|
44
|
-
VECTOR_MAX_CANDIDATES_ENV = "LATTICEAI_VECTOR_MAX_CANDIDATES"
|
|
45
|
-
DEFAULT_VECTOR_MAX_CANDIDATES = 10_000
|
|
46
|
-
#: Upper bound for a configured cap; ``0``/``None`` still means uncapped.
|
|
47
|
-
VECTOR_MAX_CANDIDATES_CEILING = 500_000
|
|
48
|
-
|
|
49
|
-
# ── scan batching (v11.1.0) ──────────────────────────────────────────────────
|
|
50
|
-
# The exact scan hands its candidates to a VectorIndex, which by definition
|
|
51
|
-
# holds what it is given. Handing it the whole result set would make peak
|
|
52
|
-
# memory O(rows × dim) floats, so the scan feeds it in fixed batches instead:
|
|
53
|
-
# exhaustive backends score every batch independently, so the union is
|
|
54
|
-
# identical to one big pass, at O(batch × dim) resident cost.
|
|
55
|
-
VECTOR_SCAN_BATCH = 512
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
def _configured_vector_max_candidates() -> Optional[int]:
|
|
59
|
-
"""Resolve the candidate cap from the environment (None = uncapped).
|
|
60
|
-
|
|
61
|
-
Never raises: an unparseable value falls back to the documented default
|
|
62
|
-
rather than breaking every search.
|
|
63
|
-
"""
|
|
64
|
-
raw = os.getenv(VECTOR_MAX_CANDIDATES_ENV)
|
|
65
|
-
if raw is None or not raw.strip():
|
|
66
|
-
return DEFAULT_VECTOR_MAX_CANDIDATES
|
|
67
|
-
try:
|
|
68
|
-
value = int(raw.strip())
|
|
69
|
-
except ValueError:
|
|
70
|
-
logging.warning(
|
|
71
|
-
"%s=%r is not an integer — using the default cap of %d",
|
|
72
|
-
VECTOR_MAX_CANDIDATES_ENV, raw, DEFAULT_VECTOR_MAX_CANDIDATES,
|
|
73
|
-
)
|
|
74
|
-
return DEFAULT_VECTOR_MAX_CANDIDATES
|
|
75
|
-
if value <= 0:
|
|
76
|
-
return None # explicit opt-in to an exhaustive scan
|
|
77
|
-
return min(value, VECTOR_MAX_CANDIDATES_CEILING)
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
class KnowledgeGraphVectorMixin(_Core):
|
|
81
|
-
"""Vector-embedding index build/status/search, split out of retrieval.
|
|
82
|
-
|
|
83
|
-
Composed into KnowledgeGraphStore alongside KnowledgeGraphRetrievalMixin;
|
|
84
|
-
both mixins share the same instance, so vector methods still reach sibling
|
|
85
|
-
retrieval/write helpers (e.g. self._vector_text_for_node) through the MRO.
|
|
86
|
-
"""
|
|
87
|
-
|
|
88
|
-
# ── embedder fingerprint (review Wave 2.2 — stale_embedder) ──────────────
|
|
89
|
-
# vector_search filters on the CURRENT model/dim, so swapping the embedder
|
|
90
|
-
# silently yields zero vector rows. The fingerprint persisted in graph_meta
|
|
91
|
-
# records which embedder actually built the index; a mismatch is surfaced
|
|
92
|
-
# as the honest ``stale_embedder`` signal instead of a silent degradation.
|
|
93
|
-
|
|
94
|
-
_EMBEDDER_FINGERPRINT_KEY = "embedder_fingerprint"
|
|
95
|
-
|
|
96
|
-
def _embedder_fingerprint_record(
|
|
97
|
-
self, conn: sqlite3.Connection
|
|
98
|
-
) -> Optional[Dict[str, Any]]:
|
|
99
|
-
"""Read the recorded embedder fingerprint from graph_meta (or None)."""
|
|
100
|
-
row = conn.execute(
|
|
101
|
-
"SELECT value FROM graph_meta WHERE key=?",
|
|
102
|
-
(self._EMBEDDER_FINGERPRINT_KEY,),
|
|
103
|
-
).fetchone()
|
|
104
|
-
if not row:
|
|
105
|
-
return None
|
|
106
|
-
payload = _safe_loads(row["value"])
|
|
107
|
-
if not isinstance(payload, dict) or not payload.get("model_id"):
|
|
108
|
-
return None
|
|
109
|
-
try:
|
|
110
|
-
dim = int(payload.get("dim") or 0)
|
|
111
|
-
except (TypeError, ValueError):
|
|
112
|
-
dim = 0
|
|
113
|
-
return {"model_id": str(payload["model_id"]), "dim": dim}
|
|
114
|
-
|
|
115
|
-
def _write_embedder_fingerprint(self, conn: sqlite3.Connection) -> Dict[str, Any]:
|
|
116
|
-
"""Persist the CURRENT embedder identity (same transaction as caller)."""
|
|
117
|
-
fingerprint = {
|
|
118
|
-
"model_id": self._embedding_model.model_id,
|
|
119
|
-
"dim": int(self._embedding_model.dim),
|
|
120
|
-
}
|
|
121
|
-
conn.execute(
|
|
122
|
-
"INSERT OR REPLACE INTO graph_meta(key, value) VALUES (?, ?)",
|
|
123
|
-
(self._EMBEDDER_FINGERPRINT_KEY, _json(fingerprint)),
|
|
124
|
-
)
|
|
125
|
-
return fingerprint
|
|
126
|
-
|
|
127
|
-
def record_embedder_fingerprint(self) -> Dict[str, Any]:
|
|
128
|
-
"""Record the current embedder (model_id + dim) as the index builder."""
|
|
129
|
-
with self._connect() as conn:
|
|
130
|
-
return self._write_embedder_fingerprint(conn)
|
|
131
|
-
|
|
132
|
-
def embedder_fingerprint_status(self) -> Dict[str, Any]:
|
|
133
|
-
"""Compare the current embedder against the recorded index fingerprint.
|
|
134
|
-
|
|
135
|
-
Returns ``{"current": {model_id, dim}, "recorded": {...} | None,
|
|
136
|
-
"stale_embedder": bool}``. ``stale_embedder`` is True only when a
|
|
137
|
-
fingerprint was recorded AND it differs from the current embedder —
|
|
138
|
-
an unrecorded index (legacy DBs, nothing indexed yet) is honestly
|
|
139
|
-
"unknown", never reported stale. Never raises.
|
|
140
|
-
"""
|
|
141
|
-
current = {
|
|
142
|
-
"model_id": self._embedding_model.model_id,
|
|
143
|
-
"dim": int(self._embedding_model.dim),
|
|
144
|
-
}
|
|
145
|
-
recorded: Optional[Dict[str, Any]] = None
|
|
146
|
-
try:
|
|
147
|
-
with self._connect() as conn:
|
|
148
|
-
recorded = self._embedder_fingerprint_record(conn)
|
|
149
|
-
except Exception: # noqa: BLE001 — status must degrade, never raise
|
|
150
|
-
recorded = None
|
|
151
|
-
stale = bool(
|
|
152
|
-
recorded is not None
|
|
153
|
-
and (
|
|
154
|
-
recorded.get("model_id") != current["model_id"]
|
|
155
|
-
or recorded.get("dim") != current["dim"]
|
|
156
|
-
)
|
|
157
|
-
)
|
|
158
|
-
return {"current": current, "recorded": recorded, "stale_embedder": stale}
|
|
159
|
-
|
|
160
|
-
def _vector_text_hashes(self, conn: sqlite3.Connection) -> Dict[str, str]:
|
|
161
|
-
"""``item_id -> text_hash`` for rows already embedded by *this* embedder.
|
|
162
|
-
|
|
163
|
-
The incremental rebuild's job is mostly deciding what it does *not*
|
|
164
|
-
have to do, and it used to ask that question with one ``SELECT`` per
|
|
165
|
-
candidate item — a round trip per node and per chunk on every run,
|
|
166
|
-
almost all of which answer "unchanged". One query returning two short
|
|
167
|
-
columns replaces all of them.
|
|
168
|
-
|
|
169
|
-
Rows written by a different embedder are left out, so they compare as
|
|
170
|
-
missing and get re-embedded, which is what an embedder swap requires.
|
|
171
|
-
"""
|
|
172
|
-
return {
|
|
173
|
-
row["item_id"]: row["text_hash"]
|
|
174
|
-
for row in conn.execute(
|
|
175
|
-
"""
|
|
176
|
-
SELECT item_id, text_hash
|
|
177
|
-
FROM vector_embeddings
|
|
178
|
-
WHERE embedding_model=? AND embedding_dim=?
|
|
179
|
-
""",
|
|
180
|
-
(self._embedding_model.model_id, self._embedding_model.dim),
|
|
181
|
-
).fetchall()
|
|
182
|
-
}
|
|
183
|
-
|
|
184
|
-
def _iter_vector_source_items(
|
|
185
|
-
self,
|
|
186
|
-
conn: sqlite3.Connection,
|
|
187
|
-
*,
|
|
188
|
-
include_nodes: bool = True,
|
|
189
|
-
include_chunks: bool = True,
|
|
190
|
-
) -> Iterator[Dict[str, Any]]:
|
|
191
|
-
"""Stream the graph's embeddable text, one item at a time.
|
|
192
|
-
|
|
193
|
-
Yields rather than returns a list. Every caller consumes this exactly
|
|
194
|
-
once in a ``for``, and building the list first meant a rebuild held
|
|
195
|
-
the full text of every node and chunk in memory simultaneously — the
|
|
196
|
-
one shape guaranteed to fail on precisely the large graph that most
|
|
197
|
-
needs the index.
|
|
198
|
-
"""
|
|
199
|
-
if include_nodes:
|
|
200
|
-
for row in conn.execute(
|
|
201
|
-
"""
|
|
202
|
-
SELECT id, type, title, summary, metadata_json
|
|
203
|
-
FROM nodes
|
|
204
|
-
WHERE type <> 'Chunk'
|
|
205
|
-
ORDER BY updated_at DESC, id ASC
|
|
206
|
-
"""
|
|
207
|
-
).fetchall():
|
|
208
|
-
metadata = _safe_loads(row["metadata_json"])
|
|
209
|
-
text = self._vector_text_for_node(
|
|
210
|
-
title=row["title"],
|
|
211
|
-
summary=row["summary"] or "",
|
|
212
|
-
metadata=metadata,
|
|
213
|
-
)
|
|
214
|
-
if text:
|
|
215
|
-
yield {
|
|
216
|
-
"item_id": row["id"],
|
|
217
|
-
"item_type": "node",
|
|
218
|
-
"source_node": row["id"],
|
|
219
|
-
"text": text,
|
|
220
|
-
"metadata": {"node_type": row["type"], **metadata},
|
|
221
|
-
}
|
|
222
|
-
if include_chunks:
|
|
223
|
-
for row in conn.execute(
|
|
224
|
-
"""
|
|
225
|
-
SELECT c.id, c.source_node AS parent_source_node, c.text, c.metadata_json
|
|
226
|
-
FROM chunks c
|
|
227
|
-
JOIN nodes n ON n.id=c.id
|
|
228
|
-
ORDER BY c.created_at DESC, c.id ASC
|
|
229
|
-
"""
|
|
230
|
-
).fetchall():
|
|
231
|
-
metadata = _safe_loads(row["metadata_json"])
|
|
232
|
-
text = _clean_text(row["text"] or "")
|
|
233
|
-
if text:
|
|
234
|
-
yield {
|
|
235
|
-
"item_id": row["id"],
|
|
236
|
-
"item_type": "chunk",
|
|
237
|
-
"source_node": row["id"],
|
|
238
|
-
"text": text,
|
|
239
|
-
"metadata": {
|
|
240
|
-
**metadata,
|
|
241
|
-
"parent_source_node": row["parent_source_node"],
|
|
242
|
-
},
|
|
243
|
-
}
|
|
244
|
-
|
|
245
|
-
def index_node_incremental(self, node_id: str) -> Dict[str, Any]:
|
|
246
|
-
"""Embed/index only ``node_id`` and its chunks (incremental sync).
|
|
247
|
-
|
|
248
|
-
The item construction mirrors :meth:`_iter_vector_source_items` exactly
|
|
249
|
-
(same ids, same ``source_node``/``parent_source_node`` semantics), so
|
|
250
|
-
anything this method indexes is indistinguishable from a full
|
|
251
|
-
:meth:`rebuild_vector_index` pass — and anything it *fails* to index
|
|
252
|
-
stays visible as ``missing``/``stale`` backlog in :meth:`index_status`,
|
|
253
|
-
where a later rebuild picks it up.
|
|
254
|
-
|
|
255
|
-
Never raises: embedding-provider or storage failures are reported as
|
|
256
|
-
``{"status": "failed", ...}`` so ingestion callers can degrade instead
|
|
257
|
-
of losing an already-persisted write.
|
|
258
|
-
"""
|
|
259
|
-
node_id = str(node_id or "").strip()
|
|
260
|
-
started = time.perf_counter()
|
|
261
|
-
summary: Dict[str, Any] = {
|
|
262
|
-
"node_id": node_id,
|
|
263
|
-
"items_total": 0,
|
|
264
|
-
"items_indexed": 0,
|
|
265
|
-
"items_skipped": 0,
|
|
266
|
-
}
|
|
267
|
-
if not node_id:
|
|
268
|
-
return {**summary, "status": "skipped", "detail": "node_id required"}
|
|
269
|
-
try:
|
|
270
|
-
with self._connect() as conn:
|
|
271
|
-
row = conn.execute(
|
|
272
|
-
"SELECT id, type, title, summary, metadata_json FROM nodes WHERE id=?",
|
|
273
|
-
(node_id,),
|
|
274
|
-
).fetchone()
|
|
275
|
-
if row is None:
|
|
276
|
-
return {**summary, "status": "skipped", "detail": "node not found"}
|
|
277
|
-
items: List[Dict[str, Any]] = []
|
|
278
|
-
if row["type"] != "Chunk":
|
|
279
|
-
metadata = _safe_loads(row["metadata_json"])
|
|
280
|
-
text = self._vector_text_for_node(
|
|
281
|
-
title=row["title"],
|
|
282
|
-
summary=row["summary"] or "",
|
|
283
|
-
metadata=metadata,
|
|
284
|
-
)
|
|
285
|
-
if text:
|
|
286
|
-
items.append(
|
|
287
|
-
{
|
|
288
|
-
"item_id": row["id"],
|
|
289
|
-
"item_type": "node",
|
|
290
|
-
"source_node": row["id"],
|
|
291
|
-
"text": text,
|
|
292
|
-
"metadata": {"node_type": row["type"], **metadata},
|
|
293
|
-
}
|
|
294
|
-
)
|
|
295
|
-
for chunk_row in conn.execute(
|
|
296
|
-
"""
|
|
297
|
-
SELECT c.id, c.source_node AS parent_source_node, c.text, c.metadata_json
|
|
298
|
-
FROM chunks c
|
|
299
|
-
JOIN nodes n ON n.id=c.id
|
|
300
|
-
WHERE c.source_node=?
|
|
301
|
-
ORDER BY c.created_at ASC, c.id ASC
|
|
302
|
-
""",
|
|
303
|
-
(node_id,),
|
|
304
|
-
).fetchall():
|
|
305
|
-
metadata = _safe_loads(chunk_row["metadata_json"])
|
|
306
|
-
text = _clean_text(chunk_row["text"] or "")
|
|
307
|
-
if text:
|
|
308
|
-
items.append(
|
|
309
|
-
{
|
|
310
|
-
"item_id": chunk_row["id"],
|
|
311
|
-
"item_type": "chunk",
|
|
312
|
-
"source_node": chunk_row["id"],
|
|
313
|
-
"text": text,
|
|
314
|
-
"metadata": {
|
|
315
|
-
**metadata,
|
|
316
|
-
"parent_source_node": chunk_row["parent_source_node"],
|
|
317
|
-
},
|
|
318
|
-
}
|
|
319
|
-
)
|
|
320
|
-
indexed = skipped = 0
|
|
321
|
-
for item in items:
|
|
322
|
-
if self._upsert_vector_item(conn, **item):
|
|
323
|
-
indexed += 1
|
|
324
|
-
else:
|
|
325
|
-
skipped += 1
|
|
326
|
-
if indexed and self._embedder_fingerprint_record(conn) is None:
|
|
327
|
-
# First successful vector write establishes the fingerprint;
|
|
328
|
-
# later incremental writes never overwrite it (only a full
|
|
329
|
-
# rebuild may flip it after an embedder swap).
|
|
330
|
-
self._write_embedder_fingerprint(conn)
|
|
331
|
-
summary.update(
|
|
332
|
-
{
|
|
333
|
-
"items_total": len(items),
|
|
334
|
-
"items_indexed": indexed,
|
|
335
|
-
"items_skipped": skipped,
|
|
336
|
-
}
|
|
337
|
-
)
|
|
338
|
-
return {
|
|
339
|
-
**summary,
|
|
340
|
-
"status": "indexed" if indexed else "noop",
|
|
341
|
-
"duration_ms": round((time.perf_counter() - started) * 1000, 2),
|
|
342
|
-
"embedding_model": self._embedding_model.model_id,
|
|
343
|
-
}
|
|
344
|
-
except Exception as exc: # noqa: BLE001 — incremental sync must never raise
|
|
345
|
-
return {
|
|
346
|
-
**summary,
|
|
347
|
-
"status": "failed",
|
|
348
|
-
"detail": str(exc),
|
|
349
|
-
"duration_ms": round((time.perf_counter() - started) * 1000, 2),
|
|
350
|
-
}
|
|
351
|
-
|
|
352
|
-
def rebuild_vector_index(
|
|
353
|
-
self,
|
|
354
|
-
*,
|
|
355
|
-
full: bool = False,
|
|
356
|
-
include_nodes: bool = True,
|
|
357
|
-
include_chunks: bool = True,
|
|
358
|
-
) -> Dict[str, Any]:
|
|
359
|
-
"""Rebuild the derived vector index without mutating graph content."""
|
|
360
|
-
op_id = f"vector-op:{_sha256_text(f'{time.time()}:{os.getpid()}')[:24]}"
|
|
361
|
-
requested_at = _now()
|
|
362
|
-
started = time.perf_counter()
|
|
363
|
-
try:
|
|
364
|
-
with self._connect() as conn:
|
|
365
|
-
conn.execute(
|
|
366
|
-
"""
|
|
367
|
-
INSERT INTO vector_index_operations(
|
|
368
|
-
id, operation, status, requested_at, started_at, metadata_json
|
|
369
|
-
)
|
|
370
|
-
VALUES (?, ?, 'running', ?, ?, ?)
|
|
371
|
-
""",
|
|
372
|
-
(
|
|
373
|
-
op_id,
|
|
374
|
-
"rebuild_full" if full else "rebuild_incremental",
|
|
375
|
-
requested_at,
|
|
376
|
-
requested_at,
|
|
377
|
-
_json(
|
|
378
|
-
{
|
|
379
|
-
"include_nodes": include_nodes,
|
|
380
|
-
"include_chunks": include_chunks,
|
|
381
|
-
}
|
|
382
|
-
),
|
|
383
|
-
),
|
|
384
|
-
)
|
|
385
|
-
if full:
|
|
386
|
-
filters = []
|
|
387
|
-
if include_nodes:
|
|
388
|
-
filters.append("'node'")
|
|
389
|
-
if include_chunks:
|
|
390
|
-
filters.append("'chunk'")
|
|
391
|
-
if filters:
|
|
392
|
-
conn.execute(
|
|
393
|
-
f"DELETE FROM vector_embeddings WHERE item_type IN ({','.join(filters)})"
|
|
394
|
-
)
|
|
395
|
-
# After a full wipe nothing is current by definition, so the
|
|
396
|
-
# prefetch would only be a wasted scan of a table we just
|
|
397
|
-
# emptied. Incremental is where it pays: it turns "one SELECT
|
|
398
|
-
# per item, nearly all of which say unchanged" into one query.
|
|
399
|
-
known = {} if full else self._vector_text_hashes(conn)
|
|
400
|
-
total = indexed = skipped = 0
|
|
401
|
-
for item in self._iter_vector_source_items(
|
|
402
|
-
conn,
|
|
403
|
-
include_nodes=include_nodes,
|
|
404
|
-
include_chunks=include_chunks,
|
|
405
|
-
):
|
|
406
|
-
total += 1
|
|
407
|
-
if known.get(item["item_id"]) == _sha256_text(_clean_text(item["text"])):
|
|
408
|
-
skipped += 1
|
|
409
|
-
continue
|
|
410
|
-
if self._upsert_vector_item(conn, **item):
|
|
411
|
-
indexed += 1
|
|
412
|
-
else:
|
|
413
|
-
skipped += 1
|
|
414
|
-
duration_ms = round((time.perf_counter() - started) * 1000, 2)
|
|
415
|
-
conn.execute(
|
|
416
|
-
"""
|
|
417
|
-
UPDATE vector_index_operations
|
|
418
|
-
SET status='completed', completed_at=?, items_total=?,
|
|
419
|
-
items_indexed=?, items_skipped=?, metadata_json=?
|
|
420
|
-
WHERE id=?
|
|
421
|
-
""",
|
|
422
|
-
(
|
|
423
|
-
_now(),
|
|
424
|
-
total,
|
|
425
|
-
indexed,
|
|
426
|
-
skipped,
|
|
427
|
-
_json(
|
|
428
|
-
{
|
|
429
|
-
"include_nodes": include_nodes,
|
|
430
|
-
"include_chunks": include_chunks,
|
|
431
|
-
"duration_ms": duration_ms,
|
|
432
|
-
"embedding_model": self._embedding_model.model_id,
|
|
433
|
-
"embedding_dim": self._embedding_model.dim,
|
|
434
|
-
}
|
|
435
|
-
),
|
|
436
|
-
op_id,
|
|
437
|
-
),
|
|
438
|
-
)
|
|
439
|
-
# A successful rebuild (re)establishes which embedder built
|
|
440
|
-
# the index — this is the only path that may flip a recorded
|
|
441
|
-
# fingerprint after an embedder swap.
|
|
442
|
-
self._write_embedder_fingerprint(conn)
|
|
443
|
-
return {
|
|
444
|
-
"status": "completed",
|
|
445
|
-
"operation_id": op_id,
|
|
446
|
-
"full": bool(full),
|
|
447
|
-
"items_total": total,
|
|
448
|
-
"items_indexed": indexed,
|
|
449
|
-
"items_skipped": skipped,
|
|
450
|
-
"duration_ms": duration_ms,
|
|
451
|
-
"embedding_model": self._embedding_model.model_id,
|
|
452
|
-
"embedding_dim": self._embedding_model.dim,
|
|
453
|
-
}
|
|
454
|
-
except Exception as exc:
|
|
455
|
-
duration_ms = round((time.perf_counter() - started) * 1000, 2)
|
|
456
|
-
with self._connect() as conn:
|
|
457
|
-
conn.execute(
|
|
458
|
-
"""
|
|
459
|
-
INSERT INTO vector_index_operations(
|
|
460
|
-
id, operation, status, requested_at, started_at, completed_at,
|
|
461
|
-
error_message, metadata_json
|
|
462
|
-
)
|
|
463
|
-
VALUES (?, ?, 'failed', ?, ?, ?, ?, ?)
|
|
464
|
-
ON CONFLICT(id) DO UPDATE SET
|
|
465
|
-
status='failed',
|
|
466
|
-
completed_at=excluded.completed_at,
|
|
467
|
-
error_message=excluded.error_message,
|
|
468
|
-
metadata_json=excluded.metadata_json
|
|
469
|
-
""",
|
|
470
|
-
(
|
|
471
|
-
op_id,
|
|
472
|
-
"rebuild_full" if full else "rebuild_incremental",
|
|
473
|
-
requested_at,
|
|
474
|
-
requested_at,
|
|
475
|
-
_now(),
|
|
476
|
-
str(exc),
|
|
477
|
-
_json({"duration_ms": duration_ms}),
|
|
478
|
-
),
|
|
479
|
-
)
|
|
480
|
-
raise
|
|
481
|
-
|
|
482
|
-
def index_status(self) -> Dict[str, Any]:
|
|
483
|
-
storage_capabilities = None
|
|
484
|
-
try:
|
|
485
|
-
storage_capabilities = self.storage_engine.capabilities().as_dict()
|
|
486
|
-
except Exception as exc:
|
|
487
|
-
storage_capabilities = {
|
|
488
|
-
"engine": "sqlite",
|
|
489
|
-
"available": False,
|
|
490
|
-
"reason": str(exc),
|
|
491
|
-
}
|
|
492
|
-
with self._connect() as conn:
|
|
493
|
-
vector_counts = {
|
|
494
|
-
row["item_type"]: row["count"]
|
|
495
|
-
for row in conn.execute(
|
|
496
|
-
"SELECT item_type, COUNT(*) AS count FROM vector_embeddings GROUP BY item_type"
|
|
497
|
-
)
|
|
498
|
-
}
|
|
499
|
-
# Materialised on purpose, unlike the rebuild path: status walks
|
|
500
|
-
# this set twice (once for ids, once to classify each item) and
|
|
501
|
-
# reports a count, none of which a one-shot iterator can serve.
|
|
502
|
-
source_items = list(self._iter_vector_source_items(conn))
|
|
503
|
-
vector_rows = {
|
|
504
|
-
row["item_id"]: row
|
|
505
|
-
for row in conn.execute(
|
|
506
|
-
"""
|
|
507
|
-
SELECT item_id, text_hash, embedding_dim, embedding_model, indexed_at
|
|
508
|
-
FROM vector_embeddings
|
|
509
|
-
"""
|
|
510
|
-
).fetchall()
|
|
511
|
-
}
|
|
512
|
-
latest_rows = conn.execute(
|
|
513
|
-
"""
|
|
514
|
-
SELECT id, operation, status, requested_at, started_at, completed_at,
|
|
515
|
-
items_total, items_indexed, items_skipped, error_message, metadata_json
|
|
516
|
-
FROM vector_index_operations
|
|
517
|
-
ORDER BY requested_at DESC, id DESC
|
|
518
|
-
LIMIT 5
|
|
519
|
-
"""
|
|
520
|
-
).fetchall()
|
|
521
|
-
missing = stale = ready = 0
|
|
522
|
-
source_item_ids = {str(item["item_id"]) for item in source_items}
|
|
523
|
-
backlog_by_type: Dict[str, int] = {}
|
|
524
|
-
backlog_reasons: Dict[str, int] = {}
|
|
525
|
-
backlog_samples: List[Dict[str, Any]] = []
|
|
526
|
-
|
|
527
|
-
def add_backlog(item: Dict[str, Any], reason: str) -> None:
|
|
528
|
-
item_type = str(item.get("item_type") or "unknown")
|
|
529
|
-
backlog_by_type[item_type] = backlog_by_type.get(item_type, 0) + 1
|
|
530
|
-
backlog_reasons[reason] = backlog_reasons.get(reason, 0) + 1
|
|
531
|
-
if len(backlog_samples) >= 20:
|
|
532
|
-
return
|
|
533
|
-
backlog_samples.append(
|
|
534
|
-
{
|
|
535
|
-
"item_id": item.get("item_id"),
|
|
536
|
-
"item_type": item_type,
|
|
537
|
-
"source_node": item.get("source_node"),
|
|
538
|
-
"reason": reason,
|
|
539
|
-
"metadata": {
|
|
540
|
-
key: value
|
|
541
|
-
for key, value in dict(item.get("metadata") or {}).items()
|
|
542
|
-
if key in {"node_type", "source", "conversation_id", "parent_source_node"}
|
|
543
|
-
},
|
|
544
|
-
}
|
|
545
|
-
)
|
|
546
|
-
|
|
547
|
-
for item in source_items:
|
|
548
|
-
vector_row = vector_rows.get(item["item_id"])
|
|
549
|
-
expected_hash = _sha256_text(_clean_text(item["text"]))
|
|
550
|
-
if not vector_row:
|
|
551
|
-
missing += 1
|
|
552
|
-
add_backlog(item, "missing_vector")
|
|
553
|
-
elif (
|
|
554
|
-
vector_row["text_hash"] != expected_hash
|
|
555
|
-
or vector_row["embedding_dim"] != self._embedding_model.dim
|
|
556
|
-
or vector_row["embedding_model"] != self._embedding_model.model_id
|
|
557
|
-
):
|
|
558
|
-
stale += 1
|
|
559
|
-
reason = "text_changed"
|
|
560
|
-
if vector_row["embedding_model"] != self._embedding_model.model_id:
|
|
561
|
-
reason = "model_changed"
|
|
562
|
-
elif vector_row["embedding_dim"] != self._embedding_model.dim:
|
|
563
|
-
reason = "dimension_changed"
|
|
564
|
-
add_backlog(item, reason)
|
|
565
|
-
else:
|
|
566
|
-
ready += 1
|
|
567
|
-
pending = missing + stale
|
|
568
|
-
orphaned_items = max(0, len(set(vector_rows) - source_item_ids))
|
|
569
|
-
coverage_ratio = round(ready / len(source_items), 6) if source_items else 1.0
|
|
570
|
-
latest_completed = None
|
|
571
|
-
for row in latest_rows:
|
|
572
|
-
if row["status"] == "completed":
|
|
573
|
-
latest_completed = row
|
|
574
|
-
break
|
|
575
|
-
latency_budget: Dict[str, Any] = {
|
|
576
|
-
"target_rebuild_ms": 10_000,
|
|
577
|
-
"last_rebuild_duration_ms": None,
|
|
578
|
-
"last_items_per_second": None,
|
|
579
|
-
"within_target": None,
|
|
580
|
-
}
|
|
581
|
-
if latest_completed is not None:
|
|
582
|
-
metadata = _safe_loads(latest_completed["metadata_json"])
|
|
583
|
-
duration_ms = metadata.get("duration_ms")
|
|
584
|
-
items_total = int(latest_completed["items_total"] or 0)
|
|
585
|
-
if isinstance(duration_ms, (int, float)) and duration_ms > 0:
|
|
586
|
-
latency_budget.update(
|
|
587
|
-
{
|
|
588
|
-
"last_rebuild_duration_ms": round(float(duration_ms), 2),
|
|
589
|
-
"last_items_per_second": round(items_total / (float(duration_ms) / 1000.0), 2),
|
|
590
|
-
"within_target": float(duration_ms) <= 10_000,
|
|
591
|
-
}
|
|
592
|
-
)
|
|
593
|
-
embedder_status = self.embedder_fingerprint_status()
|
|
594
|
-
return {
|
|
595
|
-
"status": "ready" if pending == 0 else "needs_reindex",
|
|
596
|
-
"embedder": embedder_status,
|
|
597
|
-
"storage": {
|
|
598
|
-
"db_path": str(self.db_path),
|
|
599
|
-
"backend": "sqlite",
|
|
600
|
-
"embedding_model": self._embedding_model.model_id,
|
|
601
|
-
"embedding_dim": self._embedding_model.dim,
|
|
602
|
-
# Honest capability report: trigram FTS5 keyword index, or
|
|
603
|
-
# LIKE-scan fallback when this SQLite build lacks it.
|
|
604
|
-
"fts_enabled": bool(getattr(self, "_fts_enabled", False)),
|
|
605
|
-
"engine": storage_capabilities,
|
|
606
|
-
"vector_search_backend": (
|
|
607
|
-
storage_capabilities.get("vector_backend")
|
|
608
|
-
if isinstance(storage_capabilities, dict)
|
|
609
|
-
else "bruteforce-cosine"
|
|
610
|
-
),
|
|
611
|
-
"vector_search_mode": (
|
|
612
|
-
(storage_capabilities.get("metadata") or {}).get("vector_mode")
|
|
613
|
-
if isinstance(storage_capabilities, dict)
|
|
614
|
-
else "fallback"
|
|
615
|
-
),
|
|
616
|
-
"sqlite_vec_ann_available": (
|
|
617
|
-
bool((storage_capabilities.get("metadata") or {}).get("sqlite_vec_ann_available"))
|
|
618
|
-
if isinstance(storage_capabilities, dict)
|
|
619
|
-
else False
|
|
620
|
-
),
|
|
621
|
-
# v11.1.0: which in-process index scores a search, and — when
|
|
622
|
-
# the configured one could not be used — the reason it was
|
|
623
|
-
# substituted, so an unavailable optional extra is visible
|
|
624
|
-
# here instead of only showing up as "search feels slow".
|
|
625
|
-
"vector_index": self._vector_index_selection().as_dict(),
|
|
626
|
-
},
|
|
627
|
-
"source_items": len(source_items),
|
|
628
|
-
"indexed_items": sum(vector_counts.values()),
|
|
629
|
-
"ready_items": ready,
|
|
630
|
-
"missing_items": missing,
|
|
631
|
-
"stale_items": stale,
|
|
632
|
-
"pending_items": pending,
|
|
633
|
-
"by_item_type": vector_counts,
|
|
634
|
-
"scale": {
|
|
635
|
-
"version": 1,
|
|
636
|
-
"coverage_ratio": coverage_ratio,
|
|
637
|
-
"coverage_percent": round(coverage_ratio * 100.0, 2),
|
|
638
|
-
"source_items": len(source_items),
|
|
639
|
-
"ready_items": ready,
|
|
640
|
-
"pending_items": pending,
|
|
641
|
-
"missing_items": missing,
|
|
642
|
-
"stale_items": stale,
|
|
643
|
-
"orphaned_items": orphaned_items,
|
|
644
|
-
"backlog_by_item_type": backlog_by_type,
|
|
645
|
-
"backlog_reasons": backlog_reasons,
|
|
646
|
-
"backlog_samples": backlog_samples,
|
|
647
|
-
"incremental_reindex_recommended": pending > 0,
|
|
648
|
-
# A stale embedder means every old-model row must be re-embedded;
|
|
649
|
-
# only a full rebuild (which re-records the fingerprint) heals it.
|
|
650
|
-
"full_rebuild_recommended": bool(
|
|
651
|
-
orphaned_items > 0 or embedder_status["stale_embedder"]
|
|
652
|
-
),
|
|
653
|
-
"latency_budget": latency_budget,
|
|
654
|
-
},
|
|
655
|
-
"operations": [
|
|
656
|
-
{
|
|
657
|
-
"id": row["id"],
|
|
658
|
-
"operation": row["operation"],
|
|
659
|
-
"status": row["status"],
|
|
660
|
-
"requested_at": row["requested_at"],
|
|
661
|
-
"started_at": row["started_at"],
|
|
662
|
-
"completed_at": row["completed_at"],
|
|
663
|
-
"items_total": row["items_total"],
|
|
664
|
-
"items_indexed": row["items_indexed"],
|
|
665
|
-
"items_skipped": row["items_skipped"],
|
|
666
|
-
"error_message": row["error_message"],
|
|
667
|
-
"metadata": _safe_loads(row["metadata_json"]),
|
|
668
|
-
}
|
|
669
|
-
for row in latest_rows
|
|
670
|
-
],
|
|
671
|
-
}
|
|
672
|
-
|
|
673
|
-
def vector_freshness(self) -> Dict[str, Any]:
|
|
674
|
-
"""Compact vector-index freshness summary for API surfaces (v9.8.0).
|
|
675
|
-
|
|
676
|
-
Reduces :meth:`index_status` (``pending = missing + stale``) to the
|
|
677
|
-
fixed contract ``{"status", "pending_items", "total_items", "detail"}``
|
|
678
|
-
with ``status`` in ``ready`` / ``pending`` / ``stale_embedder`` /
|
|
679
|
-
``unavailable``. ``stale_embedder`` (review Wave 2.2) is reported only
|
|
680
|
-
when the recorded embedder fingerprint differs from the current
|
|
681
|
-
embedder AND rows indexed under the old model still exist — the index
|
|
682
|
-
needs a full rebuild, not an incremental sync.
|
|
683
|
-
|
|
684
|
-
Never raises: environments where the embedding provider or index
|
|
685
|
-
storage cannot be used report ``"unavailable"`` with the cause in
|
|
686
|
-
``detail`` instead of surfacing an exception to the API layer.
|
|
687
|
-
"""
|
|
688
|
-
try:
|
|
689
|
-
status = self.index_status()
|
|
690
|
-
except Exception as exc: # noqa: BLE001 — freshness must degrade, not fail
|
|
691
|
-
return {
|
|
692
|
-
"status": "unavailable",
|
|
693
|
-
"pending_items": 0,
|
|
694
|
-
"total_items": 0,
|
|
695
|
-
"detail": f"vector index status unavailable: {exc}",
|
|
696
|
-
}
|
|
697
|
-
return self._vector_freshness_summary(status)
|
|
698
|
-
|
|
699
|
-
def _vector_freshness_summary(self, status: Dict[str, Any]) -> Dict[str, Any]:
|
|
700
|
-
"""The freshness reduction of an already-read :meth:`index_status`.
|
|
701
|
-
|
|
702
|
-
Split out so :meth:`vector_freshness_breakdown` can report both shapes
|
|
703
|
-
from one index scan; ``index_status`` walks every source item, and
|
|
704
|
-
calling it twice to answer one question about freshness would double
|
|
705
|
-
the most expensive read in this module.
|
|
706
|
-
"""
|
|
707
|
-
pending = int(status.get("pending_items") or 0)
|
|
708
|
-
total = int(status.get("source_items") or 0)
|
|
709
|
-
embedder = status.get("embedder") or {}
|
|
710
|
-
if embedder.get("stale_embedder"):
|
|
711
|
-
old_model_rows = 0
|
|
712
|
-
try:
|
|
713
|
-
with self._connect() as conn:
|
|
714
|
-
old_model_rows = int(
|
|
715
|
-
conn.execute(
|
|
716
|
-
"SELECT COUNT(*) AS c FROM vector_embeddings "
|
|
717
|
-
"WHERE embedding_model<>? OR embedding_dim<>?",
|
|
718
|
-
(
|
|
719
|
-
self._embedding_model.model_id,
|
|
720
|
-
int(self._embedding_model.dim),
|
|
721
|
-
),
|
|
722
|
-
).fetchone()["c"]
|
|
723
|
-
)
|
|
724
|
-
except Exception: # noqa: BLE001 — keep the existing statuses on failure
|
|
725
|
-
old_model_rows = 0
|
|
726
|
-
if old_model_rows > 0:
|
|
727
|
-
recorded = embedder.get("recorded") or {}
|
|
728
|
-
return {
|
|
729
|
-
"status": "stale_embedder",
|
|
730
|
-
"pending_items": pending,
|
|
731
|
-
"total_items": total,
|
|
732
|
-
"detail": (
|
|
733
|
-
f"embedding model changed ({recorded.get('model_id')} → "
|
|
734
|
-
f"{self._embedding_model.model_id}); {old_model_rows} indexed "
|
|
735
|
-
"rows still use the previous model — run a full vector index rebuild"
|
|
736
|
-
),
|
|
737
|
-
}
|
|
738
|
-
if pending > 0:
|
|
739
|
-
return {
|
|
740
|
-
"status": "pending",
|
|
741
|
-
"pending_items": pending,
|
|
742
|
-
"total_items": total,
|
|
743
|
-
"detail": (
|
|
744
|
-
f"{pending} of {total} items are missing or stale in the vector index"
|
|
745
|
-
),
|
|
746
|
-
}
|
|
747
|
-
detail = (
|
|
748
|
-
"vector index is up to date"
|
|
749
|
-
if total
|
|
750
|
-
else "vector index is empty (no indexable items yet)"
|
|
751
|
-
)
|
|
752
|
-
return {
|
|
753
|
-
"status": "ready",
|
|
754
|
-
"pending_items": 0,
|
|
755
|
-
"total_items": total,
|
|
756
|
-
"detail": detail,
|
|
757
|
-
}
|
|
758
|
-
|
|
759
|
-
@property
|
|
760
|
-
def vector_queue(self) -> VectorEmbedQueue:
|
|
761
|
-
"""This store's durable pending-embed backlog (created on demand).
|
|
762
|
-
|
|
763
|
-
Built lazily rather than in ``__init__`` so opening a graph never
|
|
764
|
-
creates a table nobody asked for, and hung off the store so the
|
|
765
|
-
ingestion pipeline and the freshness report share one backlog instead
|
|
766
|
-
of each keeping a private view of it.
|
|
767
|
-
"""
|
|
768
|
-
queue = getattr(self, "_vector_queue", None)
|
|
769
|
-
if queue is None:
|
|
770
|
-
queue = VectorEmbedQueue(
|
|
771
|
-
db_path=self.db_path, indexer=self.index_node_incremental
|
|
772
|
-
)
|
|
773
|
-
self._vector_queue = queue
|
|
774
|
-
return queue
|
|
775
|
-
|
|
776
|
-
def vector_freshness_breakdown(self) -> Dict[str, Any]:
|
|
777
|
-
"""The four numbers behind :meth:`vector_freshness` (v11.1.0).
|
|
778
|
-
|
|
779
|
-
``vector_freshness()`` answers one question — *is the index behind?* —
|
|
780
|
-
and its four keys are a frozen wire contract that surfaces already
|
|
781
|
-
read, so this is a sibling rather than an extension of it. The split
|
|
782
|
-
matters because "12 pending" hides two different situations: twelve
|
|
783
|
-
items never embedded (a new import) and twelve items whose text
|
|
784
|
-
changed under an existing embedding (edits). Only the second means
|
|
785
|
-
current answers are quietly wrong.
|
|
786
|
-
|
|
787
|
-
``queued`` counts the durable background backlog
|
|
788
|
-
(:class:`~lattice_brain.graph.vector_index.VectorEmbedQueue`), and is
|
|
789
|
-
``None`` when that queue has no database to persist to — never ``0``,
|
|
790
|
-
which would claim an empty backlog nobody measured.
|
|
791
|
-
|
|
792
|
-
Never raises: an unreadable index reports ``status="unavailable"``
|
|
793
|
-
with the cause in ``detail`` and zeroed counts.
|
|
794
|
-
"""
|
|
795
|
-
status: Dict[str, Any] = {}
|
|
796
|
-
summary: Dict[str, Any]
|
|
797
|
-
try:
|
|
798
|
-
status = self.index_status()
|
|
799
|
-
except Exception as exc: # noqa: BLE001 — freshness must degrade, not fail
|
|
800
|
-
summary = {
|
|
801
|
-
"status": "unavailable",
|
|
802
|
-
"pending_items": 0,
|
|
803
|
-
"total_items": 0,
|
|
804
|
-
"detail": f"vector index status unavailable: {exc}",
|
|
805
|
-
}
|
|
806
|
-
else:
|
|
807
|
-
summary = self._vector_freshness_summary(status)
|
|
808
|
-
breakdown: Dict[str, Any] = {
|
|
809
|
-
"status": summary["status"],
|
|
810
|
-
"detail": summary["detail"],
|
|
811
|
-
"embedded": int(status.get("ready_items") or 0),
|
|
812
|
-
"pending": int(summary["pending_items"]),
|
|
813
|
-
"missing": int(status.get("missing_items") or 0),
|
|
814
|
-
"stale": int(status.get("stale_items") or 0),
|
|
815
|
-
"total": int(summary["total_items"]),
|
|
816
|
-
"queued": None,
|
|
817
|
-
}
|
|
818
|
-
queue = self.vector_queue
|
|
819
|
-
if queue.available:
|
|
820
|
-
breakdown["queued"] = int(queue.pending_count())
|
|
821
|
-
return breakdown
|
|
822
|
-
|
|
823
|
-
def _vector_index_selection(self) -> BackendSelection:
|
|
824
|
-
"""The configured in-process index backend (``LATTICEAI_VECTOR_INDEX``).
|
|
825
|
-
|
|
826
|
-
Resolved per call rather than cached: the env var is the whole control
|
|
827
|
-
surface, and a cached selection would make a config change look like
|
|
828
|
-
it had no effect. The only expensive part — importing ``hnswlib`` —
|
|
829
|
-
is already cached by ``sys.modules``.
|
|
830
|
-
"""
|
|
831
|
-
return resolve_vector_index()
|
|
832
|
-
|
|
833
|
-
def _vector_search_backend(self) -> str:
|
|
834
|
-
"""Which backend actually scores the vectors.
|
|
835
|
-
|
|
836
|
-
An explicitly selected in-process index (quantized / hnsw) wins,
|
|
837
|
-
because it is the thing that will do the scoring. Otherwise this is
|
|
838
|
-
the storage layer's answer: sqlite-vec exposes an ANN index; without
|
|
839
|
-
it this store scores rows in Python (``bruteforce-cosine``). Never
|
|
840
|
-
raises — a capability probe failure means "we cannot claim ANN",
|
|
841
|
-
which is the brute-force answer.
|
|
842
|
-
"""
|
|
843
|
-
selection = self._vector_index_selection()
|
|
844
|
-
if selection.name != DEFAULT_VECTOR_INDEX:
|
|
845
|
-
return selection.backend
|
|
846
|
-
try:
|
|
847
|
-
capabilities = self.storage_engine.capabilities().as_dict()
|
|
848
|
-
except Exception: # noqa: BLE001 — a probe failure is not an ANN index
|
|
849
|
-
return "bruteforce-cosine"
|
|
850
|
-
backend = (capabilities or {}).get("vector_backend")
|
|
851
|
-
return str(backend) if backend else "bruteforce-cosine"
|
|
852
|
-
|
|
853
|
-
@staticmethod
|
|
854
|
-
def _recall_report(
|
|
855
|
-
*,
|
|
856
|
-
backend: str,
|
|
857
|
-
cap: Optional[int],
|
|
858
|
-
candidates_total: int,
|
|
859
|
-
candidates_scanned: int,
|
|
860
|
-
approx_detail: Optional[str] = None,
|
|
861
|
-
) -> Dict[str, Any]:
|
|
862
|
-
"""The honest answer to "did this search see the whole index?".
|
|
863
|
-
|
|
864
|
-
``approx_detail`` covers the second way recall can be incomplete: an
|
|
865
|
-
ANN backend *visits* the whole index but is not guaranteed to return
|
|
866
|
-
its true top-k. "Scanned N of N" with no detail would read as an exact
|
|
867
|
-
answer, so the approximate backends supply their caveat here.
|
|
868
|
-
"""
|
|
869
|
-
truncated = candidates_scanned < candidates_total
|
|
870
|
-
detail: Optional[str] = None
|
|
871
|
-
if truncated:
|
|
872
|
-
detail = (
|
|
873
|
-
f"partial recall: scored the {candidates_scanned} most recently "
|
|
874
|
-
f"indexed vectors of {candidates_total}. The cut is by index "
|
|
875
|
-
f"recency, not similarity, so older matches were never compared. "
|
|
876
|
-
f"Raise {VECTOR_MAX_CANDIDATES_ENV} (0 = scan everything), or "
|
|
877
|
-
f"switch to an index that covers the whole set: "
|
|
878
|
-
f"{VECTOR_INDEX_ENV}=hnsw (needs the optional hnsw extra) or "
|
|
879
|
-
f"install sqlite-vec."
|
|
880
|
-
)
|
|
881
|
-
elif approx_detail:
|
|
882
|
-
detail = approx_detail
|
|
883
|
-
return {
|
|
884
|
-
"backend": backend,
|
|
885
|
-
"max_candidates": cap,
|
|
886
|
-
"candidates_total": candidates_total,
|
|
887
|
-
"candidates_scanned": candidates_scanned,
|
|
888
|
-
"truncated": truncated,
|
|
889
|
-
"detail": detail,
|
|
890
|
-
}
|
|
891
|
-
|
|
892
|
-
def _vector_candidate_cap(
|
|
893
|
-
self, requested: Optional[int], *, limit: int
|
|
894
|
-
) -> Optional[int]:
|
|
895
|
-
"""Resolve the effective candidate cap (None = scan everything).
|
|
896
|
-
|
|
897
|
-
``requested is None`` uses the configured/default cap; an explicit
|
|
898
|
-
``<= 0`` is the caller asking for an exhaustive scan. Note the
|
|
899
|
-
``is None`` test: ``0`` is a meaningful value here, so truthiness
|
|
900
|
-
would silently turn "no cap" into "the default cap".
|
|
901
|
-
"""
|
|
902
|
-
if requested is None:
|
|
903
|
-
cap = _configured_vector_max_candidates()
|
|
904
|
-
elif int(requested) <= 0:
|
|
905
|
-
cap = None
|
|
906
|
-
else:
|
|
907
|
-
cap = min(int(requested), VECTOR_MAX_CANDIDATES_CEILING)
|
|
908
|
-
if cap is None:
|
|
909
|
-
return None
|
|
910
|
-
# Never scan fewer rows than the caller intends to receive.
|
|
911
|
-
return max(limit, cap)
|
|
912
|
-
|
|
913
|
-
# One row shape feeds every vector match, so both the exact scan and the
|
|
914
|
-
# ANN lookup project exactly the same columns; only the WHERE/ORDER tail
|
|
915
|
-
# differs. Bound values are always parameters — the interpolation below is
|
|
916
|
-
# a placeholder list, never data.
|
|
917
|
-
_VECTOR_ROW_SELECT = """
|
|
918
|
-
SELECT
|
|
919
|
-
ve.item_id, ve.item_type, ve.source_node, ve.embedding,
|
|
920
|
-
ve.embedding_dim, ve.embedding_model, ve.metadata_json AS vector_metadata,
|
|
921
|
-
n.type AS node_type, n.title AS node_title, n.summary AS node_summary,
|
|
922
|
-
n.metadata_json AS node_metadata, n.updated_at AS node_updated_at,
|
|
923
|
-
c.text AS chunk_text, c.source_node AS parent_node_id,
|
|
924
|
-
c.metadata_json AS chunk_metadata,
|
|
925
|
-
pn.type AS parent_type, pn.title AS parent_title,
|
|
926
|
-
pn.summary AS parent_summary, pn.metadata_json AS parent_metadata,
|
|
927
|
-
pn.updated_at AS parent_updated_at
|
|
928
|
-
FROM vector_embeddings ve
|
|
929
|
-
LEFT JOIN nodes n ON n.id=ve.source_node
|
|
930
|
-
LEFT JOIN chunks c ON c.id=ve.item_id
|
|
931
|
-
LEFT JOIN nodes pn ON pn.id=c.source_node
|
|
932
|
-
WHERE ve.embedding_model=? AND ve.embedding_dim=?
|
|
933
|
-
"""
|
|
934
|
-
|
|
935
|
-
@staticmethod
|
|
936
|
-
def _vector_match(row: sqlite3.Row, score: float) -> Dict[str, Any]:
|
|
937
|
-
"""One scored embedding row → one search match (pure projection)."""
|
|
938
|
-
is_chunk = row["item_type"] == "chunk"
|
|
939
|
-
summary = (
|
|
940
|
-
row["chunk_text"] if is_chunk and row["chunk_text"] else row["node_summary"]
|
|
941
|
-
)
|
|
942
|
-
parent_metadata = _safe_loads(row["parent_metadata"])
|
|
943
|
-
node_metadata = _safe_loads(row["node_metadata"])
|
|
944
|
-
# Citation precision (review 2026-07-27 P1 #4): a chunk hit used to
|
|
945
|
-
# cite only its parent document, so a 200-page PDF answered with
|
|
946
|
-
# "from report.pdf". The chunk's own provenance (section heading,
|
|
947
|
-
# page, offset) now rides along, and `locator` is the one-line
|
|
948
|
-
# human form — absent when the chunk carries no such metadata.
|
|
949
|
-
chunk_metadata = _safe_loads(row["chunk_metadata"]) if is_chunk else {}
|
|
950
|
-
locator = citation_locator(chunk_metadata)
|
|
951
|
-
return {
|
|
952
|
-
"id": row["item_id"],
|
|
953
|
-
"node_id": row["parent_node_id"]
|
|
954
|
-
if is_chunk and row["parent_node_id"]
|
|
955
|
-
else row["source_node"],
|
|
956
|
-
"item_type": row["item_type"],
|
|
957
|
-
"type": "Chunk" if is_chunk else row["node_type"],
|
|
958
|
-
"title": row["parent_title"]
|
|
959
|
-
if is_chunk and row["parent_title"]
|
|
960
|
-
else row["node_title"],
|
|
961
|
-
"summary": _clean_text(summary or "")[:1000],
|
|
962
|
-
"score": round(float(score), 6),
|
|
963
|
-
"metadata": {
|
|
964
|
-
**(parent_metadata if is_chunk else node_metadata),
|
|
965
|
-
"vector": _safe_loads(row["vector_metadata"]),
|
|
966
|
-
"parent_node_id": row["parent_node_id"],
|
|
967
|
-
"parent_type": row["parent_type"],
|
|
968
|
-
**({"chunk": chunk_metadata} if chunk_metadata else {}),
|
|
969
|
-
**({"locator": locator} if locator else {}),
|
|
970
|
-
},
|
|
971
|
-
"updated_at": row["parent_updated_at"]
|
|
972
|
-
if is_chunk and row["parent_updated_at"]
|
|
973
|
-
else row["node_updated_at"],
|
|
974
|
-
}
|
|
975
|
-
|
|
976
|
-
@staticmethod
|
|
977
|
-
def _flush_scan_batch(
|
|
978
|
-
index: VectorIndex,
|
|
979
|
-
batch: List[IndexItem],
|
|
980
|
-
query_vector: List[float],
|
|
981
|
-
min_score: float,
|
|
982
|
-
scores: Dict[str, float],
|
|
983
|
-
) -> None:
|
|
984
|
-
"""Score one batch into ``scores`` and empty it."""
|
|
985
|
-
if not batch:
|
|
986
|
-
return
|
|
987
|
-
index.rebuild(batch)
|
|
988
|
-
scores.update(
|
|
989
|
-
index.search(query_vector, len(batch), filter={"min_score": min_score})
|
|
990
|
-
)
|
|
991
|
-
batch.clear()
|
|
992
|
-
|
|
993
|
-
def _score_vector_rows(
|
|
994
|
-
self,
|
|
995
|
-
rows: List[sqlite3.Row],
|
|
996
|
-
query_vector: List[float],
|
|
997
|
-
selection: BackendSelection,
|
|
998
|
-
*,
|
|
999
|
-
min_score: float,
|
|
1000
|
-
) -> Dict[str, float]:
|
|
1001
|
-
"""``item_id -> score`` for every row that clears ``min_score``."""
|
|
1002
|
-
index = build_index(
|
|
1003
|
-
selection,
|
|
1004
|
-
dim=int(self._embedding_model.dim),
|
|
1005
|
-
similarity=self._embedding_model.similarity,
|
|
1006
|
-
)
|
|
1007
|
-
scores: Dict[str, float] = {}
|
|
1008
|
-
batch: List[IndexItem] = []
|
|
1009
|
-
for row in rows:
|
|
1010
|
-
batch.append(
|
|
1011
|
-
(
|
|
1012
|
-
str(row["item_id"]),
|
|
1013
|
-
self._embedding_model.decode(
|
|
1014
|
-
row["embedding"], row["embedding_dim"]
|
|
1015
|
-
),
|
|
1016
|
-
{"item_type": row["item_type"]},
|
|
1017
|
-
)
|
|
1018
|
-
)
|
|
1019
|
-
if len(batch) >= VECTOR_SCAN_BATCH:
|
|
1020
|
-
self._flush_scan_batch(index, batch, query_vector, min_score, scores)
|
|
1021
|
-
self._flush_scan_batch(index, batch, query_vector, min_score, scores)
|
|
1022
|
-
return scores
|
|
1023
|
-
|
|
1024
|
-
def _vector_search_scan(
|
|
1025
|
-
self,
|
|
1026
|
-
query: str,
|
|
1027
|
-
query_vector: List[float],
|
|
1028
|
-
selection: BackendSelection,
|
|
1029
|
-
*,
|
|
1030
|
-
limit: int,
|
|
1031
|
-
min_score: float,
|
|
1032
|
-
backend: str,
|
|
1033
|
-
cap: Optional[int],
|
|
1034
|
-
) -> Dict[str, Any]:
|
|
1035
|
-
"""Exhaustive scan of (at most ``cap``) rows — the historical path."""
|
|
1036
|
-
sql = self._VECTOR_ROW_SELECT + " ORDER BY ve.indexed_at DESC"
|
|
1037
|
-
params: List[Any] = [
|
|
1038
|
-
self._embedding_model.model_id,
|
|
1039
|
-
self._embedding_model.dim,
|
|
1040
|
-
]
|
|
1041
|
-
if cap is not None:
|
|
1042
|
-
sql += " LIMIT ?"
|
|
1043
|
-
params.append(cap)
|
|
1044
|
-
with self._connect() as conn:
|
|
1045
|
-
# Counted in the same transaction as the scan so "scanned N of M"
|
|
1046
|
-
# cannot describe two different index states.
|
|
1047
|
-
candidates_total = int(
|
|
1048
|
-
conn.execute(
|
|
1049
|
-
"SELECT COUNT(*) AS c FROM vector_embeddings "
|
|
1050
|
-
"WHERE embedding_model=? AND embedding_dim=?",
|
|
1051
|
-
(self._embedding_model.model_id, self._embedding_model.dim),
|
|
1052
|
-
).fetchone()["c"]
|
|
1053
|
-
)
|
|
1054
|
-
rows = conn.execute(sql, tuple(params)).fetchall()
|
|
1055
|
-
recall = self._recall_report(
|
|
1056
|
-
backend=backend,
|
|
1057
|
-
cap=cap,
|
|
1058
|
-
candidates_total=candidates_total,
|
|
1059
|
-
candidates_scanned=len(rows),
|
|
1060
|
-
approx_detail=(
|
|
1061
|
-
"approximate backend: every candidate was compared, but the "
|
|
1062
|
-
"scores are estimates, so near-ties can reorder"
|
|
1063
|
-
if selection.approx
|
|
1064
|
-
else None
|
|
1065
|
-
),
|
|
1066
|
-
)
|
|
1067
|
-
scores = self._score_vector_rows(
|
|
1068
|
-
rows, query_vector, selection, min_score=min_score
|
|
1069
|
-
)
|
|
1070
|
-
# Rows are walked in index order (not score order) so the sort below
|
|
1071
|
-
# sees exactly the input ordering the pre-11.1.0 inline loop produced:
|
|
1072
|
-
# a stable sort makes that the tie-break of last resort.
|
|
1073
|
-
scored = [
|
|
1074
|
-
self._vector_match(row, scores[str(row["item_id"])])
|
|
1075
|
-
for row in rows
|
|
1076
|
-
if str(row["item_id"]) in scores
|
|
1077
|
-
]
|
|
1078
|
-
scored.sort(
|
|
1079
|
-
key=lambda item: (item["score"], item.get("updated_at") or ""), reverse=True
|
|
1080
|
-
)
|
|
1081
|
-
return {
|
|
1082
|
-
"query": query,
|
|
1083
|
-
"embedding_model": self._embedding_model.model_id,
|
|
1084
|
-
"embedding_dim": self._embedding_model.dim,
|
|
1085
|
-
"matches": scored[:limit],
|
|
1086
|
-
"recall": recall,
|
|
1087
|
-
"index": selection.as_dict(),
|
|
1088
|
-
}
|
|
1089
|
-
|
|
1090
|
-
def _iter_vector_index_items(
|
|
1091
|
-
self, conn: sqlite3.Connection, model_id: str, dim: int
|
|
1092
|
-
) -> Iterator[IndexItem]:
|
|
1093
|
-
"""Every embedding for ``model_id``/``dim`` as index items."""
|
|
1094
|
-
for row in conn.execute(
|
|
1095
|
-
"SELECT item_id, embedding, embedding_dim FROM vector_embeddings "
|
|
1096
|
-
"WHERE embedding_model=? AND embedding_dim=? ORDER BY item_id ASC",
|
|
1097
|
-
(model_id, dim),
|
|
1098
|
-
):
|
|
1099
|
-
yield (
|
|
1100
|
-
str(row["item_id"]),
|
|
1101
|
-
self._embedding_model.decode(row["embedding"], row["embedding_dim"]),
|
|
1102
|
-
{},
|
|
1103
|
-
)
|
|
1104
|
-
|
|
1105
|
-
def _vector_rows_by_id(
|
|
1106
|
-
self, conn: sqlite3.Connection, item_ids: List[str]
|
|
1107
|
-
) -> List[sqlite3.Row]:
|
|
1108
|
-
"""Full match rows for the ids an ANN lookup returned."""
|
|
1109
|
-
if not item_ids:
|
|
1110
|
-
return []
|
|
1111
|
-
placeholders = ",".join("?" * len(item_ids))
|
|
1112
|
-
return conn.execute(
|
|
1113
|
-
self._VECTOR_ROW_SELECT + f" AND ve.item_id IN ({placeholders})",
|
|
1114
|
-
(self._embedding_model.model_id, self._embedding_model.dim, *item_ids),
|
|
1115
|
-
).fetchall()
|
|
1116
|
-
|
|
1117
|
-
def _hnsw_index(
|
|
1118
|
-
self,
|
|
1119
|
-
conn: sqlite3.Connection,
|
|
1120
|
-
fingerprint: str,
|
|
1121
|
-
model_id: str,
|
|
1122
|
-
dim: int,
|
|
1123
|
-
) -> HnswIndex:
|
|
1124
|
-
"""The live ANN graph for ``fingerprint`` — cache, sidecar, or rebuild.
|
|
1125
|
-
|
|
1126
|
-
Held on the store for the process's lifetime, because reading a
|
|
1127
|
-
50 000-vector graph off disk costs roughly as much as the search it
|
|
1128
|
-
enables: paying it per query gave back most of the speedup (105 ms
|
|
1129
|
-
instead of 15 ms at 50k). The fingerprint — model, dimension, row
|
|
1130
|
-
count, newest ``indexed_at`` — is what makes the cache safe: any write
|
|
1131
|
-
to ``vector_embeddings`` changes it, and a changed fingerprint is
|
|
1132
|
-
never served from the cache or from the sidecar.
|
|
1133
|
-
"""
|
|
1134
|
-
cached = getattr(self, "_hnsw_cached", None)
|
|
1135
|
-
if cached is not None and cached[0] == fingerprint:
|
|
1136
|
-
return cached[1]
|
|
1137
|
-
index = HnswIndex(dim=dim)
|
|
1138
|
-
if not index.load(self.db_path, fingerprint=fingerprint):
|
|
1139
|
-
index.rebuild(self._iter_vector_index_items(conn, model_id, dim))
|
|
1140
|
-
index.save(self.db_path, fingerprint=fingerprint)
|
|
1141
|
-
self._hnsw_cached = (fingerprint, index)
|
|
1142
|
-
return index
|
|
1143
|
-
|
|
1144
|
-
def _vector_search_ann(
|
|
1145
|
-
self,
|
|
1146
|
-
query: str,
|
|
1147
|
-
query_vector: List[float],
|
|
1148
|
-
selection: BackendSelection,
|
|
1149
|
-
*,
|
|
1150
|
-
limit: int,
|
|
1151
|
-
min_score: float,
|
|
1152
|
-
backend: str,
|
|
1153
|
-
) -> Optional[Dict[str, Any]]:
|
|
1154
|
-
"""Approximate top-k via the persisted HNSW sidecar.
|
|
1155
|
-
|
|
1156
|
-
Two phases instead of one: ask the graph for ids, then read only those
|
|
1157
|
-
rows. That is where the speed comes from — the exact scan pays to
|
|
1158
|
-
decode every embedding on every query, and this pays it once per
|
|
1159
|
-
index generation.
|
|
1160
|
-
|
|
1161
|
-
The sidecar is keyed by ``model:dim:rows:newest`` so any write to
|
|
1162
|
-
``vector_embeddings`` invalidates it and the next search rebuilds.
|
|
1163
|
-
Returns ``None`` when the index is empty, which the caller answers
|
|
1164
|
-
with the ordinary (equally empty, but honestly reported) scan.
|
|
1165
|
-
"""
|
|
1166
|
-
model_id = self._embedding_model.model_id
|
|
1167
|
-
dim = int(self._embedding_model.dim)
|
|
1168
|
-
with self._connect() as conn:
|
|
1169
|
-
head = conn.execute(
|
|
1170
|
-
"SELECT COUNT(*) AS c, MAX(indexed_at) AS newest FROM vector_embeddings "
|
|
1171
|
-
"WHERE embedding_model=? AND embedding_dim=?",
|
|
1172
|
-
(model_id, dim),
|
|
1173
|
-
).fetchone()
|
|
1174
|
-
candidates_total = int(head["c"])
|
|
1175
|
-
if candidates_total == 0:
|
|
1176
|
-
return None
|
|
1177
|
-
fingerprint = f"{model_id}:{dim}:{candidates_total}:{head['newest']}"
|
|
1178
|
-
index = self._hnsw_index(conn, fingerprint, model_id, dim)
|
|
1179
|
-
pairs = index.search(query_vector, limit, filter={"min_score": min_score})
|
|
1180
|
-
rows = {
|
|
1181
|
-
str(row["item_id"]): row
|
|
1182
|
-
for row in self._vector_rows_by_id(
|
|
1183
|
-
conn, [item_id for item_id, _ in pairs]
|
|
1184
|
-
)
|
|
1185
|
-
}
|
|
1186
|
-
scored = [
|
|
1187
|
-
self._vector_match(rows[item_id], score)
|
|
1188
|
-
for item_id, score in pairs
|
|
1189
|
-
if item_id in rows
|
|
1190
|
-
]
|
|
1191
|
-
scored.sort(
|
|
1192
|
-
key=lambda item: (item["score"], item.get("updated_at") or ""), reverse=True
|
|
1193
|
-
)
|
|
1194
|
-
return {
|
|
1195
|
-
"query": query,
|
|
1196
|
-
"embedding_model": model_id,
|
|
1197
|
-
"embedding_dim": dim,
|
|
1198
|
-
"matches": scored[:limit],
|
|
1199
|
-
"recall": self._recall_report(
|
|
1200
|
-
backend=backend,
|
|
1201
|
-
cap=None,
|
|
1202
|
-
candidates_total=candidates_total,
|
|
1203
|
-
candidates_scanned=candidates_total,
|
|
1204
|
-
approx_detail=(
|
|
1205
|
-
"approximate nearest-neighbour search: the whole index is "
|
|
1206
|
-
"reachable but the true top-k is not guaranteed — compare "
|
|
1207
|
-
"with scripts/bench_vector_index.py"
|
|
1208
|
-
),
|
|
1209
|
-
),
|
|
1210
|
-
"index": {**selection.as_dict(), "sidecar": index.loaded_from_sidecar},
|
|
1211
|
-
}
|
|
1212
|
-
|
|
1213
|
-
def vector_search(
|
|
1214
|
-
self,
|
|
1215
|
-
query: str,
|
|
1216
|
-
*,
|
|
1217
|
-
limit: int = 30,
|
|
1218
|
-
min_score: float = 0.0,
|
|
1219
|
-
max_candidates: Optional[int] = None,
|
|
1220
|
-
) -> Dict[str, Any]:
|
|
1221
|
-
"""Cosine search over the vector index (exact by default).
|
|
1222
|
-
|
|
1223
|
-
``max_candidates`` bounds how many indexed rows are scored; ``None``
|
|
1224
|
-
(the default) resolves it from ``LATTICEAI_VECTOR_MAX_CANDIDATES``
|
|
1225
|
-
(default 10 000), and ``0`` or a negative value scans the whole index.
|
|
1226
|
-
When the cap bites, the rows kept are the most recently indexed ones —
|
|
1227
|
-
recency, not similarity — so the result is *partial recall*. That is
|
|
1228
|
-
reported in the additive ``recall`` block
|
|
1229
|
-
(``{backend, max_candidates, candidates_total, candidates_scanned,
|
|
1230
|
-
truncated, detail}``) instead of being hidden, and callers/UIs are
|
|
1231
|
-
expected to surface ``recall.truncated``.
|
|
1232
|
-
|
|
1233
|
-
v11.1.0: the scoring itself now lives in
|
|
1234
|
-
:mod:`lattice_brain.graph.vector_index`. ``LATTICEAI_VECTOR_INDEX``
|
|
1235
|
-
picks the backend — ``brute`` (default, exact, byte-compatible with
|
|
1236
|
-
every previous release), ``quantized`` (int8, exhaustive, approximate
|
|
1237
|
-
scores) or ``hnsw`` (approximate nearest neighbour, needs the optional
|
|
1238
|
-
``hnsw`` extra). The resolved backend and any fallback reason ride
|
|
1239
|
-
along in the additive ``index`` block, whose ``approx`` flag is the
|
|
1240
|
-
one bit a caller needs to know whether "not found" is a fact or an
|
|
1241
|
-
estimate. The empty-query early return is deliberately unchanged: no
|
|
1242
|
-
query means no index was consulted, so there is nothing to report.
|
|
1243
|
-
"""
|
|
1244
|
-
query = str(query or "").strip()
|
|
1245
|
-
limit = max(1, min(int(limit or 30), 100))
|
|
1246
|
-
min_score = float(min_score or 0.0)
|
|
1247
|
-
cap = self._vector_candidate_cap(max_candidates, limit=limit)
|
|
1248
|
-
backend = self._vector_search_backend()
|
|
1249
|
-
if not query:
|
|
1250
|
-
return {
|
|
1251
|
-
"query": query,
|
|
1252
|
-
"matches": [],
|
|
1253
|
-
"recall": {
|
|
1254
|
-
"backend": backend,
|
|
1255
|
-
"max_candidates": cap,
|
|
1256
|
-
"candidates_total": 0,
|
|
1257
|
-
"candidates_scanned": 0,
|
|
1258
|
-
"truncated": False,
|
|
1259
|
-
"detail": None,
|
|
1260
|
-
},
|
|
1261
|
-
}
|
|
1262
|
-
selection = self._vector_index_selection()
|
|
1263
|
-
query_vector = self._embedding_model.embed(query)
|
|
1264
|
-
if selection.name == HNSW_BACKEND:
|
|
1265
|
-
try:
|
|
1266
|
-
approximate = self._vector_search_ann(
|
|
1267
|
-
query,
|
|
1268
|
-
query_vector,
|
|
1269
|
-
selection,
|
|
1270
|
-
limit=limit,
|
|
1271
|
-
min_score=min_score,
|
|
1272
|
-
backend=backend,
|
|
1273
|
-
)
|
|
1274
|
-
except Exception as exc: # noqa: BLE001 — a broken ANN must not lose the answer
|
|
1275
|
-
logging.warning("hnsw vector search failed: %s", exc)
|
|
1276
|
-
selection = dataclasses.replace(
|
|
1277
|
-
resolve_vector_index(DEFAULT_VECTOR_INDEX),
|
|
1278
|
-
requested=HNSW_BACKEND,
|
|
1279
|
-
detail=f"hnsw search failed ({exc}); used the exact scan instead",
|
|
1280
|
-
)
|
|
1281
|
-
backend = selection.backend
|
|
1282
|
-
else:
|
|
1283
|
-
if approximate is not None:
|
|
1284
|
-
return approximate
|
|
1285
|
-
return self._vector_search_scan(
|
|
1286
|
-
query,
|
|
1287
|
-
query_vector,
|
|
1288
|
-
selection,
|
|
1289
|
-
limit=limit,
|
|
1290
|
-
min_score=min_score,
|
|
1291
|
-
backend=backend,
|
|
1292
|
-
cap=cap,
|
|
1293
|
-
)
|