ltcai 11.2.0 → 11.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -53
- package/docs/CHANGELOG.md +61 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/MULTI_AGENT_RUNTIME.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +6 -2
- package/docs/PERMISSION_MODE.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +2 -2
- package/docs/v11.3.0_PLAN.md +202 -0
- package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +287 -0
- package/lattice_brain/graph/_kg_common/extraction.py +516 -0
- package/lattice_brain/graph/_kg_common/relations.py +161 -0
- package/lattice_brain/graph/_kg_common/text.py +479 -0
- package/lattice_brain/graph/discovery_index/__init__.py +35 -0
- package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
- package/lattice_brain/graph/discovery_index/extract.py +137 -0
- package/lattice_brain/graph/discovery_index/scan.py +411 -0
- package/lattice_brain/graph/discovery_index/upsert.py +495 -0
- package/lattice_brain/graph/projection/__init__.py +42 -0
- package/lattice_brain/graph/projection/curation.py +500 -0
- package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
- package/lattice_brain/graph/retrieval/__init__.py +54 -0
- package/lattice_brain/graph/retrieval/context.py +197 -0
- package/lattice_brain/graph/retrieval/graph_view.py +319 -0
- package/lattice_brain/graph/retrieval/hybrid.py +488 -0
- package/lattice_brain/graph/retrieval/maintenance.py +121 -0
- package/lattice_brain/graph/retrieval/signals.py +95 -0
- package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
- package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
- package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
- package/lattice_brain/graph/retrieval_vector/search.py +560 -0
- package/lattice_brain/graph/retrieval_vector/status.py +374 -0
- package/lattice_brain/ingestion/__init__.py +130 -0
- package/lattice_brain/ingestion/_contract.py +90 -0
- package/lattice_brain/ingestion/constants.py +127 -0
- package/lattice_brain/ingestion/folder_scan.py +57 -0
- package/lattice_brain/ingestion/folders.py +258 -0
- package/lattice_brain/ingestion/hashing.py +26 -0
- package/lattice_brain/ingestion/jobs_api.py +107 -0
- package/lattice_brain/ingestion/models.py +80 -0
- package/lattice_brain/ingestion/pipeline.py +486 -0
- package/lattice_brain/ingestion/quality.py +209 -0
- package/lattice_brain/ingestion/routing.py +295 -0
- package/lattice_brain/multimodal/__init__.py +164 -0
- package/lattice_brain/multimodal/audio.py +77 -0
- package/lattice_brain/multimodal/common.py +118 -0
- package/lattice_brain/multimodal/images.py +498 -0
- package/lattice_brain/multimodal/ports.py +169 -0
- package/lattice_brain/multimodal/video.py +410 -0
- package/lattice_brain/portability/__init__.py +90 -0
- package/lattice_brain/portability/_contract.py +42 -0
- package/lattice_brain/portability/backups.py +338 -0
- package/lattice_brain/portability/bundles.py +136 -0
- package/lattice_brain/portability/constants.py +93 -0
- package/lattice_brain/portability/fsops.py +138 -0
- package/lattice_brain/portability/service.py +41 -0
- package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
- package/lattice_brain/runtime/__init__.py +1 -1
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/chronicle.py +63 -0
- package/latticeai/core/agent/__init__.py +93 -0
- package/latticeai/core/agent/_contract.py +79 -0
- package/latticeai/core/agent/context.py +57 -0
- package/latticeai/core/agent/deps.py +125 -0
- package/latticeai/core/agent/execution.py +622 -0
- package/latticeai/core/agent/planning.py +145 -0
- package/latticeai/core/agent/recovery.py +157 -0
- package/latticeai/core/agent/runtime.py +210 -0
- package/latticeai/core/agent/verification.py +231 -0
- package/latticeai/core/embedding_providers/__init__.py +151 -0
- package/latticeai/core/embedding_providers/base.py +199 -0
- package/latticeai/core/embedding_providers/captions.py +162 -0
- package/latticeai/core/embedding_providers/profiles.py +126 -0
- package/latticeai/core/embedding_providers/text.py +350 -0
- package/latticeai/core/embedding_providers/vision.py +352 -0
- package/latticeai/core/file_generation/__init__.py +115 -0
- package/latticeai/core/file_generation/bundles.py +76 -0
- package/latticeai/core/file_generation/extraction.py +154 -0
- package/latticeai/core/file_generation/inference.py +235 -0
- package/latticeai/core/file_generation/orchestration.py +152 -0
- package/latticeai/core/file_generation/prompting.py +117 -0
- package/latticeai/core/file_generation/repair.py +114 -0
- package/latticeai/core/file_generation/sanitize.py +61 -0
- package/latticeai/core/file_generation/validation.py +201 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +9 -0
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/integrations/telegram_bot/__init__.py +123 -0
- package/latticeai/integrations/telegram_bot/__main__.py +17 -0
- package/latticeai/integrations/telegram_bot/config.py +86 -0
- package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
- package/latticeai/integrations/telegram_bot/flows.py +478 -0
- package/latticeai/integrations/telegram_bot/helpers.py +322 -0
- package/latticeai/integrations/telegram_bot/screens.py +394 -0
- package/latticeai/models/router/__init__.py +88 -0
- package/latticeai/models/router/_contract.py +66 -0
- package/latticeai/models/router/branding.py +56 -0
- package/latticeai/models/router/catalog.py +69 -0
- package/latticeai/models/router/documents.py +199 -0
- package/latticeai/models/router/errors.py +37 -0
- package/latticeai/models/router/generation.py +258 -0
- package/latticeai/models/router/loading.py +291 -0
- package/latticeai/models/router/local_models.py +85 -0
- package/latticeai/models/router/registry.py +147 -0
- package/latticeai/runtime/build_phases/__init__.py +82 -0
- package/latticeai/runtime/build_phases/features.py +407 -0
- package/latticeai/runtime/build_phases/foundation.py +555 -0
- package/latticeai/runtime/build_phases/web.py +492 -0
- package/latticeai/runtime/runtime_context.py +1 -0
- package/latticeai/services/architecture_readiness.py +48 -19
- package/latticeai/services/brain_intelligence/__init__.py +58 -0
- package/latticeai/services/brain_intelligence/_contract.py +71 -0
- package/latticeai/services/brain_intelligence/consistency.py +193 -0
- package/latticeai/services/brain_intelligence/constants.py +47 -0
- package/latticeai/services/brain_intelligence/digest.py +258 -0
- package/latticeai/services/brain_intelligence/health.py +331 -0
- package/latticeai/services/brain_intelligence/proposals.py +264 -0
- package/latticeai/services/brain_intelligence/sampling.py +84 -0
- package/latticeai/services/brain_intelligence/service.py +48 -0
- package/latticeai/services/chronicle.py +557 -0
- package/latticeai/services/memory_service/__init__.py +52 -0
- package/latticeai/services/memory_service/_contract.py +100 -0
- package/latticeai/services/memory_service/brief.py +431 -0
- package/latticeai/services/memory_service/constants.py +57 -0
- package/latticeai/services/memory_service/maintenance.py +138 -0
- package/latticeai/services/memory_service/manager.py +186 -0
- package/latticeai/services/memory_service/proof.py +136 -0
- package/latticeai/services/memory_service/recall.py +225 -0
- package/latticeai/services/memory_service/service.py +48 -0
- package/latticeai/services/memory_service/stores.py +110 -0
- package/latticeai/services/model_runtime/__init__.py +322 -0
- package/latticeai/services/model_runtime/cloud.py +87 -0
- package/latticeai/services/model_runtime/download.py +282 -0
- package/latticeai/services/model_runtime/engines.py +341 -0
- package/latticeai/services/model_runtime/loading.py +178 -0
- package/latticeai/services/model_runtime/service.py +129 -0
- package/latticeai/services/model_runtime/state.py +131 -0
- package/latticeai/services/model_runtime/status.py +255 -0
- package/latticeai/services/product_readiness.py +15 -7
- package/latticeai/setup/wizard/__init__.py +126 -0
- package/latticeai/setup/wizard/catalog.py +172 -0
- package/latticeai/setup/wizard/detect.py +323 -0
- package/latticeai/setup/wizard/install.py +348 -0
- package/latticeai/setup/wizard/paths.py +168 -0
- package/latticeai/setup/wizard/plans.py +74 -0
- package/latticeai/setup/wizard/recommend.py +320 -0
- package/package.json +6 -2
- package/scripts/bump_version.py +14 -0
- package/scripts/capture_release_evidence.mjs +33 -21
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_i18n_namespace_coverage.mjs +41 -4
- package/scripts/check_max_file_lines.mjs +102 -0
- package/scripts/check_release_evidence_bound.mjs +30 -15
- package/scripts/check_screenshot_pixel_delta.py +34 -4
- package/scripts/check_server_i18n.mjs +1 -0
- package/scripts/generate_rust_parity_fixtures.py +562 -0
- package/scripts/lib/mock_server_fingerprint.mjs +94 -0
- package/scripts/release_screen_claims.json +31 -2
- package/src-tauri/Cargo.lock +361 -3
- package/src-tauri/Cargo.toml +6 -1
- package/src-tauri/src/backend.rs +349 -0
- package/src-tauri/src/folder.rs +33 -0
- package/src-tauri/src/main.rs +97 -399
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +41 -37
- package/static/app/assets/Act-yYpYnn0v.js +1 -0
- package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
- package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
- package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
- package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
- package/static/app/assets/Capture-CFIRsFNE.js +1 -0
- package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
- package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
- package/static/app/assets/Library-DwO3yZST.js +1 -0
- package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
- package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
- package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
- package/static/app/assets/System-DW8F-2xL.js +1 -0
- package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
- package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
- package/static/app/assets/brain-Ci1CkWjM.js +1 -0
- package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
- package/static/app/assets/circle-check-DfInj-qD.js +1 -0
- package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
- package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
- package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
- package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
- package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
- package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
- package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
- package/static/app/assets/index-_u5iUHDr.js +10 -0
- package/static/app/assets/input-B0lPdRQZ.js +1 -0
- package/static/app/assets/link-2-CoFbooHS.js +1 -0
- package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
- package/static/app/assets/primitives-DEbN-d6p.js +1 -0
- package/static/app/assets/search-BybIWPNd.js +1 -0
- package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
- package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
- package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
- package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
- package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
- package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
- package/static/app/assets/utils-BlZr7Pd4.js +4 -0
- package/static/app/assets/workspace-jJY4RuAV.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/graph/_kg_common.py +0 -1331
- package/lattice_brain/graph/discovery_index.py +0 -1141
- package/lattice_brain/graph/retrieval.py +0 -1120
- package/lattice_brain/graph/retrieval_vector.py +0 -1293
- package/lattice_brain/ingestion.py +0 -1525
- package/lattice_brain/multimodal.py +0 -1258
- package/latticeai/core/agent.py +0 -1465
- package/latticeai/core/embedding_providers.py +0 -1196
- package/latticeai/core/file_generation.py +0 -1047
- package/latticeai/integrations/telegram_bot.py +0 -1390
- package/latticeai/models/router.py +0 -1007
- package/latticeai/runtime/build_phases.py +0 -1450
- package/latticeai/services/brain_intelligence.py +0 -1083
- package/latticeai/services/memory_service.py +0 -1177
- package/latticeai/services/model_runtime.py +0 -1281
- package/latticeai/setup/wizard.py +0 -1310
- package/static/app/assets/Act-AWf0SAKp.js +0 -1
- package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
- package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
- package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
- package/static/app/assets/Capture-CqOSzyPr.js +0 -1
- package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
- package/static/app/assets/Library-CX-bbhmK.js +0 -1
- package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
- package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
- package/static/app/assets/System-Bu2t5hn1.js +0 -1
- package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
- package/static/app/assets/brain-DJMoqrwx.js +0 -1
- package/static/app/assets/index-BpYkzcVm.js +0 -10
- package/static/app/assets/input-DSlJJxRs.js +0 -1
- package/static/app/assets/primitives-BCx6TvfG.js +0 -1
- package/static/app/assets/search-Cgy8cCFJ.js +0 -1
- package/static/app/assets/utils-zqPZJxdx.js +0 -4
- package/static/app/assets/workspace-DXTihhfU.js +0 -1
|
@@ -0,0 +1,488 @@
|
|
|
1
|
+
"""The hybrid retrieval pipeline: keyword + vector + graph expansion.
|
|
2
|
+
|
|
3
|
+
One ranking algorithm, kept in one file on purpose — the standing reason
|
|
4
|
+
recorded in ``pyproject.toml`` for this file's complexity ignores is that
|
|
5
|
+
splitting ``hybrid_search`` across files would scatter the pipeline without
|
|
6
|
+
making it clearer. Moved verbatim out of ``retrieval.py`` (v11.3.0).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from typing import TYPE_CHECKING, Sequence
|
|
12
|
+
|
|
13
|
+
# C901: `hybrid_search` is one ranking algorithm at complexity 50. The standing
|
|
14
|
+
# reason recorded for it in pyproject.toml is that splitting the pipeline across
|
|
15
|
+
# files would scatter it without making it clearer; the ignore rides with the
|
|
16
|
+
# file now that the file is the pipeline.
|
|
17
|
+
# ruff: noqa: C901,F403,F405
|
|
18
|
+
from .._kg_common import * # noqa: F403,F401
|
|
19
|
+
from ..fusion import (
|
|
20
|
+
DEFAULT_EXPANSION_CAP,
|
|
21
|
+
DEFAULT_EXPANSION_SEEDS,
|
|
22
|
+
expand_with_neighbors,
|
|
23
|
+
graph_expansion_enabled,
|
|
24
|
+
rrf_fuse,
|
|
25
|
+
)
|
|
26
|
+
from .signals import multimodal_signal
|
|
27
|
+
|
|
28
|
+
# The cross-mixin surface (`_connect`, `_upsert_node`, …) is declared in
|
|
29
|
+
# `_kg_contract.KnowledgeGraphCore`. It is a typing-only base: at runtime this
|
|
30
|
+
# is `object`, so the MRO of `KnowledgeGraphStore` is unchanged. The alias here
|
|
31
|
+
# reaches one step further: this half calls `self.search`, which the sibling
|
|
32
|
+
# half in .graph_view owns and the composed mixin puts on the same instance.
|
|
33
|
+
# Naming that sibling as the typing base states the assumption instead of
|
|
34
|
+
# re-declaring its signature where it could drift.
|
|
35
|
+
if TYPE_CHECKING:
|
|
36
|
+
from .graph_view import _GraphViewMixin as _Core
|
|
37
|
+
else:
|
|
38
|
+
_Core = object
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class _HybridSearchMixin(_Core):
|
|
42
|
+
"""Fused keyword/vector/graph retrieval. Composed into the public mixin."""
|
|
43
|
+
|
|
44
|
+
def hybrid_search(
|
|
45
|
+
self,
|
|
46
|
+
query: str,
|
|
47
|
+
*,
|
|
48
|
+
top_k: int = 20,
|
|
49
|
+
alpha: Optional[float] = None,
|
|
50
|
+
workspace_id: Optional[str] = None,
|
|
51
|
+
allowed_workspaces=None,
|
|
52
|
+
include_legacy_global: bool = False,
|
|
53
|
+
lexical_limit: Optional[int] = None,
|
|
54
|
+
vector_limit: Optional[int] = None,
|
|
55
|
+
min_vector_score: float = 0.0,
|
|
56
|
+
image_vector: Optional[Sequence[float]] = None,
|
|
57
|
+
image_fusion_weight: Optional[float] = None,
|
|
58
|
+
) -> Dict[str, Any]:
|
|
59
|
+
"""Unified lexical + vector retrieval with alpha-weighted linear fusion.
|
|
60
|
+
|
|
61
|
+
Runs the SQLite lexical :meth:`search` and the embedding-backed
|
|
62
|
+
``vector_search`` (sibling mixin via the store MRO), normalizes both
|
|
63
|
+
score spaces to ``[0, 1]``, fuses them as
|
|
64
|
+
``alpha * vector + (1 - alpha) * lexical`` (the same shape as
|
|
65
|
+
``lattice_brain.quality.HybridFusion`` — reimplemented here without
|
|
66
|
+
importing that module), and dedupes by ``node_id`` (chunk hits roll up
|
|
67
|
+
to their parent node).
|
|
68
|
+
|
|
69
|
+
Degrades gracefully: when the vector side is unavailable (mixin not
|
|
70
|
+
composed, embedder/index failure) the result falls back to
|
|
71
|
+
lexical-only ranking and reports ``mode: "lexical_only"`` with a
|
|
72
|
+
``detail`` explaining why. Each match carries per-source ``scores``
|
|
73
|
+
and a ``fusion`` field (``lexical`` / ``vector`` / ``both``).
|
|
74
|
+
|
|
75
|
+
``workspace_id`` is a convenience for single-workspace callers; the
|
|
76
|
+
richer ``allowed_workspaces`` set wins when both are provided.
|
|
77
|
+
|
|
78
|
+
``alpha=None`` (the default) resolves the vector share from the
|
|
79
|
+
single retrieval policy (:mod:`lattice_brain.graph.retrieval_policy`,
|
|
80
|
+
which wraps the query-class fusion table): fact 0.6 (the historical
|
|
81
|
+
default) / code 0.35 / person 0.45 / recency 0.5, config-overridable
|
|
82
|
+
via ``LATTICEAI_FUSION_WEIGHTS``. The policy also supplies a
|
|
83
|
+
deterministic rule-based query rewrite (echoed additively under
|
|
84
|
+
``"policy"``; the response ``"query"`` stays the original) and, for
|
|
85
|
+
the ``recency`` class only, an age-decay half-life that dampens each
|
|
86
|
+
fused score into the ``[0.5, 1.0]`` band (``scores.age_decay``).
|
|
87
|
+
Passing an explicit ``alpha`` pins it exactly as before and disables
|
|
88
|
+
rewrite + decay.
|
|
89
|
+
|
|
90
|
+
``image_vector`` (v11.1.0) is the *late fusion* seam for the separate
|
|
91
|
+
image space: the caller supplies a query vector from the same vision
|
|
92
|
+
model that embedded the pictures, its own index is ranked
|
|
93
|
+
independently, and only then are the two rankings blended
|
|
94
|
+
(``image_fusion_weight``, default 0.5). A text query never produces
|
|
95
|
+
one — it reaches images through their OCR text and captions — which is
|
|
96
|
+
exactly why the image channel has to enter at the end rather than
|
|
97
|
+
pretending to share the text index.
|
|
98
|
+
"""
|
|
99
|
+
query = str(query or "").strip()
|
|
100
|
+
try:
|
|
101
|
+
top_k = int(top_k)
|
|
102
|
+
except (TypeError, ValueError):
|
|
103
|
+
top_k = 20
|
|
104
|
+
top_k = max(1, min(top_k, 100))
|
|
105
|
+
query_class: Optional[str] = None
|
|
106
|
+
search_query = query
|
|
107
|
+
rewrite_rules: List[str] = []
|
|
108
|
+
recency_half_life_days: Optional[float] = None
|
|
109
|
+
# "alpha" is the historical linear fusion; the policy may select RRF
|
|
110
|
+
# per query class. An explicitly pinned ``alpha`` argument means the
|
|
111
|
+
# caller is asking for linear fusion by name, so it stays linear.
|
|
112
|
+
fusion_strategy = "alpha"
|
|
113
|
+
if alpha is None:
|
|
114
|
+
try:
|
|
115
|
+
from ..retrieval_policy import resolve_policy
|
|
116
|
+
|
|
117
|
+
policy = resolve_policy(query)
|
|
118
|
+
query_class = policy["query_class"]
|
|
119
|
+
alpha = float(policy["alpha"])
|
|
120
|
+
fusion_strategy = str(policy.get("fusion_strategy") or "alpha")
|
|
121
|
+
rewrite_rules = list(policy.get("rewrite_rules") or [])
|
|
122
|
+
rewritten = str(policy.get("search_query") or "")
|
|
123
|
+
if rewritten and rewritten != query:
|
|
124
|
+
search_query = rewritten
|
|
125
|
+
half_life = policy.get("recency_half_life_days")
|
|
126
|
+
if half_life is not None:
|
|
127
|
+
recency_half_life_days = float(half_life)
|
|
128
|
+
except Exception: # noqa: BLE001 — policy resolution must never break search
|
|
129
|
+
alpha = 0.6
|
|
130
|
+
try:
|
|
131
|
+
alpha = float(alpha)
|
|
132
|
+
except (TypeError, ValueError):
|
|
133
|
+
alpha = 0.6
|
|
134
|
+
alpha = max(0.0, min(alpha, 1.0))
|
|
135
|
+
if allowed_workspaces is None and workspace_id:
|
|
136
|
+
allowed_workspaces = {str(workspace_id)}
|
|
137
|
+
|
|
138
|
+
if not query:
|
|
139
|
+
return {
|
|
140
|
+
"query": query,
|
|
141
|
+
"mode": "hybrid",
|
|
142
|
+
"alpha": alpha,
|
|
143
|
+
"query_class": query_class,
|
|
144
|
+
"top_k": top_k,
|
|
145
|
+
"sources": {"lexical": 0, "vector": 0},
|
|
146
|
+
"matches": [],
|
|
147
|
+
"policy": {"search_query": search_query, "rewrite_rules": rewrite_rules},
|
|
148
|
+
"fusion_strategy": fusion_strategy,
|
|
149
|
+
"detail": None,
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
lex_fetch = max(1, min(int(lexical_limit or max(top_k * 2, 20)), 100))
|
|
153
|
+
vec_fetch = max(1, min(int(vector_limit or max(top_k * 2, 20)), 100))
|
|
154
|
+
|
|
155
|
+
lexical_matches = self.search(
|
|
156
|
+
search_query,
|
|
157
|
+
lex_fetch,
|
|
158
|
+
allowed_workspaces=allowed_workspaces,
|
|
159
|
+
include_legacy_global=include_legacy_global,
|
|
160
|
+
).get("matches", [])
|
|
161
|
+
|
|
162
|
+
mode = "hybrid"
|
|
163
|
+
detail: Optional[str] = None
|
|
164
|
+
vector_matches: List[Dict[str, Any]] = []
|
|
165
|
+
vector_recall: Optional[Dict[str, Any]] = None
|
|
166
|
+
# The vector channel's own honesty block, echoed additively so a
|
|
167
|
+
# caller can tell an exact "not found" from an approximate one.
|
|
168
|
+
vector_meta: Dict[str, Any] = {
|
|
169
|
+
"backend": None,
|
|
170
|
+
"approx": None,
|
|
171
|
+
"exhaustive": None,
|
|
172
|
+
"truncated": None,
|
|
173
|
+
"embedded_rows": None,
|
|
174
|
+
"degraded": None,
|
|
175
|
+
}
|
|
176
|
+
vector_fn = getattr(self, "vector_search", None)
|
|
177
|
+
if not callable(vector_fn):
|
|
178
|
+
mode = "lexical_only"
|
|
179
|
+
detail = "vector search is not available on this store"
|
|
180
|
+
else:
|
|
181
|
+
try:
|
|
182
|
+
vector_payload = (
|
|
183
|
+
vector_fn(search_query, limit=vec_fetch, min_score=min_vector_score)
|
|
184
|
+
or {}
|
|
185
|
+
)
|
|
186
|
+
vector_matches = list(vector_payload.get("matches", []))
|
|
187
|
+
# Partial recall must reach the caller: the vector channel can
|
|
188
|
+
# only score a capped slice of a large index (see
|
|
189
|
+
# retrieval_vector.vector_search), and a fused answer built on
|
|
190
|
+
# a truncated scan is not the same claim as a complete one.
|
|
191
|
+
recall = vector_payload.get("recall")
|
|
192
|
+
if isinstance(recall, dict):
|
|
193
|
+
vector_meta["backend"] = recall.get("backend")
|
|
194
|
+
vector_meta["truncated"] = bool(recall.get("truncated"))
|
|
195
|
+
vector_meta["embedded_rows"] = recall.get("candidates_total")
|
|
196
|
+
if recall.get("truncated"):
|
|
197
|
+
vector_recall = dict(recall)
|
|
198
|
+
index_block = vector_payload.get("index")
|
|
199
|
+
if isinstance(index_block, dict):
|
|
200
|
+
vector_meta["approx"] = bool(index_block.get("approx"))
|
|
201
|
+
vector_meta["exhaustive"] = bool(index_block.get("exhaustive"))
|
|
202
|
+
except Exception as exc: # noqa: BLE001 — degrade, never fail the search
|
|
203
|
+
mode = "lexical_only"
|
|
204
|
+
detail = f"vector index unavailable: {exc}"
|
|
205
|
+
vector_matches = []
|
|
206
|
+
# An embedder swap makes the vector channel silently return zero rows
|
|
207
|
+
# (vector_search filters on the CURRENT model/dim). Surface the honest
|
|
208
|
+
# cause additively without changing the mode string.
|
|
209
|
+
vector_degraded: Optional[str] = None
|
|
210
|
+
if mode == "hybrid" and not vector_matches:
|
|
211
|
+
try:
|
|
212
|
+
fingerprint_fn = getattr(self, "embedder_fingerprint_status", None)
|
|
213
|
+
if callable(fingerprint_fn) and fingerprint_fn().get("stale_embedder"):
|
|
214
|
+
vector_degraded = "stale_embedder"
|
|
215
|
+
except Exception: # noqa: BLE001 — fingerprint status must never break search
|
|
216
|
+
vector_degraded = None
|
|
217
|
+
if vector_matches and allowed_workspaces is not None:
|
|
218
|
+
vector_matches = self.filter_scoped_nodes(
|
|
219
|
+
vector_matches,
|
|
220
|
+
allowed_workspaces,
|
|
221
|
+
id_key="node_id",
|
|
222
|
+
include_legacy_global=include_legacy_global,
|
|
223
|
+
)
|
|
224
|
+
|
|
225
|
+
def _parent_node_id(match: Dict[str, Any]) -> str:
|
|
226
|
+
# Chunk-level hits dedupe to their parent content node.
|
|
227
|
+
if match.get("type") == "Chunk":
|
|
228
|
+
meta = match.get("metadata") or {}
|
|
229
|
+
parent = meta.get("source_node") or meta.get("parent_source_node")
|
|
230
|
+
if parent:
|
|
231
|
+
return str(parent)
|
|
232
|
+
return str(match.get("node_id") or match.get("id") or "")
|
|
233
|
+
|
|
234
|
+
entries: Dict[str, Dict[str, Any]] = {}
|
|
235
|
+
|
|
236
|
+
def _entry_for(node_id: str, match: Dict[str, Any]) -> Dict[str, Any]:
|
|
237
|
+
entry = entries.get(node_id)
|
|
238
|
+
if entry is None:
|
|
239
|
+
entry = {
|
|
240
|
+
"node_id": node_id,
|
|
241
|
+
"id": match.get("id") or node_id,
|
|
242
|
+
"type": match.get("type"),
|
|
243
|
+
"title": match.get("title"),
|
|
244
|
+
"summary": match.get("summary"),
|
|
245
|
+
"metadata": match.get("metadata") or {},
|
|
246
|
+
"updated_at": match.get("updated_at"),
|
|
247
|
+
"scores": {"lexical": 0.0, "vector": 0.0},
|
|
248
|
+
"_lexical": False,
|
|
249
|
+
"_vector": False,
|
|
250
|
+
}
|
|
251
|
+
entries[node_id] = entry
|
|
252
|
+
return entry
|
|
253
|
+
|
|
254
|
+
# Per-channel id order (best first) — the only input RRF needs, and
|
|
255
|
+
# the one thing a normalized score cannot reconstruct.
|
|
256
|
+
lexical_order: List[str] = []
|
|
257
|
+
vector_order: List[str] = []
|
|
258
|
+
|
|
259
|
+
for rank, match in enumerate(lexical_matches, start=1):
|
|
260
|
+
node_id = _parent_node_id(match)
|
|
261
|
+
if not node_id:
|
|
262
|
+
continue
|
|
263
|
+
entry = _entry_for(node_id, match)
|
|
264
|
+
entry["scores"]["lexical"] = max(
|
|
265
|
+
entry["scores"]["lexical"], round(1.0 / rank, 6)
|
|
266
|
+
)
|
|
267
|
+
entry["_lexical"] = True
|
|
268
|
+
lexical_order.append(node_id)
|
|
269
|
+
|
|
270
|
+
# Max-normalize cosine scores into [0, 1] (guard the score-0 falsy trap
|
|
271
|
+
# by comparing explicitly, never with truthiness).
|
|
272
|
+
max_vec = 0.0
|
|
273
|
+
for match in vector_matches:
|
|
274
|
+
raw = match.get("score")
|
|
275
|
+
if raw is not None and float(raw) > max_vec:
|
|
276
|
+
max_vec = float(raw)
|
|
277
|
+
for match in vector_matches:
|
|
278
|
+
node_id = _parent_node_id(match)
|
|
279
|
+
if not node_id:
|
|
280
|
+
continue
|
|
281
|
+
raw = float(match.get("score") or 0.0)
|
|
282
|
+
vec_norm = max(0.0, raw) / max_vec if max_vec > 0 else 0.0
|
|
283
|
+
entry = _entry_for(node_id, match)
|
|
284
|
+
entry["scores"]["vector"] = max(entry["scores"]["vector"], round(vec_norm, 6))
|
|
285
|
+
entry["_vector"] = True
|
|
286
|
+
vector_order.append(node_id)
|
|
287
|
+
# Prefer a real snippet when the lexical row had no summary.
|
|
288
|
+
if not entry.get("summary") and match.get("summary"):
|
|
289
|
+
entry["summary"] = match.get("summary")
|
|
290
|
+
|
|
291
|
+
# Graph traversal candidate expansion (opt-in, capped, counted): pull
|
|
292
|
+
# the one-hop neighbours of the strongest hits into the candidate pool
|
|
293
|
+
# so an answer that is adjacent to the match — not in it — is
|
|
294
|
+
# reachable at all. Off by default; see fusion.GRAPH_EXPANSION_ENV.
|
|
295
|
+
expansion_report: Dict[str, Any] = {
|
|
296
|
+
"enabled": False,
|
|
297
|
+
"seeds": 0,
|
|
298
|
+
"added": 0,
|
|
299
|
+
"cap": DEFAULT_EXPANSION_CAP,
|
|
300
|
+
"truncated": False,
|
|
301
|
+
"failed_seeds": 0,
|
|
302
|
+
}
|
|
303
|
+
if entries and graph_expansion_enabled():
|
|
304
|
+
seeds = sorted(
|
|
305
|
+
(
|
|
306
|
+
(node_id, float(entry["scores"]["vector"]))
|
|
307
|
+
for node_id, entry in entries.items()
|
|
308
|
+
),
|
|
309
|
+
key=lambda pair: -pair[1],
|
|
310
|
+
)[:DEFAULT_EXPANSION_SEEDS]
|
|
311
|
+
expanded, expansion_report = expand_with_neighbors(
|
|
312
|
+
seeds,
|
|
313
|
+
self.neighbors,
|
|
314
|
+
exclude=list(entries),
|
|
315
|
+
cap=DEFAULT_EXPANSION_CAP,
|
|
316
|
+
)
|
|
317
|
+
for candidate in expanded:
|
|
318
|
+
node = candidate["node"]
|
|
319
|
+
entry = _entry_for(str(node.get("id")), dict(node))
|
|
320
|
+
entry["scores"]["graph"] = candidate["score"]
|
|
321
|
+
entry["metadata"] = {
|
|
322
|
+
**(entry.get("metadata") or {}),
|
|
323
|
+
"expanded_from": candidate["seed"],
|
|
324
|
+
}
|
|
325
|
+
entry["_graph"] = True
|
|
326
|
+
|
|
327
|
+
rrf_normalized: Dict[str, float] = {}
|
|
328
|
+
if fusion_strategy == "rrf":
|
|
329
|
+
raw_rrf = rrf_fuse(
|
|
330
|
+
{
|
|
331
|
+
"lexical": list(dict.fromkeys(lexical_order)),
|
|
332
|
+
"vector": list(dict.fromkeys(vector_order)),
|
|
333
|
+
}
|
|
334
|
+
)
|
|
335
|
+
peak = max(raw_rrf.values(), default=0.0)
|
|
336
|
+
if peak > 0:
|
|
337
|
+
# Rescale to [0, 1] so the score column keeps the same meaning
|
|
338
|
+
# across strategies; RRF's raw values live around 1/60.
|
|
339
|
+
rrf_normalized = {key: value / peak for key, value in raw_rrf.items()}
|
|
340
|
+
|
|
341
|
+
matches: List[Dict[str, Any]] = []
|
|
342
|
+
for entry in entries.values():
|
|
343
|
+
lex_score = float(entry["scores"]["lexical"])
|
|
344
|
+
vec_score = float(entry["scores"]["vector"])
|
|
345
|
+
if mode == "lexical_only":
|
|
346
|
+
fused = lex_score
|
|
347
|
+
elif fusion_strategy == "rrf":
|
|
348
|
+
fused = float(rrf_normalized.get(entry["node_id"], 0.0))
|
|
349
|
+
entry["scores"]["rrf"] = round(fused, 6)
|
|
350
|
+
else:
|
|
351
|
+
fused = alpha * vec_score + (1.0 - alpha) * lex_score
|
|
352
|
+
from_lexical = bool(entry.pop("_lexical", False))
|
|
353
|
+
from_vector = bool(entry.pop("_vector", False))
|
|
354
|
+
if entry.pop("_graph", False):
|
|
355
|
+
# A one-hop neighbour of a hit: related to the answer, never
|
|
356
|
+
# itself a match, so it carries only its damped seed score.
|
|
357
|
+
fused = float(entry["scores"]["graph"])
|
|
358
|
+
entry["fusion"] = "graph"
|
|
359
|
+
elif from_lexical and from_vector:
|
|
360
|
+
entry["fusion"] = "both"
|
|
361
|
+
elif from_vector:
|
|
362
|
+
entry["fusion"] = "vector"
|
|
363
|
+
else:
|
|
364
|
+
entry["fusion"] = "lexical"
|
|
365
|
+
entry["score"] = round(fused, 6)
|
|
366
|
+
matches.append(entry)
|
|
367
|
+
|
|
368
|
+
# Recency-class age decay (retrieval_policy): dampen each fused score
|
|
369
|
+
# into the [0.5, 1.0] band so old-but-relevant items sink without ever
|
|
370
|
+
# being zeroed. Other classes skip this block byte-identically.
|
|
371
|
+
if recency_half_life_days is not None:
|
|
372
|
+
decay_now = datetime.now()
|
|
373
|
+
for match in matches:
|
|
374
|
+
stamp = match.get("updated_at")
|
|
375
|
+
if _parse_iso(stamp):
|
|
376
|
+
multiplier = 0.5 + 0.5 * _recency_score(
|
|
377
|
+
stamp, now=decay_now, half_life_days=recency_half_life_days
|
|
378
|
+
)
|
|
379
|
+
else:
|
|
380
|
+
# Unknown age is not evidence of staleness — never dampen.
|
|
381
|
+
multiplier = 1.0
|
|
382
|
+
match["scores"]["age_decay"] = round(multiplier, 6)
|
|
383
|
+
match["score"] = round(float(match["score"]) * multiplier, 6)
|
|
384
|
+
|
|
385
|
+
# Late fusion of the image space (v11.1.0). Runs after the text
|
|
386
|
+
# channels have produced a ranking and before the cut, so image
|
|
387
|
+
# evidence can lift a picture into the answer without ever having been
|
|
388
|
+
# compared against a text vector.
|
|
389
|
+
image_fusion: Optional[Dict[str, Any]] = None
|
|
390
|
+
if image_vector is not None:
|
|
391
|
+
image_fusion = self._fuse_image_channel(
|
|
392
|
+
matches, image_vector, top_k=top_k, weight=image_fusion_weight
|
|
393
|
+
)
|
|
394
|
+
|
|
395
|
+
matches.sort(key=lambda item: (-item["score"], item["node_id"]))
|
|
396
|
+
# Optional cross-encoder rerank (v9.9.5). Off by default; when the
|
|
397
|
+
# env kill-switch is set and the model loads, pair scores reorder the
|
|
398
|
+
# fused list. Failures degrade to identity and never break search.
|
|
399
|
+
rerank_meta: Dict[str, Any]
|
|
400
|
+
try:
|
|
401
|
+
from ..rerank import rerank_matches
|
|
402
|
+
|
|
403
|
+
# Rerank a slightly wider window, then cut to top_k.
|
|
404
|
+
window = matches[: max(top_k * 2, top_k)]
|
|
405
|
+
reranked = rerank_matches(search_query, window, top_k=top_k)
|
|
406
|
+
matches = list(reranked.get("matches") or matches[:top_k])
|
|
407
|
+
rerank_meta = {
|
|
408
|
+
"mode": reranked.get("mode") or "identity",
|
|
409
|
+
"model": reranked.get("model"),
|
|
410
|
+
"detail": reranked.get("detail"),
|
|
411
|
+
}
|
|
412
|
+
except Exception as exc: # noqa: BLE001 — rerank must never break search
|
|
413
|
+
matches = matches[:top_k]
|
|
414
|
+
rerank_meta = {"mode": "identity", "model": None, "detail": str(exc)}
|
|
415
|
+
for rank, match in enumerate(matches, start=1):
|
|
416
|
+
match["rank"] = rank
|
|
417
|
+
result = {
|
|
418
|
+
"query": query,
|
|
419
|
+
"mode": mode,
|
|
420
|
+
"alpha": alpha,
|
|
421
|
+
"query_class": query_class,
|
|
422
|
+
"top_k": top_k,
|
|
423
|
+
"sources": {"lexical": len(lexical_matches), "vector": len(vector_matches)},
|
|
424
|
+
"matches": matches,
|
|
425
|
+
"policy": {"search_query": search_query, "rewrite_rules": rewrite_rules},
|
|
426
|
+
"fusion_strategy": fusion_strategy,
|
|
427
|
+
"graph_expansion": expansion_report,
|
|
428
|
+
"rerank": rerank_meta,
|
|
429
|
+
"detail": detail,
|
|
430
|
+
}
|
|
431
|
+
if vector_degraded is not None:
|
|
432
|
+
result["vector_degraded"] = vector_degraded
|
|
433
|
+
if vector_recall is not None:
|
|
434
|
+
result["vector_recall"] = vector_recall
|
|
435
|
+
if vector_degraded is None:
|
|
436
|
+
result["vector_degraded"] = "partial_recall"
|
|
437
|
+
vector_meta["degraded"] = result.get("vector_degraded")
|
|
438
|
+
result["vector"] = vector_meta
|
|
439
|
+
multimodal = multimodal_signal(matches)
|
|
440
|
+
if multimodal is not None or image_fusion is not None:
|
|
441
|
+
result["multimodal"] = {
|
|
442
|
+
**(multimodal or {"images": 0, "types": []}),
|
|
443
|
+
**({"image_fusion": image_fusion} if image_fusion is not None else {}),
|
|
444
|
+
}
|
|
445
|
+
return result
|
|
446
|
+
|
|
447
|
+
def _fuse_image_channel(
|
|
448
|
+
self,
|
|
449
|
+
matches: List[Dict[str, Any]],
|
|
450
|
+
image_vector: Sequence[float],
|
|
451
|
+
*,
|
|
452
|
+
top_k: int,
|
|
453
|
+
weight: Optional[float],
|
|
454
|
+
) -> Dict[str, Any]:
|
|
455
|
+
"""Rank the image index separately, then blend it into ``matches``.
|
|
456
|
+
|
|
457
|
+
Any failure degrades to "the image channel contributed nothing" with
|
|
458
|
+
the reason attached — an image index that cannot be read is not a
|
|
459
|
+
reason to lose the text answer.
|
|
460
|
+
"""
|
|
461
|
+
from ..image_vectors import (
|
|
462
|
+
DEFAULT_IMAGE_FUSION_WEIGHT,
|
|
463
|
+
fuse_image_scores,
|
|
464
|
+
image_similarity_search,
|
|
465
|
+
)
|
|
466
|
+
|
|
467
|
+
share = DEFAULT_IMAGE_FUSION_WEIGHT if weight is None else float(weight)
|
|
468
|
+
report: Dict[str, Any] = {
|
|
469
|
+
"weight": round(max(0.0, min(1.0, share)), 4),
|
|
470
|
+
"candidates": 0,
|
|
471
|
+
"fused": 0,
|
|
472
|
+
"detail": None,
|
|
473
|
+
}
|
|
474
|
+
try:
|
|
475
|
+
found = image_similarity_search(
|
|
476
|
+
self, image_vector, top_k=max(1, int(top_k) * 2)
|
|
477
|
+
)
|
|
478
|
+
except Exception as exc: # noqa: BLE001 — never fail the text answer
|
|
479
|
+
report["detail"] = f"image index unavailable: {exc}"
|
|
480
|
+
return report
|
|
481
|
+
report["candidates"] = int(found.get("candidates") or 0)
|
|
482
|
+
report["detail"] = found.get("detail")
|
|
483
|
+
scores = {
|
|
484
|
+
str(row.get("node_id")): float(row.get("score") or 0.0)
|
|
485
|
+
for row in found.get("matches") or []
|
|
486
|
+
}
|
|
487
|
+
report["fused"] = fuse_image_scores(matches, scores, weight=share)
|
|
488
|
+
return report
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
"""Destructive graph maintenance: drop one conversation, or everything.
|
|
2
|
+
|
|
3
|
+
Both methods gate their ``nodes_v2`` work on ``KGStoreV2``. That name is
|
|
4
|
+
star-imported from ``_kg_common`` into **this** module's globals, so the
|
|
5
|
+
patch target for "pretend the v2 projection is unavailable" is
|
|
6
|
+
``lattice_brain.graph.retrieval.maintenance.KGStoreV2``.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from typing import TYPE_CHECKING
|
|
12
|
+
|
|
13
|
+
# ruff: noqa: F403,F405
|
|
14
|
+
from .._kg_common import * # noqa: F403,F401
|
|
15
|
+
|
|
16
|
+
# The cross-mixin surface (`_connect`, `_upsert_node`, …) is declared in
|
|
17
|
+
# `_kg_contract.KnowledgeGraphCore`. It is a typing-only base: at runtime this
|
|
18
|
+
# is `object`, so the MRO of `KnowledgeGraphStore` is unchanged.
|
|
19
|
+
if TYPE_CHECKING:
|
|
20
|
+
from .._kg_contract import KnowledgeGraphCore as _Core
|
|
21
|
+
else:
|
|
22
|
+
_Core = object
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class _MaintenanceMixin(_Core):
|
|
26
|
+
"""Conversation/graph deletion. Composed into the public mixin."""
|
|
27
|
+
|
|
28
|
+
def delete_conversation(self, conversation_id: str) -> Dict[str, Any]:
|
|
29
|
+
conversation_id = str(conversation_id or "").strip()
|
|
30
|
+
if not conversation_id:
|
|
31
|
+
return {"status": "skipped", "removed_nodes": 0}
|
|
32
|
+
conv_id = f"conversation:{_slug(conversation_id)}"
|
|
33
|
+
with self._connect() as conn:
|
|
34
|
+
# Edge rows may carry the legacy lowercase label (pre-v4) or the
|
|
35
|
+
# canonical EdgeType value (v4 write door) — match both.
|
|
36
|
+
direct_ids = [
|
|
37
|
+
row["to_node"]
|
|
38
|
+
for row in conn.execute(
|
|
39
|
+
"SELECT to_node FROM edges WHERE from_node=? AND type IN ('contains', 'CONTAINS')",
|
|
40
|
+
(conv_id,),
|
|
41
|
+
)
|
|
42
|
+
]
|
|
43
|
+
remove_ids = set(direct_ids)
|
|
44
|
+
child_types = [
|
|
45
|
+
"has_chunk",
|
|
46
|
+
"implies",
|
|
47
|
+
"contains_signal",
|
|
48
|
+
"has_page",
|
|
49
|
+
"has_slide",
|
|
50
|
+
"has_sheet",
|
|
51
|
+
"contains_image",
|
|
52
|
+
]
|
|
53
|
+
child_types += [t.upper() for t in child_types]
|
|
54
|
+
placeholders = ",".join("?" for _ in child_types)
|
|
55
|
+
for source_id in list(direct_ids):
|
|
56
|
+
for row in conn.execute(
|
|
57
|
+
f"SELECT to_node FROM edges WHERE from_node=? AND type IN ({placeholders})",
|
|
58
|
+
(source_id, *child_types),
|
|
59
|
+
):
|
|
60
|
+
remove_ids.add(row["to_node"])
|
|
61
|
+
remove_ids.add(conv_id)
|
|
62
|
+
for node_id in remove_ids:
|
|
63
|
+
conn.execute("DELETE FROM nodes WHERE id=?", (node_id,))
|
|
64
|
+
if KGStoreV2 is not None:
|
|
65
|
+
conn.execute(
|
|
66
|
+
"DELETE FROM nodes_v2 WHERE id=?", (node_id,)
|
|
67
|
+
) # edges_v2 cascade
|
|
68
|
+
conn.execute(
|
|
69
|
+
"""
|
|
70
|
+
DELETE FROM nodes
|
|
71
|
+
WHERE type='Topic'
|
|
72
|
+
AND id NOT IN (SELECT to_node FROM edges)
|
|
73
|
+
AND id NOT IN (SELECT from_node FROM edges)
|
|
74
|
+
"""
|
|
75
|
+
)
|
|
76
|
+
if KGStoreV2 is not None:
|
|
77
|
+
conn.execute(
|
|
78
|
+
"""
|
|
79
|
+
DELETE FROM nodes_v2
|
|
80
|
+
WHERE legacy_type='Topic'
|
|
81
|
+
AND id NOT IN (SELECT target FROM edges_v2)
|
|
82
|
+
AND id NOT IN (SELECT source FROM edges_v2)
|
|
83
|
+
"""
|
|
84
|
+
)
|
|
85
|
+
return {
|
|
86
|
+
"status": "ok",
|
|
87
|
+
"conversation_id": conversation_id,
|
|
88
|
+
"removed_nodes": len(remove_ids),
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
def clear_all(self) -> Dict[str, Any]:
|
|
92
|
+
with self._connect() as conn:
|
|
93
|
+
counts = {
|
|
94
|
+
"nodes": conn.execute("SELECT COUNT(*) AS c FROM nodes").fetchone()[
|
|
95
|
+
"c"
|
|
96
|
+
],
|
|
97
|
+
"edges": conn.execute("SELECT COUNT(*) AS c FROM edges").fetchone()[
|
|
98
|
+
"c"
|
|
99
|
+
],
|
|
100
|
+
"chunks": conn.execute("SELECT COUNT(*) AS c FROM chunks").fetchone()[
|
|
101
|
+
"c"
|
|
102
|
+
],
|
|
103
|
+
"knowledge_sources": conn.execute(
|
|
104
|
+
"SELECT COUNT(*) AS c FROM knowledge_sources"
|
|
105
|
+
).fetchone()["c"],
|
|
106
|
+
"local_file_index": conn.execute(
|
|
107
|
+
"SELECT COUNT(*) AS c FROM local_file_index"
|
|
108
|
+
).fetchone()["c"],
|
|
109
|
+
}
|
|
110
|
+
conn.execute("DELETE FROM local_file_index")
|
|
111
|
+
conn.execute("DELETE FROM knowledge_sources")
|
|
112
|
+
conn.execute("DELETE FROM chunks")
|
|
113
|
+
conn.execute("DELETE FROM edges")
|
|
114
|
+
conn.execute("DELETE FROM nodes")
|
|
115
|
+
if KGStoreV2 is not None:
|
|
116
|
+
conn.execute("DELETE FROM edges_v2")
|
|
117
|
+
conn.execute("DELETE FROM nodes_v2")
|
|
118
|
+
if self.blob_dir.exists():
|
|
119
|
+
shutil.rmtree(self.blob_dir, ignore_errors=True)
|
|
120
|
+
self.blob_dir.mkdir(parents=True, exist_ok=True)
|
|
121
|
+
return {"status": "ok", "removed": counts}
|