ltcai 11.2.0 → 11.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -53
- package/docs/CHANGELOG.md +61 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/MULTI_AGENT_RUNTIME.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +6 -2
- package/docs/PERMISSION_MODE.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +2 -2
- package/docs/v11.3.0_PLAN.md +202 -0
- package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +287 -0
- package/lattice_brain/graph/_kg_common/extraction.py +516 -0
- package/lattice_brain/graph/_kg_common/relations.py +161 -0
- package/lattice_brain/graph/_kg_common/text.py +479 -0
- package/lattice_brain/graph/discovery_index/__init__.py +35 -0
- package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
- package/lattice_brain/graph/discovery_index/extract.py +137 -0
- package/lattice_brain/graph/discovery_index/scan.py +411 -0
- package/lattice_brain/graph/discovery_index/upsert.py +495 -0
- package/lattice_brain/graph/projection/__init__.py +42 -0
- package/lattice_brain/graph/projection/curation.py +500 -0
- package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
- package/lattice_brain/graph/retrieval/__init__.py +54 -0
- package/lattice_brain/graph/retrieval/context.py +197 -0
- package/lattice_brain/graph/retrieval/graph_view.py +319 -0
- package/lattice_brain/graph/retrieval/hybrid.py +488 -0
- package/lattice_brain/graph/retrieval/maintenance.py +121 -0
- package/lattice_brain/graph/retrieval/signals.py +95 -0
- package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
- package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
- package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
- package/lattice_brain/graph/retrieval_vector/search.py +560 -0
- package/lattice_brain/graph/retrieval_vector/status.py +374 -0
- package/lattice_brain/ingestion/__init__.py +130 -0
- package/lattice_brain/ingestion/_contract.py +90 -0
- package/lattice_brain/ingestion/constants.py +127 -0
- package/lattice_brain/ingestion/folder_scan.py +57 -0
- package/lattice_brain/ingestion/folders.py +258 -0
- package/lattice_brain/ingestion/hashing.py +26 -0
- package/lattice_brain/ingestion/jobs_api.py +107 -0
- package/lattice_brain/ingestion/models.py +80 -0
- package/lattice_brain/ingestion/pipeline.py +486 -0
- package/lattice_brain/ingestion/quality.py +209 -0
- package/lattice_brain/ingestion/routing.py +295 -0
- package/lattice_brain/multimodal/__init__.py +164 -0
- package/lattice_brain/multimodal/audio.py +77 -0
- package/lattice_brain/multimodal/common.py +118 -0
- package/lattice_brain/multimodal/images.py +498 -0
- package/lattice_brain/multimodal/ports.py +169 -0
- package/lattice_brain/multimodal/video.py +410 -0
- package/lattice_brain/portability/__init__.py +90 -0
- package/lattice_brain/portability/_contract.py +42 -0
- package/lattice_brain/portability/backups.py +338 -0
- package/lattice_brain/portability/bundles.py +136 -0
- package/lattice_brain/portability/constants.py +93 -0
- package/lattice_brain/portability/fsops.py +138 -0
- package/lattice_brain/portability/service.py +41 -0
- package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
- package/lattice_brain/runtime/__init__.py +1 -1
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/chronicle.py +63 -0
- package/latticeai/core/agent/__init__.py +93 -0
- package/latticeai/core/agent/_contract.py +79 -0
- package/latticeai/core/agent/context.py +57 -0
- package/latticeai/core/agent/deps.py +125 -0
- package/latticeai/core/agent/execution.py +622 -0
- package/latticeai/core/agent/planning.py +145 -0
- package/latticeai/core/agent/recovery.py +157 -0
- package/latticeai/core/agent/runtime.py +210 -0
- package/latticeai/core/agent/verification.py +231 -0
- package/latticeai/core/embedding_providers/__init__.py +151 -0
- package/latticeai/core/embedding_providers/base.py +199 -0
- package/latticeai/core/embedding_providers/captions.py +162 -0
- package/latticeai/core/embedding_providers/profiles.py +126 -0
- package/latticeai/core/embedding_providers/text.py +350 -0
- package/latticeai/core/embedding_providers/vision.py +352 -0
- package/latticeai/core/file_generation/__init__.py +115 -0
- package/latticeai/core/file_generation/bundles.py +76 -0
- package/latticeai/core/file_generation/extraction.py +154 -0
- package/latticeai/core/file_generation/inference.py +235 -0
- package/latticeai/core/file_generation/orchestration.py +152 -0
- package/latticeai/core/file_generation/prompting.py +117 -0
- package/latticeai/core/file_generation/repair.py +114 -0
- package/latticeai/core/file_generation/sanitize.py +61 -0
- package/latticeai/core/file_generation/validation.py +201 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +9 -0
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/integrations/telegram_bot/__init__.py +123 -0
- package/latticeai/integrations/telegram_bot/__main__.py +17 -0
- package/latticeai/integrations/telegram_bot/config.py +86 -0
- package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
- package/latticeai/integrations/telegram_bot/flows.py +478 -0
- package/latticeai/integrations/telegram_bot/helpers.py +322 -0
- package/latticeai/integrations/telegram_bot/screens.py +394 -0
- package/latticeai/models/router/__init__.py +88 -0
- package/latticeai/models/router/_contract.py +66 -0
- package/latticeai/models/router/branding.py +56 -0
- package/latticeai/models/router/catalog.py +69 -0
- package/latticeai/models/router/documents.py +199 -0
- package/latticeai/models/router/errors.py +37 -0
- package/latticeai/models/router/generation.py +258 -0
- package/latticeai/models/router/loading.py +291 -0
- package/latticeai/models/router/local_models.py +85 -0
- package/latticeai/models/router/registry.py +147 -0
- package/latticeai/runtime/build_phases/__init__.py +82 -0
- package/latticeai/runtime/build_phases/features.py +407 -0
- package/latticeai/runtime/build_phases/foundation.py +555 -0
- package/latticeai/runtime/build_phases/web.py +492 -0
- package/latticeai/runtime/runtime_context.py +1 -0
- package/latticeai/services/architecture_readiness.py +48 -19
- package/latticeai/services/brain_intelligence/__init__.py +58 -0
- package/latticeai/services/brain_intelligence/_contract.py +71 -0
- package/latticeai/services/brain_intelligence/consistency.py +193 -0
- package/latticeai/services/brain_intelligence/constants.py +47 -0
- package/latticeai/services/brain_intelligence/digest.py +258 -0
- package/latticeai/services/brain_intelligence/health.py +331 -0
- package/latticeai/services/brain_intelligence/proposals.py +264 -0
- package/latticeai/services/brain_intelligence/sampling.py +84 -0
- package/latticeai/services/brain_intelligence/service.py +48 -0
- package/latticeai/services/chronicle.py +557 -0
- package/latticeai/services/memory_service/__init__.py +52 -0
- package/latticeai/services/memory_service/_contract.py +100 -0
- package/latticeai/services/memory_service/brief.py +431 -0
- package/latticeai/services/memory_service/constants.py +57 -0
- package/latticeai/services/memory_service/maintenance.py +138 -0
- package/latticeai/services/memory_service/manager.py +186 -0
- package/latticeai/services/memory_service/proof.py +136 -0
- package/latticeai/services/memory_service/recall.py +225 -0
- package/latticeai/services/memory_service/service.py +48 -0
- package/latticeai/services/memory_service/stores.py +110 -0
- package/latticeai/services/model_runtime/__init__.py +322 -0
- package/latticeai/services/model_runtime/cloud.py +87 -0
- package/latticeai/services/model_runtime/download.py +282 -0
- package/latticeai/services/model_runtime/engines.py +341 -0
- package/latticeai/services/model_runtime/loading.py +178 -0
- package/latticeai/services/model_runtime/service.py +129 -0
- package/latticeai/services/model_runtime/state.py +131 -0
- package/latticeai/services/model_runtime/status.py +255 -0
- package/latticeai/services/product_readiness.py +15 -7
- package/latticeai/setup/wizard/__init__.py +126 -0
- package/latticeai/setup/wizard/catalog.py +172 -0
- package/latticeai/setup/wizard/detect.py +323 -0
- package/latticeai/setup/wizard/install.py +348 -0
- package/latticeai/setup/wizard/paths.py +168 -0
- package/latticeai/setup/wizard/plans.py +74 -0
- package/latticeai/setup/wizard/recommend.py +320 -0
- package/package.json +6 -2
- package/scripts/bump_version.py +14 -0
- package/scripts/capture_release_evidence.mjs +33 -21
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_i18n_namespace_coverage.mjs +41 -4
- package/scripts/check_max_file_lines.mjs +102 -0
- package/scripts/check_release_evidence_bound.mjs +30 -15
- package/scripts/check_screenshot_pixel_delta.py +34 -4
- package/scripts/check_server_i18n.mjs +1 -0
- package/scripts/generate_rust_parity_fixtures.py +562 -0
- package/scripts/lib/mock_server_fingerprint.mjs +94 -0
- package/scripts/release_screen_claims.json +31 -2
- package/src-tauri/Cargo.lock +361 -3
- package/src-tauri/Cargo.toml +6 -1
- package/src-tauri/src/backend.rs +349 -0
- package/src-tauri/src/folder.rs +33 -0
- package/src-tauri/src/main.rs +97 -399
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +41 -37
- package/static/app/assets/Act-yYpYnn0v.js +1 -0
- package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
- package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
- package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
- package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
- package/static/app/assets/Capture-CFIRsFNE.js +1 -0
- package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
- package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
- package/static/app/assets/Library-DwO3yZST.js +1 -0
- package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
- package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
- package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
- package/static/app/assets/System-DW8F-2xL.js +1 -0
- package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
- package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
- package/static/app/assets/brain-Ci1CkWjM.js +1 -0
- package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
- package/static/app/assets/circle-check-DfInj-qD.js +1 -0
- package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
- package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
- package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
- package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
- package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
- package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
- package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
- package/static/app/assets/index-_u5iUHDr.js +10 -0
- package/static/app/assets/input-B0lPdRQZ.js +1 -0
- package/static/app/assets/link-2-CoFbooHS.js +1 -0
- package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
- package/static/app/assets/primitives-DEbN-d6p.js +1 -0
- package/static/app/assets/search-BybIWPNd.js +1 -0
- package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
- package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
- package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
- package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
- package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
- package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
- package/static/app/assets/utils-BlZr7Pd4.js +4 -0
- package/static/app/assets/workspace-jJY4RuAV.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/graph/_kg_common.py +0 -1331
- package/lattice_brain/graph/discovery_index.py +0 -1141
- package/lattice_brain/graph/retrieval.py +0 -1120
- package/lattice_brain/graph/retrieval_vector.py +0 -1293
- package/lattice_brain/ingestion.py +0 -1525
- package/lattice_brain/multimodal.py +0 -1258
- package/latticeai/core/agent.py +0 -1465
- package/latticeai/core/embedding_providers.py +0 -1196
- package/latticeai/core/file_generation.py +0 -1047
- package/latticeai/integrations/telegram_bot.py +0 -1390
- package/latticeai/models/router.py +0 -1007
- package/latticeai/runtime/build_phases.py +0 -1450
- package/latticeai/services/brain_intelligence.py +0 -1083
- package/latticeai/services/memory_service.py +0 -1177
- package/latticeai/services/model_runtime.py +0 -1281
- package/latticeai/setup/wizard.py +0 -1310
- package/static/app/assets/Act-AWf0SAKp.js +0 -1
- package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
- package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
- package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
- package/static/app/assets/Capture-CqOSzyPr.js +0 -1
- package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
- package/static/app/assets/Library-CX-bbhmK.js +0 -1
- package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
- package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
- package/static/app/assets/System-Bu2t5hn1.js +0 -1
- package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
- package/static/app/assets/brain-DJMoqrwx.js +0 -1
- package/static/app/assets/index-BpYkzcVm.js +0 -10
- package/static/app/assets/input-DSlJJxRs.js +0 -1
- package/static/app/assets/primitives-BCx6TvfG.js +0 -1
- package/static/app/assets/search-Cgy8cCFJ.js +0 -1
- package/static/app/assets/utils-zqPZJxdx.js +0 -4
- package/static/app/assets/workspace-DXTihhfU.js +0 -1
|
@@ -1,1120 +0,0 @@
|
|
|
1
|
-
from __future__ import annotations
|
|
2
|
-
|
|
3
|
-
from typing import TYPE_CHECKING, Sequence
|
|
4
|
-
|
|
5
|
-
# ruff: noqa: F403,F405
|
|
6
|
-
from ._kg_common import * # noqa: F403,F401
|
|
7
|
-
|
|
8
|
-
# The cross-mixin surface (`_connect`, `_upsert_node`, …) is declared in
|
|
9
|
-
# `_kg_contract.KnowledgeGraphCore`. It is a typing-only base: at runtime this
|
|
10
|
-
# is `object`, so the MRO of `KnowledgeGraphStore` is unchanged.
|
|
11
|
-
if TYPE_CHECKING:
|
|
12
|
-
from ._kg_contract import KnowledgeGraphCore as _Core
|
|
13
|
-
else:
|
|
14
|
-
_Core = object
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
# --- Compat seam (v9.9.5 decomposition) -------------------------------------
|
|
18
|
-
# The non-search read surface (list_documents / workspaces_of /
|
|
19
|
-
# filter_scoped_nodes / neighbors / get_node / relationship_search /
|
|
20
|
-
# traverse / stats) moved byte-identically to .retrieval_reads as
|
|
21
|
-
# KnowledgeGraphReadsMixin. Re-exported here so any legacy
|
|
22
|
-
# ``from lattice_brain.graph.retrieval import ...`` site keeps resolving.
|
|
23
|
-
from .fusion import (
|
|
24
|
-
DEFAULT_EXPANSION_CAP,
|
|
25
|
-
DEFAULT_EXPANSION_SEEDS,
|
|
26
|
-
expand_with_neighbors,
|
|
27
|
-
graph_expansion_enabled,
|
|
28
|
-
rrf_fuse,
|
|
29
|
-
)
|
|
30
|
-
from .retrieval_reads import KnowledgeGraphReadsMixin # noqa: F401
|
|
31
|
-
|
|
32
|
-
#: Node types that are a *thing you can look at or listen to*, not prose. A
|
|
33
|
-
#: match of one of these means the answer rests on more than text.
|
|
34
|
-
MULTIMODAL_NODE_TYPES = ("Image", "ImageText")
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
def multimodal_signal(matches: Iterable[Dict[str, Any]]) -> Optional[Dict[str, Any]]:
|
|
38
|
-
"""``{"images": n, "types": [...]}`` when a result set includes pictures.
|
|
39
|
-
|
|
40
|
-
``None`` when it does not: the context-quality contract stays four keys
|
|
41
|
-
wide for the ordinary all-text case, and a caller that sees the key knows
|
|
42
|
-
it means something rather than having to compare a zero.
|
|
43
|
-
"""
|
|
44
|
-
images = 0
|
|
45
|
-
seen: List[str] = []
|
|
46
|
-
for match in matches:
|
|
47
|
-
node_type = str(match.get("type") or "")
|
|
48
|
-
if node_type in MULTIMODAL_NODE_TYPES:
|
|
49
|
-
images += 1
|
|
50
|
-
if node_type not in seen:
|
|
51
|
-
seen.append(node_type)
|
|
52
|
-
if not images:
|
|
53
|
-
return None
|
|
54
|
-
return {"images": images, "types": seen}
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
def context_quality_signal(
|
|
58
|
-
mode: str,
|
|
59
|
-
nodes: int,
|
|
60
|
-
*,
|
|
61
|
-
reason: Optional[str] = None,
|
|
62
|
-
vector: Optional[Dict[str, Any]] = None,
|
|
63
|
-
multimodal: Optional[Dict[str, Any]] = None,
|
|
64
|
-
) -> Dict[str, Any]:
|
|
65
|
-
"""Honest RAG context-quality signal (v9.8.0, additive contract).
|
|
66
|
-
|
|
67
|
-
Shape consumed by the chat metadata channel:
|
|
68
|
-
``{"mode": "hybrid"|"lexical_only"|"none", "nodes": int, "limited": bool,
|
|
69
|
-
"reason": str|None}``. ``nodes == 0`` always collapses ``mode`` to
|
|
70
|
-
``"none"``; ``limited`` is true whenever the context is thin (0–1 nodes)
|
|
71
|
-
or the vector side fell back to lexical-only retrieval. ``reason`` is a
|
|
72
|
-
short human-readable Korean phrase, only present when limited.
|
|
73
|
-
|
|
74
|
-
``vector`` (v11.1.0) carries the vector channel's own honesty block —
|
|
75
|
-
which backend scored, whether it was approximate, whether the candidate
|
|
76
|
-
scan was truncated. "hybrid, 6 nodes" describes two different answers
|
|
77
|
-
depending on those bits, and the caller that has to say "I did not find
|
|
78
|
-
it" deserves to know which one it got. The key is present **only when
|
|
79
|
-
there is a caveat to report**: an exact, complete vector scan is the
|
|
80
|
-
contract's baseline assumption, so annotating it would be noise, and the
|
|
81
|
-
four-key shape stays exactly what existing consumers pin.
|
|
82
|
-
|
|
83
|
-
``multimodal`` (v11.1.0) follows the same present-only-when-true rule and
|
|
84
|
-
says that part of this context is a picture. "6 nodes" reads differently
|
|
85
|
-
when two of them are screenshots whose text came out of OCR, and the
|
|
86
|
-
surface that has to explain the answer deserves to know.
|
|
87
|
-
"""
|
|
88
|
-
nodes = max(0, int(nodes or 0))
|
|
89
|
-
mode = str(mode or "none")
|
|
90
|
-
if nodes == 0:
|
|
91
|
-
mode = "none"
|
|
92
|
-
if mode not in ("hybrid", "lexical_only", "none"):
|
|
93
|
-
mode = "lexical_only"
|
|
94
|
-
limited = nodes <= 1 or mode != "hybrid"
|
|
95
|
-
if reason is None and limited:
|
|
96
|
-
if nodes == 0:
|
|
97
|
-
reason = "그래프에서 관련 지식을 찾지 못했습니다"
|
|
98
|
-
elif mode == "lexical_only":
|
|
99
|
-
reason = "벡터 검색을 사용할 수 없어 키워드 검색 결과만 사용했습니다"
|
|
100
|
-
else:
|
|
101
|
-
reason = "그래프 기반 컨텍스트가 제한적입니다"
|
|
102
|
-
if not limited:
|
|
103
|
-
reason = None
|
|
104
|
-
signal: Dict[str, Any] = {
|
|
105
|
-
"mode": mode,
|
|
106
|
-
"nodes": nodes,
|
|
107
|
-
"limited": limited,
|
|
108
|
-
"reason": reason,
|
|
109
|
-
}
|
|
110
|
-
if vector is not None:
|
|
111
|
-
signal["vector"] = dict(vector)
|
|
112
|
-
if multimodal is not None:
|
|
113
|
-
signal["multimodal"] = dict(multimodal)
|
|
114
|
-
return signal
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
class KnowledgeGraphRetrievalMixin(_Core):
|
|
118
|
-
_GRAPH_VISIBLE_TYPES = (
|
|
119
|
-
"Computer", # 내 컴퓨터
|
|
120
|
-
"Drive", # 드라이브 / 볼륨
|
|
121
|
-
"Folder", # 폴더
|
|
122
|
-
"File", # 일반 파일
|
|
123
|
-
"Chat", # 대화 세션
|
|
124
|
-
"Document", # 파일 (PDF·PPT·Word·Excel·이미지)
|
|
125
|
-
"CodeFile", # 코드 파일
|
|
126
|
-
"Spreadsheet", # 엑셀/CSV
|
|
127
|
-
"SlideDeck", # 프레젠테이션
|
|
128
|
-
"Image", # 이미지
|
|
129
|
-
"ImageText", # OCR 텍스트
|
|
130
|
-
"Audio", # 녹음 / 음성 메모 (11.1.0)
|
|
131
|
-
"Concept", # 개념 / 아이디어 / 기술 용어
|
|
132
|
-
"Person", # 사람
|
|
133
|
-
"Error", # 오류 / 버그
|
|
134
|
-
"Code", # 코드 / 함수
|
|
135
|
-
"Feature", # 소프트웨어 기능
|
|
136
|
-
"Task", # 할 일
|
|
137
|
-
"Decision", # 결정 사항
|
|
138
|
-
# v3.6.0 Knowledge Graph First — 1급 엔티티를 그래프에 노출
|
|
139
|
-
"Source", # 수집 출처 (파일/URL/브라우저 탭/git)
|
|
140
|
-
"Repository", # git 저장소
|
|
141
|
-
"Meeting", # 회의
|
|
142
|
-
"Organization", # 조직
|
|
143
|
-
"Workflow", # 워크플로우
|
|
144
|
-
"Agent", # 에이전트
|
|
145
|
-
)
|
|
146
|
-
|
|
147
|
-
def graph(
|
|
148
|
-
self,
|
|
149
|
-
limit: int = 300,
|
|
150
|
-
*,
|
|
151
|
-
allowed_workspaces=None,
|
|
152
|
-
include_legacy_global: bool = False,
|
|
153
|
-
) -> Dict[str, Any]:
|
|
154
|
-
limit = max(1, min(int(limit or 300), 2000))
|
|
155
|
-
visible = ",".join(f"'{t}'" for t in self._GRAPH_VISIBLE_TYPES)
|
|
156
|
-
nt, et = self._read_tables()
|
|
157
|
-
with self._connect() as conn:
|
|
158
|
-
nodes = [
|
|
159
|
-
{
|
|
160
|
-
"id": row["id"],
|
|
161
|
-
"type": row["type"],
|
|
162
|
-
"title": row["title"],
|
|
163
|
-
"summary": row["summary"],
|
|
164
|
-
"metadata": _safe_loads(row["metadata_json"]),
|
|
165
|
-
"updated_at": row["updated_at"],
|
|
166
|
-
}
|
|
167
|
-
for row in conn.execute(
|
|
168
|
-
f"SELECT id, type, title, summary, metadata_json, updated_at FROM {nt} WHERE type IN ({visible}) ORDER BY updated_at DESC, id ASC LIMIT ?",
|
|
169
|
-
(limit,),
|
|
170
|
-
)
|
|
171
|
-
]
|
|
172
|
-
node_ids = {node["id"] for node in nodes}
|
|
173
|
-
edges: List[Dict[str, Any]] = []
|
|
174
|
-
if node_ids:
|
|
175
|
-
edge_rows = conn.execute(
|
|
176
|
-
f"""
|
|
177
|
-
SELECT id, from_node, to_node, type, weight, metadata_json
|
|
178
|
-
FROM {et}
|
|
179
|
-
WHERE from_node IN (
|
|
180
|
-
SELECT id FROM {nt} WHERE type IN ({visible})
|
|
181
|
-
ORDER BY updated_at DESC, id ASC LIMIT ?
|
|
182
|
-
)
|
|
183
|
-
AND to_node IN (
|
|
184
|
-
SELECT id FROM {nt} WHERE type IN ({visible})
|
|
185
|
-
ORDER BY updated_at DESC, id ASC LIMIT ?
|
|
186
|
-
)
|
|
187
|
-
ORDER BY weight DESC, created_at DESC, id ASC
|
|
188
|
-
""",
|
|
189
|
-
(limit, limit),
|
|
190
|
-
).fetchall()
|
|
191
|
-
edges = [
|
|
192
|
-
{
|
|
193
|
-
"id": row["id"],
|
|
194
|
-
"from": row["from_node"],
|
|
195
|
-
"to": row["to_node"],
|
|
196
|
-
"type": row["type"],
|
|
197
|
-
"weight": row["weight"],
|
|
198
|
-
"metadata": _safe_loads(row["metadata_json"]),
|
|
199
|
-
}
|
|
200
|
-
for row in edge_rows
|
|
201
|
-
]
|
|
202
|
-
|
|
203
|
-
if allowed_workspaces is not None:
|
|
204
|
-
nodes = self.filter_scoped_nodes(
|
|
205
|
-
nodes,
|
|
206
|
-
allowed_workspaces,
|
|
207
|
-
include_legacy_global=include_legacy_global,
|
|
208
|
-
)
|
|
209
|
-
kept_ids = {node["id"] for node in nodes}
|
|
210
|
-
edges = [e for e in edges if e["from"] in kept_ids and e["to"] in kept_ids]
|
|
211
|
-
|
|
212
|
-
degree_map: Dict[str, int] = {}
|
|
213
|
-
now = datetime.now()
|
|
214
|
-
node_by_id = {node["id"]: node for node in nodes}
|
|
215
|
-
topic_metrics: Dict[str, Dict[str, Any]] = {}
|
|
216
|
-
|
|
217
|
-
for edge in edges:
|
|
218
|
-
degree_map[edge["from"]] = degree_map.get(edge["from"], 0) + 1
|
|
219
|
-
degree_map[edge["to"]] = degree_map.get(edge["to"], 0) + 1
|
|
220
|
-
from_node = node_by_id.get(edge["from"])
|
|
221
|
-
to_node = node_by_id.get(edge["to"])
|
|
222
|
-
if not from_node or not to_node:
|
|
223
|
-
continue # pragma: no cover — unreachable: the edge query selects endpoints from the same node window
|
|
224
|
-
for topic_node, other_node in ((from_node, to_node), (to_node, from_node)):
|
|
225
|
-
if topic_node["type"] != "Topic":
|
|
226
|
-
continue
|
|
227
|
-
metrics = topic_metrics.setdefault(
|
|
228
|
-
topic_node["id"],
|
|
229
|
-
{
|
|
230
|
-
"mention_count": 0.0,
|
|
231
|
-
"conversation_ids": set(),
|
|
232
|
-
},
|
|
233
|
-
)
|
|
234
|
-
if edge["type"] in {"mentions", "discusses"}:
|
|
235
|
-
metrics["mention_count"] += max(
|
|
236
|
-
0.5, float(edge.get("weight") or 1.0)
|
|
237
|
-
)
|
|
238
|
-
other_meta = other_node.get("metadata") or {}
|
|
239
|
-
conversation_id = other_meta.get("conversation_id")
|
|
240
|
-
if other_node["type"] == "Conversation":
|
|
241
|
-
conversation_id = other_node["id"]
|
|
242
|
-
if conversation_id:
|
|
243
|
-
metrics["conversation_ids"].add(str(conversation_id))
|
|
244
|
-
|
|
245
|
-
type_max_raw: Dict[str, float] = {}
|
|
246
|
-
for node in nodes:
|
|
247
|
-
degree = degree_map.get(node["id"], 0)
|
|
248
|
-
recency = _recency_score(node.get("updated_at"), now=now)
|
|
249
|
-
metrics = {
|
|
250
|
-
"degree": degree,
|
|
251
|
-
"recency_score": round(recency, 4),
|
|
252
|
-
}
|
|
253
|
-
if node["type"] == "Topic":
|
|
254
|
-
topic_stat = topic_metrics.get(node["id"], {})
|
|
255
|
-
mention_count = float(topic_stat.get("mention_count") or 0.0)
|
|
256
|
-
conversation_count = len(topic_stat.get("conversation_ids") or ())
|
|
257
|
-
raw_importance = (
|
|
258
|
-
math.log1p(mention_count) * 2.8
|
|
259
|
-
+ math.log1p(conversation_count) * 2.2
|
|
260
|
-
+ recency * 1.4
|
|
261
|
-
+ math.sqrt(max(0, degree)) * 0.45
|
|
262
|
-
)
|
|
263
|
-
metrics.update(
|
|
264
|
-
{
|
|
265
|
-
"mention_count": round(mention_count, 2),
|
|
266
|
-
"conversation_count": conversation_count,
|
|
267
|
-
}
|
|
268
|
-
)
|
|
269
|
-
else:
|
|
270
|
-
raw_importance = math.log1p(max(0, degree)) * 1.4 + recency * 0.9
|
|
271
|
-
|
|
272
|
-
metrics["importance_raw"] = round(raw_importance, 4)
|
|
273
|
-
node["importance"] = round(raw_importance, 4)
|
|
274
|
-
node["_raw_importance"] = raw_importance
|
|
275
|
-
node["metadata"] = {
|
|
276
|
-
**(node.get("metadata") or {}),
|
|
277
|
-
"graph_metrics": metrics,
|
|
278
|
-
}
|
|
279
|
-
type_max_raw[node["type"]] = max(
|
|
280
|
-
type_max_raw.get(node["type"], 0.0), raw_importance
|
|
281
|
-
)
|
|
282
|
-
|
|
283
|
-
for node in nodes:
|
|
284
|
-
max_raw = max(type_max_raw.get(node["type"], 0.0), 0.0001)
|
|
285
|
-
importance_norm = min(1.0, (node.get("_raw_importance") or 0.0) / max_raw)
|
|
286
|
-
node["importance_norm"] = round(importance_norm, 4)
|
|
287
|
-
node["metadata"]["graph_metrics"]["importance_norm"] = node[
|
|
288
|
-
"importance_norm"
|
|
289
|
-
]
|
|
290
|
-
node.pop("_raw_importance", None)
|
|
291
|
-
return {"nodes": nodes, "edges": edges}
|
|
292
|
-
|
|
293
|
-
def search(
|
|
294
|
-
self,
|
|
295
|
-
query: str,
|
|
296
|
-
limit: int = 30,
|
|
297
|
-
*,
|
|
298
|
-
allowed_workspaces=None,
|
|
299
|
-
include_legacy_global: bool = False,
|
|
300
|
-
) -> Dict[str, Any]:
|
|
301
|
-
query = str(query or "").strip()
|
|
302
|
-
q = f"%{query}%"
|
|
303
|
-
limit = max(1, min(int(limit or 30), 100))
|
|
304
|
-
nt, et = self._read_tables()
|
|
305
|
-
with self._connect() as conn:
|
|
306
|
-
rows = []
|
|
307
|
-
if query:
|
|
308
|
-
fts_ids = self._fts_match_ids(conn, query, limit)
|
|
309
|
-
if fts_ids:
|
|
310
|
-
placeholders = ",".join("?" for _ in fts_ids)
|
|
311
|
-
by_id = {
|
|
312
|
-
row["id"]: row
|
|
313
|
-
for row in conn.execute(
|
|
314
|
-
f"""
|
|
315
|
-
SELECT id, type, title, summary, metadata_json, updated_at
|
|
316
|
-
FROM {nt} WHERE id IN ({placeholders})
|
|
317
|
-
""",
|
|
318
|
-
fts_ids,
|
|
319
|
-
).fetchall()
|
|
320
|
-
}
|
|
321
|
-
# Preserve FTS bm25 rank order.
|
|
322
|
-
rows = [by_id[i] for i in fts_ids if i in by_id]
|
|
323
|
-
else:
|
|
324
|
-
rows = conn.execute(
|
|
325
|
-
f"""
|
|
326
|
-
SELECT id, type, title, summary, metadata_json, updated_at
|
|
327
|
-
FROM {nt}
|
|
328
|
-
WHERE title LIKE ? OR summary LIKE ? OR metadata_json LIKE ?
|
|
329
|
-
ORDER BY updated_at DESC, id ASC
|
|
330
|
-
LIMIT ?
|
|
331
|
-
""",
|
|
332
|
-
(q, q, q, limit),
|
|
333
|
-
).fetchall()
|
|
334
|
-
|
|
335
|
-
if len(rows) < limit:
|
|
336
|
-
terms = _topic_candidates(query, limit=8)
|
|
337
|
-
if terms:
|
|
338
|
-
clauses = []
|
|
339
|
-
params: List[str] = []
|
|
340
|
-
for term in terms:
|
|
341
|
-
clauses.append(
|
|
342
|
-
"(title LIKE ? OR summary LIKE ? OR metadata_json LIKE ?)"
|
|
343
|
-
)
|
|
344
|
-
params.extend([f"%{term}%", f"%{term}%", f"%{term}%"])
|
|
345
|
-
extra = conn.execute(
|
|
346
|
-
f"""
|
|
347
|
-
SELECT id, type, title, summary, metadata_json, updated_at
|
|
348
|
-
FROM {nt}
|
|
349
|
-
WHERE {" OR ".join(clauses)}
|
|
350
|
-
ORDER BY updated_at DESC, id ASC
|
|
351
|
-
LIMIT ?
|
|
352
|
-
""",
|
|
353
|
-
(*params, limit * 3),
|
|
354
|
-
).fetchall()
|
|
355
|
-
by_id = {row["id"]: row for row in rows}
|
|
356
|
-
for row in extra:
|
|
357
|
-
by_id.setdefault(row["id"], row)
|
|
358
|
-
rows = list(by_id.values())
|
|
359
|
-
|
|
360
|
-
terms_for_score = set(_topic_candidates(query, limit=12))
|
|
361
|
-
|
|
362
|
-
def score(row: sqlite3.Row) -> tuple:
|
|
363
|
-
haystack = (
|
|
364
|
-
f"{row['title']} {row['summary']} {row['metadata_json']}".lower()
|
|
365
|
-
)
|
|
366
|
-
hits = sum(1 for term in terms_for_score if term.lower() in haystack)
|
|
367
|
-
type_boost = (
|
|
368
|
-
1
|
|
369
|
-
if row["type"]
|
|
370
|
-
in {
|
|
371
|
-
"Decision",
|
|
372
|
-
"Task",
|
|
373
|
-
"File",
|
|
374
|
-
"Document",
|
|
375
|
-
"CodeFile",
|
|
376
|
-
"Spreadsheet",
|
|
377
|
-
"SlideDeck",
|
|
378
|
-
"Image",
|
|
379
|
-
"ImageText",
|
|
380
|
-
"Audio",
|
|
381
|
-
"Page",
|
|
382
|
-
"Slide",
|
|
383
|
-
}
|
|
384
|
-
else 0
|
|
385
|
-
)
|
|
386
|
-
return (hits, type_boost, row["updated_at"] or "")
|
|
387
|
-
|
|
388
|
-
# Deterministic contract: rows with equal relevance order by id ASC
|
|
389
|
-
# (stable sort preserves the pre-sort under reverse=True), matching
|
|
390
|
-
# the legacy LIKE path regardless of FTS bm25 tie ordering.
|
|
391
|
-
rows = sorted(rows, key=lambda r: r["id"])
|
|
392
|
-
rows = sorted(rows, key=score, reverse=True)[:limit]
|
|
393
|
-
matches = [
|
|
394
|
-
{
|
|
395
|
-
"id": row["id"],
|
|
396
|
-
"type": row["type"],
|
|
397
|
-
"title": row["title"],
|
|
398
|
-
"summary": row["summary"],
|
|
399
|
-
"metadata": _safe_loads(row["metadata_json"]),
|
|
400
|
-
"updated_at": row["updated_at"],
|
|
401
|
-
}
|
|
402
|
-
for row in rows
|
|
403
|
-
]
|
|
404
|
-
if allowed_workspaces is not None:
|
|
405
|
-
matches = self.filter_scoped_nodes(
|
|
406
|
-
matches,
|
|
407
|
-
allowed_workspaces,
|
|
408
|
-
include_legacy_global=include_legacy_global,
|
|
409
|
-
)
|
|
410
|
-
return {"query": query, "matches": matches}
|
|
411
|
-
|
|
412
|
-
def hybrid_search(
|
|
413
|
-
self,
|
|
414
|
-
query: str,
|
|
415
|
-
*,
|
|
416
|
-
top_k: int = 20,
|
|
417
|
-
alpha: Optional[float] = None,
|
|
418
|
-
workspace_id: Optional[str] = None,
|
|
419
|
-
allowed_workspaces=None,
|
|
420
|
-
include_legacy_global: bool = False,
|
|
421
|
-
lexical_limit: Optional[int] = None,
|
|
422
|
-
vector_limit: Optional[int] = None,
|
|
423
|
-
min_vector_score: float = 0.0,
|
|
424
|
-
image_vector: Optional[Sequence[float]] = None,
|
|
425
|
-
image_fusion_weight: Optional[float] = None,
|
|
426
|
-
) -> Dict[str, Any]:
|
|
427
|
-
"""Unified lexical + vector retrieval with alpha-weighted linear fusion.
|
|
428
|
-
|
|
429
|
-
Runs the SQLite lexical :meth:`search` and the embedding-backed
|
|
430
|
-
``vector_search`` (sibling mixin via the store MRO), normalizes both
|
|
431
|
-
score spaces to ``[0, 1]``, fuses them as
|
|
432
|
-
``alpha * vector + (1 - alpha) * lexical`` (the same shape as
|
|
433
|
-
``lattice_brain.quality.HybridFusion`` — reimplemented here without
|
|
434
|
-
importing that module), and dedupes by ``node_id`` (chunk hits roll up
|
|
435
|
-
to their parent node).
|
|
436
|
-
|
|
437
|
-
Degrades gracefully: when the vector side is unavailable (mixin not
|
|
438
|
-
composed, embedder/index failure) the result falls back to
|
|
439
|
-
lexical-only ranking and reports ``mode: "lexical_only"`` with a
|
|
440
|
-
``detail`` explaining why. Each match carries per-source ``scores``
|
|
441
|
-
and a ``fusion`` field (``lexical`` / ``vector`` / ``both``).
|
|
442
|
-
|
|
443
|
-
``workspace_id`` is a convenience for single-workspace callers; the
|
|
444
|
-
richer ``allowed_workspaces`` set wins when both are provided.
|
|
445
|
-
|
|
446
|
-
``alpha=None`` (the default) resolves the vector share from the
|
|
447
|
-
single retrieval policy (:mod:`lattice_brain.graph.retrieval_policy`,
|
|
448
|
-
which wraps the query-class fusion table): fact 0.6 (the historical
|
|
449
|
-
default) / code 0.35 / person 0.45 / recency 0.5, config-overridable
|
|
450
|
-
via ``LATTICEAI_FUSION_WEIGHTS``. The policy also supplies a
|
|
451
|
-
deterministic rule-based query rewrite (echoed additively under
|
|
452
|
-
``"policy"``; the response ``"query"`` stays the original) and, for
|
|
453
|
-
the ``recency`` class only, an age-decay half-life that dampens each
|
|
454
|
-
fused score into the ``[0.5, 1.0]`` band (``scores.age_decay``).
|
|
455
|
-
Passing an explicit ``alpha`` pins it exactly as before and disables
|
|
456
|
-
rewrite + decay.
|
|
457
|
-
|
|
458
|
-
``image_vector`` (v11.1.0) is the *late fusion* seam for the separate
|
|
459
|
-
image space: the caller supplies a query vector from the same vision
|
|
460
|
-
model that embedded the pictures, its own index is ranked
|
|
461
|
-
independently, and only then are the two rankings blended
|
|
462
|
-
(``image_fusion_weight``, default 0.5). A text query never produces
|
|
463
|
-
one — it reaches images through their OCR text and captions — which is
|
|
464
|
-
exactly why the image channel has to enter at the end rather than
|
|
465
|
-
pretending to share the text index.
|
|
466
|
-
"""
|
|
467
|
-
query = str(query or "").strip()
|
|
468
|
-
try:
|
|
469
|
-
top_k = int(top_k)
|
|
470
|
-
except (TypeError, ValueError):
|
|
471
|
-
top_k = 20
|
|
472
|
-
top_k = max(1, min(top_k, 100))
|
|
473
|
-
query_class: Optional[str] = None
|
|
474
|
-
search_query = query
|
|
475
|
-
rewrite_rules: List[str] = []
|
|
476
|
-
recency_half_life_days: Optional[float] = None
|
|
477
|
-
# "alpha" is the historical linear fusion; the policy may select RRF
|
|
478
|
-
# per query class. An explicitly pinned ``alpha`` argument means the
|
|
479
|
-
# caller is asking for linear fusion by name, so it stays linear.
|
|
480
|
-
fusion_strategy = "alpha"
|
|
481
|
-
if alpha is None:
|
|
482
|
-
try:
|
|
483
|
-
from .retrieval_policy import resolve_policy
|
|
484
|
-
|
|
485
|
-
policy = resolve_policy(query)
|
|
486
|
-
query_class = policy["query_class"]
|
|
487
|
-
alpha = float(policy["alpha"])
|
|
488
|
-
fusion_strategy = str(policy.get("fusion_strategy") or "alpha")
|
|
489
|
-
rewrite_rules = list(policy.get("rewrite_rules") or [])
|
|
490
|
-
rewritten = str(policy.get("search_query") or "")
|
|
491
|
-
if rewritten and rewritten != query:
|
|
492
|
-
search_query = rewritten
|
|
493
|
-
half_life = policy.get("recency_half_life_days")
|
|
494
|
-
if half_life is not None:
|
|
495
|
-
recency_half_life_days = float(half_life)
|
|
496
|
-
except Exception: # noqa: BLE001 — policy resolution must never break search
|
|
497
|
-
alpha = 0.6
|
|
498
|
-
try:
|
|
499
|
-
alpha = float(alpha)
|
|
500
|
-
except (TypeError, ValueError):
|
|
501
|
-
alpha = 0.6
|
|
502
|
-
alpha = max(0.0, min(alpha, 1.0))
|
|
503
|
-
if allowed_workspaces is None and workspace_id:
|
|
504
|
-
allowed_workspaces = {str(workspace_id)}
|
|
505
|
-
|
|
506
|
-
if not query:
|
|
507
|
-
return {
|
|
508
|
-
"query": query,
|
|
509
|
-
"mode": "hybrid",
|
|
510
|
-
"alpha": alpha,
|
|
511
|
-
"query_class": query_class,
|
|
512
|
-
"top_k": top_k,
|
|
513
|
-
"sources": {"lexical": 0, "vector": 0},
|
|
514
|
-
"matches": [],
|
|
515
|
-
"policy": {"search_query": search_query, "rewrite_rules": rewrite_rules},
|
|
516
|
-
"fusion_strategy": fusion_strategy,
|
|
517
|
-
"detail": None,
|
|
518
|
-
}
|
|
519
|
-
|
|
520
|
-
lex_fetch = max(1, min(int(lexical_limit or max(top_k * 2, 20)), 100))
|
|
521
|
-
vec_fetch = max(1, min(int(vector_limit or max(top_k * 2, 20)), 100))
|
|
522
|
-
|
|
523
|
-
lexical_matches = self.search(
|
|
524
|
-
search_query,
|
|
525
|
-
lex_fetch,
|
|
526
|
-
allowed_workspaces=allowed_workspaces,
|
|
527
|
-
include_legacy_global=include_legacy_global,
|
|
528
|
-
).get("matches", [])
|
|
529
|
-
|
|
530
|
-
mode = "hybrid"
|
|
531
|
-
detail: Optional[str] = None
|
|
532
|
-
vector_matches: List[Dict[str, Any]] = []
|
|
533
|
-
vector_recall: Optional[Dict[str, Any]] = None
|
|
534
|
-
# The vector channel's own honesty block, echoed additively so a
|
|
535
|
-
# caller can tell an exact "not found" from an approximate one.
|
|
536
|
-
vector_meta: Dict[str, Any] = {
|
|
537
|
-
"backend": None,
|
|
538
|
-
"approx": None,
|
|
539
|
-
"exhaustive": None,
|
|
540
|
-
"truncated": None,
|
|
541
|
-
"embedded_rows": None,
|
|
542
|
-
"degraded": None,
|
|
543
|
-
}
|
|
544
|
-
vector_fn = getattr(self, "vector_search", None)
|
|
545
|
-
if not callable(vector_fn):
|
|
546
|
-
mode = "lexical_only"
|
|
547
|
-
detail = "vector search is not available on this store"
|
|
548
|
-
else:
|
|
549
|
-
try:
|
|
550
|
-
vector_payload = (
|
|
551
|
-
vector_fn(search_query, limit=vec_fetch, min_score=min_vector_score)
|
|
552
|
-
or {}
|
|
553
|
-
)
|
|
554
|
-
vector_matches = list(vector_payload.get("matches", []))
|
|
555
|
-
# Partial recall must reach the caller: the vector channel can
|
|
556
|
-
# only score a capped slice of a large index (see
|
|
557
|
-
# retrieval_vector.vector_search), and a fused answer built on
|
|
558
|
-
# a truncated scan is not the same claim as a complete one.
|
|
559
|
-
recall = vector_payload.get("recall")
|
|
560
|
-
if isinstance(recall, dict):
|
|
561
|
-
vector_meta["backend"] = recall.get("backend")
|
|
562
|
-
vector_meta["truncated"] = bool(recall.get("truncated"))
|
|
563
|
-
vector_meta["embedded_rows"] = recall.get("candidates_total")
|
|
564
|
-
if recall.get("truncated"):
|
|
565
|
-
vector_recall = dict(recall)
|
|
566
|
-
index_block = vector_payload.get("index")
|
|
567
|
-
if isinstance(index_block, dict):
|
|
568
|
-
vector_meta["approx"] = bool(index_block.get("approx"))
|
|
569
|
-
vector_meta["exhaustive"] = bool(index_block.get("exhaustive"))
|
|
570
|
-
except Exception as exc: # noqa: BLE001 — degrade, never fail the search
|
|
571
|
-
mode = "lexical_only"
|
|
572
|
-
detail = f"vector index unavailable: {exc}"
|
|
573
|
-
vector_matches = []
|
|
574
|
-
# An embedder swap makes the vector channel silently return zero rows
|
|
575
|
-
# (vector_search filters on the CURRENT model/dim). Surface the honest
|
|
576
|
-
# cause additively without changing the mode string.
|
|
577
|
-
vector_degraded: Optional[str] = None
|
|
578
|
-
if mode == "hybrid" and not vector_matches:
|
|
579
|
-
try:
|
|
580
|
-
fingerprint_fn = getattr(self, "embedder_fingerprint_status", None)
|
|
581
|
-
if callable(fingerprint_fn) and fingerprint_fn().get("stale_embedder"):
|
|
582
|
-
vector_degraded = "stale_embedder"
|
|
583
|
-
except Exception: # noqa: BLE001 — fingerprint status must never break search
|
|
584
|
-
vector_degraded = None
|
|
585
|
-
if vector_matches and allowed_workspaces is not None:
|
|
586
|
-
vector_matches = self.filter_scoped_nodes(
|
|
587
|
-
vector_matches,
|
|
588
|
-
allowed_workspaces,
|
|
589
|
-
id_key="node_id",
|
|
590
|
-
include_legacy_global=include_legacy_global,
|
|
591
|
-
)
|
|
592
|
-
|
|
593
|
-
def _parent_node_id(match: Dict[str, Any]) -> str:
|
|
594
|
-
# Chunk-level hits dedupe to their parent content node.
|
|
595
|
-
if match.get("type") == "Chunk":
|
|
596
|
-
meta = match.get("metadata") or {}
|
|
597
|
-
parent = meta.get("source_node") or meta.get("parent_source_node")
|
|
598
|
-
if parent:
|
|
599
|
-
return str(parent)
|
|
600
|
-
return str(match.get("node_id") or match.get("id") or "")
|
|
601
|
-
|
|
602
|
-
entries: Dict[str, Dict[str, Any]] = {}
|
|
603
|
-
|
|
604
|
-
def _entry_for(node_id: str, match: Dict[str, Any]) -> Dict[str, Any]:
|
|
605
|
-
entry = entries.get(node_id)
|
|
606
|
-
if entry is None:
|
|
607
|
-
entry = {
|
|
608
|
-
"node_id": node_id,
|
|
609
|
-
"id": match.get("id") or node_id,
|
|
610
|
-
"type": match.get("type"),
|
|
611
|
-
"title": match.get("title"),
|
|
612
|
-
"summary": match.get("summary"),
|
|
613
|
-
"metadata": match.get("metadata") or {},
|
|
614
|
-
"updated_at": match.get("updated_at"),
|
|
615
|
-
"scores": {"lexical": 0.0, "vector": 0.0},
|
|
616
|
-
"_lexical": False,
|
|
617
|
-
"_vector": False,
|
|
618
|
-
}
|
|
619
|
-
entries[node_id] = entry
|
|
620
|
-
return entry
|
|
621
|
-
|
|
622
|
-
# Per-channel id order (best first) — the only input RRF needs, and
|
|
623
|
-
# the one thing a normalized score cannot reconstruct.
|
|
624
|
-
lexical_order: List[str] = []
|
|
625
|
-
vector_order: List[str] = []
|
|
626
|
-
|
|
627
|
-
for rank, match in enumerate(lexical_matches, start=1):
|
|
628
|
-
node_id = _parent_node_id(match)
|
|
629
|
-
if not node_id:
|
|
630
|
-
continue
|
|
631
|
-
entry = _entry_for(node_id, match)
|
|
632
|
-
entry["scores"]["lexical"] = max(
|
|
633
|
-
entry["scores"]["lexical"], round(1.0 / rank, 6)
|
|
634
|
-
)
|
|
635
|
-
entry["_lexical"] = True
|
|
636
|
-
lexical_order.append(node_id)
|
|
637
|
-
|
|
638
|
-
# Max-normalize cosine scores into [0, 1] (guard the score-0 falsy trap
|
|
639
|
-
# by comparing explicitly, never with truthiness).
|
|
640
|
-
max_vec = 0.0
|
|
641
|
-
for match in vector_matches:
|
|
642
|
-
raw = match.get("score")
|
|
643
|
-
if raw is not None and float(raw) > max_vec:
|
|
644
|
-
max_vec = float(raw)
|
|
645
|
-
for match in vector_matches:
|
|
646
|
-
node_id = _parent_node_id(match)
|
|
647
|
-
if not node_id:
|
|
648
|
-
continue
|
|
649
|
-
raw = float(match.get("score") or 0.0)
|
|
650
|
-
vec_norm = max(0.0, raw) / max_vec if max_vec > 0 else 0.0
|
|
651
|
-
entry = _entry_for(node_id, match)
|
|
652
|
-
entry["scores"]["vector"] = max(entry["scores"]["vector"], round(vec_norm, 6))
|
|
653
|
-
entry["_vector"] = True
|
|
654
|
-
vector_order.append(node_id)
|
|
655
|
-
# Prefer a real snippet when the lexical row had no summary.
|
|
656
|
-
if not entry.get("summary") and match.get("summary"):
|
|
657
|
-
entry["summary"] = match.get("summary")
|
|
658
|
-
|
|
659
|
-
# Graph traversal candidate expansion (opt-in, capped, counted): pull
|
|
660
|
-
# the one-hop neighbours of the strongest hits into the candidate pool
|
|
661
|
-
# so an answer that is adjacent to the match — not in it — is
|
|
662
|
-
# reachable at all. Off by default; see fusion.GRAPH_EXPANSION_ENV.
|
|
663
|
-
expansion_report: Dict[str, Any] = {
|
|
664
|
-
"enabled": False,
|
|
665
|
-
"seeds": 0,
|
|
666
|
-
"added": 0,
|
|
667
|
-
"cap": DEFAULT_EXPANSION_CAP,
|
|
668
|
-
"truncated": False,
|
|
669
|
-
"failed_seeds": 0,
|
|
670
|
-
}
|
|
671
|
-
if entries and graph_expansion_enabled():
|
|
672
|
-
seeds = sorted(
|
|
673
|
-
(
|
|
674
|
-
(node_id, float(entry["scores"]["vector"]))
|
|
675
|
-
for node_id, entry in entries.items()
|
|
676
|
-
),
|
|
677
|
-
key=lambda pair: -pair[1],
|
|
678
|
-
)[:DEFAULT_EXPANSION_SEEDS]
|
|
679
|
-
expanded, expansion_report = expand_with_neighbors(
|
|
680
|
-
seeds,
|
|
681
|
-
self.neighbors,
|
|
682
|
-
exclude=list(entries),
|
|
683
|
-
cap=DEFAULT_EXPANSION_CAP,
|
|
684
|
-
)
|
|
685
|
-
for candidate in expanded:
|
|
686
|
-
node = candidate["node"]
|
|
687
|
-
entry = _entry_for(str(node.get("id")), dict(node))
|
|
688
|
-
entry["scores"]["graph"] = candidate["score"]
|
|
689
|
-
entry["metadata"] = {
|
|
690
|
-
**(entry.get("metadata") or {}),
|
|
691
|
-
"expanded_from": candidate["seed"],
|
|
692
|
-
}
|
|
693
|
-
entry["_graph"] = True
|
|
694
|
-
|
|
695
|
-
rrf_normalized: Dict[str, float] = {}
|
|
696
|
-
if fusion_strategy == "rrf":
|
|
697
|
-
raw_rrf = rrf_fuse(
|
|
698
|
-
{
|
|
699
|
-
"lexical": list(dict.fromkeys(lexical_order)),
|
|
700
|
-
"vector": list(dict.fromkeys(vector_order)),
|
|
701
|
-
}
|
|
702
|
-
)
|
|
703
|
-
peak = max(raw_rrf.values(), default=0.0)
|
|
704
|
-
if peak > 0:
|
|
705
|
-
# Rescale to [0, 1] so the score column keeps the same meaning
|
|
706
|
-
# across strategies; RRF's raw values live around 1/60.
|
|
707
|
-
rrf_normalized = {key: value / peak for key, value in raw_rrf.items()}
|
|
708
|
-
|
|
709
|
-
matches: List[Dict[str, Any]] = []
|
|
710
|
-
for entry in entries.values():
|
|
711
|
-
lex_score = float(entry["scores"]["lexical"])
|
|
712
|
-
vec_score = float(entry["scores"]["vector"])
|
|
713
|
-
if mode == "lexical_only":
|
|
714
|
-
fused = lex_score
|
|
715
|
-
elif fusion_strategy == "rrf":
|
|
716
|
-
fused = float(rrf_normalized.get(entry["node_id"], 0.0))
|
|
717
|
-
entry["scores"]["rrf"] = round(fused, 6)
|
|
718
|
-
else:
|
|
719
|
-
fused = alpha * vec_score + (1.0 - alpha) * lex_score
|
|
720
|
-
from_lexical = bool(entry.pop("_lexical", False))
|
|
721
|
-
from_vector = bool(entry.pop("_vector", False))
|
|
722
|
-
if entry.pop("_graph", False):
|
|
723
|
-
# A one-hop neighbour of a hit: related to the answer, never
|
|
724
|
-
# itself a match, so it carries only its damped seed score.
|
|
725
|
-
fused = float(entry["scores"]["graph"])
|
|
726
|
-
entry["fusion"] = "graph"
|
|
727
|
-
elif from_lexical and from_vector:
|
|
728
|
-
entry["fusion"] = "both"
|
|
729
|
-
elif from_vector:
|
|
730
|
-
entry["fusion"] = "vector"
|
|
731
|
-
else:
|
|
732
|
-
entry["fusion"] = "lexical"
|
|
733
|
-
entry["score"] = round(fused, 6)
|
|
734
|
-
matches.append(entry)
|
|
735
|
-
|
|
736
|
-
# Recency-class age decay (retrieval_policy): dampen each fused score
|
|
737
|
-
# into the [0.5, 1.0] band so old-but-relevant items sink without ever
|
|
738
|
-
# being zeroed. Other classes skip this block byte-identically.
|
|
739
|
-
if recency_half_life_days is not None:
|
|
740
|
-
decay_now = datetime.now()
|
|
741
|
-
for match in matches:
|
|
742
|
-
stamp = match.get("updated_at")
|
|
743
|
-
if _parse_iso(stamp):
|
|
744
|
-
multiplier = 0.5 + 0.5 * _recency_score(
|
|
745
|
-
stamp, now=decay_now, half_life_days=recency_half_life_days
|
|
746
|
-
)
|
|
747
|
-
else:
|
|
748
|
-
# Unknown age is not evidence of staleness — never dampen.
|
|
749
|
-
multiplier = 1.0
|
|
750
|
-
match["scores"]["age_decay"] = round(multiplier, 6)
|
|
751
|
-
match["score"] = round(float(match["score"]) * multiplier, 6)
|
|
752
|
-
|
|
753
|
-
# Late fusion of the image space (v11.1.0). Runs after the text
|
|
754
|
-
# channels have produced a ranking and before the cut, so image
|
|
755
|
-
# evidence can lift a picture into the answer without ever having been
|
|
756
|
-
# compared against a text vector.
|
|
757
|
-
image_fusion: Optional[Dict[str, Any]] = None
|
|
758
|
-
if image_vector is not None:
|
|
759
|
-
image_fusion = self._fuse_image_channel(
|
|
760
|
-
matches, image_vector, top_k=top_k, weight=image_fusion_weight
|
|
761
|
-
)
|
|
762
|
-
|
|
763
|
-
matches.sort(key=lambda item: (-item["score"], item["node_id"]))
|
|
764
|
-
# Optional cross-encoder rerank (v9.9.5). Off by default; when the
|
|
765
|
-
# env kill-switch is set and the model loads, pair scores reorder the
|
|
766
|
-
# fused list. Failures degrade to identity and never break search.
|
|
767
|
-
rerank_meta: Dict[str, Any]
|
|
768
|
-
try:
|
|
769
|
-
from .rerank import rerank_matches
|
|
770
|
-
|
|
771
|
-
# Rerank a slightly wider window, then cut to top_k.
|
|
772
|
-
window = matches[: max(top_k * 2, top_k)]
|
|
773
|
-
reranked = rerank_matches(search_query, window, top_k=top_k)
|
|
774
|
-
matches = list(reranked.get("matches") or matches[:top_k])
|
|
775
|
-
rerank_meta = {
|
|
776
|
-
"mode": reranked.get("mode") or "identity",
|
|
777
|
-
"model": reranked.get("model"),
|
|
778
|
-
"detail": reranked.get("detail"),
|
|
779
|
-
}
|
|
780
|
-
except Exception as exc: # noqa: BLE001 — rerank must never break search
|
|
781
|
-
matches = matches[:top_k]
|
|
782
|
-
rerank_meta = {"mode": "identity", "model": None, "detail": str(exc)}
|
|
783
|
-
for rank, match in enumerate(matches, start=1):
|
|
784
|
-
match["rank"] = rank
|
|
785
|
-
result = {
|
|
786
|
-
"query": query,
|
|
787
|
-
"mode": mode,
|
|
788
|
-
"alpha": alpha,
|
|
789
|
-
"query_class": query_class,
|
|
790
|
-
"top_k": top_k,
|
|
791
|
-
"sources": {"lexical": len(lexical_matches), "vector": len(vector_matches)},
|
|
792
|
-
"matches": matches,
|
|
793
|
-
"policy": {"search_query": search_query, "rewrite_rules": rewrite_rules},
|
|
794
|
-
"fusion_strategy": fusion_strategy,
|
|
795
|
-
"graph_expansion": expansion_report,
|
|
796
|
-
"rerank": rerank_meta,
|
|
797
|
-
"detail": detail,
|
|
798
|
-
}
|
|
799
|
-
if vector_degraded is not None:
|
|
800
|
-
result["vector_degraded"] = vector_degraded
|
|
801
|
-
if vector_recall is not None:
|
|
802
|
-
result["vector_recall"] = vector_recall
|
|
803
|
-
if vector_degraded is None:
|
|
804
|
-
result["vector_degraded"] = "partial_recall"
|
|
805
|
-
vector_meta["degraded"] = result.get("vector_degraded")
|
|
806
|
-
result["vector"] = vector_meta
|
|
807
|
-
multimodal = multimodal_signal(matches)
|
|
808
|
-
if multimodal is not None or image_fusion is not None:
|
|
809
|
-
result["multimodal"] = {
|
|
810
|
-
**(multimodal or {"images": 0, "types": []}),
|
|
811
|
-
**({"image_fusion": image_fusion} if image_fusion is not None else {}),
|
|
812
|
-
}
|
|
813
|
-
return result
|
|
814
|
-
|
|
815
|
-
def _fuse_image_channel(
|
|
816
|
-
self,
|
|
817
|
-
matches: List[Dict[str, Any]],
|
|
818
|
-
image_vector: Sequence[float],
|
|
819
|
-
*,
|
|
820
|
-
top_k: int,
|
|
821
|
-
weight: Optional[float],
|
|
822
|
-
) -> Dict[str, Any]:
|
|
823
|
-
"""Rank the image index separately, then blend it into ``matches``.
|
|
824
|
-
|
|
825
|
-
Any failure degrades to "the image channel contributed nothing" with
|
|
826
|
-
the reason attached — an image index that cannot be read is not a
|
|
827
|
-
reason to lose the text answer.
|
|
828
|
-
"""
|
|
829
|
-
from .image_vectors import (
|
|
830
|
-
DEFAULT_IMAGE_FUSION_WEIGHT,
|
|
831
|
-
fuse_image_scores,
|
|
832
|
-
image_similarity_search,
|
|
833
|
-
)
|
|
834
|
-
|
|
835
|
-
share = DEFAULT_IMAGE_FUSION_WEIGHT if weight is None else float(weight)
|
|
836
|
-
report: Dict[str, Any] = {
|
|
837
|
-
"weight": round(max(0.0, min(1.0, share)), 4),
|
|
838
|
-
"candidates": 0,
|
|
839
|
-
"fused": 0,
|
|
840
|
-
"detail": None,
|
|
841
|
-
}
|
|
842
|
-
try:
|
|
843
|
-
found = image_similarity_search(
|
|
844
|
-
self, image_vector, top_k=max(1, int(top_k) * 2)
|
|
845
|
-
)
|
|
846
|
-
except Exception as exc: # noqa: BLE001 — never fail the text answer
|
|
847
|
-
report["detail"] = f"image index unavailable: {exc}"
|
|
848
|
-
return report
|
|
849
|
-
report["candidates"] = int(found.get("candidates") or 0)
|
|
850
|
-
report["detail"] = found.get("detail")
|
|
851
|
-
scores = {
|
|
852
|
-
str(row.get("node_id")): float(row.get("score") or 0.0)
|
|
853
|
-
for row in found.get("matches") or []
|
|
854
|
-
}
|
|
855
|
-
report["fused"] = fuse_image_scores(matches, scores, weight=share)
|
|
856
|
-
return report
|
|
857
|
-
|
|
858
|
-
def context_for_query(
|
|
859
|
-
self,
|
|
860
|
-
query: str,
|
|
861
|
-
limit: int = 6,
|
|
862
|
-
*,
|
|
863
|
-
allowed_workspaces=None,
|
|
864
|
-
include_legacy_global: bool = False,
|
|
865
|
-
use_hybrid: bool = False,
|
|
866
|
-
with_meta: bool = False,
|
|
867
|
-
):
|
|
868
|
-
"""Return compact graph-backed RAG context for chat generation.
|
|
869
|
-
|
|
870
|
-
``use_hybrid=True`` sources the matches from :meth:`hybrid_search`
|
|
871
|
-
(lexical + vector fusion) instead of the lexical-only :meth:`search`.
|
|
872
|
-
Default behavior is unchanged, and any hybrid failure silently falls
|
|
873
|
-
back to the legacy lexical path.
|
|
874
|
-
|
|
875
|
-
``with_meta=True`` (additive, v9.8.0) returns
|
|
876
|
-
``{"context": str, "quality": {...}}`` instead of the bare string.
|
|
877
|
-
``quality`` follows :func:`context_quality_signal` and honestly
|
|
878
|
-
reports how the context was retrieved (hybrid vs lexical-only
|
|
879
|
-
fallback vs nothing). The ``context`` value is byte-identical to the
|
|
880
|
-
default ``with_meta=False`` return for the same arguments.
|
|
881
|
-
"""
|
|
882
|
-
query = str(query or "").strip()
|
|
883
|
-
if not query:
|
|
884
|
-
if with_meta:
|
|
885
|
-
return {
|
|
886
|
-
"context": "",
|
|
887
|
-
"quality": context_quality_signal(
|
|
888
|
-
"none", 0, reason="질의가 비어 있습니다"
|
|
889
|
-
),
|
|
890
|
-
}
|
|
891
|
-
return ""
|
|
892
|
-
matches: List[Dict[str, Any]] = []
|
|
893
|
-
retrieval_mode = "none"
|
|
894
|
-
vector_meta: Optional[Dict[str, Any]] = None
|
|
895
|
-
multimodal_meta: Optional[Dict[str, Any]] = None
|
|
896
|
-
if use_hybrid:
|
|
897
|
-
try:
|
|
898
|
-
hybrid = self.hybrid_search(
|
|
899
|
-
query,
|
|
900
|
-
top_k=limit,
|
|
901
|
-
allowed_workspaces=allowed_workspaces,
|
|
902
|
-
include_legacy_global=include_legacy_global,
|
|
903
|
-
)
|
|
904
|
-
matches = hybrid.get("matches", [])
|
|
905
|
-
vector_block = hybrid.get("vector") or {}
|
|
906
|
-
# Only a caveat is worth carrying: approximate scoring, a
|
|
907
|
-
# truncated candidate scan, or an already-flagged degradation.
|
|
908
|
-
if (
|
|
909
|
-
vector_block.get("approx")
|
|
910
|
-
or vector_block.get("truncated")
|
|
911
|
-
or vector_block.get("degraded")
|
|
912
|
-
):
|
|
913
|
-
vector_meta = dict(vector_block)
|
|
914
|
-
if matches:
|
|
915
|
-
retrieval_mode = str(hybrid.get("mode") or "hybrid")
|
|
916
|
-
except Exception: # noqa: BLE001 — context building must never fail
|
|
917
|
-
matches = []
|
|
918
|
-
if not matches:
|
|
919
|
-
matches = self.search(
|
|
920
|
-
query,
|
|
921
|
-
limit,
|
|
922
|
-
allowed_workspaces=allowed_workspaces,
|
|
923
|
-
include_legacy_global=include_legacy_global,
|
|
924
|
-
).get("matches", [])
|
|
925
|
-
if matches:
|
|
926
|
-
retrieval_mode = "lexical_only"
|
|
927
|
-
if not matches:
|
|
928
|
-
topics = _topic_candidates(query, limit=4)
|
|
929
|
-
if topics:
|
|
930
|
-
nt, et = self._read_tables()
|
|
931
|
-
with self._connect() as conn:
|
|
932
|
-
rows = []
|
|
933
|
-
for topic in topics:
|
|
934
|
-
rows.extend(
|
|
935
|
-
conn.execute(
|
|
936
|
-
f"""
|
|
937
|
-
SELECT id, type, title, summary, metadata_json
|
|
938
|
-
FROM {nt}
|
|
939
|
-
WHERE title LIKE ? OR metadata_json LIKE ?
|
|
940
|
-
ORDER BY updated_at DESC, id ASC
|
|
941
|
-
LIMIT 3
|
|
942
|
-
""",
|
|
943
|
-
(f"%{topic}%", f"%{topic}%"),
|
|
944
|
-
).fetchall()
|
|
945
|
-
)
|
|
946
|
-
seen = set()
|
|
947
|
-
matches = []
|
|
948
|
-
for row in rows:
|
|
949
|
-
if row["id"] in seen:
|
|
950
|
-
continue
|
|
951
|
-
seen.add(row["id"])
|
|
952
|
-
matches.append(
|
|
953
|
-
{
|
|
954
|
-
"id": row["id"],
|
|
955
|
-
"type": row["type"],
|
|
956
|
-
"title": row["title"],
|
|
957
|
-
"summary": row["summary"],
|
|
958
|
-
"metadata": _safe_loads(row["metadata_json"]),
|
|
959
|
-
}
|
|
960
|
-
)
|
|
961
|
-
if len(matches) >= limit:
|
|
962
|
-
break
|
|
963
|
-
if allowed_workspaces is not None:
|
|
964
|
-
matches = self.filter_scoped_nodes(
|
|
965
|
-
matches,
|
|
966
|
-
allowed_workspaces,
|
|
967
|
-
include_legacy_global=include_legacy_global,
|
|
968
|
-
)
|
|
969
|
-
if matches:
|
|
970
|
-
retrieval_mode = "lexical_only"
|
|
971
|
-
lines = []
|
|
972
|
-
for match in matches[:limit]:
|
|
973
|
-
meta = match.get("metadata") or {}
|
|
974
|
-
source = (
|
|
975
|
-
meta.get("relative_path")
|
|
976
|
-
or meta.get("filename")
|
|
977
|
-
or meta.get("conversation_id")
|
|
978
|
-
or meta.get("source")
|
|
979
|
-
or match["id"]
|
|
980
|
-
)
|
|
981
|
-
summary = _clean_text(match.get("summary") or "")[:700]
|
|
982
|
-
lines.append(
|
|
983
|
-
f"- [{match['type']}] {match['title']} | source={source} | {summary}"
|
|
984
|
-
)
|
|
985
|
-
context = "\n".join(lines)
|
|
986
|
-
if not with_meta:
|
|
987
|
-
return context
|
|
988
|
-
# Only the context that actually reached the model counts as
|
|
989
|
-
# multimodal — matches trimmed by ``limit`` are not in the answer.
|
|
990
|
-
multimodal_meta = multimodal_signal(matches[:limit])
|
|
991
|
-
return {
|
|
992
|
-
"context": context,
|
|
993
|
-
"quality": context_quality_signal(
|
|
994
|
-
retrieval_mode,
|
|
995
|
-
len(matches[:limit]),
|
|
996
|
-
vector=vector_meta,
|
|
997
|
-
multimodal=multimodal_meta,
|
|
998
|
-
),
|
|
999
|
-
}
|
|
1000
|
-
|
|
1001
|
-
def context_for_query_with_meta(
|
|
1002
|
-
self,
|
|
1003
|
-
query: str,
|
|
1004
|
-
limit: int = 6,
|
|
1005
|
-
*,
|
|
1006
|
-
allowed_workspaces=None,
|
|
1007
|
-
include_legacy_global: bool = False,
|
|
1008
|
-
use_hybrid: bool = True,
|
|
1009
|
-
) -> Dict[str, Any]:
|
|
1010
|
-
"""Additive companion to :meth:`context_for_query` (v9.8.0).
|
|
1011
|
-
|
|
1012
|
-
Always returns ``{"context": str, "quality": {...}}`` so chat callers
|
|
1013
|
-
can surface an honest retrieval signal without changing the legacy
|
|
1014
|
-
string-returning contract. Defaults to hybrid retrieval because meta
|
|
1015
|
-
consumers want the vector-fallback signal; pass ``use_hybrid=False``
|
|
1016
|
-
for the legacy lexical-only sourcing.
|
|
1017
|
-
"""
|
|
1018
|
-
return self.context_for_query(
|
|
1019
|
-
query,
|
|
1020
|
-
limit,
|
|
1021
|
-
allowed_workspaces=allowed_workspaces,
|
|
1022
|
-
include_legacy_global=include_legacy_global,
|
|
1023
|
-
use_hybrid=use_hybrid,
|
|
1024
|
-
with_meta=True,
|
|
1025
|
-
)
|
|
1026
|
-
|
|
1027
|
-
def delete_conversation(self, conversation_id: str) -> Dict[str, Any]:
|
|
1028
|
-
conversation_id = str(conversation_id or "").strip()
|
|
1029
|
-
if not conversation_id:
|
|
1030
|
-
return {"status": "skipped", "removed_nodes": 0}
|
|
1031
|
-
conv_id = f"conversation:{_slug(conversation_id)}"
|
|
1032
|
-
with self._connect() as conn:
|
|
1033
|
-
# Edge rows may carry the legacy lowercase label (pre-v4) or the
|
|
1034
|
-
# canonical EdgeType value (v4 write door) — match both.
|
|
1035
|
-
direct_ids = [
|
|
1036
|
-
row["to_node"]
|
|
1037
|
-
for row in conn.execute(
|
|
1038
|
-
"SELECT to_node FROM edges WHERE from_node=? AND type IN ('contains', 'CONTAINS')",
|
|
1039
|
-
(conv_id,),
|
|
1040
|
-
)
|
|
1041
|
-
]
|
|
1042
|
-
remove_ids = set(direct_ids)
|
|
1043
|
-
child_types = [
|
|
1044
|
-
"has_chunk",
|
|
1045
|
-
"implies",
|
|
1046
|
-
"contains_signal",
|
|
1047
|
-
"has_page",
|
|
1048
|
-
"has_slide",
|
|
1049
|
-
"has_sheet",
|
|
1050
|
-
"contains_image",
|
|
1051
|
-
]
|
|
1052
|
-
child_types += [t.upper() for t in child_types]
|
|
1053
|
-
placeholders = ",".join("?" for _ in child_types)
|
|
1054
|
-
for source_id in list(direct_ids):
|
|
1055
|
-
for row in conn.execute(
|
|
1056
|
-
f"SELECT to_node FROM edges WHERE from_node=? AND type IN ({placeholders})",
|
|
1057
|
-
(source_id, *child_types),
|
|
1058
|
-
):
|
|
1059
|
-
remove_ids.add(row["to_node"])
|
|
1060
|
-
remove_ids.add(conv_id)
|
|
1061
|
-
for node_id in remove_ids:
|
|
1062
|
-
conn.execute("DELETE FROM nodes WHERE id=?", (node_id,))
|
|
1063
|
-
if KGStoreV2 is not None:
|
|
1064
|
-
conn.execute(
|
|
1065
|
-
"DELETE FROM nodes_v2 WHERE id=?", (node_id,)
|
|
1066
|
-
) # edges_v2 cascade
|
|
1067
|
-
conn.execute(
|
|
1068
|
-
"""
|
|
1069
|
-
DELETE FROM nodes
|
|
1070
|
-
WHERE type='Topic'
|
|
1071
|
-
AND id NOT IN (SELECT to_node FROM edges)
|
|
1072
|
-
AND id NOT IN (SELECT from_node FROM edges)
|
|
1073
|
-
"""
|
|
1074
|
-
)
|
|
1075
|
-
if KGStoreV2 is not None:
|
|
1076
|
-
conn.execute(
|
|
1077
|
-
"""
|
|
1078
|
-
DELETE FROM nodes_v2
|
|
1079
|
-
WHERE legacy_type='Topic'
|
|
1080
|
-
AND id NOT IN (SELECT target FROM edges_v2)
|
|
1081
|
-
AND id NOT IN (SELECT source FROM edges_v2)
|
|
1082
|
-
"""
|
|
1083
|
-
)
|
|
1084
|
-
return {
|
|
1085
|
-
"status": "ok",
|
|
1086
|
-
"conversation_id": conversation_id,
|
|
1087
|
-
"removed_nodes": len(remove_ids),
|
|
1088
|
-
}
|
|
1089
|
-
|
|
1090
|
-
def clear_all(self) -> Dict[str, Any]:
|
|
1091
|
-
with self._connect() as conn:
|
|
1092
|
-
counts = {
|
|
1093
|
-
"nodes": conn.execute("SELECT COUNT(*) AS c FROM nodes").fetchone()[
|
|
1094
|
-
"c"
|
|
1095
|
-
],
|
|
1096
|
-
"edges": conn.execute("SELECT COUNT(*) AS c FROM edges").fetchone()[
|
|
1097
|
-
"c"
|
|
1098
|
-
],
|
|
1099
|
-
"chunks": conn.execute("SELECT COUNT(*) AS c FROM chunks").fetchone()[
|
|
1100
|
-
"c"
|
|
1101
|
-
],
|
|
1102
|
-
"knowledge_sources": conn.execute(
|
|
1103
|
-
"SELECT COUNT(*) AS c FROM knowledge_sources"
|
|
1104
|
-
).fetchone()["c"],
|
|
1105
|
-
"local_file_index": conn.execute(
|
|
1106
|
-
"SELECT COUNT(*) AS c FROM local_file_index"
|
|
1107
|
-
).fetchone()["c"],
|
|
1108
|
-
}
|
|
1109
|
-
conn.execute("DELETE FROM local_file_index")
|
|
1110
|
-
conn.execute("DELETE FROM knowledge_sources")
|
|
1111
|
-
conn.execute("DELETE FROM chunks")
|
|
1112
|
-
conn.execute("DELETE FROM edges")
|
|
1113
|
-
conn.execute("DELETE FROM nodes")
|
|
1114
|
-
if KGStoreV2 is not None:
|
|
1115
|
-
conn.execute("DELETE FROM edges_v2")
|
|
1116
|
-
conn.execute("DELETE FROM nodes_v2")
|
|
1117
|
-
if self.blob_dir.exists():
|
|
1118
|
-
shutil.rmtree(self.blob_dir, ignore_errors=True)
|
|
1119
|
-
self.blob_dir.mkdir(parents=True, exist_ok=True)
|
|
1120
|
-
return {"status": "ok", "removed": counts}
|