ltcai 10.9.0 → 11.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -64
- package/docs/CHANGELOG.md +72 -237
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +1 -1
- package/docs/PERFORMANCE.md +78 -7
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +1 -1
- package/docs/v11.1.0_PRODUCT_INTELLIGENCE_PLAN.md +313 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_contract.py +11 -0
- package/lattice_brain/graph/curator.py +1 -1
- package/lattice_brain/graph/fusion.py +184 -3
- package/lattice_brain/graph/proactive.py +138 -1
- package/lattice_brain/graph/projection.py +10 -2
- package/lattice_brain/graph/retrieval.py +137 -7
- package/lattice_brain/graph/retrieval_docgen.py +20 -20
- package/lattice_brain/graph/retrieval_policy.py +6 -0
- package/lattice_brain/graph/retrieval_reads.py +188 -1
- package/lattice_brain/graph/retrieval_vector.py +474 -110
- package/lattice_brain/graph/schema.py +125 -2
- package/lattice_brain/graph/vector_index/__init__.py +85 -0
- package/lattice_brain/graph/vector_index/base.py +170 -0
- package/lattice_brain/graph/vector_index/brute_force.py +114 -0
- package/lattice_brain/graph/vector_index/hnsw.py +293 -0
- package/lattice_brain/graph/vector_index/jobs.py +287 -0
- package/lattice_brain/graph/vector_index/quantized.py +151 -0
- package/lattice_brain/graph/vector_index/selector.py +131 -0
- package/lattice_brain/ingestion.py +50 -3
- package/lattice_brain/portability.py +654 -2
- package/lattice_brain/runtime/agent_runtime.py +1 -1
- package/lattice_brain/runtime/contracts.py +1 -1
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/lattice_brain/self_model.py +620 -0
- package/lattice_brain/synthesis.py +801 -0
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/brain_intelligence.py +83 -1
- package/latticeai/api/chat_stream.py +4 -1
- package/latticeai/api/local_files.py +62 -0
- package/latticeai/api/models.py +1 -1
- package/latticeai/api/portability.py +130 -1
- package/latticeai/api/security_dashboard.py +48 -13
- package/latticeai/api/voice_capture.py +4 -1
- package/latticeai/api/workspace.py +11 -5
- package/latticeai/core/embedding_providers.py +20 -1
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +14 -0
- package/latticeai/core/model_compat.py +2 -2
- package/latticeai/core/tool_registry.py +0 -7
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/core/workspace_os_utils.py +4 -50
- package/latticeai/core/workspace_review_items.py +12 -1
- package/latticeai/integrations/telegram_bot.py +28 -10
- package/latticeai/models/router.py +1 -1
- package/latticeai/runtime/access_runtime.py +1 -1
- package/latticeai/runtime/network_boundary_wiring.py +9 -5
- package/latticeai/runtime/permission_mode_wiring.py +9 -6
- package/latticeai/runtime/router_registration.py +3 -0
- package/latticeai/services/architecture_readiness.py +1 -1
- package/latticeai/services/brain_intelligence.py +253 -0
- package/latticeai/services/memory_service.py +1 -1
- package/latticeai/services/model_catalog.py +4 -3
- package/latticeai/services/model_engines.py +28 -14
- package/latticeai/services/obsidian_bridge.py +618 -0
- package/latticeai/services/product_readiness.py +5 -3
- package/latticeai/tools/filesystem.py +4 -1
- package/package.json +1 -1
- package/scripts/bench_vector_index.py +295 -0
- package/scripts/check_current_release_docs.mjs +4 -2
- package/scripts/release_screen_claims.json +31 -0
- package/src-tauri/Cargo.lock +1 -1
- package/src-tauri/Cargo.toml +1 -1
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +37 -37
- package/static/app/assets/{Act-CS9IeqUX.js → Act-D4zSxFR-.js} +1 -1
- package/static/app/assets/{AdminConsole-3UkIEWGA.js → AdminConsole-w5jBfPt2.js} +1 -1
- package/static/app/assets/{Brain-B22EmNqS.js → Brain-C2EqQg74.js} +2 -2
- package/static/app/assets/BrainHome-CvXS6XiQ.js +2 -0
- package/static/app/assets/BrainSignals-DOE_KhOU.js +1 -0
- package/static/app/assets/Capture-DPqpGK8d.js +1 -0
- package/static/app/assets/{CommandPalette-86m4FCcN.js → CommandPalette-CNf7h5fp.js} +1 -1
- package/static/app/assets/Library-BN0HYOfc.js +1 -0
- package/static/app/assets/LivingBrain-Dfq_wEDI.js +1 -0
- package/static/app/assets/ProductFlow-B-3O0rNV.js +1 -0
- package/static/app/assets/{ReviewCard-BepjSDpN.js → ReviewCard-gZ-tdqFM.js} +1 -1
- package/static/app/assets/System-BElUcSSw.js +1 -0
- package/static/app/assets/arrow-left-CFNIMjhv.js +1 -0
- package/static/app/assets/{bot-DQj0-LkM.js → bot--qYHMtkP.js} +1 -1
- package/static/app/assets/brain-DDCLjRqO.js +1 -0
- package/static/app/assets/{button-CmTknyAP.js → button-51Z3rsuv.js} +1 -1
- package/static/app/assets/{circle-pause-yTCWRziJ.js → circle-pause-CMIiMaQl.js} +1 -1
- package/static/app/assets/{circle-play-Ccrva84R.js → circle-play-DZoO_cfG.js} +1 -1
- package/static/app/assets/{cpu-CbJqWTlS.js → cpu-Bs6uc9W9.js} +1 -1
- package/static/app/assets/{download-CkSzbzU-.js → download-G-2olkWz.js} +1 -1
- package/static/app/assets/{folder-open-CKyjQ4PU.js → folder-open-CTOspnmb.js} +1 -1
- package/static/app/assets/{hard-drive-DAzk9um0.js → hard-drive-CewHWJhn.js} +1 -1
- package/static/app/assets/index-CkzokZAj.css +2 -0
- package/static/app/assets/{index-CxOcwsHV.js → index-D7Rr-J2Y.js} +3 -3
- package/static/app/assets/{input-DcMETmZ7.js → input-D4w_BZWl.js} +1 -1
- package/static/app/assets/{permissionCopy-BVf13_25.js → permissionCopy-CosBEXAZ.js} +1 -1
- package/static/app/assets/primitives-d0g9pvzS.js +1 -0
- package/static/app/assets/search-BLCYt75v.js +1 -0
- package/static/app/assets/{share-2-COWCHNZm.js → share-2-NmD7e_oV.js} +1 -1
- package/static/app/assets/{shield-alert-BDrvilyK.js → shield-alert-CcQeMuju.js} +1 -1
- package/static/app/assets/{textarea-rUmsc8cP.js → textarea-BPAJDc-0.js} +1 -1
- package/static/app/assets/{useFocusTrap-Bi5UY_8v.js → useFocusTrap-C7YLdTBC.js} +1 -1
- package/static/app/assets/{useQuery-C-AicB-3.js → useQuery-DRyD9opW.js} +1 -1
- package/static/app/assets/{utils-BMwWg78e.js → utils-DG1_ExrP.js} +3 -3
- package/static/app/assets/{workspace-Y93tls8P.js → workspace-CWVf3gsI.js} +1 -1
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/static/app/assets/BrainHome-CMDqgJF4.js +0 -2
- package/static/app/assets/BrainSignals-BeE8RJo3.js +0 -1
- package/static/app/assets/Capture-ZX9bQh68.js +0 -1
- package/static/app/assets/Library-Bhz5LUca.js +0 -1
- package/static/app/assets/LivingBrain-BSa0wpFG.js +0 -1
- package/static/app/assets/ProductFlow-IP4Q-5aQ.js +0 -1
- package/static/app/assets/System-Bwx1h_jT.js +0 -1
- package/static/app/assets/arrow-left-ig7AZU8B.js +0 -1
- package/static/app/assets/brain-D8OEwmVj.js +0 -1
- package/static/app/assets/index-BfD-jhA9.css +0 -2
- package/static/app/assets/primitives-CQV9Q2YM.js +0 -1
- package/static/app/assets/search-CXGASMMH.js +0 -1
|
@@ -48,6 +48,21 @@ _DEFAULT_CONTRADICTION_NODES = 300
|
|
|
48
48
|
_DEFAULT_NEAR_THRESHOLD = 0.75
|
|
49
49
|
_DEFAULT_MAX_PAIRS = 200
|
|
50
50
|
_DEFAULT_STALE_DAYS = 90 # mirrors MemoryQualityManager.apply_retention
|
|
51
|
+
_DEFAULT_HALF_LIFE_DAYS = 30.0
|
|
52
|
+
# Node types that record *what happened* rather than *what is known*. Only
|
|
53
|
+
# these are offered for consolidation: folding a Decision or a Document into a
|
|
54
|
+
# summary would lose the thing the user actually keeps a Brain for.
|
|
55
|
+
_EPISODIC_TYPES = frozenset(
|
|
56
|
+
{
|
|
57
|
+
"chat",
|
|
58
|
+
"conversation",
|
|
59
|
+
"message",
|
|
60
|
+
"airesponse",
|
|
61
|
+
"ai_response",
|
|
62
|
+
"event",
|
|
63
|
+
"chunk",
|
|
64
|
+
}
|
|
65
|
+
)
|
|
51
66
|
|
|
52
67
|
|
|
53
68
|
def _parse_ts(value: Any) -> Optional[datetime]:
|
|
@@ -78,6 +93,28 @@ def _node_text(node: Dict[str, Any]) -> str:
|
|
|
78
93
|
return f"{title} {summary}".strip()
|
|
79
94
|
|
|
80
95
|
|
|
96
|
+
def _is_episodic(node: Dict[str, Any]) -> bool:
|
|
97
|
+
return str(node.get("type") or "").strip().lower() in _EPISODIC_TYPES
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _access_count(node: Dict[str, Any], stored: Optional[Dict[str, Any]]) -> float:
|
|
101
|
+
"""Access count for a node: ingested metadata first, then the store counter.
|
|
102
|
+
|
|
103
|
+
A surface that already tracks reads (``metadata.access_count``) is more
|
|
104
|
+
accurate than our own read-path counter, so it wins; ``0`` from metadata is
|
|
105
|
+
a real answer and is not treated as "missing".
|
|
106
|
+
"""
|
|
107
|
+
metadata = node.get("metadata")
|
|
108
|
+
if isinstance(metadata, dict):
|
|
109
|
+
for key in ("access_count", "accesses", "access"):
|
|
110
|
+
value = metadata.get(key)
|
|
111
|
+
if isinstance(value, (int, float)) and not isinstance(value, bool):
|
|
112
|
+
return float(value)
|
|
113
|
+
if stored is None:
|
|
114
|
+
return 0.0
|
|
115
|
+
return float(stored.get("accesses") or 0.0)
|
|
116
|
+
|
|
117
|
+
|
|
81
118
|
def _slim(node: Dict[str, Any]) -> Dict[str, Any]:
|
|
82
119
|
return {
|
|
83
120
|
"id": node.get("id"),
|
|
@@ -129,6 +166,17 @@ class ProactiveBrain:
|
|
|
129
166
|
edges.append(normalized)
|
|
130
167
|
return {"nodes": nodes, "edges": edges}
|
|
131
168
|
|
|
169
|
+
def sample(
|
|
170
|
+
self, *, workspace_id: Optional[str] = None, limit: Optional[int] = None
|
|
171
|
+
) -> Dict[str, List[Dict[str, Any]]]:
|
|
172
|
+
"""One normalized graph sample (``source``/``target`` edge keys).
|
|
173
|
+
|
|
174
|
+
Public seam for :mod:`lattice_brain.synthesis`, which needs the *same*
|
|
175
|
+
sample for several passes; taking it once keeps a synthesis run to one
|
|
176
|
+
graph read and keeps every pass looking at identical data.
|
|
177
|
+
"""
|
|
178
|
+
return self._sample(workspace_id=workspace_id, limit=limit)
|
|
179
|
+
|
|
132
180
|
# ── duplicates ───────────────────────────────────────────────────────
|
|
133
181
|
|
|
134
182
|
def find_duplicates(
|
|
@@ -246,6 +294,16 @@ class ProactiveBrain:
|
|
|
246
294
|
sample["nodes"], sample["edges"], max_nodes=max_nodes
|
|
247
295
|
)
|
|
248
296
|
|
|
297
|
+
def contradictions_in(
|
|
298
|
+
self,
|
|
299
|
+
nodes: List[Dict[str, Any]],
|
|
300
|
+
edges: List[Dict[str, Any]],
|
|
301
|
+
*,
|
|
302
|
+
max_nodes: int = _DEFAULT_CONTRADICTION_NODES,
|
|
303
|
+
) -> Dict[str, Any]:
|
|
304
|
+
"""Contradiction signals over an already-taken :meth:`sample`."""
|
|
305
|
+
return self._detect_contradictions_in(nodes, edges, max_nodes=max_nodes)
|
|
306
|
+
|
|
249
307
|
def _detect_contradictions_in(
|
|
250
308
|
self,
|
|
251
309
|
nodes: List[Dict[str, Any]],
|
|
@@ -403,6 +461,85 @@ class ProactiveBrain:
|
|
|
403
461
|
"generated_at": datetime.now(timezone.utc).isoformat(),
|
|
404
462
|
}
|
|
405
463
|
|
|
464
|
+
# ── importance & decay (v11.1.0) ─────────────────────────────────────
|
|
465
|
+
|
|
466
|
+
def importance_report(
|
|
467
|
+
self,
|
|
468
|
+
*,
|
|
469
|
+
workspace_id: Optional[str] = None,
|
|
470
|
+
limit: Optional[int] = None,
|
|
471
|
+
half_life_days: float = _DEFAULT_HALF_LIFE_DAYS,
|
|
472
|
+
max_candidates: int = 20,
|
|
473
|
+
sample: Optional[Dict[str, List[Dict[str, Any]]]] = None,
|
|
474
|
+
) -> Dict[str, Any]:
|
|
475
|
+
"""Score every sampled node by use, then name the weakest episodic ones.
|
|
476
|
+
|
|
477
|
+
The score is deliberately boring and reproducible — no model, no
|
|
478
|
+
randomness::
|
|
479
|
+
|
|
480
|
+
score = (1 + accesses + degree) * 0.5 ** (age_days / half_life)
|
|
481
|
+
|
|
482
|
+
*accesses* prefers a real counter: ``metadata.access_count`` when the
|
|
483
|
+
ingesting surface recorded one, otherwise the store's own read-path
|
|
484
|
+
counter (``access_stats``), otherwise zero. *Episodic* types (chats,
|
|
485
|
+
messages, events, chunks) are the only consolidation candidates —
|
|
486
|
+
a decayed Document or Decision is stale knowledge to review, not
|
|
487
|
+
noise to fold away.
|
|
488
|
+
"""
|
|
489
|
+
data = sample if sample is not None else self._sample(
|
|
490
|
+
workspace_id=workspace_id, limit=limit
|
|
491
|
+
)
|
|
492
|
+
nodes, edges = data["nodes"], data["edges"]
|
|
493
|
+
degree: Dict[str, int] = {}
|
|
494
|
+
for edge in edges:
|
|
495
|
+
for key in ("source", "target"):
|
|
496
|
+
node_id = str(edge.get(key) or "")
|
|
497
|
+
if node_id:
|
|
498
|
+
degree[node_id] = degree.get(node_id, 0) + 1
|
|
499
|
+
|
|
500
|
+
stats_fn = getattr(self._store, "access_stats", None)
|
|
501
|
+
stored: Dict[str, Any] = {}
|
|
502
|
+
if callable(stats_fn):
|
|
503
|
+
try:
|
|
504
|
+
stored = dict(stats_fn([n.get("id") for n in nodes]) or {})
|
|
505
|
+
except Exception: # noqa: BLE001 — the report degrades, never fails
|
|
506
|
+
logger.exception("access stats read failed")
|
|
507
|
+
|
|
508
|
+
now = datetime.now(timezone.utc)
|
|
509
|
+
half_life = max(0.5, float(half_life_days))
|
|
510
|
+
scored: List[Dict[str, Any]] = []
|
|
511
|
+
for node in nodes:
|
|
512
|
+
node_id = str(node.get("id") or "")
|
|
513
|
+
accesses = _access_count(node, stored.get(node_id))
|
|
514
|
+
ts = _parse_ts(node.get("updated_at"))
|
|
515
|
+
age_days = 0.0 if ts is None else max(
|
|
516
|
+
0.0, (now - ts).total_seconds() / 86400.0
|
|
517
|
+
)
|
|
518
|
+
decay = 0.5 ** (age_days / half_life)
|
|
519
|
+
scored.append(
|
|
520
|
+
{
|
|
521
|
+
**_slim(node),
|
|
522
|
+
"accesses": accesses,
|
|
523
|
+
"degree": degree.get(node_id, 0),
|
|
524
|
+
"age_days": round(age_days, 2),
|
|
525
|
+
"score": round((1.0 + accesses + degree.get(node_id, 0)) * decay, 4),
|
|
526
|
+
"episodic": _is_episodic(node),
|
|
527
|
+
}
|
|
528
|
+
)
|
|
529
|
+
scored.sort(key=lambda item: (item["score"], str(item.get("id") or "")))
|
|
530
|
+
candidates = [item for item in scored if item["episodic"]][
|
|
531
|
+
: max(1, int(max_candidates))
|
|
532
|
+
]
|
|
533
|
+
return {
|
|
534
|
+
"nodes_scanned": len(nodes),
|
|
535
|
+
"half_life_days": half_life,
|
|
536
|
+
"access_source": "store" if stored else "metadata",
|
|
537
|
+
"candidates": candidates,
|
|
538
|
+
"candidate_count": len(candidates),
|
|
539
|
+
"strongest": list(reversed(scored[-5:])),
|
|
540
|
+
"generated_at": datetime.now(timezone.utc).isoformat(),
|
|
541
|
+
}
|
|
542
|
+
|
|
406
543
|
# ── consolidation ────────────────────────────────────────────────────
|
|
407
544
|
|
|
408
545
|
def consolidate_duplicates(
|
|
@@ -470,7 +607,7 @@ class ProactiveBrain:
|
|
|
470
607
|
for group in groups:
|
|
471
608
|
try:
|
|
472
609
|
if merge_fn is None: # guarded by apply_supported above
|
|
473
|
-
raise RuntimeError("store has no merge_nodes")
|
|
610
|
+
raise RuntimeError("store has no merge_nodes") # pragma: no cover — unreachable: callable(merge_fn) above proves it is not None
|
|
474
611
|
result = merge_fn(group["keep"], group["remove"])
|
|
475
612
|
applied.append({"keep": group["keep"], "result": result})
|
|
476
613
|
except Exception as exc: # keep going; report per-group failure
|
|
@@ -52,6 +52,12 @@ class KnowledgeGraphProjectionMixin(_Core):
|
|
|
52
52
|
END;
|
|
53
53
|
"""
|
|
54
54
|
|
|
55
|
+
# The temporal columns pass through *raw* (v11.1.0). ``type`` is COALESCEd
|
|
56
|
+
# because ``legacy_type`` carries the label a reader expects; validity is
|
|
57
|
+
# not a label — a COALESCE there would turn "still valid" (NULL) into a
|
|
58
|
+
# value, which is exactly the ``kgv2_edges`` trap noted in the 11.0.1
|
|
59
|
+
# review. NULL in, NULL out; the fallback to ``created_at`` belongs to the
|
|
60
|
+
# read predicate (``schema.TEMPORAL_PREDICATE_SQL``), not to the view.
|
|
55
61
|
_V2_VIEWS_SQL = """
|
|
56
62
|
CREATE VIEW IF NOT EXISTS kgv2_nodes AS
|
|
57
63
|
SELECT id,
|
|
@@ -59,14 +65,16 @@ class KnowledgeGraphProjectionMixin(_Core):
|
|
|
59
65
|
label AS title,
|
|
60
66
|
summary,
|
|
61
67
|
attrs AS metadata_json,
|
|
62
|
-
created_at, updated_at
|
|
68
|
+
created_at, updated_at,
|
|
69
|
+
valid_from, valid_to, superseded_by
|
|
63
70
|
FROM nodes_v2;
|
|
64
71
|
CREATE VIEW IF NOT EXISTS kgv2_edges AS
|
|
65
72
|
SELECT id, source AS from_node, target AS to_node,
|
|
66
73
|
COALESCE(legacy_type, type) AS type,
|
|
67
74
|
weight,
|
|
68
75
|
metadata AS metadata_json,
|
|
69
|
-
created_at
|
|
76
|
+
created_at,
|
|
77
|
+
valid_from, valid_to, superseded_by
|
|
70
78
|
FROM edges_v2;
|
|
71
79
|
"""
|
|
72
80
|
|
|
@@ -20,6 +20,13 @@ else:
|
|
|
20
20
|
# traverse / stats) moved byte-identically to .retrieval_reads as
|
|
21
21
|
# KnowledgeGraphReadsMixin. Re-exported here so any legacy
|
|
22
22
|
# ``from lattice_brain.graph.retrieval import ...`` site keeps resolving.
|
|
23
|
+
from .fusion import (
|
|
24
|
+
DEFAULT_EXPANSION_CAP,
|
|
25
|
+
DEFAULT_EXPANSION_SEEDS,
|
|
26
|
+
expand_with_neighbors,
|
|
27
|
+
graph_expansion_enabled,
|
|
28
|
+
rrf_fuse,
|
|
29
|
+
)
|
|
23
30
|
from .retrieval_reads import KnowledgeGraphReadsMixin # noqa: F401
|
|
24
31
|
|
|
25
32
|
|
|
@@ -28,6 +35,7 @@ def context_quality_signal(
|
|
|
28
35
|
nodes: int,
|
|
29
36
|
*,
|
|
30
37
|
reason: Optional[str] = None,
|
|
38
|
+
vector: Optional[Dict[str, Any]] = None,
|
|
31
39
|
) -> Dict[str, Any]:
|
|
32
40
|
"""Honest RAG context-quality signal (v9.8.0, additive contract).
|
|
33
41
|
|
|
@@ -37,6 +45,15 @@ def context_quality_signal(
|
|
|
37
45
|
``"none"``; ``limited`` is true whenever the context is thin (0–1 nodes)
|
|
38
46
|
or the vector side fell back to lexical-only retrieval. ``reason`` is a
|
|
39
47
|
short human-readable Korean phrase, only present when limited.
|
|
48
|
+
|
|
49
|
+
``vector`` (v11.1.0) carries the vector channel's own honesty block —
|
|
50
|
+
which backend scored, whether it was approximate, whether the candidate
|
|
51
|
+
scan was truncated. "hybrid, 6 nodes" describes two different answers
|
|
52
|
+
depending on those bits, and the caller that has to say "I did not find
|
|
53
|
+
it" deserves to know which one it got. The key is present **only when
|
|
54
|
+
there is a caveat to report**: an exact, complete vector scan is the
|
|
55
|
+
contract's baseline assumption, so annotating it would be noise, and the
|
|
56
|
+
four-key shape stays exactly what existing consumers pin.
|
|
40
57
|
"""
|
|
41
58
|
nodes = max(0, int(nodes or 0))
|
|
42
59
|
mode = str(mode or "none")
|
|
@@ -54,7 +71,15 @@ def context_quality_signal(
|
|
|
54
71
|
reason = "그래프 기반 컨텍스트가 제한적입니다"
|
|
55
72
|
if not limited:
|
|
56
73
|
reason = None
|
|
57
|
-
|
|
74
|
+
signal: Dict[str, Any] = {
|
|
75
|
+
"mode": mode,
|
|
76
|
+
"nodes": nodes,
|
|
77
|
+
"limited": limited,
|
|
78
|
+
"reason": reason,
|
|
79
|
+
}
|
|
80
|
+
if vector is not None:
|
|
81
|
+
signal["vector"] = dict(vector)
|
|
82
|
+
return signal
|
|
58
83
|
|
|
59
84
|
|
|
60
85
|
class KnowledgeGraphRetrievalMixin(_Core):
|
|
@@ -162,7 +187,7 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
162
187
|
from_node = node_by_id.get(edge["from"])
|
|
163
188
|
to_node = node_by_id.get(edge["to"])
|
|
164
189
|
if not from_node or not to_node:
|
|
165
|
-
continue
|
|
190
|
+
continue # pragma: no cover — unreachable: the edge query selects endpoints from the same node window
|
|
166
191
|
for topic_node, other_node in ((from_node, to_node), (to_node, from_node)):
|
|
167
192
|
if topic_node["type"] != "Topic":
|
|
168
193
|
continue
|
|
@@ -404,6 +429,10 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
404
429
|
search_query = query
|
|
405
430
|
rewrite_rules: List[str] = []
|
|
406
431
|
recency_half_life_days: Optional[float] = None
|
|
432
|
+
# "alpha" is the historical linear fusion; the policy may select RRF
|
|
433
|
+
# per query class. An explicitly pinned ``alpha`` argument means the
|
|
434
|
+
# caller is asking for linear fusion by name, so it stays linear.
|
|
435
|
+
fusion_strategy = "alpha"
|
|
407
436
|
if alpha is None:
|
|
408
437
|
try:
|
|
409
438
|
from .retrieval_policy import resolve_policy
|
|
@@ -411,6 +440,7 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
411
440
|
policy = resolve_policy(query)
|
|
412
441
|
query_class = policy["query_class"]
|
|
413
442
|
alpha = float(policy["alpha"])
|
|
443
|
+
fusion_strategy = str(policy.get("fusion_strategy") or "alpha")
|
|
414
444
|
rewrite_rules = list(policy.get("rewrite_rules") or [])
|
|
415
445
|
rewritten = str(policy.get("search_query") or "")
|
|
416
446
|
if rewritten and rewritten != query:
|
|
@@ -438,6 +468,7 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
438
468
|
"sources": {"lexical": 0, "vector": 0},
|
|
439
469
|
"matches": [],
|
|
440
470
|
"policy": {"search_query": search_query, "rewrite_rules": rewrite_rules},
|
|
471
|
+
"fusion_strategy": fusion_strategy,
|
|
441
472
|
"detail": None,
|
|
442
473
|
}
|
|
443
474
|
|
|
@@ -455,6 +486,16 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
455
486
|
detail: Optional[str] = None
|
|
456
487
|
vector_matches: List[Dict[str, Any]] = []
|
|
457
488
|
vector_recall: Optional[Dict[str, Any]] = None
|
|
489
|
+
# The vector channel's own honesty block, echoed additively so a
|
|
490
|
+
# caller can tell an exact "not found" from an approximate one.
|
|
491
|
+
vector_meta: Dict[str, Any] = {
|
|
492
|
+
"backend": None,
|
|
493
|
+
"approx": None,
|
|
494
|
+
"exhaustive": None,
|
|
495
|
+
"truncated": None,
|
|
496
|
+
"embedded_rows": None,
|
|
497
|
+
"degraded": None,
|
|
498
|
+
}
|
|
458
499
|
vector_fn = getattr(self, "vector_search", None)
|
|
459
500
|
if not callable(vector_fn):
|
|
460
501
|
mode = "lexical_only"
|
|
@@ -471,8 +512,16 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
471
512
|
# retrieval_vector.vector_search), and a fused answer built on
|
|
472
513
|
# a truncated scan is not the same claim as a complete one.
|
|
473
514
|
recall = vector_payload.get("recall")
|
|
474
|
-
if isinstance(recall, dict)
|
|
475
|
-
|
|
515
|
+
if isinstance(recall, dict):
|
|
516
|
+
vector_meta["backend"] = recall.get("backend")
|
|
517
|
+
vector_meta["truncated"] = bool(recall.get("truncated"))
|
|
518
|
+
vector_meta["embedded_rows"] = recall.get("candidates_total")
|
|
519
|
+
if recall.get("truncated"):
|
|
520
|
+
vector_recall = dict(recall)
|
|
521
|
+
index_block = vector_payload.get("index")
|
|
522
|
+
if isinstance(index_block, dict):
|
|
523
|
+
vector_meta["approx"] = bool(index_block.get("approx"))
|
|
524
|
+
vector_meta["exhaustive"] = bool(index_block.get("exhaustive"))
|
|
476
525
|
except Exception as exc: # noqa: BLE001 — degrade, never fail the search
|
|
477
526
|
mode = "lexical_only"
|
|
478
527
|
detail = f"vector index unavailable: {exc}"
|
|
@@ -525,6 +574,11 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
525
574
|
entries[node_id] = entry
|
|
526
575
|
return entry
|
|
527
576
|
|
|
577
|
+
# Per-channel id order (best first) — the only input RRF needs, and
|
|
578
|
+
# the one thing a normalized score cannot reconstruct.
|
|
579
|
+
lexical_order: List[str] = []
|
|
580
|
+
vector_order: List[str] = []
|
|
581
|
+
|
|
528
582
|
for rank, match in enumerate(lexical_matches, start=1):
|
|
529
583
|
node_id = _parent_node_id(match)
|
|
530
584
|
if not node_id:
|
|
@@ -534,6 +588,7 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
534
588
|
entry["scores"]["lexical"], round(1.0 / rank, 6)
|
|
535
589
|
)
|
|
536
590
|
entry["_lexical"] = True
|
|
591
|
+
lexical_order.append(node_id)
|
|
537
592
|
|
|
538
593
|
# Max-normalize cosine scores into [0, 1] (guard the score-0 falsy trap
|
|
539
594
|
# by comparing explicitly, never with truthiness).
|
|
@@ -551,27 +606,86 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
551
606
|
entry = _entry_for(node_id, match)
|
|
552
607
|
entry["scores"]["vector"] = max(entry["scores"]["vector"], round(vec_norm, 6))
|
|
553
608
|
entry["_vector"] = True
|
|
609
|
+
vector_order.append(node_id)
|
|
554
610
|
# Prefer a real snippet when the lexical row had no summary.
|
|
555
611
|
if not entry.get("summary") and match.get("summary"):
|
|
556
612
|
entry["summary"] = match.get("summary")
|
|
557
613
|
|
|
614
|
+
# Graph traversal candidate expansion (opt-in, capped, counted): pull
|
|
615
|
+
# the one-hop neighbours of the strongest hits into the candidate pool
|
|
616
|
+
# so an answer that is adjacent to the match — not in it — is
|
|
617
|
+
# reachable at all. Off by default; see fusion.GRAPH_EXPANSION_ENV.
|
|
618
|
+
expansion_report: Dict[str, Any] = {
|
|
619
|
+
"enabled": False,
|
|
620
|
+
"seeds": 0,
|
|
621
|
+
"added": 0,
|
|
622
|
+
"cap": DEFAULT_EXPANSION_CAP,
|
|
623
|
+
"truncated": False,
|
|
624
|
+
"failed_seeds": 0,
|
|
625
|
+
}
|
|
626
|
+
if entries and graph_expansion_enabled():
|
|
627
|
+
seeds = sorted(
|
|
628
|
+
(
|
|
629
|
+
(node_id, float(entry["scores"]["vector"]))
|
|
630
|
+
for node_id, entry in entries.items()
|
|
631
|
+
),
|
|
632
|
+
key=lambda pair: -pair[1],
|
|
633
|
+
)[:DEFAULT_EXPANSION_SEEDS]
|
|
634
|
+
expanded, expansion_report = expand_with_neighbors(
|
|
635
|
+
seeds,
|
|
636
|
+
self.neighbors,
|
|
637
|
+
exclude=list(entries),
|
|
638
|
+
cap=DEFAULT_EXPANSION_CAP,
|
|
639
|
+
)
|
|
640
|
+
for candidate in expanded:
|
|
641
|
+
node = candidate["node"]
|
|
642
|
+
entry = _entry_for(str(node.get("id")), dict(node))
|
|
643
|
+
entry["scores"]["graph"] = candidate["score"]
|
|
644
|
+
entry["metadata"] = {
|
|
645
|
+
**(entry.get("metadata") or {}),
|
|
646
|
+
"expanded_from": candidate["seed"],
|
|
647
|
+
}
|
|
648
|
+
entry["_graph"] = True
|
|
649
|
+
|
|
650
|
+
rrf_normalized: Dict[str, float] = {}
|
|
651
|
+
if fusion_strategy == "rrf":
|
|
652
|
+
raw_rrf = rrf_fuse(
|
|
653
|
+
{
|
|
654
|
+
"lexical": list(dict.fromkeys(lexical_order)),
|
|
655
|
+
"vector": list(dict.fromkeys(vector_order)),
|
|
656
|
+
}
|
|
657
|
+
)
|
|
658
|
+
peak = max(raw_rrf.values(), default=0.0)
|
|
659
|
+
if peak > 0:
|
|
660
|
+
# Rescale to [0, 1] so the score column keeps the same meaning
|
|
661
|
+
# across strategies; RRF's raw values live around 1/60.
|
|
662
|
+
rrf_normalized = {key: value / peak for key, value in raw_rrf.items()}
|
|
663
|
+
|
|
558
664
|
matches: List[Dict[str, Any]] = []
|
|
559
665
|
for entry in entries.values():
|
|
560
666
|
lex_score = float(entry["scores"]["lexical"])
|
|
561
667
|
vec_score = float(entry["scores"]["vector"])
|
|
562
668
|
if mode == "lexical_only":
|
|
563
669
|
fused = lex_score
|
|
670
|
+
elif fusion_strategy == "rrf":
|
|
671
|
+
fused = float(rrf_normalized.get(entry["node_id"], 0.0))
|
|
672
|
+
entry["scores"]["rrf"] = round(fused, 6)
|
|
564
673
|
else:
|
|
565
674
|
fused = alpha * vec_score + (1.0 - alpha) * lex_score
|
|
566
|
-
entry["score"] = round(fused, 6)
|
|
567
675
|
from_lexical = bool(entry.pop("_lexical", False))
|
|
568
676
|
from_vector = bool(entry.pop("_vector", False))
|
|
569
|
-
if
|
|
677
|
+
if entry.pop("_graph", False):
|
|
678
|
+
# A one-hop neighbour of a hit: related to the answer, never
|
|
679
|
+
# itself a match, so it carries only its damped seed score.
|
|
680
|
+
fused = float(entry["scores"]["graph"])
|
|
681
|
+
entry["fusion"] = "graph"
|
|
682
|
+
elif from_lexical and from_vector:
|
|
570
683
|
entry["fusion"] = "both"
|
|
571
684
|
elif from_vector:
|
|
572
685
|
entry["fusion"] = "vector"
|
|
573
686
|
else:
|
|
574
687
|
entry["fusion"] = "lexical"
|
|
688
|
+
entry["score"] = round(fused, 6)
|
|
575
689
|
matches.append(entry)
|
|
576
690
|
|
|
577
691
|
# Recency-class age decay (retrieval_policy): dampen each fused score
|
|
@@ -622,6 +736,8 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
622
736
|
"sources": {"lexical": len(lexical_matches), "vector": len(vector_matches)},
|
|
623
737
|
"matches": matches,
|
|
624
738
|
"policy": {"search_query": search_query, "rewrite_rules": rewrite_rules},
|
|
739
|
+
"fusion_strategy": fusion_strategy,
|
|
740
|
+
"graph_expansion": expansion_report,
|
|
625
741
|
"rerank": rerank_meta,
|
|
626
742
|
"detail": detail,
|
|
627
743
|
}
|
|
@@ -631,6 +747,8 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
631
747
|
result["vector_recall"] = vector_recall
|
|
632
748
|
if vector_degraded is None:
|
|
633
749
|
result["vector_degraded"] = "partial_recall"
|
|
750
|
+
vector_meta["degraded"] = result.get("vector_degraded")
|
|
751
|
+
result["vector"] = vector_meta
|
|
634
752
|
return result
|
|
635
753
|
|
|
636
754
|
def context_for_query(
|
|
@@ -669,6 +787,7 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
669
787
|
return ""
|
|
670
788
|
matches: List[Dict[str, Any]] = []
|
|
671
789
|
retrieval_mode = "none"
|
|
790
|
+
vector_meta: Optional[Dict[str, Any]] = None
|
|
672
791
|
if use_hybrid:
|
|
673
792
|
try:
|
|
674
793
|
hybrid = self.hybrid_search(
|
|
@@ -678,6 +797,15 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
678
797
|
include_legacy_global=include_legacy_global,
|
|
679
798
|
)
|
|
680
799
|
matches = hybrid.get("matches", [])
|
|
800
|
+
vector_block = hybrid.get("vector") or {}
|
|
801
|
+
# Only a caveat is worth carrying: approximate scoring, a
|
|
802
|
+
# truncated candidate scan, or an already-flagged degradation.
|
|
803
|
+
if (
|
|
804
|
+
vector_block.get("approx")
|
|
805
|
+
or vector_block.get("truncated")
|
|
806
|
+
or vector_block.get("degraded")
|
|
807
|
+
):
|
|
808
|
+
vector_meta = dict(vector_block)
|
|
681
809
|
if matches:
|
|
682
810
|
retrieval_mode = str(hybrid.get("mode") or "hybrid")
|
|
683
811
|
except Exception: # noqa: BLE001 — context building must never fail
|
|
@@ -754,7 +882,9 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
754
882
|
return context
|
|
755
883
|
return {
|
|
756
884
|
"context": context,
|
|
757
|
-
"quality": context_quality_signal(
|
|
885
|
+
"quality": context_quality_signal(
|
|
886
|
+
retrieval_mode, len(matches[:limit]), vector=vector_meta
|
|
887
|
+
),
|
|
758
888
|
}
|
|
759
889
|
|
|
760
890
|
def context_for_query_with_meta(
|
|
@@ -46,26 +46,26 @@ class KnowledgeGraphDocGenMixin(_Core):
|
|
|
46
46
|
candidate_rows = []
|
|
47
47
|
seen_ids = set()
|
|
48
48
|
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
49
|
+
# `query` is non-empty here — the early return above took the blank case.
|
|
50
|
+
q = f"%{query}%"
|
|
51
|
+
rows = conn.execute(
|
|
52
|
+
f"""
|
|
53
|
+
SELECT id, type, title, summary, metadata_json, updated_at
|
|
54
|
+
FROM {nt}
|
|
55
|
+
WHERE (title LIKE ? OR summary LIKE ? OR metadata_json LIKE ?)
|
|
56
|
+
AND type IN ('Document', 'File', 'CodeFile', 'SlideDeck',
|
|
57
|
+
'Spreadsheet', 'Image', 'ImageText', 'Chat',
|
|
58
|
+
'Decision', 'Task', 'Concept', 'Feature',
|
|
59
|
+
'Page', 'Slide')
|
|
60
|
+
ORDER BY updated_at DESC, id ASC
|
|
61
|
+
LIMIT ?
|
|
62
|
+
""",
|
|
63
|
+
(q, q, q, limit * 5),
|
|
64
|
+
).fetchall()
|
|
65
|
+
for row in rows:
|
|
66
|
+
if row["id"] not in seen_ids:
|
|
67
|
+
seen_ids.add(row["id"])
|
|
68
|
+
candidate_rows.append(row)
|
|
69
69
|
|
|
70
70
|
for term in terms:
|
|
71
71
|
t = f"%{term}%"
|
|
@@ -139,12 +139,17 @@ def resolve_policy(
|
|
|
139
139
|
"query_class": "fact" | "code" | "person" | "recency",
|
|
140
140
|
"weights": {"keyword", "vector", "graph"}, # service fusion
|
|
141
141
|
"alpha": float, # graph-layer fusion
|
|
142
|
+
"fusion_strategy": "alpha" | "rrf", # how they combine
|
|
142
143
|
"original_query": str,
|
|
143
144
|
"search_query": str, # the rewritten form to search with
|
|
144
145
|
"rewrite_rules": [str],
|
|
145
146
|
"recency_half_life_days": float | None, # 14.0 only for recency
|
|
146
147
|
}
|
|
147
148
|
|
|
149
|
+
``fusion_strategy`` is ``"alpha"`` for every class unless
|
|
150
|
+
``LATTICEAI_FUSION_STRATEGY`` says otherwise, so the default policy is
|
|
151
|
+
byte-identical to the pre-11.1.0 one.
|
|
152
|
+
|
|
148
153
|
``recency_half_life_days`` is non-``None`` only for the ``recency``
|
|
149
154
|
class — the honest contract that age decay applies exactly where the
|
|
150
155
|
fusion layers wire it, and nowhere else.
|
|
@@ -157,6 +162,7 @@ def resolve_policy(
|
|
|
157
162
|
"query_class": query_class,
|
|
158
163
|
"weights": dict(profile["weights"]),
|
|
159
164
|
"alpha": float(profile["alpha"]),
|
|
165
|
+
"fusion_strategy": str(profile["strategy"]),
|
|
160
166
|
"original_query": rewrite["original"],
|
|
161
167
|
"search_query": search_query,
|
|
162
168
|
"rewrite_rules": list(rewrite["rules"]),
|