ltcai 11.0.1 → 11.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +55 -43
- package/docs/CHANGELOG.md +61 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/FEATURE_AUDIT_v11.2.0.md +393 -0
- package/docs/LAYOUT_REBUILD_SPEC.md +9 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +1 -1
- package/docs/PERFORMANCE.md +71 -18
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/architecture.md +6 -2
- package/docs/kg-schema.md +1 -1
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/embeddings.py +12 -37
- package/lattice_brain/gates.py +125 -0
- package/lattice_brain/graph/discovery_index.py +30 -32
- package/lattice_brain/graph/fusion.py +35 -4
- package/lattice_brain/graph/image_vectors.py +230 -0
- package/lattice_brain/graph/ingest.py +11 -5
- package/lattice_brain/graph/projection.py +66 -8
- package/lattice_brain/graph/provenance.py +27 -2
- package/lattice_brain/graph/retrieval.py +113 -2
- package/lattice_brain/graph/retrieval_docgen.py +6 -6
- package/lattice_brain/graph/schema.py +18 -0
- package/lattice_brain/graph/store.py +9 -0
- package/lattice_brain/graph/vector_index/selector.py +32 -2
- package/lattice_brain/ingestion.py +363 -10
- package/lattice_brain/multimodal.py +1258 -0
- package/lattice_brain/portability.py +169 -32
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/lattice_brain/sealed_box.py +244 -0
- package/lattice_brain/self_model.py +77 -22
- package/lattice_brain/synthesis.py +24 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/brain_intelligence.py +4 -0
- package/latticeai/api/chat.py +11 -0
- package/latticeai/api/chat_helpers.py +16 -3
- package/latticeai/api/chat_hybrid.py +32 -1
- package/latticeai/api/features.py +70 -0
- package/latticeai/api/local_files.py +102 -0
- package/latticeai/api/memory.py +128 -1
- package/latticeai/api/portability.py +39 -4
- package/latticeai/api/review_queue.py +126 -0
- package/latticeai/api/search.py +16 -2
- package/latticeai/core/agent.py +59 -2
- package/latticeai/core/agent_prompts.py +66 -0
- package/latticeai/core/config.py +4 -1
- package/latticeai/core/context_builder.py +98 -11
- package/latticeai/core/embedding_providers.py +528 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +180 -0
- package/latticeai/core/model_compat.py +73 -2
- package/latticeai/core/workspace_os.py +43 -0
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/core/workspace_reorganization.py +335 -0
- package/latticeai/models/model_providers.py +12 -4
- package/latticeai/runtime/build_phases.py +33 -2
- package/latticeai/runtime/chat_wiring.py +4 -0
- package/latticeai/runtime/feature_toggle_wiring.py +163 -0
- package/latticeai/runtime/persistence_runtime.py +41 -4
- package/latticeai/runtime/router_registration.py +11 -0
- package/latticeai/runtime/runtime_context.py +1 -0
- package/latticeai/services/app_context.py +8 -0
- package/latticeai/services/architecture_readiness.py +1 -1
- package/latticeai/services/automation_intelligence.py +22 -2
- package/latticeai/services/brain_intelligence.py +123 -7
- package/latticeai/services/change_proposals.py +50 -10
- package/latticeai/services/command_center.py +10 -4
- package/latticeai/services/feature_toggles.py +502 -0
- package/latticeai/services/folder_watch.py +122 -1
- package/latticeai/services/hybrid_chat.py +56 -5
- package/latticeai/services/interop_bridges.py +978 -0
- package/latticeai/services/memory_service.py +34 -0
- package/latticeai/services/model_capability_registry.py +434 -261
- package/latticeai/services/model_catalog.py +95 -61
- package/latticeai/services/model_recommendation.py +18 -11
- package/latticeai/services/model_runtime.py +1 -1
- package/latticeai/services/multimodal_ports.py +112 -0
- package/latticeai/services/obsidian_bridge.py +16 -25
- package/latticeai/services/product_readiness.py +1 -1
- package/latticeai/services/search_service.py +149 -2
- package/latticeai/services/self_model_service.py +171 -0
- package/latticeai/services/tool_dispatch.py +4 -0
- package/latticeai/services/voice_capture.py +27 -1
- package/latticeai/setup/auto_setup.py +27 -30
- package/latticeai/setup/wizard.py +77 -44
- package/package.json +1 -1
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_server_i18n.mjs +1 -0
- package/scripts/release_screen_claims.json +22 -0
- package/scripts/verify_hf_model_registry.py +253 -218
- package/src-tauri/Cargo.lock +1 -1
- package/src-tauri/Cargo.toml +1 -1
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +37 -37
- package/static/app/assets/{Act-D4zSxFR-.js → Act-AWf0SAKp.js} +1 -1
- package/static/app/assets/{AdminConsole-w5jBfPt2.js → AdminConsole-D0u8Tiyj.js} +1 -1
- package/static/app/assets/{Brain-C2EqQg74.js → Brain-tuhI4sOC.js} +1 -1
- package/static/app/assets/BrainHome-Ts7G_Ila.js +2 -0
- package/static/app/assets/BrainSignals-jMYgQ2Ar.js +1 -0
- package/static/app/assets/{Capture-DPqpGK8d.js → Capture-CqOSzyPr.js} +1 -1
- package/static/app/assets/{CommandPalette-CNf7h5fp.js → CommandPalette-DC0Bzh-I.js} +1 -1
- package/static/app/assets/{Library-BN0HYOfc.js → Library-CX-bbhmK.js} +1 -1
- package/static/app/assets/{LivingBrain-Dfq_wEDI.js → LivingBrain-DBwhto14.js} +1 -1
- package/static/app/assets/{ProductFlow-B-3O0rNV.js → ProductFlow-BHA2cfKI.js} +1 -1
- package/static/app/assets/{ReviewCard-gZ-tdqFM.js → ReviewCard-BUhCKRNM.js} +1 -1
- package/static/app/assets/{System-BElUcSSw.js → System-Bu2t5hn1.js} +1 -1
- package/static/app/assets/arrow-left-Dzwa5zRb.js +1 -0
- package/static/app/assets/{bot--qYHMtkP.js → bot-Cia42c2h.js} +1 -1
- package/static/app/assets/brain-DJMoqrwx.js +1 -0
- package/static/app/assets/{button-51Z3rsuv.js → button-2j2Ijzgq.js} +1 -1
- package/static/app/assets/{circle-pause-CMIiMaQl.js → circle-pause-BEFeWpVW.js} +1 -1
- package/static/app/assets/{circle-play-DZoO_cfG.js → circle-play-ujXMcHxl.js} +1 -1
- package/static/app/assets/{cpu-Bs6uc9W9.js → cpu-k4awryFq.js} +1 -1
- package/static/app/assets/{download-G-2olkWz.js → download-DFbLJ_ig.js} +1 -1
- package/static/app/assets/{folder-open-CTOspnmb.js → folder-open-7y_b6xkM.js} +1 -1
- package/static/app/assets/{hard-drive-CewHWJhn.js → hard-drive-Bidh02Kr.js} +1 -1
- package/static/app/assets/{index-D7Rr-J2Y.js → index-BpYkzcVm.js} +3 -3
- package/static/app/assets/index-DwDl9-8Y.css +2 -0
- package/static/app/assets/{input-D4w_BZWl.js → input-DSlJJxRs.js} +1 -1
- package/static/app/assets/{permissionCopy-CosBEXAZ.js → permissionCopy-Bpb83Hx9.js} +1 -1
- package/static/app/assets/{primitives-d0g9pvzS.js → primitives-BCx6TvfG.js} +1 -1
- package/static/app/assets/search-Cgy8cCFJ.js +1 -0
- package/static/app/assets/{share-2-NmD7e_oV.js → share-2-BH1M-WNi.js} +1 -1
- package/static/app/assets/{shield-alert-CcQeMuju.js → shield-alert-BlKdBXcG.js} +1 -1
- package/static/app/assets/{textarea-BPAJDc-0.js → textarea-CCWbUfFB.js} +1 -1
- package/static/app/assets/{useFocusTrap-C7YLdTBC.js → useFocusTrap-YdHQ7pJ1.js} +1 -1
- package/static/app/assets/{useQuery-DRyD9opW.js → useQuery-CXQiwbVT.js} +1 -1
- package/static/app/assets/{utils-DG1_ExrP.js → utils-zqPZJxdx.js} +2 -2
- package/static/app/assets/{workspace-CWVf3gsI.js → workspace-DXTihhfU.js} +1 -1
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/static/app/assets/BrainHome-CvXS6XiQ.js +0 -2
- package/static/app/assets/BrainSignals-DOE_KhOU.js +0 -1
- package/static/app/assets/arrow-left-CFNIMjhv.js +0 -1
- package/static/app/assets/brain-DDCLjRqO.js +0 -1
- package/static/app/assets/index-CkzokZAj.css +0 -2
- package/static/app/assets/search-BLCYt75v.js +0 -1
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
"""The image vector space — stored apart, joined by late fusion (v11.1.0).
|
|
2
|
+
|
|
3
|
+
A CLIP image vector and a BGE text vector are both "1024 floats", and that is
|
|
4
|
+
the entire extent of what they have in common. Filing them in one index means
|
|
5
|
+
every cosine between them is a number with no meaning, and the ranking that
|
|
6
|
+
comes out looks exactly like a working search.
|
|
7
|
+
|
|
8
|
+
So image vectors get their own table (``image_embeddings``) keyed by the vision
|
|
9
|
+
model that produced them, and their own search. Text queries never touch this
|
|
10
|
+
index: they reach pictures through OCR text and captions, which are text and
|
|
11
|
+
live in the text index like everything else. This index answers one question —
|
|
12
|
+
*which stored images look like this vector* — and its scores enter
|
|
13
|
+
``hybrid_search`` by late fusion, as a separate channel with its own weight,
|
|
14
|
+
rather than by pretending to be comparable up front.
|
|
15
|
+
|
|
16
|
+
The table is created on demand, is derivable from the images themselves, and is
|
|
17
|
+
never the source of truth: dropping it costs a re-embed, not a memory.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import sqlite3
|
|
23
|
+
import struct
|
|
24
|
+
from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple
|
|
25
|
+
|
|
26
|
+
from .vector_index import build_index, resolve_vector_index
|
|
27
|
+
|
|
28
|
+
IMAGE_VECTOR_TABLE = "image_embeddings"
|
|
29
|
+
#: Default weight the image channel carries when fused with text scores.
|
|
30
|
+
DEFAULT_IMAGE_FUSION_WEIGHT = 0.5
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def ensure_image_vector_table(conn: sqlite3.Connection) -> None:
|
|
34
|
+
"""Create the image-vector table if this graph has never held one."""
|
|
35
|
+
conn.execute(
|
|
36
|
+
f"""
|
|
37
|
+
CREATE TABLE IF NOT EXISTS {IMAGE_VECTOR_TABLE} (
|
|
38
|
+
node_id TEXT PRIMARY KEY,
|
|
39
|
+
model_id TEXT NOT NULL,
|
|
40
|
+
dim INTEGER NOT NULL,
|
|
41
|
+
space TEXT NOT NULL DEFAULT 'image',
|
|
42
|
+
vector BLOB NOT NULL,
|
|
43
|
+
updated_at TEXT
|
|
44
|
+
)
|
|
45
|
+
"""
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def encode_vector(vector: Sequence[float]) -> bytes:
|
|
50
|
+
values = [float(value) for value in vector]
|
|
51
|
+
return struct.pack(f"<{len(values)}f", *values)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def decode_vector(payload: bytes) -> List[float]:
|
|
55
|
+
count = len(payload) // 4
|
|
56
|
+
return list(struct.unpack(f"<{count}f", payload[: count * 4]))
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def record_image_vector(
|
|
60
|
+
store: Any,
|
|
61
|
+
*,
|
|
62
|
+
node_id: str,
|
|
63
|
+
vector: Sequence[float],
|
|
64
|
+
model_id: str,
|
|
65
|
+
space: str = "image",
|
|
66
|
+
updated_at: Optional[str] = None,
|
|
67
|
+
) -> bool:
|
|
68
|
+
"""Persist one image vector. Returns False when there is nothing to store.
|
|
69
|
+
|
|
70
|
+
Never raises: an image whose vector could not be filed is still a memory,
|
|
71
|
+
and :func:`image_index_status` reports it as backlog.
|
|
72
|
+
"""
|
|
73
|
+
values = [float(value) for value in (vector or [])]
|
|
74
|
+
if not node_id or not values or not model_id:
|
|
75
|
+
return False
|
|
76
|
+
try:
|
|
77
|
+
with store._connect() as conn:
|
|
78
|
+
ensure_image_vector_table(conn)
|
|
79
|
+
conn.execute(
|
|
80
|
+
f"""
|
|
81
|
+
INSERT INTO {IMAGE_VECTOR_TABLE}
|
|
82
|
+
(node_id, model_id, dim, space, vector, updated_at)
|
|
83
|
+
VALUES (?, ?, ?, ?, ?, ?)
|
|
84
|
+
ON CONFLICT(node_id) DO UPDATE SET
|
|
85
|
+
model_id=excluded.model_id,
|
|
86
|
+
dim=excluded.dim,
|
|
87
|
+
space=excluded.space,
|
|
88
|
+
vector=excluded.vector,
|
|
89
|
+
updated_at=excluded.updated_at
|
|
90
|
+
""",
|
|
91
|
+
(
|
|
92
|
+
str(node_id),
|
|
93
|
+
str(model_id),
|
|
94
|
+
len(values),
|
|
95
|
+
str(space or "image"),
|
|
96
|
+
encode_vector(values),
|
|
97
|
+
updated_at,
|
|
98
|
+
),
|
|
99
|
+
)
|
|
100
|
+
return True
|
|
101
|
+
except Exception: # noqa: BLE001 — a vector is derivable; the image is not
|
|
102
|
+
return False
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _rows(
|
|
106
|
+
store: Any, model_id: Optional[str], dim: Optional[int]
|
|
107
|
+
) -> List[Tuple[str, List[float]]]:
|
|
108
|
+
with store._connect() as conn:
|
|
109
|
+
ensure_image_vector_table(conn)
|
|
110
|
+
if model_id:
|
|
111
|
+
cursor = conn.execute(
|
|
112
|
+
f"SELECT node_id, vector, dim FROM {IMAGE_VECTOR_TABLE} WHERE model_id=? ORDER BY node_id ASC",
|
|
113
|
+
(str(model_id),),
|
|
114
|
+
)
|
|
115
|
+
else:
|
|
116
|
+
cursor = conn.execute(
|
|
117
|
+
f"SELECT node_id, vector, dim FROM {IMAGE_VECTOR_TABLE} ORDER BY node_id ASC"
|
|
118
|
+
)
|
|
119
|
+
rows = cursor.fetchall()
|
|
120
|
+
out: List[Tuple[str, List[float]]] = []
|
|
121
|
+
for row in rows:
|
|
122
|
+
if dim is not None and int(row["dim"]) != dim:
|
|
123
|
+
# A different width means a different model. Comparing across it is
|
|
124
|
+
# the exact silent wrongness this whole module exists to prevent.
|
|
125
|
+
continue
|
|
126
|
+
out.append((str(row["node_id"]), decode_vector(row["vector"])))
|
|
127
|
+
return out
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def image_similarity_search(
|
|
131
|
+
store: Any,
|
|
132
|
+
vector: Sequence[float],
|
|
133
|
+
*,
|
|
134
|
+
top_k: int = 10,
|
|
135
|
+
model_id: Optional[str] = None,
|
|
136
|
+
min_score: float = 0.0,
|
|
137
|
+
) -> Dict[str, Any]:
|
|
138
|
+
"""Rank stored images against a query *image* vector.
|
|
139
|
+
|
|
140
|
+
Scoring is delegated to the pluggable backend layer (Track 1), so this
|
|
141
|
+
index picks up ``LATTICEAI_VECTOR_INDEX`` and reports the same
|
|
142
|
+
``approx``/``exhaustive`` honesty bits as the text index.
|
|
143
|
+
"""
|
|
144
|
+
query = [float(value) for value in (vector or [])]
|
|
145
|
+
result: Dict[str, Any] = {
|
|
146
|
+
"matches": [],
|
|
147
|
+
"count": 0,
|
|
148
|
+
"candidates": 0,
|
|
149
|
+
"model_id": model_id,
|
|
150
|
+
"detail": None,
|
|
151
|
+
}
|
|
152
|
+
if not query:
|
|
153
|
+
result["detail"] = "an image query needs an image vector"
|
|
154
|
+
return result
|
|
155
|
+
try:
|
|
156
|
+
rows = _rows(store, model_id, len(query))
|
|
157
|
+
except Exception as exc: # noqa: BLE001 — degrade, never fail the search
|
|
158
|
+
result["detail"] = f"image vector index unavailable: {exc}"
|
|
159
|
+
return result
|
|
160
|
+
result["candidates"] = len(rows)
|
|
161
|
+
selection = resolve_vector_index()
|
|
162
|
+
index = build_index(selection, dim=len(query))
|
|
163
|
+
index.rebuild((node_id, values, {}) for node_id, values in rows)
|
|
164
|
+
scored = index.search(query, max(1, int(top_k)), {"min_score": min_score})
|
|
165
|
+
result["matches"] = [
|
|
166
|
+
{"node_id": node_id, "score": round(float(score), 6)} for node_id, score in scored
|
|
167
|
+
]
|
|
168
|
+
result["count"] = len(result["matches"])
|
|
169
|
+
result["index"] = selection.as_dict()
|
|
170
|
+
return result
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def image_index_status(store: Any) -> Dict[str, Any]:
|
|
174
|
+
"""How many images carry a vector, and under which model."""
|
|
175
|
+
status: Dict[str, Any] = {"vectors": 0, "models": {}, "detail": None}
|
|
176
|
+
try:
|
|
177
|
+
with store._connect() as conn:
|
|
178
|
+
ensure_image_vector_table(conn)
|
|
179
|
+
rows = conn.execute(
|
|
180
|
+
f"SELECT model_id, COUNT(*) AS total FROM {IMAGE_VECTOR_TABLE} GROUP BY model_id"
|
|
181
|
+
).fetchall()
|
|
182
|
+
except Exception as exc: # noqa: BLE001 — status must never raise
|
|
183
|
+
status["detail"] = str(exc)
|
|
184
|
+
return status
|
|
185
|
+
models = {str(row["model_id"]): int(row["total"]) for row in rows}
|
|
186
|
+
status["models"] = models
|
|
187
|
+
status["vectors"] = sum(models.values())
|
|
188
|
+
return status
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def fuse_image_scores(
|
|
192
|
+
matches: Iterable[Dict[str, Any]],
|
|
193
|
+
image_scores: Dict[str, float],
|
|
194
|
+
*,
|
|
195
|
+
weight: float = DEFAULT_IMAGE_FUSION_WEIGHT,
|
|
196
|
+
) -> int:
|
|
197
|
+
"""Late-fuse image-space scores into an already-fused match list.
|
|
198
|
+
|
|
199
|
+
Mutates each match in place — ``scores.image`` records the raw image-space
|
|
200
|
+
similarity and ``score`` blends it in — and returns how many matches the
|
|
201
|
+
image channel actually touched. Late fusion is the point: the two spaces
|
|
202
|
+
are ranked separately and combined at the end, so neither is ever asked to
|
|
203
|
+
interpret the other's numbers.
|
|
204
|
+
"""
|
|
205
|
+
weight = max(0.0, min(1.0, float(weight)))
|
|
206
|
+
touched = 0
|
|
207
|
+
for match in matches:
|
|
208
|
+
node_id = str(match.get("node_id") or "")
|
|
209
|
+
raw = image_scores.get(node_id)
|
|
210
|
+
if raw is None:
|
|
211
|
+
continue
|
|
212
|
+
touched += 1
|
|
213
|
+
scores = match.setdefault("scores", {})
|
|
214
|
+
scores["image"] = round(float(raw), 6)
|
|
215
|
+
blended = (1.0 - weight) * float(match.get("score") or 0.0) + weight * float(raw)
|
|
216
|
+
match["score"] = round(blended, 6)
|
|
217
|
+
return touched
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
__all__ = [
|
|
221
|
+
"DEFAULT_IMAGE_FUSION_WEIGHT",
|
|
222
|
+
"IMAGE_VECTOR_TABLE",
|
|
223
|
+
"decode_vector",
|
|
224
|
+
"encode_vector",
|
|
225
|
+
"ensure_image_vector_table",
|
|
226
|
+
"fuse_image_scores",
|
|
227
|
+
"image_index_status",
|
|
228
|
+
"image_similarity_search",
|
|
229
|
+
"record_image_vector",
|
|
230
|
+
]
|
|
@@ -636,14 +636,20 @@ class KnowledgeGraphIngestMixin(_Core):
|
|
|
636
636
|
modified_at: Optional[str] = None,
|
|
637
637
|
conversation_id: Optional[str] = None,
|
|
638
638
|
metadata: Optional[Dict[str, Any]] = None,
|
|
639
|
+
node_type: str = "Document",
|
|
639
640
|
) -> Dict[str, Any]:
|
|
640
641
|
"""Unified text/web ingestion: one shape for URL, browser tab, note, text.
|
|
641
642
|
|
|
642
|
-
Creates a content
|
|
643
|
-
|
|
643
|
+
Creates a content node (idempotent by content hash), a ``Source`` node
|
|
644
|
+
linked via ``indexed_from``, RAG chunks, and extracted
|
|
644
645
|
Concept/Task/Decision nodes — mirroring ingest_document for non-file
|
|
645
646
|
sources. Returns the full set of ids the caller needs to record
|
|
646
647
|
provenance, including ``duplicate`` (was the content already indexed).
|
|
648
|
+
|
|
649
|
+
``node_type`` names the content node. It defaults to ``Document``, so
|
|
650
|
+
every historical caller is unchanged; a door that knows its material is
|
|
651
|
+
something else (``Audio`` for a recording) says so here rather than
|
|
652
|
+
writing a second ingest path to say it.
|
|
647
653
|
"""
|
|
648
654
|
source_type = str(source_type or "text")
|
|
649
655
|
text = str(text or "")
|
|
@@ -674,11 +680,11 @@ class KnowledgeGraphIngestMixin(_Core):
|
|
|
674
680
|
|
|
675
681
|
with self._connect() as conn:
|
|
676
682
|
duplicate = self._node_exists(conn, content_id)
|
|
677
|
-
# ── 콘텐츠 노드 (점: 명사 — 문서)
|
|
683
|
+
# ── 콘텐츠 노드 (점: 명사 — 기본은 문서) ─────────────────────────
|
|
678
684
|
self._upsert_node(
|
|
679
685
|
conn,
|
|
680
686
|
content_id,
|
|
681
|
-
|
|
687
|
+
node_type,
|
|
682
688
|
title,
|
|
683
689
|
summary=(text or title)[:500],
|
|
684
690
|
metadata=node_meta,
|
|
@@ -813,7 +819,7 @@ class KnowledgeGraphIngestMixin(_Core):
|
|
|
813
819
|
|
|
814
820
|
return {
|
|
815
821
|
"node_id": content_id,
|
|
816
|
-
"type":
|
|
822
|
+
"type": node_type,
|
|
817
823
|
"source_node_id": source_node_id,
|
|
818
824
|
"content_hash": content_hash,
|
|
819
825
|
"chunk_ids": chunk_ids,
|
|
@@ -52,16 +52,29 @@ class KnowledgeGraphProjectionMixin(_Core):
|
|
|
52
52
|
END;
|
|
53
53
|
"""
|
|
54
54
|
|
|
55
|
-
#
|
|
56
|
-
#
|
|
57
|
-
#
|
|
58
|
-
#
|
|
59
|
-
#
|
|
60
|
-
#
|
|
55
|
+
# ``type`` is reconstructed from ``legacy_type`` when there is one, because
|
|
56
|
+
# that column carries the label a reader expects. The trap the 11.0.1
|
|
57
|
+
# review recorded — and which 11.2.0 fixes — is that ``edges_v2.legacy_type``
|
|
58
|
+
# is ``NOT NULL DEFAULT ''``: a natively-canonical edge is written with
|
|
59
|
+
# ``legacy_type=''`` so its identity is effectively (source, target, type),
|
|
60
|
+
# and ``COALESCE('', type)`` returns ``''`` because COALESCE only skips
|
|
61
|
+
# NULL. Every canonical edge therefore read back with an **empty type**.
|
|
62
|
+
# ``NULLIF(legacy_type, '')`` is the fix: empty means "no legacy label",
|
|
63
|
+
# which is precisely what the write side meant by it.
|
|
64
|
+
#
|
|
65
|
+
# The empty string stays the write-side sentinel on purpose. SQLite treats
|
|
66
|
+
# NULLs as distinct in a UNIQUE index, so moving to NULL would silently
|
|
67
|
+
# disable the (source, target, type, legacy_type) dedupe and let the same
|
|
68
|
+
# relation land twice.
|
|
69
|
+
#
|
|
70
|
+
# The temporal columns pass through *raw* (v11.1.0): validity is not a
|
|
71
|
+
# label, and a COALESCE there would turn "still valid" (NULL) into a value.
|
|
72
|
+
# NULL in, NULL out; the fallback to ``created_at`` belongs to the read
|
|
73
|
+
# predicate (``schema.TEMPORAL_PREDICATE_SQL``), not to the view.
|
|
61
74
|
_V2_VIEWS_SQL = """
|
|
62
75
|
CREATE VIEW IF NOT EXISTS kgv2_nodes AS
|
|
63
76
|
SELECT id,
|
|
64
|
-
COALESCE(legacy_type, type) AS type,
|
|
77
|
+
COALESCE(NULLIF(legacy_type, ''), type) AS type,
|
|
65
78
|
label AS title,
|
|
66
79
|
summary,
|
|
67
80
|
attrs AS metadata_json,
|
|
@@ -70,7 +83,7 @@ class KnowledgeGraphProjectionMixin(_Core):
|
|
|
70
83
|
FROM nodes_v2;
|
|
71
84
|
CREATE VIEW IF NOT EXISTS kgv2_edges AS
|
|
72
85
|
SELECT id, source AS from_node, target AS to_node,
|
|
73
|
-
COALESCE(legacy_type, type) AS type,
|
|
86
|
+
COALESCE(NULLIF(legacy_type, ''), type) AS type,
|
|
74
87
|
weight,
|
|
75
88
|
metadata AS metadata_json,
|
|
76
89
|
created_at,
|
|
@@ -156,6 +169,7 @@ class KnowledgeGraphProjectionMixin(_Core):
|
|
|
156
169
|
KGStoreV2(self.db_path).init_schema(conn=conn)
|
|
157
170
|
_exec_script(conn, self._V2_VIEWS_SQL)
|
|
158
171
|
self._backfill_v2_on(conn, force=stale)
|
|
172
|
+
self._normalize_v2_legacy_types(conn)
|
|
159
173
|
# version stamp commits together with the backfill — never stranded
|
|
160
174
|
conn.execute(
|
|
161
175
|
"INSERT OR REPLACE INTO kg_meta(key, value) VALUES ('projection_version', ?)",
|
|
@@ -177,6 +191,50 @@ class KnowledgeGraphProjectionMixin(_Core):
|
|
|
177
191
|
except Exception as e:
|
|
178
192
|
logging.warning("knowledge_graph: v2 schema init/backfill skipped: %s", e)
|
|
179
193
|
|
|
194
|
+
def _normalize_v2_legacy_types(self, conn: sqlite3.Connection) -> Dict[str, int]:
|
|
195
|
+
"""Bring pre-11.2.0 rows onto the current ``legacy_type`` convention.
|
|
196
|
+
|
|
197
|
+
The convention is: ``legacy_type`` holds the *raw* label only when it
|
|
198
|
+
differs from the canonical type, and ``''`` (edges) / ``NULL`` (nodes)
|
|
199
|
+
otherwise. Rows written by older builds could carry the canonical value
|
|
200
|
+
in both columns, which is redundant and — because the dedupe key
|
|
201
|
+
includes ``legacy_type`` — splits one relation across two rows.
|
|
202
|
+
|
|
203
|
+
Idempotent by construction: a second run matches nothing. ``UPDATE OR
|
|
204
|
+
IGNORE`` is used because collapsing a redundant row can collide with
|
|
205
|
+
the canonical one that already exists; those survivors are counted and
|
|
206
|
+
reported rather than deleted, since the fixed view reads both
|
|
207
|
+
identically and deleting an edge is not a migration's business.
|
|
208
|
+
"""
|
|
209
|
+
report = {"edges": 0, "nodes": 0, "collisions": 0}
|
|
210
|
+
try:
|
|
211
|
+
before = int(
|
|
212
|
+
conn.execute(
|
|
213
|
+
"SELECT COUNT(*) FROM edges_v2 WHERE legacy_type = type"
|
|
214
|
+
).fetchone()[0]
|
|
215
|
+
or 0
|
|
216
|
+
)
|
|
217
|
+
conn.execute(
|
|
218
|
+
"UPDATE OR IGNORE edges_v2 SET legacy_type='' WHERE legacy_type = type"
|
|
219
|
+
)
|
|
220
|
+
after = int(
|
|
221
|
+
conn.execute(
|
|
222
|
+
"SELECT COUNT(*) FROM edges_v2 WHERE legacy_type = type"
|
|
223
|
+
).fetchone()[0]
|
|
224
|
+
or 0
|
|
225
|
+
)
|
|
226
|
+
report["edges"] = before - after
|
|
227
|
+
report["collisions"] = after
|
|
228
|
+
cursor = conn.execute(
|
|
229
|
+
"UPDATE nodes_v2 SET legacy_type=NULL WHERE legacy_type = ''"
|
|
230
|
+
)
|
|
231
|
+
report["nodes"] = int(cursor.rowcount or 0)
|
|
232
|
+
except sqlite3.Error as exc:
|
|
233
|
+
# The projection is derived; a migration that cannot run leaves the
|
|
234
|
+
# data exactly as it was and the fixed view still reads it right.
|
|
235
|
+
logging.debug("knowledge_graph: legacy_type normalization skipped: %s", exc)
|
|
236
|
+
return report
|
|
237
|
+
|
|
180
238
|
def _backup_before_v2_flip(self) -> Optional[str]:
|
|
181
239
|
"""Create one local SQLite backup before the v2 write-master flip."""
|
|
182
240
|
if not self.db_path.exists() or self.db_path.stat().st_size == 0:
|
|
@@ -37,9 +37,34 @@ class KnowledgeGraphProvenanceMixin(_Core):
|
|
|
37
37
|
permissions: Optional[Dict[str, Any]] = None,
|
|
38
38
|
metadata: Optional[Dict[str, Any]] = None,
|
|
39
39
|
) -> Dict[str, Any]:
|
|
40
|
-
"""
|
|
40
|
+
"""Record where an ingested node came from (upsert by origin).
|
|
41
|
+
|
|
42
|
+
Row identity is ``(node, content, source_type, source_uri, pipeline)``
|
|
43
|
+
— deliberately *not* the wall clock. Through 11.0.x the basis included
|
|
44
|
+
a second-resolution timestamp, which made the record's identity depend
|
|
45
|
+
on when it happened: re-ingesting unchanged content twice inside the
|
|
46
|
+
same second collapsed onto one row, and one second later appended a
|
|
47
|
+
duplicate. That is not an audit trail, it is a race — the same class of
|
|
48
|
+
defect as the 11.0.0 review-item ids — and it grew this table (and the
|
|
49
|
+
"recent ingestions" list built from it) without bound on every re-scan
|
|
50
|
+
of an unchanged folder or vault.
|
|
51
|
+
|
|
52
|
+
With the clock out of the basis, re-ingesting the same content from the
|
|
53
|
+
same origin *updates* one record (``created_at`` moves to the latest
|
|
54
|
+
sighting, so "recently seen" stays true), while genuinely new content or
|
|
55
|
+
a genuinely different origin — another source URI, another pipeline —
|
|
56
|
+
still appends its own record. The timestamp is data on the row, never
|
|
57
|
+
part of its identity. Every individual ingest event remains visible in
|
|
58
|
+
the audit log (``kg_ingest``), which is where per-event history belongs.
|
|
59
|
+
"""
|
|
41
60
|
now = _now()
|
|
42
|
-
prov_basis =
|
|
61
|
+
prov_basis = "|".join([
|
|
62
|
+
node_id,
|
|
63
|
+
content_hash or "",
|
|
64
|
+
source_type,
|
|
65
|
+
source_uri or "",
|
|
66
|
+
pipeline,
|
|
67
|
+
])
|
|
43
68
|
prov_id = f"prov:{_sha256_text(prov_basis)[:24]}"
|
|
44
69
|
with self._connect() as conn:
|
|
45
70
|
conn.execute(
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
|
-
from typing import TYPE_CHECKING
|
|
3
|
+
from typing import TYPE_CHECKING, Sequence
|
|
4
4
|
|
|
5
5
|
# ruff: noqa: F403,F405
|
|
6
6
|
from ._kg_common import * # noqa: F403,F401
|
|
@@ -29,6 +29,30 @@ from .fusion import (
|
|
|
29
29
|
)
|
|
30
30
|
from .retrieval_reads import KnowledgeGraphReadsMixin # noqa: F401
|
|
31
31
|
|
|
32
|
+
#: Node types that are a *thing you can look at or listen to*, not prose. A
|
|
33
|
+
#: match of one of these means the answer rests on more than text.
|
|
34
|
+
MULTIMODAL_NODE_TYPES = ("Image", "ImageText")
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def multimodal_signal(matches: Iterable[Dict[str, Any]]) -> Optional[Dict[str, Any]]:
|
|
38
|
+
"""``{"images": n, "types": [...]}`` when a result set includes pictures.
|
|
39
|
+
|
|
40
|
+
``None`` when it does not: the context-quality contract stays four keys
|
|
41
|
+
wide for the ordinary all-text case, and a caller that sees the key knows
|
|
42
|
+
it means something rather than having to compare a zero.
|
|
43
|
+
"""
|
|
44
|
+
images = 0
|
|
45
|
+
seen: List[str] = []
|
|
46
|
+
for match in matches:
|
|
47
|
+
node_type = str(match.get("type") or "")
|
|
48
|
+
if node_type in MULTIMODAL_NODE_TYPES:
|
|
49
|
+
images += 1
|
|
50
|
+
if node_type not in seen:
|
|
51
|
+
seen.append(node_type)
|
|
52
|
+
if not images:
|
|
53
|
+
return None
|
|
54
|
+
return {"images": images, "types": seen}
|
|
55
|
+
|
|
32
56
|
|
|
33
57
|
def context_quality_signal(
|
|
34
58
|
mode: str,
|
|
@@ -36,6 +60,7 @@ def context_quality_signal(
|
|
|
36
60
|
*,
|
|
37
61
|
reason: Optional[str] = None,
|
|
38
62
|
vector: Optional[Dict[str, Any]] = None,
|
|
63
|
+
multimodal: Optional[Dict[str, Any]] = None,
|
|
39
64
|
) -> Dict[str, Any]:
|
|
40
65
|
"""Honest RAG context-quality signal (v9.8.0, additive contract).
|
|
41
66
|
|
|
@@ -54,6 +79,11 @@ def context_quality_signal(
|
|
|
54
79
|
there is a caveat to report**: an exact, complete vector scan is the
|
|
55
80
|
contract's baseline assumption, so annotating it would be noise, and the
|
|
56
81
|
four-key shape stays exactly what existing consumers pin.
|
|
82
|
+
|
|
83
|
+
``multimodal`` (v11.1.0) follows the same present-only-when-true rule and
|
|
84
|
+
says that part of this context is a picture. "6 nodes" reads differently
|
|
85
|
+
when two of them are screenshots whose text came out of OCR, and the
|
|
86
|
+
surface that has to explain the answer deserves to know.
|
|
57
87
|
"""
|
|
58
88
|
nodes = max(0, int(nodes or 0))
|
|
59
89
|
mode = str(mode or "none")
|
|
@@ -79,6 +109,8 @@ def context_quality_signal(
|
|
|
79
109
|
}
|
|
80
110
|
if vector is not None:
|
|
81
111
|
signal["vector"] = dict(vector)
|
|
112
|
+
if multimodal is not None:
|
|
113
|
+
signal["multimodal"] = dict(multimodal)
|
|
82
114
|
return signal
|
|
83
115
|
|
|
84
116
|
|
|
@@ -95,6 +127,7 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
95
127
|
"SlideDeck", # 프레젠테이션
|
|
96
128
|
"Image", # 이미지
|
|
97
129
|
"ImageText", # OCR 텍스트
|
|
130
|
+
"Audio", # 녹음 / 음성 메모 (11.1.0)
|
|
98
131
|
"Concept", # 개념 / 아이디어 / 기술 용어
|
|
99
132
|
"Person", # 사람
|
|
100
133
|
"Error", # 오류 / 버그
|
|
@@ -344,6 +377,7 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
344
377
|
"SlideDeck",
|
|
345
378
|
"Image",
|
|
346
379
|
"ImageText",
|
|
380
|
+
"Audio",
|
|
347
381
|
"Page",
|
|
348
382
|
"Slide",
|
|
349
383
|
}
|
|
@@ -387,6 +421,8 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
387
421
|
lexical_limit: Optional[int] = None,
|
|
388
422
|
vector_limit: Optional[int] = None,
|
|
389
423
|
min_vector_score: float = 0.0,
|
|
424
|
+
image_vector: Optional[Sequence[float]] = None,
|
|
425
|
+
image_fusion_weight: Optional[float] = None,
|
|
390
426
|
) -> Dict[str, Any]:
|
|
391
427
|
"""Unified lexical + vector retrieval with alpha-weighted linear fusion.
|
|
392
428
|
|
|
@@ -418,6 +454,15 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
418
454
|
fused score into the ``[0.5, 1.0]`` band (``scores.age_decay``).
|
|
419
455
|
Passing an explicit ``alpha`` pins it exactly as before and disables
|
|
420
456
|
rewrite + decay.
|
|
457
|
+
|
|
458
|
+
``image_vector`` (v11.1.0) is the *late fusion* seam for the separate
|
|
459
|
+
image space: the caller supplies a query vector from the same vision
|
|
460
|
+
model that embedded the pictures, its own index is ranked
|
|
461
|
+
independently, and only then are the two rankings blended
|
|
462
|
+
(``image_fusion_weight``, default 0.5). A text query never produces
|
|
463
|
+
one — it reaches images through their OCR text and captions — which is
|
|
464
|
+
exactly why the image channel has to enter at the end rather than
|
|
465
|
+
pretending to share the text index.
|
|
421
466
|
"""
|
|
422
467
|
query = str(query or "").strip()
|
|
423
468
|
try:
|
|
@@ -705,6 +750,16 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
705
750
|
match["scores"]["age_decay"] = round(multiplier, 6)
|
|
706
751
|
match["score"] = round(float(match["score"]) * multiplier, 6)
|
|
707
752
|
|
|
753
|
+
# Late fusion of the image space (v11.1.0). Runs after the text
|
|
754
|
+
# channels have produced a ranking and before the cut, so image
|
|
755
|
+
# evidence can lift a picture into the answer without ever having been
|
|
756
|
+
# compared against a text vector.
|
|
757
|
+
image_fusion: Optional[Dict[str, Any]] = None
|
|
758
|
+
if image_vector is not None:
|
|
759
|
+
image_fusion = self._fuse_image_channel(
|
|
760
|
+
matches, image_vector, top_k=top_k, weight=image_fusion_weight
|
|
761
|
+
)
|
|
762
|
+
|
|
708
763
|
matches.sort(key=lambda item: (-item["score"], item["node_id"]))
|
|
709
764
|
# Optional cross-encoder rerank (v9.9.5). Off by default; when the
|
|
710
765
|
# env kill-switch is set and the model loads, pair scores reorder the
|
|
@@ -749,8 +804,57 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
749
804
|
result["vector_degraded"] = "partial_recall"
|
|
750
805
|
vector_meta["degraded"] = result.get("vector_degraded")
|
|
751
806
|
result["vector"] = vector_meta
|
|
807
|
+
multimodal = multimodal_signal(matches)
|
|
808
|
+
if multimodal is not None or image_fusion is not None:
|
|
809
|
+
result["multimodal"] = {
|
|
810
|
+
**(multimodal or {"images": 0, "types": []}),
|
|
811
|
+
**({"image_fusion": image_fusion} if image_fusion is not None else {}),
|
|
812
|
+
}
|
|
752
813
|
return result
|
|
753
814
|
|
|
815
|
+
def _fuse_image_channel(
|
|
816
|
+
self,
|
|
817
|
+
matches: List[Dict[str, Any]],
|
|
818
|
+
image_vector: Sequence[float],
|
|
819
|
+
*,
|
|
820
|
+
top_k: int,
|
|
821
|
+
weight: Optional[float],
|
|
822
|
+
) -> Dict[str, Any]:
|
|
823
|
+
"""Rank the image index separately, then blend it into ``matches``.
|
|
824
|
+
|
|
825
|
+
Any failure degrades to "the image channel contributed nothing" with
|
|
826
|
+
the reason attached — an image index that cannot be read is not a
|
|
827
|
+
reason to lose the text answer.
|
|
828
|
+
"""
|
|
829
|
+
from .image_vectors import (
|
|
830
|
+
DEFAULT_IMAGE_FUSION_WEIGHT,
|
|
831
|
+
fuse_image_scores,
|
|
832
|
+
image_similarity_search,
|
|
833
|
+
)
|
|
834
|
+
|
|
835
|
+
share = DEFAULT_IMAGE_FUSION_WEIGHT if weight is None else float(weight)
|
|
836
|
+
report: Dict[str, Any] = {
|
|
837
|
+
"weight": round(max(0.0, min(1.0, share)), 4),
|
|
838
|
+
"candidates": 0,
|
|
839
|
+
"fused": 0,
|
|
840
|
+
"detail": None,
|
|
841
|
+
}
|
|
842
|
+
try:
|
|
843
|
+
found = image_similarity_search(
|
|
844
|
+
self, image_vector, top_k=max(1, int(top_k) * 2)
|
|
845
|
+
)
|
|
846
|
+
except Exception as exc: # noqa: BLE001 — never fail the text answer
|
|
847
|
+
report["detail"] = f"image index unavailable: {exc}"
|
|
848
|
+
return report
|
|
849
|
+
report["candidates"] = int(found.get("candidates") or 0)
|
|
850
|
+
report["detail"] = found.get("detail")
|
|
851
|
+
scores = {
|
|
852
|
+
str(row.get("node_id")): float(row.get("score") or 0.0)
|
|
853
|
+
for row in found.get("matches") or []
|
|
854
|
+
}
|
|
855
|
+
report["fused"] = fuse_image_scores(matches, scores, weight=share)
|
|
856
|
+
return report
|
|
857
|
+
|
|
754
858
|
def context_for_query(
|
|
755
859
|
self,
|
|
756
860
|
query: str,
|
|
@@ -788,6 +892,7 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
788
892
|
matches: List[Dict[str, Any]] = []
|
|
789
893
|
retrieval_mode = "none"
|
|
790
894
|
vector_meta: Optional[Dict[str, Any]] = None
|
|
895
|
+
multimodal_meta: Optional[Dict[str, Any]] = None
|
|
791
896
|
if use_hybrid:
|
|
792
897
|
try:
|
|
793
898
|
hybrid = self.hybrid_search(
|
|
@@ -880,10 +985,16 @@ class KnowledgeGraphRetrievalMixin(_Core):
|
|
|
880
985
|
context = "\n".join(lines)
|
|
881
986
|
if not with_meta:
|
|
882
987
|
return context
|
|
988
|
+
# Only the context that actually reached the model counts as
|
|
989
|
+
# multimodal — matches trimmed by ``limit`` are not in the answer.
|
|
990
|
+
multimodal_meta = multimodal_signal(matches[:limit])
|
|
883
991
|
return {
|
|
884
992
|
"context": context,
|
|
885
993
|
"quality": context_quality_signal(
|
|
886
|
-
retrieval_mode,
|
|
994
|
+
retrieval_mode,
|
|
995
|
+
len(matches[:limit]),
|
|
996
|
+
vector=vector_meta,
|
|
997
|
+
multimodal=multimodal_meta,
|
|
887
998
|
),
|
|
888
999
|
}
|
|
889
1000
|
|
|
@@ -54,9 +54,9 @@ class KnowledgeGraphDocGenMixin(_Core):
|
|
|
54
54
|
FROM {nt}
|
|
55
55
|
WHERE (title LIKE ? OR summary LIKE ? OR metadata_json LIKE ?)
|
|
56
56
|
AND type IN ('Document', 'File', 'CodeFile', 'SlideDeck',
|
|
57
|
-
'Spreadsheet', 'Image', 'ImageText', '
|
|
58
|
-
'
|
|
59
|
-
'Page', 'Slide')
|
|
57
|
+
'Spreadsheet', 'Image', 'ImageText', 'Audio',
|
|
58
|
+
'Chat', 'Decision', 'Task', 'Concept',
|
|
59
|
+
'Feature', 'Page', 'Slide')
|
|
60
60
|
ORDER BY updated_at DESC, id ASC
|
|
61
61
|
LIMIT ?
|
|
62
62
|
""",
|
|
@@ -75,9 +75,9 @@ class KnowledgeGraphDocGenMixin(_Core):
|
|
|
75
75
|
FROM {nt}
|
|
76
76
|
WHERE (title LIKE ? OR summary LIKE ? OR metadata_json LIKE ?)
|
|
77
77
|
AND type IN ('Document', 'File', 'CodeFile', 'SlideDeck',
|
|
78
|
-
'Spreadsheet', 'Image', 'ImageText', '
|
|
79
|
-
'
|
|
80
|
-
'Page', 'Slide')
|
|
78
|
+
'Spreadsheet', 'Image', 'ImageText', 'Audio',
|
|
79
|
+
'Chat', 'Decision', 'Task', 'Concept',
|
|
80
|
+
'Feature', 'Page', 'Slide')
|
|
81
81
|
ORDER BY updated_at DESC, id ASC
|
|
82
82
|
LIMIT ?
|
|
83
83
|
""",
|