okfgraph 0.2.4__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,354 @@
1
+ from __future__ import annotations
2
+
3
+ import base64
4
+ import hashlib
5
+ import heapq
6
+ import json
7
+ import logging
8
+ import math
9
+ import os
10
+ import re
11
+ import time
12
+ import uuid
13
+ from contextlib import contextmanager
14
+ from pathlib import Path
15
+ from typing import Any, Callable, Dict, List, Optional, Tuple, Union, Set
16
+ from urllib.parse import urlparse
17
+
18
+ import mordant
19
+ import numpy as np
20
+ import yaml
21
+ import frontmatter
22
+ from okfgraph.models import ChunkModel, ConceptModel
23
+
24
+ logger = logging.getLogger(__name__)
25
+
26
+ class ExportManager:
27
+ def __init__(self, conn, search_engine):
28
+ self.conn = conn
29
+ self.search_engine = search_engine
30
+
31
+ def _enrich_body_with_graph_links(
32
+ self, concept_id: str, body: str, flavor: str = "okf",
33
+ title_counts: Optional[Dict[str, int]] = None,
34
+ ) -> str:
35
+ """Enrich body with graph-derived links so the exported markdown
36
+ faithfully reflects the LINKS_TO graph.
37
+
38
+ Strategy (Option A — append, never replace):
39
+ 1. Query all outgoing LINKS_TO edges from this concept.
40
+ 2. For each target, check if a link to that target already exists
41
+ in the body (by matching the target_id in link URLs).
42
+ 3. If not already linked, append a "See Also" bullet.
43
+ 4. Query all incoming LINKS_TO edges (concepts that link TO this one).
44
+ 5. If any exist, append a "Cited By" bullet list.
45
+
46
+ This preserves the original body's links (which may have richer anchor
47
+ text) while ensuring the graph structure is expressed in the export.
48
+
49
+ ``flavor="obsidian"`` renders appended links as ``[[Title]]``
50
+ (unique titles) or ``[[id|Title]]`` instead of ``[t](id.md)``; the
51
+ original body is never rewritten in either flavor. The obsidian
52
+ flavor omits the "Cited By" section: backlinks re-imported as
53
+ forward wikilinks would reverse their direction, and Obsidian
54
+ renders backlinks natively anyway.
55
+ """
56
+ import re
57
+
58
+ wiki_refs = {
59
+ m.split("|", 1)[0].strip()
60
+ for m in re.findall(r"\[\[([^\]]+)\]\]", body or "")
61
+ }
62
+ wiki_refs_lower = {r.lower() for r in wiki_refs}
63
+
64
+ def _link(target_id: str, title: str) -> str:
65
+ label = title or target_id.split("/")[-1]
66
+ if flavor == "obsidian":
67
+ if label and (title_counts or {}).get(label.lower(), 0) == 1:
68
+ return f"- [[{label}]]"
69
+ return f"- [[{target_id}|{label}]]"
70
+ return f"- [{label}]({target_id}.md)"
71
+
72
+ parts: List[str] = []
73
+
74
+ # --- Outgoing links (See Also) ---
75
+ result = self.conn.execute("""
76
+ MATCH (s:Concept {id: $cid})-[:LINKS_TO]->(t:Concept)
77
+ RETURN t.id AS target_id, t.title AS title, t.type AS type
78
+ ORDER BY t.title
79
+ """, {"cid": concept_id})
80
+ outgoing_rows = result.rows_as_dict().get_all()
81
+
82
+ if outgoing_rows:
83
+ # Determine which targets are already linked in the body:
84
+ # either a [t](...target...) URL or an equivalent [[ref]].
85
+ existing_link_targets = set()
86
+ for row in outgoing_rows:
87
+ target_id = row["target_id"]
88
+ title = row["title"] or ""
89
+ if target_id in wiki_refs or title.lower() in wiki_refs_lower:
90
+ existing_link_targets.add(target_id)
91
+ continue
92
+ # Check if target_id appears in any link URL in the body
93
+ link_pattern = re.compile(
94
+ r"\]\(([^)]*?" + re.escape(target_id) + r"[^)]*)\)"
95
+ )
96
+ if link_pattern.search(body):
97
+ existing_link_targets.add(target_id)
98
+
99
+ # Collect targets that need a link added
100
+ new_links = []
101
+ for row in outgoing_rows:
102
+ target_id = row["target_id"]
103
+ if target_id not in existing_link_targets:
104
+ title = row["title"] or target_id.split("/")[-1]
105
+ new_links.append(_link(target_id, title))
106
+
107
+ if new_links:
108
+ parts.append("\n## See Also\n" + "\n".join(new_links))
109
+
110
+ # --- Incoming links (Cited By: okf flavor only) ---
111
+ # Obsidian renders backlinks natively; emitting them as forward
112
+ # [[wikilinks]] would reverse the edge on re-import (round-trip loss).
113
+ result = self.conn.execute("""
114
+ MATCH (s:Concept)-[:LINKS_TO]->(t:Concept {id: $cid})
115
+ RETURN s.id AS source_id, s.title AS title, s.type AS type
116
+ ORDER BY s.title
117
+ """, {"cid": concept_id})
118
+ incoming_rows = result.rows_as_dict().get_all()
119
+
120
+ if incoming_rows and flavor == "okf":
121
+ cited_lines = ["\n## Cited By\n"]
122
+ for row in incoming_rows:
123
+ source_id = row["source_id"]
124
+ title = row["title"] or source_id.split("/")[-1]
125
+ cited_lines.append(_link(source_id, title))
126
+ parts.append("\n".join(cited_lines))
127
+
128
+ return body + "".join(parts)
129
+
130
+
131
+ def _fetch_concepts(
132
+ self,
133
+ concept_type: Optional[str] = None,
134
+ tags: Optional[List[str]] = None,
135
+ ) -> Dict[str, ConceptModel]:
136
+ """Fetch all concepts, optionally filtered by type and tags."""
137
+ where_clauses: list[str] = []
138
+ params: Dict[str, Any] = {}
139
+
140
+ if concept_type:
141
+ where_clauses.append("c.type = $type")
142
+ params["type"] = concept_type
143
+ if tags:
144
+ where_clauses.append("ALL(tag IN $tags WHERE tag IN c.tags)")
145
+ params["tags"] = tags
146
+
147
+ where_str = " AND ".join(where_clauses) if where_clauses else "true"
148
+ query = f"""
149
+ MATCH (c:Concept)
150
+ WHERE {where_str}
151
+ RETURN c.id, c.type, c.title, c.description, c.resource,
152
+ c.tags, c.timestamp, c.body, c.embedding, c.extra
153
+ """
154
+ results = self.conn.execute(query, params)
155
+ rows = results.rows_as_dict().get_all()
156
+
157
+ concepts: Dict[str, ConceptModel] = {}
158
+ for row in rows:
159
+ data: Dict[str, Any] = {}
160
+ for key, val in row.items():
161
+ col = key.split(".", 1)[-1] # strip 'c.' prefix
162
+ if col != "extra":
163
+ data[col] = val
164
+
165
+ # Decode extra MAP fields
166
+ extra = row.get("c.extra") or {}
167
+ for k, v in extra.items():
168
+ if isinstance(v, str) and v.startswith(("{", "[")):
169
+ try:
170
+ data[k] = json.loads(v)
171
+ except json.JSONDecodeError:
172
+ data[k] = v
173
+ else:
174
+ data[k] = v
175
+
176
+ try:
177
+ concepts[data["id"]] = ConceptModel.model_validate(data)
178
+ except Exception:
179
+ pass # Skip malformed concepts
180
+
181
+ return concepts
182
+
183
+
184
+ def _generate_index_files(
185
+ self, output_dir: Path, concepts: Dict[str, ConceptModel]
186
+ ) -> None:
187
+ """Generate index.md files for every directory in the bundle.
188
+
189
+ Each index.md lists the children (concepts and subdirectories) of that
190
+ directory, enabling progressive disclosure for OKF consumers.
191
+ """
192
+ # Build a map of directory_id → list of (title, relative_path) children
193
+ dir_children: Dict[str, List[Tuple[str, str]]] = {}
194
+
195
+ for cid, concept in concepts.items():
196
+ parts = cid.split("/")
197
+ for i in range(1, len(parts)):
198
+ dir_id = "/".join(parts[:i])
199
+ dir_children.setdefault(dir_id, [])
200
+ child_title = concept.title or parts[i]
201
+ child_rel = cid.replace("/", os.sep) + ".md"
202
+ dir_children[dir_id].append((child_title, child_rel))
203
+
204
+ # Write index.md for each directory
205
+ for dir_id, children in dir_children.items():
206
+ # Sort children by title
207
+ children.sort(key=lambda x: x[0])
208
+ lines = [
209
+ f"# {dir_id.split('/')[-1] or '(root)'}\n",
210
+ "",
211
+ ]
212
+ for title, rel_path in children:
213
+ lines.append(f"- [{title}]({rel_path})")
214
+ lines.append("")
215
+
216
+ # Create parent directories if needed
217
+ dir_path = output_dir / dir_id.replace("/", os.sep)
218
+ dir_path.mkdir(parents=True, exist_ok=True)
219
+ (dir_path / "index.md").write_text("\n".join(lines), encoding="utf-8")
220
+
221
+
222
+ def _is_under_directory(self, concept_id: str, directory_id: str) -> bool:
223
+ """Check if a concept is under a given directory (via CONTAINS graph)."""
224
+ result = self.conn.execute("""
225
+ MATCH (d:Directory {id: $dir_id})-[:CONTAINS*1..5]->(c:Concept {id: $cid})
226
+ RETURN count(c) AS cnt
227
+ """, {"dir_id": directory_id, "cid": concept_id})
228
+ rows = result.rows_as_dict().get_all()
229
+ return rows[0]["cnt"] > 0 if rows else False
230
+
231
+
232
+ def _title_counts(self) -> Dict[str, int]:
233
+ """Count concepts per (lowercased) title.
234
+
235
+ Lowercased to match the case-insensitive name index on import: two
236
+ titles differing only by case are ambiguous as ``[[Title]]`` and
237
+ must both export disambiguated.
238
+ """
239
+ counts: Dict[str, int] = {}
240
+ for r in self.conn.execute(
241
+ "MATCH (c:Concept) RETURN c.title"
242
+ ).rows_as_dict().get_all():
243
+ title = (r.get("c.title") or "").lower()
244
+ counts[title] = counts.get(title, 0) + 1
245
+ return counts
246
+
247
+ def _write_okf(
248
+ self, concept: ConceptModel, output_path: Path,
249
+ flavor: str = "okf", title_counts: Optional[Dict[str, int]] = None,
250
+ ) -> None:
251
+ """Internal: serialize a ConceptModel to an OKF .md file.
252
+
253
+ Enriches the body with LINKS_TO relationships from the graph so that
254
+ exported markdown faithfully reflects the graph structure.
255
+
256
+ ``flavor="obsidian"`` renders appended graph links as ``[[Title]]``
257
+ wikilinks; the default ``okf`` flavor uses ``[t](id.md)`` links.
258
+ Either way the stored ``uid`` (frontmatter ``id:`` on import) is
259
+ written back as ``id:`` so re-import is lossless.
260
+ """
261
+ data, body = concept.export_frontmatter()
262
+
263
+ yaml_str = yaml.dump(
264
+ data, default_flow_style=False, allow_unicode=True, sort_keys=False
265
+ )
266
+
267
+ # ENRICH: add graph-derived links to the body
268
+ body = self._enrich_body_with_graph_links(
269
+ concept.id, body, flavor=flavor, title_counts=title_counts,
270
+ )
271
+
272
+ output_path.parent.mkdir(parents=True, exist_ok=True)
273
+ output_path.write_text(f"---\n{yaml_str}---\n\n{body}", encoding="utf-8")
274
+
275
+
276
+ def export_bundle(
277
+ self,
278
+ output_dir: Path,
279
+ directory_id: Optional[str] = None,
280
+ concept_type: Optional[str] = None,
281
+ tags: Optional[List[str]] = None,
282
+ flavor: str = "okf",
283
+ ) -> List[str]:
284
+ """Export concepts from the graph back to an OKF bundle directory.
285
+
286
+ Reconstructs the full directory hierarchy from CONTAINS relationships.
287
+ Supports filtering by directory subtree, concept type, or tags.
288
+
289
+ Args:
290
+ output_dir: Root directory to write the bundle into.
291
+ directory_id: If set, only export concepts under this directory.
292
+ concept_type: If set, only export concepts of this type.
293
+ tags: If set, only export concepts with ALL these tags.
294
+ flavor: ``"okf"`` (``[t](id.md)`` graph links + index.md files)
295
+ or ``"obsidian"`` (``[[Title]]`` wikilinks, no index files —
296
+ a vault re-imports losslessly via the name index).
297
+
298
+ Returns:
299
+ List of exported concept IDs.
300
+ """
301
+ if flavor not in ("okf", "obsidian"):
302
+ raise ValueError(f"flavor must be 'okf' or 'obsidian', got {flavor!r}")
303
+ output_dir.mkdir(parents=True, exist_ok=True)
304
+
305
+ # Fetch all concepts (optionally filtered)
306
+ concepts = self._fetch_concepts(
307
+ concept_type=concept_type,
308
+ tags=tags,
309
+ )
310
+
311
+ # If directory_id specified, filter to subtree
312
+ if directory_id:
313
+ concepts = {
314
+ cid: c for cid, c in concepts.items()
315
+ if self._is_under_directory(cid, directory_id)
316
+ }
317
+
318
+ if not concepts:
319
+ return []
320
+
321
+ title_counts = self._title_counts() if flavor == "obsidian" else None
322
+
323
+ # Export each concept, reconstructing path from its ID
324
+ exported: List[str] = []
325
+ for cid, concept in sorted(concepts.items()):
326
+ # Concept IDs use forward slashes; convert to OS path separator
327
+ rel_path = cid.replace("/", os.sep)
328
+ file_path = output_dir / (rel_path + ".md")
329
+ try:
330
+ self._write_okf(concept, file_path, flavor=flavor,
331
+ title_counts=title_counts)
332
+ exported.append(cid)
333
+ except Exception as e:
334
+ print(f" [WARN] Failed to export {cid}: {e}")
335
+
336
+ if flavor == "okf":
337
+ # Generate index.md files for progressive disclosure.
338
+ # Obsidian vaults don't have them (they'd import as stray concepts).
339
+ self._generate_index_files(output_dir, concepts)
340
+
341
+ return exported
342
+
343
+ def export_to_okf(
344
+ self, concept_id: str, output_path: Path, flavor: str = "okf",
345
+ ) -> None:
346
+ """Export a concept back to an OKF .md file (see ``export_bundle``)."""
347
+ concept = self.search_engine.get_by_id(concept_id)
348
+ if not concept:
349
+ raise FileNotFoundError(f"Concept {concept_id} not found")
350
+
351
+ title_counts = self._title_counts() if flavor == "obsidian" else None
352
+ self._write_okf(concept, output_path, flavor=flavor,
353
+ title_counts=title_counts)
354
+
@@ -0,0 +1,319 @@
1
+ """ImageAssetManager — okf-asset:// URI storage, content-hash deduplication,
2
+ and text-based image search.
3
+
4
+ Encoding delegates to the injected EmbeddingEngine; write-epoch bumps route
5
+ to the injected SchemaManager.
6
+ """
7
+
8
+ from __future__ import annotations
9
+ import base64
10
+ import hashlib
11
+ import logging
12
+ import math
13
+ import mimetypes
14
+ import re
15
+ import urllib.parse
16
+ import uuid
17
+ from typing import Any, Callable, Dict, List, Optional, Tuple
18
+ from pathlib import Path
19
+
20
+ from okfgraph.images import (
21
+ EmbedRoute,
22
+ IngestMode,
23
+ build_extracted_images,
24
+ plan_embedding,
25
+ )
26
+ from okfgraph.models import ChunkModel, ConceptModel
27
+
28
+ logger = logging.getLogger(__name__)
29
+
30
+ class ImageAssetManager:
31
+ def __init__(
32
+ self,
33
+ conn,
34
+ embed_engine,
35
+ schema_mgr,
36
+ allow_remote_images: bool,
37
+ allowed_image_domains: List[str],
38
+ bundle_root,
39
+ db=None,
40
+ ):
41
+ self.conn = conn
42
+ # Ladybug Database handle for short-lived index connections
43
+ # (same QUERY_*_INDEX segfault workaround as SearchEngine).
44
+ self._db = db
45
+ self.embed_engine = embed_engine
46
+ self.schema_mgr = schema_mgr
47
+ self.allow_remote_images = allow_remote_images
48
+ self.allowed_image_domains = allowed_image_domains or []
49
+ self.bundle_root = bundle_root
50
+
51
+ def _ingest_concept_images(
52
+ self,
53
+ concept_id: str,
54
+ body: str,
55
+ base_dir: Path,
56
+ mode: "str | IngestMode",
57
+ ) -> Dict[str, int]:
58
+ """Extract, embed, and store the images referenced by a concept.
59
+
60
+ Per-image routing follows ``mode``:
61
+ * ``text`` — alt-text (or filename + image-number fallback), text model
62
+ * ``optional`` — alt-text via text model; images without alt-text via omni
63
+ * ``omni`` — every image via the omni model
64
+
65
+ Unchanged images (same content hash) are skipped so the omni model is
66
+ not re-run on re-import. Images removed from the document are pruned.
67
+ Returns a small stats dict.
68
+ """
69
+ mode = IngestMode.coerce(mode)
70
+
71
+ # Resolve relative image paths against the file's dir, then bundle root.
72
+ search_dirs: List[Path] = []
73
+ for d in (Path(base_dir), self.bundle_root):
74
+ if d not in search_dirs:
75
+ search_dirs.append(d)
76
+
77
+ images = build_extracted_images(
78
+ concept_id, body, search_dirs=search_dirs,
79
+ allow_remote=self.allow_remote_images,
80
+ allowed_domains=self.allowed_image_domains,
81
+ bundle_root=self.bundle_root,
82
+ )
83
+
84
+ stats = {"total": len(images), "text": 0, "omni": 0, "reused": 0, "pruned": 0}
85
+ if not images and not self._concept_has_assets(concept_id):
86
+ return stats
87
+
88
+ existing = self._existing_asset_hashes(concept_id) # {asset_id: content_hash}
89
+
90
+ # --- Encode outside any DB transaction (omni can be slow) ---
91
+ pending: List[Dict[str, Any]] = []
92
+ planned_ids = set()
93
+ for img in images:
94
+ route, caption = plan_embedding(img, mode)
95
+ payload = img.data if route is EmbedRoute.OMNI else (caption or "").encode("utf-8")
96
+ content_hash = self._content_hash(route, payload)
97
+ planned_ids.add(img.asset_id)
98
+
99
+ if existing.get(img.asset_id) == content_hash:
100
+ stats["reused"] += 1
101
+ continue
102
+
103
+ if route is EmbedRoute.OMNI:
104
+ embedding = self.embed_engine._encode_image(img.data)
105
+ stats["omni"] += 1
106
+ else:
107
+ embedding = self.embed_engine._encode(caption or img.filename, task="Document")
108
+ stats["text"] += 1
109
+
110
+ pending.append({
111
+ "img": img,
112
+ "route": route.value,
113
+ "caption": caption or "",
114
+ "content_hash": content_hash,
115
+ "embedding": embedding,
116
+ })
117
+
118
+ stale_ids = [aid for aid in existing if aid not in planned_ids]
119
+ stats["pruned"] = len(stale_ids)
120
+
121
+ if not pending and not stale_ids:
122
+ return stats
123
+
124
+ # --- Write everything atomically ---
125
+ self.conn.execute("BEGIN TRANSACTION")
126
+ try:
127
+ for aid in stale_ids:
128
+ self._delete_image_asset(concept_id, aid)
129
+ for item in pending:
130
+ self._upsert_image_asset(concept_id, item)
131
+ self.conn.execute("COMMIT")
132
+ except Exception:
133
+ try:
134
+ self.conn.execute("ROLLBACK")
135
+ except Exception:
136
+ pass
137
+ raise
138
+
139
+ return stats
140
+
141
+
142
+ @staticmethod
143
+ def _content_hash(route: EmbedRoute, payload: bytes) -> str:
144
+ """Hash that changes whenever the embedding should be recomputed."""
145
+ h = hashlib.sha256()
146
+ h.update(route.value.encode("utf-8"))
147
+ h.update(b"|")
148
+ h.update(payload or b"")
149
+ return h.hexdigest()
150
+
151
+
152
+ def _concept_has_assets(self, concept_id: str) -> bool:
153
+ result = self.conn.execute(
154
+ """
155
+ MATCH (c:Concept {id: $cid})-[:INCLUDES_ASSET]->(i:ImageAsset)
156
+ RETURN count(i) AS cnt
157
+ """,
158
+ {"cid": concept_id},
159
+ )
160
+ rows = result.rows_as_dict().get_all()
161
+ return bool(rows) and rows[0]["cnt"] > 0
162
+
163
+
164
+ def _existing_asset_hashes(self, concept_id: str) -> Dict[str, str]:
165
+ result = self.conn.execute(
166
+ """
167
+ MATCH (c:Concept {id: $cid})-[:INCLUDES_ASSET]->(i:ImageAsset)
168
+ RETURN i.id AS id, i.content_hash AS content_hash
169
+ """,
170
+ {"cid": concept_id},
171
+ )
172
+ return {
173
+ r["id"]: r["content_hash"]
174
+ for r in result.rows_as_dict().get_all()
175
+ }
176
+
177
+
178
+ def _delete_image_asset(self, concept_id: str, asset_id: str) -> None:
179
+ """Unlink an asset from this concept, and delete the node if now orphaned.
180
+
181
+ The concept→asset edge is always removed. The ImageAsset node itself is
182
+ only deleted when no other concept still references it — otherwise a
183
+ shared asset id (e.g. an ``okf-asset://`` passthrough reused by several
184
+ concepts) would be clobbered, or a plain DELETE would fail because the
185
+ node still has edges.
186
+ """
187
+ self.conn.execute(
188
+ """
189
+ MATCH (c:Concept {id: $cid})-[r:INCLUDES_ASSET]->(i:ImageAsset {id: $iid})
190
+ DELETE r
191
+ """,
192
+ {"cid": concept_id, "iid": asset_id},
193
+ )
194
+ self.conn.execute(
195
+ """
196
+ MATCH (i:ImageAsset {id: $iid})
197
+ WHERE NOT EXISTS { MATCH (i)<-[:INCLUDES_ASSET]-(:Concept) }
198
+ DETACH DELETE i
199
+ """,
200
+ {"iid": asset_id},
201
+ )
202
+ self.schema_mgr._bump_write_epoch() # image set changed -> image index dirty
203
+
204
+
205
+ def _upsert_image_asset(self, concept_id: str, item: Dict[str, Any]) -> None:
206
+ """Delete-then-create the ImageAsset, then (re)link it to the concept."""
207
+ img = item["img"]
208
+ # Clear any prior version (edge first, then node).
209
+ self._delete_image_asset(concept_id, img.asset_id)
210
+ self.conn.execute(
211
+ """
212
+ CREATE (i:ImageAsset {
213
+ id: $id, file_name: $file_name, mime_type: $mime_type,
214
+ alt_text: $alt_text, caption: $caption, embed_route: $embed_route,
215
+ content_hash: $content_hash, data: $data, embedding: $embedding
216
+ })
217
+ """,
218
+ {
219
+ "id": img.asset_id,
220
+ "file_name": img.filename,
221
+ "mime_type": img.mime_type,
222
+ "alt_text": img.alt_text or "",
223
+ "caption": item["caption"],
224
+ "embed_route": item["route"],
225
+ "content_hash": item["content_hash"],
226
+ "data": img.data if img.data is not None else b"",
227
+ "embedding": item["embedding"],
228
+ },
229
+ )
230
+ self.conn.execute(
231
+ """
232
+ MATCH (c:Concept {id: $cid}), (i:ImageAsset {id: $iid})
233
+ MERGE (c)-[:INCLUDES_ASSET]->(i)
234
+ """,
235
+ {"cid": concept_id, "iid": img.asset_id},
236
+ )
237
+ self.schema_mgr._bump_write_epoch() # new/updated image -> image index dirty
238
+
239
+
240
+ def list_images(self, concept_id: str) -> List[Dict[str, Any]]:
241
+ """List the image assets attached to a concept (no BLOB payloads)."""
242
+ result = self.conn.execute(
243
+ """
244
+ MATCH (c:Concept {id: $cid})-[:INCLUDES_ASSET]->(i:ImageAsset)
245
+ RETURN i.id AS id, i.file_name AS file_name, i.mime_type AS mime_type,
246
+ i.alt_text AS alt_text, i.embed_route AS embed_route
247
+ """,
248
+ {"cid": concept_id},
249
+ )
250
+ return result.rows_as_dict().get_all()
251
+
252
+
253
+ def get_image_data(self, asset_id: str) -> Optional[Dict[str, Any]]:
254
+ """Fetch a single image asset including its raw BLOB bytes."""
255
+ result = self.conn.execute(
256
+ """
257
+ MATCH (i:ImageAsset {id: $iid})
258
+ RETURN i.id AS id, i.file_name AS file_name, i.mime_type AS mime_type,
259
+ i.alt_text AS alt_text, i.embed_route AS embed_route, i.data AS data
260
+ """,
261
+ {"iid": asset_id},
262
+ )
263
+ rows = result.rows_as_dict().get_all()
264
+ return rows[0] if rows else None
265
+
266
+
267
+ def search_images_with_text(
268
+ self,
269
+ text_query: str,
270
+ use_text_model: bool = True,
271
+ limit: int = 10,
272
+ ) -> List[Dict[str, Any]]:
273
+ """Find image assets from a text query via the unified vector index.
274
+
275
+ ``use_text_model=True`` (default) encodes the query with the lightweight
276
+ text model — no omni load required, since both models share the vector
277
+ space. Set it to ``False`` to route the query through the omni text side.
278
+ """
279
+ if use_text_model:
280
+ query_vec = self.embed_engine._encode(text_query, task="Query")
281
+ else:
282
+ query_vec = self.embed_engine._encode_omni_text(text_query, task="Query")
283
+
284
+ if self._db is None:
285
+ result = self.conn.execute(
286
+ "CALL QUERY_VECTOR_INDEX('ImageAsset', 'image_omni_idx', $vec, $k) "
287
+ "RETURN node, distance",
288
+ {"vec": query_vec, "k": limit},
289
+ )
290
+ rows = result.rows_as_dict().get_all()
291
+ else:
292
+ import ladybug as lb
293
+
294
+ c = lb.Connection(self._db)
295
+ try:
296
+ c.execute("LOAD EXTENSION VECTOR")
297
+ result = c.execute(
298
+ "CALL QUERY_VECTOR_INDEX('ImageAsset', 'image_omni_idx', $vec, $k) "
299
+ "RETURN node, distance",
300
+ {"vec": query_vec, "k": limit},
301
+ )
302
+ rows = result.rows_as_dict().get_all()
303
+ finally:
304
+ c.close()
305
+ out: List[Dict[str, Any]] = []
306
+ for row in rows:
307
+ node = row.get("node", {})
308
+ if not isinstance(node, dict):
309
+ continue
310
+ out.append({
311
+ "id": node.get("id"),
312
+ "file_name": node.get("file_name"),
313
+ "alt_text": node.get("alt_text"),
314
+ "embed_route": node.get("embed_route"),
315
+ "distance": row.get("distance"),
316
+ "relevance_score": 1 - row.get("distance", 0),
317
+ })
318
+ return out
319
+