okfgraph 0.2.4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- okfgraph/__init__.py +7 -0
- okfgraph/cli.py +1364 -0
- okfgraph/components/__init__.py +61 -0
- okfgraph/components/converters.py +128 -0
- okfgraph/components/delta.py +271 -0
- okfgraph/components/diff.py +172 -0
- okfgraph/components/doctor.py +216 -0
- okfgraph/components/embedding.py +323 -0
- okfgraph/components/export.py +354 -0
- okfgraph/components/image_assets.py +319 -0
- okfgraph/components/import_.py +1110 -0
- okfgraph/components/ingest.py +495 -0
- okfgraph/components/links.py +133 -0
- okfgraph/components/lint.py +118 -0
- okfgraph/components/purge.py +342 -0
- okfgraph/components/ranking.py +168 -0
- okfgraph/components/schema.py +443 -0
- okfgraph/components/search.py +1006 -0
- okfgraph/config.py +393 -0
- okfgraph/images.py +435 -0
- okfgraph/mcp_server.py +461 -0
- okfgraph/models.py +128 -0
- okfgraph/router.py +485 -0
- okfgraph/security.py +269 -0
- okfgraph/tools.py +460 -0
- okfgraph-0.2.4.dist-info/METADATA +25 -0
- okfgraph-0.2.4.dist-info/RECORD +31 -0
- okfgraph-0.2.4.dist-info/WHEEL +5 -0
- okfgraph-0.2.4.dist-info/entry_points.txt +3 -0
- okfgraph-0.2.4.dist-info/licenses/LICENSE +6 -0
- okfgraph-0.2.4.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,354 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import base64
|
|
4
|
+
import hashlib
|
|
5
|
+
import heapq
|
|
6
|
+
import json
|
|
7
|
+
import logging
|
|
8
|
+
import math
|
|
9
|
+
import os
|
|
10
|
+
import re
|
|
11
|
+
import time
|
|
12
|
+
import uuid
|
|
13
|
+
from contextlib import contextmanager
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Any, Callable, Dict, List, Optional, Tuple, Union, Set
|
|
16
|
+
from urllib.parse import urlparse
|
|
17
|
+
|
|
18
|
+
import mordant
|
|
19
|
+
import numpy as np
|
|
20
|
+
import yaml
|
|
21
|
+
import frontmatter
|
|
22
|
+
from okfgraph.models import ChunkModel, ConceptModel
|
|
23
|
+
|
|
24
|
+
logger = logging.getLogger(__name__)
|
|
25
|
+
|
|
26
|
+
class ExportManager:
|
|
27
|
+
def __init__(self, conn, search_engine):
|
|
28
|
+
self.conn = conn
|
|
29
|
+
self.search_engine = search_engine
|
|
30
|
+
|
|
31
|
+
def _enrich_body_with_graph_links(
|
|
32
|
+
self, concept_id: str, body: str, flavor: str = "okf",
|
|
33
|
+
title_counts: Optional[Dict[str, int]] = None,
|
|
34
|
+
) -> str:
|
|
35
|
+
"""Enrich body with graph-derived links so the exported markdown
|
|
36
|
+
faithfully reflects the LINKS_TO graph.
|
|
37
|
+
|
|
38
|
+
Strategy (Option A — append, never replace):
|
|
39
|
+
1. Query all outgoing LINKS_TO edges from this concept.
|
|
40
|
+
2. For each target, check if a link to that target already exists
|
|
41
|
+
in the body (by matching the target_id in link URLs).
|
|
42
|
+
3. If not already linked, append a "See Also" bullet.
|
|
43
|
+
4. Query all incoming LINKS_TO edges (concepts that link TO this one).
|
|
44
|
+
5. If any exist, append a "Cited By" bullet list.
|
|
45
|
+
|
|
46
|
+
This preserves the original body's links (which may have richer anchor
|
|
47
|
+
text) while ensuring the graph structure is expressed in the export.
|
|
48
|
+
|
|
49
|
+
``flavor="obsidian"`` renders appended links as ``[[Title]]``
|
|
50
|
+
(unique titles) or ``[[id|Title]]`` instead of ``[t](id.md)``; the
|
|
51
|
+
original body is never rewritten in either flavor. The obsidian
|
|
52
|
+
flavor omits the "Cited By" section: backlinks re-imported as
|
|
53
|
+
forward wikilinks would reverse their direction, and Obsidian
|
|
54
|
+
renders backlinks natively anyway.
|
|
55
|
+
"""
|
|
56
|
+
import re
|
|
57
|
+
|
|
58
|
+
wiki_refs = {
|
|
59
|
+
m.split("|", 1)[0].strip()
|
|
60
|
+
for m in re.findall(r"\[\[([^\]]+)\]\]", body or "")
|
|
61
|
+
}
|
|
62
|
+
wiki_refs_lower = {r.lower() for r in wiki_refs}
|
|
63
|
+
|
|
64
|
+
def _link(target_id: str, title: str) -> str:
|
|
65
|
+
label = title or target_id.split("/")[-1]
|
|
66
|
+
if flavor == "obsidian":
|
|
67
|
+
if label and (title_counts or {}).get(label.lower(), 0) == 1:
|
|
68
|
+
return f"- [[{label}]]"
|
|
69
|
+
return f"- [[{target_id}|{label}]]"
|
|
70
|
+
return f"- [{label}]({target_id}.md)"
|
|
71
|
+
|
|
72
|
+
parts: List[str] = []
|
|
73
|
+
|
|
74
|
+
# --- Outgoing links (See Also) ---
|
|
75
|
+
result = self.conn.execute("""
|
|
76
|
+
MATCH (s:Concept {id: $cid})-[:LINKS_TO]->(t:Concept)
|
|
77
|
+
RETURN t.id AS target_id, t.title AS title, t.type AS type
|
|
78
|
+
ORDER BY t.title
|
|
79
|
+
""", {"cid": concept_id})
|
|
80
|
+
outgoing_rows = result.rows_as_dict().get_all()
|
|
81
|
+
|
|
82
|
+
if outgoing_rows:
|
|
83
|
+
# Determine which targets are already linked in the body:
|
|
84
|
+
# either a [t](...target...) URL or an equivalent [[ref]].
|
|
85
|
+
existing_link_targets = set()
|
|
86
|
+
for row in outgoing_rows:
|
|
87
|
+
target_id = row["target_id"]
|
|
88
|
+
title = row["title"] or ""
|
|
89
|
+
if target_id in wiki_refs or title.lower() in wiki_refs_lower:
|
|
90
|
+
existing_link_targets.add(target_id)
|
|
91
|
+
continue
|
|
92
|
+
# Check if target_id appears in any link URL in the body
|
|
93
|
+
link_pattern = re.compile(
|
|
94
|
+
r"\]\(([^)]*?" + re.escape(target_id) + r"[^)]*)\)"
|
|
95
|
+
)
|
|
96
|
+
if link_pattern.search(body):
|
|
97
|
+
existing_link_targets.add(target_id)
|
|
98
|
+
|
|
99
|
+
# Collect targets that need a link added
|
|
100
|
+
new_links = []
|
|
101
|
+
for row in outgoing_rows:
|
|
102
|
+
target_id = row["target_id"]
|
|
103
|
+
if target_id not in existing_link_targets:
|
|
104
|
+
title = row["title"] or target_id.split("/")[-1]
|
|
105
|
+
new_links.append(_link(target_id, title))
|
|
106
|
+
|
|
107
|
+
if new_links:
|
|
108
|
+
parts.append("\n## See Also\n" + "\n".join(new_links))
|
|
109
|
+
|
|
110
|
+
# --- Incoming links (Cited By: okf flavor only) ---
|
|
111
|
+
# Obsidian renders backlinks natively; emitting them as forward
|
|
112
|
+
# [[wikilinks]] would reverse the edge on re-import (round-trip loss).
|
|
113
|
+
result = self.conn.execute("""
|
|
114
|
+
MATCH (s:Concept)-[:LINKS_TO]->(t:Concept {id: $cid})
|
|
115
|
+
RETURN s.id AS source_id, s.title AS title, s.type AS type
|
|
116
|
+
ORDER BY s.title
|
|
117
|
+
""", {"cid": concept_id})
|
|
118
|
+
incoming_rows = result.rows_as_dict().get_all()
|
|
119
|
+
|
|
120
|
+
if incoming_rows and flavor == "okf":
|
|
121
|
+
cited_lines = ["\n## Cited By\n"]
|
|
122
|
+
for row in incoming_rows:
|
|
123
|
+
source_id = row["source_id"]
|
|
124
|
+
title = row["title"] or source_id.split("/")[-1]
|
|
125
|
+
cited_lines.append(_link(source_id, title))
|
|
126
|
+
parts.append("\n".join(cited_lines))
|
|
127
|
+
|
|
128
|
+
return body + "".join(parts)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _fetch_concepts(
|
|
132
|
+
self,
|
|
133
|
+
concept_type: Optional[str] = None,
|
|
134
|
+
tags: Optional[List[str]] = None,
|
|
135
|
+
) -> Dict[str, ConceptModel]:
|
|
136
|
+
"""Fetch all concepts, optionally filtered by type and tags."""
|
|
137
|
+
where_clauses: list[str] = []
|
|
138
|
+
params: Dict[str, Any] = {}
|
|
139
|
+
|
|
140
|
+
if concept_type:
|
|
141
|
+
where_clauses.append("c.type = $type")
|
|
142
|
+
params["type"] = concept_type
|
|
143
|
+
if tags:
|
|
144
|
+
where_clauses.append("ALL(tag IN $tags WHERE tag IN c.tags)")
|
|
145
|
+
params["tags"] = tags
|
|
146
|
+
|
|
147
|
+
where_str = " AND ".join(where_clauses) if where_clauses else "true"
|
|
148
|
+
query = f"""
|
|
149
|
+
MATCH (c:Concept)
|
|
150
|
+
WHERE {where_str}
|
|
151
|
+
RETURN c.id, c.type, c.title, c.description, c.resource,
|
|
152
|
+
c.tags, c.timestamp, c.body, c.embedding, c.extra
|
|
153
|
+
"""
|
|
154
|
+
results = self.conn.execute(query, params)
|
|
155
|
+
rows = results.rows_as_dict().get_all()
|
|
156
|
+
|
|
157
|
+
concepts: Dict[str, ConceptModel] = {}
|
|
158
|
+
for row in rows:
|
|
159
|
+
data: Dict[str, Any] = {}
|
|
160
|
+
for key, val in row.items():
|
|
161
|
+
col = key.split(".", 1)[-1] # strip 'c.' prefix
|
|
162
|
+
if col != "extra":
|
|
163
|
+
data[col] = val
|
|
164
|
+
|
|
165
|
+
# Decode extra MAP fields
|
|
166
|
+
extra = row.get("c.extra") or {}
|
|
167
|
+
for k, v in extra.items():
|
|
168
|
+
if isinstance(v, str) and v.startswith(("{", "[")):
|
|
169
|
+
try:
|
|
170
|
+
data[k] = json.loads(v)
|
|
171
|
+
except json.JSONDecodeError:
|
|
172
|
+
data[k] = v
|
|
173
|
+
else:
|
|
174
|
+
data[k] = v
|
|
175
|
+
|
|
176
|
+
try:
|
|
177
|
+
concepts[data["id"]] = ConceptModel.model_validate(data)
|
|
178
|
+
except Exception:
|
|
179
|
+
pass # Skip malformed concepts
|
|
180
|
+
|
|
181
|
+
return concepts
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _generate_index_files(
|
|
185
|
+
self, output_dir: Path, concepts: Dict[str, ConceptModel]
|
|
186
|
+
) -> None:
|
|
187
|
+
"""Generate index.md files for every directory in the bundle.
|
|
188
|
+
|
|
189
|
+
Each index.md lists the children (concepts and subdirectories) of that
|
|
190
|
+
directory, enabling progressive disclosure for OKF consumers.
|
|
191
|
+
"""
|
|
192
|
+
# Build a map of directory_id → list of (title, relative_path) children
|
|
193
|
+
dir_children: Dict[str, List[Tuple[str, str]]] = {}
|
|
194
|
+
|
|
195
|
+
for cid, concept in concepts.items():
|
|
196
|
+
parts = cid.split("/")
|
|
197
|
+
for i in range(1, len(parts)):
|
|
198
|
+
dir_id = "/".join(parts[:i])
|
|
199
|
+
dir_children.setdefault(dir_id, [])
|
|
200
|
+
child_title = concept.title or parts[i]
|
|
201
|
+
child_rel = cid.replace("/", os.sep) + ".md"
|
|
202
|
+
dir_children[dir_id].append((child_title, child_rel))
|
|
203
|
+
|
|
204
|
+
# Write index.md for each directory
|
|
205
|
+
for dir_id, children in dir_children.items():
|
|
206
|
+
# Sort children by title
|
|
207
|
+
children.sort(key=lambda x: x[0])
|
|
208
|
+
lines = [
|
|
209
|
+
f"# {dir_id.split('/')[-1] or '(root)'}\n",
|
|
210
|
+
"",
|
|
211
|
+
]
|
|
212
|
+
for title, rel_path in children:
|
|
213
|
+
lines.append(f"- [{title}]({rel_path})")
|
|
214
|
+
lines.append("")
|
|
215
|
+
|
|
216
|
+
# Create parent directories if needed
|
|
217
|
+
dir_path = output_dir / dir_id.replace("/", os.sep)
|
|
218
|
+
dir_path.mkdir(parents=True, exist_ok=True)
|
|
219
|
+
(dir_path / "index.md").write_text("\n".join(lines), encoding="utf-8")
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _is_under_directory(self, concept_id: str, directory_id: str) -> bool:
|
|
223
|
+
"""Check if a concept is under a given directory (via CONTAINS graph)."""
|
|
224
|
+
result = self.conn.execute("""
|
|
225
|
+
MATCH (d:Directory {id: $dir_id})-[:CONTAINS*1..5]->(c:Concept {id: $cid})
|
|
226
|
+
RETURN count(c) AS cnt
|
|
227
|
+
""", {"dir_id": directory_id, "cid": concept_id})
|
|
228
|
+
rows = result.rows_as_dict().get_all()
|
|
229
|
+
return rows[0]["cnt"] > 0 if rows else False
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def _title_counts(self) -> Dict[str, int]:
|
|
233
|
+
"""Count concepts per (lowercased) title.
|
|
234
|
+
|
|
235
|
+
Lowercased to match the case-insensitive name index on import: two
|
|
236
|
+
titles differing only by case are ambiguous as ``[[Title]]`` and
|
|
237
|
+
must both export disambiguated.
|
|
238
|
+
"""
|
|
239
|
+
counts: Dict[str, int] = {}
|
|
240
|
+
for r in self.conn.execute(
|
|
241
|
+
"MATCH (c:Concept) RETURN c.title"
|
|
242
|
+
).rows_as_dict().get_all():
|
|
243
|
+
title = (r.get("c.title") or "").lower()
|
|
244
|
+
counts[title] = counts.get(title, 0) + 1
|
|
245
|
+
return counts
|
|
246
|
+
|
|
247
|
+
def _write_okf(
|
|
248
|
+
self, concept: ConceptModel, output_path: Path,
|
|
249
|
+
flavor: str = "okf", title_counts: Optional[Dict[str, int]] = None,
|
|
250
|
+
) -> None:
|
|
251
|
+
"""Internal: serialize a ConceptModel to an OKF .md file.
|
|
252
|
+
|
|
253
|
+
Enriches the body with LINKS_TO relationships from the graph so that
|
|
254
|
+
exported markdown faithfully reflects the graph structure.
|
|
255
|
+
|
|
256
|
+
``flavor="obsidian"`` renders appended graph links as ``[[Title]]``
|
|
257
|
+
wikilinks; the default ``okf`` flavor uses ``[t](id.md)`` links.
|
|
258
|
+
Either way the stored ``uid`` (frontmatter ``id:`` on import) is
|
|
259
|
+
written back as ``id:`` so re-import is lossless.
|
|
260
|
+
"""
|
|
261
|
+
data, body = concept.export_frontmatter()
|
|
262
|
+
|
|
263
|
+
yaml_str = yaml.dump(
|
|
264
|
+
data, default_flow_style=False, allow_unicode=True, sort_keys=False
|
|
265
|
+
)
|
|
266
|
+
|
|
267
|
+
# ENRICH: add graph-derived links to the body
|
|
268
|
+
body = self._enrich_body_with_graph_links(
|
|
269
|
+
concept.id, body, flavor=flavor, title_counts=title_counts,
|
|
270
|
+
)
|
|
271
|
+
|
|
272
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
273
|
+
output_path.write_text(f"---\n{yaml_str}---\n\n{body}", encoding="utf-8")
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def export_bundle(
|
|
277
|
+
self,
|
|
278
|
+
output_dir: Path,
|
|
279
|
+
directory_id: Optional[str] = None,
|
|
280
|
+
concept_type: Optional[str] = None,
|
|
281
|
+
tags: Optional[List[str]] = None,
|
|
282
|
+
flavor: str = "okf",
|
|
283
|
+
) -> List[str]:
|
|
284
|
+
"""Export concepts from the graph back to an OKF bundle directory.
|
|
285
|
+
|
|
286
|
+
Reconstructs the full directory hierarchy from CONTAINS relationships.
|
|
287
|
+
Supports filtering by directory subtree, concept type, or tags.
|
|
288
|
+
|
|
289
|
+
Args:
|
|
290
|
+
output_dir: Root directory to write the bundle into.
|
|
291
|
+
directory_id: If set, only export concepts under this directory.
|
|
292
|
+
concept_type: If set, only export concepts of this type.
|
|
293
|
+
tags: If set, only export concepts with ALL these tags.
|
|
294
|
+
flavor: ``"okf"`` (``[t](id.md)`` graph links + index.md files)
|
|
295
|
+
or ``"obsidian"`` (``[[Title]]`` wikilinks, no index files —
|
|
296
|
+
a vault re-imports losslessly via the name index).
|
|
297
|
+
|
|
298
|
+
Returns:
|
|
299
|
+
List of exported concept IDs.
|
|
300
|
+
"""
|
|
301
|
+
if flavor not in ("okf", "obsidian"):
|
|
302
|
+
raise ValueError(f"flavor must be 'okf' or 'obsidian', got {flavor!r}")
|
|
303
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
304
|
+
|
|
305
|
+
# Fetch all concepts (optionally filtered)
|
|
306
|
+
concepts = self._fetch_concepts(
|
|
307
|
+
concept_type=concept_type,
|
|
308
|
+
tags=tags,
|
|
309
|
+
)
|
|
310
|
+
|
|
311
|
+
# If directory_id specified, filter to subtree
|
|
312
|
+
if directory_id:
|
|
313
|
+
concepts = {
|
|
314
|
+
cid: c for cid, c in concepts.items()
|
|
315
|
+
if self._is_under_directory(cid, directory_id)
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
if not concepts:
|
|
319
|
+
return []
|
|
320
|
+
|
|
321
|
+
title_counts = self._title_counts() if flavor == "obsidian" else None
|
|
322
|
+
|
|
323
|
+
# Export each concept, reconstructing path from its ID
|
|
324
|
+
exported: List[str] = []
|
|
325
|
+
for cid, concept in sorted(concepts.items()):
|
|
326
|
+
# Concept IDs use forward slashes; convert to OS path separator
|
|
327
|
+
rel_path = cid.replace("/", os.sep)
|
|
328
|
+
file_path = output_dir / (rel_path + ".md")
|
|
329
|
+
try:
|
|
330
|
+
self._write_okf(concept, file_path, flavor=flavor,
|
|
331
|
+
title_counts=title_counts)
|
|
332
|
+
exported.append(cid)
|
|
333
|
+
except Exception as e:
|
|
334
|
+
print(f" [WARN] Failed to export {cid}: {e}")
|
|
335
|
+
|
|
336
|
+
if flavor == "okf":
|
|
337
|
+
# Generate index.md files for progressive disclosure.
|
|
338
|
+
# Obsidian vaults don't have them (they'd import as stray concepts).
|
|
339
|
+
self._generate_index_files(output_dir, concepts)
|
|
340
|
+
|
|
341
|
+
return exported
|
|
342
|
+
|
|
343
|
+
def export_to_okf(
|
|
344
|
+
self, concept_id: str, output_path: Path, flavor: str = "okf",
|
|
345
|
+
) -> None:
|
|
346
|
+
"""Export a concept back to an OKF .md file (see ``export_bundle``)."""
|
|
347
|
+
concept = self.search_engine.get_by_id(concept_id)
|
|
348
|
+
if not concept:
|
|
349
|
+
raise FileNotFoundError(f"Concept {concept_id} not found")
|
|
350
|
+
|
|
351
|
+
title_counts = self._title_counts() if flavor == "obsidian" else None
|
|
352
|
+
self._write_okf(concept, output_path, flavor=flavor,
|
|
353
|
+
title_counts=title_counts)
|
|
354
|
+
|
|
@@ -0,0 +1,319 @@
|
|
|
1
|
+
"""ImageAssetManager — okf-asset:// URI storage, content-hash deduplication,
|
|
2
|
+
and text-based image search.
|
|
3
|
+
|
|
4
|
+
Encoding delegates to the injected EmbeddingEngine; write-epoch bumps route
|
|
5
|
+
to the injected SchemaManager.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
import base64
|
|
10
|
+
import hashlib
|
|
11
|
+
import logging
|
|
12
|
+
import math
|
|
13
|
+
import mimetypes
|
|
14
|
+
import re
|
|
15
|
+
import urllib.parse
|
|
16
|
+
import uuid
|
|
17
|
+
from typing import Any, Callable, Dict, List, Optional, Tuple
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
|
|
20
|
+
from okfgraph.images import (
|
|
21
|
+
EmbedRoute,
|
|
22
|
+
IngestMode,
|
|
23
|
+
build_extracted_images,
|
|
24
|
+
plan_embedding,
|
|
25
|
+
)
|
|
26
|
+
from okfgraph.models import ChunkModel, ConceptModel
|
|
27
|
+
|
|
28
|
+
logger = logging.getLogger(__name__)
|
|
29
|
+
|
|
30
|
+
class ImageAssetManager:
|
|
31
|
+
def __init__(
|
|
32
|
+
self,
|
|
33
|
+
conn,
|
|
34
|
+
embed_engine,
|
|
35
|
+
schema_mgr,
|
|
36
|
+
allow_remote_images: bool,
|
|
37
|
+
allowed_image_domains: List[str],
|
|
38
|
+
bundle_root,
|
|
39
|
+
db=None,
|
|
40
|
+
):
|
|
41
|
+
self.conn = conn
|
|
42
|
+
# Ladybug Database handle for short-lived index connections
|
|
43
|
+
# (same QUERY_*_INDEX segfault workaround as SearchEngine).
|
|
44
|
+
self._db = db
|
|
45
|
+
self.embed_engine = embed_engine
|
|
46
|
+
self.schema_mgr = schema_mgr
|
|
47
|
+
self.allow_remote_images = allow_remote_images
|
|
48
|
+
self.allowed_image_domains = allowed_image_domains or []
|
|
49
|
+
self.bundle_root = bundle_root
|
|
50
|
+
|
|
51
|
+
def _ingest_concept_images(
|
|
52
|
+
self,
|
|
53
|
+
concept_id: str,
|
|
54
|
+
body: str,
|
|
55
|
+
base_dir: Path,
|
|
56
|
+
mode: "str | IngestMode",
|
|
57
|
+
) -> Dict[str, int]:
|
|
58
|
+
"""Extract, embed, and store the images referenced by a concept.
|
|
59
|
+
|
|
60
|
+
Per-image routing follows ``mode``:
|
|
61
|
+
* ``text`` — alt-text (or filename + image-number fallback), text model
|
|
62
|
+
* ``optional`` — alt-text via text model; images without alt-text via omni
|
|
63
|
+
* ``omni`` — every image via the omni model
|
|
64
|
+
|
|
65
|
+
Unchanged images (same content hash) are skipped so the omni model is
|
|
66
|
+
not re-run on re-import. Images removed from the document are pruned.
|
|
67
|
+
Returns a small stats dict.
|
|
68
|
+
"""
|
|
69
|
+
mode = IngestMode.coerce(mode)
|
|
70
|
+
|
|
71
|
+
# Resolve relative image paths against the file's dir, then bundle root.
|
|
72
|
+
search_dirs: List[Path] = []
|
|
73
|
+
for d in (Path(base_dir), self.bundle_root):
|
|
74
|
+
if d not in search_dirs:
|
|
75
|
+
search_dirs.append(d)
|
|
76
|
+
|
|
77
|
+
images = build_extracted_images(
|
|
78
|
+
concept_id, body, search_dirs=search_dirs,
|
|
79
|
+
allow_remote=self.allow_remote_images,
|
|
80
|
+
allowed_domains=self.allowed_image_domains,
|
|
81
|
+
bundle_root=self.bundle_root,
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
stats = {"total": len(images), "text": 0, "omni": 0, "reused": 0, "pruned": 0}
|
|
85
|
+
if not images and not self._concept_has_assets(concept_id):
|
|
86
|
+
return stats
|
|
87
|
+
|
|
88
|
+
existing = self._existing_asset_hashes(concept_id) # {asset_id: content_hash}
|
|
89
|
+
|
|
90
|
+
# --- Encode outside any DB transaction (omni can be slow) ---
|
|
91
|
+
pending: List[Dict[str, Any]] = []
|
|
92
|
+
planned_ids = set()
|
|
93
|
+
for img in images:
|
|
94
|
+
route, caption = plan_embedding(img, mode)
|
|
95
|
+
payload = img.data if route is EmbedRoute.OMNI else (caption or "").encode("utf-8")
|
|
96
|
+
content_hash = self._content_hash(route, payload)
|
|
97
|
+
planned_ids.add(img.asset_id)
|
|
98
|
+
|
|
99
|
+
if existing.get(img.asset_id) == content_hash:
|
|
100
|
+
stats["reused"] += 1
|
|
101
|
+
continue
|
|
102
|
+
|
|
103
|
+
if route is EmbedRoute.OMNI:
|
|
104
|
+
embedding = self.embed_engine._encode_image(img.data)
|
|
105
|
+
stats["omni"] += 1
|
|
106
|
+
else:
|
|
107
|
+
embedding = self.embed_engine._encode(caption or img.filename, task="Document")
|
|
108
|
+
stats["text"] += 1
|
|
109
|
+
|
|
110
|
+
pending.append({
|
|
111
|
+
"img": img,
|
|
112
|
+
"route": route.value,
|
|
113
|
+
"caption": caption or "",
|
|
114
|
+
"content_hash": content_hash,
|
|
115
|
+
"embedding": embedding,
|
|
116
|
+
})
|
|
117
|
+
|
|
118
|
+
stale_ids = [aid for aid in existing if aid not in planned_ids]
|
|
119
|
+
stats["pruned"] = len(stale_ids)
|
|
120
|
+
|
|
121
|
+
if not pending and not stale_ids:
|
|
122
|
+
return stats
|
|
123
|
+
|
|
124
|
+
# --- Write everything atomically ---
|
|
125
|
+
self.conn.execute("BEGIN TRANSACTION")
|
|
126
|
+
try:
|
|
127
|
+
for aid in stale_ids:
|
|
128
|
+
self._delete_image_asset(concept_id, aid)
|
|
129
|
+
for item in pending:
|
|
130
|
+
self._upsert_image_asset(concept_id, item)
|
|
131
|
+
self.conn.execute("COMMIT")
|
|
132
|
+
except Exception:
|
|
133
|
+
try:
|
|
134
|
+
self.conn.execute("ROLLBACK")
|
|
135
|
+
except Exception:
|
|
136
|
+
pass
|
|
137
|
+
raise
|
|
138
|
+
|
|
139
|
+
return stats
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
@staticmethod
|
|
143
|
+
def _content_hash(route: EmbedRoute, payload: bytes) -> str:
|
|
144
|
+
"""Hash that changes whenever the embedding should be recomputed."""
|
|
145
|
+
h = hashlib.sha256()
|
|
146
|
+
h.update(route.value.encode("utf-8"))
|
|
147
|
+
h.update(b"|")
|
|
148
|
+
h.update(payload or b"")
|
|
149
|
+
return h.hexdigest()
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _concept_has_assets(self, concept_id: str) -> bool:
|
|
153
|
+
result = self.conn.execute(
|
|
154
|
+
"""
|
|
155
|
+
MATCH (c:Concept {id: $cid})-[:INCLUDES_ASSET]->(i:ImageAsset)
|
|
156
|
+
RETURN count(i) AS cnt
|
|
157
|
+
""",
|
|
158
|
+
{"cid": concept_id},
|
|
159
|
+
)
|
|
160
|
+
rows = result.rows_as_dict().get_all()
|
|
161
|
+
return bool(rows) and rows[0]["cnt"] > 0
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def _existing_asset_hashes(self, concept_id: str) -> Dict[str, str]:
|
|
165
|
+
result = self.conn.execute(
|
|
166
|
+
"""
|
|
167
|
+
MATCH (c:Concept {id: $cid})-[:INCLUDES_ASSET]->(i:ImageAsset)
|
|
168
|
+
RETURN i.id AS id, i.content_hash AS content_hash
|
|
169
|
+
""",
|
|
170
|
+
{"cid": concept_id},
|
|
171
|
+
)
|
|
172
|
+
return {
|
|
173
|
+
r["id"]: r["content_hash"]
|
|
174
|
+
for r in result.rows_as_dict().get_all()
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _delete_image_asset(self, concept_id: str, asset_id: str) -> None:
|
|
179
|
+
"""Unlink an asset from this concept, and delete the node if now orphaned.
|
|
180
|
+
|
|
181
|
+
The concept→asset edge is always removed. The ImageAsset node itself is
|
|
182
|
+
only deleted when no other concept still references it — otherwise a
|
|
183
|
+
shared asset id (e.g. an ``okf-asset://`` passthrough reused by several
|
|
184
|
+
concepts) would be clobbered, or a plain DELETE would fail because the
|
|
185
|
+
node still has edges.
|
|
186
|
+
"""
|
|
187
|
+
self.conn.execute(
|
|
188
|
+
"""
|
|
189
|
+
MATCH (c:Concept {id: $cid})-[r:INCLUDES_ASSET]->(i:ImageAsset {id: $iid})
|
|
190
|
+
DELETE r
|
|
191
|
+
""",
|
|
192
|
+
{"cid": concept_id, "iid": asset_id},
|
|
193
|
+
)
|
|
194
|
+
self.conn.execute(
|
|
195
|
+
"""
|
|
196
|
+
MATCH (i:ImageAsset {id: $iid})
|
|
197
|
+
WHERE NOT EXISTS { MATCH (i)<-[:INCLUDES_ASSET]-(:Concept) }
|
|
198
|
+
DETACH DELETE i
|
|
199
|
+
""",
|
|
200
|
+
{"iid": asset_id},
|
|
201
|
+
)
|
|
202
|
+
self.schema_mgr._bump_write_epoch() # image set changed -> image index dirty
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def _upsert_image_asset(self, concept_id: str, item: Dict[str, Any]) -> None:
|
|
206
|
+
"""Delete-then-create the ImageAsset, then (re)link it to the concept."""
|
|
207
|
+
img = item["img"]
|
|
208
|
+
# Clear any prior version (edge first, then node).
|
|
209
|
+
self._delete_image_asset(concept_id, img.asset_id)
|
|
210
|
+
self.conn.execute(
|
|
211
|
+
"""
|
|
212
|
+
CREATE (i:ImageAsset {
|
|
213
|
+
id: $id, file_name: $file_name, mime_type: $mime_type,
|
|
214
|
+
alt_text: $alt_text, caption: $caption, embed_route: $embed_route,
|
|
215
|
+
content_hash: $content_hash, data: $data, embedding: $embedding
|
|
216
|
+
})
|
|
217
|
+
""",
|
|
218
|
+
{
|
|
219
|
+
"id": img.asset_id,
|
|
220
|
+
"file_name": img.filename,
|
|
221
|
+
"mime_type": img.mime_type,
|
|
222
|
+
"alt_text": img.alt_text or "",
|
|
223
|
+
"caption": item["caption"],
|
|
224
|
+
"embed_route": item["route"],
|
|
225
|
+
"content_hash": item["content_hash"],
|
|
226
|
+
"data": img.data if img.data is not None else b"",
|
|
227
|
+
"embedding": item["embedding"],
|
|
228
|
+
},
|
|
229
|
+
)
|
|
230
|
+
self.conn.execute(
|
|
231
|
+
"""
|
|
232
|
+
MATCH (c:Concept {id: $cid}), (i:ImageAsset {id: $iid})
|
|
233
|
+
MERGE (c)-[:INCLUDES_ASSET]->(i)
|
|
234
|
+
""",
|
|
235
|
+
{"cid": concept_id, "iid": img.asset_id},
|
|
236
|
+
)
|
|
237
|
+
self.schema_mgr._bump_write_epoch() # new/updated image -> image index dirty
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def list_images(self, concept_id: str) -> List[Dict[str, Any]]:
|
|
241
|
+
"""List the image assets attached to a concept (no BLOB payloads)."""
|
|
242
|
+
result = self.conn.execute(
|
|
243
|
+
"""
|
|
244
|
+
MATCH (c:Concept {id: $cid})-[:INCLUDES_ASSET]->(i:ImageAsset)
|
|
245
|
+
RETURN i.id AS id, i.file_name AS file_name, i.mime_type AS mime_type,
|
|
246
|
+
i.alt_text AS alt_text, i.embed_route AS embed_route
|
|
247
|
+
""",
|
|
248
|
+
{"cid": concept_id},
|
|
249
|
+
)
|
|
250
|
+
return result.rows_as_dict().get_all()
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def get_image_data(self, asset_id: str) -> Optional[Dict[str, Any]]:
|
|
254
|
+
"""Fetch a single image asset including its raw BLOB bytes."""
|
|
255
|
+
result = self.conn.execute(
|
|
256
|
+
"""
|
|
257
|
+
MATCH (i:ImageAsset {id: $iid})
|
|
258
|
+
RETURN i.id AS id, i.file_name AS file_name, i.mime_type AS mime_type,
|
|
259
|
+
i.alt_text AS alt_text, i.embed_route AS embed_route, i.data AS data
|
|
260
|
+
""",
|
|
261
|
+
{"iid": asset_id},
|
|
262
|
+
)
|
|
263
|
+
rows = result.rows_as_dict().get_all()
|
|
264
|
+
return rows[0] if rows else None
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def search_images_with_text(
|
|
268
|
+
self,
|
|
269
|
+
text_query: str,
|
|
270
|
+
use_text_model: bool = True,
|
|
271
|
+
limit: int = 10,
|
|
272
|
+
) -> List[Dict[str, Any]]:
|
|
273
|
+
"""Find image assets from a text query via the unified vector index.
|
|
274
|
+
|
|
275
|
+
``use_text_model=True`` (default) encodes the query with the lightweight
|
|
276
|
+
text model — no omni load required, since both models share the vector
|
|
277
|
+
space. Set it to ``False`` to route the query through the omni text side.
|
|
278
|
+
"""
|
|
279
|
+
if use_text_model:
|
|
280
|
+
query_vec = self.embed_engine._encode(text_query, task="Query")
|
|
281
|
+
else:
|
|
282
|
+
query_vec = self.embed_engine._encode_omni_text(text_query, task="Query")
|
|
283
|
+
|
|
284
|
+
if self._db is None:
|
|
285
|
+
result = self.conn.execute(
|
|
286
|
+
"CALL QUERY_VECTOR_INDEX('ImageAsset', 'image_omni_idx', $vec, $k) "
|
|
287
|
+
"RETURN node, distance",
|
|
288
|
+
{"vec": query_vec, "k": limit},
|
|
289
|
+
)
|
|
290
|
+
rows = result.rows_as_dict().get_all()
|
|
291
|
+
else:
|
|
292
|
+
import ladybug as lb
|
|
293
|
+
|
|
294
|
+
c = lb.Connection(self._db)
|
|
295
|
+
try:
|
|
296
|
+
c.execute("LOAD EXTENSION VECTOR")
|
|
297
|
+
result = c.execute(
|
|
298
|
+
"CALL QUERY_VECTOR_INDEX('ImageAsset', 'image_omni_idx', $vec, $k) "
|
|
299
|
+
"RETURN node, distance",
|
|
300
|
+
{"vec": query_vec, "k": limit},
|
|
301
|
+
)
|
|
302
|
+
rows = result.rows_as_dict().get_all()
|
|
303
|
+
finally:
|
|
304
|
+
c.close()
|
|
305
|
+
out: List[Dict[str, Any]] = []
|
|
306
|
+
for row in rows:
|
|
307
|
+
node = row.get("node", {})
|
|
308
|
+
if not isinstance(node, dict):
|
|
309
|
+
continue
|
|
310
|
+
out.append({
|
|
311
|
+
"id": node.get("id"),
|
|
312
|
+
"file_name": node.get("file_name"),
|
|
313
|
+
"alt_text": node.get("alt_text"),
|
|
314
|
+
"embed_route": node.get("embed_route"),
|
|
315
|
+
"distance": row.get("distance"),
|
|
316
|
+
"relevance_score": 1 - row.get("distance", 0),
|
|
317
|
+
})
|
|
318
|
+
return out
|
|
319
|
+
|