okfgraph 0.2.7__tar.gz → 0.2.11__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. {okfgraph-0.2.7 → okfgraph-0.2.11}/PKG-INFO +5 -1
  2. {okfgraph-0.2.7 → okfgraph-0.2.11}/README.md +4 -0
  3. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/__init__.py +2 -1
  4. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/embedding.py +232 -13
  5. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/router.py +77 -15
  6. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph.egg-info/PKG-INFO +5 -1
  7. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph.egg-info/SOURCES.txt +3 -0
  8. {okfgraph-0.2.7 → okfgraph-0.2.11}/pyproject.toml +1 -1
  9. okfgraph-0.2.11/tests/test_explicit_files.py +147 -0
  10. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_gpu_integration.py +5 -1
  11. okfgraph-0.2.11/tests/test_lazy_encoder.py +202 -0
  12. okfgraph-0.2.11/tests/test_ort.py +196 -0
  13. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_rust_backend.py +32 -1
  14. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_rust_e2e.py +4 -1
  15. {okfgraph-0.2.7 → okfgraph-0.2.11}/LICENSE +0 -0
  16. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/__init__.py +0 -0
  17. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/cli.py +0 -0
  18. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/converters.py +0 -0
  19. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/delta.py +0 -0
  20. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/diff.py +0 -0
  21. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/doctor.py +0 -0
  22. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/export.py +0 -0
  23. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/image_assets.py +0 -0
  24. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/import_.py +0 -0
  25. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/ingest.py +0 -0
  26. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/links.py +0 -0
  27. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/lint.py +0 -0
  28. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/purge.py +0 -0
  29. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/ranking.py +0 -0
  30. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/schema.py +0 -0
  31. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/search.py +0 -0
  32. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/config.py +0 -0
  33. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/images.py +0 -0
  34. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/mcp_server.py +0 -0
  35. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/models.py +0 -0
  36. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/security.py +0 -0
  37. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/tools.py +0 -0
  38. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph.egg-info/dependency_links.txt +0 -0
  39. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph.egg-info/entry_points.txt +0 -0
  40. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph.egg-info/requires.txt +0 -0
  41. {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph.egg-info/top_level.txt +0 -0
  42. {okfgraph-0.2.7 → okfgraph-0.2.11}/setup.cfg +0 -0
  43. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_chunk_search.py +0 -0
  44. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_chunking.py +0 -0
  45. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_cli.py +0 -0
  46. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_config.py +0 -0
  47. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_converter.py +0 -0
  48. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_delta.py +0 -0
  49. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_diff.py +0 -0
  50. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_directory_hash.py +0 -0
  51. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_doctor.py +0 -0
  52. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_export_compliance.py +0 -0
  53. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_graph_enrichment.py +0 -0
  54. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_images.py +0 -0
  55. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_ingest.py +0 -0
  56. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_ingest_tool.py +0 -0
  57. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_integration.py +0 -0
  58. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_lint.py +0 -0
  59. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_logging.py +0 -0
  60. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_mcp_server.py +0 -0
  61. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_models.py +0 -0
  62. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_obsidian.py +0 -0
  63. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_okf_ingest_tool.py +0 -0
  64. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_packaging.py +0 -0
  65. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_parity.py +0 -0
  66. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_pdf_e2e.py +0 -0
  67. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_ppr_search.py +0 -0
  68. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_ranking.py +0 -0
  69. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_reconstruction.py +0 -0
  70. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_reserved.py +0 -0
  71. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_roundup_cli.py +0 -0
  72. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_router.py +0 -0
  73. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_router_misc.py +0 -0
  74. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_sanitize.py +0 -0
  75. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_search_browser.py +0 -0
  76. {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_security.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: okfgraph
3
- Version: 0.2.7
3
+ Version: 0.2.11
4
4
  Summary: Ladybug-backed OKF knowledge graph with ONNX + Jina v5 embeddings
5
5
  License-Expression: Apache-2.0 OR MIT
6
6
  Requires-Python: >=3.11
@@ -259,6 +259,10 @@ Key design decisions:
259
259
 
260
260
  - **Rust-only embeddings, fail-fast** — no Python fallback; a mid-run stack
261
261
  switch would silently mix vector spaces in one index.
262
+ - **Lazy session init** — router construction never downloads the model or
263
+ builds the ONNX session; first encode opens once (thread-safe), token
264
+ counting uses a tokenizer-only handle, so PPR search, budgeted reads,
265
+ diff, and doctor stay cold.
262
266
  - **Last-token pooling** (not mean) — required by Jina v5; mean pooling
263
267
  breaks alignment with omni image embeddings.
264
268
  - **Single pinned ORT** (`onnxruntime==1.29.0`, `ORT_DYLIB_PATH`-overridable)
@@ -232,6 +232,10 @@ Key design decisions:
232
232
 
233
233
  - **Rust-only embeddings, fail-fast** — no Python fallback; a mid-run stack
234
234
  switch would silently mix vector spaces in one index.
235
+ - **Lazy session init** — router construction never downloads the model or
236
+ builds the ONNX session; first encode opens once (thread-safe), token
237
+ counting uses a tokenizer-only handle, so PPR search, budgeted reads,
238
+ diff, and doctor stay cold.
235
239
  - **Last-token pooling** (not mean) — required by Jina v5; mean pooling
236
240
  breaks alignment with omni image embeddings.
237
241
  - **Single pinned ORT** (`onnxruntime==1.29.0`, `ORT_DYLIB_PATH`-overridable)
@@ -13,7 +13,7 @@ stubs (``...``) until their respective phase moves the implementation over.
13
13
  from okfgraph.components.schema import SchemaManager
14
14
  from okfgraph.components.delta import DeltaDetector
15
15
  from okfgraph.components.purge import PurgeManager
16
- from okfgraph.components.embedding import EmbeddingEngine
16
+ from okfgraph.components.embedding import EmbeddingEngine, LazyRustEncoder
17
17
  from okfgraph.components.image_assets import ImageAssetManager
18
18
  from okfgraph.components.search import SearchEngine
19
19
  from okfgraph.components.import_ import ImportManager, parse_source_file
@@ -37,6 +37,7 @@ __all__ = [
37
37
  "DeltaDetector",
38
38
  "PurgeManager",
39
39
  "EmbeddingEngine",
40
+ "LazyRustEncoder",
40
41
  "ImageAssetManager",
41
42
  "SearchEngine",
42
43
  "ImportManager",
@@ -6,33 +6,250 @@ here. Public callers reach these via router.<method> (component bridge).
6
6
  """
7
7
  import logging
8
8
  import math
9
+ import threading
9
10
  from pathlib import Path
10
11
 
11
12
  import mordant
12
13
  from typing import Any, Dict, List, Optional
13
14
  logger = logging.getLogger(__name__)
14
15
 
15
- def resolve_ort_dylib() -> Optional[str]:
16
- """Point ``ORT_DYLIB_PATH`` at the pip-installed ORT build when unset.
16
+ _ORT_MODULE_NAMES = ("onnxruntime", "onnxruntime-gpu")
17
+
18
+
19
+ def _candidate_ort_library_names(os_name=None, sys_platform=None):
20
+ """Return the ORT library filenames for an OS/platform pair.
21
+
22
+ Takes explicit arguments so unit tests can cover Windows/macOS/Linux
23
+ from any host. Linux uses a versioned ``libonnxruntime.so.*`` glob
24
+ (handled by :func:`_find_runtime_in_package`); this helper returns the
25
+ unversioned fallback name for that platform.
26
+ """
27
+ import os
28
+ import sys
29
+ if os_name is None:
30
+ os_name = os.name
31
+ if sys_platform is None:
32
+ sys_platform = sys.platform
33
+ if os_name == "nt":
34
+ return ("onnxruntime.dll",)
35
+ if sys_platform == "darwin":
36
+ return ("libonnxruntime.dylib",)
37
+ return ("libonnxruntime.so",)
38
+
39
+
40
+ def _find_runtime_in_package(package_dir, os_name=None, sys_platform=None):
41
+ """Find an ORT shared library inside a pip-installed package directory.
42
+
43
+ Returns a :class:`Path` or ``None``. Prefers versioned Linux libraries
44
+ (``libonnxruntime.so.*``) over the unversioned ``libonnxruntime.so``.
45
+ """
46
+ import os
47
+ import sys
48
+ if os_name is None:
49
+ os_name = os.name
50
+ if sys_platform is None:
51
+ sys_platform = sys.platform
52
+ capi_dir = Path(package_dir) / "capi"
53
+ if not capi_dir.is_dir():
54
+ return None
55
+ if os_name == "nt" or sys_platform == "darwin":
56
+ candidate = capi_dir / _candidate_ort_library_names(os_name, sys_platform)[0]
57
+ return candidate if candidate.is_file() else None
58
+ versioned = sorted(capi_dir.glob("libonnxruntime.so.*"))
59
+ if versioned:
60
+ return versioned[-1]
61
+ candidate = capi_dir / "libonnxruntime.so"
62
+ return candidate if candidate.is_file() else None
63
+
64
+
65
+ def _configure_windows_ort_dll_directory(runtime_path) -> bool:
66
+ """Add the ORT ``capi`` directory to Windows DLL resolution.
67
+
68
+ Best-effort only: returns ``False`` (never raises) when unavailable.
69
+ """
70
+ import os
71
+ if os.name != "nt":
72
+ return False
73
+ add_dll_directory = getattr(os, "add_dll_directory", None)
74
+ if not callable(add_dll_directory):
75
+ return False
76
+ try:
77
+ add_dll_directory(str(Path(runtime_path).parent))
78
+ return True
79
+ except OSError as exc:
80
+ logger.debug("ORT DLL directory configuration failed: %s", exc)
81
+ return False
82
+
83
+
84
+ def _warm_ort_gpu_dlls(module) -> bool:
85
+ """Preload NVIDIA DLLs when the installed ORT package exposes CUDA.
86
+
87
+ This mirrors bobine's GPU bootstrap: only attempt the preload when the
88
+ package reports a CUDA execution provider, and never let it fail
89
+ runtime discovery.
90
+ """
91
+ preload = getattr(module, "preload_dlls", None)
92
+ if not callable(preload):
93
+ return False
94
+ try:
95
+ providers = list(module.get_available_providers())
96
+ except Exception as exc:
97
+ logger.debug("ORT provider query failed: %s", exc)
98
+ return False
99
+ if "CUDAExecutionProvider" not in providers:
100
+ return False
101
+ try:
102
+ preload()
103
+ return True
104
+ except Exception as exc:
105
+ logger.debug("ORT GPU DLL preload failed: %s", exc)
106
+ return False
107
+
108
+
109
+ def resolve_ort_dylib(*, warm_gpu: bool = True, os_name=None, sys_platform=None) -> Optional[str]:
110
+ """Point ``ORT_DYLIB_PATH`` at a pip-installed ORT build when unset.
17
111
 
18
112
  Both bobine and okf-embed load ONNX Runtime dynamically; sharing one
19
113
  binary avoids version/CUDA drift between the two runtimes. Explicit
20
114
  user configuration always wins — this only fills the gap.
115
+
116
+ Resolution order:
117
+
118
+ 1. Existing ``ORT_DYLIB_PATH``.
119
+ 2. Pip-installed ``onnxruntime`` or ``onnxruntime-gpu`` package.
120
+ 3. OS loader path (represented by returning ``None``).
121
+
122
+ Missing runtimes never raise here; session creation or encoding is the
123
+ fail-fast boundary. ``os_name``/``sys_platform`` are explicit so unit
124
+ tests can cover every platform from any host (patching ``os.name``
125
+ would break ``pathlib`` dispatch on Windows).
21
126
  """
22
127
  import os
23
- if os.environ.get("ORT_DYLIB_PATH"):
24
- return os.environ["ORT_DYLIB_PATH"]
25
- try:
26
- import onnxruntime
27
- dll = Path(str(onnxruntime.__file__)).parent / "capi" / "onnxruntime.dll"
28
- if dll.exists():
29
- os.environ["ORT_DYLIB_PATH"] = str(dll)
30
- return str(dll)
31
- except ImportError:
32
- pass
128
+ explicit = os.environ.get("ORT_DYLIB_PATH")
129
+ if explicit:
130
+ return explicit
131
+ for module_name in _ORT_MODULE_NAMES:
132
+ try:
133
+ module = __import__(module_name)
134
+ except ImportError:
135
+ continue
136
+ try:
137
+ package_dir = Path(str(module.__file__)).parent
138
+ except Exception:
139
+ continue
140
+ found = _find_runtime_in_package(package_dir, os_name, sys_platform)
141
+ if found is None:
142
+ continue
143
+ os.environ["ORT_DYLIB_PATH"] = str(found)
144
+ _configure_windows_ort_dll_directory(found)
145
+ if warm_gpu:
146
+ _warm_ort_gpu_dlls(module)
147
+ logger.debug("resolved ORT runtime: %s", found)
148
+ return str(found)
33
149
  return None
34
150
 
35
151
 
152
+ class LazyRustEncoder:
153
+ """Defers the ONNX session open until the first real encode.
154
+
155
+ Router construction stays cheap: the ``okf-embed`` wheel import is still
156
+ validated eagerly (fail fast on a missing install), but ``JinaV5.open``
157
+ — model download + session build — waits for the first ``encode`` /
158
+ ``encode_batch`` / ``used_cuda`` access. Token counting uses the
159
+ separate lightweight ``JinaTokenizer`` handle, so budgeted reads and the
160
+ context-window guard stay cold too.
161
+
162
+ A failed session open is cached and re-raised: configuration errors stay
163
+ fail-fast (once, at first encode) instead of retrying network/model
164
+ acquisition on every call. Thread-safe: concurrent first encodes open
165
+ exactly one session.
166
+ """
167
+
168
+ def __init__(self, *, model_id, truncate_dim, device,
169
+ session_factory, tokenizer_factory, on_open=None):
170
+ self._model_id = model_id
171
+ self._truncate_dim = truncate_dim
172
+ self._device = device
173
+ self._session_factory = session_factory
174
+ self._tokenizer_factory = tokenizer_factory
175
+ self._on_open = on_open
176
+ self._lock = threading.Lock()
177
+ self._encoder = None
178
+ self._encoder_error = None
179
+ self._open_reported = False
180
+ self._tokenizer = None
181
+
182
+ @property
183
+ def is_loaded(self) -> bool:
184
+ """True once the ONNX session has been opened."""
185
+ return self._encoder is not None
186
+
187
+ @property
188
+ def model_id(self) -> str:
189
+ return self._model_id
190
+
191
+ @property
192
+ def dim(self) -> int:
193
+ return self._truncate_dim
194
+
195
+ @property
196
+ def used_cuda(self) -> bool:
197
+ """Effective device — opens the session on first access."""
198
+ return bool(self._get_encoder().used_cuda)
199
+
200
+ def encode(self, text: str, task: str = "Document"):
201
+ return self._get_encoder().encode(text, task=task)
202
+
203
+ def encode_batch(self, texts, task: str = "Document"):
204
+ return self._get_encoder().encode_batch(texts, task=task)
205
+
206
+ def count_tokens(self, text: str) -> int:
207
+ """Exact count via the tokenizer-only handle (never opens the session)."""
208
+ return int(self._get_tokenizer().count_tokens(text))
209
+
210
+ def _get_encoder(self):
211
+ encoder = self._encoder
212
+ if encoder is not None:
213
+ return encoder
214
+ with self._lock:
215
+ if self._encoder is not None:
216
+ return self._encoder
217
+ if self._encoder_error is not None:
218
+ raise self._encoder_error
219
+ try:
220
+ encoder = self._session_factory()
221
+ except Exception as exc:
222
+ self._encoder_error = exc
223
+ raise
224
+ self._encoder = encoder
225
+ if self._on_open is not None and not self._open_reported:
226
+ self._open_reported = True
227
+ self._on_open(encoder)
228
+ return encoder
229
+
230
+ def _get_tokenizer(self):
231
+ tokenizer = self._tokenizer
232
+ if tokenizer is not None:
233
+ return tokenizer
234
+ with self._lock:
235
+ if self._tokenizer is not None:
236
+ return self._tokenizer
237
+ try:
238
+ tokenizer = self._tokenizer_factory()
239
+ except Exception as exc:
240
+ logger.debug("tokenizer-only open failed: %s", exc)
241
+ raise
242
+ self._tokenizer = tokenizer
243
+ return tokenizer
244
+
245
+ def __repr__(self) -> str:
246
+ state = "loaded" if self._encoder is not None else "cold"
247
+ return (
248
+ f"LazyRustEncoder({self._model_id}, "
249
+ f"dim={self._truncate_dim}, {state})"
250
+ )
251
+
252
+
36
253
  class EmbeddingEngine:
37
254
  """Owns the embedding model and chunking logic.
38
255
 
@@ -43,7 +260,8 @@ class EmbeddingEngine:
43
260
 
44
261
  def __init__(self, rust_encoder, embedding_dim, device,
45
262
  cache_dir, model_id, omni_model_id, omni,
46
- chunk_size, chunk_overlap, enable_chunking, conn):
263
+ chunk_size, chunk_overlap, enable_chunking, conn,
264
+ ort_dylib=None):
47
265
  self.encoder = rust_encoder
48
266
  self.embedding_dim = embedding_dim
49
267
  self.device = device
@@ -55,6 +273,7 @@ class EmbeddingEngine:
55
273
  self.chunk_overlap = chunk_overlap
56
274
  self.enable_chunking = enable_chunking
57
275
  self.conn = conn
276
+ self.ort_dylib = ort_dylib
58
277
 
59
278
  def _encode(self, text: str, task: str = "Document") -> List[float]:
60
279
  """Encode text with the Rust Jina v5 encoder.
@@ -106,6 +106,8 @@ class OKFRouter:
106
106
  omni_model_id: str = "jinaai/jina-embeddings-v5-omni-small-retrieval",
107
107
  embedding_dim: int = 512,
108
108
  cache_dir: Optional[str] = None,
109
+ model_path: Optional[str] = None,
110
+ tokenizer_path: Optional[str] = None,
109
111
  device: str = "cpu",
110
112
  allow_remote_images: bool = False,
111
113
  allowed_image_domains: Optional[List[str]] = None,
@@ -124,6 +126,10 @@ class OKFRouter:
124
126
  omni_model_id: HuggingFace ID of the Jina v5 omni model (images).
125
127
  embedding_dim: Truncated Matryoshka dimension (<= 1024).
126
128
  cache_dir: Model cache directory (defaults per-platform).
129
+ model_path: Explicit local ONNX file (with `tokenizer_path`);
130
+ skips every download — air-gapped / reproducible installs.
131
+ An external-data sidecar must sit next to this file.
132
+ tokenizer_path: Explicit local `tokenizer.json` (with `model_path`).
127
133
  device: "cpu" (CUDA is opportunistic inside the Rust loader).
128
134
  allow_remote_images: Whether http(s) image URLs may be fetched.
129
135
  allowed_image_domains: Domain allowlist for remote images.
@@ -186,7 +192,10 @@ class OKFRouter:
186
192
  # Text embeddings: Rust okf_embed wheel (Jina v5 via ORT). No Python
187
193
  # fallback — a mid-run stack switch would silently mix vector spaces
188
194
  # in one index.
189
- from okfgraph.components.embedding import resolve_ort_dylib
195
+ from okfgraph.components.embedding import LazyRustEncoder, resolve_ort_dylib
196
+ # Resolve first: the native module loads ORT dynamically, so the
197
+ # shared runtime choice must be fixed before importing it.
198
+ self.ort_dylib = resolve_ort_dylib()
190
199
  try:
191
200
  import okf_embed
192
201
  except ImportError:
@@ -194,24 +203,77 @@ class OKFRouter:
194
203
  "the okf-embed wheel is required for text embeddings: "
195
204
  "build rust/okf-embed (maturin build --release) and install it"
196
205
  ) from None
197
- resolve_ort_dylib()
198
206
  # The Rust crate knows auto/cpu/cuda; map torch-style aliases.
207
+ # Validate eagerly so a bad device still fails at construction —
208
+ # the session itself opens lazily on first encode (see below).
199
209
  rust_device = {"mps": "auto"}.get(device, device)
200
- self.encoder = okf_embed.JinaV5.open(
201
- model_id,
210
+ if rust_device not in ("auto", "cpu", "cuda"):
211
+ raise ValueError(
212
+ f"device must be 'auto', 'cpu' or 'cuda', got '{device}'"
213
+ )
214
+ if (model_path is None) != (tokenizer_path is None):
215
+ raise ValueError(
216
+ "model_path and tokenizer_path must be given together "
217
+ "(explicit local files skip all downloads)"
218
+ )
219
+ explicit_files = model_path is not None
220
+ if explicit_files:
221
+ for label, path in (
222
+ ("model_path", model_path),
223
+ ("tokenizer_path", tokenizer_path),
224
+ ):
225
+ if not Path(path).is_file():
226
+ raise FileNotFoundError(
227
+ f"{label} not found: {path} "
228
+ "(explicit paths skip all downloads)"
229
+ )
230
+ logger.debug("using explicit model files: %s", model_path)
231
+
232
+ def _report_encoder_open(encoder) -> None:
233
+ logger.info(
234
+ "text embeddings: %s dim=%d cuda=%s",
235
+ model_id, embedding_dim, encoder.used_cuda,
236
+ )
237
+ if device == "cuda" and not encoder.used_cuda:
238
+ logger.warning(
239
+ "CUDA requested but the loaded ONNX Runtime has no CUDA execution "
240
+ "provider — running on CPU. Install onnxruntime-gpu for acceleration."
241
+ )
242
+
243
+ # Text embeddings: Rust okf_embed wheel (Jina v5 via ORT). No Python
244
+ # fallback — a mid-run stack switch would silently mix vector spaces
245
+ # in one index. The wheel import above stays fail-fast; the session
246
+ # open is lazy so model-free commands (PPR search, budgeted reads,
247
+ # diff, doctor) never pay model-download/session-build costs.
248
+ if explicit_files:
249
+ session_factory = lambda: okf_embed.JinaV5.open_files(
250
+ str(model_path),
251
+ str(tokenizer_path),
252
+ truncate_dim=embedding_dim,
253
+ device=rust_device,
254
+ )
255
+ tokenizer_factory = lambda: okf_embed.JinaTokenizer.open_files(
256
+ str(tokenizer_path),
257
+ )
258
+ else:
259
+ session_factory = lambda: okf_embed.JinaV5.open(
260
+ model_id,
261
+ truncate_dim=embedding_dim,
262
+ device=rust_device,
263
+ cache_dir=cache_dir,
264
+ )
265
+ tokenizer_factory = lambda: okf_embed.JinaTokenizer.open(
266
+ model_id,
267
+ cache_dir=cache_dir,
268
+ )
269
+ self.encoder = LazyRustEncoder(
270
+ model_id=model_id,
202
271
  truncate_dim=embedding_dim,
203
272
  device=rust_device,
204
- cache_dir=cache_dir,
273
+ session_factory=session_factory,
274
+ tokenizer_factory=tokenizer_factory,
275
+ on_open=_report_encoder_open,
205
276
  )
206
- logger.info(
207
- "text embeddings: %s dim=%d cuda=%s",
208
- model_id, embedding_dim, self.encoder.used_cuda,
209
- )
210
- if device == "cuda" and not self.encoder.used_cuda:
211
- logger.warning(
212
- "CUDA requested but the loaded ONNX Runtime has no CUDA execution "
213
- "provider — running on CPU. Install onnxruntime-gpu for acceleration."
214
- )
215
277
 
216
278
  # ── Component wiring (Phase 1-3 refactor) ───────────────────
217
279
  # The facade owns the resources (conn, encoder, lock) and injects
@@ -232,7 +294,7 @@ class OKFRouter:
232
294
  self.encoder, self.embedding_dim,
233
295
  self.device, self.cache_dir, self.model_id, self.omni_model_id,
234
296
  self._omni, self.chunk_size, self.chunk_overlap, self.enable_chunking,
235
- self.conn,
297
+ self.conn, self.ort_dylib,
236
298
  )
237
299
  self.image_mgr = ImageAssetManager(
238
300
  self.conn, self.embed_engine, self.schema_mgr,
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: okfgraph
3
- Version: 0.2.7
3
+ Version: 0.2.11
4
4
  Summary: Ladybug-backed OKF knowledge graph with ONNX + Jina v5 embeddings
5
5
  License-Expression: Apache-2.0 OR MIT
6
6
  Requires-Python: >=3.11
@@ -259,6 +259,10 @@ Key design decisions:
259
259
 
260
260
  - **Rust-only embeddings, fail-fast** — no Python fallback; a mid-run stack
261
261
  switch would silently mix vector spaces in one index.
262
+ - **Lazy session init** — router construction never downloads the model or
263
+ builds the ONNX session; first encode opens once (thread-safe), token
264
+ counting uses a tokenizer-only handle, so PPR search, budgeted reads,
265
+ diff, and doctor stay cold.
262
266
  - **Last-token pooling** (not mean) — required by Jina v5; mean pooling
263
267
  breaks alignment with omni image embeddings.
264
268
  - **Single pinned ORT** (`onnxruntime==1.29.0`, `ORT_DYLIB_PATH`-overridable)
@@ -41,6 +41,7 @@ tests/test_delta.py
41
41
  tests/test_diff.py
42
42
  tests/test_directory_hash.py
43
43
  tests/test_doctor.py
44
+ tests/test_explicit_files.py
44
45
  tests/test_export_compliance.py
45
46
  tests/test_gpu_integration.py
46
47
  tests/test_graph_enrichment.py
@@ -48,12 +49,14 @@ tests/test_images.py
48
49
  tests/test_ingest.py
49
50
  tests/test_ingest_tool.py
50
51
  tests/test_integration.py
52
+ tests/test_lazy_encoder.py
51
53
  tests/test_lint.py
52
54
  tests/test_logging.py
53
55
  tests/test_mcp_server.py
54
56
  tests/test_models.py
55
57
  tests/test_obsidian.py
56
58
  tests/test_okf_ingest_tool.py
59
+ tests/test_ort.py
57
60
  tests/test_packaging.py
58
61
  tests/test_parity.py
59
62
  tests/test_pdf_e2e.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "okfgraph"
7
- version = "0.2.7"
7
+ version = "0.2.11"
8
8
  description = "Ladybug-backed OKF knowledge graph with ONNX + Jina v5 embeddings"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.11"
@@ -0,0 +1,147 @@
1
+ """Explicit local model-file tests — no downloads, no ORT session.
2
+
3
+ Validation (pairing, existence) fires at router construction. Routing to
4
+ `open_files` vs `open` is proven with a stubbed `okf_embed` module; numeric
5
+ equivalence of the two acquisition paths is pinned in test_rust_backend.py.
6
+ """
7
+
8
+ import sys
9
+ import types
10
+
11
+ import pytest
12
+
13
+ from okfgraph.router import OKFRouter
14
+
15
+
16
+ def _stub_embed(monkeypatch, calls):
17
+ class _Session:
18
+ def __init__(self, via):
19
+ self.via = via
20
+ self.used_cuda = False
21
+
22
+ def encode(self, text, task="Document"):
23
+ calls["encode"] += 1
24
+ return [float(len(text))]
25
+
26
+ def encode_batch(self, texts, task="Document"):
27
+ return [self.encode(t, task=task) for t in texts]
28
+
29
+ class _Tokenizer:
30
+ def __init__(self, via):
31
+ self.via = via
32
+
33
+ def count_tokens(self, text):
34
+ return len(text.split())
35
+
36
+ stub = types.SimpleNamespace(
37
+ JinaV5=types.SimpleNamespace(
38
+ open=lambda *a, **k: (_ for _ in ()).throw(
39
+ AssertionError(f"unexpected open: {a} {k}")
40
+ ),
41
+ open_files=lambda onnx, tok, **k: (
42
+ calls["open_files"].append((onnx, tok, k)),
43
+ _Session("files"),
44
+ )[1],
45
+ ),
46
+ JinaTokenizer=types.SimpleNamespace(
47
+ open=lambda *a, **k: (_ for _ in ()).throw(
48
+ AssertionError(f"unexpected tokenizer open: {a} {k}")
49
+ ),
50
+ open_files=lambda tok: (
51
+ calls["tok_files"].append(tok),
52
+ _Tokenizer("files"),
53
+ )[1],
54
+ ),
55
+ MAX_LENGTH=8192,
56
+ )
57
+ monkeypatch.setitem(sys.modules, "okf_embed", stub)
58
+ return stub
59
+
60
+
61
+ def _router(tmp_path, **kwargs):
62
+ return OKFRouter(
63
+ db_path=str(tmp_path / "explicit.db"),
64
+ bundle_root=str(tmp_path),
65
+ device="cpu",
66
+ **kwargs,
67
+ )
68
+
69
+
70
+ def test_paths_must_be_given_together(tmp_path):
71
+ with pytest.raises(ValueError, match="must be given together"):
72
+ _router(tmp_path, model_path="model.onnx")
73
+ with pytest.raises(ValueError, match="must be given together"):
74
+ _router(tmp_path, tokenizer_path="tokenizer.json")
75
+
76
+
77
+ def test_missing_files_fail_fast(tmp_path):
78
+ with pytest.raises(FileNotFoundError, match="model_path not found"):
79
+ _router(
80
+ tmp_path,
81
+ model_path=str(tmp_path / "nope.onnx"),
82
+ tokenizer_path=str(tmp_path / "tok.json"),
83
+ )
84
+
85
+
86
+ def test_encode_routes_to_open_files(tmp_path, monkeypatch):
87
+ calls = {"open_files": [], "tok_files": [], "encode": 0}
88
+ _stub_embed(monkeypatch, calls)
89
+ onnx = tmp_path / "model.onnx"
90
+ tok = tmp_path / "tokenizer.json"
91
+ onnx.write_bytes(b"fake")
92
+ tok.write_text("{}", encoding="utf-8")
93
+
94
+ router = _router(tmp_path, model_path=str(onnx), tokenizer_path=str(tok))
95
+ try:
96
+ assert router.encoder.is_loaded is False
97
+ assert router.embed_engine._encode("hi") == [2.0]
98
+ assert router.embed_engine.count_tokens("one two") == 2
99
+ finally:
100
+ router.close()
101
+
102
+ assert calls["open_files"] == [
103
+ (str(onnx), str(tok), {"truncate_dim": 512, "device": "cpu"})
104
+ ]
105
+ assert calls["tok_files"] == [str(tok)]
106
+ assert calls["encode"] == 1
107
+
108
+
109
+ def test_default_path_still_uses_open(tmp_path, monkeypatch):
110
+ import okf_embed as real_embed
111
+
112
+ opened = {}
113
+
114
+ class _Session:
115
+ used_cuda = False
116
+
117
+ def encode(self, text, task="Document"):
118
+ return [1.0]
119
+
120
+ def encode_batch(self, texts, task="Document"):
121
+ return [[1.0] for _ in texts]
122
+
123
+ stub = types.SimpleNamespace(
124
+ JinaV5=types.SimpleNamespace(
125
+ open=lambda *a, **k: (opened.setdefault("open", (a, k)), _Session())[1],
126
+ open_files=lambda *a, **k: (_ for _ in ()).throw(
127
+ AssertionError("open_files must not be used by default")
128
+ ),
129
+ ),
130
+ JinaTokenizer=types.SimpleNamespace(
131
+ open=lambda *a, **k: types.SimpleNamespace(
132
+ count_tokens=lambda t: 1
133
+ ),
134
+ open_files=lambda *a, **k: (_ for _ in ()).throw(
135
+ AssertionError("tokenizer open_files must not be used by default")
136
+ ),
137
+ ),
138
+ MAX_LENGTH=real_embed.MAX_LENGTH,
139
+ )
140
+ monkeypatch.setitem(sys.modules, "okf_embed", stub)
141
+
142
+ router = _router(tmp_path)
143
+ try:
144
+ assert router.embed_engine._encode("hi") == [1.0]
145
+ finally:
146
+ router.close()
147
+ assert opened["open"][1]["device"] == "cpu"
@@ -165,13 +165,17 @@ class TestGPUInitialization:
165
165
 
166
166
  @pytest.mark.skipif(_has_onnxruntime_gpu(), reason="onnxruntime-gpu is installed")
167
167
  def test_no_gpu_runtime_warns_on_cuda_request(self, tmp_dir, capsys):
168
- """When the loaded ORT has no CUDA EP and device='cuda', warn on stderr."""
168
+ """When the loaded ORT has no CUDA EP and device='cuda', the first
169
+ encode warns on stderr and falls back to CPU (the session now opens
170
+ lazily, so construction alone stays silent)."""
169
171
  r = OKFRouter(
170
172
  db_path=str(Path(tmp_dir) / "test_no_gpu.db"),
171
173
  bundle_root=tmp_dir,
172
174
  embedding_dim=512,
173
175
  device="cuda",
174
176
  )
177
+ assert r.encoder.is_loaded is False
178
+ _ = r.embed_engine._encode("warmup", task="Document")
175
179
  assert r.encoder.used_cuda is False
176
180
  assert "CUDA" in capsys.readouterr().err
177
181
  r.close()