okfgraph 0.2.7__tar.gz → 0.2.11__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {okfgraph-0.2.7 → okfgraph-0.2.11}/PKG-INFO +5 -1
- {okfgraph-0.2.7 → okfgraph-0.2.11}/README.md +4 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/__init__.py +2 -1
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/embedding.py +232 -13
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/router.py +77 -15
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph.egg-info/PKG-INFO +5 -1
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph.egg-info/SOURCES.txt +3 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/pyproject.toml +1 -1
- okfgraph-0.2.11/tests/test_explicit_files.py +147 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_gpu_integration.py +5 -1
- okfgraph-0.2.11/tests/test_lazy_encoder.py +202 -0
- okfgraph-0.2.11/tests/test_ort.py +196 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_rust_backend.py +32 -1
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_rust_e2e.py +4 -1
- {okfgraph-0.2.7 → okfgraph-0.2.11}/LICENSE +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/__init__.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/cli.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/converters.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/delta.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/diff.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/doctor.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/export.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/image_assets.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/import_.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/ingest.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/links.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/lint.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/purge.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/ranking.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/schema.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/components/search.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/config.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/images.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/mcp_server.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/models.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/security.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph/tools.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph.egg-info/dependency_links.txt +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph.egg-info/entry_points.txt +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph.egg-info/requires.txt +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/okfgraph.egg-info/top_level.txt +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/setup.cfg +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_chunk_search.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_chunking.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_cli.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_config.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_converter.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_delta.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_diff.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_directory_hash.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_doctor.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_export_compliance.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_graph_enrichment.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_images.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_ingest.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_ingest_tool.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_integration.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_lint.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_logging.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_mcp_server.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_models.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_obsidian.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_okf_ingest_tool.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_packaging.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_parity.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_pdf_e2e.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_ppr_search.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_ranking.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_reconstruction.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_reserved.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_roundup_cli.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_router.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_router_misc.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_sanitize.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_search_browser.py +0 -0
- {okfgraph-0.2.7 → okfgraph-0.2.11}/tests/test_security.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: okfgraph
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.11
|
|
4
4
|
Summary: Ladybug-backed OKF knowledge graph with ONNX + Jina v5 embeddings
|
|
5
5
|
License-Expression: Apache-2.0 OR MIT
|
|
6
6
|
Requires-Python: >=3.11
|
|
@@ -259,6 +259,10 @@ Key design decisions:
|
|
|
259
259
|
|
|
260
260
|
- **Rust-only embeddings, fail-fast** — no Python fallback; a mid-run stack
|
|
261
261
|
switch would silently mix vector spaces in one index.
|
|
262
|
+
- **Lazy session init** — router construction never downloads the model or
|
|
263
|
+
builds the ONNX session; first encode opens once (thread-safe), token
|
|
264
|
+
counting uses a tokenizer-only handle, so PPR search, budgeted reads,
|
|
265
|
+
diff, and doctor stay cold.
|
|
262
266
|
- **Last-token pooling** (not mean) — required by Jina v5; mean pooling
|
|
263
267
|
breaks alignment with omni image embeddings.
|
|
264
268
|
- **Single pinned ORT** (`onnxruntime==1.29.0`, `ORT_DYLIB_PATH`-overridable)
|
|
@@ -232,6 +232,10 @@ Key design decisions:
|
|
|
232
232
|
|
|
233
233
|
- **Rust-only embeddings, fail-fast** — no Python fallback; a mid-run stack
|
|
234
234
|
switch would silently mix vector spaces in one index.
|
|
235
|
+
- **Lazy session init** — router construction never downloads the model or
|
|
236
|
+
builds the ONNX session; first encode opens once (thread-safe), token
|
|
237
|
+
counting uses a tokenizer-only handle, so PPR search, budgeted reads,
|
|
238
|
+
diff, and doctor stay cold.
|
|
235
239
|
- **Last-token pooling** (not mean) — required by Jina v5; mean pooling
|
|
236
240
|
breaks alignment with omni image embeddings.
|
|
237
241
|
- **Single pinned ORT** (`onnxruntime==1.29.0`, `ORT_DYLIB_PATH`-overridable)
|
|
@@ -13,7 +13,7 @@ stubs (``...``) until their respective phase moves the implementation over.
|
|
|
13
13
|
from okfgraph.components.schema import SchemaManager
|
|
14
14
|
from okfgraph.components.delta import DeltaDetector
|
|
15
15
|
from okfgraph.components.purge import PurgeManager
|
|
16
|
-
from okfgraph.components.embedding import EmbeddingEngine
|
|
16
|
+
from okfgraph.components.embedding import EmbeddingEngine, LazyRustEncoder
|
|
17
17
|
from okfgraph.components.image_assets import ImageAssetManager
|
|
18
18
|
from okfgraph.components.search import SearchEngine
|
|
19
19
|
from okfgraph.components.import_ import ImportManager, parse_source_file
|
|
@@ -37,6 +37,7 @@ __all__ = [
|
|
|
37
37
|
"DeltaDetector",
|
|
38
38
|
"PurgeManager",
|
|
39
39
|
"EmbeddingEngine",
|
|
40
|
+
"LazyRustEncoder",
|
|
40
41
|
"ImageAssetManager",
|
|
41
42
|
"SearchEngine",
|
|
42
43
|
"ImportManager",
|
|
@@ -6,33 +6,250 @@ here. Public callers reach these via router.<method> (component bridge).
|
|
|
6
6
|
"""
|
|
7
7
|
import logging
|
|
8
8
|
import math
|
|
9
|
+
import threading
|
|
9
10
|
from pathlib import Path
|
|
10
11
|
|
|
11
12
|
import mordant
|
|
12
13
|
from typing import Any, Dict, List, Optional
|
|
13
14
|
logger = logging.getLogger(__name__)
|
|
14
15
|
|
|
15
|
-
|
|
16
|
-
|
|
16
|
+
_ORT_MODULE_NAMES = ("onnxruntime", "onnxruntime-gpu")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _candidate_ort_library_names(os_name=None, sys_platform=None):
|
|
20
|
+
"""Return the ORT library filenames for an OS/platform pair.
|
|
21
|
+
|
|
22
|
+
Takes explicit arguments so unit tests can cover Windows/macOS/Linux
|
|
23
|
+
from any host. Linux uses a versioned ``libonnxruntime.so.*`` glob
|
|
24
|
+
(handled by :func:`_find_runtime_in_package`); this helper returns the
|
|
25
|
+
unversioned fallback name for that platform.
|
|
26
|
+
"""
|
|
27
|
+
import os
|
|
28
|
+
import sys
|
|
29
|
+
if os_name is None:
|
|
30
|
+
os_name = os.name
|
|
31
|
+
if sys_platform is None:
|
|
32
|
+
sys_platform = sys.platform
|
|
33
|
+
if os_name == "nt":
|
|
34
|
+
return ("onnxruntime.dll",)
|
|
35
|
+
if sys_platform == "darwin":
|
|
36
|
+
return ("libonnxruntime.dylib",)
|
|
37
|
+
return ("libonnxruntime.so",)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _find_runtime_in_package(package_dir, os_name=None, sys_platform=None):
|
|
41
|
+
"""Find an ORT shared library inside a pip-installed package directory.
|
|
42
|
+
|
|
43
|
+
Returns a :class:`Path` or ``None``. Prefers versioned Linux libraries
|
|
44
|
+
(``libonnxruntime.so.*``) over the unversioned ``libonnxruntime.so``.
|
|
45
|
+
"""
|
|
46
|
+
import os
|
|
47
|
+
import sys
|
|
48
|
+
if os_name is None:
|
|
49
|
+
os_name = os.name
|
|
50
|
+
if sys_platform is None:
|
|
51
|
+
sys_platform = sys.platform
|
|
52
|
+
capi_dir = Path(package_dir) / "capi"
|
|
53
|
+
if not capi_dir.is_dir():
|
|
54
|
+
return None
|
|
55
|
+
if os_name == "nt" or sys_platform == "darwin":
|
|
56
|
+
candidate = capi_dir / _candidate_ort_library_names(os_name, sys_platform)[0]
|
|
57
|
+
return candidate if candidate.is_file() else None
|
|
58
|
+
versioned = sorted(capi_dir.glob("libonnxruntime.so.*"))
|
|
59
|
+
if versioned:
|
|
60
|
+
return versioned[-1]
|
|
61
|
+
candidate = capi_dir / "libonnxruntime.so"
|
|
62
|
+
return candidate if candidate.is_file() else None
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _configure_windows_ort_dll_directory(runtime_path) -> bool:
|
|
66
|
+
"""Add the ORT ``capi`` directory to Windows DLL resolution.
|
|
67
|
+
|
|
68
|
+
Best-effort only: returns ``False`` (never raises) when unavailable.
|
|
69
|
+
"""
|
|
70
|
+
import os
|
|
71
|
+
if os.name != "nt":
|
|
72
|
+
return False
|
|
73
|
+
add_dll_directory = getattr(os, "add_dll_directory", None)
|
|
74
|
+
if not callable(add_dll_directory):
|
|
75
|
+
return False
|
|
76
|
+
try:
|
|
77
|
+
add_dll_directory(str(Path(runtime_path).parent))
|
|
78
|
+
return True
|
|
79
|
+
except OSError as exc:
|
|
80
|
+
logger.debug("ORT DLL directory configuration failed: %s", exc)
|
|
81
|
+
return False
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _warm_ort_gpu_dlls(module) -> bool:
|
|
85
|
+
"""Preload NVIDIA DLLs when the installed ORT package exposes CUDA.
|
|
86
|
+
|
|
87
|
+
This mirrors bobine's GPU bootstrap: only attempt the preload when the
|
|
88
|
+
package reports a CUDA execution provider, and never let it fail
|
|
89
|
+
runtime discovery.
|
|
90
|
+
"""
|
|
91
|
+
preload = getattr(module, "preload_dlls", None)
|
|
92
|
+
if not callable(preload):
|
|
93
|
+
return False
|
|
94
|
+
try:
|
|
95
|
+
providers = list(module.get_available_providers())
|
|
96
|
+
except Exception as exc:
|
|
97
|
+
logger.debug("ORT provider query failed: %s", exc)
|
|
98
|
+
return False
|
|
99
|
+
if "CUDAExecutionProvider" not in providers:
|
|
100
|
+
return False
|
|
101
|
+
try:
|
|
102
|
+
preload()
|
|
103
|
+
return True
|
|
104
|
+
except Exception as exc:
|
|
105
|
+
logger.debug("ORT GPU DLL preload failed: %s", exc)
|
|
106
|
+
return False
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def resolve_ort_dylib(*, warm_gpu: bool = True, os_name=None, sys_platform=None) -> Optional[str]:
|
|
110
|
+
"""Point ``ORT_DYLIB_PATH`` at a pip-installed ORT build when unset.
|
|
17
111
|
|
|
18
112
|
Both bobine and okf-embed load ONNX Runtime dynamically; sharing one
|
|
19
113
|
binary avoids version/CUDA drift between the two runtimes. Explicit
|
|
20
114
|
user configuration always wins — this only fills the gap.
|
|
115
|
+
|
|
116
|
+
Resolution order:
|
|
117
|
+
|
|
118
|
+
1. Existing ``ORT_DYLIB_PATH``.
|
|
119
|
+
2. Pip-installed ``onnxruntime`` or ``onnxruntime-gpu`` package.
|
|
120
|
+
3. OS loader path (represented by returning ``None``).
|
|
121
|
+
|
|
122
|
+
Missing runtimes never raise here; session creation or encoding is the
|
|
123
|
+
fail-fast boundary. ``os_name``/``sys_platform`` are explicit so unit
|
|
124
|
+
tests can cover every platform from any host (patching ``os.name``
|
|
125
|
+
would break ``pathlib`` dispatch on Windows).
|
|
21
126
|
"""
|
|
22
127
|
import os
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
128
|
+
explicit = os.environ.get("ORT_DYLIB_PATH")
|
|
129
|
+
if explicit:
|
|
130
|
+
return explicit
|
|
131
|
+
for module_name in _ORT_MODULE_NAMES:
|
|
132
|
+
try:
|
|
133
|
+
module = __import__(module_name)
|
|
134
|
+
except ImportError:
|
|
135
|
+
continue
|
|
136
|
+
try:
|
|
137
|
+
package_dir = Path(str(module.__file__)).parent
|
|
138
|
+
except Exception:
|
|
139
|
+
continue
|
|
140
|
+
found = _find_runtime_in_package(package_dir, os_name, sys_platform)
|
|
141
|
+
if found is None:
|
|
142
|
+
continue
|
|
143
|
+
os.environ["ORT_DYLIB_PATH"] = str(found)
|
|
144
|
+
_configure_windows_ort_dll_directory(found)
|
|
145
|
+
if warm_gpu:
|
|
146
|
+
_warm_ort_gpu_dlls(module)
|
|
147
|
+
logger.debug("resolved ORT runtime: %s", found)
|
|
148
|
+
return str(found)
|
|
33
149
|
return None
|
|
34
150
|
|
|
35
151
|
|
|
152
|
+
class LazyRustEncoder:
|
|
153
|
+
"""Defers the ONNX session open until the first real encode.
|
|
154
|
+
|
|
155
|
+
Router construction stays cheap: the ``okf-embed`` wheel import is still
|
|
156
|
+
validated eagerly (fail fast on a missing install), but ``JinaV5.open``
|
|
157
|
+
— model download + session build — waits for the first ``encode`` /
|
|
158
|
+
``encode_batch`` / ``used_cuda`` access. Token counting uses the
|
|
159
|
+
separate lightweight ``JinaTokenizer`` handle, so budgeted reads and the
|
|
160
|
+
context-window guard stay cold too.
|
|
161
|
+
|
|
162
|
+
A failed session open is cached and re-raised: configuration errors stay
|
|
163
|
+
fail-fast (once, at first encode) instead of retrying network/model
|
|
164
|
+
acquisition on every call. Thread-safe: concurrent first encodes open
|
|
165
|
+
exactly one session.
|
|
166
|
+
"""
|
|
167
|
+
|
|
168
|
+
def __init__(self, *, model_id, truncate_dim, device,
|
|
169
|
+
session_factory, tokenizer_factory, on_open=None):
|
|
170
|
+
self._model_id = model_id
|
|
171
|
+
self._truncate_dim = truncate_dim
|
|
172
|
+
self._device = device
|
|
173
|
+
self._session_factory = session_factory
|
|
174
|
+
self._tokenizer_factory = tokenizer_factory
|
|
175
|
+
self._on_open = on_open
|
|
176
|
+
self._lock = threading.Lock()
|
|
177
|
+
self._encoder = None
|
|
178
|
+
self._encoder_error = None
|
|
179
|
+
self._open_reported = False
|
|
180
|
+
self._tokenizer = None
|
|
181
|
+
|
|
182
|
+
@property
|
|
183
|
+
def is_loaded(self) -> bool:
|
|
184
|
+
"""True once the ONNX session has been opened."""
|
|
185
|
+
return self._encoder is not None
|
|
186
|
+
|
|
187
|
+
@property
|
|
188
|
+
def model_id(self) -> str:
|
|
189
|
+
return self._model_id
|
|
190
|
+
|
|
191
|
+
@property
|
|
192
|
+
def dim(self) -> int:
|
|
193
|
+
return self._truncate_dim
|
|
194
|
+
|
|
195
|
+
@property
|
|
196
|
+
def used_cuda(self) -> bool:
|
|
197
|
+
"""Effective device — opens the session on first access."""
|
|
198
|
+
return bool(self._get_encoder().used_cuda)
|
|
199
|
+
|
|
200
|
+
def encode(self, text: str, task: str = "Document"):
|
|
201
|
+
return self._get_encoder().encode(text, task=task)
|
|
202
|
+
|
|
203
|
+
def encode_batch(self, texts, task: str = "Document"):
|
|
204
|
+
return self._get_encoder().encode_batch(texts, task=task)
|
|
205
|
+
|
|
206
|
+
def count_tokens(self, text: str) -> int:
|
|
207
|
+
"""Exact count via the tokenizer-only handle (never opens the session)."""
|
|
208
|
+
return int(self._get_tokenizer().count_tokens(text))
|
|
209
|
+
|
|
210
|
+
def _get_encoder(self):
|
|
211
|
+
encoder = self._encoder
|
|
212
|
+
if encoder is not None:
|
|
213
|
+
return encoder
|
|
214
|
+
with self._lock:
|
|
215
|
+
if self._encoder is not None:
|
|
216
|
+
return self._encoder
|
|
217
|
+
if self._encoder_error is not None:
|
|
218
|
+
raise self._encoder_error
|
|
219
|
+
try:
|
|
220
|
+
encoder = self._session_factory()
|
|
221
|
+
except Exception as exc:
|
|
222
|
+
self._encoder_error = exc
|
|
223
|
+
raise
|
|
224
|
+
self._encoder = encoder
|
|
225
|
+
if self._on_open is not None and not self._open_reported:
|
|
226
|
+
self._open_reported = True
|
|
227
|
+
self._on_open(encoder)
|
|
228
|
+
return encoder
|
|
229
|
+
|
|
230
|
+
def _get_tokenizer(self):
|
|
231
|
+
tokenizer = self._tokenizer
|
|
232
|
+
if tokenizer is not None:
|
|
233
|
+
return tokenizer
|
|
234
|
+
with self._lock:
|
|
235
|
+
if self._tokenizer is not None:
|
|
236
|
+
return self._tokenizer
|
|
237
|
+
try:
|
|
238
|
+
tokenizer = self._tokenizer_factory()
|
|
239
|
+
except Exception as exc:
|
|
240
|
+
logger.debug("tokenizer-only open failed: %s", exc)
|
|
241
|
+
raise
|
|
242
|
+
self._tokenizer = tokenizer
|
|
243
|
+
return tokenizer
|
|
244
|
+
|
|
245
|
+
def __repr__(self) -> str:
|
|
246
|
+
state = "loaded" if self._encoder is not None else "cold"
|
|
247
|
+
return (
|
|
248
|
+
f"LazyRustEncoder({self._model_id}, "
|
|
249
|
+
f"dim={self._truncate_dim}, {state})"
|
|
250
|
+
)
|
|
251
|
+
|
|
252
|
+
|
|
36
253
|
class EmbeddingEngine:
|
|
37
254
|
"""Owns the embedding model and chunking logic.
|
|
38
255
|
|
|
@@ -43,7 +260,8 @@ class EmbeddingEngine:
|
|
|
43
260
|
|
|
44
261
|
def __init__(self, rust_encoder, embedding_dim, device,
|
|
45
262
|
cache_dir, model_id, omni_model_id, omni,
|
|
46
|
-
chunk_size, chunk_overlap, enable_chunking, conn
|
|
263
|
+
chunk_size, chunk_overlap, enable_chunking, conn,
|
|
264
|
+
ort_dylib=None):
|
|
47
265
|
self.encoder = rust_encoder
|
|
48
266
|
self.embedding_dim = embedding_dim
|
|
49
267
|
self.device = device
|
|
@@ -55,6 +273,7 @@ class EmbeddingEngine:
|
|
|
55
273
|
self.chunk_overlap = chunk_overlap
|
|
56
274
|
self.enable_chunking = enable_chunking
|
|
57
275
|
self.conn = conn
|
|
276
|
+
self.ort_dylib = ort_dylib
|
|
58
277
|
|
|
59
278
|
def _encode(self, text: str, task: str = "Document") -> List[float]:
|
|
60
279
|
"""Encode text with the Rust Jina v5 encoder.
|
|
@@ -106,6 +106,8 @@ class OKFRouter:
|
|
|
106
106
|
omni_model_id: str = "jinaai/jina-embeddings-v5-omni-small-retrieval",
|
|
107
107
|
embedding_dim: int = 512,
|
|
108
108
|
cache_dir: Optional[str] = None,
|
|
109
|
+
model_path: Optional[str] = None,
|
|
110
|
+
tokenizer_path: Optional[str] = None,
|
|
109
111
|
device: str = "cpu",
|
|
110
112
|
allow_remote_images: bool = False,
|
|
111
113
|
allowed_image_domains: Optional[List[str]] = None,
|
|
@@ -124,6 +126,10 @@ class OKFRouter:
|
|
|
124
126
|
omni_model_id: HuggingFace ID of the Jina v5 omni model (images).
|
|
125
127
|
embedding_dim: Truncated Matryoshka dimension (<= 1024).
|
|
126
128
|
cache_dir: Model cache directory (defaults per-platform).
|
|
129
|
+
model_path: Explicit local ONNX file (with `tokenizer_path`);
|
|
130
|
+
skips every download — air-gapped / reproducible installs.
|
|
131
|
+
An external-data sidecar must sit next to this file.
|
|
132
|
+
tokenizer_path: Explicit local `tokenizer.json` (with `model_path`).
|
|
127
133
|
device: "cpu" (CUDA is opportunistic inside the Rust loader).
|
|
128
134
|
allow_remote_images: Whether http(s) image URLs may be fetched.
|
|
129
135
|
allowed_image_domains: Domain allowlist for remote images.
|
|
@@ -186,7 +192,10 @@ class OKFRouter:
|
|
|
186
192
|
# Text embeddings: Rust okf_embed wheel (Jina v5 via ORT). No Python
|
|
187
193
|
# fallback — a mid-run stack switch would silently mix vector spaces
|
|
188
194
|
# in one index.
|
|
189
|
-
from okfgraph.components.embedding import resolve_ort_dylib
|
|
195
|
+
from okfgraph.components.embedding import LazyRustEncoder, resolve_ort_dylib
|
|
196
|
+
# Resolve first: the native module loads ORT dynamically, so the
|
|
197
|
+
# shared runtime choice must be fixed before importing it.
|
|
198
|
+
self.ort_dylib = resolve_ort_dylib()
|
|
190
199
|
try:
|
|
191
200
|
import okf_embed
|
|
192
201
|
except ImportError:
|
|
@@ -194,24 +203,77 @@ class OKFRouter:
|
|
|
194
203
|
"the okf-embed wheel is required for text embeddings: "
|
|
195
204
|
"build rust/okf-embed (maturin build --release) and install it"
|
|
196
205
|
) from None
|
|
197
|
-
resolve_ort_dylib()
|
|
198
206
|
# The Rust crate knows auto/cpu/cuda; map torch-style aliases.
|
|
207
|
+
# Validate eagerly so a bad device still fails at construction —
|
|
208
|
+
# the session itself opens lazily on first encode (see below).
|
|
199
209
|
rust_device = {"mps": "auto"}.get(device, device)
|
|
200
|
-
|
|
201
|
-
|
|
210
|
+
if rust_device not in ("auto", "cpu", "cuda"):
|
|
211
|
+
raise ValueError(
|
|
212
|
+
f"device must be 'auto', 'cpu' or 'cuda', got '{device}'"
|
|
213
|
+
)
|
|
214
|
+
if (model_path is None) != (tokenizer_path is None):
|
|
215
|
+
raise ValueError(
|
|
216
|
+
"model_path and tokenizer_path must be given together "
|
|
217
|
+
"(explicit local files skip all downloads)"
|
|
218
|
+
)
|
|
219
|
+
explicit_files = model_path is not None
|
|
220
|
+
if explicit_files:
|
|
221
|
+
for label, path in (
|
|
222
|
+
("model_path", model_path),
|
|
223
|
+
("tokenizer_path", tokenizer_path),
|
|
224
|
+
):
|
|
225
|
+
if not Path(path).is_file():
|
|
226
|
+
raise FileNotFoundError(
|
|
227
|
+
f"{label} not found: {path} "
|
|
228
|
+
"(explicit paths skip all downloads)"
|
|
229
|
+
)
|
|
230
|
+
logger.debug("using explicit model files: %s", model_path)
|
|
231
|
+
|
|
232
|
+
def _report_encoder_open(encoder) -> None:
|
|
233
|
+
logger.info(
|
|
234
|
+
"text embeddings: %s dim=%d cuda=%s",
|
|
235
|
+
model_id, embedding_dim, encoder.used_cuda,
|
|
236
|
+
)
|
|
237
|
+
if device == "cuda" and not encoder.used_cuda:
|
|
238
|
+
logger.warning(
|
|
239
|
+
"CUDA requested but the loaded ONNX Runtime has no CUDA execution "
|
|
240
|
+
"provider — running on CPU. Install onnxruntime-gpu for acceleration."
|
|
241
|
+
)
|
|
242
|
+
|
|
243
|
+
# Text embeddings: Rust okf_embed wheel (Jina v5 via ORT). No Python
|
|
244
|
+
# fallback — a mid-run stack switch would silently mix vector spaces
|
|
245
|
+
# in one index. The wheel import above stays fail-fast; the session
|
|
246
|
+
# open is lazy so model-free commands (PPR search, budgeted reads,
|
|
247
|
+
# diff, doctor) never pay model-download/session-build costs.
|
|
248
|
+
if explicit_files:
|
|
249
|
+
session_factory = lambda: okf_embed.JinaV5.open_files(
|
|
250
|
+
str(model_path),
|
|
251
|
+
str(tokenizer_path),
|
|
252
|
+
truncate_dim=embedding_dim,
|
|
253
|
+
device=rust_device,
|
|
254
|
+
)
|
|
255
|
+
tokenizer_factory = lambda: okf_embed.JinaTokenizer.open_files(
|
|
256
|
+
str(tokenizer_path),
|
|
257
|
+
)
|
|
258
|
+
else:
|
|
259
|
+
session_factory = lambda: okf_embed.JinaV5.open(
|
|
260
|
+
model_id,
|
|
261
|
+
truncate_dim=embedding_dim,
|
|
262
|
+
device=rust_device,
|
|
263
|
+
cache_dir=cache_dir,
|
|
264
|
+
)
|
|
265
|
+
tokenizer_factory = lambda: okf_embed.JinaTokenizer.open(
|
|
266
|
+
model_id,
|
|
267
|
+
cache_dir=cache_dir,
|
|
268
|
+
)
|
|
269
|
+
self.encoder = LazyRustEncoder(
|
|
270
|
+
model_id=model_id,
|
|
202
271
|
truncate_dim=embedding_dim,
|
|
203
272
|
device=rust_device,
|
|
204
|
-
|
|
273
|
+
session_factory=session_factory,
|
|
274
|
+
tokenizer_factory=tokenizer_factory,
|
|
275
|
+
on_open=_report_encoder_open,
|
|
205
276
|
)
|
|
206
|
-
logger.info(
|
|
207
|
-
"text embeddings: %s dim=%d cuda=%s",
|
|
208
|
-
model_id, embedding_dim, self.encoder.used_cuda,
|
|
209
|
-
)
|
|
210
|
-
if device == "cuda" and not self.encoder.used_cuda:
|
|
211
|
-
logger.warning(
|
|
212
|
-
"CUDA requested but the loaded ONNX Runtime has no CUDA execution "
|
|
213
|
-
"provider — running on CPU. Install onnxruntime-gpu for acceleration."
|
|
214
|
-
)
|
|
215
277
|
|
|
216
278
|
# ── Component wiring (Phase 1-3 refactor) ───────────────────
|
|
217
279
|
# The facade owns the resources (conn, encoder, lock) and injects
|
|
@@ -232,7 +294,7 @@ class OKFRouter:
|
|
|
232
294
|
self.encoder, self.embedding_dim,
|
|
233
295
|
self.device, self.cache_dir, self.model_id, self.omni_model_id,
|
|
234
296
|
self._omni, self.chunk_size, self.chunk_overlap, self.enable_chunking,
|
|
235
|
-
self.conn,
|
|
297
|
+
self.conn, self.ort_dylib,
|
|
236
298
|
)
|
|
237
299
|
self.image_mgr = ImageAssetManager(
|
|
238
300
|
self.conn, self.embed_engine, self.schema_mgr,
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: okfgraph
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.11
|
|
4
4
|
Summary: Ladybug-backed OKF knowledge graph with ONNX + Jina v5 embeddings
|
|
5
5
|
License-Expression: Apache-2.0 OR MIT
|
|
6
6
|
Requires-Python: >=3.11
|
|
@@ -259,6 +259,10 @@ Key design decisions:
|
|
|
259
259
|
|
|
260
260
|
- **Rust-only embeddings, fail-fast** — no Python fallback; a mid-run stack
|
|
261
261
|
switch would silently mix vector spaces in one index.
|
|
262
|
+
- **Lazy session init** — router construction never downloads the model or
|
|
263
|
+
builds the ONNX session; first encode opens once (thread-safe), token
|
|
264
|
+
counting uses a tokenizer-only handle, so PPR search, budgeted reads,
|
|
265
|
+
diff, and doctor stay cold.
|
|
262
266
|
- **Last-token pooling** (not mean) — required by Jina v5; mean pooling
|
|
263
267
|
breaks alignment with omni image embeddings.
|
|
264
268
|
- **Single pinned ORT** (`onnxruntime==1.29.0`, `ORT_DYLIB_PATH`-overridable)
|
|
@@ -41,6 +41,7 @@ tests/test_delta.py
|
|
|
41
41
|
tests/test_diff.py
|
|
42
42
|
tests/test_directory_hash.py
|
|
43
43
|
tests/test_doctor.py
|
|
44
|
+
tests/test_explicit_files.py
|
|
44
45
|
tests/test_export_compliance.py
|
|
45
46
|
tests/test_gpu_integration.py
|
|
46
47
|
tests/test_graph_enrichment.py
|
|
@@ -48,12 +49,14 @@ tests/test_images.py
|
|
|
48
49
|
tests/test_ingest.py
|
|
49
50
|
tests/test_ingest_tool.py
|
|
50
51
|
tests/test_integration.py
|
|
52
|
+
tests/test_lazy_encoder.py
|
|
51
53
|
tests/test_lint.py
|
|
52
54
|
tests/test_logging.py
|
|
53
55
|
tests/test_mcp_server.py
|
|
54
56
|
tests/test_models.py
|
|
55
57
|
tests/test_obsidian.py
|
|
56
58
|
tests/test_okf_ingest_tool.py
|
|
59
|
+
tests/test_ort.py
|
|
57
60
|
tests/test_packaging.py
|
|
58
61
|
tests/test_parity.py
|
|
59
62
|
tests/test_pdf_e2e.py
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
"""Explicit local model-file tests — no downloads, no ORT session.
|
|
2
|
+
|
|
3
|
+
Validation (pairing, existence) fires at router construction. Routing to
|
|
4
|
+
`open_files` vs `open` is proven with a stubbed `okf_embed` module; numeric
|
|
5
|
+
equivalence of the two acquisition paths is pinned in test_rust_backend.py.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import sys
|
|
9
|
+
import types
|
|
10
|
+
|
|
11
|
+
import pytest
|
|
12
|
+
|
|
13
|
+
from okfgraph.router import OKFRouter
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _stub_embed(monkeypatch, calls):
|
|
17
|
+
class _Session:
|
|
18
|
+
def __init__(self, via):
|
|
19
|
+
self.via = via
|
|
20
|
+
self.used_cuda = False
|
|
21
|
+
|
|
22
|
+
def encode(self, text, task="Document"):
|
|
23
|
+
calls["encode"] += 1
|
|
24
|
+
return [float(len(text))]
|
|
25
|
+
|
|
26
|
+
def encode_batch(self, texts, task="Document"):
|
|
27
|
+
return [self.encode(t, task=task) for t in texts]
|
|
28
|
+
|
|
29
|
+
class _Tokenizer:
|
|
30
|
+
def __init__(self, via):
|
|
31
|
+
self.via = via
|
|
32
|
+
|
|
33
|
+
def count_tokens(self, text):
|
|
34
|
+
return len(text.split())
|
|
35
|
+
|
|
36
|
+
stub = types.SimpleNamespace(
|
|
37
|
+
JinaV5=types.SimpleNamespace(
|
|
38
|
+
open=lambda *a, **k: (_ for _ in ()).throw(
|
|
39
|
+
AssertionError(f"unexpected open: {a} {k}")
|
|
40
|
+
),
|
|
41
|
+
open_files=lambda onnx, tok, **k: (
|
|
42
|
+
calls["open_files"].append((onnx, tok, k)),
|
|
43
|
+
_Session("files"),
|
|
44
|
+
)[1],
|
|
45
|
+
),
|
|
46
|
+
JinaTokenizer=types.SimpleNamespace(
|
|
47
|
+
open=lambda *a, **k: (_ for _ in ()).throw(
|
|
48
|
+
AssertionError(f"unexpected tokenizer open: {a} {k}")
|
|
49
|
+
),
|
|
50
|
+
open_files=lambda tok: (
|
|
51
|
+
calls["tok_files"].append(tok),
|
|
52
|
+
_Tokenizer("files"),
|
|
53
|
+
)[1],
|
|
54
|
+
),
|
|
55
|
+
MAX_LENGTH=8192,
|
|
56
|
+
)
|
|
57
|
+
monkeypatch.setitem(sys.modules, "okf_embed", stub)
|
|
58
|
+
return stub
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _router(tmp_path, **kwargs):
|
|
62
|
+
return OKFRouter(
|
|
63
|
+
db_path=str(tmp_path / "explicit.db"),
|
|
64
|
+
bundle_root=str(tmp_path),
|
|
65
|
+
device="cpu",
|
|
66
|
+
**kwargs,
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def test_paths_must_be_given_together(tmp_path):
|
|
71
|
+
with pytest.raises(ValueError, match="must be given together"):
|
|
72
|
+
_router(tmp_path, model_path="model.onnx")
|
|
73
|
+
with pytest.raises(ValueError, match="must be given together"):
|
|
74
|
+
_router(tmp_path, tokenizer_path="tokenizer.json")
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def test_missing_files_fail_fast(tmp_path):
|
|
78
|
+
with pytest.raises(FileNotFoundError, match="model_path not found"):
|
|
79
|
+
_router(
|
|
80
|
+
tmp_path,
|
|
81
|
+
model_path=str(tmp_path / "nope.onnx"),
|
|
82
|
+
tokenizer_path=str(tmp_path / "tok.json"),
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def test_encode_routes_to_open_files(tmp_path, monkeypatch):
|
|
87
|
+
calls = {"open_files": [], "tok_files": [], "encode": 0}
|
|
88
|
+
_stub_embed(monkeypatch, calls)
|
|
89
|
+
onnx = tmp_path / "model.onnx"
|
|
90
|
+
tok = tmp_path / "tokenizer.json"
|
|
91
|
+
onnx.write_bytes(b"fake")
|
|
92
|
+
tok.write_text("{}", encoding="utf-8")
|
|
93
|
+
|
|
94
|
+
router = _router(tmp_path, model_path=str(onnx), tokenizer_path=str(tok))
|
|
95
|
+
try:
|
|
96
|
+
assert router.encoder.is_loaded is False
|
|
97
|
+
assert router.embed_engine._encode("hi") == [2.0]
|
|
98
|
+
assert router.embed_engine.count_tokens("one two") == 2
|
|
99
|
+
finally:
|
|
100
|
+
router.close()
|
|
101
|
+
|
|
102
|
+
assert calls["open_files"] == [
|
|
103
|
+
(str(onnx), str(tok), {"truncate_dim": 512, "device": "cpu"})
|
|
104
|
+
]
|
|
105
|
+
assert calls["tok_files"] == [str(tok)]
|
|
106
|
+
assert calls["encode"] == 1
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def test_default_path_still_uses_open(tmp_path, monkeypatch):
|
|
110
|
+
import okf_embed as real_embed
|
|
111
|
+
|
|
112
|
+
opened = {}
|
|
113
|
+
|
|
114
|
+
class _Session:
|
|
115
|
+
used_cuda = False
|
|
116
|
+
|
|
117
|
+
def encode(self, text, task="Document"):
|
|
118
|
+
return [1.0]
|
|
119
|
+
|
|
120
|
+
def encode_batch(self, texts, task="Document"):
|
|
121
|
+
return [[1.0] for _ in texts]
|
|
122
|
+
|
|
123
|
+
stub = types.SimpleNamespace(
|
|
124
|
+
JinaV5=types.SimpleNamespace(
|
|
125
|
+
open=lambda *a, **k: (opened.setdefault("open", (a, k)), _Session())[1],
|
|
126
|
+
open_files=lambda *a, **k: (_ for _ in ()).throw(
|
|
127
|
+
AssertionError("open_files must not be used by default")
|
|
128
|
+
),
|
|
129
|
+
),
|
|
130
|
+
JinaTokenizer=types.SimpleNamespace(
|
|
131
|
+
open=lambda *a, **k: types.SimpleNamespace(
|
|
132
|
+
count_tokens=lambda t: 1
|
|
133
|
+
),
|
|
134
|
+
open_files=lambda *a, **k: (_ for _ in ()).throw(
|
|
135
|
+
AssertionError("tokenizer open_files must not be used by default")
|
|
136
|
+
),
|
|
137
|
+
),
|
|
138
|
+
MAX_LENGTH=real_embed.MAX_LENGTH,
|
|
139
|
+
)
|
|
140
|
+
monkeypatch.setitem(sys.modules, "okf_embed", stub)
|
|
141
|
+
|
|
142
|
+
router = _router(tmp_path)
|
|
143
|
+
try:
|
|
144
|
+
assert router.embed_engine._encode("hi") == [1.0]
|
|
145
|
+
finally:
|
|
146
|
+
router.close()
|
|
147
|
+
assert opened["open"][1]["device"] == "cpu"
|
|
@@ -165,13 +165,17 @@ class TestGPUInitialization:
|
|
|
165
165
|
|
|
166
166
|
@pytest.mark.skipif(_has_onnxruntime_gpu(), reason="onnxruntime-gpu is installed")
|
|
167
167
|
def test_no_gpu_runtime_warns_on_cuda_request(self, tmp_dir, capsys):
|
|
168
|
-
"""When the loaded ORT has no CUDA EP and device='cuda',
|
|
168
|
+
"""When the loaded ORT has no CUDA EP and device='cuda', the first
|
|
169
|
+
encode warns on stderr and falls back to CPU (the session now opens
|
|
170
|
+
lazily, so construction alone stays silent)."""
|
|
169
171
|
r = OKFRouter(
|
|
170
172
|
db_path=str(Path(tmp_dir) / "test_no_gpu.db"),
|
|
171
173
|
bundle_root=tmp_dir,
|
|
172
174
|
embedding_dim=512,
|
|
173
175
|
device="cuda",
|
|
174
176
|
)
|
|
177
|
+
assert r.encoder.is_loaded is False
|
|
178
|
+
_ = r.embed_engine._encode("warmup", task="Document")
|
|
175
179
|
assert r.encoder.used_cuda is False
|
|
176
180
|
assert "CUDA" in capsys.readouterr().err
|
|
177
181
|
r.close()
|