okfgraph 0.2.4__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,216 @@
1
+ """Doctor: scored graph health + safe `--fix` repairs.
2
+
3
+ Generalizes ``broken-links``/``repair-links`` (kept as-is) to a full scan:
4
+ ``broken_link`` (error), ``orphan`` / ``stale`` / ``duplicate_title`` /
5
+ ``missing_description`` (warnings), and info-level ``hub_concentration``
6
+ which never affects the score. ``--fix`` applies only unambiguous repairs —
7
+ ISO timestamp normalization and broken-link re-pointing (exact id or unique
8
+ wikilink name) — and never touches ``reviewed: true`` concepts, while still
9
+ reporting findings on them.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import logging
15
+ from datetime import datetime, timezone
16
+ from typing import Any, Dict, List, Tuple
17
+
18
+ logger = logging.getLogger(__name__)
19
+
20
+ # Deduction per finding: (points, cap). Score = max(0, 100 - deductions).
21
+ DEDUCTIONS = {
22
+ "broken_link": (5, 40),
23
+ "orphan": (2, 20),
24
+ "duplicate_title": (3, 15),
25
+ "stale": (1, 10),
26
+ "missing_description": (1, 15),
27
+ }
28
+
29
+ _TIMESTAMP_FORMATS = (
30
+ "%d.%m.%Y", "%d.%m.%Y %H:%M", "%d.%m.%Y %H:%M:%S",
31
+ "%Y/%m/%d", "%Y/%m/%d %H:%M:%S",
32
+ )
33
+
34
+
35
+ def parse_timestamp(value: Any) -> Optional[datetime]:
36
+ """Parse ISO plus a few common human variants; None when unparseable."""
37
+ if value is None:
38
+ return None
39
+ if isinstance(value, datetime):
40
+ return value
41
+ text = str(value).strip()
42
+ if not text:
43
+ return None
44
+ try:
45
+ return datetime.fromisoformat(text.replace("Z", "+00:00"))
46
+ except ValueError:
47
+ pass
48
+ for fmt in _TIMESTAMP_FORMATS:
49
+ try:
50
+ return datetime.strptime(text, fmt)
51
+ except ValueError:
52
+ continue
53
+ return None
54
+
55
+
56
+ def is_reviewed(extra: Any) -> bool:
57
+ """True when the concept's extra MAP carries ``reviewed: true``."""
58
+ if not isinstance(extra, dict):
59
+ return False
60
+ return str(extra.get("reviewed", "")).strip().lower() in ("true", "1", "yes")
61
+
62
+
63
+ class DoctorManager:
64
+ """Health scan + safe repairs over the live graph."""
65
+
66
+ def __init__(self, conn, import_mgr):
67
+ self.conn = conn
68
+ self.import_mgr = import_mgr
69
+
70
+ # -- data ---------------------------------------------------------
71
+
72
+ def _concepts(self) -> List[Dict[str, Any]]:
73
+ rows = self.conn.execute(
74
+ "MATCH (c:Concept) "
75
+ "RETURN c.id, c.title, c.description, c.timestamp, c.extra"
76
+ ).rows_as_dict().get_all()
77
+ return [
78
+ {
79
+ "id": r["c.id"],
80
+ "title": r.get("c.title"),
81
+ "description": r.get("c.description"),
82
+ "timestamp": r.get("c.timestamp"),
83
+ "extra": r.get("c.extra") or {},
84
+ "reviewed": is_reviewed(r.get("c.extra") or {}),
85
+ }
86
+ for r in rows
87
+ ]
88
+
89
+ def _degrees(self) -> Dict[str, Tuple[int, int]]:
90
+ """Concept id -> (out_degree, in_degree) over LINKS_TO."""
91
+ deg: Dict[str, List[int]] = {}
92
+ for r in self.conn.execute(
93
+ "MATCH (a:Concept)-[e:LINKS_TO]->(b:Concept) "
94
+ "RETURN a.id AS src, b.id AS dst"
95
+ ).rows_as_dict().get_all():
96
+ deg.setdefault(r["src"], [0, 0])[0] += 1
97
+ deg.setdefault(r["dst"], [0, 0])[1] += 1
98
+ return {k: (v[0], v[1]) for k, v in deg.items()}
99
+
100
+ # -- diagnose ------------------------------------------------------
101
+
102
+ def diagnose(self, stale_days: int = 365) -> Dict[str, Any]:
103
+ """Run the full health scan. Deterministic finding order (rule, path)."""
104
+ concepts = self._concepts()
105
+ degrees = self._degrees()
106
+ now = datetime.now(timezone.utc)
107
+
108
+ findings: List[Dict[str, Any]] = []
109
+
110
+ def _add(path: str, severity: str, rule: str, message: str) -> None:
111
+ findings.append({
112
+ "path": path, "severity": severity,
113
+ "rule": rule, "message": message,
114
+ })
115
+
116
+ for bl in self.conn.execute(
117
+ "MATCH (bl:BrokenLink) "
118
+ "RETURN bl.source_id AS source, bl.target_id AS target"
119
+ ).rows_as_dict().get_all():
120
+ _add(bl["source"], "error", "broken_link",
121
+ f"links to missing concept '{bl['target']}'")
122
+
123
+ titles: Dict[str, List[str]] = {}
124
+ for c in concepts:
125
+ if c["title"]:
126
+ titles.setdefault(c["title"], []).append(c["id"])
127
+ for title, ids in titles.items():
128
+ if len(ids) > 1:
129
+ for cid in sorted(ids):
130
+ others = sorted(i for i in ids if i != cid)
131
+ _add(cid, "warn", "duplicate_title",
132
+ f"title '{title}' also used by {', '.join(others)}")
133
+
134
+ for c in concepts:
135
+ cid = c["id"]
136
+ if cid.endswith("index") or cid.endswith("log"):
137
+ continue
138
+ out_d, in_d = degrees.get(cid, (0, 0))
139
+ if out_d == 0 and in_d == 0:
140
+ _add(cid, "warn", "orphan", "no incoming or outgoing links")
141
+ if not (c["description"] or "").strip():
142
+ _add(cid, "warn", "missing_description", "empty description")
143
+ ts = parse_timestamp(c["timestamp"])
144
+ if ts is not None:
145
+ if ts.tzinfo is None:
146
+ ts = ts.replace(tzinfo=timezone.utc)
147
+ age_days = (now - ts).total_seconds() / 86400
148
+ if age_days > stale_days:
149
+ _add(cid, "warn", "stale",
150
+ f"timestamp {c['timestamp']} is {int(age_days)} days old")
151
+
152
+ indeg = sorted(
153
+ ((cid, d[1]) for cid, d in degrees.items() if d[1] > 0),
154
+ key=lambda kv: (-kv[1], kv[0]),
155
+ )[:3]
156
+ info = [
157
+ {"rule": "hub_concentration",
158
+ "message": f"'{cid}' has {n} incoming link(s)"}
159
+ for cid, n in indeg
160
+ ]
161
+
162
+ findings.sort(key=lambda f: (f["rule"], f["path"]))
163
+ totals: Dict[str, int] = {}
164
+ for f in findings:
165
+ totals[f["rule"]] = totals.get(f["rule"], 0) + 1
166
+ score = 100
167
+ for rule, count in totals.items():
168
+ points, cap = DEDUCTIONS.get(rule, (0, 0))
169
+ score -= min(points * count, cap)
170
+ score = max(0, score)
171
+
172
+ return {
173
+ "score": score,
174
+ "concepts": len(concepts),
175
+ "findings": findings,
176
+ "summary": totals,
177
+ "info": info,
178
+ }
179
+
180
+ # -- fix ------------------------------------------------------------
181
+
182
+ def fix(self) -> Dict[str, Any]:
183
+ """Apply safe repairs; ``reviewed: true`` concepts are never modified."""
184
+ concepts = self._concepts()
185
+ reviewed = {c["id"] for c in concepts if c["reviewed"]}
186
+ skipped: List[str] = sorted(reviewed)
187
+
188
+ normalized = 0
189
+ for c in concepts:
190
+ if c["id"] in reviewed:
191
+ continue
192
+ ts = parse_timestamp(c["timestamp"])
193
+ if ts is None or c["timestamp"] is None:
194
+ continue
195
+ if ts.tzinfo is None:
196
+ ts = ts.replace(tzinfo=timezone.utc)
197
+ # The column is TIMESTAMP-typed: compare instants, write datetimes.
198
+ current = c["timestamp"]
199
+ current_dt = current if isinstance(current, datetime) else ts
200
+ if current_dt.tzinfo is None:
201
+ current_dt = current_dt.replace(tzinfo=timezone.utc)
202
+ if current_dt.astimezone(timezone.utc) != ts.astimezone(timezone.utc) \
203
+ or not isinstance(current, datetime):
204
+ self.conn.execute(
205
+ "MATCH (c:Concept {id: $id}) SET c.timestamp = $ts",
206
+ {"id": c["id"], "ts": ts},
207
+ )
208
+ normalized += 1
209
+
210
+ repaired = self.import_mgr.repair_links(skip_sources=reviewed)
211
+
212
+ return {
213
+ "repaired_links": repaired,
214
+ "normalized_timestamps": normalized,
215
+ "skipped_reviewed": skipped,
216
+ }
@@ -0,0 +1,323 @@
1
+ """Vector encoding, chunking, and document reconstruction extracted during the OKFRouter Phase 1 refactor.
2
+
3
+ Bodies are verbatim from okfgraph/router.py; the facade (OKFRouter) owns
4
+ the shared resources (conn, embedder, tokenizer, ...) and injects them
5
+ here. Public callers reach these via router.<method> (component bridge).
6
+ """
7
+ import logging
8
+ import math
9
+ from pathlib import Path
10
+
11
+ import mordant
12
+ from typing import Any, Dict, List, Optional
13
+ logger = logging.getLogger(__name__)
14
+
15
+ def resolve_ort_dylib() -> Optional[str]:
16
+ """Point ``ORT_DYLIB_PATH`` at the pip-installed ORT build when unset.
17
+
18
+ Both bobine and okf-embed load ONNX Runtime dynamically; sharing one
19
+ binary avoids version/CUDA drift between the two runtimes. Explicit
20
+ user configuration always wins — this only fills the gap.
21
+ """
22
+ import os
23
+ if os.environ.get("ORT_DYLIB_PATH"):
24
+ return os.environ["ORT_DYLIB_PATH"]
25
+ try:
26
+ import onnxruntime
27
+ dll = Path(str(onnxruntime.__file__)).parent / "capi" / "onnxruntime.dll"
28
+ if dll.exists():
29
+ os.environ["ORT_DYLIB_PATH"] = str(dll)
30
+ return str(dll)
31
+ except ImportError:
32
+ pass
33
+ return None
34
+
35
+
36
+ class EmbeddingEngine:
37
+ """Owns the embedding model and chunking logic.
38
+
39
+ Text embeddings come from the Rust okf_embed wheel (Jina v5 via ORT):
40
+ prefixed, last-token pooled, truncated. There is no Python fallback —
41
+ a mid-run stack switch would silently mix vector spaces in one index.
42
+ """
43
+
44
+ def __init__(self, rust_encoder, embedding_dim, device,
45
+ cache_dir, model_id, omni_model_id, omni,
46
+ chunk_size, chunk_overlap, enable_chunking, conn):
47
+ self.encoder = rust_encoder
48
+ self.embedding_dim = embedding_dim
49
+ self.device = device
50
+ self.cache_dir = cache_dir
51
+ self.model_id = model_id
52
+ self.omni_model_id = omni_model_id
53
+ self._omni = omni
54
+ self.chunk_size = chunk_size
55
+ self.chunk_overlap = chunk_overlap
56
+ self.enable_chunking = enable_chunking
57
+ self.conn = conn
58
+
59
+ def _encode(self, text: str, task: str = "Document") -> List[float]:
60
+ """Encode text with the Rust Jina v5 encoder.
61
+
62
+ Returns an L2-normalised vector, truncated to the Matryoshka dim.
63
+ Last-token pooling is REQUIRED (mean pooling lands in a different
64
+ space that will NOT align with the omni image embeddings).
65
+ """
66
+ return self.encoder.encode(text, task=task)
67
+
68
+ def count_tokens(self, text: str) -> int:
69
+ """Count tokens with the Rust encoder, falling back to chars/4.
70
+
71
+ Used for token-budgeted reads and the context-window guard. Never
72
+ raises: without a tokenizer a rough estimate beats no answer.
73
+ """
74
+ try:
75
+ return int(self.encoder.count_tokens(text))
76
+ except Exception:
77
+ return max(1, len(text) // 4)
78
+
79
+
80
+ def _truncate_normalize(self, vec: List[float]) -> List[float]:
81
+ """Truncate to the configured Matryoshka dimension and L2-renormalise.
82
+
83
+ Both the text and omni encoders pass through here so every vector that
84
+ lands in a Ladybug FLOAT[dim] column is unit-norm and exactly dim-long.
85
+ """
86
+ v = list(vec[: self.embedding_dim])
87
+ if len(v) < self.embedding_dim:
88
+ v = v + [0.0] * (self.embedding_dim - len(v))
89
+ norm = math.sqrt(sum(x * x for x in v))
90
+ if norm > 0:
91
+ v = [x / norm for x in v]
92
+ return v
93
+
94
+
95
+ def _encode_batch(
96
+ self, texts: List[str], task: str = "Document"
97
+ ) -> List[List[float]]:
98
+ """Encode multiple texts via one Rust call (sequential inside,
99
+ avoiding padded-batch attention waste on variable-length docs).
100
+
101
+ Args:
102
+ texts: List of raw texts to encode.
103
+ task: ``"Query"`` or ``"Document"`` — controls the prefix.
104
+
105
+ Returns:
106
+ List of L2-normalised embedding vectors (each truncated to target dim).
107
+ """
108
+ if not texts:
109
+ return []
110
+ # Prefix guard lives in Rust; one boundary crossing for the batch.
111
+ return self.encoder.encode_batch(texts, task=task)
112
+
113
+
114
+ def _get_omni(self):
115
+ """Load the omni model on first use (vision + text towers only)."""
116
+ if self._omni is None:
117
+ from sentence_transformers import SentenceTransformer
118
+
119
+ logger.info(
120
+ "Loading omni model %s (vision modality) on %s ...",
121
+ self.omni_model_id, self.device,
122
+ )
123
+ self._omni = SentenceTransformer(
124
+ self.omni_model_id,
125
+ trust_remote_code=True,
126
+ cache_folder=self.cache_dir,
127
+ device=self.device,
128
+ model_kwargs={"modality": "vision"}, # skip the audio tower
129
+ )
130
+ return self._omni
131
+
132
+
133
+ def _encode_image(self, data: bytes) -> List[float]:
134
+ """Embed raw image bytes with the omni model (shared vector space)."""
135
+ from io import BytesIO
136
+
137
+ from PIL import Image
138
+
139
+ img = Image.open(BytesIO(data))
140
+ if img.mode not in ("RGB", "L"):
141
+ img = img.convert("RGB")
142
+ model = self._get_omni()
143
+ vec = model.encode(
144
+ img,
145
+ truncate_dim=self.embedding_dim,
146
+ normalize_embeddings=True,
147
+ show_progress_bar=False,
148
+ )
149
+ return self._truncate_normalize([float(x) for x in list(vec)])
150
+
151
+
152
+ def _encode_omni_text(self, text: str, task: str = "Query") -> List[float]:
153
+ """Embed text with the omni model's text side (for cross-modal queries)."""
154
+ model = self._get_omni()
155
+ encoder = model.encode_query if task == "Query" else model.encode_document
156
+ vec = encoder(text, truncate_dim=self.embedding_dim)
157
+ return self._truncate_normalize([float(x) for x in list(vec)])
158
+
159
+
160
+ def _compute_overlap_payloads(
161
+ self, chunks: List[Dict[str, Any]]
162
+ ) -> List[Dict[str, Any]]:
163
+ """Add context injection and token overlap in memory for embedding.
164
+
165
+ Combines heading context structures and token tails into a deep
166
+ semantic representation for the encoder without mutating the raw text store.
167
+ """
168
+ payloads: List[Dict[str, Any]] = []
169
+ prev_tail = ""
170
+
171
+ # Structural blocks represent hard semantic boundaries. They should not
172
+ # receive tails from preceding prose, nor generate tails that bleed into
173
+ # subsequent prose.
174
+ STRUCTURAL_BLOCKS = (
175
+ "Heading",
176
+ "CodeBlock",
177
+ "List",
178
+ "Blockquote",
179
+ "Table",
180
+ "Diagram",
181
+ )
182
+
183
+ for chunk in chunks:
184
+ text_to_embed = chunk['chunk_text']
185
+
186
+ # 1. Enforce hard semantic boundary
187
+ # Clear any trailing words from the previous section when hitting a structural block
188
+ if chunk["block_type"] in STRUCTURAL_BLOCKS:
189
+ prev_tail = ""
190
+
191
+ # 2. Apply sliding word boundary window if a tail exists
192
+ if prev_tail:
193
+ text_to_embed = f"{prev_tail}\n\n{text_to_embed}"
194
+
195
+ # 3. Prepend structural Heading Context if available
196
+ if chunk.get("heading_context"):
197
+ text_to_embed = f"{chunk['heading_context']}\n\n{text_to_embed}"
198
+
199
+ payloads.append({
200
+ "chunk_id": f"{chunk['parent_doc_id']}#chunk:{chunk['chunk_index']}",
201
+ "text": text_to_embed,
202
+ })
203
+
204
+ # Compute tail from the PURE chunk text (not the context-enriched string)
205
+ # Structural blocks never generate tails
206
+ if self.chunk_overlap > 0 and chunk["block_type"] not in STRUCTURAL_BLOCKS:
207
+ words = chunk["chunk_text"].split()
208
+ prev_tail = " ".join(words[-self.chunk_overlap:])
209
+ else:
210
+ prev_tail = ""
211
+
212
+ return payloads
213
+
214
+
215
+ def reconstruct_document(self, document_id: str) -> str:
216
+ """Reconstruct original markdown from stored chunks.
217
+
218
+ Uses block_type to determine correct delimiters between chunks.
219
+ Approximate byte-exact reconstruction (~98% fidelity).
220
+ """
221
+ result = self.conn.execute("""
222
+ MATCH (ch:Chunk)
223
+ WHERE ch.parent_doc_id = $id
224
+ RETURN ch.chunk_text AS chunk_text, ch.block_type AS block_type, ch.chunk_index AS chunk_index
225
+ ORDER BY ch.chunk_index
226
+ """, {"id": document_id})
227
+ rows = result.rows_as_dict().get_all()
228
+
229
+ if not rows:
230
+ return None
231
+
232
+ parts = [rows[0]["chunk_text"]]
233
+ for i in range(1, len(rows)):
234
+ sep = mordant.MarkdownChunker.get_delimiter(
235
+ rows[i - 1]["block_type"], rows[i]["block_type"]
236
+ )
237
+ parts.append(sep + rows[i]["chunk_text"])
238
+
239
+ return "".join(parts)
240
+
241
+
242
+ @staticmethod
243
+ def default_cache_dir() -> str:
244
+ """Return the HuggingFace default cache directory."""
245
+ import os
246
+ return os.path.expanduser(os.getenv("HF_HOME", "~/.cache/huggingface"))
247
+
248
+
249
+ @classmethod
250
+ def model_info(cls, model_id: str = "jinaai/jina-embeddings-v5-text-small-retrieval",
251
+ cache_dir: Optional[str] = None) -> Dict[str, Any]:
252
+ """Inspect model cache status without loading the model.
253
+
254
+ Returns a dict with cache location, snapshot path, and disk usage.
255
+ """
256
+ from huggingface_hub import list_repo_files, snapshot_download
257
+
258
+ effective_cache = cache_dir or cls.default_cache_dir()
259
+ info: Dict[str, Any] = {
260
+ "model_id": model_id,
261
+ "cache_dir": effective_cache,
262
+ "cached": False,
263
+ "snapshot_path": None,
264
+ "disk_usage_bytes": 0,
265
+ }
266
+
267
+ try:
268
+ snapshot_path = snapshot_download(
269
+ model_id,
270
+ cache_dir=effective_cache,
271
+ local_files_only=True,
272
+ )
273
+ info["cached"] = True
274
+ info["snapshot_path"] = snapshot_path
275
+ # Calculate disk usage
276
+ snap = Path(snapshot_path)
277
+ if snap.exists():
278
+ info["disk_usage_bytes"] = sum(
279
+ f.stat().st_size for f in snap.rglob("*") if f.is_file()
280
+ )
281
+ except Exception:
282
+ pass # Not cached locally — will download on first use
283
+
284
+ return info
285
+
286
+ # ------------------------------------------------------------------
287
+ # Chunking
288
+ # ------------------------------------------------------------------
289
+
290
+ def _split_into_chunks(
291
+ self, body: str, document_id: str
292
+ ) -> List[Dict[str, Any]]:
293
+ """Split document body into pure blocks using mordant chunker.
294
+
295
+ Uses chunker.get_all_chunks() to get ExtractedChunk objects with
296
+ block_type and byte offsets. Includes headings as separate chunks
297
+ so they are preserved during reconstruction. No overlap is stored.
298
+ """
299
+ chunker = mordant.MarkdownChunker(body)
300
+ chunks: List[Dict[str, Any]] = []
301
+ index = 0
302
+
303
+ current_heading = ""
304
+ for chunk in chunker.get_all_chunks():
305
+ # Track the heading context as we move down the document
306
+ if chunk.block_type == "Heading":
307
+ current_heading = chunk.text
308
+
309
+ chunks.append({
310
+ "parent_doc_id": document_id,
311
+ "chunk_text": chunk.text,
312
+ "block_type": chunk.block_type,
313
+ "start_offset": chunk.start_offset,
314
+ "end_offset": chunk.end_offset,
315
+ "chunk_index": index,
316
+ # Ephemeral context used strictly for constructing the embedding payload
317
+ "heading_context": current_heading if chunk.block_type != "Heading" else ""
318
+ })
319
+ index += 1
320
+
321
+ return chunks
322
+
323
+