okfgraph 0.2.4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- okfgraph/__init__.py +7 -0
- okfgraph/cli.py +1364 -0
- okfgraph/components/__init__.py +61 -0
- okfgraph/components/converters.py +128 -0
- okfgraph/components/delta.py +271 -0
- okfgraph/components/diff.py +172 -0
- okfgraph/components/doctor.py +216 -0
- okfgraph/components/embedding.py +323 -0
- okfgraph/components/export.py +354 -0
- okfgraph/components/image_assets.py +319 -0
- okfgraph/components/import_.py +1110 -0
- okfgraph/components/ingest.py +495 -0
- okfgraph/components/links.py +133 -0
- okfgraph/components/lint.py +118 -0
- okfgraph/components/purge.py +342 -0
- okfgraph/components/ranking.py +168 -0
- okfgraph/components/schema.py +443 -0
- okfgraph/components/search.py +1006 -0
- okfgraph/config.py +393 -0
- okfgraph/images.py +435 -0
- okfgraph/mcp_server.py +461 -0
- okfgraph/models.py +128 -0
- okfgraph/router.py +485 -0
- okfgraph/security.py +269 -0
- okfgraph/tools.py +460 -0
- okfgraph-0.2.4.dist-info/METADATA +25 -0
- okfgraph-0.2.4.dist-info/RECORD +31 -0
- okfgraph-0.2.4.dist-info/WHEEL +5 -0
- okfgraph-0.2.4.dist-info/entry_points.txt +3 -0
- okfgraph-0.2.4.dist-info/licenses/LICENSE +6 -0
- okfgraph-0.2.4.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
"""Doctor: scored graph health + safe `--fix` repairs.
|
|
2
|
+
|
|
3
|
+
Generalizes ``broken-links``/``repair-links`` (kept as-is) to a full scan:
|
|
4
|
+
``broken_link`` (error), ``orphan`` / ``stale`` / ``duplicate_title`` /
|
|
5
|
+
``missing_description`` (warnings), and info-level ``hub_concentration``
|
|
6
|
+
which never affects the score. ``--fix`` applies only unambiguous repairs —
|
|
7
|
+
ISO timestamp normalization and broken-link re-pointing (exact id or unique
|
|
8
|
+
wikilink name) — and never touches ``reviewed: true`` concepts, while still
|
|
9
|
+
reporting findings on them.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import logging
|
|
15
|
+
from datetime import datetime, timezone
|
|
16
|
+
from typing import Any, Dict, List, Tuple
|
|
17
|
+
|
|
18
|
+
logger = logging.getLogger(__name__)
|
|
19
|
+
|
|
20
|
+
# Deduction per finding: (points, cap). Score = max(0, 100 - deductions).
|
|
21
|
+
DEDUCTIONS = {
|
|
22
|
+
"broken_link": (5, 40),
|
|
23
|
+
"orphan": (2, 20),
|
|
24
|
+
"duplicate_title": (3, 15),
|
|
25
|
+
"stale": (1, 10),
|
|
26
|
+
"missing_description": (1, 15),
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
_TIMESTAMP_FORMATS = (
|
|
30
|
+
"%d.%m.%Y", "%d.%m.%Y %H:%M", "%d.%m.%Y %H:%M:%S",
|
|
31
|
+
"%Y/%m/%d", "%Y/%m/%d %H:%M:%S",
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def parse_timestamp(value: Any) -> Optional[datetime]:
|
|
36
|
+
"""Parse ISO plus a few common human variants; None when unparseable."""
|
|
37
|
+
if value is None:
|
|
38
|
+
return None
|
|
39
|
+
if isinstance(value, datetime):
|
|
40
|
+
return value
|
|
41
|
+
text = str(value).strip()
|
|
42
|
+
if not text:
|
|
43
|
+
return None
|
|
44
|
+
try:
|
|
45
|
+
return datetime.fromisoformat(text.replace("Z", "+00:00"))
|
|
46
|
+
except ValueError:
|
|
47
|
+
pass
|
|
48
|
+
for fmt in _TIMESTAMP_FORMATS:
|
|
49
|
+
try:
|
|
50
|
+
return datetime.strptime(text, fmt)
|
|
51
|
+
except ValueError:
|
|
52
|
+
continue
|
|
53
|
+
return None
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def is_reviewed(extra: Any) -> bool:
|
|
57
|
+
"""True when the concept's extra MAP carries ``reviewed: true``."""
|
|
58
|
+
if not isinstance(extra, dict):
|
|
59
|
+
return False
|
|
60
|
+
return str(extra.get("reviewed", "")).strip().lower() in ("true", "1", "yes")
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class DoctorManager:
|
|
64
|
+
"""Health scan + safe repairs over the live graph."""
|
|
65
|
+
|
|
66
|
+
def __init__(self, conn, import_mgr):
|
|
67
|
+
self.conn = conn
|
|
68
|
+
self.import_mgr = import_mgr
|
|
69
|
+
|
|
70
|
+
# -- data ---------------------------------------------------------
|
|
71
|
+
|
|
72
|
+
def _concepts(self) -> List[Dict[str, Any]]:
|
|
73
|
+
rows = self.conn.execute(
|
|
74
|
+
"MATCH (c:Concept) "
|
|
75
|
+
"RETURN c.id, c.title, c.description, c.timestamp, c.extra"
|
|
76
|
+
).rows_as_dict().get_all()
|
|
77
|
+
return [
|
|
78
|
+
{
|
|
79
|
+
"id": r["c.id"],
|
|
80
|
+
"title": r.get("c.title"),
|
|
81
|
+
"description": r.get("c.description"),
|
|
82
|
+
"timestamp": r.get("c.timestamp"),
|
|
83
|
+
"extra": r.get("c.extra") or {},
|
|
84
|
+
"reviewed": is_reviewed(r.get("c.extra") or {}),
|
|
85
|
+
}
|
|
86
|
+
for r in rows
|
|
87
|
+
]
|
|
88
|
+
|
|
89
|
+
def _degrees(self) -> Dict[str, Tuple[int, int]]:
|
|
90
|
+
"""Concept id -> (out_degree, in_degree) over LINKS_TO."""
|
|
91
|
+
deg: Dict[str, List[int]] = {}
|
|
92
|
+
for r in self.conn.execute(
|
|
93
|
+
"MATCH (a:Concept)-[e:LINKS_TO]->(b:Concept) "
|
|
94
|
+
"RETURN a.id AS src, b.id AS dst"
|
|
95
|
+
).rows_as_dict().get_all():
|
|
96
|
+
deg.setdefault(r["src"], [0, 0])[0] += 1
|
|
97
|
+
deg.setdefault(r["dst"], [0, 0])[1] += 1
|
|
98
|
+
return {k: (v[0], v[1]) for k, v in deg.items()}
|
|
99
|
+
|
|
100
|
+
# -- diagnose ------------------------------------------------------
|
|
101
|
+
|
|
102
|
+
def diagnose(self, stale_days: int = 365) -> Dict[str, Any]:
|
|
103
|
+
"""Run the full health scan. Deterministic finding order (rule, path)."""
|
|
104
|
+
concepts = self._concepts()
|
|
105
|
+
degrees = self._degrees()
|
|
106
|
+
now = datetime.now(timezone.utc)
|
|
107
|
+
|
|
108
|
+
findings: List[Dict[str, Any]] = []
|
|
109
|
+
|
|
110
|
+
def _add(path: str, severity: str, rule: str, message: str) -> None:
|
|
111
|
+
findings.append({
|
|
112
|
+
"path": path, "severity": severity,
|
|
113
|
+
"rule": rule, "message": message,
|
|
114
|
+
})
|
|
115
|
+
|
|
116
|
+
for bl in self.conn.execute(
|
|
117
|
+
"MATCH (bl:BrokenLink) "
|
|
118
|
+
"RETURN bl.source_id AS source, bl.target_id AS target"
|
|
119
|
+
).rows_as_dict().get_all():
|
|
120
|
+
_add(bl["source"], "error", "broken_link",
|
|
121
|
+
f"links to missing concept '{bl['target']}'")
|
|
122
|
+
|
|
123
|
+
titles: Dict[str, List[str]] = {}
|
|
124
|
+
for c in concepts:
|
|
125
|
+
if c["title"]:
|
|
126
|
+
titles.setdefault(c["title"], []).append(c["id"])
|
|
127
|
+
for title, ids in titles.items():
|
|
128
|
+
if len(ids) > 1:
|
|
129
|
+
for cid in sorted(ids):
|
|
130
|
+
others = sorted(i for i in ids if i != cid)
|
|
131
|
+
_add(cid, "warn", "duplicate_title",
|
|
132
|
+
f"title '{title}' also used by {', '.join(others)}")
|
|
133
|
+
|
|
134
|
+
for c in concepts:
|
|
135
|
+
cid = c["id"]
|
|
136
|
+
if cid.endswith("index") or cid.endswith("log"):
|
|
137
|
+
continue
|
|
138
|
+
out_d, in_d = degrees.get(cid, (0, 0))
|
|
139
|
+
if out_d == 0 and in_d == 0:
|
|
140
|
+
_add(cid, "warn", "orphan", "no incoming or outgoing links")
|
|
141
|
+
if not (c["description"] or "").strip():
|
|
142
|
+
_add(cid, "warn", "missing_description", "empty description")
|
|
143
|
+
ts = parse_timestamp(c["timestamp"])
|
|
144
|
+
if ts is not None:
|
|
145
|
+
if ts.tzinfo is None:
|
|
146
|
+
ts = ts.replace(tzinfo=timezone.utc)
|
|
147
|
+
age_days = (now - ts).total_seconds() / 86400
|
|
148
|
+
if age_days > stale_days:
|
|
149
|
+
_add(cid, "warn", "stale",
|
|
150
|
+
f"timestamp {c['timestamp']} is {int(age_days)} days old")
|
|
151
|
+
|
|
152
|
+
indeg = sorted(
|
|
153
|
+
((cid, d[1]) for cid, d in degrees.items() if d[1] > 0),
|
|
154
|
+
key=lambda kv: (-kv[1], kv[0]),
|
|
155
|
+
)[:3]
|
|
156
|
+
info = [
|
|
157
|
+
{"rule": "hub_concentration",
|
|
158
|
+
"message": f"'{cid}' has {n} incoming link(s)"}
|
|
159
|
+
for cid, n in indeg
|
|
160
|
+
]
|
|
161
|
+
|
|
162
|
+
findings.sort(key=lambda f: (f["rule"], f["path"]))
|
|
163
|
+
totals: Dict[str, int] = {}
|
|
164
|
+
for f in findings:
|
|
165
|
+
totals[f["rule"]] = totals.get(f["rule"], 0) + 1
|
|
166
|
+
score = 100
|
|
167
|
+
for rule, count in totals.items():
|
|
168
|
+
points, cap = DEDUCTIONS.get(rule, (0, 0))
|
|
169
|
+
score -= min(points * count, cap)
|
|
170
|
+
score = max(0, score)
|
|
171
|
+
|
|
172
|
+
return {
|
|
173
|
+
"score": score,
|
|
174
|
+
"concepts": len(concepts),
|
|
175
|
+
"findings": findings,
|
|
176
|
+
"summary": totals,
|
|
177
|
+
"info": info,
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
# -- fix ------------------------------------------------------------
|
|
181
|
+
|
|
182
|
+
def fix(self) -> Dict[str, Any]:
|
|
183
|
+
"""Apply safe repairs; ``reviewed: true`` concepts are never modified."""
|
|
184
|
+
concepts = self._concepts()
|
|
185
|
+
reviewed = {c["id"] for c in concepts if c["reviewed"]}
|
|
186
|
+
skipped: List[str] = sorted(reviewed)
|
|
187
|
+
|
|
188
|
+
normalized = 0
|
|
189
|
+
for c in concepts:
|
|
190
|
+
if c["id"] in reviewed:
|
|
191
|
+
continue
|
|
192
|
+
ts = parse_timestamp(c["timestamp"])
|
|
193
|
+
if ts is None or c["timestamp"] is None:
|
|
194
|
+
continue
|
|
195
|
+
if ts.tzinfo is None:
|
|
196
|
+
ts = ts.replace(tzinfo=timezone.utc)
|
|
197
|
+
# The column is TIMESTAMP-typed: compare instants, write datetimes.
|
|
198
|
+
current = c["timestamp"]
|
|
199
|
+
current_dt = current if isinstance(current, datetime) else ts
|
|
200
|
+
if current_dt.tzinfo is None:
|
|
201
|
+
current_dt = current_dt.replace(tzinfo=timezone.utc)
|
|
202
|
+
if current_dt.astimezone(timezone.utc) != ts.astimezone(timezone.utc) \
|
|
203
|
+
or not isinstance(current, datetime):
|
|
204
|
+
self.conn.execute(
|
|
205
|
+
"MATCH (c:Concept {id: $id}) SET c.timestamp = $ts",
|
|
206
|
+
{"id": c["id"], "ts": ts},
|
|
207
|
+
)
|
|
208
|
+
normalized += 1
|
|
209
|
+
|
|
210
|
+
repaired = self.import_mgr.repair_links(skip_sources=reviewed)
|
|
211
|
+
|
|
212
|
+
return {
|
|
213
|
+
"repaired_links": repaired,
|
|
214
|
+
"normalized_timestamps": normalized,
|
|
215
|
+
"skipped_reviewed": skipped,
|
|
216
|
+
}
|
|
@@ -0,0 +1,323 @@
|
|
|
1
|
+
"""Vector encoding, chunking, and document reconstruction extracted during the OKFRouter Phase 1 refactor.
|
|
2
|
+
|
|
3
|
+
Bodies are verbatim from okfgraph/router.py; the facade (OKFRouter) owns
|
|
4
|
+
the shared resources (conn, embedder, tokenizer, ...) and injects them
|
|
5
|
+
here. Public callers reach these via router.<method> (component bridge).
|
|
6
|
+
"""
|
|
7
|
+
import logging
|
|
8
|
+
import math
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
import mordant
|
|
12
|
+
from typing import Any, Dict, List, Optional
|
|
13
|
+
logger = logging.getLogger(__name__)
|
|
14
|
+
|
|
15
|
+
def resolve_ort_dylib() -> Optional[str]:
|
|
16
|
+
"""Point ``ORT_DYLIB_PATH`` at the pip-installed ORT build when unset.
|
|
17
|
+
|
|
18
|
+
Both bobine and okf-embed load ONNX Runtime dynamically; sharing one
|
|
19
|
+
binary avoids version/CUDA drift between the two runtimes. Explicit
|
|
20
|
+
user configuration always wins — this only fills the gap.
|
|
21
|
+
"""
|
|
22
|
+
import os
|
|
23
|
+
if os.environ.get("ORT_DYLIB_PATH"):
|
|
24
|
+
return os.environ["ORT_DYLIB_PATH"]
|
|
25
|
+
try:
|
|
26
|
+
import onnxruntime
|
|
27
|
+
dll = Path(str(onnxruntime.__file__)).parent / "capi" / "onnxruntime.dll"
|
|
28
|
+
if dll.exists():
|
|
29
|
+
os.environ["ORT_DYLIB_PATH"] = str(dll)
|
|
30
|
+
return str(dll)
|
|
31
|
+
except ImportError:
|
|
32
|
+
pass
|
|
33
|
+
return None
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class EmbeddingEngine:
|
|
37
|
+
"""Owns the embedding model and chunking logic.
|
|
38
|
+
|
|
39
|
+
Text embeddings come from the Rust okf_embed wheel (Jina v5 via ORT):
|
|
40
|
+
prefixed, last-token pooled, truncated. There is no Python fallback —
|
|
41
|
+
a mid-run stack switch would silently mix vector spaces in one index.
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
def __init__(self, rust_encoder, embedding_dim, device,
|
|
45
|
+
cache_dir, model_id, omni_model_id, omni,
|
|
46
|
+
chunk_size, chunk_overlap, enable_chunking, conn):
|
|
47
|
+
self.encoder = rust_encoder
|
|
48
|
+
self.embedding_dim = embedding_dim
|
|
49
|
+
self.device = device
|
|
50
|
+
self.cache_dir = cache_dir
|
|
51
|
+
self.model_id = model_id
|
|
52
|
+
self.omni_model_id = omni_model_id
|
|
53
|
+
self._omni = omni
|
|
54
|
+
self.chunk_size = chunk_size
|
|
55
|
+
self.chunk_overlap = chunk_overlap
|
|
56
|
+
self.enable_chunking = enable_chunking
|
|
57
|
+
self.conn = conn
|
|
58
|
+
|
|
59
|
+
def _encode(self, text: str, task: str = "Document") -> List[float]:
|
|
60
|
+
"""Encode text with the Rust Jina v5 encoder.
|
|
61
|
+
|
|
62
|
+
Returns an L2-normalised vector, truncated to the Matryoshka dim.
|
|
63
|
+
Last-token pooling is REQUIRED (mean pooling lands in a different
|
|
64
|
+
space that will NOT align with the omni image embeddings).
|
|
65
|
+
"""
|
|
66
|
+
return self.encoder.encode(text, task=task)
|
|
67
|
+
|
|
68
|
+
def count_tokens(self, text: str) -> int:
|
|
69
|
+
"""Count tokens with the Rust encoder, falling back to chars/4.
|
|
70
|
+
|
|
71
|
+
Used for token-budgeted reads and the context-window guard. Never
|
|
72
|
+
raises: without a tokenizer a rough estimate beats no answer.
|
|
73
|
+
"""
|
|
74
|
+
try:
|
|
75
|
+
return int(self.encoder.count_tokens(text))
|
|
76
|
+
except Exception:
|
|
77
|
+
return max(1, len(text) // 4)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _truncate_normalize(self, vec: List[float]) -> List[float]:
|
|
81
|
+
"""Truncate to the configured Matryoshka dimension and L2-renormalise.
|
|
82
|
+
|
|
83
|
+
Both the text and omni encoders pass through here so every vector that
|
|
84
|
+
lands in a Ladybug FLOAT[dim] column is unit-norm and exactly dim-long.
|
|
85
|
+
"""
|
|
86
|
+
v = list(vec[: self.embedding_dim])
|
|
87
|
+
if len(v) < self.embedding_dim:
|
|
88
|
+
v = v + [0.0] * (self.embedding_dim - len(v))
|
|
89
|
+
norm = math.sqrt(sum(x * x for x in v))
|
|
90
|
+
if norm > 0:
|
|
91
|
+
v = [x / norm for x in v]
|
|
92
|
+
return v
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _encode_batch(
|
|
96
|
+
self, texts: List[str], task: str = "Document"
|
|
97
|
+
) -> List[List[float]]:
|
|
98
|
+
"""Encode multiple texts via one Rust call (sequential inside,
|
|
99
|
+
avoiding padded-batch attention waste on variable-length docs).
|
|
100
|
+
|
|
101
|
+
Args:
|
|
102
|
+
texts: List of raw texts to encode.
|
|
103
|
+
task: ``"Query"`` or ``"Document"`` — controls the prefix.
|
|
104
|
+
|
|
105
|
+
Returns:
|
|
106
|
+
List of L2-normalised embedding vectors (each truncated to target dim).
|
|
107
|
+
"""
|
|
108
|
+
if not texts:
|
|
109
|
+
return []
|
|
110
|
+
# Prefix guard lives in Rust; one boundary crossing for the batch.
|
|
111
|
+
return self.encoder.encode_batch(texts, task=task)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _get_omni(self):
|
|
115
|
+
"""Load the omni model on first use (vision + text towers only)."""
|
|
116
|
+
if self._omni is None:
|
|
117
|
+
from sentence_transformers import SentenceTransformer
|
|
118
|
+
|
|
119
|
+
logger.info(
|
|
120
|
+
"Loading omni model %s (vision modality) on %s ...",
|
|
121
|
+
self.omni_model_id, self.device,
|
|
122
|
+
)
|
|
123
|
+
self._omni = SentenceTransformer(
|
|
124
|
+
self.omni_model_id,
|
|
125
|
+
trust_remote_code=True,
|
|
126
|
+
cache_folder=self.cache_dir,
|
|
127
|
+
device=self.device,
|
|
128
|
+
model_kwargs={"modality": "vision"}, # skip the audio tower
|
|
129
|
+
)
|
|
130
|
+
return self._omni
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _encode_image(self, data: bytes) -> List[float]:
|
|
134
|
+
"""Embed raw image bytes with the omni model (shared vector space)."""
|
|
135
|
+
from io import BytesIO
|
|
136
|
+
|
|
137
|
+
from PIL import Image
|
|
138
|
+
|
|
139
|
+
img = Image.open(BytesIO(data))
|
|
140
|
+
if img.mode not in ("RGB", "L"):
|
|
141
|
+
img = img.convert("RGB")
|
|
142
|
+
model = self._get_omni()
|
|
143
|
+
vec = model.encode(
|
|
144
|
+
img,
|
|
145
|
+
truncate_dim=self.embedding_dim,
|
|
146
|
+
normalize_embeddings=True,
|
|
147
|
+
show_progress_bar=False,
|
|
148
|
+
)
|
|
149
|
+
return self._truncate_normalize([float(x) for x in list(vec)])
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _encode_omni_text(self, text: str, task: str = "Query") -> List[float]:
|
|
153
|
+
"""Embed text with the omni model's text side (for cross-modal queries)."""
|
|
154
|
+
model = self._get_omni()
|
|
155
|
+
encoder = model.encode_query if task == "Query" else model.encode_document
|
|
156
|
+
vec = encoder(text, truncate_dim=self.embedding_dim)
|
|
157
|
+
return self._truncate_normalize([float(x) for x in list(vec)])
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _compute_overlap_payloads(
|
|
161
|
+
self, chunks: List[Dict[str, Any]]
|
|
162
|
+
) -> List[Dict[str, Any]]:
|
|
163
|
+
"""Add context injection and token overlap in memory for embedding.
|
|
164
|
+
|
|
165
|
+
Combines heading context structures and token tails into a deep
|
|
166
|
+
semantic representation for the encoder without mutating the raw text store.
|
|
167
|
+
"""
|
|
168
|
+
payloads: List[Dict[str, Any]] = []
|
|
169
|
+
prev_tail = ""
|
|
170
|
+
|
|
171
|
+
# Structural blocks represent hard semantic boundaries. They should not
|
|
172
|
+
# receive tails from preceding prose, nor generate tails that bleed into
|
|
173
|
+
# subsequent prose.
|
|
174
|
+
STRUCTURAL_BLOCKS = (
|
|
175
|
+
"Heading",
|
|
176
|
+
"CodeBlock",
|
|
177
|
+
"List",
|
|
178
|
+
"Blockquote",
|
|
179
|
+
"Table",
|
|
180
|
+
"Diagram",
|
|
181
|
+
)
|
|
182
|
+
|
|
183
|
+
for chunk in chunks:
|
|
184
|
+
text_to_embed = chunk['chunk_text']
|
|
185
|
+
|
|
186
|
+
# 1. Enforce hard semantic boundary
|
|
187
|
+
# Clear any trailing words from the previous section when hitting a structural block
|
|
188
|
+
if chunk["block_type"] in STRUCTURAL_BLOCKS:
|
|
189
|
+
prev_tail = ""
|
|
190
|
+
|
|
191
|
+
# 2. Apply sliding word boundary window if a tail exists
|
|
192
|
+
if prev_tail:
|
|
193
|
+
text_to_embed = f"{prev_tail}\n\n{text_to_embed}"
|
|
194
|
+
|
|
195
|
+
# 3. Prepend structural Heading Context if available
|
|
196
|
+
if chunk.get("heading_context"):
|
|
197
|
+
text_to_embed = f"{chunk['heading_context']}\n\n{text_to_embed}"
|
|
198
|
+
|
|
199
|
+
payloads.append({
|
|
200
|
+
"chunk_id": f"{chunk['parent_doc_id']}#chunk:{chunk['chunk_index']}",
|
|
201
|
+
"text": text_to_embed,
|
|
202
|
+
})
|
|
203
|
+
|
|
204
|
+
# Compute tail from the PURE chunk text (not the context-enriched string)
|
|
205
|
+
# Structural blocks never generate tails
|
|
206
|
+
if self.chunk_overlap > 0 and chunk["block_type"] not in STRUCTURAL_BLOCKS:
|
|
207
|
+
words = chunk["chunk_text"].split()
|
|
208
|
+
prev_tail = " ".join(words[-self.chunk_overlap:])
|
|
209
|
+
else:
|
|
210
|
+
prev_tail = ""
|
|
211
|
+
|
|
212
|
+
return payloads
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def reconstruct_document(self, document_id: str) -> str:
|
|
216
|
+
"""Reconstruct original markdown from stored chunks.
|
|
217
|
+
|
|
218
|
+
Uses block_type to determine correct delimiters between chunks.
|
|
219
|
+
Approximate byte-exact reconstruction (~98% fidelity).
|
|
220
|
+
"""
|
|
221
|
+
result = self.conn.execute("""
|
|
222
|
+
MATCH (ch:Chunk)
|
|
223
|
+
WHERE ch.parent_doc_id = $id
|
|
224
|
+
RETURN ch.chunk_text AS chunk_text, ch.block_type AS block_type, ch.chunk_index AS chunk_index
|
|
225
|
+
ORDER BY ch.chunk_index
|
|
226
|
+
""", {"id": document_id})
|
|
227
|
+
rows = result.rows_as_dict().get_all()
|
|
228
|
+
|
|
229
|
+
if not rows:
|
|
230
|
+
return None
|
|
231
|
+
|
|
232
|
+
parts = [rows[0]["chunk_text"]]
|
|
233
|
+
for i in range(1, len(rows)):
|
|
234
|
+
sep = mordant.MarkdownChunker.get_delimiter(
|
|
235
|
+
rows[i - 1]["block_type"], rows[i]["block_type"]
|
|
236
|
+
)
|
|
237
|
+
parts.append(sep + rows[i]["chunk_text"])
|
|
238
|
+
|
|
239
|
+
return "".join(parts)
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
@staticmethod
|
|
243
|
+
def default_cache_dir() -> str:
|
|
244
|
+
"""Return the HuggingFace default cache directory."""
|
|
245
|
+
import os
|
|
246
|
+
return os.path.expanduser(os.getenv("HF_HOME", "~/.cache/huggingface"))
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
@classmethod
|
|
250
|
+
def model_info(cls, model_id: str = "jinaai/jina-embeddings-v5-text-small-retrieval",
|
|
251
|
+
cache_dir: Optional[str] = None) -> Dict[str, Any]:
|
|
252
|
+
"""Inspect model cache status without loading the model.
|
|
253
|
+
|
|
254
|
+
Returns a dict with cache location, snapshot path, and disk usage.
|
|
255
|
+
"""
|
|
256
|
+
from huggingface_hub import list_repo_files, snapshot_download
|
|
257
|
+
|
|
258
|
+
effective_cache = cache_dir or cls.default_cache_dir()
|
|
259
|
+
info: Dict[str, Any] = {
|
|
260
|
+
"model_id": model_id,
|
|
261
|
+
"cache_dir": effective_cache,
|
|
262
|
+
"cached": False,
|
|
263
|
+
"snapshot_path": None,
|
|
264
|
+
"disk_usage_bytes": 0,
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
try:
|
|
268
|
+
snapshot_path = snapshot_download(
|
|
269
|
+
model_id,
|
|
270
|
+
cache_dir=effective_cache,
|
|
271
|
+
local_files_only=True,
|
|
272
|
+
)
|
|
273
|
+
info["cached"] = True
|
|
274
|
+
info["snapshot_path"] = snapshot_path
|
|
275
|
+
# Calculate disk usage
|
|
276
|
+
snap = Path(snapshot_path)
|
|
277
|
+
if snap.exists():
|
|
278
|
+
info["disk_usage_bytes"] = sum(
|
|
279
|
+
f.stat().st_size for f in snap.rglob("*") if f.is_file()
|
|
280
|
+
)
|
|
281
|
+
except Exception:
|
|
282
|
+
pass # Not cached locally — will download on first use
|
|
283
|
+
|
|
284
|
+
return info
|
|
285
|
+
|
|
286
|
+
# ------------------------------------------------------------------
|
|
287
|
+
# Chunking
|
|
288
|
+
# ------------------------------------------------------------------
|
|
289
|
+
|
|
290
|
+
def _split_into_chunks(
|
|
291
|
+
self, body: str, document_id: str
|
|
292
|
+
) -> List[Dict[str, Any]]:
|
|
293
|
+
"""Split document body into pure blocks using mordant chunker.
|
|
294
|
+
|
|
295
|
+
Uses chunker.get_all_chunks() to get ExtractedChunk objects with
|
|
296
|
+
block_type and byte offsets. Includes headings as separate chunks
|
|
297
|
+
so they are preserved during reconstruction. No overlap is stored.
|
|
298
|
+
"""
|
|
299
|
+
chunker = mordant.MarkdownChunker(body)
|
|
300
|
+
chunks: List[Dict[str, Any]] = []
|
|
301
|
+
index = 0
|
|
302
|
+
|
|
303
|
+
current_heading = ""
|
|
304
|
+
for chunk in chunker.get_all_chunks():
|
|
305
|
+
# Track the heading context as we move down the document
|
|
306
|
+
if chunk.block_type == "Heading":
|
|
307
|
+
current_heading = chunk.text
|
|
308
|
+
|
|
309
|
+
chunks.append({
|
|
310
|
+
"parent_doc_id": document_id,
|
|
311
|
+
"chunk_text": chunk.text,
|
|
312
|
+
"block_type": chunk.block_type,
|
|
313
|
+
"start_offset": chunk.start_offset,
|
|
314
|
+
"end_offset": chunk.end_offset,
|
|
315
|
+
"chunk_index": index,
|
|
316
|
+
# Ephemeral context used strictly for constructing the embedding payload
|
|
317
|
+
"heading_context": current_heading if chunk.block_type != "Heading" else ""
|
|
318
|
+
})
|
|
319
|
+
index += 1
|
|
320
|
+
|
|
321
|
+
return chunks
|
|
322
|
+
|
|
323
|
+
|