okfgraph 0.2.4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- okfgraph/__init__.py +7 -0
- okfgraph/cli.py +1364 -0
- okfgraph/components/__init__.py +61 -0
- okfgraph/components/converters.py +128 -0
- okfgraph/components/delta.py +271 -0
- okfgraph/components/diff.py +172 -0
- okfgraph/components/doctor.py +216 -0
- okfgraph/components/embedding.py +323 -0
- okfgraph/components/export.py +354 -0
- okfgraph/components/image_assets.py +319 -0
- okfgraph/components/import_.py +1110 -0
- okfgraph/components/ingest.py +495 -0
- okfgraph/components/links.py +133 -0
- okfgraph/components/lint.py +118 -0
- okfgraph/components/purge.py +342 -0
- okfgraph/components/ranking.py +168 -0
- okfgraph/components/schema.py +443 -0
- okfgraph/components/search.py +1006 -0
- okfgraph/config.py +393 -0
- okfgraph/images.py +435 -0
- okfgraph/mcp_server.py +461 -0
- okfgraph/models.py +128 -0
- okfgraph/router.py +485 -0
- okfgraph/security.py +269 -0
- okfgraph/tools.py +460 -0
- okfgraph-0.2.4.dist-info/METADATA +25 -0
- okfgraph-0.2.4.dist-info/RECORD +31 -0
- okfgraph-0.2.4.dist-info/WHEEL +5 -0
- okfgraph-0.2.4.dist-info/entry_points.txt +3 -0
- okfgraph-0.2.4.dist-info/licenses/LICENSE +6 -0
- okfgraph-0.2.4.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""OKFRouter component objects.
|
|
2
|
+
|
|
3
|
+
These are extracted from the monolithic ``okfgraph.router.OKFRouter`` as part
|
|
4
|
+
of the Phase 0–4 refactor (see ``docs/plan-router-refactor-components.md``).
|
|
5
|
+
Each component receives its dependencies explicitly via ``__init__`` (dependency
|
|
6
|
+
injection) rather than reaching into a shared ``self``.
|
|
7
|
+
|
|
8
|
+
Phase 0 status: classes and method signatures are scaffolded; method bodies are
|
|
9
|
+
stubs (``...``) until their respective phase moves the implementation over.
|
|
10
|
+
``lint`` is fully implemented (self-contained, no router state required).
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from okfgraph.components.schema import SchemaManager
|
|
14
|
+
from okfgraph.components.delta import DeltaDetector
|
|
15
|
+
from okfgraph.components.purge import PurgeManager
|
|
16
|
+
from okfgraph.components.embedding import EmbeddingEngine
|
|
17
|
+
from okfgraph.components.image_assets import ImageAssetManager
|
|
18
|
+
from okfgraph.components.search import SearchEngine
|
|
19
|
+
from okfgraph.components.import_ import ImportManager, parse_source_file
|
|
20
|
+
from okfgraph.components.export import ExportManager
|
|
21
|
+
from okfgraph.components.ingest import IngestManager
|
|
22
|
+
from okfgraph.components.converters import BobineConverter, ConvertedDocument
|
|
23
|
+
from okfgraph.components.ranking import ppr, seed_ranked_ppr, seeds
|
|
24
|
+
from okfgraph.components.links import (
|
|
25
|
+
build_name_index,
|
|
26
|
+
extract_md_links,
|
|
27
|
+
extract_wikilinks,
|
|
28
|
+
normalize_path_link,
|
|
29
|
+
resolve_wiki,
|
|
30
|
+
)
|
|
31
|
+
from okfgraph.components.diff import DiffManager, DiffState, state_of_dir
|
|
32
|
+
from okfgraph.components.doctor import DoctorManager
|
|
33
|
+
from okfgraph.components.lint import lint_bundle
|
|
34
|
+
|
|
35
|
+
__all__ = [
|
|
36
|
+
"SchemaManager",
|
|
37
|
+
"DeltaDetector",
|
|
38
|
+
"PurgeManager",
|
|
39
|
+
"EmbeddingEngine",
|
|
40
|
+
"ImageAssetManager",
|
|
41
|
+
"SearchEngine",
|
|
42
|
+
"ImportManager",
|
|
43
|
+
"ExportManager",
|
|
44
|
+
"IngestManager",
|
|
45
|
+
"BobineConverter",
|
|
46
|
+
"ConvertedDocument",
|
|
47
|
+
"parse_source_file",
|
|
48
|
+
"seeds",
|
|
49
|
+
"ppr",
|
|
50
|
+
"seed_ranked_ppr",
|
|
51
|
+
"build_name_index",
|
|
52
|
+
"extract_md_links",
|
|
53
|
+
"extract_wikilinks",
|
|
54
|
+
"normalize_path_link",
|
|
55
|
+
"resolve_wiki",
|
|
56
|
+
"DiffManager",
|
|
57
|
+
"DiffState",
|
|
58
|
+
"state_of_dir",
|
|
59
|
+
"DoctorManager",
|
|
60
|
+
"lint_bundle",
|
|
61
|
+
]
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""Document-converter interface for PDF ingestion.
|
|
2
|
+
|
|
3
|
+
The knowledge graph never talks to a conversion engine directly — it talks
|
|
4
|
+
to a :class:`DocumentConverter`. Bobine is just the default implementation;
|
|
5
|
+
any pipeline that turns a PDF into markdown plus staged images can plug in
|
|
6
|
+
by implementing the one-method protocol::
|
|
7
|
+
|
|
8
|
+
class MyConverter:
|
|
9
|
+
def convert(self, pdf_path, output_dir, *, on_page=None):
|
|
10
|
+
...
|
|
11
|
+
return ConvertedDocument(md_path=..., image_dir=..., page_count=...)
|
|
12
|
+
|
|
13
|
+
Device selection for ONNX-backed converters is ORT-level
|
|
14
|
+
(``ORT_DYLIB_PATH``) and intentionally not part of this interface.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import logging
|
|
20
|
+
from dataclasses import dataclass
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from typing import Callable, Optional, Protocol
|
|
23
|
+
|
|
24
|
+
logger = logging.getLogger(__name__)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass(frozen=True)
|
|
28
|
+
class ConvertedDocument:
|
|
29
|
+
"""The only thing the graph needs from a conversion pipeline.
|
|
30
|
+
|
|
31
|
+
Attributes:
|
|
32
|
+
md_path: The converted ``<stem>.md`` file inside ``output_dir``.
|
|
33
|
+
Transient when produced in a temp dir (auto-import) — the graph
|
|
34
|
+
keeps the content, not the file.
|
|
35
|
+
image_dir: Directory holding the staged images (may be empty).
|
|
36
|
+
page_count: Number of pages in the source PDF.
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
md_path: Path
|
|
40
|
+
image_dir: Path
|
|
41
|
+
page_count: int
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class DocumentConverter(Protocol):
|
|
45
|
+
"""One-method protocol every ingestion pipeline must satisfy."""
|
|
46
|
+
|
|
47
|
+
def convert(
|
|
48
|
+
self,
|
|
49
|
+
pdf_path: str | Path,
|
|
50
|
+
output_dir: str | Path,
|
|
51
|
+
*,
|
|
52
|
+
on_page: Callable[[int, int], None] | None = None,
|
|
53
|
+
) -> ConvertedDocument:
|
|
54
|
+
"""Convert ``pdf_path`` into ``output_dir``.
|
|
55
|
+
|
|
56
|
+
Must write ``<stem>.md`` plus any staged images into ``output_dir``
|
|
57
|
+
and return their locations. ``on_page`` receives 0-based
|
|
58
|
+
``(page_index, page_total)`` progress callbacks.
|
|
59
|
+
"""
|
|
60
|
+
... # pragma: no cover - protocol stub
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
# Bobine routing names are Capitalized; the public API takes lowercase.
|
|
64
|
+
_BOBINE_ROUTING = {
|
|
65
|
+
"auto": "Auto",
|
|
66
|
+
"surgical": "Surgical",
|
|
67
|
+
"always": "Always",
|
|
68
|
+
"never": "Never",
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
class BobineConverter:
|
|
73
|
+
"""The bobine (Rust) implementation of :class:`DocumentConverter`.
|
|
74
|
+
|
|
75
|
+
Args:
|
|
76
|
+
routing_mode: "auto" | "surgical" | "always" | "never". Controls
|
|
77
|
+
when bobine invokes ONNX models ("never" = pdf_oxide fast path).
|
|
78
|
+
extract_images: Whether to extract embedded images from the PDF.
|
|
79
|
+
|
|
80
|
+
Raises:
|
|
81
|
+
ValueError: On an unknown routing_mode (at construction — no work done).
|
|
82
|
+
RuntimeError: When converting without bobine installed.
|
|
83
|
+
"""
|
|
84
|
+
|
|
85
|
+
def __init__(self, routing_mode: str = "auto", extract_images: bool = True):
|
|
86
|
+
if routing_mode not in _BOBINE_ROUTING:
|
|
87
|
+
raise ValueError(
|
|
88
|
+
f"routing_mode must be one of {sorted(_BOBINE_ROUTING)}, "
|
|
89
|
+
f"got '{routing_mode}'"
|
|
90
|
+
)
|
|
91
|
+
self.routing_mode = routing_mode
|
|
92
|
+
self.extract_images = extract_images
|
|
93
|
+
|
|
94
|
+
def convert(
|
|
95
|
+
self,
|
|
96
|
+
pdf_path: str | Path,
|
|
97
|
+
output_dir: str | Path,
|
|
98
|
+
*,
|
|
99
|
+
on_page: Callable[[int, int], None] | None = None,
|
|
100
|
+
) -> ConvertedDocument:
|
|
101
|
+
try:
|
|
102
|
+
import bobine # noqa: PLC0415 — optional dependency, lazy import
|
|
103
|
+
except ImportError:
|
|
104
|
+
raise RuntimeError(
|
|
105
|
+
"bobine is required for PDF ingestion: pip install bobine"
|
|
106
|
+
) from None
|
|
107
|
+
|
|
108
|
+
mode = getattr(bobine.RoutingMode, _BOBINE_ROUTING[self.routing_mode])
|
|
109
|
+
config = bobine.ConverterConfig(
|
|
110
|
+
routing_mode=mode,
|
|
111
|
+
extract_images=self.extract_images,
|
|
112
|
+
)
|
|
113
|
+
doc = bobine.ingest_document(
|
|
114
|
+
str(pdf_path),
|
|
115
|
+
str(output_dir),
|
|
116
|
+
config,
|
|
117
|
+
should_continue=lambda: True,
|
|
118
|
+
on_page=on_page
|
|
119
|
+
or (lambda idx, total: logger.info("page %d/%d", idx + 1, total)),
|
|
120
|
+
)
|
|
121
|
+
return ConvertedDocument(
|
|
122
|
+
md_path=Path(doc.md_path),
|
|
123
|
+
image_dir=Path(doc.image_dir),
|
|
124
|
+
page_count=doc.page_count,
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
__all__ = ["ConvertedDocument", "DocumentConverter", "BobineConverter"]
|
|
@@ -0,0 +1,271 @@
|
|
|
1
|
+
"""Change-detection (directory/file hashing) extracted during the OKFRouter Phase 1 refactor.
|
|
2
|
+
|
|
3
|
+
Bodies are verbatim from okfgraph/router.py; the facade (OKFRouter) owns
|
|
4
|
+
the shared resources (conn, embedder, tokenizer, ...) and injects them
|
|
5
|
+
here. Public callers reach these via router.<method> (component bridge).
|
|
6
|
+
"""
|
|
7
|
+
import hashlib
|
|
8
|
+
import json
|
|
9
|
+
import logging
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import Dict, List
|
|
12
|
+
|
|
13
|
+
from okfgraph.components.import_ import is_concept_file
|
|
14
|
+
|
|
15
|
+
logger = logging.getLogger(__name__)
|
|
16
|
+
|
|
17
|
+
class DeltaDetector:
|
|
18
|
+
"""Detects which source files/directories changed since last ingest."""
|
|
19
|
+
|
|
20
|
+
SUPPORTED_SOURCE_EXTS = (".md", ".markdown", ".txt")
|
|
21
|
+
|
|
22
|
+
def __init__(self, conn, bundle_root):
|
|
23
|
+
self.conn = conn
|
|
24
|
+
self.bundle_root = bundle_root
|
|
25
|
+
|
|
26
|
+
def _file_hash(self, file_path: Path) -> str:
|
|
27
|
+
"""SHA-256 hex digest of a file's raw bytes."""
|
|
28
|
+
return hashlib.sha256(file_path.read_bytes()).hexdigest()
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _compute_directory_hash(self, dir_path: Path) -> str:
|
|
32
|
+
"""Compute a combined hash for a directory's contents.
|
|
33
|
+
|
|
34
|
+
Hashes all .md/.txt files in the directory (recursively) sorted by relative path,
|
|
35
|
+
then hashes the concatenation of all file hashes. Returns a single
|
|
36
|
+
SHA-256 hex digest that changes if any file in the subtree changes.
|
|
37
|
+
"""
|
|
38
|
+
file_hashes = []
|
|
39
|
+
for fp in sorted(dir_path.rglob("*")):
|
|
40
|
+
if is_concept_file(fp):
|
|
41
|
+
rel = str(fp.relative_to(dir_path))
|
|
42
|
+
fh = self._file_hash(fp)
|
|
43
|
+
file_hashes.append((rel, fh))
|
|
44
|
+
combined = "|".join(f"{rel}:{fh}" for rel, fh in file_hashes)
|
|
45
|
+
return hashlib.sha256(combined.encode("utf-8")).hexdigest()
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _compute_directory_hash_with_files(self, dir_path: Path) -> tuple[str, List[str]]:
|
|
49
|
+
"""Compute a combined hash for a directory's contents and return file paths.
|
|
50
|
+
|
|
51
|
+
Returns (hash, [relative_file_paths]) where file_paths can be used to
|
|
52
|
+
identify deleted files when the directory is removed.
|
|
53
|
+
"""
|
|
54
|
+
file_hashes = []
|
|
55
|
+
file_paths = []
|
|
56
|
+
for fp in sorted(dir_path.rglob("*")):
|
|
57
|
+
if is_concept_file(fp):
|
|
58
|
+
rel = str(fp.relative_to(dir_path))
|
|
59
|
+
fh = self._file_hash(fp)
|
|
60
|
+
file_hashes.append((rel, fh))
|
|
61
|
+
file_paths.append(rel)
|
|
62
|
+
combined = "|".join(f"{rel}:{fh}" for rel, fh in file_hashes)
|
|
63
|
+
return hashlib.sha256(combined.encode("utf-8")).hexdigest(), file_paths
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _load_directory_hashes(self) -> Dict[str, Dict]:
|
|
67
|
+
"""Load the persisted directory→{hash, files} mapping, or return empty dict."""
|
|
68
|
+
try:
|
|
69
|
+
rows = self.conn.execute(
|
|
70
|
+
"MATCH (d:DirHash) RETURN d.path AS p, d.hash AS h, d.files AS f"
|
|
71
|
+
).rows_as_dict().get_all()
|
|
72
|
+
if rows:
|
|
73
|
+
result = {}
|
|
74
|
+
for r in rows:
|
|
75
|
+
files_str = r.get("f") or ""
|
|
76
|
+
try:
|
|
77
|
+
files = json.loads(files_str) if files_str else []
|
|
78
|
+
except (json.JSONDecodeError, TypeError):
|
|
79
|
+
files = []
|
|
80
|
+
result[r["p"]] = {"hash": r["h"], "files": files}
|
|
81
|
+
return result
|
|
82
|
+
except Exception as exc:
|
|
83
|
+
logger.debug("could not load directory hashes: %s", exc)
|
|
84
|
+
return {}
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _store_directory_hashes(self, hashes: Dict[str, Dict]) -> None:
|
|
88
|
+
"""Persist a path→{hash, files} mapping in the DirHash table."""
|
|
89
|
+
try:
|
|
90
|
+
for path, data in hashes.items():
|
|
91
|
+
files_str = json.dumps(data.get("files", []))
|
|
92
|
+
self.conn.execute(
|
|
93
|
+
"""
|
|
94
|
+
MERGE (d:DirHash {path: $p})
|
|
95
|
+
SET d.hash = $h, d.files = $f
|
|
96
|
+
""",
|
|
97
|
+
{"p": path, "h": data["hash"], "f": files_str},
|
|
98
|
+
)
|
|
99
|
+
except Exception as exc:
|
|
100
|
+
logger.debug("could not store directory hashes: %s", exc)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _changed_directories(self, source_files: List[Path]) -> tuple[List[Path], List[str]]:
|
|
104
|
+
"""Return (changed_files, deleted_paths) using directory-level hash aggregation.
|
|
105
|
+
|
|
106
|
+
Groups files by parent directory, computes combined directory hashes,
|
|
107
|
+
and skips entire subtrees when the directory hash hasn't changed.
|
|
108
|
+
Only files in changed directories are returned as changed.
|
|
109
|
+
Deleted paths are relative file paths from the stored map of deleted directories.
|
|
110
|
+
"""
|
|
111
|
+
stored_dir_hashes = self._load_directory_hashes()
|
|
112
|
+
|
|
113
|
+
# Group files by parent directory
|
|
114
|
+
dir_files: Dict[str, List[Path]] = {}
|
|
115
|
+
for fp in source_files:
|
|
116
|
+
parent = str(fp.parent.relative_to(self.bundle_root))
|
|
117
|
+
dir_files.setdefault(parent, []).append(fp)
|
|
118
|
+
|
|
119
|
+
# Compute current directory hashes with file paths
|
|
120
|
+
current_dir_hashes: Dict[str, Dict] = {}
|
|
121
|
+
changed: List[Path] = []
|
|
122
|
+
for dir_rel, files in dir_files.items():
|
|
123
|
+
dir_path = self.bundle_root / dir_rel
|
|
124
|
+
if dir_path.exists():
|
|
125
|
+
dir_hash, file_paths = self._compute_directory_hash_with_files(dir_path)
|
|
126
|
+
else:
|
|
127
|
+
dir_hash, file_paths = "", []
|
|
128
|
+
current_dir_hashes[dir_rel] = {"hash": dir_hash, "files": file_paths}
|
|
129
|
+
|
|
130
|
+
# Check if directory hash changed
|
|
131
|
+
stored_data = stored_dir_hashes.get(dir_rel, {})
|
|
132
|
+
stored_hash = stored_data.get("hash") if stored_data else None
|
|
133
|
+
if dir_hash != stored_hash:
|
|
134
|
+
# Directory changed — add all files in it
|
|
135
|
+
changed.extend(files)
|
|
136
|
+
|
|
137
|
+
# Detect deleted directories: dirs in stored map but not on disk
|
|
138
|
+
stored_dirs = set(stored_dir_hashes.keys())
|
|
139
|
+
current_dirs = set(current_dir_hashes.keys())
|
|
140
|
+
deleted_dirs = sorted(stored_dirs - current_dirs)
|
|
141
|
+
|
|
142
|
+
# Collect deleted file paths from deleted directories
|
|
143
|
+
deleted_paths: List[str] = []
|
|
144
|
+
for del_dir in deleted_dirs:
|
|
145
|
+
stored_data = stored_dir_hashes.get(del_dir, {})
|
|
146
|
+
stored_files = stored_data.get("files", []) if stored_data else []
|
|
147
|
+
for rel_file in stored_files:
|
|
148
|
+
# Use native path separator to match FileHash entries
|
|
149
|
+
deleted_paths.append(str(Path(del_dir) / rel_file))
|
|
150
|
+
|
|
151
|
+
# Persist the new directory hashes for the next run
|
|
152
|
+
self._store_directory_hashes(current_dir_hashes)
|
|
153
|
+
|
|
154
|
+
if changed:
|
|
155
|
+
logger.info(
|
|
156
|
+
"directory-delta: %d changed dir(s), %d changed files out of %d total",
|
|
157
|
+
len([d for d in dir_files if current_dir_hashes.get(d, {}).get("hash") != stored_dir_hashes.get(d, {}).get("hash")]),
|
|
158
|
+
len(changed),
|
|
159
|
+
len(source_files),
|
|
160
|
+
)
|
|
161
|
+
else:
|
|
162
|
+
logger.info("directory-delta: no changes detected")
|
|
163
|
+
|
|
164
|
+
if deleted_paths:
|
|
165
|
+
logger.info(
|
|
166
|
+
"directory-delta: %d deleted file(s) in %d deleted directory(s): %s",
|
|
167
|
+
len(deleted_paths),
|
|
168
|
+
len(deleted_dirs),
|
|
169
|
+
deleted_paths[:5], # Limit output
|
|
170
|
+
)
|
|
171
|
+
|
|
172
|
+
return changed, deleted_paths
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def _store_file_hashes(self, hashes: Dict[str, str]) -> None:
|
|
176
|
+
"""Persist a path→hash mapping in the FileHash table.
|
|
177
|
+
|
|
178
|
+
Upserts each row so the table always reflects the current state.
|
|
179
|
+
Also stores the concept_id for each file to enable safe purge.
|
|
180
|
+
"""
|
|
181
|
+
try:
|
|
182
|
+
for path, h in hashes.items():
|
|
183
|
+
# Derive concept_id from path: strip extension, normalise separators.
|
|
184
|
+
concept_id = str(Path(path).with_suffix("")).replace("\\", "/") # remove .md / .txt suffix, use forward slashes
|
|
185
|
+
self.conn.execute(
|
|
186
|
+
"""
|
|
187
|
+
MERGE (f:FileHash {path: $p})
|
|
188
|
+
SET f.hash = $h, f.concept_id = $c
|
|
189
|
+
""",
|
|
190
|
+
{"p": path, "h": h, "c": concept_id},
|
|
191
|
+
)
|
|
192
|
+
except Exception as exc:
|
|
193
|
+
logger.debug("could not store file hashes: %s", exc)
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _load_file_hashes(self) -> Dict[str, str]:
|
|
197
|
+
"""Load the persisted path→hash mapping, or return empty dict."""
|
|
198
|
+
try:
|
|
199
|
+
rows = self.conn.execute(
|
|
200
|
+
"MATCH (f:FileHash) RETURN f.path AS p, f.hash AS h"
|
|
201
|
+
).rows_as_dict().get_all()
|
|
202
|
+
if rows:
|
|
203
|
+
return {r["p"]: r["h"] for r in rows}
|
|
204
|
+
except Exception as exc:
|
|
205
|
+
logger.debug("could not load file hashes: %s", exc)
|
|
206
|
+
return {}
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _load_file_hash_concept_ids(self) -> Dict[str, str]:
|
|
210
|
+
"""Load the persisted path→concept_id mapping, or return empty dict."""
|
|
211
|
+
try:
|
|
212
|
+
rows = self.conn.execute(
|
|
213
|
+
"MATCH (f:FileHash) RETURN f.path AS p, f.concept_id AS c"
|
|
214
|
+
).rows_as_dict().get_all()
|
|
215
|
+
if rows:
|
|
216
|
+
return {r["p"]: r["c"] for r in rows}
|
|
217
|
+
except Exception as exc:
|
|
218
|
+
logger.debug("could not load file hash concept ids: %s", exc)
|
|
219
|
+
return {}
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _changed_files(
|
|
223
|
+
self, source_files: List[Path]
|
|
224
|
+
) -> tuple[List[Path], List[str]]:
|
|
225
|
+
"""Return (changed_files, deleted_paths) since last import.
|
|
226
|
+
|
|
227
|
+
Compares SHA-256 of each file against the hashes persisted from the
|
|
228
|
+
previous ``import_bundle()`` call. New files (not in the stored map)
|
|
229
|
+
are treated as changed. Files in the stored map but absent from disk
|
|
230
|
+
are returned as deleted_paths.
|
|
231
|
+
|
|
232
|
+
The stored map is updated after the check.
|
|
233
|
+
"""
|
|
234
|
+
stored = self._load_file_hashes()
|
|
235
|
+
|
|
236
|
+
current: Dict[str, str] = {}
|
|
237
|
+
changed: List[Path] = []
|
|
238
|
+
for fp in source_files:
|
|
239
|
+
rel = str(fp.relative_to(self.bundle_root))
|
|
240
|
+
h = self._file_hash(fp)
|
|
241
|
+
current[rel] = h
|
|
242
|
+
if h != stored.get(rel):
|
|
243
|
+
changed.append(fp)
|
|
244
|
+
|
|
245
|
+
# Detect deleted files: paths in stored map but not on disk.
|
|
246
|
+
stored_paths = set(stored.keys())
|
|
247
|
+
current_paths = set(current.keys())
|
|
248
|
+
deleted = sorted(stored_paths - current_paths)
|
|
249
|
+
|
|
250
|
+
# Persist the new mapping for the next run.
|
|
251
|
+
self._store_file_hashes(current)
|
|
252
|
+
|
|
253
|
+
if changed:
|
|
254
|
+
logger.info(
|
|
255
|
+
"delta: %d changed / %d total files",
|
|
256
|
+
len(changed),
|
|
257
|
+
len(source_files),
|
|
258
|
+
)
|
|
259
|
+
else:
|
|
260
|
+
logger.info("delta: no changes detected")
|
|
261
|
+
|
|
262
|
+
if deleted:
|
|
263
|
+
logger.info(
|
|
264
|
+
"delta: %d deleted files detected: %s",
|
|
265
|
+
len(deleted),
|
|
266
|
+
deleted,
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
return changed, deleted
|
|
270
|
+
|
|
271
|
+
|
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
"""Structural diff: knowledge-structure comparison, not text hunks.
|
|
2
|
+
|
|
3
|
+
Compares two states — a bundle directory (parsed, never imported) or the live
|
|
4
|
+
graph — and reports concepts added/removed/changed, retitles/retypes, edge
|
|
5
|
+
deltas, and newly broken vs fixed links. Pure hash/set comparison; every list
|
|
6
|
+
sorted. Exit-code friendly (``identical`` flag for CI gates).
|
|
7
|
+
|
|
8
|
+
Drift mode (graph vs its bundle dir) answers "what changed since last import";
|
|
9
|
+
snapshot mode (dir vs dir) previews an import or diffs two vaults.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import hashlib
|
|
15
|
+
import logging
|
|
16
|
+
from dataclasses import dataclass, field
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Any, Dict, List, Set, Tuple
|
|
19
|
+
|
|
20
|
+
from okfgraph.components.import_ import is_concept_file, parse_source_file
|
|
21
|
+
from okfgraph.components.links import (
|
|
22
|
+
build_name_index,
|
|
23
|
+
extract_md_links,
|
|
24
|
+
extract_wikilinks,
|
|
25
|
+
is_external,
|
|
26
|
+
normalize_path_link,
|
|
27
|
+
resolve_wiki,
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
logger = logging.getLogger(__name__)
|
|
31
|
+
|
|
32
|
+
# NOTE: file enumeration uses is_concept_file() from import_ (extensions +
|
|
33
|
+
# reserved-name skip) so diff and import agree on what a concept file is.
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def content_hash(body: str) -> str:
|
|
37
|
+
"""SHA-256 over newline-normalized, stripped body text."""
|
|
38
|
+
norm = (body or "").replace("\r\n", "\n").strip()
|
|
39
|
+
return hashlib.sha256(norm.encode("utf-8")).hexdigest()
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class DiffState:
|
|
44
|
+
"""One diffable side: concept meta, resolved edges, unresolved links."""
|
|
45
|
+
|
|
46
|
+
concepts: Dict[str, Dict[str, Any]] = field(default_factory=dict)
|
|
47
|
+
edges: Set[Tuple[str, str]] = field(default_factory=set)
|
|
48
|
+
broken: Set[Tuple[str, str]] = field(default_factory=set)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def state_of_dir(bundle_dir: Path) -> DiffState:
|
|
52
|
+
"""Parse a bundle directory into a DiffState (no database, no import)."""
|
|
53
|
+
state = DiffState()
|
|
54
|
+
files = sorted(fp for fp in Path(bundle_dir).rglob("*") if is_concept_file(fp))
|
|
55
|
+
raw_links: Dict[str, Dict[str, List[str]]] = {}
|
|
56
|
+
index_concepts = []
|
|
57
|
+
for fp in files:
|
|
58
|
+
try:
|
|
59
|
+
concept, body, cid = parse_source_file(fp, Path(bundle_dir))
|
|
60
|
+
except Exception as e:
|
|
61
|
+
logger.warning("diff: skipping unparsable %s: %s", fp, e)
|
|
62
|
+
continue
|
|
63
|
+
extra = concept.model_extra or {}
|
|
64
|
+
state.concepts[cid] = {
|
|
65
|
+
"title": concept.title,
|
|
66
|
+
"type": concept.type,
|
|
67
|
+
"hash": content_hash(body),
|
|
68
|
+
}
|
|
69
|
+
index_concepts.append({
|
|
70
|
+
"id": cid,
|
|
71
|
+
"title": concept.title,
|
|
72
|
+
"uid": extra.get("uid"),
|
|
73
|
+
"aliases": extra.get("aliases"),
|
|
74
|
+
"alias": extra.get("alias"),
|
|
75
|
+
})
|
|
76
|
+
raw_links[cid] = {
|
|
77
|
+
"md": extract_md_links(body),
|
|
78
|
+
"wiki": extract_wikilinks(body),
|
|
79
|
+
}
|
|
80
|
+
maps, _ = build_name_index(index_concepts)
|
|
81
|
+
known = set(state.concepts)
|
|
82
|
+
for cid, links in raw_links.items():
|
|
83
|
+
for raw in links["md"]:
|
|
84
|
+
if is_external(raw):
|
|
85
|
+
continue
|
|
86
|
+
target = normalize_path_link(raw.split("#", 1)[0])
|
|
87
|
+
if target in known:
|
|
88
|
+
state.edges.add((cid, target))
|
|
89
|
+
else:
|
|
90
|
+
state.broken.add((cid, target))
|
|
91
|
+
for raw in links["wiki"]:
|
|
92
|
+
if not raw or is_external(raw):
|
|
93
|
+
continue
|
|
94
|
+
target = resolve_wiki(raw, maps, known)
|
|
95
|
+
if target is not None:
|
|
96
|
+
state.edges.add((cid, target))
|
|
97
|
+
else:
|
|
98
|
+
state.broken.add((cid, raw))
|
|
99
|
+
return state
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
class DiffManager:
|
|
103
|
+
"""Diffs involving the live graph (dir-vs-dir uses ``state_of_dir``)."""
|
|
104
|
+
|
|
105
|
+
def __init__(self, conn):
|
|
106
|
+
self.conn = conn
|
|
107
|
+
|
|
108
|
+
def state_of_db(self) -> DiffState:
|
|
109
|
+
"""Read the live graph into a DiffState."""
|
|
110
|
+
state = DiffState()
|
|
111
|
+
for r in self.conn.execute(
|
|
112
|
+
"MATCH (c:Concept) RETURN c.id, c.title, c.type, c.body"
|
|
113
|
+
).rows_as_dict().get_all():
|
|
114
|
+
state.concepts[r["c.id"]] = {
|
|
115
|
+
"title": r.get("c.title"),
|
|
116
|
+
"type": r.get("c.type"),
|
|
117
|
+
"hash": content_hash(r.get("c.body") or ""),
|
|
118
|
+
}
|
|
119
|
+
for r in self.conn.execute(
|
|
120
|
+
"MATCH (a:Concept)-[:LINKS_TO]->(b:Concept) "
|
|
121
|
+
"RETURN a.id AS src, b.id AS dst"
|
|
122
|
+
).rows_as_dict().get_all():
|
|
123
|
+
state.edges.add((r["src"], r["dst"]))
|
|
124
|
+
for r in self.conn.execute(
|
|
125
|
+
"MATCH (bl:BrokenLink) RETURN bl.source_id AS source, bl.target_id AS target"
|
|
126
|
+
).rows_as_dict().get_all():
|
|
127
|
+
state.broken.add((r["source"], r["target"]))
|
|
128
|
+
return state
|
|
129
|
+
|
|
130
|
+
@staticmethod
|
|
131
|
+
def compare(a: DiffState, b: DiffState) -> Dict[str, Any]:
|
|
132
|
+
"""Pure comparison of two states (a = old, b = new). All lists sorted."""
|
|
133
|
+
a_ids, b_ids = set(a.concepts), set(b.concepts)
|
|
134
|
+
added = sorted(b_ids - a_ids)
|
|
135
|
+
removed = sorted(a_ids - b_ids)
|
|
136
|
+
changed, retitled, retyped = [], [], []
|
|
137
|
+
for cid in sorted(a_ids & b_ids):
|
|
138
|
+
old, new = a.concepts[cid], b.concepts[cid]
|
|
139
|
+
if old["hash"] != new["hash"]:
|
|
140
|
+
changed.append(cid)
|
|
141
|
+
if (old["title"] or "") != (new["title"] or ""):
|
|
142
|
+
retitled.append({"id": cid, "old": old["title"], "new": new["title"]})
|
|
143
|
+
if (old["type"] or "") != (new["type"] or ""):
|
|
144
|
+
retyped.append({"id": cid, "old": old["type"], "new": new["type"]})
|
|
145
|
+
edges_added = sorted(b.edges - a.edges)
|
|
146
|
+
edges_removed = sorted(a.edges - b.edges)
|
|
147
|
+
broken_new = sorted(b.broken - a.broken)
|
|
148
|
+
broken_fixed = sorted(a.broken - b.broken)
|
|
149
|
+
identical = not (
|
|
150
|
+
added or removed or changed or retitled or retyped
|
|
151
|
+
or edges_added or edges_removed or broken_new or broken_fixed
|
|
152
|
+
)
|
|
153
|
+
return {
|
|
154
|
+
"added": added,
|
|
155
|
+
"removed": removed,
|
|
156
|
+
"changed": changed,
|
|
157
|
+
"retitled": retitled,
|
|
158
|
+
"retyped": retyped,
|
|
159
|
+
"edges_added": [list(e) for e in edges_added],
|
|
160
|
+
"edges_removed": [list(e) for e in edges_removed],
|
|
161
|
+
"broken_new": [list(e) for e in broken_new],
|
|
162
|
+
"broken_fixed": [list(e) for e in broken_fixed],
|
|
163
|
+
"identical": identical,
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
def diff_dirs(self, old: Path, new: Path) -> Dict[str, Any]:
|
|
167
|
+
"""Snapshot mode: two bundle directories, no database needed."""
|
|
168
|
+
return self.compare(state_of_dir(old), state_of_dir(new))
|
|
169
|
+
|
|
170
|
+
def diff_db_dir(self, bundle_dir: Path) -> Dict[str, Any]:
|
|
171
|
+
"""Drift mode: live graph (old) vs bundle directory (new)."""
|
|
172
|
+
return self.compare(self.state_of_db(), state_of_dir(bundle_dir))
|