okfgraph 0.2.4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- okfgraph/__init__.py +7 -0
- okfgraph/cli.py +1364 -0
- okfgraph/components/__init__.py +61 -0
- okfgraph/components/converters.py +128 -0
- okfgraph/components/delta.py +271 -0
- okfgraph/components/diff.py +172 -0
- okfgraph/components/doctor.py +216 -0
- okfgraph/components/embedding.py +323 -0
- okfgraph/components/export.py +354 -0
- okfgraph/components/image_assets.py +319 -0
- okfgraph/components/import_.py +1110 -0
- okfgraph/components/ingest.py +495 -0
- okfgraph/components/links.py +133 -0
- okfgraph/components/lint.py +118 -0
- okfgraph/components/purge.py +342 -0
- okfgraph/components/ranking.py +168 -0
- okfgraph/components/schema.py +443 -0
- okfgraph/components/search.py +1006 -0
- okfgraph/config.py +393 -0
- okfgraph/images.py +435 -0
- okfgraph/mcp_server.py +461 -0
- okfgraph/models.py +128 -0
- okfgraph/router.py +485 -0
- okfgraph/security.py +269 -0
- okfgraph/tools.py +460 -0
- okfgraph-0.2.4.dist-info/METADATA +25 -0
- okfgraph-0.2.4.dist-info/RECORD +31 -0
- okfgraph-0.2.4.dist-info/WHEEL +5 -0
- okfgraph-0.2.4.dist-info/entry_points.txt +3 -0
- okfgraph-0.2.4.dist-info/licenses/LICENSE +6 -0
- okfgraph-0.2.4.dist-info/top_level.txt +1 -0
okfgraph/cli.py
ADDED
|
@@ -0,0 +1,1364 @@
|
|
|
1
|
+
"""OKF CLI — Command-line interface for the OKF knowledge graph."""
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
import cProfile
|
|
5
|
+
import io
|
|
6
|
+
import json
|
|
7
|
+
import logging
|
|
8
|
+
import sys
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
from okfgraph.router import OKFRouter
|
|
12
|
+
from okfgraph.config import OKFConfig
|
|
13
|
+
from okfgraph.components.converters import BobineConverter
|
|
14
|
+
|
|
15
|
+
# ── Logging setup ──────────────────────────────────────────────────────────
|
|
16
|
+
# Structured logging with stdlib (Gap #10).
|
|
17
|
+
# Loguru was rejected: third-party dependency for a CLI tool where stdlib
|
|
18
|
+
# logging is sufficient. The goal is consistent, structured, debuggable
|
|
19
|
+
# logging without adding extra dependencies.
|
|
20
|
+
|
|
21
|
+
_LOG_HANDLER = None
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _setup_logging(verbose: bool = False, quiet: bool = False, log_file: str = "") -> None:
|
|
25
|
+
"""Configure logging for the CLI.
|
|
26
|
+
|
|
27
|
+
Precedence: quiet > verbose > default.
|
|
28
|
+
- quiet: ERROR and above only
|
|
29
|
+
- default: INFO
|
|
30
|
+
- verbose: DEBUG
|
|
31
|
+
"""
|
|
32
|
+
global _LOG_HANDLER
|
|
33
|
+
|
|
34
|
+
if quiet:
|
|
35
|
+
level = logging.ERROR
|
|
36
|
+
elif verbose:
|
|
37
|
+
level = logging.DEBUG
|
|
38
|
+
else:
|
|
39
|
+
level = logging.INFO
|
|
40
|
+
|
|
41
|
+
# Configure root logger
|
|
42
|
+
root = logging.getLogger()
|
|
43
|
+
root.setLevel(level)
|
|
44
|
+
|
|
45
|
+
# Remove any existing handlers to avoid duplicates across invocations
|
|
46
|
+
for h in root.handlers[:]:
|
|
47
|
+
root.removeHandler(h)
|
|
48
|
+
|
|
49
|
+
# Console handler with structured format
|
|
50
|
+
fmt = logging.Formatter(
|
|
51
|
+
"%(asctime)s [%(levelname)s] %(name)s: %(message)s",
|
|
52
|
+
datefmt="%H:%M:%S",
|
|
53
|
+
)
|
|
54
|
+
console = logging.StreamHandler(sys.stderr)
|
|
55
|
+
console.setFormatter(fmt)
|
|
56
|
+
console.setLevel(level)
|
|
57
|
+
root.addHandler(console)
|
|
58
|
+
_LOG_HANDLER = console
|
|
59
|
+
|
|
60
|
+
# Optional file handler with rotation
|
|
61
|
+
if log_file:
|
|
62
|
+
from logging.handlers import RotatingFileHandler
|
|
63
|
+
file_handler = RotatingFileHandler(
|
|
64
|
+
log_file,
|
|
65
|
+
maxBytes=5 * 1024 * 1024, # 5MB
|
|
66
|
+
backupCount=3,
|
|
67
|
+
)
|
|
68
|
+
file_handler.setFormatter(fmt)
|
|
69
|
+
file_handler.setLevel(level)
|
|
70
|
+
root.addHandler(file_handler)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _teardown_logging() -> None:
|
|
74
|
+
"""Remove handlers to avoid leaks across CLI invocations."""
|
|
75
|
+
global _LOG_HANDLER
|
|
76
|
+
root = logging.getLogger()
|
|
77
|
+
if _LOG_HANDLER and _LOG_HANDLER in root.handlers:
|
|
78
|
+
root.removeHandler(_LOG_HANDLER)
|
|
79
|
+
_LOG_HANDLER = None
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
# ── helpers ────────────────────────────────────────────────────────────────
|
|
83
|
+
|
|
84
|
+
class _SlimHelpFormatter(argparse.HelpFormatter):
|
|
85
|
+
"""Subcommand help without the repeated global flags.
|
|
86
|
+
|
|
87
|
+
Global options (connection, models, logging) are identical on every
|
|
88
|
+
command, so printing them 15 times costs agents ~4k tokens to learn
|
|
89
|
+
nothing. They are documented once in top-level ``okf --help`` and
|
|
90
|
+
remain fully functional on every subcommand.
|
|
91
|
+
"""
|
|
92
|
+
|
|
93
|
+
def add_arguments(self, actions):
|
|
94
|
+
super().add_arguments(
|
|
95
|
+
[a for a in actions if not getattr(a, "_okf_global", False)]
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
def add_usage(self, usage, actions, groups=(), prefix=None):
|
|
99
|
+
super().add_usage(
|
|
100
|
+
usage,
|
|
101
|
+
[a for a in actions if not getattr(a, "_okf_global", False)],
|
|
102
|
+
groups,
|
|
103
|
+
prefix,
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
class _SubParser(argparse.ArgumentParser):
|
|
108
|
+
"""Subcommand parser with slim help + pointer to global options."""
|
|
109
|
+
|
|
110
|
+
def __init__(self, *args, **kwargs):
|
|
111
|
+
kwargs.setdefault("formatter_class", _SlimHelpFormatter)
|
|
112
|
+
kwargs.setdefault(
|
|
113
|
+
"epilog",
|
|
114
|
+
"Global options hidden; see 'okf --help' (or okfgraph.toml).",
|
|
115
|
+
)
|
|
116
|
+
super().__init__(*args, **kwargs)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _add_global(parser, mark=True):
|
|
120
|
+
"""Add --db / --bundle / --dim / --cache-dir / --device to any subparser.
|
|
121
|
+
|
|
122
|
+
With mark=True (subcommands) the flags are tagged so _SlimHelpFormatter
|
|
123
|
+
hides them from per-command help; they stay functional and are shown
|
|
124
|
+
once in top-level help (mark=False there).
|
|
125
|
+
"""
|
|
126
|
+
def _add(*a, **k):
|
|
127
|
+
act = parser.add_argument(*a, **k)
|
|
128
|
+
if mark:
|
|
129
|
+
act._okf_global = True # noqa: SLF001 — our own marker
|
|
130
|
+
return act
|
|
131
|
+
|
|
132
|
+
_add("--db", default=None, help="Database path (default: okfgraph.db, or from okfgraph.toml)")
|
|
133
|
+
_add("--bundle", default=None, help="Bundle root directory (default: ., or from okfgraph.toml)")
|
|
134
|
+
_add("--dim", type=int, default=None, help="Embedding dimension (Matryoshka; default: 512, or from okfgraph.toml)")
|
|
135
|
+
_add("--cache-dir", default=None, help="HuggingFace model cache directory (default: ~/.cache/huggingface, or from okfgraph.toml)")
|
|
136
|
+
_add("--device", default=None, choices=["cpu", "cuda"], help="Inference device: cpu or cuda (default: cpu, or from okfgraph.toml)")
|
|
137
|
+
_add("--omni-model-id", default=None, help="Multimodal model ID for image embeddings (default from okfgraph.toml)")
|
|
138
|
+
_add("--chunk-size", type=int, default=None, help="Chunk size in words for overlap (default: 512, or from okfgraph.toml)")
|
|
139
|
+
_add("--chunk-overlap", type=int, default=None, help="Overlap in words between chunks (default: 40, or from okfgraph.toml)")
|
|
140
|
+
_add("--no-chunking", action="store_true", help="Disable chunking during ingestion")
|
|
141
|
+
_add("--wal-mode", action="store_true", help="Enable SQLite WAL mode for concurrent reads (Gap #7a)")
|
|
142
|
+
_add("--allow-remote-images", action="store_true", help="Allow fetching remote images (SSRF risk — use with caution)")
|
|
143
|
+
_add("--allowed-image-domains", default=None, help="Comma-separated list of allowed domains for remote images (Gap #9a)")
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _add_logging_flags(parser, mark=True):
|
|
147
|
+
"""Add --verbose / --quiet / --log-file / --profile to a subparser."""
|
|
148
|
+
for args, kwargs in [
|
|
149
|
+
(("--verbose", "-v"), {"action": "store_true", "help": "Enable debug logging"}),
|
|
150
|
+
(("--quiet", "-q"), {"action": "store_true", "help": "Suppress all logging except errors"}),
|
|
151
|
+
(("--log-file",), {"default": "", "help": "Write logs to file (with 5MB rotation)"}),
|
|
152
|
+
(("--profile",), {"action": "store_true", "help": "Enable cProfile for the current invocation (outputs to stdout)"}),
|
|
153
|
+
]:
|
|
154
|
+
act = parser.add_argument(*args, **kwargs)
|
|
155
|
+
if mark:
|
|
156
|
+
act._okf_global = True # noqa: SLF001 — our own marker
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
# Routers opened during a CLI invocation, closed (checkpointed) on exit so a
|
|
160
|
+
# writer never leaves an un-checkpointed WAL that a later open would reject.
|
|
161
|
+
_OPEN_ROUTERS = []
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def _router(args):
|
|
165
|
+
"""Build an OKFRouter from parsed args (registered for cleanup on exit).
|
|
166
|
+
|
|
167
|
+
Uses the config module to merge CLI args with TOML file and env vars.
|
|
168
|
+
Precedence: CLI > env > file > defaults.
|
|
169
|
+
"""
|
|
170
|
+
# Build CLI args dict (only non-None values override config)
|
|
171
|
+
cli_dict = {}
|
|
172
|
+
for attr in ("db", "bundle", "dim", "cache_dir", "device",
|
|
173
|
+
"omni_model_id", "chunk_size", "chunk_overlap",
|
|
174
|
+
"no_chunking", "mode", "batch_size",
|
|
175
|
+
"allow_remote_images", "wal_mode", "allowed_image_domains"):
|
|
176
|
+
val = getattr(args, attr, None)
|
|
177
|
+
if val is not None:
|
|
178
|
+
cli_dict[attr] = val
|
|
179
|
+
|
|
180
|
+
# Resolve bundle root for TOML lookup
|
|
181
|
+
bundle_root = cli_dict.get("bundle") or "."
|
|
182
|
+
|
|
183
|
+
# Load merged config
|
|
184
|
+
config = OKFConfig.load(bundle_root=bundle_root, cli_args=cli_dict)
|
|
185
|
+
|
|
186
|
+
# Build allowed_image_domains list
|
|
187
|
+
allowed_domains = config.import_config.allowed_image_domains
|
|
188
|
+
if getattr(args, "allowed_image_domains", None):
|
|
189
|
+
allowed_domains = [d.strip() for d in args.allowed_image_domains.split(",") if d.strip()]
|
|
190
|
+
|
|
191
|
+
router = OKFRouter(
|
|
192
|
+
db_path=config.database.path,
|
|
193
|
+
bundle_root=str(config.bundle),
|
|
194
|
+
embedding_dim=config.database.dim,
|
|
195
|
+
omni_model_id=config.embedding.omni_model_id,
|
|
196
|
+
cache_dir=config.embedding.cache_dir,
|
|
197
|
+
device=config.embedding.device,
|
|
198
|
+
allow_remote_images=config.import_config.allow_remote_images,
|
|
199
|
+
allowed_image_domains=allowed_domains,
|
|
200
|
+
chunk_size=config.import_config.chunk_size,
|
|
201
|
+
chunk_overlap=config.import_config.chunk_overlap,
|
|
202
|
+
enable_chunking=not config.import_config.no_chunking,
|
|
203
|
+
wal_mode=config.database.wal_mode,
|
|
204
|
+
)
|
|
205
|
+
_OPEN_ROUTERS.append(router)
|
|
206
|
+
return router
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _close_routers():
|
|
210
|
+
"""Checkpoint + close every router opened this invocation."""
|
|
211
|
+
while _OPEN_ROUTERS:
|
|
212
|
+
router = _OPEN_ROUTERS.pop()
|
|
213
|
+
try:
|
|
214
|
+
router.close()
|
|
215
|
+
except Exception:
|
|
216
|
+
pass
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
# ── command handlers ───────────────────────────────────────────────────────
|
|
220
|
+
|
|
221
|
+
def _init(args):
|
|
222
|
+
db_path = str(args.db)
|
|
223
|
+
logger = logging.getLogger("cli")
|
|
224
|
+
logger.info("initializing database at %s (dim=%d)", db_path, args.dim)
|
|
225
|
+
_router(args)
|
|
226
|
+
logger.info("database initialized (embedding_dim=%d)", args.dim)
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def _model_info(args):
|
|
230
|
+
"""Show model cache status without loading the model."""
|
|
231
|
+
logger = logging.getLogger("cli")
|
|
232
|
+
info = OKFRouter.model_info(
|
|
233
|
+
model_id=getattr(args, "model_id", "jinaai/jina-embeddings-v5-text-small-retrieval"),
|
|
234
|
+
cache_dir=getattr(args, "cache_dir", None),
|
|
235
|
+
)
|
|
236
|
+
logger.info("model: %s", info['model_id'])
|
|
237
|
+
logger.info("cache: %s", info['cache_dir'])
|
|
238
|
+
if info["cached"]:
|
|
239
|
+
logger.info("status: cached")
|
|
240
|
+
logger.info("path: %s", info['snapshot_path'])
|
|
241
|
+
size_gb = info["disk_usage_bytes"] / (1024 ** 3)
|
|
242
|
+
logger.info("size: %.2f GB", size_gb)
|
|
243
|
+
else:
|
|
244
|
+
default_cache = OKFRouter.default_cache_dir()
|
|
245
|
+
logger.info("status: not cached (will download on first use)")
|
|
246
|
+
logger.info("will use: %s", default_cache)
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def _import(args):
|
|
250
|
+
logger = logging.getLogger("cli")
|
|
251
|
+
router = _router(args)
|
|
252
|
+
mode = getattr(args, "mode", "text")
|
|
253
|
+
purge = getattr(args, "purge", False)
|
|
254
|
+
if getattr(args, "import_all", False):
|
|
255
|
+
bundle_path = Path(args.bundle) if args.bundle else None
|
|
256
|
+
ids = router.import_mgr.import_bundle(
|
|
257
|
+
bundle_path,
|
|
258
|
+
batch_size=getattr(args, "batch_size", 32) or 32,
|
|
259
|
+
mode=mode,
|
|
260
|
+
purge_deleted=purge,
|
|
261
|
+
)
|
|
262
|
+
logger.info("imported %d concept(s) (mode: %s)", len(ids), mode)
|
|
263
|
+
for cid in ids:
|
|
264
|
+
n = len(router.image_mgr.list_images(cid))
|
|
265
|
+
suffix = f" [{n} image(s)]" if n else ""
|
|
266
|
+
logger.info(" %s%s", cid, suffix)
|
|
267
|
+
else:
|
|
268
|
+
for fp in args.files:
|
|
269
|
+
path = Path(fp)
|
|
270
|
+
if not path.exists():
|
|
271
|
+
logger.warning("skipping %s: file not found", fp)
|
|
272
|
+
continue
|
|
273
|
+
cid = router.import_from_okf(path, mode=mode)
|
|
274
|
+
imgs = router.image_mgr.list_images(cid)
|
|
275
|
+
suffix = f" ({len(imgs)} image(s), mode: {mode})" if imgs else ""
|
|
276
|
+
logger.info("imported: %s%s", cid, suffix)
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
def _search(args):
|
|
280
|
+
"""Unified search: concepts (default), chunks, or images."""
|
|
281
|
+
router = _router(args)
|
|
282
|
+
target = getattr(args, "target", "concepts") or "concepts"
|
|
283
|
+
limit = getattr(args, "limit", 10) or 10
|
|
284
|
+
if target == "images":
|
|
285
|
+
results = router.image_mgr.search_images_with_text(
|
|
286
|
+
text_query=args.query,
|
|
287
|
+
use_text_model=not getattr(args, "use_omni", False),
|
|
288
|
+
limit=limit,
|
|
289
|
+
)
|
|
290
|
+
if not results:
|
|
291
|
+
print("No image results found.")
|
|
292
|
+
return
|
|
293
|
+
print(f"Found {len(results)} image(s):\n")
|
|
294
|
+
for i, r in enumerate(results, 1):
|
|
295
|
+
label = r.get("alt_text") or r.get("file_name") or r.get("id")
|
|
296
|
+
print(f" {i}. [{r['relevance_score']:.4f}] {label} ({r.get('embed_route')})")
|
|
297
|
+
print(f" file: {r.get('file_name')}")
|
|
298
|
+
print(f" id: {r['id']}")
|
|
299
|
+
print()
|
|
300
|
+
return
|
|
301
|
+
tags = args.tags.split(",") if getattr(args, "tags", None) else None
|
|
302
|
+
filt = {
|
|
303
|
+
"concept_type": getattr(args, "type", None),
|
|
304
|
+
"tags": tags,
|
|
305
|
+
"parent_id": getattr(args, "parent", None),
|
|
306
|
+
}
|
|
307
|
+
filt = {k: v for k, v in filt.items() if v is not None}
|
|
308
|
+
rank = getattr(args, "rank", "none") or "none"
|
|
309
|
+
if target == "chunks":
|
|
310
|
+
if rank != "none":
|
|
311
|
+
print("[ERROR] --rank is concepts-only; for chunks use --hub-rerank/--expand")
|
|
312
|
+
return
|
|
313
|
+
if getattr(args, "hub_rerank", False):
|
|
314
|
+
results = router.search_engine.search_chunks_with_hub_score(
|
|
315
|
+
query=args.query,
|
|
316
|
+
limit=limit,
|
|
317
|
+
hub_weight=getattr(args, "hub_weight", 0.3),
|
|
318
|
+
)
|
|
319
|
+
if not results:
|
|
320
|
+
print("No results found.")
|
|
321
|
+
return
|
|
322
|
+
print(f"Found {len(results)} result(s):\n")
|
|
323
|
+
for i, r in enumerate(results, 1):
|
|
324
|
+
print(f" {i}. [{r['final_score']:.4f}] {r['parent_title']} §{r['chunk_index']}")
|
|
325
|
+
print(f" hub={r['hub_score']:.2f} rrf={r['rrf_score']:.4f}")
|
|
326
|
+
print(f" {r['chunk_text'][:150]}")
|
|
327
|
+
print()
|
|
328
|
+
return
|
|
329
|
+
if getattr(args, "expand", False):
|
|
330
|
+
results = router.search_engine.search_with_context(
|
|
331
|
+
query=args.query,
|
|
332
|
+
limit=min(limit, 20),
|
|
333
|
+
context_hops=getattr(args, "context_hops", 1),
|
|
334
|
+
)
|
|
335
|
+
if not results:
|
|
336
|
+
print("No results found.")
|
|
337
|
+
return
|
|
338
|
+
print(f"Found {len(results)} result(s):\n")
|
|
339
|
+
for i, r in enumerate(results, 1):
|
|
340
|
+
chunk = r["chunk"]
|
|
341
|
+
print(f" {i}. [{chunk['rrf_score']:.4f}] {chunk['parent_title']} §{chunk['chunk_index']}")
|
|
342
|
+
print(f" {chunk['chunk_text'][:150]}")
|
|
343
|
+
if r["incoming_links"]:
|
|
344
|
+
titles = [l.get("title", l.get("id", "?")) for l in r["incoming_links"][:3]]
|
|
345
|
+
print(f" ← linked by: {', '.join(titles)}")
|
|
346
|
+
if r["outgoing_links"]:
|
|
347
|
+
titles = [l.get("title", l.get("id", "?")) for l in r["outgoing_links"][:3]]
|
|
348
|
+
print(f" → links to: {', '.join(titles)}")
|
|
349
|
+
if r["ancestry"]:
|
|
350
|
+
print(f" path: {' → '.join(a['title'] for a in r['ancestry'])}")
|
|
351
|
+
if r["siblings"]:
|
|
352
|
+
print(f" siblings: {', '.join(s['title'] for s in r['siblings'][:3])}")
|
|
353
|
+
print()
|
|
354
|
+
return
|
|
355
|
+
results = router.search_engine.search_chunks(
|
|
356
|
+
query=args.query,
|
|
357
|
+
limit=limit,
|
|
358
|
+
**filt,
|
|
359
|
+
)
|
|
360
|
+
if not results:
|
|
361
|
+
print("No chunk results found.")
|
|
362
|
+
return
|
|
363
|
+
print(f"Found {len(results)} chunk(s):\n")
|
|
364
|
+
for i, r in enumerate(results, 1):
|
|
365
|
+
print(f" {i}. [{r['rrf_score']:.4f}] {r['block_type']} #{r['chunk_index']}")
|
|
366
|
+
text = r.get("chunk_text", "")
|
|
367
|
+
print(f" {text[:150]}")
|
|
368
|
+
if r.get("parent_title"):
|
|
369
|
+
print(f" parent: {r['parent_title']}")
|
|
370
|
+
print(f" id: {r['chunk_id']}")
|
|
371
|
+
print()
|
|
372
|
+
return
|
|
373
|
+
results = router.search_hybrid(
|
|
374
|
+
query=args.query,
|
|
375
|
+
limit=limit,
|
|
376
|
+
include_chunks=getattr(args, "chunks", False),
|
|
377
|
+
rank=rank,
|
|
378
|
+
hub_weight=getattr(args, "hub_weight", 0.3) or 0.3,
|
|
379
|
+
**filt,
|
|
380
|
+
)
|
|
381
|
+
if not results:
|
|
382
|
+
print("No results found.")
|
|
383
|
+
return
|
|
384
|
+
print(f"Found {len(results)} result(s):\n")
|
|
385
|
+
for i, r in enumerate(results, 1):
|
|
386
|
+
print(f" {i}. [{r['relevance_score']:.4f}] {r['title']} ({r['type']})")
|
|
387
|
+
desc = r.get("description") or ""
|
|
388
|
+
if desc:
|
|
389
|
+
print(f" {desc[:120]}")
|
|
390
|
+
if r.get("tags"):
|
|
391
|
+
print(f" tags: {', '.join(r['tags'])}")
|
|
392
|
+
print(f" id: {r['id']}")
|
|
393
|
+
# Print matched chunks if requested
|
|
394
|
+
if r.get("matched_chunks"):
|
|
395
|
+
print(f" matched chunks:")
|
|
396
|
+
for mc in r["matched_chunks"][:3]:
|
|
397
|
+
print(f" [{mc['rrf_score']:.4f}] {mc['block_type']} #{mc['chunk_index']}")
|
|
398
|
+
print(f" {mc.get('chunk_text', '')[:100]}")
|
|
399
|
+
print()
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
def _traverse(args):
|
|
403
|
+
"""Unified traversal: relationships, directory listing, or shortest path."""
|
|
404
|
+
router = _router(args)
|
|
405
|
+
target = getattr(args, "target", None)
|
|
406
|
+
start = getattr(args, "start_id", "") or ""
|
|
407
|
+
if not start:
|
|
408
|
+
items = router.list_directory("")
|
|
409
|
+
if not items:
|
|
410
|
+
print("Directory is empty.")
|
|
411
|
+
return
|
|
412
|
+
print("Contents of '(root)':\n")
|
|
413
|
+
for item in items:
|
|
414
|
+
icon = "[D]" if item["type"] == "Directory" else "[F]"
|
|
415
|
+
print(f" {icon} {item['title']} ({item['type']})")
|
|
416
|
+
print(f" id: {item['id']}")
|
|
417
|
+
return
|
|
418
|
+
if target:
|
|
419
|
+
nodes = router.search_engine.find_path(
|
|
420
|
+
start, target, max_length=getattr(args, "max_path_length", 6)
|
|
421
|
+
)
|
|
422
|
+
if not nodes:
|
|
423
|
+
print(f"No path found between '{start}' and '{target}'.")
|
|
424
|
+
return
|
|
425
|
+
print(f"Path ({len(nodes)} nodes):")
|
|
426
|
+
for i, n in enumerate(nodes, 1):
|
|
427
|
+
print(f" {i}. {n.get('title', '?')} ({n.get('type', '?')})")
|
|
428
|
+
print(f" id: {n['id']}")
|
|
429
|
+
return
|
|
430
|
+
results = router.traverse(
|
|
431
|
+
start_id=start,
|
|
432
|
+
relationship=args.relationship,
|
|
433
|
+
direction=args.direction,
|
|
434
|
+
depth=args.depth,
|
|
435
|
+
node_type=getattr(args, "type", None),
|
|
436
|
+
)
|
|
437
|
+
if not results:
|
|
438
|
+
print("No results found.")
|
|
439
|
+
return
|
|
440
|
+
print(f"Found {len(results)} node(s):\n")
|
|
441
|
+
for r in results:
|
|
442
|
+
print(f" {r['id']} ({r['type']})")
|
|
443
|
+
if r.get("title"):
|
|
444
|
+
print(f" title: {r['title']}")
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
def _read(args):
|
|
448
|
+
"""Unified read: body (default), chunks, document, or context."""
|
|
449
|
+
router = _router(args)
|
|
450
|
+
include = getattr(args, "include", "body") or "body"
|
|
451
|
+
cid = args.concept_id
|
|
452
|
+
max_tokens = getattr(args, "max_tokens", None)
|
|
453
|
+
if max_tokens:
|
|
454
|
+
try:
|
|
455
|
+
reading = router.search_engine.read_with_budget(
|
|
456
|
+
cid, include=include, max_tokens=max_tokens,
|
|
457
|
+
)
|
|
458
|
+
except KeyError:
|
|
459
|
+
print(f"Concept '{cid}' not found.")
|
|
460
|
+
return
|
|
461
|
+
flag = " (truncated)" if reading["truncated"] else ""
|
|
462
|
+
print(f"[{reading['used']}/{reading['budget']} tokens{flag}] {cid}\n")
|
|
463
|
+
for sec in reading["sections"]:
|
|
464
|
+
print(f"## {sec['title'] or sec['id']} [{sec['kind']} | {sec['id']}]")
|
|
465
|
+
print(sec["text"])
|
|
466
|
+
print()
|
|
467
|
+
return
|
|
468
|
+
if include == "chunks":
|
|
469
|
+
chunks = router.search_engine.get_chunks(cid)
|
|
470
|
+
if not chunks:
|
|
471
|
+
print("No chunks found for this concept.")
|
|
472
|
+
return
|
|
473
|
+
print(f"Chunks for '{cid}' ({len(chunks)} total):\n")
|
|
474
|
+
for c in chunks:
|
|
475
|
+
text = c.chunk_text[:120]
|
|
476
|
+
print(f" #{c.chunk_index} [{c.block_type}] {text}")
|
|
477
|
+
return
|
|
478
|
+
if include == "document":
|
|
479
|
+
text = router.embed_engine.reconstruct_document(cid)
|
|
480
|
+
if not text:
|
|
481
|
+
print("No chunks found for this concept.")
|
|
482
|
+
return
|
|
483
|
+
if getattr(args, "output", None):
|
|
484
|
+
Path(args.output).write_text(text, encoding="utf-8")
|
|
485
|
+
print(f"[OK] Reconstructed document written to {args.output}")
|
|
486
|
+
else:
|
|
487
|
+
print(text)
|
|
488
|
+
return
|
|
489
|
+
if include == "context":
|
|
490
|
+
incoming = router.traverse(cid, "LINKS_TO", "INCOMING", 1)[:10]
|
|
491
|
+
outgoing = router.traverse(cid, "LINKS_TO", "OUTGOING", 1)[:10]
|
|
492
|
+
ancestry = router.search_engine._get_ancestry(cid)
|
|
493
|
+
siblings = router.search_engine._get_siblings(cid)[:10]
|
|
494
|
+
if incoming:
|
|
495
|
+
print("Linked by:")
|
|
496
|
+
for l in incoming:
|
|
497
|
+
print(f" {l.get('title', l.get('id', '?'))} (id: {l['id']})")
|
|
498
|
+
if outgoing:
|
|
499
|
+
print("Links to:")
|
|
500
|
+
for l in outgoing:
|
|
501
|
+
print(f" {l.get('title', l.get('id', '?'))} (id: {l['id']})")
|
|
502
|
+
if ancestry:
|
|
503
|
+
print(f"Path: {' → '.join(a['title'] for a in ancestry)}")
|
|
504
|
+
if siblings:
|
|
505
|
+
print("Siblings:")
|
|
506
|
+
for s in siblings:
|
|
507
|
+
print(f" {s['title']} ({s['type']})")
|
|
508
|
+
print(f" id: {s['id']}")
|
|
509
|
+
if not (incoming or outgoing or ancestry or siblings):
|
|
510
|
+
print(f"No context found for '{cid}'.")
|
|
511
|
+
return
|
|
512
|
+
concept = router.get_by_id(cid)
|
|
513
|
+
if not concept:
|
|
514
|
+
print(f"Concept '{cid}' not found.")
|
|
515
|
+
return
|
|
516
|
+
data = concept.public_dict()
|
|
517
|
+
body = data.pop("body", "")
|
|
518
|
+
print(json.dumps(data, indent=2, default=str))
|
|
519
|
+
if body:
|
|
520
|
+
print(f"\n--- BODY ---\n{body}")
|
|
521
|
+
|
|
522
|
+
|
|
523
|
+
def _export(args):
|
|
524
|
+
router = _router(args)
|
|
525
|
+
flavor = getattr(args, "flavor", "okf") or "okf"
|
|
526
|
+
if getattr(args, "export_all", False):
|
|
527
|
+
tags = args.tags.split(",") if args.tags else None
|
|
528
|
+
ids = router.export_mgr.export_bundle(
|
|
529
|
+
output_dir=Path(args.output),
|
|
530
|
+
directory_id=args.parent,
|
|
531
|
+
concept_type=args.type,
|
|
532
|
+
tags=tags,
|
|
533
|
+
flavor=flavor,
|
|
534
|
+
)
|
|
535
|
+
print(f"[OK] Exported {len(ids)} concepts to {args.output} (flavor: {flavor})")
|
|
536
|
+
else:
|
|
537
|
+
cid = args.concept_id
|
|
538
|
+
output_path = Path(args.output) / f"{cid}.md"
|
|
539
|
+
router.export_mgr.export_to_okf(cid, output_path, flavor=flavor)
|
|
540
|
+
print(f"[OK] Exported {cid} → {output_path} (flavor: {flavor})")
|
|
541
|
+
|
|
542
|
+
|
|
543
|
+
def _broken_links(args):
|
|
544
|
+
logger = logging.getLogger("cli")
|
|
545
|
+
router = _router(args)
|
|
546
|
+
broken = router.list_broken_links()
|
|
547
|
+
if not broken:
|
|
548
|
+
logger.info("no broken links found")
|
|
549
|
+
return
|
|
550
|
+
logger.info("found %d broken link(s)", len(broken))
|
|
551
|
+
for link in broken:
|
|
552
|
+
logger.info(" %s → %s", link['source'], link['target'])
|
|
553
|
+
|
|
554
|
+
|
|
555
|
+
def _repair_links(args):
|
|
556
|
+
logger = logging.getLogger("cli")
|
|
557
|
+
router = _router(args)
|
|
558
|
+
count = router.repair_links()
|
|
559
|
+
logger.info("repaired %d link(s)", count)
|
|
560
|
+
|
|
561
|
+
|
|
562
|
+
def _lint(args):
|
|
563
|
+
"""Pre-import bundle gate: frontmatter + link validation.
|
|
564
|
+
|
|
565
|
+
Deliberately router-free (no DB, no ~30s model cold-boot): lint answers
|
|
566
|
+
"is this bundle well-formed?" before an import cycle is spent.
|
|
567
|
+
Exit 0 = clean (warnings ok), 1 = errors, 2 = usage (bad dir).
|
|
568
|
+
"""
|
|
569
|
+
from okfgraph.components.lint import lint_bundle
|
|
570
|
+
from okfgraph.config import OKFConfig
|
|
571
|
+
|
|
572
|
+
given = getattr(args, "dir", None) or getattr(args, "bundle", None) or "."
|
|
573
|
+
# bundle_root is only the TOML lookup location; the value itself rides
|
|
574
|
+
# in cli_args (same split as _router's cli_dict).
|
|
575
|
+
config = OKFConfig.load(bundle_root=given, cli_args={"bundle": given})
|
|
576
|
+
target = Path(str(config.bundle))
|
|
577
|
+
if not target.is_dir():
|
|
578
|
+
print(f"[ERROR] not a bundle directory: {target}")
|
|
579
|
+
return 2
|
|
580
|
+
report = lint_bundle(target)
|
|
581
|
+
if getattr(args, "json", False):
|
|
582
|
+
print(json.dumps(report, indent=2, default=str))
|
|
583
|
+
else:
|
|
584
|
+
print(f"{report['files']} file(s): "
|
|
585
|
+
f"{len(report['errors'])} error(s), "
|
|
586
|
+
f"{len(report['warnings'])} warning(s)")
|
|
587
|
+
for e in report["errors"]:
|
|
588
|
+
print(f" [ERROR] {e['file']} {e['rule']}: {e['message']}")
|
|
589
|
+
for w in report["warnings"]:
|
|
590
|
+
print(f" [warn] {w['file']} {w['rule']}: {w['message']}")
|
|
591
|
+
if report["clean"]:
|
|
592
|
+
print("Bundle is lint-clean (safe to import).")
|
|
593
|
+
return 0 if report["clean"] else 1
|
|
594
|
+
|
|
595
|
+
|
|
596
|
+
def _diff(args):
|
|
597
|
+
"""Structural diff: snapshot (dir vs dir) or drift (graph vs dir).
|
|
598
|
+
|
|
599
|
+
Returns an exit code (0 = identical, 1 = different) for CI gating;
|
|
600
|
+
main() propagates int returns to sys.exit.
|
|
601
|
+
"""
|
|
602
|
+
from okfgraph.components.diff import DiffManager
|
|
603
|
+
|
|
604
|
+
old = getattr(args, "old", None)
|
|
605
|
+
new = getattr(args, "new", None)
|
|
606
|
+
as_json = getattr(args, "json", False)
|
|
607
|
+
if old and new and Path(old).is_dir() and Path(new).is_dir():
|
|
608
|
+
# Snapshot mode needs no database (and no model load).
|
|
609
|
+
result = DiffManager(None).diff_dirs(Path(old), Path(new))
|
|
610
|
+
elif old and new:
|
|
611
|
+
print("[ERROR] diff needs two bundle directories (or one + --db/--bundle)")
|
|
612
|
+
return 2
|
|
613
|
+
else:
|
|
614
|
+
router = _router(args)
|
|
615
|
+
side = Path(old or new) if (old or new) else router.bundle_root
|
|
616
|
+
if not side.is_dir():
|
|
617
|
+
print(f"[ERROR] not a bundle directory: {side}")
|
|
618
|
+
return 2
|
|
619
|
+
result = router.diff_db_dir(side)
|
|
620
|
+
if as_json:
|
|
621
|
+
print(json.dumps(result, indent=2, default=str))
|
|
622
|
+
else:
|
|
623
|
+
_print_diff(result)
|
|
624
|
+
return 0 if result["identical"] else 1
|
|
625
|
+
|
|
626
|
+
|
|
627
|
+
def _print_diff(result) -> None:
|
|
628
|
+
"""Human-readable rendering of a structural diff report."""
|
|
629
|
+
if result["identical"]:
|
|
630
|
+
print("No structural differences.")
|
|
631
|
+
return
|
|
632
|
+
for cid in result["added"]:
|
|
633
|
+
print(f" + concept {cid}")
|
|
634
|
+
for cid in result["removed"]:
|
|
635
|
+
print(f" - concept {cid}")
|
|
636
|
+
for cid in result["changed"]:
|
|
637
|
+
print(f" ~ body {cid}")
|
|
638
|
+
for r in result["retitled"]:
|
|
639
|
+
print(f" ~ title {r['id']}: {r['old']!r} -> {r['new']!r}")
|
|
640
|
+
for r in result["retyped"]:
|
|
641
|
+
print(f" ~ type {r['id']}: {r['old']!r} -> {r['new']!r}")
|
|
642
|
+
for s, t in result["edges_added"]:
|
|
643
|
+
print(f" + edge {s} -> {t}")
|
|
644
|
+
for s, t in result["edges_removed"]:
|
|
645
|
+
print(f" - edge {s} -> {t}")
|
|
646
|
+
for s, t in result["broken_new"]:
|
|
647
|
+
print(f" + broken {s} -> {t}")
|
|
648
|
+
for s, t in result["broken_fixed"]:
|
|
649
|
+
print(f" - broken {s} -> {t}")
|
|
650
|
+
|
|
651
|
+
|
|
652
|
+
def _doctor(args):
|
|
653
|
+
"""Scored health scan, optionally with safe --fix repairs.
|
|
654
|
+
|
|
655
|
+
Returns an exit code with --strict (1 when any finding exists).
|
|
656
|
+
"""
|
|
657
|
+
router = _router(args)
|
|
658
|
+
if getattr(args, "fix", False):
|
|
659
|
+
fixed = router.doctor_fix()
|
|
660
|
+
print(f"[OK] repaired {fixed['repaired_links']} link(s), "
|
|
661
|
+
f"normalized {fixed['normalized_timestamps']} timestamp(s)")
|
|
662
|
+
if fixed["skipped_reviewed"]:
|
|
663
|
+
print(f" skipped reviewed: {', '.join(fixed['skipped_reviewed'])}")
|
|
664
|
+
report = router.diagnose(
|
|
665
|
+
stale_days=getattr(args, "stale_days", 365) or 365,
|
|
666
|
+
)
|
|
667
|
+
if getattr(args, "json", False):
|
|
668
|
+
print(json.dumps(report, indent=2, default=str))
|
|
669
|
+
else:
|
|
670
|
+
print(f"Health score: {report['score']}/100 ({report['concepts']} concepts)")
|
|
671
|
+
for f in report["findings"]:
|
|
672
|
+
print(f" [{f['severity']}] {f['rule']} {f['path']}: {f['message']}")
|
|
673
|
+
for i in report["info"]:
|
|
674
|
+
print(f" (info) {i['rule']}: {i['message']}")
|
|
675
|
+
if getattr(args, "strict", False) and report["findings"]:
|
|
676
|
+
return 1
|
|
677
|
+
return 0
|
|
678
|
+
|
|
679
|
+
|
|
680
|
+
def _reindex(args):
|
|
681
|
+
logger = logging.getLogger("cli")
|
|
682
|
+
router = _router(args)
|
|
683
|
+
ran = router.schema_mgr.reindex(force=not getattr(args, "if_dirty", False))
|
|
684
|
+
if ran:
|
|
685
|
+
logger.info("search indexes rebuilt")
|
|
686
|
+
else:
|
|
687
|
+
logger.info("search indexes already up to date; nothing to do.")
|
|
688
|
+
|
|
689
|
+
|
|
690
|
+
def _ingest(args):
|
|
691
|
+
"""Unified ingest: markdown file, PDF (default), or raw thoughts.
|
|
692
|
+
|
|
693
|
+
Delegates to IngestManager.ingest_md / ingest_pdf / ingest_thoughts.
|
|
694
|
+
PDF conversion uses the configured DocumentConverter (bobine default).
|
|
695
|
+
"""
|
|
696
|
+
logger = logging.getLogger("cli")
|
|
697
|
+
router = _router(args)
|
|
698
|
+
kind = getattr(args, "kind", "pdf") or "pdf"
|
|
699
|
+
tags = args.tags.split(",") if getattr(args, "tags", None) else None
|
|
700
|
+
if kind == "md":
|
|
701
|
+
md_file = getattr(args, "md_file", None)
|
|
702
|
+
if not md_file:
|
|
703
|
+
print("[ERROR] --md-file is required for --kind md")
|
|
704
|
+
return
|
|
705
|
+
result = router.ingest_mgr.ingest_md(
|
|
706
|
+
md_path=md_file,
|
|
707
|
+
concept_id=getattr(args, "concept_id", None),
|
|
708
|
+
title=getattr(args, "title", None),
|
|
709
|
+
description=getattr(args, "description", None),
|
|
710
|
+
tags=tags,
|
|
711
|
+
mode=getattr(args, "mode", "text") or "text",
|
|
712
|
+
)
|
|
713
|
+
print(f"[OK] Imported {result['concept_id']} ({result['chunk_count']} chunks)")
|
|
714
|
+
return
|
|
715
|
+
if kind == "thoughts":
|
|
716
|
+
if not getattr(args, "thoughts", None) or not getattr(args, "topic", None):
|
|
717
|
+
print("[ERROR] --thoughts and --topic are required for --kind thoughts")
|
|
718
|
+
return
|
|
719
|
+
result = router.ingest_mgr.ingest_thoughts(
|
|
720
|
+
args.thoughts,
|
|
721
|
+
topic=args.topic,
|
|
722
|
+
concept_id=getattr(args, "concept_id", None),
|
|
723
|
+
tags=tags,
|
|
724
|
+
)
|
|
725
|
+
print(f"[OK] Stored thought {result['concept_id']}")
|
|
726
|
+
return
|
|
727
|
+
pdf_path = Path(getattr(args, "pdf_file", None) or "")
|
|
728
|
+
if not pdf_path.name or not pdf_path.exists():
|
|
729
|
+
print(f"[ERROR] File not found: {pdf_path}")
|
|
730
|
+
return
|
|
731
|
+
|
|
732
|
+
auto_import = getattr(args, "auto_import", False)
|
|
733
|
+
output_dir = getattr(args, "output", None)
|
|
734
|
+
if not auto_import and not output_dir:
|
|
735
|
+
output_dir = Path(".")
|
|
736
|
+
|
|
737
|
+
def on_page(idx, total):
|
|
738
|
+
print(f" page {idx + 1}/{total}", end="\r")
|
|
739
|
+
|
|
740
|
+
try:
|
|
741
|
+
from okfgraph.components.converters import BobineConverter
|
|
742
|
+
converter = BobineConverter(
|
|
743
|
+
routing_mode=getattr(args, "routing_mode", "auto") or "auto",
|
|
744
|
+
extract_images=not getattr(args, "no_extract_images", False),
|
|
745
|
+
)
|
|
746
|
+
result = router.ingest_mgr.ingest_pdf(
|
|
747
|
+
pdf_path,
|
|
748
|
+
auto_import=auto_import,
|
|
749
|
+
output_dir=output_dir,
|
|
750
|
+
mode=getattr(args, "mode", "text") or "text",
|
|
751
|
+
batch_size=getattr(args, "batch_size", 32) or 32,
|
|
752
|
+
purge_deleted=getattr(args, "purge", False),
|
|
753
|
+
on_page=on_page,
|
|
754
|
+
converter=converter,
|
|
755
|
+
)
|
|
756
|
+
except RuntimeError as e:
|
|
757
|
+
print(f"[ERROR] {e}")
|
|
758
|
+
return
|
|
759
|
+
print() # newline after progress
|
|
760
|
+
|
|
761
|
+
logger.info("written %s", result["md_path"])
|
|
762
|
+
logger.info("assets in %s", result["image_dir"])
|
|
763
|
+
if auto_import:
|
|
764
|
+
for cid in result["concept_ids"]:
|
|
765
|
+
n = len(router.image_mgr.list_images(cid))
|
|
766
|
+
suffix = f" [{n} image(s)]" if n else ""
|
|
767
|
+
logger.info(" %s%s", cid, suffix)
|
|
768
|
+
else:
|
|
769
|
+
logger.info("run 'okf import --all --bundle %s' to import.", output_dir)
|
|
770
|
+
|
|
771
|
+
|
|
772
|
+
def _deleted_list(args):
|
|
773
|
+
"""List soft-deleted concepts with recovery status."""
|
|
774
|
+
router = _router(args)
|
|
775
|
+
deleted = router.purge_mgr.list_deleted_concepts()
|
|
776
|
+
if not deleted:
|
|
777
|
+
print("No soft-deleted concepts found.")
|
|
778
|
+
return
|
|
779
|
+
print(f"Soft-deleted concepts ({len(deleted)} total):\n")
|
|
780
|
+
for d in deleted:
|
|
781
|
+
status = "recoverable" if d["recoverable"] else "expired"
|
|
782
|
+
print(f" [{status}] {d['concept_id']}")
|
|
783
|
+
print(f" title: {d['title']}")
|
|
784
|
+
print(f" type: {d['type']}")
|
|
785
|
+
print(f" deleted: {d['deleted_at']} ({d['age_seconds']:.0f}s ago)")
|
|
786
|
+
print()
|
|
787
|
+
|
|
788
|
+
|
|
789
|
+
def _deleted_recover(args):
|
|
790
|
+
"""Recover a soft-deleted concept."""
|
|
791
|
+
router = _router(args)
|
|
792
|
+
success = router.purge_mgr._recover_concept(args.concept_id)
|
|
793
|
+
if success:
|
|
794
|
+
print(f"[OK] Recovered concept '{args.concept_id}'.")
|
|
795
|
+
else:
|
|
796
|
+
print(f"[ERROR] Concept '{args.concept_id}' not found or past recovery window.")
|
|
797
|
+
|
|
798
|
+
|
|
799
|
+
def _deleted_purge(args):
|
|
800
|
+
"""Permanently delete expired soft-deleted concepts."""
|
|
801
|
+
router = _router(args)
|
|
802
|
+
older_than = getattr(args, "older_than", None)
|
|
803
|
+
count = router.purge_mgr.purge_deleted_concepts(older_than=older_than)
|
|
804
|
+
print(f"[OK] Permanently deleted {count} expired concept(s).")
|
|
805
|
+
|
|
806
|
+
|
|
807
|
+
def _shell(args):
|
|
808
|
+
router = _router(args)
|
|
809
|
+
banner = """OKF Interactive Shell
|
|
810
|
+
========================================
|
|
811
|
+
Commands:
|
|
812
|
+
import <file> [mode] — import single OKF file (mode: text|optional|omni)
|
|
813
|
+
import-bundle [path] [mode]— import entire bundle (mode: text|optional|omni)
|
|
814
|
+
search [target:]<query> — search concepts (default), chunks:, images:
|
|
815
|
+
search <query> expand — chunk hits + graph neighborhood
|
|
816
|
+
search <query> hub — chunk hits reranked by hub score
|
|
817
|
+
read <id> [chunks|document|context] — read a concept (default: body)
|
|
818
|
+
traverse [id] [rel] [dir] [depth] — traverse (no id = root listing)
|
|
819
|
+
traverse <id1> <id2> — shortest path between two concepts
|
|
820
|
+
images <concept_id> — list images attached to a concept
|
|
821
|
+
export-bundle <output_dir> — export all concepts
|
|
822
|
+
export <id> <output_dir> — export single concept
|
|
823
|
+
ingest <file> [--auto-import] — ingest .md or .pdf (auto-import PDFs)
|
|
824
|
+
model-info — show model cache status
|
|
825
|
+
help — show this help
|
|
826
|
+
quit / exit — exit shell
|
|
827
|
+
========================================"""
|
|
828
|
+
|
|
829
|
+
print(banner)
|
|
830
|
+
|
|
831
|
+
while True:
|
|
832
|
+
try:
|
|
833
|
+
line = input("\n> ").strip()
|
|
834
|
+
except (EOFError, KeyboardInterrupt):
|
|
835
|
+
print("\nBye.")
|
|
836
|
+
break
|
|
837
|
+
|
|
838
|
+
if not line:
|
|
839
|
+
continue
|
|
840
|
+
|
|
841
|
+
parts = line.split(None, 1)
|
|
842
|
+
cmd = parts[0].lower()
|
|
843
|
+
rest = parts[1] if len(parts) > 1 else ""
|
|
844
|
+
|
|
845
|
+
if cmd in ("quit", "exit", "q"):
|
|
846
|
+
print("Bye.")
|
|
847
|
+
break
|
|
848
|
+
|
|
849
|
+
elif cmd == "help":
|
|
850
|
+
print(banner)
|
|
851
|
+
|
|
852
|
+
elif cmd == "import" and rest:
|
|
853
|
+
tokens = rest.strip().split()
|
|
854
|
+
mode = "text"
|
|
855
|
+
if tokens and tokens[-1].lower() in ("text", "optional", "omni"):
|
|
856
|
+
mode = tokens[-1].lower()
|
|
857
|
+
tokens = tokens[:-1]
|
|
858
|
+
fp = Path(" ".join(tokens))
|
|
859
|
+
if not fp.exists():
|
|
860
|
+
print(f"Error: {fp} not found")
|
|
861
|
+
continue
|
|
862
|
+
cid = router.import_from_okf(fp, mode=mode)
|
|
863
|
+
imgs = router.image_mgr.list_images(cid)
|
|
864
|
+
suffix = f" ({len(imgs)} image(s), mode: {mode})" if imgs else ""
|
|
865
|
+
print(f"[OK] Imported: {cid}{suffix}")
|
|
866
|
+
|
|
867
|
+
elif cmd == "import-bundle":
|
|
868
|
+
tokens = rest.strip().split()
|
|
869
|
+
mode = "text"
|
|
870
|
+
if tokens and tokens[-1].lower() in ("text", "optional", "omni"):
|
|
871
|
+
mode = tokens[-1].lower()
|
|
872
|
+
tokens = tokens[:-1]
|
|
873
|
+
bundle_path = Path(" ".join(tokens)) if tokens else None
|
|
874
|
+
ids = router.import_mgr.import_bundle(bundle_path, mode=mode)
|
|
875
|
+
print(f"[OK] Imported {len(ids)} concepts (image mode: {mode})")
|
|
876
|
+
|
|
877
|
+
elif cmd == "search" and rest:
|
|
878
|
+
tokens = rest.strip().split()
|
|
879
|
+
first = tokens[0]
|
|
880
|
+
target = "concepts"
|
|
881
|
+
if ":" in first and first.split(":")[0] in ("concepts", "chunks", "images"):
|
|
882
|
+
target, first = first.split(":", 1)
|
|
883
|
+
tokens[0] = first
|
|
884
|
+
query = tokens[0]
|
|
885
|
+
expand = "expand" in tokens[1:]
|
|
886
|
+
hub = "hub" in tokens[1:]
|
|
887
|
+
type_filter = tags_filter = parent_filter = None
|
|
888
|
+
for t in tokens[1:]:
|
|
889
|
+
if t.startswith("type:"):
|
|
890
|
+
type_filter = t[5:]
|
|
891
|
+
elif t.startswith("tags:"):
|
|
892
|
+
tags_filter = t[5:].split(",")
|
|
893
|
+
elif t.startswith("parent:"):
|
|
894
|
+
parent_filter = t[7:]
|
|
895
|
+
if target == "images":
|
|
896
|
+
results = router.image_mgr.search_images_with_text(query)
|
|
897
|
+
for i, r in enumerate(results, 1):
|
|
898
|
+
label = r.get("alt_text") or r.get("file_name") or r.get("id")
|
|
899
|
+
print(f" {i}. [{r['relevance_score']:.4f}] {label} ({r.get('embed_route')})")
|
|
900
|
+
print(f" id: {r['id']}")
|
|
901
|
+
elif target == "chunks":
|
|
902
|
+
if hub:
|
|
903
|
+
results = router.search_engine.search_chunks_with_hub_score(query)
|
|
904
|
+
elif expand:
|
|
905
|
+
results = router.search_engine.search_with_context(query)
|
|
906
|
+
else:
|
|
907
|
+
results = router.search_engine.search_chunks(query)
|
|
908
|
+
for i, r in enumerate(results, 1):
|
|
909
|
+
chunk = r.get("chunk", r)
|
|
910
|
+
score = r.get("final_score", chunk.get("rrf_score", 0))
|
|
911
|
+
print(f" {i}. [{score:.4f}] {chunk.get('parent_title', '?')} §{chunk.get('chunk_index', '?')}")
|
|
912
|
+
print(f" {chunk.get('chunk_text', '')[:150]}")
|
|
913
|
+
else:
|
|
914
|
+
results = router.search_hybrid(
|
|
915
|
+
query=query, concept_type=type_filter,
|
|
916
|
+
tags=tags_filter, parent_id=parent_filter,
|
|
917
|
+
)
|
|
918
|
+
for i, r in enumerate(results, 1):
|
|
919
|
+
print(f" {i}. [{r['relevance_score']:.4f}] {r['title']} ({r['type']})")
|
|
920
|
+
desc = r.get("description") or ""
|
|
921
|
+
if desc:
|
|
922
|
+
print(f" {desc[:120]}")
|
|
923
|
+
|
|
924
|
+
elif cmd == "read" and rest:
|
|
925
|
+
tokens = rest.strip().split()
|
|
926
|
+
cid = tokens[0]
|
|
927
|
+
include = tokens[1] if len(tokens) > 1 else "body"
|
|
928
|
+
if include == "chunks":
|
|
929
|
+
chunks = router.search_engine.get_chunks(cid)
|
|
930
|
+
if not chunks:
|
|
931
|
+
print("No chunks found.")
|
|
932
|
+
for c in chunks:
|
|
933
|
+
print(f" #{c.chunk_index} [{c.block_type}] {c.chunk_text[:120]}")
|
|
934
|
+
elif include == "document":
|
|
935
|
+
text = router.embed_engine.reconstruct_document(cid)
|
|
936
|
+
print(text if text else "No chunks found for this concept.")
|
|
937
|
+
elif include == "context":
|
|
938
|
+
for l in router.traverse(cid, "LINKS_TO", "INCOMING", 1)[:10]:
|
|
939
|
+
print(f" ← {l.get('title', l.get('id', '?'))}")
|
|
940
|
+
for l in router.traverse(cid, "LINKS_TO", "OUTGOING", 1)[:10]:
|
|
941
|
+
print(f" → {l.get('title', l.get('id', '?'))}")
|
|
942
|
+
for a in router.search_engine._get_ancestry(cid):
|
|
943
|
+
print(f" ↑ {a['title']}")
|
|
944
|
+
else:
|
|
945
|
+
concept = router.get_by_id(cid)
|
|
946
|
+
if concept:
|
|
947
|
+
data = concept.public_dict()
|
|
948
|
+
body = data.pop("body", "")
|
|
949
|
+
print(json.dumps(data, indent=2, default=str))
|
|
950
|
+
if body:
|
|
951
|
+
print(f"\n--- BODY ---\n{body}")
|
|
952
|
+
else:
|
|
953
|
+
print(f"Concept '{cid}' not found")
|
|
954
|
+
|
|
955
|
+
elif cmd == "traverse":
|
|
956
|
+
tokens = rest.strip().split()
|
|
957
|
+
if not tokens:
|
|
958
|
+
for item in router.list_directory(""):
|
|
959
|
+
icon = "[D]" if item["type"] == "Directory" else "[F]"
|
|
960
|
+
print(f" {icon} {item['title']} ({item['type']})")
|
|
961
|
+
elif len(tokens) == 2 and tokens[1] not in ("CONTAINS", "LINKS_TO", "PART_OF", "INCLUDES_ASSET"):
|
|
962
|
+
nodes = router.search_engine.find_path(tokens[0], tokens[1])
|
|
963
|
+
if not nodes:
|
|
964
|
+
print(f"No path found between '{tokens[0]}' and '{tokens[1]}'.")
|
|
965
|
+
else:
|
|
966
|
+
print(f"Path ({len(nodes)} nodes):")
|
|
967
|
+
for i, n in enumerate(nodes, 1):
|
|
968
|
+
print(f" {i}. {n.get('title', '?')} ({n.get('type', '?')})")
|
|
969
|
+
print(f" id: {n['id']}")
|
|
970
|
+
else:
|
|
971
|
+
start_id = tokens[0]
|
|
972
|
+
rel = tokens[1] if len(tokens) > 1 else "CONTAINS"
|
|
973
|
+
direction = tokens[2] if len(tokens) > 2 else "OUTGOING"
|
|
974
|
+
depth = int(tokens[3]) if len(tokens) > 3 else 1
|
|
975
|
+
results = router.traverse(start_id, rel, direction, depth)
|
|
976
|
+
for r in results:
|
|
977
|
+
print(f" {r['id']} ({r['type']}) — {r.get('title', '')}")
|
|
978
|
+
|
|
979
|
+
elif cmd == "images" and rest:
|
|
980
|
+
imgs = router.image_mgr.list_images(rest.strip())
|
|
981
|
+
if not imgs:
|
|
982
|
+
print("No images attached.")
|
|
983
|
+
for im in imgs:
|
|
984
|
+
alt = im.get("alt_text") or "(no alt-text)"
|
|
985
|
+
print(f" [{im.get('embed_route')}] {im.get('file_name')} — {alt}")
|
|
986
|
+
print(f" id: {im.get('id')}")
|
|
987
|
+
|
|
988
|
+
elif cmd == "export-bundle" and rest:
|
|
989
|
+
ids = router.export_mgr.export_bundle(Path(rest.strip()))
|
|
990
|
+
print(f"[OK] Exported {len(ids)} concepts to {rest.strip()}")
|
|
991
|
+
|
|
992
|
+
elif cmd == "export" and rest:
|
|
993
|
+
tokens = rest.strip().split(None, 1)
|
|
994
|
+
if len(tokens) == 2:
|
|
995
|
+
cid, out_dir = tokens
|
|
996
|
+
output_path = Path(out_dir) / f"{cid}.md"
|
|
997
|
+
router.export_to_okf(cid, output_path)
|
|
998
|
+
print(f"[OK] Exported {cid} → {output_path}")
|
|
999
|
+
else:
|
|
1000
|
+
print("Usage: export <concept_id> <output_dir>")
|
|
1001
|
+
|
|
1002
|
+
elif cmd == "model-info":
|
|
1003
|
+
info = OKFRouter.model_info(cache_dir=router.cache_dir)
|
|
1004
|
+
print(f"Model: {info['model_id']}")
|
|
1005
|
+
print(f"Cache: {info['cache_dir']}")
|
|
1006
|
+
if info["cached"]:
|
|
1007
|
+
size_gb = info["disk_usage_bytes"] / (1024 ** 3)
|
|
1008
|
+
print(f"Status: cached ({size_gb:.2f} GB)")
|
|
1009
|
+
print(f"Path: {info['snapshot_path']}")
|
|
1010
|
+
else:
|
|
1011
|
+
print("Status: not cached (will download on first use)")
|
|
1012
|
+
|
|
1013
|
+
elif cmd == "broken-links":
|
|
1014
|
+
broken = router.list_broken_links()
|
|
1015
|
+
if not broken:
|
|
1016
|
+
print("No broken links found.")
|
|
1017
|
+
else:
|
|
1018
|
+
print(f"Found {len(broken)} broken link(s):")
|
|
1019
|
+
for link in broken:
|
|
1020
|
+
print(f" {link['source']} → {link['target']}")
|
|
1021
|
+
|
|
1022
|
+
elif cmd == "repair-links":
|
|
1023
|
+
count = router.repair_links()
|
|
1024
|
+
print(f"[OK] Repaired {count} link(s)")
|
|
1025
|
+
|
|
1026
|
+
elif cmd == "ingest" and rest:
|
|
1027
|
+
# Minimal shell dispatch for ingest — delegates to the CLI handler.
|
|
1028
|
+
from okfgraph.cli import _ingest
|
|
1029
|
+
from types import SimpleNamespace
|
|
1030
|
+
src_path = rest.strip()
|
|
1031
|
+
is_md = src_path.lower().endswith(".md")
|
|
1032
|
+
shell_args = SimpleNamespace(
|
|
1033
|
+
kind="md" if is_md else "pdf",
|
|
1034
|
+
md_file=src_path if is_md else None,
|
|
1035
|
+
pdf_file=None if is_md else src_path,
|
|
1036
|
+
thoughts=None,
|
|
1037
|
+
topic=None,
|
|
1038
|
+
concept_id=None,
|
|
1039
|
+
title=None,
|
|
1040
|
+
description=None,
|
|
1041
|
+
tags=None,
|
|
1042
|
+
auto_import=False,
|
|
1043
|
+
output=None,
|
|
1044
|
+
routing_mode="auto",
|
|
1045
|
+
mode="text",
|
|
1046
|
+
batch_size=32,
|
|
1047
|
+
purge=False,
|
|
1048
|
+
no_extract_images=False,
|
|
1049
|
+
db=args.db,
|
|
1050
|
+
bundle=args.bundle,
|
|
1051
|
+
dim=args.dim,
|
|
1052
|
+
cache_dir=getattr(args, "cache_dir", None),
|
|
1053
|
+
device=getattr(args, "device", "cpu"),
|
|
1054
|
+
omni_model_id=getattr(args, "omni_model_id", None),
|
|
1055
|
+
chunk_size=getattr(args, "chunk_size", 512),
|
|
1056
|
+
chunk_overlap=getattr(args, "chunk_overlap", 40),
|
|
1057
|
+
no_chunking=False,
|
|
1058
|
+
allow_remote_images=False,
|
|
1059
|
+
)
|
|
1060
|
+
_ingest(shell_args)
|
|
1061
|
+
|
|
1062
|
+
else:
|
|
1063
|
+
print(f"Unknown command: {cmd}. Type 'help' for usage.")
|
|
1064
|
+
|
|
1065
|
+
|
|
1066
|
+
# ── argument parser ────────────────────────────────────────────────────────
|
|
1067
|
+
|
|
1068
|
+
def _global_options_epilog() -> str:
|
|
1069
|
+
"""Render the global flags once for top-level ``okf --help``.
|
|
1070
|
+
|
|
1071
|
+
Single source of truth: builds a throwaway parser with the same helpers
|
|
1072
|
+
(unmarked, so nothing is hidden) and reuses its options section.
|
|
1073
|
+
"""
|
|
1074
|
+
probe = argparse.ArgumentParser(prog="okf")
|
|
1075
|
+
_add_global(probe, mark=False)
|
|
1076
|
+
_add_logging_flags(probe, mark=False)
|
|
1077
|
+
text = probe.format_help()
|
|
1078
|
+
try:
|
|
1079
|
+
body = text.split("options:", 1)[1]
|
|
1080
|
+
except IndexError: # pragma: no cover - Python <3.11 wording
|
|
1081
|
+
body = text.split("optional arguments:", 1)[1]
|
|
1082
|
+
return "Global options (every command; may also come from okfgraph.toml):" + body
|
|
1083
|
+
|
|
1084
|
+
|
|
1085
|
+
def build_parser():
|
|
1086
|
+
parser = argparse.ArgumentParser(
|
|
1087
|
+
prog="okf",
|
|
1088
|
+
description="OKF Knowledge Graph CLI — LadybugDB + Jina v5 embeddings",
|
|
1089
|
+
epilog=_global_options_epilog(),
|
|
1090
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
1091
|
+
)
|
|
1092
|
+
sub = parser.add_subparsers(dest="command", help="Command to run", parser_class=_SubParser)
|
|
1093
|
+
|
|
1094
|
+
# init
|
|
1095
|
+
p = sub.add_parser("init", help="Initialize database and schema")
|
|
1096
|
+
_add_global(p)
|
|
1097
|
+
_add_logging_flags(p)
|
|
1098
|
+
|
|
1099
|
+
# model-info
|
|
1100
|
+
p = sub.add_parser("model-info", help="Show model cache status")
|
|
1101
|
+
_add_global(p)
|
|
1102
|
+
_add_logging_flags(p)
|
|
1103
|
+
p.add_argument("--model-id", default="jinaai/jina-embeddings-v5-text-small-retrieval", help="Model ID to inspect")
|
|
1104
|
+
|
|
1105
|
+
# import
|
|
1106
|
+
p = sub.add_parser("import", help="Import OKF files")
|
|
1107
|
+
_add_global(p)
|
|
1108
|
+
_add_logging_flags(p)
|
|
1109
|
+
p.add_argument("files", nargs="*", help="Files to import")
|
|
1110
|
+
p.add_argument("--all", action="store_true", dest="import_all", help="Import entire bundle")
|
|
1111
|
+
p.add_argument("--batch-size", type=int, default=32, help="Batch size for encoding (default: 32)")
|
|
1112
|
+
p.add_argument(
|
|
1113
|
+
"--mode", default="text", choices=["text", "optional", "omni"],
|
|
1114
|
+
help="Image ingestion mode: text (alt-text/filename, no omni), "
|
|
1115
|
+
"optional (omni only for images lacking alt-text), "
|
|
1116
|
+
"omni (omni for every image). Default: text",
|
|
1117
|
+
)
|
|
1118
|
+
p.add_argument(
|
|
1119
|
+
"--purge", action="store_true", default=False,
|
|
1120
|
+
help="Also purge concepts whose source files were deleted from disk "
|
|
1121
|
+
"(removes concept, chunks, links, and orphaned image assets)",
|
|
1122
|
+
)
|
|
1123
|
+
|
|
1124
|
+
# search (unified: concepts, chunks, images)
|
|
1125
|
+
p = sub.add_parser("search", help="Search concepts, chunks, or images")
|
|
1126
|
+
_add_global(p)
|
|
1127
|
+
_add_logging_flags(p)
|
|
1128
|
+
p.add_argument("query", help="Search query")
|
|
1129
|
+
p.add_argument("--target", default="concepts", choices=["concepts", "chunks", "images"],
|
|
1130
|
+
help="What to search (default: concepts)")
|
|
1131
|
+
p.add_argument("--limit", type=int, default=10, help="Max results (default: 10)")
|
|
1132
|
+
p.add_argument("--type", help="Concept type filter (concepts/chunks)")
|
|
1133
|
+
p.add_argument("--tags", help="Comma-separated tag filters (concepts/chunks)")
|
|
1134
|
+
p.add_argument("--parent", help="Parent directory ID (concepts/chunks)")
|
|
1135
|
+
p.add_argument("--chunks", action="store_true", help="Include matched chunks per concept result")
|
|
1136
|
+
p.add_argument("--expand", action="store_true", help="Chunks: attach graph neighborhood to each hit")
|
|
1137
|
+
p.add_argument("--context-hops", type=int, default=1, help="Expansion hops with --expand (default: 1)")
|
|
1138
|
+
p.add_argument("--hub-rerank", action="store_true", help="Chunks: rerank by graph hub score")
|
|
1139
|
+
p.add_argument("--hub-weight", type=float, default=0.3, help="Hub weight with --hub-rerank (default: 0.3)")
|
|
1140
|
+
p.add_argument("--use-omni", action="store_true", help="Images: encode query with omni text side")
|
|
1141
|
+
p.add_argument("--rank", default="none", choices=["none", "hub", "ppr"],
|
|
1142
|
+
help="Concepts: ranking — none (RRF order), hub (blend incoming-link "
|
|
1143
|
+
"authority), ppr (model-free lexical-seed PPR, no ONNX load). "
|
|
1144
|
+
"Default: none")
|
|
1145
|
+
|
|
1146
|
+
# read (unified: body, chunks, document, context)
|
|
1147
|
+
p = sub.add_parser("read", help="Read a concept: body, chunks, document, or context")
|
|
1148
|
+
_add_global(p)
|
|
1149
|
+
_add_logging_flags(p)
|
|
1150
|
+
p.add_argument("concept_id", help="Concept ID")
|
|
1151
|
+
p.add_argument("--include", default="body", choices=["body", "chunks", "document", "context"],
|
|
1152
|
+
help="What to return (default: body)")
|
|
1153
|
+
p.add_argument("--output", help="Output file for --include document (default: stdout)")
|
|
1154
|
+
p.add_argument("--max-tokens", type=int, default=None,
|
|
1155
|
+
help="Token budget: assemble self + PPR-ranked neighbours "
|
|
1156
|
+
"(index-first for context), truncating to fit")
|
|
1157
|
+
|
|
1158
|
+
# traverse (unified: relationships, directory listing, shortest path)
|
|
1159
|
+
p = sub.add_parser("traverse", help="Traverse relationships, list directories, find paths")
|
|
1160
|
+
_add_global(p)
|
|
1161
|
+
_add_logging_flags(p)
|
|
1162
|
+
p.add_argument("start_id", nargs="?", default="", help="Starting concept or directory ID (empty = root listing)")
|
|
1163
|
+
p.add_argument("--relationship", default="CONTAINS", choices=["CONTAINS", "LINKS_TO", "PART_OF", "INCLUDES_ASSET"])
|
|
1164
|
+
p.add_argument("--direction", default="OUTGOING", choices=["OUTGOING", "INCOMING", "BOTH"])
|
|
1165
|
+
p.add_argument("--depth", type=int, default=1, help="Max depth (1-5)")
|
|
1166
|
+
p.add_argument("--type", help="Target node type filter")
|
|
1167
|
+
p.add_argument("--target", default=None, help="Find shortest path from start_id to this ID instead of traversing")
|
|
1168
|
+
p.add_argument("--max-path-length", type=int, default=6, help="Max path length with --target (default: 6)")
|
|
1169
|
+
|
|
1170
|
+
# ingest (unified: markdown, PDF, thoughts)
|
|
1171
|
+
p = sub.add_parser("ingest", help="Add content: markdown file, PDF, or thoughts")
|
|
1172
|
+
_add_global(p)
|
|
1173
|
+
_add_logging_flags(p)
|
|
1174
|
+
p.add_argument("--kind", default="pdf", choices=["md", "pdf", "thoughts"],
|
|
1175
|
+
help="What to ingest (default: pdf)")
|
|
1176
|
+
p.add_argument("--md-file", default=None, help="Markdown file (--kind md)")
|
|
1177
|
+
p.add_argument("--pdf-file", default=None, help="PDF file (--kind pdf)")
|
|
1178
|
+
p.add_argument("--thoughts", default=None, help="Raw reasoning text (--kind thoughts)")
|
|
1179
|
+
p.add_argument("--topic", default=None, help="Topic (--kind thoughts)")
|
|
1180
|
+
p.add_argument("--concept-id", default=None, help="Explicit concept ID (--kind md/thoughts)")
|
|
1181
|
+
p.add_argument("--title", default=None, help="Title override (--kind md)")
|
|
1182
|
+
p.add_argument("--description", default=None, help="Description override (--kind md)")
|
|
1183
|
+
p.add_argument("--tags", default=None, help="Comma-separated tags (--kind md/thoughts)")
|
|
1184
|
+
p.add_argument(
|
|
1185
|
+
"--auto-import", action="store_true",
|
|
1186
|
+
help="Auto-import converted markdown into the graph (--kind pdf)",
|
|
1187
|
+
)
|
|
1188
|
+
p.add_argument(
|
|
1189
|
+
"--output", default=None,
|
|
1190
|
+
help="Output directory for converted markdown (default: current dir)",
|
|
1191
|
+
)
|
|
1192
|
+
p.add_argument(
|
|
1193
|
+
"--routing-mode", default="auto",
|
|
1194
|
+
choices=["auto", "surgical", "always", "never"],
|
|
1195
|
+
help="ONNX routing mode (--kind pdf, default: auto)",
|
|
1196
|
+
)
|
|
1197
|
+
p.add_argument(
|
|
1198
|
+
"--mode", default="text", choices=["text", "optional", "omni"],
|
|
1199
|
+
help="Image ingestion mode (default: text)",
|
|
1200
|
+
)
|
|
1201
|
+
p.add_argument(
|
|
1202
|
+
"--batch-size", type=int, default=32,
|
|
1203
|
+
help="Batch size for encoding during auto-import (default: 32)",
|
|
1204
|
+
)
|
|
1205
|
+
p.add_argument(
|
|
1206
|
+
"--purge", action="store_true", default=False,
|
|
1207
|
+
help="Purge deleted concepts during auto-import",
|
|
1208
|
+
)
|
|
1209
|
+
p.add_argument(
|
|
1210
|
+
"--no-extract-images", action="store_true",
|
|
1211
|
+
help="Do not extract embedded images from the PDF",
|
|
1212
|
+
)
|
|
1213
|
+
|
|
1214
|
+
# export
|
|
1215
|
+
p = sub.add_parser("export", help="Export concepts")
|
|
1216
|
+
_add_global(p)
|
|
1217
|
+
_add_logging_flags(p)
|
|
1218
|
+
p.add_argument("--all", action="store_true", dest="export_all", help="Export entire bundle")
|
|
1219
|
+
p.add_argument("--output", required=True, help="Output directory")
|
|
1220
|
+
p.add_argument("--concept-id", help="Concept ID (for single export)")
|
|
1221
|
+
p.add_argument("--type", help="Concept type filter")
|
|
1222
|
+
p.add_argument("--tags", help="Comma-separated tag filters")
|
|
1223
|
+
p.add_argument("--parent", help="Parent directory ID")
|
|
1224
|
+
p.add_argument("--flavor", default="okf", choices=["okf", "obsidian"],
|
|
1225
|
+
help="Link flavor: okf ([t](id.md) + index files) or obsidian "
|
|
1226
|
+
"([[Title]] wikilinks, no index files). Default: okf")
|
|
1227
|
+
|
|
1228
|
+
# diff (structural: snapshot dir-vs-dir, or drift graph-vs-dir)
|
|
1229
|
+
p = sub.add_parser("diff", help="Structural diff: concepts/edges/broken-link deltas")
|
|
1230
|
+
_add_global(p)
|
|
1231
|
+
_add_logging_flags(p)
|
|
1232
|
+
p.add_argument("old", nargs="?", default=None,
|
|
1233
|
+
help="Old side: bundle dir (with NEW: snapshot; alone: drift vs graph)")
|
|
1234
|
+
p.add_argument("new", nargs="?", default=None, help="New side: bundle dir (snapshot mode)")
|
|
1235
|
+
p.add_argument("--json", action="store_true", help="Machine-readable report")
|
|
1236
|
+
|
|
1237
|
+
# doctor (scored health + safe repairs)
|
|
1238
|
+
p = sub.add_parser("doctor", help="Health scan: score, findings, safe --fix")
|
|
1239
|
+
_add_global(p)
|
|
1240
|
+
_add_logging_flags(p)
|
|
1241
|
+
p.add_argument("--strict", action="store_true",
|
|
1242
|
+
help="Exit 1 when any finding exists (CI gate)")
|
|
1243
|
+
p.add_argument("--fix", action="store_true",
|
|
1244
|
+
help="Apply safe repairs (link re-points, timestamp normalization; "
|
|
1245
|
+
"never touches reviewed:true concepts)")
|
|
1246
|
+
p.add_argument("--stale-days", type=int, default=365,
|
|
1247
|
+
help="Age threshold for 'stale' findings (default: 365)")
|
|
1248
|
+
p.add_argument("--json", action="store_true", help="Machine-readable report")
|
|
1249
|
+
|
|
1250
|
+
# lint (pre-import bundle gate: no DB, no model load)
|
|
1251
|
+
p = sub.add_parser("lint", help="Validate bundle frontmatter + links before import")
|
|
1252
|
+
_add_global(p)
|
|
1253
|
+
_add_logging_flags(p)
|
|
1254
|
+
p.add_argument("dir", nargs="?", default=None,
|
|
1255
|
+
help="Bundle directory (default: --bundle, okfgraph.toml, or .)")
|
|
1256
|
+
p.add_argument("--json", action="store_true", help="Machine-readable report")
|
|
1257
|
+
|
|
1258
|
+
# shell
|
|
1259
|
+
p = sub.add_parser("shell", help="Interactive REPL")
|
|
1260
|
+
_add_global(p)
|
|
1261
|
+
_add_logging_flags(p)
|
|
1262
|
+
|
|
1263
|
+
# broken-links
|
|
1264
|
+
p = sub.add_parser("broken-links", help="List broken (orphan) links")
|
|
1265
|
+
_add_global(p)
|
|
1266
|
+
_add_logging_flags(p)
|
|
1267
|
+
|
|
1268
|
+
# repair-links
|
|
1269
|
+
p = sub.add_parser("repair-links", help="Repair broken links by re-checking targets")
|
|
1270
|
+
_add_global(p)
|
|
1271
|
+
_add_logging_flags(p)
|
|
1272
|
+
|
|
1273
|
+
# reindex
|
|
1274
|
+
p = sub.add_parser("reindex", help="Rebuild vector + FTS search indexes")
|
|
1275
|
+
_add_global(p)
|
|
1276
|
+
_add_logging_flags(p)
|
|
1277
|
+
p.add_argument("--if-dirty", action="store_true",
|
|
1278
|
+
help="Only rebuild if data changed since the last index build")
|
|
1279
|
+
|
|
1280
|
+
# Soft-delete commands (Gap #1d)
|
|
1281
|
+
p = sub.add_parser("deleted-list", help="List soft-deleted concepts")
|
|
1282
|
+
_add_global(p)
|
|
1283
|
+
_add_logging_flags(p)
|
|
1284
|
+
|
|
1285
|
+
p = sub.add_parser("deleted-recover", help="Recover a soft-deleted concept")
|
|
1286
|
+
_add_global(p)
|
|
1287
|
+
_add_logging_flags(p)
|
|
1288
|
+
p.add_argument("concept_id", help="Concept ID to recover")
|
|
1289
|
+
|
|
1290
|
+
p = sub.add_parser("deleted-purge", help="Permanently delete expired soft-deleted concepts")
|
|
1291
|
+
_add_global(p)
|
|
1292
|
+
_add_logging_flags(p)
|
|
1293
|
+
p.add_argument("--older-than", type=int, default=None, help="Override recovery window (seconds)")
|
|
1294
|
+
|
|
1295
|
+
return parser
|
|
1296
|
+
|
|
1297
|
+
|
|
1298
|
+
def main():
|
|
1299
|
+
parser = build_parser()
|
|
1300
|
+
args = parser.parse_args()
|
|
1301
|
+
|
|
1302
|
+
if not args.command:
|
|
1303
|
+
parser.print_help()
|
|
1304
|
+
sys.exit(0)
|
|
1305
|
+
|
|
1306
|
+
# Setup logging (Gap #10)
|
|
1307
|
+
_setup_logging(
|
|
1308
|
+
verbose=getattr(args, "verbose", False),
|
|
1309
|
+
quiet=getattr(args, "quiet", False),
|
|
1310
|
+
log_file=getattr(args, "log_file", ""),
|
|
1311
|
+
)
|
|
1312
|
+
|
|
1313
|
+
logger = logging.getLogger("cli")
|
|
1314
|
+
|
|
1315
|
+
# Profile flag (Gap #10C)
|
|
1316
|
+
if getattr(args, "profile", False):
|
|
1317
|
+
logger.info("profiling enabled")
|
|
1318
|
+
profiler = cProfile.Profile()
|
|
1319
|
+
profiler.enable()
|
|
1320
|
+
|
|
1321
|
+
commands = {
|
|
1322
|
+
"init": _init,
|
|
1323
|
+
"model-info": _model_info,
|
|
1324
|
+
"import": _import,
|
|
1325
|
+
"ingest": _ingest,
|
|
1326
|
+
"search": _search,
|
|
1327
|
+
"read": _read,
|
|
1328
|
+
"traverse": _traverse,
|
|
1329
|
+
"export": _export,
|
|
1330
|
+
"shell": _shell,
|
|
1331
|
+
"broken-links": _broken_links,
|
|
1332
|
+
"repair-links": _repair_links,
|
|
1333
|
+
"diff": _diff,
|
|
1334
|
+
"doctor": _doctor,
|
|
1335
|
+
"lint": _lint,
|
|
1336
|
+
"reindex": _reindex,
|
|
1337
|
+
"deleted-list": _deleted_list,
|
|
1338
|
+
"deleted-recover": _deleted_recover,
|
|
1339
|
+
"deleted-purge": _deleted_purge,
|
|
1340
|
+
}
|
|
1341
|
+
|
|
1342
|
+
try:
|
|
1343
|
+
ret = commands[args.command](args)
|
|
1344
|
+
finally:
|
|
1345
|
+
_close_routers()
|
|
1346
|
+
_teardown_logging()
|
|
1347
|
+
# Handlers return an int exit code when CI-gateable (diff/doctor);
|
|
1348
|
+
# everything else returns None (= success).
|
|
1349
|
+
if isinstance(ret, int) and ret != 0:
|
|
1350
|
+
sys.exit(ret)
|
|
1351
|
+
|
|
1352
|
+
# Profile output (Gap #10C)
|
|
1353
|
+
if getattr(args, "profile", False):
|
|
1354
|
+
profiler.disable()
|
|
1355
|
+
import pstats
|
|
1356
|
+
stream = io.StringIO()
|
|
1357
|
+
stats = pstats.Stats(profiler, stream=stream)
|
|
1358
|
+
stats.sort_stats("cumulative")
|
|
1359
|
+
stats.print_stats(20)
|
|
1360
|
+
print(stream.getvalue(), file=sys.stderr)
|
|
1361
|
+
|
|
1362
|
+
|
|
1363
|
+
if __name__ == "__main__":
|
|
1364
|
+
main()
|