okfgraph 0.2.4__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
okfgraph/cli.py ADDED
@@ -0,0 +1,1364 @@
1
+ """OKF CLI — Command-line interface for the OKF knowledge graph."""
2
+
3
+ import argparse
4
+ import cProfile
5
+ import io
6
+ import json
7
+ import logging
8
+ import sys
9
+ from pathlib import Path
10
+
11
+ from okfgraph.router import OKFRouter
12
+ from okfgraph.config import OKFConfig
13
+ from okfgraph.components.converters import BobineConverter
14
+
15
+ # ── Logging setup ──────────────────────────────────────────────────────────
16
+ # Structured logging with stdlib (Gap #10).
17
+ # Loguru was rejected: third-party dependency for a CLI tool where stdlib
18
+ # logging is sufficient. The goal is consistent, structured, debuggable
19
+ # logging without adding extra dependencies.
20
+
21
+ _LOG_HANDLER = None
22
+
23
+
24
+ def _setup_logging(verbose: bool = False, quiet: bool = False, log_file: str = "") -> None:
25
+ """Configure logging for the CLI.
26
+
27
+ Precedence: quiet > verbose > default.
28
+ - quiet: ERROR and above only
29
+ - default: INFO
30
+ - verbose: DEBUG
31
+ """
32
+ global _LOG_HANDLER
33
+
34
+ if quiet:
35
+ level = logging.ERROR
36
+ elif verbose:
37
+ level = logging.DEBUG
38
+ else:
39
+ level = logging.INFO
40
+
41
+ # Configure root logger
42
+ root = logging.getLogger()
43
+ root.setLevel(level)
44
+
45
+ # Remove any existing handlers to avoid duplicates across invocations
46
+ for h in root.handlers[:]:
47
+ root.removeHandler(h)
48
+
49
+ # Console handler with structured format
50
+ fmt = logging.Formatter(
51
+ "%(asctime)s [%(levelname)s] %(name)s: %(message)s",
52
+ datefmt="%H:%M:%S",
53
+ )
54
+ console = logging.StreamHandler(sys.stderr)
55
+ console.setFormatter(fmt)
56
+ console.setLevel(level)
57
+ root.addHandler(console)
58
+ _LOG_HANDLER = console
59
+
60
+ # Optional file handler with rotation
61
+ if log_file:
62
+ from logging.handlers import RotatingFileHandler
63
+ file_handler = RotatingFileHandler(
64
+ log_file,
65
+ maxBytes=5 * 1024 * 1024, # 5MB
66
+ backupCount=3,
67
+ )
68
+ file_handler.setFormatter(fmt)
69
+ file_handler.setLevel(level)
70
+ root.addHandler(file_handler)
71
+
72
+
73
+ def _teardown_logging() -> None:
74
+ """Remove handlers to avoid leaks across CLI invocations."""
75
+ global _LOG_HANDLER
76
+ root = logging.getLogger()
77
+ if _LOG_HANDLER and _LOG_HANDLER in root.handlers:
78
+ root.removeHandler(_LOG_HANDLER)
79
+ _LOG_HANDLER = None
80
+
81
+
82
+ # ── helpers ────────────────────────────────────────────────────────────────
83
+
84
+ class _SlimHelpFormatter(argparse.HelpFormatter):
85
+ """Subcommand help without the repeated global flags.
86
+
87
+ Global options (connection, models, logging) are identical on every
88
+ command, so printing them 15 times costs agents ~4k tokens to learn
89
+ nothing. They are documented once in top-level ``okf --help`` and
90
+ remain fully functional on every subcommand.
91
+ """
92
+
93
+ def add_arguments(self, actions):
94
+ super().add_arguments(
95
+ [a for a in actions if not getattr(a, "_okf_global", False)]
96
+ )
97
+
98
+ def add_usage(self, usage, actions, groups=(), prefix=None):
99
+ super().add_usage(
100
+ usage,
101
+ [a for a in actions if not getattr(a, "_okf_global", False)],
102
+ groups,
103
+ prefix,
104
+ )
105
+
106
+
107
+ class _SubParser(argparse.ArgumentParser):
108
+ """Subcommand parser with slim help + pointer to global options."""
109
+
110
+ def __init__(self, *args, **kwargs):
111
+ kwargs.setdefault("formatter_class", _SlimHelpFormatter)
112
+ kwargs.setdefault(
113
+ "epilog",
114
+ "Global options hidden; see 'okf --help' (or okfgraph.toml).",
115
+ )
116
+ super().__init__(*args, **kwargs)
117
+
118
+
119
+ def _add_global(parser, mark=True):
120
+ """Add --db / --bundle / --dim / --cache-dir / --device to any subparser.
121
+
122
+ With mark=True (subcommands) the flags are tagged so _SlimHelpFormatter
123
+ hides them from per-command help; they stay functional and are shown
124
+ once in top-level help (mark=False there).
125
+ """
126
+ def _add(*a, **k):
127
+ act = parser.add_argument(*a, **k)
128
+ if mark:
129
+ act._okf_global = True # noqa: SLF001 — our own marker
130
+ return act
131
+
132
+ _add("--db", default=None, help="Database path (default: okfgraph.db, or from okfgraph.toml)")
133
+ _add("--bundle", default=None, help="Bundle root directory (default: ., or from okfgraph.toml)")
134
+ _add("--dim", type=int, default=None, help="Embedding dimension (Matryoshka; default: 512, or from okfgraph.toml)")
135
+ _add("--cache-dir", default=None, help="HuggingFace model cache directory (default: ~/.cache/huggingface, or from okfgraph.toml)")
136
+ _add("--device", default=None, choices=["cpu", "cuda"], help="Inference device: cpu or cuda (default: cpu, or from okfgraph.toml)")
137
+ _add("--omni-model-id", default=None, help="Multimodal model ID for image embeddings (default from okfgraph.toml)")
138
+ _add("--chunk-size", type=int, default=None, help="Chunk size in words for overlap (default: 512, or from okfgraph.toml)")
139
+ _add("--chunk-overlap", type=int, default=None, help="Overlap in words between chunks (default: 40, or from okfgraph.toml)")
140
+ _add("--no-chunking", action="store_true", help="Disable chunking during ingestion")
141
+ _add("--wal-mode", action="store_true", help="Enable SQLite WAL mode for concurrent reads (Gap #7a)")
142
+ _add("--allow-remote-images", action="store_true", help="Allow fetching remote images (SSRF risk — use with caution)")
143
+ _add("--allowed-image-domains", default=None, help="Comma-separated list of allowed domains for remote images (Gap #9a)")
144
+
145
+
146
+ def _add_logging_flags(parser, mark=True):
147
+ """Add --verbose / --quiet / --log-file / --profile to a subparser."""
148
+ for args, kwargs in [
149
+ (("--verbose", "-v"), {"action": "store_true", "help": "Enable debug logging"}),
150
+ (("--quiet", "-q"), {"action": "store_true", "help": "Suppress all logging except errors"}),
151
+ (("--log-file",), {"default": "", "help": "Write logs to file (with 5MB rotation)"}),
152
+ (("--profile",), {"action": "store_true", "help": "Enable cProfile for the current invocation (outputs to stdout)"}),
153
+ ]:
154
+ act = parser.add_argument(*args, **kwargs)
155
+ if mark:
156
+ act._okf_global = True # noqa: SLF001 — our own marker
157
+
158
+
159
+ # Routers opened during a CLI invocation, closed (checkpointed) on exit so a
160
+ # writer never leaves an un-checkpointed WAL that a later open would reject.
161
+ _OPEN_ROUTERS = []
162
+
163
+
164
+ def _router(args):
165
+ """Build an OKFRouter from parsed args (registered for cleanup on exit).
166
+
167
+ Uses the config module to merge CLI args with TOML file and env vars.
168
+ Precedence: CLI > env > file > defaults.
169
+ """
170
+ # Build CLI args dict (only non-None values override config)
171
+ cli_dict = {}
172
+ for attr in ("db", "bundle", "dim", "cache_dir", "device",
173
+ "omni_model_id", "chunk_size", "chunk_overlap",
174
+ "no_chunking", "mode", "batch_size",
175
+ "allow_remote_images", "wal_mode", "allowed_image_domains"):
176
+ val = getattr(args, attr, None)
177
+ if val is not None:
178
+ cli_dict[attr] = val
179
+
180
+ # Resolve bundle root for TOML lookup
181
+ bundle_root = cli_dict.get("bundle") or "."
182
+
183
+ # Load merged config
184
+ config = OKFConfig.load(bundle_root=bundle_root, cli_args=cli_dict)
185
+
186
+ # Build allowed_image_domains list
187
+ allowed_domains = config.import_config.allowed_image_domains
188
+ if getattr(args, "allowed_image_domains", None):
189
+ allowed_domains = [d.strip() for d in args.allowed_image_domains.split(",") if d.strip()]
190
+
191
+ router = OKFRouter(
192
+ db_path=config.database.path,
193
+ bundle_root=str(config.bundle),
194
+ embedding_dim=config.database.dim,
195
+ omni_model_id=config.embedding.omni_model_id,
196
+ cache_dir=config.embedding.cache_dir,
197
+ device=config.embedding.device,
198
+ allow_remote_images=config.import_config.allow_remote_images,
199
+ allowed_image_domains=allowed_domains,
200
+ chunk_size=config.import_config.chunk_size,
201
+ chunk_overlap=config.import_config.chunk_overlap,
202
+ enable_chunking=not config.import_config.no_chunking,
203
+ wal_mode=config.database.wal_mode,
204
+ )
205
+ _OPEN_ROUTERS.append(router)
206
+ return router
207
+
208
+
209
+ def _close_routers():
210
+ """Checkpoint + close every router opened this invocation."""
211
+ while _OPEN_ROUTERS:
212
+ router = _OPEN_ROUTERS.pop()
213
+ try:
214
+ router.close()
215
+ except Exception:
216
+ pass
217
+
218
+
219
+ # ── command handlers ───────────────────────────────────────────────────────
220
+
221
+ def _init(args):
222
+ db_path = str(args.db)
223
+ logger = logging.getLogger("cli")
224
+ logger.info("initializing database at %s (dim=%d)", db_path, args.dim)
225
+ _router(args)
226
+ logger.info("database initialized (embedding_dim=%d)", args.dim)
227
+
228
+
229
+ def _model_info(args):
230
+ """Show model cache status without loading the model."""
231
+ logger = logging.getLogger("cli")
232
+ info = OKFRouter.model_info(
233
+ model_id=getattr(args, "model_id", "jinaai/jina-embeddings-v5-text-small-retrieval"),
234
+ cache_dir=getattr(args, "cache_dir", None),
235
+ )
236
+ logger.info("model: %s", info['model_id'])
237
+ logger.info("cache: %s", info['cache_dir'])
238
+ if info["cached"]:
239
+ logger.info("status: cached")
240
+ logger.info("path: %s", info['snapshot_path'])
241
+ size_gb = info["disk_usage_bytes"] / (1024 ** 3)
242
+ logger.info("size: %.2f GB", size_gb)
243
+ else:
244
+ default_cache = OKFRouter.default_cache_dir()
245
+ logger.info("status: not cached (will download on first use)")
246
+ logger.info("will use: %s", default_cache)
247
+
248
+
249
+ def _import(args):
250
+ logger = logging.getLogger("cli")
251
+ router = _router(args)
252
+ mode = getattr(args, "mode", "text")
253
+ purge = getattr(args, "purge", False)
254
+ if getattr(args, "import_all", False):
255
+ bundle_path = Path(args.bundle) if args.bundle else None
256
+ ids = router.import_mgr.import_bundle(
257
+ bundle_path,
258
+ batch_size=getattr(args, "batch_size", 32) or 32,
259
+ mode=mode,
260
+ purge_deleted=purge,
261
+ )
262
+ logger.info("imported %d concept(s) (mode: %s)", len(ids), mode)
263
+ for cid in ids:
264
+ n = len(router.image_mgr.list_images(cid))
265
+ suffix = f" [{n} image(s)]" if n else ""
266
+ logger.info(" %s%s", cid, suffix)
267
+ else:
268
+ for fp in args.files:
269
+ path = Path(fp)
270
+ if not path.exists():
271
+ logger.warning("skipping %s: file not found", fp)
272
+ continue
273
+ cid = router.import_from_okf(path, mode=mode)
274
+ imgs = router.image_mgr.list_images(cid)
275
+ suffix = f" ({len(imgs)} image(s), mode: {mode})" if imgs else ""
276
+ logger.info("imported: %s%s", cid, suffix)
277
+
278
+
279
+ def _search(args):
280
+ """Unified search: concepts (default), chunks, or images."""
281
+ router = _router(args)
282
+ target = getattr(args, "target", "concepts") or "concepts"
283
+ limit = getattr(args, "limit", 10) or 10
284
+ if target == "images":
285
+ results = router.image_mgr.search_images_with_text(
286
+ text_query=args.query,
287
+ use_text_model=not getattr(args, "use_omni", False),
288
+ limit=limit,
289
+ )
290
+ if not results:
291
+ print("No image results found.")
292
+ return
293
+ print(f"Found {len(results)} image(s):\n")
294
+ for i, r in enumerate(results, 1):
295
+ label = r.get("alt_text") or r.get("file_name") or r.get("id")
296
+ print(f" {i}. [{r['relevance_score']:.4f}] {label} ({r.get('embed_route')})")
297
+ print(f" file: {r.get('file_name')}")
298
+ print(f" id: {r['id']}")
299
+ print()
300
+ return
301
+ tags = args.tags.split(",") if getattr(args, "tags", None) else None
302
+ filt = {
303
+ "concept_type": getattr(args, "type", None),
304
+ "tags": tags,
305
+ "parent_id": getattr(args, "parent", None),
306
+ }
307
+ filt = {k: v for k, v in filt.items() if v is not None}
308
+ rank = getattr(args, "rank", "none") or "none"
309
+ if target == "chunks":
310
+ if rank != "none":
311
+ print("[ERROR] --rank is concepts-only; for chunks use --hub-rerank/--expand")
312
+ return
313
+ if getattr(args, "hub_rerank", False):
314
+ results = router.search_engine.search_chunks_with_hub_score(
315
+ query=args.query,
316
+ limit=limit,
317
+ hub_weight=getattr(args, "hub_weight", 0.3),
318
+ )
319
+ if not results:
320
+ print("No results found.")
321
+ return
322
+ print(f"Found {len(results)} result(s):\n")
323
+ for i, r in enumerate(results, 1):
324
+ print(f" {i}. [{r['final_score']:.4f}] {r['parent_title']} §{r['chunk_index']}")
325
+ print(f" hub={r['hub_score']:.2f} rrf={r['rrf_score']:.4f}")
326
+ print(f" {r['chunk_text'][:150]}")
327
+ print()
328
+ return
329
+ if getattr(args, "expand", False):
330
+ results = router.search_engine.search_with_context(
331
+ query=args.query,
332
+ limit=min(limit, 20),
333
+ context_hops=getattr(args, "context_hops", 1),
334
+ )
335
+ if not results:
336
+ print("No results found.")
337
+ return
338
+ print(f"Found {len(results)} result(s):\n")
339
+ for i, r in enumerate(results, 1):
340
+ chunk = r["chunk"]
341
+ print(f" {i}. [{chunk['rrf_score']:.4f}] {chunk['parent_title']} §{chunk['chunk_index']}")
342
+ print(f" {chunk['chunk_text'][:150]}")
343
+ if r["incoming_links"]:
344
+ titles = [l.get("title", l.get("id", "?")) for l in r["incoming_links"][:3]]
345
+ print(f" ← linked by: {', '.join(titles)}")
346
+ if r["outgoing_links"]:
347
+ titles = [l.get("title", l.get("id", "?")) for l in r["outgoing_links"][:3]]
348
+ print(f" → links to: {', '.join(titles)}")
349
+ if r["ancestry"]:
350
+ print(f" path: {' → '.join(a['title'] for a in r['ancestry'])}")
351
+ if r["siblings"]:
352
+ print(f" siblings: {', '.join(s['title'] for s in r['siblings'][:3])}")
353
+ print()
354
+ return
355
+ results = router.search_engine.search_chunks(
356
+ query=args.query,
357
+ limit=limit,
358
+ **filt,
359
+ )
360
+ if not results:
361
+ print("No chunk results found.")
362
+ return
363
+ print(f"Found {len(results)} chunk(s):\n")
364
+ for i, r in enumerate(results, 1):
365
+ print(f" {i}. [{r['rrf_score']:.4f}] {r['block_type']} #{r['chunk_index']}")
366
+ text = r.get("chunk_text", "")
367
+ print(f" {text[:150]}")
368
+ if r.get("parent_title"):
369
+ print(f" parent: {r['parent_title']}")
370
+ print(f" id: {r['chunk_id']}")
371
+ print()
372
+ return
373
+ results = router.search_hybrid(
374
+ query=args.query,
375
+ limit=limit,
376
+ include_chunks=getattr(args, "chunks", False),
377
+ rank=rank,
378
+ hub_weight=getattr(args, "hub_weight", 0.3) or 0.3,
379
+ **filt,
380
+ )
381
+ if not results:
382
+ print("No results found.")
383
+ return
384
+ print(f"Found {len(results)} result(s):\n")
385
+ for i, r in enumerate(results, 1):
386
+ print(f" {i}. [{r['relevance_score']:.4f}] {r['title']} ({r['type']})")
387
+ desc = r.get("description") or ""
388
+ if desc:
389
+ print(f" {desc[:120]}")
390
+ if r.get("tags"):
391
+ print(f" tags: {', '.join(r['tags'])}")
392
+ print(f" id: {r['id']}")
393
+ # Print matched chunks if requested
394
+ if r.get("matched_chunks"):
395
+ print(f" matched chunks:")
396
+ for mc in r["matched_chunks"][:3]:
397
+ print(f" [{mc['rrf_score']:.4f}] {mc['block_type']} #{mc['chunk_index']}")
398
+ print(f" {mc.get('chunk_text', '')[:100]}")
399
+ print()
400
+
401
+
402
+ def _traverse(args):
403
+ """Unified traversal: relationships, directory listing, or shortest path."""
404
+ router = _router(args)
405
+ target = getattr(args, "target", None)
406
+ start = getattr(args, "start_id", "") or ""
407
+ if not start:
408
+ items = router.list_directory("")
409
+ if not items:
410
+ print("Directory is empty.")
411
+ return
412
+ print("Contents of '(root)':\n")
413
+ for item in items:
414
+ icon = "[D]" if item["type"] == "Directory" else "[F]"
415
+ print(f" {icon} {item['title']} ({item['type']})")
416
+ print(f" id: {item['id']}")
417
+ return
418
+ if target:
419
+ nodes = router.search_engine.find_path(
420
+ start, target, max_length=getattr(args, "max_path_length", 6)
421
+ )
422
+ if not nodes:
423
+ print(f"No path found between '{start}' and '{target}'.")
424
+ return
425
+ print(f"Path ({len(nodes)} nodes):")
426
+ for i, n in enumerate(nodes, 1):
427
+ print(f" {i}. {n.get('title', '?')} ({n.get('type', '?')})")
428
+ print(f" id: {n['id']}")
429
+ return
430
+ results = router.traverse(
431
+ start_id=start,
432
+ relationship=args.relationship,
433
+ direction=args.direction,
434
+ depth=args.depth,
435
+ node_type=getattr(args, "type", None),
436
+ )
437
+ if not results:
438
+ print("No results found.")
439
+ return
440
+ print(f"Found {len(results)} node(s):\n")
441
+ for r in results:
442
+ print(f" {r['id']} ({r['type']})")
443
+ if r.get("title"):
444
+ print(f" title: {r['title']}")
445
+
446
+
447
+ def _read(args):
448
+ """Unified read: body (default), chunks, document, or context."""
449
+ router = _router(args)
450
+ include = getattr(args, "include", "body") or "body"
451
+ cid = args.concept_id
452
+ max_tokens = getattr(args, "max_tokens", None)
453
+ if max_tokens:
454
+ try:
455
+ reading = router.search_engine.read_with_budget(
456
+ cid, include=include, max_tokens=max_tokens,
457
+ )
458
+ except KeyError:
459
+ print(f"Concept '{cid}' not found.")
460
+ return
461
+ flag = " (truncated)" if reading["truncated"] else ""
462
+ print(f"[{reading['used']}/{reading['budget']} tokens{flag}] {cid}\n")
463
+ for sec in reading["sections"]:
464
+ print(f"## {sec['title'] or sec['id']} [{sec['kind']} | {sec['id']}]")
465
+ print(sec["text"])
466
+ print()
467
+ return
468
+ if include == "chunks":
469
+ chunks = router.search_engine.get_chunks(cid)
470
+ if not chunks:
471
+ print("No chunks found for this concept.")
472
+ return
473
+ print(f"Chunks for '{cid}' ({len(chunks)} total):\n")
474
+ for c in chunks:
475
+ text = c.chunk_text[:120]
476
+ print(f" #{c.chunk_index} [{c.block_type}] {text}")
477
+ return
478
+ if include == "document":
479
+ text = router.embed_engine.reconstruct_document(cid)
480
+ if not text:
481
+ print("No chunks found for this concept.")
482
+ return
483
+ if getattr(args, "output", None):
484
+ Path(args.output).write_text(text, encoding="utf-8")
485
+ print(f"[OK] Reconstructed document written to {args.output}")
486
+ else:
487
+ print(text)
488
+ return
489
+ if include == "context":
490
+ incoming = router.traverse(cid, "LINKS_TO", "INCOMING", 1)[:10]
491
+ outgoing = router.traverse(cid, "LINKS_TO", "OUTGOING", 1)[:10]
492
+ ancestry = router.search_engine._get_ancestry(cid)
493
+ siblings = router.search_engine._get_siblings(cid)[:10]
494
+ if incoming:
495
+ print("Linked by:")
496
+ for l in incoming:
497
+ print(f" {l.get('title', l.get('id', '?'))} (id: {l['id']})")
498
+ if outgoing:
499
+ print("Links to:")
500
+ for l in outgoing:
501
+ print(f" {l.get('title', l.get('id', '?'))} (id: {l['id']})")
502
+ if ancestry:
503
+ print(f"Path: {' → '.join(a['title'] for a in ancestry)}")
504
+ if siblings:
505
+ print("Siblings:")
506
+ for s in siblings:
507
+ print(f" {s['title']} ({s['type']})")
508
+ print(f" id: {s['id']}")
509
+ if not (incoming or outgoing or ancestry or siblings):
510
+ print(f"No context found for '{cid}'.")
511
+ return
512
+ concept = router.get_by_id(cid)
513
+ if not concept:
514
+ print(f"Concept '{cid}' not found.")
515
+ return
516
+ data = concept.public_dict()
517
+ body = data.pop("body", "")
518
+ print(json.dumps(data, indent=2, default=str))
519
+ if body:
520
+ print(f"\n--- BODY ---\n{body}")
521
+
522
+
523
+ def _export(args):
524
+ router = _router(args)
525
+ flavor = getattr(args, "flavor", "okf") or "okf"
526
+ if getattr(args, "export_all", False):
527
+ tags = args.tags.split(",") if args.tags else None
528
+ ids = router.export_mgr.export_bundle(
529
+ output_dir=Path(args.output),
530
+ directory_id=args.parent,
531
+ concept_type=args.type,
532
+ tags=tags,
533
+ flavor=flavor,
534
+ )
535
+ print(f"[OK] Exported {len(ids)} concepts to {args.output} (flavor: {flavor})")
536
+ else:
537
+ cid = args.concept_id
538
+ output_path = Path(args.output) / f"{cid}.md"
539
+ router.export_mgr.export_to_okf(cid, output_path, flavor=flavor)
540
+ print(f"[OK] Exported {cid} → {output_path} (flavor: {flavor})")
541
+
542
+
543
+ def _broken_links(args):
544
+ logger = logging.getLogger("cli")
545
+ router = _router(args)
546
+ broken = router.list_broken_links()
547
+ if not broken:
548
+ logger.info("no broken links found")
549
+ return
550
+ logger.info("found %d broken link(s)", len(broken))
551
+ for link in broken:
552
+ logger.info(" %s → %s", link['source'], link['target'])
553
+
554
+
555
+ def _repair_links(args):
556
+ logger = logging.getLogger("cli")
557
+ router = _router(args)
558
+ count = router.repair_links()
559
+ logger.info("repaired %d link(s)", count)
560
+
561
+
562
+ def _lint(args):
563
+ """Pre-import bundle gate: frontmatter + link validation.
564
+
565
+ Deliberately router-free (no DB, no ~30s model cold-boot): lint answers
566
+ "is this bundle well-formed?" before an import cycle is spent.
567
+ Exit 0 = clean (warnings ok), 1 = errors, 2 = usage (bad dir).
568
+ """
569
+ from okfgraph.components.lint import lint_bundle
570
+ from okfgraph.config import OKFConfig
571
+
572
+ given = getattr(args, "dir", None) or getattr(args, "bundle", None) or "."
573
+ # bundle_root is only the TOML lookup location; the value itself rides
574
+ # in cli_args (same split as _router's cli_dict).
575
+ config = OKFConfig.load(bundle_root=given, cli_args={"bundle": given})
576
+ target = Path(str(config.bundle))
577
+ if not target.is_dir():
578
+ print(f"[ERROR] not a bundle directory: {target}")
579
+ return 2
580
+ report = lint_bundle(target)
581
+ if getattr(args, "json", False):
582
+ print(json.dumps(report, indent=2, default=str))
583
+ else:
584
+ print(f"{report['files']} file(s): "
585
+ f"{len(report['errors'])} error(s), "
586
+ f"{len(report['warnings'])} warning(s)")
587
+ for e in report["errors"]:
588
+ print(f" [ERROR] {e['file']} {e['rule']}: {e['message']}")
589
+ for w in report["warnings"]:
590
+ print(f" [warn] {w['file']} {w['rule']}: {w['message']}")
591
+ if report["clean"]:
592
+ print("Bundle is lint-clean (safe to import).")
593
+ return 0 if report["clean"] else 1
594
+
595
+
596
+ def _diff(args):
597
+ """Structural diff: snapshot (dir vs dir) or drift (graph vs dir).
598
+
599
+ Returns an exit code (0 = identical, 1 = different) for CI gating;
600
+ main() propagates int returns to sys.exit.
601
+ """
602
+ from okfgraph.components.diff import DiffManager
603
+
604
+ old = getattr(args, "old", None)
605
+ new = getattr(args, "new", None)
606
+ as_json = getattr(args, "json", False)
607
+ if old and new and Path(old).is_dir() and Path(new).is_dir():
608
+ # Snapshot mode needs no database (and no model load).
609
+ result = DiffManager(None).diff_dirs(Path(old), Path(new))
610
+ elif old and new:
611
+ print("[ERROR] diff needs two bundle directories (or one + --db/--bundle)")
612
+ return 2
613
+ else:
614
+ router = _router(args)
615
+ side = Path(old or new) if (old or new) else router.bundle_root
616
+ if not side.is_dir():
617
+ print(f"[ERROR] not a bundle directory: {side}")
618
+ return 2
619
+ result = router.diff_db_dir(side)
620
+ if as_json:
621
+ print(json.dumps(result, indent=2, default=str))
622
+ else:
623
+ _print_diff(result)
624
+ return 0 if result["identical"] else 1
625
+
626
+
627
+ def _print_diff(result) -> None:
628
+ """Human-readable rendering of a structural diff report."""
629
+ if result["identical"]:
630
+ print("No structural differences.")
631
+ return
632
+ for cid in result["added"]:
633
+ print(f" + concept {cid}")
634
+ for cid in result["removed"]:
635
+ print(f" - concept {cid}")
636
+ for cid in result["changed"]:
637
+ print(f" ~ body {cid}")
638
+ for r in result["retitled"]:
639
+ print(f" ~ title {r['id']}: {r['old']!r} -> {r['new']!r}")
640
+ for r in result["retyped"]:
641
+ print(f" ~ type {r['id']}: {r['old']!r} -> {r['new']!r}")
642
+ for s, t in result["edges_added"]:
643
+ print(f" + edge {s} -> {t}")
644
+ for s, t in result["edges_removed"]:
645
+ print(f" - edge {s} -> {t}")
646
+ for s, t in result["broken_new"]:
647
+ print(f" + broken {s} -> {t}")
648
+ for s, t in result["broken_fixed"]:
649
+ print(f" - broken {s} -> {t}")
650
+
651
+
652
+ def _doctor(args):
653
+ """Scored health scan, optionally with safe --fix repairs.
654
+
655
+ Returns an exit code with --strict (1 when any finding exists).
656
+ """
657
+ router = _router(args)
658
+ if getattr(args, "fix", False):
659
+ fixed = router.doctor_fix()
660
+ print(f"[OK] repaired {fixed['repaired_links']} link(s), "
661
+ f"normalized {fixed['normalized_timestamps']} timestamp(s)")
662
+ if fixed["skipped_reviewed"]:
663
+ print(f" skipped reviewed: {', '.join(fixed['skipped_reviewed'])}")
664
+ report = router.diagnose(
665
+ stale_days=getattr(args, "stale_days", 365) or 365,
666
+ )
667
+ if getattr(args, "json", False):
668
+ print(json.dumps(report, indent=2, default=str))
669
+ else:
670
+ print(f"Health score: {report['score']}/100 ({report['concepts']} concepts)")
671
+ for f in report["findings"]:
672
+ print(f" [{f['severity']}] {f['rule']} {f['path']}: {f['message']}")
673
+ for i in report["info"]:
674
+ print(f" (info) {i['rule']}: {i['message']}")
675
+ if getattr(args, "strict", False) and report["findings"]:
676
+ return 1
677
+ return 0
678
+
679
+
680
+ def _reindex(args):
681
+ logger = logging.getLogger("cli")
682
+ router = _router(args)
683
+ ran = router.schema_mgr.reindex(force=not getattr(args, "if_dirty", False))
684
+ if ran:
685
+ logger.info("search indexes rebuilt")
686
+ else:
687
+ logger.info("search indexes already up to date; nothing to do.")
688
+
689
+
690
+ def _ingest(args):
691
+ """Unified ingest: markdown file, PDF (default), or raw thoughts.
692
+
693
+ Delegates to IngestManager.ingest_md / ingest_pdf / ingest_thoughts.
694
+ PDF conversion uses the configured DocumentConverter (bobine default).
695
+ """
696
+ logger = logging.getLogger("cli")
697
+ router = _router(args)
698
+ kind = getattr(args, "kind", "pdf") or "pdf"
699
+ tags = args.tags.split(",") if getattr(args, "tags", None) else None
700
+ if kind == "md":
701
+ md_file = getattr(args, "md_file", None)
702
+ if not md_file:
703
+ print("[ERROR] --md-file is required for --kind md")
704
+ return
705
+ result = router.ingest_mgr.ingest_md(
706
+ md_path=md_file,
707
+ concept_id=getattr(args, "concept_id", None),
708
+ title=getattr(args, "title", None),
709
+ description=getattr(args, "description", None),
710
+ tags=tags,
711
+ mode=getattr(args, "mode", "text") or "text",
712
+ )
713
+ print(f"[OK] Imported {result['concept_id']} ({result['chunk_count']} chunks)")
714
+ return
715
+ if kind == "thoughts":
716
+ if not getattr(args, "thoughts", None) or not getattr(args, "topic", None):
717
+ print("[ERROR] --thoughts and --topic are required for --kind thoughts")
718
+ return
719
+ result = router.ingest_mgr.ingest_thoughts(
720
+ args.thoughts,
721
+ topic=args.topic,
722
+ concept_id=getattr(args, "concept_id", None),
723
+ tags=tags,
724
+ )
725
+ print(f"[OK] Stored thought {result['concept_id']}")
726
+ return
727
+ pdf_path = Path(getattr(args, "pdf_file", None) or "")
728
+ if not pdf_path.name or not pdf_path.exists():
729
+ print(f"[ERROR] File not found: {pdf_path}")
730
+ return
731
+
732
+ auto_import = getattr(args, "auto_import", False)
733
+ output_dir = getattr(args, "output", None)
734
+ if not auto_import and not output_dir:
735
+ output_dir = Path(".")
736
+
737
+ def on_page(idx, total):
738
+ print(f" page {idx + 1}/{total}", end="\r")
739
+
740
+ try:
741
+ from okfgraph.components.converters import BobineConverter
742
+ converter = BobineConverter(
743
+ routing_mode=getattr(args, "routing_mode", "auto") or "auto",
744
+ extract_images=not getattr(args, "no_extract_images", False),
745
+ )
746
+ result = router.ingest_mgr.ingest_pdf(
747
+ pdf_path,
748
+ auto_import=auto_import,
749
+ output_dir=output_dir,
750
+ mode=getattr(args, "mode", "text") or "text",
751
+ batch_size=getattr(args, "batch_size", 32) or 32,
752
+ purge_deleted=getattr(args, "purge", False),
753
+ on_page=on_page,
754
+ converter=converter,
755
+ )
756
+ except RuntimeError as e:
757
+ print(f"[ERROR] {e}")
758
+ return
759
+ print() # newline after progress
760
+
761
+ logger.info("written %s", result["md_path"])
762
+ logger.info("assets in %s", result["image_dir"])
763
+ if auto_import:
764
+ for cid in result["concept_ids"]:
765
+ n = len(router.image_mgr.list_images(cid))
766
+ suffix = f" [{n} image(s)]" if n else ""
767
+ logger.info(" %s%s", cid, suffix)
768
+ else:
769
+ logger.info("run 'okf import --all --bundle %s' to import.", output_dir)
770
+
771
+
772
+ def _deleted_list(args):
773
+ """List soft-deleted concepts with recovery status."""
774
+ router = _router(args)
775
+ deleted = router.purge_mgr.list_deleted_concepts()
776
+ if not deleted:
777
+ print("No soft-deleted concepts found.")
778
+ return
779
+ print(f"Soft-deleted concepts ({len(deleted)} total):\n")
780
+ for d in deleted:
781
+ status = "recoverable" if d["recoverable"] else "expired"
782
+ print(f" [{status}] {d['concept_id']}")
783
+ print(f" title: {d['title']}")
784
+ print(f" type: {d['type']}")
785
+ print(f" deleted: {d['deleted_at']} ({d['age_seconds']:.0f}s ago)")
786
+ print()
787
+
788
+
789
+ def _deleted_recover(args):
790
+ """Recover a soft-deleted concept."""
791
+ router = _router(args)
792
+ success = router.purge_mgr._recover_concept(args.concept_id)
793
+ if success:
794
+ print(f"[OK] Recovered concept '{args.concept_id}'.")
795
+ else:
796
+ print(f"[ERROR] Concept '{args.concept_id}' not found or past recovery window.")
797
+
798
+
799
+ def _deleted_purge(args):
800
+ """Permanently delete expired soft-deleted concepts."""
801
+ router = _router(args)
802
+ older_than = getattr(args, "older_than", None)
803
+ count = router.purge_mgr.purge_deleted_concepts(older_than=older_than)
804
+ print(f"[OK] Permanently deleted {count} expired concept(s).")
805
+
806
+
807
+ def _shell(args):
808
+ router = _router(args)
809
+ banner = """OKF Interactive Shell
810
+ ========================================
811
+ Commands:
812
+ import <file> [mode] — import single OKF file (mode: text|optional|omni)
813
+ import-bundle [path] [mode]— import entire bundle (mode: text|optional|omni)
814
+ search [target:]<query> — search concepts (default), chunks:, images:
815
+ search <query> expand — chunk hits + graph neighborhood
816
+ search <query> hub — chunk hits reranked by hub score
817
+ read <id> [chunks|document|context] — read a concept (default: body)
818
+ traverse [id] [rel] [dir] [depth] — traverse (no id = root listing)
819
+ traverse <id1> <id2> — shortest path between two concepts
820
+ images <concept_id> — list images attached to a concept
821
+ export-bundle <output_dir> — export all concepts
822
+ export <id> <output_dir> — export single concept
823
+ ingest <file> [--auto-import] — ingest .md or .pdf (auto-import PDFs)
824
+ model-info — show model cache status
825
+ help — show this help
826
+ quit / exit — exit shell
827
+ ========================================"""
828
+
829
+ print(banner)
830
+
831
+ while True:
832
+ try:
833
+ line = input("\n> ").strip()
834
+ except (EOFError, KeyboardInterrupt):
835
+ print("\nBye.")
836
+ break
837
+
838
+ if not line:
839
+ continue
840
+
841
+ parts = line.split(None, 1)
842
+ cmd = parts[0].lower()
843
+ rest = parts[1] if len(parts) > 1 else ""
844
+
845
+ if cmd in ("quit", "exit", "q"):
846
+ print("Bye.")
847
+ break
848
+
849
+ elif cmd == "help":
850
+ print(banner)
851
+
852
+ elif cmd == "import" and rest:
853
+ tokens = rest.strip().split()
854
+ mode = "text"
855
+ if tokens and tokens[-1].lower() in ("text", "optional", "omni"):
856
+ mode = tokens[-1].lower()
857
+ tokens = tokens[:-1]
858
+ fp = Path(" ".join(tokens))
859
+ if not fp.exists():
860
+ print(f"Error: {fp} not found")
861
+ continue
862
+ cid = router.import_from_okf(fp, mode=mode)
863
+ imgs = router.image_mgr.list_images(cid)
864
+ suffix = f" ({len(imgs)} image(s), mode: {mode})" if imgs else ""
865
+ print(f"[OK] Imported: {cid}{suffix}")
866
+
867
+ elif cmd == "import-bundle":
868
+ tokens = rest.strip().split()
869
+ mode = "text"
870
+ if tokens and tokens[-1].lower() in ("text", "optional", "omni"):
871
+ mode = tokens[-1].lower()
872
+ tokens = tokens[:-1]
873
+ bundle_path = Path(" ".join(tokens)) if tokens else None
874
+ ids = router.import_mgr.import_bundle(bundle_path, mode=mode)
875
+ print(f"[OK] Imported {len(ids)} concepts (image mode: {mode})")
876
+
877
+ elif cmd == "search" and rest:
878
+ tokens = rest.strip().split()
879
+ first = tokens[0]
880
+ target = "concepts"
881
+ if ":" in first and first.split(":")[0] in ("concepts", "chunks", "images"):
882
+ target, first = first.split(":", 1)
883
+ tokens[0] = first
884
+ query = tokens[0]
885
+ expand = "expand" in tokens[1:]
886
+ hub = "hub" in tokens[1:]
887
+ type_filter = tags_filter = parent_filter = None
888
+ for t in tokens[1:]:
889
+ if t.startswith("type:"):
890
+ type_filter = t[5:]
891
+ elif t.startswith("tags:"):
892
+ tags_filter = t[5:].split(",")
893
+ elif t.startswith("parent:"):
894
+ parent_filter = t[7:]
895
+ if target == "images":
896
+ results = router.image_mgr.search_images_with_text(query)
897
+ for i, r in enumerate(results, 1):
898
+ label = r.get("alt_text") or r.get("file_name") or r.get("id")
899
+ print(f" {i}. [{r['relevance_score']:.4f}] {label} ({r.get('embed_route')})")
900
+ print(f" id: {r['id']}")
901
+ elif target == "chunks":
902
+ if hub:
903
+ results = router.search_engine.search_chunks_with_hub_score(query)
904
+ elif expand:
905
+ results = router.search_engine.search_with_context(query)
906
+ else:
907
+ results = router.search_engine.search_chunks(query)
908
+ for i, r in enumerate(results, 1):
909
+ chunk = r.get("chunk", r)
910
+ score = r.get("final_score", chunk.get("rrf_score", 0))
911
+ print(f" {i}. [{score:.4f}] {chunk.get('parent_title', '?')} §{chunk.get('chunk_index', '?')}")
912
+ print(f" {chunk.get('chunk_text', '')[:150]}")
913
+ else:
914
+ results = router.search_hybrid(
915
+ query=query, concept_type=type_filter,
916
+ tags=tags_filter, parent_id=parent_filter,
917
+ )
918
+ for i, r in enumerate(results, 1):
919
+ print(f" {i}. [{r['relevance_score']:.4f}] {r['title']} ({r['type']})")
920
+ desc = r.get("description") or ""
921
+ if desc:
922
+ print(f" {desc[:120]}")
923
+
924
+ elif cmd == "read" and rest:
925
+ tokens = rest.strip().split()
926
+ cid = tokens[0]
927
+ include = tokens[1] if len(tokens) > 1 else "body"
928
+ if include == "chunks":
929
+ chunks = router.search_engine.get_chunks(cid)
930
+ if not chunks:
931
+ print("No chunks found.")
932
+ for c in chunks:
933
+ print(f" #{c.chunk_index} [{c.block_type}] {c.chunk_text[:120]}")
934
+ elif include == "document":
935
+ text = router.embed_engine.reconstruct_document(cid)
936
+ print(text if text else "No chunks found for this concept.")
937
+ elif include == "context":
938
+ for l in router.traverse(cid, "LINKS_TO", "INCOMING", 1)[:10]:
939
+ print(f" ← {l.get('title', l.get('id', '?'))}")
940
+ for l in router.traverse(cid, "LINKS_TO", "OUTGOING", 1)[:10]:
941
+ print(f" → {l.get('title', l.get('id', '?'))}")
942
+ for a in router.search_engine._get_ancestry(cid):
943
+ print(f" ↑ {a['title']}")
944
+ else:
945
+ concept = router.get_by_id(cid)
946
+ if concept:
947
+ data = concept.public_dict()
948
+ body = data.pop("body", "")
949
+ print(json.dumps(data, indent=2, default=str))
950
+ if body:
951
+ print(f"\n--- BODY ---\n{body}")
952
+ else:
953
+ print(f"Concept '{cid}' not found")
954
+
955
+ elif cmd == "traverse":
956
+ tokens = rest.strip().split()
957
+ if not tokens:
958
+ for item in router.list_directory(""):
959
+ icon = "[D]" if item["type"] == "Directory" else "[F]"
960
+ print(f" {icon} {item['title']} ({item['type']})")
961
+ elif len(tokens) == 2 and tokens[1] not in ("CONTAINS", "LINKS_TO", "PART_OF", "INCLUDES_ASSET"):
962
+ nodes = router.search_engine.find_path(tokens[0], tokens[1])
963
+ if not nodes:
964
+ print(f"No path found between '{tokens[0]}' and '{tokens[1]}'.")
965
+ else:
966
+ print(f"Path ({len(nodes)} nodes):")
967
+ for i, n in enumerate(nodes, 1):
968
+ print(f" {i}. {n.get('title', '?')} ({n.get('type', '?')})")
969
+ print(f" id: {n['id']}")
970
+ else:
971
+ start_id = tokens[0]
972
+ rel = tokens[1] if len(tokens) > 1 else "CONTAINS"
973
+ direction = tokens[2] if len(tokens) > 2 else "OUTGOING"
974
+ depth = int(tokens[3]) if len(tokens) > 3 else 1
975
+ results = router.traverse(start_id, rel, direction, depth)
976
+ for r in results:
977
+ print(f" {r['id']} ({r['type']}) — {r.get('title', '')}")
978
+
979
+ elif cmd == "images" and rest:
980
+ imgs = router.image_mgr.list_images(rest.strip())
981
+ if not imgs:
982
+ print("No images attached.")
983
+ for im in imgs:
984
+ alt = im.get("alt_text") or "(no alt-text)"
985
+ print(f" [{im.get('embed_route')}] {im.get('file_name')} — {alt}")
986
+ print(f" id: {im.get('id')}")
987
+
988
+ elif cmd == "export-bundle" and rest:
989
+ ids = router.export_mgr.export_bundle(Path(rest.strip()))
990
+ print(f"[OK] Exported {len(ids)} concepts to {rest.strip()}")
991
+
992
+ elif cmd == "export" and rest:
993
+ tokens = rest.strip().split(None, 1)
994
+ if len(tokens) == 2:
995
+ cid, out_dir = tokens
996
+ output_path = Path(out_dir) / f"{cid}.md"
997
+ router.export_to_okf(cid, output_path)
998
+ print(f"[OK] Exported {cid} → {output_path}")
999
+ else:
1000
+ print("Usage: export <concept_id> <output_dir>")
1001
+
1002
+ elif cmd == "model-info":
1003
+ info = OKFRouter.model_info(cache_dir=router.cache_dir)
1004
+ print(f"Model: {info['model_id']}")
1005
+ print(f"Cache: {info['cache_dir']}")
1006
+ if info["cached"]:
1007
+ size_gb = info["disk_usage_bytes"] / (1024 ** 3)
1008
+ print(f"Status: cached ({size_gb:.2f} GB)")
1009
+ print(f"Path: {info['snapshot_path']}")
1010
+ else:
1011
+ print("Status: not cached (will download on first use)")
1012
+
1013
+ elif cmd == "broken-links":
1014
+ broken = router.list_broken_links()
1015
+ if not broken:
1016
+ print("No broken links found.")
1017
+ else:
1018
+ print(f"Found {len(broken)} broken link(s):")
1019
+ for link in broken:
1020
+ print(f" {link['source']} → {link['target']}")
1021
+
1022
+ elif cmd == "repair-links":
1023
+ count = router.repair_links()
1024
+ print(f"[OK] Repaired {count} link(s)")
1025
+
1026
+ elif cmd == "ingest" and rest:
1027
+ # Minimal shell dispatch for ingest — delegates to the CLI handler.
1028
+ from okfgraph.cli import _ingest
1029
+ from types import SimpleNamespace
1030
+ src_path = rest.strip()
1031
+ is_md = src_path.lower().endswith(".md")
1032
+ shell_args = SimpleNamespace(
1033
+ kind="md" if is_md else "pdf",
1034
+ md_file=src_path if is_md else None,
1035
+ pdf_file=None if is_md else src_path,
1036
+ thoughts=None,
1037
+ topic=None,
1038
+ concept_id=None,
1039
+ title=None,
1040
+ description=None,
1041
+ tags=None,
1042
+ auto_import=False,
1043
+ output=None,
1044
+ routing_mode="auto",
1045
+ mode="text",
1046
+ batch_size=32,
1047
+ purge=False,
1048
+ no_extract_images=False,
1049
+ db=args.db,
1050
+ bundle=args.bundle,
1051
+ dim=args.dim,
1052
+ cache_dir=getattr(args, "cache_dir", None),
1053
+ device=getattr(args, "device", "cpu"),
1054
+ omni_model_id=getattr(args, "omni_model_id", None),
1055
+ chunk_size=getattr(args, "chunk_size", 512),
1056
+ chunk_overlap=getattr(args, "chunk_overlap", 40),
1057
+ no_chunking=False,
1058
+ allow_remote_images=False,
1059
+ )
1060
+ _ingest(shell_args)
1061
+
1062
+ else:
1063
+ print(f"Unknown command: {cmd}. Type 'help' for usage.")
1064
+
1065
+
1066
+ # ── argument parser ────────────────────────────────────────────────────────
1067
+
1068
+ def _global_options_epilog() -> str:
1069
+ """Render the global flags once for top-level ``okf --help``.
1070
+
1071
+ Single source of truth: builds a throwaway parser with the same helpers
1072
+ (unmarked, so nothing is hidden) and reuses its options section.
1073
+ """
1074
+ probe = argparse.ArgumentParser(prog="okf")
1075
+ _add_global(probe, mark=False)
1076
+ _add_logging_flags(probe, mark=False)
1077
+ text = probe.format_help()
1078
+ try:
1079
+ body = text.split("options:", 1)[1]
1080
+ except IndexError: # pragma: no cover - Python <3.11 wording
1081
+ body = text.split("optional arguments:", 1)[1]
1082
+ return "Global options (every command; may also come from okfgraph.toml):" + body
1083
+
1084
+
1085
+ def build_parser():
1086
+ parser = argparse.ArgumentParser(
1087
+ prog="okf",
1088
+ description="OKF Knowledge Graph CLI — LadybugDB + Jina v5 embeddings",
1089
+ epilog=_global_options_epilog(),
1090
+ formatter_class=argparse.RawDescriptionHelpFormatter,
1091
+ )
1092
+ sub = parser.add_subparsers(dest="command", help="Command to run", parser_class=_SubParser)
1093
+
1094
+ # init
1095
+ p = sub.add_parser("init", help="Initialize database and schema")
1096
+ _add_global(p)
1097
+ _add_logging_flags(p)
1098
+
1099
+ # model-info
1100
+ p = sub.add_parser("model-info", help="Show model cache status")
1101
+ _add_global(p)
1102
+ _add_logging_flags(p)
1103
+ p.add_argument("--model-id", default="jinaai/jina-embeddings-v5-text-small-retrieval", help="Model ID to inspect")
1104
+
1105
+ # import
1106
+ p = sub.add_parser("import", help="Import OKF files")
1107
+ _add_global(p)
1108
+ _add_logging_flags(p)
1109
+ p.add_argument("files", nargs="*", help="Files to import")
1110
+ p.add_argument("--all", action="store_true", dest="import_all", help="Import entire bundle")
1111
+ p.add_argument("--batch-size", type=int, default=32, help="Batch size for encoding (default: 32)")
1112
+ p.add_argument(
1113
+ "--mode", default="text", choices=["text", "optional", "omni"],
1114
+ help="Image ingestion mode: text (alt-text/filename, no omni), "
1115
+ "optional (omni only for images lacking alt-text), "
1116
+ "omni (omni for every image). Default: text",
1117
+ )
1118
+ p.add_argument(
1119
+ "--purge", action="store_true", default=False,
1120
+ help="Also purge concepts whose source files were deleted from disk "
1121
+ "(removes concept, chunks, links, and orphaned image assets)",
1122
+ )
1123
+
1124
+ # search (unified: concepts, chunks, images)
1125
+ p = sub.add_parser("search", help="Search concepts, chunks, or images")
1126
+ _add_global(p)
1127
+ _add_logging_flags(p)
1128
+ p.add_argument("query", help="Search query")
1129
+ p.add_argument("--target", default="concepts", choices=["concepts", "chunks", "images"],
1130
+ help="What to search (default: concepts)")
1131
+ p.add_argument("--limit", type=int, default=10, help="Max results (default: 10)")
1132
+ p.add_argument("--type", help="Concept type filter (concepts/chunks)")
1133
+ p.add_argument("--tags", help="Comma-separated tag filters (concepts/chunks)")
1134
+ p.add_argument("--parent", help="Parent directory ID (concepts/chunks)")
1135
+ p.add_argument("--chunks", action="store_true", help="Include matched chunks per concept result")
1136
+ p.add_argument("--expand", action="store_true", help="Chunks: attach graph neighborhood to each hit")
1137
+ p.add_argument("--context-hops", type=int, default=1, help="Expansion hops with --expand (default: 1)")
1138
+ p.add_argument("--hub-rerank", action="store_true", help="Chunks: rerank by graph hub score")
1139
+ p.add_argument("--hub-weight", type=float, default=0.3, help="Hub weight with --hub-rerank (default: 0.3)")
1140
+ p.add_argument("--use-omni", action="store_true", help="Images: encode query with omni text side")
1141
+ p.add_argument("--rank", default="none", choices=["none", "hub", "ppr"],
1142
+ help="Concepts: ranking — none (RRF order), hub (blend incoming-link "
1143
+ "authority), ppr (model-free lexical-seed PPR, no ONNX load). "
1144
+ "Default: none")
1145
+
1146
+ # read (unified: body, chunks, document, context)
1147
+ p = sub.add_parser("read", help="Read a concept: body, chunks, document, or context")
1148
+ _add_global(p)
1149
+ _add_logging_flags(p)
1150
+ p.add_argument("concept_id", help="Concept ID")
1151
+ p.add_argument("--include", default="body", choices=["body", "chunks", "document", "context"],
1152
+ help="What to return (default: body)")
1153
+ p.add_argument("--output", help="Output file for --include document (default: stdout)")
1154
+ p.add_argument("--max-tokens", type=int, default=None,
1155
+ help="Token budget: assemble self + PPR-ranked neighbours "
1156
+ "(index-first for context), truncating to fit")
1157
+
1158
+ # traverse (unified: relationships, directory listing, shortest path)
1159
+ p = sub.add_parser("traverse", help="Traverse relationships, list directories, find paths")
1160
+ _add_global(p)
1161
+ _add_logging_flags(p)
1162
+ p.add_argument("start_id", nargs="?", default="", help="Starting concept or directory ID (empty = root listing)")
1163
+ p.add_argument("--relationship", default="CONTAINS", choices=["CONTAINS", "LINKS_TO", "PART_OF", "INCLUDES_ASSET"])
1164
+ p.add_argument("--direction", default="OUTGOING", choices=["OUTGOING", "INCOMING", "BOTH"])
1165
+ p.add_argument("--depth", type=int, default=1, help="Max depth (1-5)")
1166
+ p.add_argument("--type", help="Target node type filter")
1167
+ p.add_argument("--target", default=None, help="Find shortest path from start_id to this ID instead of traversing")
1168
+ p.add_argument("--max-path-length", type=int, default=6, help="Max path length with --target (default: 6)")
1169
+
1170
+ # ingest (unified: markdown, PDF, thoughts)
1171
+ p = sub.add_parser("ingest", help="Add content: markdown file, PDF, or thoughts")
1172
+ _add_global(p)
1173
+ _add_logging_flags(p)
1174
+ p.add_argument("--kind", default="pdf", choices=["md", "pdf", "thoughts"],
1175
+ help="What to ingest (default: pdf)")
1176
+ p.add_argument("--md-file", default=None, help="Markdown file (--kind md)")
1177
+ p.add_argument("--pdf-file", default=None, help="PDF file (--kind pdf)")
1178
+ p.add_argument("--thoughts", default=None, help="Raw reasoning text (--kind thoughts)")
1179
+ p.add_argument("--topic", default=None, help="Topic (--kind thoughts)")
1180
+ p.add_argument("--concept-id", default=None, help="Explicit concept ID (--kind md/thoughts)")
1181
+ p.add_argument("--title", default=None, help="Title override (--kind md)")
1182
+ p.add_argument("--description", default=None, help="Description override (--kind md)")
1183
+ p.add_argument("--tags", default=None, help="Comma-separated tags (--kind md/thoughts)")
1184
+ p.add_argument(
1185
+ "--auto-import", action="store_true",
1186
+ help="Auto-import converted markdown into the graph (--kind pdf)",
1187
+ )
1188
+ p.add_argument(
1189
+ "--output", default=None,
1190
+ help="Output directory for converted markdown (default: current dir)",
1191
+ )
1192
+ p.add_argument(
1193
+ "--routing-mode", default="auto",
1194
+ choices=["auto", "surgical", "always", "never"],
1195
+ help="ONNX routing mode (--kind pdf, default: auto)",
1196
+ )
1197
+ p.add_argument(
1198
+ "--mode", default="text", choices=["text", "optional", "omni"],
1199
+ help="Image ingestion mode (default: text)",
1200
+ )
1201
+ p.add_argument(
1202
+ "--batch-size", type=int, default=32,
1203
+ help="Batch size for encoding during auto-import (default: 32)",
1204
+ )
1205
+ p.add_argument(
1206
+ "--purge", action="store_true", default=False,
1207
+ help="Purge deleted concepts during auto-import",
1208
+ )
1209
+ p.add_argument(
1210
+ "--no-extract-images", action="store_true",
1211
+ help="Do not extract embedded images from the PDF",
1212
+ )
1213
+
1214
+ # export
1215
+ p = sub.add_parser("export", help="Export concepts")
1216
+ _add_global(p)
1217
+ _add_logging_flags(p)
1218
+ p.add_argument("--all", action="store_true", dest="export_all", help="Export entire bundle")
1219
+ p.add_argument("--output", required=True, help="Output directory")
1220
+ p.add_argument("--concept-id", help="Concept ID (for single export)")
1221
+ p.add_argument("--type", help="Concept type filter")
1222
+ p.add_argument("--tags", help="Comma-separated tag filters")
1223
+ p.add_argument("--parent", help="Parent directory ID")
1224
+ p.add_argument("--flavor", default="okf", choices=["okf", "obsidian"],
1225
+ help="Link flavor: okf ([t](id.md) + index files) or obsidian "
1226
+ "([[Title]] wikilinks, no index files). Default: okf")
1227
+
1228
+ # diff (structural: snapshot dir-vs-dir, or drift graph-vs-dir)
1229
+ p = sub.add_parser("diff", help="Structural diff: concepts/edges/broken-link deltas")
1230
+ _add_global(p)
1231
+ _add_logging_flags(p)
1232
+ p.add_argument("old", nargs="?", default=None,
1233
+ help="Old side: bundle dir (with NEW: snapshot; alone: drift vs graph)")
1234
+ p.add_argument("new", nargs="?", default=None, help="New side: bundle dir (snapshot mode)")
1235
+ p.add_argument("--json", action="store_true", help="Machine-readable report")
1236
+
1237
+ # doctor (scored health + safe repairs)
1238
+ p = sub.add_parser("doctor", help="Health scan: score, findings, safe --fix")
1239
+ _add_global(p)
1240
+ _add_logging_flags(p)
1241
+ p.add_argument("--strict", action="store_true",
1242
+ help="Exit 1 when any finding exists (CI gate)")
1243
+ p.add_argument("--fix", action="store_true",
1244
+ help="Apply safe repairs (link re-points, timestamp normalization; "
1245
+ "never touches reviewed:true concepts)")
1246
+ p.add_argument("--stale-days", type=int, default=365,
1247
+ help="Age threshold for 'stale' findings (default: 365)")
1248
+ p.add_argument("--json", action="store_true", help="Machine-readable report")
1249
+
1250
+ # lint (pre-import bundle gate: no DB, no model load)
1251
+ p = sub.add_parser("lint", help="Validate bundle frontmatter + links before import")
1252
+ _add_global(p)
1253
+ _add_logging_flags(p)
1254
+ p.add_argument("dir", nargs="?", default=None,
1255
+ help="Bundle directory (default: --bundle, okfgraph.toml, or .)")
1256
+ p.add_argument("--json", action="store_true", help="Machine-readable report")
1257
+
1258
+ # shell
1259
+ p = sub.add_parser("shell", help="Interactive REPL")
1260
+ _add_global(p)
1261
+ _add_logging_flags(p)
1262
+
1263
+ # broken-links
1264
+ p = sub.add_parser("broken-links", help="List broken (orphan) links")
1265
+ _add_global(p)
1266
+ _add_logging_flags(p)
1267
+
1268
+ # repair-links
1269
+ p = sub.add_parser("repair-links", help="Repair broken links by re-checking targets")
1270
+ _add_global(p)
1271
+ _add_logging_flags(p)
1272
+
1273
+ # reindex
1274
+ p = sub.add_parser("reindex", help="Rebuild vector + FTS search indexes")
1275
+ _add_global(p)
1276
+ _add_logging_flags(p)
1277
+ p.add_argument("--if-dirty", action="store_true",
1278
+ help="Only rebuild if data changed since the last index build")
1279
+
1280
+ # Soft-delete commands (Gap #1d)
1281
+ p = sub.add_parser("deleted-list", help="List soft-deleted concepts")
1282
+ _add_global(p)
1283
+ _add_logging_flags(p)
1284
+
1285
+ p = sub.add_parser("deleted-recover", help="Recover a soft-deleted concept")
1286
+ _add_global(p)
1287
+ _add_logging_flags(p)
1288
+ p.add_argument("concept_id", help="Concept ID to recover")
1289
+
1290
+ p = sub.add_parser("deleted-purge", help="Permanently delete expired soft-deleted concepts")
1291
+ _add_global(p)
1292
+ _add_logging_flags(p)
1293
+ p.add_argument("--older-than", type=int, default=None, help="Override recovery window (seconds)")
1294
+
1295
+ return parser
1296
+
1297
+
1298
+ def main():
1299
+ parser = build_parser()
1300
+ args = parser.parse_args()
1301
+
1302
+ if not args.command:
1303
+ parser.print_help()
1304
+ sys.exit(0)
1305
+
1306
+ # Setup logging (Gap #10)
1307
+ _setup_logging(
1308
+ verbose=getattr(args, "verbose", False),
1309
+ quiet=getattr(args, "quiet", False),
1310
+ log_file=getattr(args, "log_file", ""),
1311
+ )
1312
+
1313
+ logger = logging.getLogger("cli")
1314
+
1315
+ # Profile flag (Gap #10C)
1316
+ if getattr(args, "profile", False):
1317
+ logger.info("profiling enabled")
1318
+ profiler = cProfile.Profile()
1319
+ profiler.enable()
1320
+
1321
+ commands = {
1322
+ "init": _init,
1323
+ "model-info": _model_info,
1324
+ "import": _import,
1325
+ "ingest": _ingest,
1326
+ "search": _search,
1327
+ "read": _read,
1328
+ "traverse": _traverse,
1329
+ "export": _export,
1330
+ "shell": _shell,
1331
+ "broken-links": _broken_links,
1332
+ "repair-links": _repair_links,
1333
+ "diff": _diff,
1334
+ "doctor": _doctor,
1335
+ "lint": _lint,
1336
+ "reindex": _reindex,
1337
+ "deleted-list": _deleted_list,
1338
+ "deleted-recover": _deleted_recover,
1339
+ "deleted-purge": _deleted_purge,
1340
+ }
1341
+
1342
+ try:
1343
+ ret = commands[args.command](args)
1344
+ finally:
1345
+ _close_routers()
1346
+ _teardown_logging()
1347
+ # Handlers return an int exit code when CI-gateable (diff/doctor);
1348
+ # everything else returns None (= success).
1349
+ if isinstance(ret, int) and ret != 0:
1350
+ sys.exit(ret)
1351
+
1352
+ # Profile output (Gap #10C)
1353
+ if getattr(args, "profile", False):
1354
+ profiler.disable()
1355
+ import pstats
1356
+ stream = io.StringIO()
1357
+ stats = pstats.Stats(profiler, stream=stream)
1358
+ stats.sort_stats("cumulative")
1359
+ stats.print_stats(20)
1360
+ print(stream.getvalue(), file=sys.stderr)
1361
+
1362
+
1363
+ if __name__ == "__main__":
1364
+ main()