okfsmith 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
okfsmith/__init__.py ADDED
@@ -0,0 +1,8 @@
1
+ """okfsmith — Documents → OKF knowledge bundles.
2
+
3
+ Converts messy documents (PDFs, markdown, wikis, exports) into Google's
4
+ Open Knowledge Format (OKF) v0.2 knowledge bundles: markdown files with
5
+ YAML frontmatter, plus reserved ``index.md`` / ``log.md`` files.
6
+ """
7
+
8
+ __version__ = "0.1.0"
@@ -0,0 +1,5 @@
1
+ """okfsmith CLI package."""
2
+
3
+ # Importing commands registers them on the Typer app defined in
4
+ # ``okfsmith.cli.app`` (the app module itself stays command-free).
5
+ from okfsmith.cli import commands # noqa: F401
okfsmith/cli/app.py ADDED
@@ -0,0 +1,36 @@
1
+ """okfsmith command-line interface.
2
+
3
+ Minimal Typer app: only the ``--version`` callback lives here for now.
4
+ The CLI engineer adds commands later — do not add commands in this module.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import typer
10
+
11
+ from okfsmith import __version__
12
+
13
+ app = typer.Typer(
14
+ help="Convert messy documents into OKF v0.2 knowledge bundles.",
15
+ add_completion=False,
16
+ )
17
+
18
+
19
+ def _version_callback(value: bool) -> None:
20
+ """Print the version and exit when ``--version`` is passed."""
21
+ if value:
22
+ typer.echo(f"okfsmith {__version__}")
23
+ raise typer.Exit()
24
+
25
+
26
+ @app.callback()
27
+ def main(
28
+ version: bool | None = typer.Option(
29
+ None,
30
+ "--version",
31
+ callback=_version_callback,
32
+ is_eager=True,
33
+ help="Show the okfsmith version and exit.",
34
+ ),
35
+ ) -> None:
36
+ """okfsmith — Documents → OKF knowledge bundles."""
@@ -0,0 +1,483 @@
1
+ """Typer commands for okfsmith.
2
+
3
+ This module only wires user input to business logic: the real work lives in
4
+ ``okfsmith.core`` and the sibling slices (``parsers``, ``extract``,
5
+ ``validate``, ``viz``, ``mcp_server``), which are imported lazily so each
6
+ command fails cleanly when its slice is not installed.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import importlib
12
+ import json
13
+ from pathlib import Path
14
+ from typing import Any
15
+
16
+ import typer
17
+ from rich.console import Console
18
+ from rich.table import Table
19
+
20
+ from okfsmith import links as _links
21
+ from okfsmith.cli.app import app
22
+ from okfsmith.core import Bundle, indexlog
23
+ from okfsmith.core import frontmatter as _fm
24
+ from okfsmith.core import spec as _spec
25
+
26
+ console = Console()
27
+
28
+
29
+ def _lazy_attr(module_name: str, attr: str) -> Any:
30
+ """Import *attr* from *module_name*, failing cleanly if the slice is absent.
31
+
32
+ Sibling slices (``okfsmith.parsers``, ``okfsmith.extract``,
33
+ ``okfsmith.validate``, ``okfsmith.viz``, ``okfsmith.mcp_server``) are
34
+ imported lazily; when one is not installed the command exits 1 with a
35
+ clear message instead of an ImportError traceback. *module_name* may be a
36
+ dotted submodule path (e.g. ``okfsmith.parsers.ingest_no_llm``).
37
+ """
38
+ try:
39
+ module = importlib.import_module(module_name)
40
+ except ModuleNotFoundError as exc:
41
+ missing = exc.name or ""
42
+ # The slice itself (or one of its parents under okfsmith) is absent.
43
+ if module_name == missing or module_name.startswith(missing + "."):
44
+ typer.echo(
45
+ f"error: '{module_name}' is not available in this installation.",
46
+ err=True,
47
+ )
48
+ raise typer.Exit(code=1)
49
+ raise
50
+ try:
51
+ return getattr(module, attr)
52
+ except AttributeError:
53
+ typer.echo(
54
+ f"error: '{module_name}' does not provide '{attr}'.", err=True
55
+ )
56
+ raise typer.Exit(code=1)
57
+
58
+
59
+ def _require_dir(path: Path, what: str = "directory") -> None:
60
+ if not path.is_dir():
61
+ typer.echo(f"error: {what} '{path}' does not exist.", err=True)
62
+ raise typer.Exit(code=1)
63
+
64
+
65
+ def _collect_inputs(source: Path, recursive: bool) -> list[Path]:
66
+ if source.is_file():
67
+ return [source]
68
+ iterator = source.rglob("*") if recursive else source.iterdir()
69
+ return sorted(p for p in iterator if p.is_file())
70
+
71
+
72
+ # ---------------------------------------------------------------------------
73
+ # init
74
+ # ---------------------------------------------------------------------------
75
+
76
+
77
+ @app.command()
78
+ def init(
79
+ directory: Path = typer.Argument(
80
+ ..., help="Directory to scaffold the bundle in."
81
+ ),
82
+ force: bool = typer.Option(
83
+ False, "--force", help="Scaffold even if the directory exists and is non-empty."
84
+ ),
85
+ ) -> None:
86
+ """Create a new, empty OKF bundle in DIRECTORY."""
87
+ if directory.exists() and any(directory.iterdir()) and not force:
88
+ typer.echo(
89
+ f"error: '{directory}' exists and is not empty "
90
+ "(use --force to scaffold anyway).",
91
+ err=True,
92
+ )
93
+ raise typer.Exit(code=1)
94
+ directory.mkdir(parents=True, exist_ok=True)
95
+ bundle = Bundle(directory)
96
+ index_path = indexlog.ensure_index(bundle)
97
+ log_path = indexlog.append_log(
98
+ bundle, kind="Creation", message="Bundle created with `okfsmith init`."
99
+ )
100
+ typer.echo(f"Initialized OKF bundle in {directory}")
101
+ typer.echo(f" index: {index_path}")
102
+ typer.echo(f" log: {log_path}")
103
+
104
+
105
+ # ---------------------------------------------------------------------------
106
+ # ingest
107
+ # ---------------------------------------------------------------------------
108
+
109
+
110
+ def _ingest_no_llm_one(
111
+ path: Path,
112
+ target: Bundle,
113
+ *,
114
+ parse_file: Any,
115
+ ingest_no_llm: Any,
116
+ sha256_of: Any,
117
+ already_ingested: Any,
118
+ record_ingested: Any,
119
+ ) -> tuple[str, int]:
120
+ """Ingest one file via deterministic sectioning. Returns (status, count)."""
121
+ digest = sha256_of(path)
122
+ if already_ingested(target, digest):
123
+ return "skipped (already ingested)", 0
124
+ parsed = parse_file(path)
125
+ created = ingest_no_llm(target, parsed, str(path))
126
+ record_ingested(target, digest, str(path))
127
+ return "ok", len(created)
128
+
129
+
130
+ def _ingest_llm_one(
131
+ path: Path,
132
+ target: Bundle,
133
+ *,
134
+ parse_file: Any,
135
+ section: Any,
136
+ SectionInput: Any,
137
+ run: Any,
138
+ model: str | None,
139
+ ) -> tuple[str, int]:
140
+ """Ingest one file via the LLM extraction pipeline. Returns (status, count)."""
141
+ parsed = parse_file(path)
142
+ sectioned = section(parsed)
143
+ doc_title = (parsed.meta or {}).get("title") or path.stem
144
+ sections = [
145
+ SectionInput(
146
+ title=sec.title,
147
+ level=sec.level,
148
+ text=sec.text,
149
+ page_span=sec.page_span,
150
+ tables=list(sec.tables or []),
151
+ source_id=str(path),
152
+ source_path=str(path),
153
+ doc_title=doc_title,
154
+ )
155
+ for sec in sectioned.sections
156
+ ]
157
+ created = run(target, sections, model=model)
158
+ return "ok", len(created)
159
+
160
+
161
+ @app.command()
162
+ def ingest(
163
+ source: Path = typer.Argument(..., help="File or directory to ingest."),
164
+ bundle: Path = typer.Option(..., "--bundle", help="Target bundle directory."),
165
+ model: str | None = typer.Option(
166
+ None, "--model", help="Model to use for LLM extraction."
167
+ ),
168
+ no_llm: bool = typer.Option(
169
+ False,
170
+ "--no-llm",
171
+ help="Use deterministic sectioning (okfsmith.parsers) instead of an LLM.",
172
+ ),
173
+ recursive: bool = typer.Option(
174
+ False,
175
+ "--recursive",
176
+ help="Recurse into subdirectories when SOURCE is a directory.",
177
+ ),
178
+ ) -> None:
179
+ """Ingest documents into the bundle as draft concepts.
180
+
181
+ With ``--no-llm`` the deterministic sectioning path is used
182
+ (``okfsmith.parsers.ingest_no_llm``) with per-file SHA-256 dedup via the
183
+ bundle manifest; otherwise the LLM extraction path
184
+ (``okfsmith.extract.run``) is used.
185
+ """
186
+ if not source.exists():
187
+ typer.echo(f"error: source '{source}' does not exist.", err=True)
188
+ raise typer.Exit(code=1)
189
+ files = _collect_inputs(source, recursive)
190
+ if not files:
191
+ typer.echo(f"error: no input files found under '{source}'.", err=True)
192
+ raise typer.Exit(code=1)
193
+
194
+ parse_file = _lazy_attr("okfsmith.parsers", "parse_file")
195
+ if no_llm:
196
+ ingest_no_llm = _lazy_attr(
197
+ "okfsmith.parsers.ingest_no_llm", "ingest_no_llm"
198
+ )
199
+ sha256_of = _lazy_attr("okfsmith.parsers.dedup", "sha256_of")
200
+ already_ingested = _lazy_attr("okfsmith.parsers.dedup", "already_ingested")
201
+ record_ingested = _lazy_attr("okfsmith.parsers.dedup", "record_ingested")
202
+ mode = "sectioning (no LLM)"
203
+ else:
204
+ section = _lazy_attr("okfsmith.parsers.sectioning", "section")
205
+ SectionInput = _lazy_attr("okfsmith.extract", "SectionInput")
206
+ run = _lazy_attr("okfsmith.extract", "run")
207
+ LLMUnavailableError = _lazy_attr("okfsmith.extract", "LLMUnavailableError")
208
+ mode = f"LLM extraction (model={model or 'default'})"
209
+
210
+ target = Bundle.load(bundle)
211
+ table = Table(title=f"Ingest summary — {mode}")
212
+ table.add_column("File")
213
+ table.add_column("SHA-256")
214
+ table.add_column("Concepts", justify="right")
215
+ table.add_column("Status")
216
+
217
+ digest_of = _lazy_attr("okfsmith.parsers.dedup", "sha256_of")
218
+ failures: list[Path] = []
219
+ created_total = 0
220
+ for path in files:
221
+ digest = digest_of(path)[:12]
222
+ try:
223
+ if no_llm:
224
+ status, count = _ingest_no_llm_one(
225
+ path,
226
+ target,
227
+ parse_file=parse_file,
228
+ ingest_no_llm=ingest_no_llm,
229
+ sha256_of=digest_of,
230
+ already_ingested=already_ingested,
231
+ record_ingested=record_ingested,
232
+ )
233
+ style = "green" if status == "ok" else "yellow"
234
+ table.add_row(str(path), digest, str(count), f"[{style}]{status}[/{style}]")
235
+ else:
236
+ try:
237
+ status, count = _ingest_llm_one(
238
+ path,
239
+ target,
240
+ parse_file=parse_file,
241
+ section=section,
242
+ SectionInput=SectionInput,
243
+ run=run,
244
+ model=model,
245
+ )
246
+ except LLMUnavailableError as exc:
247
+ typer.echo(f"error: LLM unavailable: {exc}", err=True)
248
+ raise typer.Exit(code=1)
249
+ table.add_row(str(path), digest, str(count), "[green]ok[/green]")
250
+ created_total += count
251
+ except typer.Exit:
252
+ raise
253
+ except Exception as exc: # noqa: BLE001 — per-file failure, keep going
254
+ failures.append(path)
255
+ table.add_row(str(path), digest, "0", f"[red]failed: {exc}[/red]")
256
+ console.print(table)
257
+
258
+ if created_total:
259
+ indexlog.ensure_index(target)
260
+ indexlog.append_log(
261
+ target,
262
+ kind="Update",
263
+ message=(
264
+ f"Ingested {created_total} draft concept(s) from "
265
+ f"{len(files) - len(failures)} source file(s)."
266
+ ),
267
+ )
268
+ typer.echo(f"Wrote {created_total} draft concept(s) to {bundle}")
269
+ if failures and len(failures) == len(files):
270
+ typer.echo("error: all inputs failed to ingest.", err=True)
271
+ raise typer.Exit(code=1)
272
+ if failures:
273
+ typer.echo(
274
+ f"warning: {len(failures)} of {len(files)} input(s) failed.", err=True
275
+ )
276
+
277
+
278
+ # ---------------------------------------------------------------------------
279
+ # validate
280
+ # ---------------------------------------------------------------------------
281
+
282
+
283
+ @app.command()
284
+ def validate(
285
+ directory: Path = typer.Argument(..., help="Bundle directory to validate."),
286
+ strict: bool = typer.Option(
287
+ False, "--strict", help="Treat warnings as failures."
288
+ ),
289
+ output_format: str = typer.Option(
290
+ "text", "--format", help="Output format: text or json."
291
+ ),
292
+ ) -> None:
293
+ """Validate a bundle against OKF v0.2 (E001–E004 / W001–W015)."""
294
+ _require_dir(directory)
295
+ # check(bundle_path) -> ValidationReport with .errors / .warnings as
296
+ # Finding objects; serialize each via Finding.as_dict().
297
+ check = _lazy_attr("okfsmith.validate", "check")
298
+ result = check(directory)
299
+ errors = [finding.as_dict() for finding in result.errors]
300
+ warnings = [finding.as_dict() for finding in result.warnings]
301
+
302
+ if output_format == "json":
303
+ typer.echo(json.dumps({"errors": errors, "warnings": warnings}, indent=2))
304
+ elif output_format == "text":
305
+ _print_report(errors, warnings)
306
+ else:
307
+ typer.echo(f"error: unknown --format '{output_format}' (use text or json).", err=True)
308
+ raise typer.Exit(code=1)
309
+
310
+ if errors or (strict and warnings):
311
+ raise typer.Exit(code=1)
312
+
313
+
314
+ def _print_report(errors: list[dict], warnings: list[dict]) -> None:
315
+ for title, items, style in (
316
+ ("Errors", errors, "red"),
317
+ ("Warnings", warnings, "yellow"),
318
+ ):
319
+ table = Table(title=f"{title} ({len(items)})")
320
+ table.add_column("Code", style=style)
321
+ table.add_column("File")
322
+ table.add_column("Message")
323
+ for item in items:
324
+ table.add_row(
325
+ str(item.get("code", "")),
326
+ str(item.get("file", "")),
327
+ str(item.get("message", "")),
328
+ )
329
+ console.print(table)
330
+ if errors:
331
+ typer.echo(f"INVALID: {len(errors)} error(s), {len(warnings)} warning(s).")
332
+ elif warnings:
333
+ typer.echo(f"Conformant with {len(warnings)} warning(s).")
334
+ else:
335
+ typer.echo("Conformant: no errors, no warnings.")
336
+
337
+
338
+ # ---------------------------------------------------------------------------
339
+ # list
340
+ # ---------------------------------------------------------------------------
341
+
342
+
343
+ @app.command(name="list")
344
+ def list_concepts(
345
+ directory: Path = typer.Argument(..., help="Bundle directory to list."),
346
+ type_filter: str | None = typer.Option(
347
+ None, "--type", help="Only show concepts of this type."
348
+ ),
349
+ tier_filter: str | None = typer.Option(
350
+ None, "--tier", help="Only show concepts with this trust tier."
351
+ ),
352
+ ) -> None:
353
+ """List concepts in the bundle: id, type, title, trust tier."""
354
+ _require_dir(directory)
355
+ bundle = Bundle.load(directory)
356
+ table = Table(title=f"Concepts in {directory}")
357
+ table.add_column("ID")
358
+ table.add_column("Type")
359
+ table.add_column("Title")
360
+ table.add_column("Trust tier")
361
+ shown = 0
362
+ for concept in bundle.iter_concepts():
363
+ ctype = str(concept.frontmatter.get("type") or "")
364
+ tier = _spec.trust_tier(concept.frontmatter)
365
+ if type_filter and ctype.casefold() != type_filter.casefold():
366
+ continue
367
+ if tier_filter and tier.casefold() != tier_filter.casefold():
368
+ continue
369
+ table.add_row(
370
+ concept.id, ctype, str(concept.frontmatter.get("title") or ""), tier
371
+ )
372
+ shown += 1
373
+ console.print(table)
374
+ typer.echo(f"{shown} concept(s)")
375
+
376
+
377
+ # ---------------------------------------------------------------------------
378
+ # read
379
+ # ---------------------------------------------------------------------------
380
+
381
+
382
+ @app.command()
383
+ def read(
384
+ directory: Path = typer.Argument(..., help="Bundle directory."),
385
+ concept_id: str = typer.Argument(..., help="Concept id, e.g. finance/revenue."),
386
+ ) -> None:
387
+ """Print a concept: frontmatter as YAML, then the body."""
388
+ _require_dir(directory)
389
+ bundle = Bundle.load(directory)
390
+ concept = bundle.get(concept_id)
391
+ if concept is None:
392
+ typer.echo(
393
+ f"error: concept '{concept_id}' not found in '{directory}'.", err=True
394
+ )
395
+ raise typer.Exit(code=1)
396
+ typer.echo(_fm.serialize_frontmatter(concept.frontmatter, concept.body))
397
+
398
+
399
+ # ---------------------------------------------------------------------------
400
+ # graph
401
+ # ---------------------------------------------------------------------------
402
+
403
+
404
+ @app.command()
405
+ def graph(
406
+ directory: Path = typer.Argument(..., help="Bundle directory."),
407
+ output_format: str = typer.Option(
408
+ "text", "--format", help="Output format: text, json, mermaid, or html."
409
+ ),
410
+ output: Path | None = typer.Option(
411
+ None,
412
+ "--output",
413
+ help="Output file for --format html (default: <bundle>/viz.html).",
414
+ ),
415
+ ) -> None:
416
+ """Show the concept link graph (nodes, edges, orphans, dead links)."""
417
+ _require_dir(directory)
418
+ bundle = Bundle.load(directory)
419
+ data = _links.build_graph(bundle)
420
+
421
+ if output_format == "json":
422
+ adjacency: dict[str, list[str]] = {}
423
+ for node in data["nodes"]:
424
+ adjacency[node["id"]] = []
425
+ for edge in data["edges"]:
426
+ adjacency.setdefault(edge["from"], []).append(edge["to"])
427
+ typer.echo(
428
+ json.dumps({"nodes": data["nodes"], "adjacency": adjacency}, indent=2)
429
+ )
430
+ elif output_format == "mermaid":
431
+ typer.echo(_links.mermaid_flowchart(data), nl=False)
432
+ elif output_format == "html":
433
+ # render_html(root, output) -> Path
434
+ render_html = _lazy_attr("okfsmith.viz", "render_html")
435
+ out_path = output or (Path(directory) / "viz.html")
436
+ written = render_html(Path(directory), out_path)
437
+ typer.echo(f"Wrote {written}")
438
+ elif output_format == "text":
439
+ _print_graph_text(bundle, data)
440
+ else:
441
+ typer.echo(
442
+ f"error: unknown --format '{output_format}' "
443
+ "(use text, json, mermaid, or html).",
444
+ err=True,
445
+ )
446
+ raise typer.Exit(code=1)
447
+
448
+
449
+ def _print_graph_text(bundle: Bundle, data: dict) -> None:
450
+ typer.echo(
451
+ f"{len(data['nodes'])} concept(s), {len(data['edges'])} link(s), "
452
+ f"{len(data['dead_links'])} dead link(s)."
453
+ )
454
+ orphan_ids = _links.orphans(bundle)
455
+ typer.echo(f"\nOrphans ({len(orphan_ids)}):")
456
+ for oid in orphan_ids:
457
+ typer.echo(f" - {oid}")
458
+ if not orphan_ids:
459
+ typer.echo(" (none)")
460
+ typer.echo(f"\nDead links ({len(data['dead_links'])}):")
461
+ for item in data["dead_links"]:
462
+ typer.echo(f" - {item['source']}: {item['target']}")
463
+ if not data["dead_links"]:
464
+ typer.echo(" (none)")
465
+
466
+
467
+ # ---------------------------------------------------------------------------
468
+ # mcp
469
+ # ---------------------------------------------------------------------------
470
+
471
+
472
+ @app.command()
473
+ def mcp(
474
+ bundle: Path = typer.Option(..., "--bundle", help="Bundle directory to serve."),
475
+ transport: str = typer.Option(
476
+ "stdio", "--transport", help="MCP transport to use."
477
+ ),
478
+ ) -> None:
479
+ """Serve the bundle over MCP (Model Context Protocol)."""
480
+ _require_dir(bundle, "bundle")
481
+ # serve(bundle_path, transport) -> None
482
+ serve = _lazy_attr("okfsmith.mcp_server", "serve")
483
+ serve(bundle, transport)
@@ -0,0 +1,20 @@
1
+ """okfsmith.core — bundle I/O, frontmatter, index/log, and OKF v0.2 spec constants.
2
+
3
+ This is the API contract the rest of the team codes against:
4
+
5
+ - :mod:`okfsmith.core.spec` — OKF v0.2 constants (reserved files, required
6
+ key, statuses, trust-tier derivation, actor helpers, timestamps).
7
+ - :mod:`okfsmith.core.frontmatter` — YAML frontmatter parse/serialize
8
+ (unknown keys preserved, key order preserved).
9
+ - :mod:`okfsmith.core.bundle` — :class:`Bundle` / :class:`Concept`: load and
10
+ write concept documents on disk.
11
+ - :mod:`okfsmith.core.indexlog` — generate ``index.md`` and append to
12
+ ``log.md``.
13
+
14
+ No network calls anywhere in core; stdlib + declared dependencies only.
15
+ """
16
+
17
+ from okfsmith.core import frontmatter, indexlog, spec
18
+ from okfsmith.core.bundle import Bundle, Concept
19
+
20
+ __all__ = ["Bundle", "Concept", "frontmatter", "indexlog", "spec"]
@@ -0,0 +1,123 @@
1
+ """Bundle: the on-disk OKF knowledge bundle (concepts + index.md + log.md).
2
+
3
+ A bundle is a directory tree of ``*.md`` concept documents. Reserved files
4
+ (``index.md``, ``log.md`` — see :mod:`okfsmith.core.spec`) are never concepts.
5
+ A concept's id is its path relative to the bundle root, minus the ``.md``
6
+ suffix, with forward slashes (e.g. ``finance/revenue``).
7
+
8
+ No network calls; stdlib only.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import re
14
+ from dataclasses import dataclass, field
15
+ from pathlib import Path
16
+ from typing import Iterator
17
+
18
+ from okfsmith.core import frontmatter as _fm
19
+ from okfsmith.core.spec import RESERVED_FILES
20
+
21
+ _SUFFIX = ".md"
22
+
23
+
24
+ def slugify(value: str) -> str:
25
+ """Slugify *value*: lowercase, runs of non-alphanumerics become ``-``.
26
+
27
+ >>> slugify("Hello, World!")
28
+ 'hello-world'
29
+ """
30
+ slug = re.sub(r"[^a-z0-9]+", "-", value.lower()).strip("-")
31
+ return slug or "untitled"
32
+
33
+
34
+ def concept_path_for(root: Path, concept_id: str) -> Path:
35
+ """Map a concept id to its on-disk path, slugifying each ``/`` segment.
36
+
37
+ ``"Notes/Hello World"`` under ``root`` → ``root/notes/hello-world.md``.
38
+ """
39
+ parts = [slugify(part) for part in concept_id.replace("\\", "/").split("/")]
40
+ return root.joinpath(*parts).with_suffix(_SUFFIX)
41
+
42
+
43
+ @dataclass
44
+ class Concept:
45
+ """A single OKF concept document."""
46
+
47
+ id: str
48
+ """Concept id: path relative to the bundle root, minus ``.md``, forward slashes."""
49
+
50
+ path: Path
51
+ """Absolute path of the ``.md`` file on disk."""
52
+
53
+ frontmatter: dict = field(default_factory=dict)
54
+ """Parsed YAML frontmatter; unknown keys preserved as-is."""
55
+
56
+ body: str = ""
57
+ """Markdown body after the frontmatter block."""
58
+
59
+
60
+ class Bundle:
61
+ """An OKF knowledge bundle rooted at a directory on disk."""
62
+
63
+ def __init__(self, root: str | Path) -> None:
64
+ """Create a bundle handle for *root* (nothing is read yet)."""
65
+ self.root = Path(root).resolve()
66
+ self._concepts: dict[str, Concept] = {}
67
+ self.index_text: str | None = None
68
+ """Raw text of the root ``index.md``, if present when loaded."""
69
+ self.log_text: str | None = None
70
+ """Raw text of the root ``log.md``, if present when loaded."""
71
+
72
+ @classmethod
73
+ def load(cls, root: str | Path) -> "Bundle":
74
+ """Walk *root* and parse every ``*.md`` file into a :class:`Concept`.
75
+
76
+ Files named ``index.md`` / ``log.md`` are skipped as concepts; the
77
+ root ``index.md`` / ``log.md`` are read into ``index_text`` /
78
+ ``log_text``. Missing reserved files are fine (``None``).
79
+ """
80
+ bundle = cls(root)
81
+ bundle.root.mkdir(parents=True, exist_ok=True)
82
+ for md in sorted(bundle.root.rglob(f"*{_SUFFIX}")):
83
+ if md.name in RESERVED_FILES:
84
+ continue
85
+ rel = md.relative_to(bundle.root)
86
+ concept_id = rel.with_suffix("").as_posix()
87
+ data, body = _fm.parse_frontmatter(md.read_text(encoding="utf-8"))
88
+ bundle._concepts[concept_id] = Concept(
89
+ id=concept_id, path=md, frontmatter=data, body=body
90
+ )
91
+ for filename, attr in (("index.md", "index_text"), ("log.md", "log_text")):
92
+ candidate = bundle.root / filename
93
+ if candidate.is_file():
94
+ setattr(bundle, attr, candidate.read_text(encoding="utf-8"))
95
+ return bundle
96
+
97
+ def write_concept(self, concept_id: str, frontmatter: dict, body: str) -> Concept:
98
+ """Write a concept document to disk and register it in this bundle.
99
+
100
+ The id is slugified segment-by-segment to derive the file path
101
+ (parent directories are created). The returned :class:`Concept`'s
102
+ ``id`` is the slugified id, so it round-trips through :meth:`load`.
103
+ Unknown frontmatter keys are written back untouched.
104
+ """
105
+ self.root.mkdir(parents=True, exist_ok=True)
106
+ path = concept_path_for(self.root, concept_id)
107
+ path.parent.mkdir(parents=True, exist_ok=True)
108
+ path.write_text(_fm.serialize_frontmatter(frontmatter, body), encoding="utf-8")
109
+ stored_id = path.relative_to(self.root).with_suffix("").as_posix()
110
+ concept = Concept(
111
+ id=stored_id, path=path, frontmatter=dict(frontmatter), body=body
112
+ )
113
+ self._concepts[stored_id] = concept
114
+ return concept
115
+
116
+ def iter_concepts(self) -> Iterator[Concept]:
117
+ """Yield all concepts, sorted by id for deterministic output."""
118
+ for concept_id in sorted(self._concepts):
119
+ yield self._concepts[concept_id]
120
+
121
+ def get(self, concept_id: str) -> Concept | None:
122
+ """Return the concept with *concept_id*, or ``None`` if absent."""
123
+ return self._concepts.get(concept_id)