okfsmith 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- okfsmith/__init__.py +8 -0
- okfsmith/cli/__init__.py +5 -0
- okfsmith/cli/app.py +36 -0
- okfsmith/cli/commands.py +483 -0
- okfsmith/core/__init__.py +20 -0
- okfsmith/core/bundle.py +123 -0
- okfsmith/core/frontmatter.py +58 -0
- okfsmith/core/indexlog.py +129 -0
- okfsmith/core/spec.py +107 -0
- okfsmith/extract/__init__.py +47 -0
- okfsmith/extract/human_review.py +52 -0
- okfsmith/extract/llm.py +260 -0
- okfsmith/extract/pipeline.py +668 -0
- okfsmith/extract/prompts.py +147 -0
- okfsmith/links.py +133 -0
- okfsmith/mcp_server/__init__.py +19 -0
- okfsmith/mcp_server/server.py +314 -0
- okfsmith/parsers/__init__.py +75 -0
- okfsmith/parsers/dedup.py +81 -0
- okfsmith/parsers/ingest_no_llm.py +120 -0
- okfsmith/parsers/notion.py +157 -0
- okfsmith/parsers/office.py +203 -0
- okfsmith/parsers/pdf.py +91 -0
- okfsmith/parsers/router.py +69 -0
- okfsmith/parsers/sectioning.py +116 -0
- okfsmith/validate/__init__.py +95 -0
- okfsmith/validate/rules.py +728 -0
- okfsmith/viz/__init__.py +644 -0
- okfsmith-0.1.0.dist-info/METADATA +171 -0
- okfsmith-0.1.0.dist-info/RECORD +34 -0
- okfsmith-0.1.0.dist-info/WHEEL +5 -0
- okfsmith-0.1.0.dist-info/entry_points.txt +2 -0
- okfsmith-0.1.0.dist-info/licenses/LICENSE +201 -0
- okfsmith-0.1.0.dist-info/top_level.txt +1 -0
okfsmith/__init__.py
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""okfsmith — Documents → OKF knowledge bundles.
|
|
2
|
+
|
|
3
|
+
Converts messy documents (PDFs, markdown, wikis, exports) into Google's
|
|
4
|
+
Open Knowledge Format (OKF) v0.2 knowledge bundles: markdown files with
|
|
5
|
+
YAML frontmatter, plus reserved ``index.md`` / ``log.md`` files.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
__version__ = "0.1.0"
|
okfsmith/cli/__init__.py
ADDED
okfsmith/cli/app.py
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""okfsmith command-line interface.
|
|
2
|
+
|
|
3
|
+
Minimal Typer app: only the ``--version`` callback lives here for now.
|
|
4
|
+
The CLI engineer adds commands later — do not add commands in this module.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import typer
|
|
10
|
+
|
|
11
|
+
from okfsmith import __version__
|
|
12
|
+
|
|
13
|
+
app = typer.Typer(
|
|
14
|
+
help="Convert messy documents into OKF v0.2 knowledge bundles.",
|
|
15
|
+
add_completion=False,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _version_callback(value: bool) -> None:
|
|
20
|
+
"""Print the version and exit when ``--version`` is passed."""
|
|
21
|
+
if value:
|
|
22
|
+
typer.echo(f"okfsmith {__version__}")
|
|
23
|
+
raise typer.Exit()
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@app.callback()
|
|
27
|
+
def main(
|
|
28
|
+
version: bool | None = typer.Option(
|
|
29
|
+
None,
|
|
30
|
+
"--version",
|
|
31
|
+
callback=_version_callback,
|
|
32
|
+
is_eager=True,
|
|
33
|
+
help="Show the okfsmith version and exit.",
|
|
34
|
+
),
|
|
35
|
+
) -> None:
|
|
36
|
+
"""okfsmith — Documents → OKF knowledge bundles."""
|
okfsmith/cli/commands.py
ADDED
|
@@ -0,0 +1,483 @@
|
|
|
1
|
+
"""Typer commands for okfsmith.
|
|
2
|
+
|
|
3
|
+
This module only wires user input to business logic: the real work lives in
|
|
4
|
+
``okfsmith.core`` and the sibling slices (``parsers``, ``extract``,
|
|
5
|
+
``validate``, ``viz``, ``mcp_server``), which are imported lazily so each
|
|
6
|
+
command fails cleanly when its slice is not installed.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import importlib
|
|
12
|
+
import json
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import Any
|
|
15
|
+
|
|
16
|
+
import typer
|
|
17
|
+
from rich.console import Console
|
|
18
|
+
from rich.table import Table
|
|
19
|
+
|
|
20
|
+
from okfsmith import links as _links
|
|
21
|
+
from okfsmith.cli.app import app
|
|
22
|
+
from okfsmith.core import Bundle, indexlog
|
|
23
|
+
from okfsmith.core import frontmatter as _fm
|
|
24
|
+
from okfsmith.core import spec as _spec
|
|
25
|
+
|
|
26
|
+
console = Console()
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _lazy_attr(module_name: str, attr: str) -> Any:
|
|
30
|
+
"""Import *attr* from *module_name*, failing cleanly if the slice is absent.
|
|
31
|
+
|
|
32
|
+
Sibling slices (``okfsmith.parsers``, ``okfsmith.extract``,
|
|
33
|
+
``okfsmith.validate``, ``okfsmith.viz``, ``okfsmith.mcp_server``) are
|
|
34
|
+
imported lazily; when one is not installed the command exits 1 with a
|
|
35
|
+
clear message instead of an ImportError traceback. *module_name* may be a
|
|
36
|
+
dotted submodule path (e.g. ``okfsmith.parsers.ingest_no_llm``).
|
|
37
|
+
"""
|
|
38
|
+
try:
|
|
39
|
+
module = importlib.import_module(module_name)
|
|
40
|
+
except ModuleNotFoundError as exc:
|
|
41
|
+
missing = exc.name or ""
|
|
42
|
+
# The slice itself (or one of its parents under okfsmith) is absent.
|
|
43
|
+
if module_name == missing or module_name.startswith(missing + "."):
|
|
44
|
+
typer.echo(
|
|
45
|
+
f"error: '{module_name}' is not available in this installation.",
|
|
46
|
+
err=True,
|
|
47
|
+
)
|
|
48
|
+
raise typer.Exit(code=1)
|
|
49
|
+
raise
|
|
50
|
+
try:
|
|
51
|
+
return getattr(module, attr)
|
|
52
|
+
except AttributeError:
|
|
53
|
+
typer.echo(
|
|
54
|
+
f"error: '{module_name}' does not provide '{attr}'.", err=True
|
|
55
|
+
)
|
|
56
|
+
raise typer.Exit(code=1)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _require_dir(path: Path, what: str = "directory") -> None:
|
|
60
|
+
if not path.is_dir():
|
|
61
|
+
typer.echo(f"error: {what} '{path}' does not exist.", err=True)
|
|
62
|
+
raise typer.Exit(code=1)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _collect_inputs(source: Path, recursive: bool) -> list[Path]:
|
|
66
|
+
if source.is_file():
|
|
67
|
+
return [source]
|
|
68
|
+
iterator = source.rglob("*") if recursive else source.iterdir()
|
|
69
|
+
return sorted(p for p in iterator if p.is_file())
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
# ---------------------------------------------------------------------------
|
|
73
|
+
# init
|
|
74
|
+
# ---------------------------------------------------------------------------
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
@app.command()
|
|
78
|
+
def init(
|
|
79
|
+
directory: Path = typer.Argument(
|
|
80
|
+
..., help="Directory to scaffold the bundle in."
|
|
81
|
+
),
|
|
82
|
+
force: bool = typer.Option(
|
|
83
|
+
False, "--force", help="Scaffold even if the directory exists and is non-empty."
|
|
84
|
+
),
|
|
85
|
+
) -> None:
|
|
86
|
+
"""Create a new, empty OKF bundle in DIRECTORY."""
|
|
87
|
+
if directory.exists() and any(directory.iterdir()) and not force:
|
|
88
|
+
typer.echo(
|
|
89
|
+
f"error: '{directory}' exists and is not empty "
|
|
90
|
+
"(use --force to scaffold anyway).",
|
|
91
|
+
err=True,
|
|
92
|
+
)
|
|
93
|
+
raise typer.Exit(code=1)
|
|
94
|
+
directory.mkdir(parents=True, exist_ok=True)
|
|
95
|
+
bundle = Bundle(directory)
|
|
96
|
+
index_path = indexlog.ensure_index(bundle)
|
|
97
|
+
log_path = indexlog.append_log(
|
|
98
|
+
bundle, kind="Creation", message="Bundle created with `okfsmith init`."
|
|
99
|
+
)
|
|
100
|
+
typer.echo(f"Initialized OKF bundle in {directory}")
|
|
101
|
+
typer.echo(f" index: {index_path}")
|
|
102
|
+
typer.echo(f" log: {log_path}")
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
# ---------------------------------------------------------------------------
|
|
106
|
+
# ingest
|
|
107
|
+
# ---------------------------------------------------------------------------
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _ingest_no_llm_one(
|
|
111
|
+
path: Path,
|
|
112
|
+
target: Bundle,
|
|
113
|
+
*,
|
|
114
|
+
parse_file: Any,
|
|
115
|
+
ingest_no_llm: Any,
|
|
116
|
+
sha256_of: Any,
|
|
117
|
+
already_ingested: Any,
|
|
118
|
+
record_ingested: Any,
|
|
119
|
+
) -> tuple[str, int]:
|
|
120
|
+
"""Ingest one file via deterministic sectioning. Returns (status, count)."""
|
|
121
|
+
digest = sha256_of(path)
|
|
122
|
+
if already_ingested(target, digest):
|
|
123
|
+
return "skipped (already ingested)", 0
|
|
124
|
+
parsed = parse_file(path)
|
|
125
|
+
created = ingest_no_llm(target, parsed, str(path))
|
|
126
|
+
record_ingested(target, digest, str(path))
|
|
127
|
+
return "ok", len(created)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _ingest_llm_one(
|
|
131
|
+
path: Path,
|
|
132
|
+
target: Bundle,
|
|
133
|
+
*,
|
|
134
|
+
parse_file: Any,
|
|
135
|
+
section: Any,
|
|
136
|
+
SectionInput: Any,
|
|
137
|
+
run: Any,
|
|
138
|
+
model: str | None,
|
|
139
|
+
) -> tuple[str, int]:
|
|
140
|
+
"""Ingest one file via the LLM extraction pipeline. Returns (status, count)."""
|
|
141
|
+
parsed = parse_file(path)
|
|
142
|
+
sectioned = section(parsed)
|
|
143
|
+
doc_title = (parsed.meta or {}).get("title") or path.stem
|
|
144
|
+
sections = [
|
|
145
|
+
SectionInput(
|
|
146
|
+
title=sec.title,
|
|
147
|
+
level=sec.level,
|
|
148
|
+
text=sec.text,
|
|
149
|
+
page_span=sec.page_span,
|
|
150
|
+
tables=list(sec.tables or []),
|
|
151
|
+
source_id=str(path),
|
|
152
|
+
source_path=str(path),
|
|
153
|
+
doc_title=doc_title,
|
|
154
|
+
)
|
|
155
|
+
for sec in sectioned.sections
|
|
156
|
+
]
|
|
157
|
+
created = run(target, sections, model=model)
|
|
158
|
+
return "ok", len(created)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
@app.command()
|
|
162
|
+
def ingest(
|
|
163
|
+
source: Path = typer.Argument(..., help="File or directory to ingest."),
|
|
164
|
+
bundle: Path = typer.Option(..., "--bundle", help="Target bundle directory."),
|
|
165
|
+
model: str | None = typer.Option(
|
|
166
|
+
None, "--model", help="Model to use for LLM extraction."
|
|
167
|
+
),
|
|
168
|
+
no_llm: bool = typer.Option(
|
|
169
|
+
False,
|
|
170
|
+
"--no-llm",
|
|
171
|
+
help="Use deterministic sectioning (okfsmith.parsers) instead of an LLM.",
|
|
172
|
+
),
|
|
173
|
+
recursive: bool = typer.Option(
|
|
174
|
+
False,
|
|
175
|
+
"--recursive",
|
|
176
|
+
help="Recurse into subdirectories when SOURCE is a directory.",
|
|
177
|
+
),
|
|
178
|
+
) -> None:
|
|
179
|
+
"""Ingest documents into the bundle as draft concepts.
|
|
180
|
+
|
|
181
|
+
With ``--no-llm`` the deterministic sectioning path is used
|
|
182
|
+
(``okfsmith.parsers.ingest_no_llm``) with per-file SHA-256 dedup via the
|
|
183
|
+
bundle manifest; otherwise the LLM extraction path
|
|
184
|
+
(``okfsmith.extract.run``) is used.
|
|
185
|
+
"""
|
|
186
|
+
if not source.exists():
|
|
187
|
+
typer.echo(f"error: source '{source}' does not exist.", err=True)
|
|
188
|
+
raise typer.Exit(code=1)
|
|
189
|
+
files = _collect_inputs(source, recursive)
|
|
190
|
+
if not files:
|
|
191
|
+
typer.echo(f"error: no input files found under '{source}'.", err=True)
|
|
192
|
+
raise typer.Exit(code=1)
|
|
193
|
+
|
|
194
|
+
parse_file = _lazy_attr("okfsmith.parsers", "parse_file")
|
|
195
|
+
if no_llm:
|
|
196
|
+
ingest_no_llm = _lazy_attr(
|
|
197
|
+
"okfsmith.parsers.ingest_no_llm", "ingest_no_llm"
|
|
198
|
+
)
|
|
199
|
+
sha256_of = _lazy_attr("okfsmith.parsers.dedup", "sha256_of")
|
|
200
|
+
already_ingested = _lazy_attr("okfsmith.parsers.dedup", "already_ingested")
|
|
201
|
+
record_ingested = _lazy_attr("okfsmith.parsers.dedup", "record_ingested")
|
|
202
|
+
mode = "sectioning (no LLM)"
|
|
203
|
+
else:
|
|
204
|
+
section = _lazy_attr("okfsmith.parsers.sectioning", "section")
|
|
205
|
+
SectionInput = _lazy_attr("okfsmith.extract", "SectionInput")
|
|
206
|
+
run = _lazy_attr("okfsmith.extract", "run")
|
|
207
|
+
LLMUnavailableError = _lazy_attr("okfsmith.extract", "LLMUnavailableError")
|
|
208
|
+
mode = f"LLM extraction (model={model or 'default'})"
|
|
209
|
+
|
|
210
|
+
target = Bundle.load(bundle)
|
|
211
|
+
table = Table(title=f"Ingest summary — {mode}")
|
|
212
|
+
table.add_column("File")
|
|
213
|
+
table.add_column("SHA-256")
|
|
214
|
+
table.add_column("Concepts", justify="right")
|
|
215
|
+
table.add_column("Status")
|
|
216
|
+
|
|
217
|
+
digest_of = _lazy_attr("okfsmith.parsers.dedup", "sha256_of")
|
|
218
|
+
failures: list[Path] = []
|
|
219
|
+
created_total = 0
|
|
220
|
+
for path in files:
|
|
221
|
+
digest = digest_of(path)[:12]
|
|
222
|
+
try:
|
|
223
|
+
if no_llm:
|
|
224
|
+
status, count = _ingest_no_llm_one(
|
|
225
|
+
path,
|
|
226
|
+
target,
|
|
227
|
+
parse_file=parse_file,
|
|
228
|
+
ingest_no_llm=ingest_no_llm,
|
|
229
|
+
sha256_of=digest_of,
|
|
230
|
+
already_ingested=already_ingested,
|
|
231
|
+
record_ingested=record_ingested,
|
|
232
|
+
)
|
|
233
|
+
style = "green" if status == "ok" else "yellow"
|
|
234
|
+
table.add_row(str(path), digest, str(count), f"[{style}]{status}[/{style}]")
|
|
235
|
+
else:
|
|
236
|
+
try:
|
|
237
|
+
status, count = _ingest_llm_one(
|
|
238
|
+
path,
|
|
239
|
+
target,
|
|
240
|
+
parse_file=parse_file,
|
|
241
|
+
section=section,
|
|
242
|
+
SectionInput=SectionInput,
|
|
243
|
+
run=run,
|
|
244
|
+
model=model,
|
|
245
|
+
)
|
|
246
|
+
except LLMUnavailableError as exc:
|
|
247
|
+
typer.echo(f"error: LLM unavailable: {exc}", err=True)
|
|
248
|
+
raise typer.Exit(code=1)
|
|
249
|
+
table.add_row(str(path), digest, str(count), "[green]ok[/green]")
|
|
250
|
+
created_total += count
|
|
251
|
+
except typer.Exit:
|
|
252
|
+
raise
|
|
253
|
+
except Exception as exc: # noqa: BLE001 — per-file failure, keep going
|
|
254
|
+
failures.append(path)
|
|
255
|
+
table.add_row(str(path), digest, "0", f"[red]failed: {exc}[/red]")
|
|
256
|
+
console.print(table)
|
|
257
|
+
|
|
258
|
+
if created_total:
|
|
259
|
+
indexlog.ensure_index(target)
|
|
260
|
+
indexlog.append_log(
|
|
261
|
+
target,
|
|
262
|
+
kind="Update",
|
|
263
|
+
message=(
|
|
264
|
+
f"Ingested {created_total} draft concept(s) from "
|
|
265
|
+
f"{len(files) - len(failures)} source file(s)."
|
|
266
|
+
),
|
|
267
|
+
)
|
|
268
|
+
typer.echo(f"Wrote {created_total} draft concept(s) to {bundle}")
|
|
269
|
+
if failures and len(failures) == len(files):
|
|
270
|
+
typer.echo("error: all inputs failed to ingest.", err=True)
|
|
271
|
+
raise typer.Exit(code=1)
|
|
272
|
+
if failures:
|
|
273
|
+
typer.echo(
|
|
274
|
+
f"warning: {len(failures)} of {len(files)} input(s) failed.", err=True
|
|
275
|
+
)
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
# ---------------------------------------------------------------------------
|
|
279
|
+
# validate
|
|
280
|
+
# ---------------------------------------------------------------------------
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
@app.command()
|
|
284
|
+
def validate(
|
|
285
|
+
directory: Path = typer.Argument(..., help="Bundle directory to validate."),
|
|
286
|
+
strict: bool = typer.Option(
|
|
287
|
+
False, "--strict", help="Treat warnings as failures."
|
|
288
|
+
),
|
|
289
|
+
output_format: str = typer.Option(
|
|
290
|
+
"text", "--format", help="Output format: text or json."
|
|
291
|
+
),
|
|
292
|
+
) -> None:
|
|
293
|
+
"""Validate a bundle against OKF v0.2 (E001–E004 / W001–W015)."""
|
|
294
|
+
_require_dir(directory)
|
|
295
|
+
# check(bundle_path) -> ValidationReport with .errors / .warnings as
|
|
296
|
+
# Finding objects; serialize each via Finding.as_dict().
|
|
297
|
+
check = _lazy_attr("okfsmith.validate", "check")
|
|
298
|
+
result = check(directory)
|
|
299
|
+
errors = [finding.as_dict() for finding in result.errors]
|
|
300
|
+
warnings = [finding.as_dict() for finding in result.warnings]
|
|
301
|
+
|
|
302
|
+
if output_format == "json":
|
|
303
|
+
typer.echo(json.dumps({"errors": errors, "warnings": warnings}, indent=2))
|
|
304
|
+
elif output_format == "text":
|
|
305
|
+
_print_report(errors, warnings)
|
|
306
|
+
else:
|
|
307
|
+
typer.echo(f"error: unknown --format '{output_format}' (use text or json).", err=True)
|
|
308
|
+
raise typer.Exit(code=1)
|
|
309
|
+
|
|
310
|
+
if errors or (strict and warnings):
|
|
311
|
+
raise typer.Exit(code=1)
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
def _print_report(errors: list[dict], warnings: list[dict]) -> None:
|
|
315
|
+
for title, items, style in (
|
|
316
|
+
("Errors", errors, "red"),
|
|
317
|
+
("Warnings", warnings, "yellow"),
|
|
318
|
+
):
|
|
319
|
+
table = Table(title=f"{title} ({len(items)})")
|
|
320
|
+
table.add_column("Code", style=style)
|
|
321
|
+
table.add_column("File")
|
|
322
|
+
table.add_column("Message")
|
|
323
|
+
for item in items:
|
|
324
|
+
table.add_row(
|
|
325
|
+
str(item.get("code", "")),
|
|
326
|
+
str(item.get("file", "")),
|
|
327
|
+
str(item.get("message", "")),
|
|
328
|
+
)
|
|
329
|
+
console.print(table)
|
|
330
|
+
if errors:
|
|
331
|
+
typer.echo(f"INVALID: {len(errors)} error(s), {len(warnings)} warning(s).")
|
|
332
|
+
elif warnings:
|
|
333
|
+
typer.echo(f"Conformant with {len(warnings)} warning(s).")
|
|
334
|
+
else:
|
|
335
|
+
typer.echo("Conformant: no errors, no warnings.")
|
|
336
|
+
|
|
337
|
+
|
|
338
|
+
# ---------------------------------------------------------------------------
|
|
339
|
+
# list
|
|
340
|
+
# ---------------------------------------------------------------------------
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
@app.command(name="list")
|
|
344
|
+
def list_concepts(
|
|
345
|
+
directory: Path = typer.Argument(..., help="Bundle directory to list."),
|
|
346
|
+
type_filter: str | None = typer.Option(
|
|
347
|
+
None, "--type", help="Only show concepts of this type."
|
|
348
|
+
),
|
|
349
|
+
tier_filter: str | None = typer.Option(
|
|
350
|
+
None, "--tier", help="Only show concepts with this trust tier."
|
|
351
|
+
),
|
|
352
|
+
) -> None:
|
|
353
|
+
"""List concepts in the bundle: id, type, title, trust tier."""
|
|
354
|
+
_require_dir(directory)
|
|
355
|
+
bundle = Bundle.load(directory)
|
|
356
|
+
table = Table(title=f"Concepts in {directory}")
|
|
357
|
+
table.add_column("ID")
|
|
358
|
+
table.add_column("Type")
|
|
359
|
+
table.add_column("Title")
|
|
360
|
+
table.add_column("Trust tier")
|
|
361
|
+
shown = 0
|
|
362
|
+
for concept in bundle.iter_concepts():
|
|
363
|
+
ctype = str(concept.frontmatter.get("type") or "")
|
|
364
|
+
tier = _spec.trust_tier(concept.frontmatter)
|
|
365
|
+
if type_filter and ctype.casefold() != type_filter.casefold():
|
|
366
|
+
continue
|
|
367
|
+
if tier_filter and tier.casefold() != tier_filter.casefold():
|
|
368
|
+
continue
|
|
369
|
+
table.add_row(
|
|
370
|
+
concept.id, ctype, str(concept.frontmatter.get("title") or ""), tier
|
|
371
|
+
)
|
|
372
|
+
shown += 1
|
|
373
|
+
console.print(table)
|
|
374
|
+
typer.echo(f"{shown} concept(s)")
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
# ---------------------------------------------------------------------------
|
|
378
|
+
# read
|
|
379
|
+
# ---------------------------------------------------------------------------
|
|
380
|
+
|
|
381
|
+
|
|
382
|
+
@app.command()
|
|
383
|
+
def read(
|
|
384
|
+
directory: Path = typer.Argument(..., help="Bundle directory."),
|
|
385
|
+
concept_id: str = typer.Argument(..., help="Concept id, e.g. finance/revenue."),
|
|
386
|
+
) -> None:
|
|
387
|
+
"""Print a concept: frontmatter as YAML, then the body."""
|
|
388
|
+
_require_dir(directory)
|
|
389
|
+
bundle = Bundle.load(directory)
|
|
390
|
+
concept = bundle.get(concept_id)
|
|
391
|
+
if concept is None:
|
|
392
|
+
typer.echo(
|
|
393
|
+
f"error: concept '{concept_id}' not found in '{directory}'.", err=True
|
|
394
|
+
)
|
|
395
|
+
raise typer.Exit(code=1)
|
|
396
|
+
typer.echo(_fm.serialize_frontmatter(concept.frontmatter, concept.body))
|
|
397
|
+
|
|
398
|
+
|
|
399
|
+
# ---------------------------------------------------------------------------
|
|
400
|
+
# graph
|
|
401
|
+
# ---------------------------------------------------------------------------
|
|
402
|
+
|
|
403
|
+
|
|
404
|
+
@app.command()
|
|
405
|
+
def graph(
|
|
406
|
+
directory: Path = typer.Argument(..., help="Bundle directory."),
|
|
407
|
+
output_format: str = typer.Option(
|
|
408
|
+
"text", "--format", help="Output format: text, json, mermaid, or html."
|
|
409
|
+
),
|
|
410
|
+
output: Path | None = typer.Option(
|
|
411
|
+
None,
|
|
412
|
+
"--output",
|
|
413
|
+
help="Output file for --format html (default: <bundle>/viz.html).",
|
|
414
|
+
),
|
|
415
|
+
) -> None:
|
|
416
|
+
"""Show the concept link graph (nodes, edges, orphans, dead links)."""
|
|
417
|
+
_require_dir(directory)
|
|
418
|
+
bundle = Bundle.load(directory)
|
|
419
|
+
data = _links.build_graph(bundle)
|
|
420
|
+
|
|
421
|
+
if output_format == "json":
|
|
422
|
+
adjacency: dict[str, list[str]] = {}
|
|
423
|
+
for node in data["nodes"]:
|
|
424
|
+
adjacency[node["id"]] = []
|
|
425
|
+
for edge in data["edges"]:
|
|
426
|
+
adjacency.setdefault(edge["from"], []).append(edge["to"])
|
|
427
|
+
typer.echo(
|
|
428
|
+
json.dumps({"nodes": data["nodes"], "adjacency": adjacency}, indent=2)
|
|
429
|
+
)
|
|
430
|
+
elif output_format == "mermaid":
|
|
431
|
+
typer.echo(_links.mermaid_flowchart(data), nl=False)
|
|
432
|
+
elif output_format == "html":
|
|
433
|
+
# render_html(root, output) -> Path
|
|
434
|
+
render_html = _lazy_attr("okfsmith.viz", "render_html")
|
|
435
|
+
out_path = output or (Path(directory) / "viz.html")
|
|
436
|
+
written = render_html(Path(directory), out_path)
|
|
437
|
+
typer.echo(f"Wrote {written}")
|
|
438
|
+
elif output_format == "text":
|
|
439
|
+
_print_graph_text(bundle, data)
|
|
440
|
+
else:
|
|
441
|
+
typer.echo(
|
|
442
|
+
f"error: unknown --format '{output_format}' "
|
|
443
|
+
"(use text, json, mermaid, or html).",
|
|
444
|
+
err=True,
|
|
445
|
+
)
|
|
446
|
+
raise typer.Exit(code=1)
|
|
447
|
+
|
|
448
|
+
|
|
449
|
+
def _print_graph_text(bundle: Bundle, data: dict) -> None:
|
|
450
|
+
typer.echo(
|
|
451
|
+
f"{len(data['nodes'])} concept(s), {len(data['edges'])} link(s), "
|
|
452
|
+
f"{len(data['dead_links'])} dead link(s)."
|
|
453
|
+
)
|
|
454
|
+
orphan_ids = _links.orphans(bundle)
|
|
455
|
+
typer.echo(f"\nOrphans ({len(orphan_ids)}):")
|
|
456
|
+
for oid in orphan_ids:
|
|
457
|
+
typer.echo(f" - {oid}")
|
|
458
|
+
if not orphan_ids:
|
|
459
|
+
typer.echo(" (none)")
|
|
460
|
+
typer.echo(f"\nDead links ({len(data['dead_links'])}):")
|
|
461
|
+
for item in data["dead_links"]:
|
|
462
|
+
typer.echo(f" - {item['source']}: {item['target']}")
|
|
463
|
+
if not data["dead_links"]:
|
|
464
|
+
typer.echo(" (none)")
|
|
465
|
+
|
|
466
|
+
|
|
467
|
+
# ---------------------------------------------------------------------------
|
|
468
|
+
# mcp
|
|
469
|
+
# ---------------------------------------------------------------------------
|
|
470
|
+
|
|
471
|
+
|
|
472
|
+
@app.command()
|
|
473
|
+
def mcp(
|
|
474
|
+
bundle: Path = typer.Option(..., "--bundle", help="Bundle directory to serve."),
|
|
475
|
+
transport: str = typer.Option(
|
|
476
|
+
"stdio", "--transport", help="MCP transport to use."
|
|
477
|
+
),
|
|
478
|
+
) -> None:
|
|
479
|
+
"""Serve the bundle over MCP (Model Context Protocol)."""
|
|
480
|
+
_require_dir(bundle, "bundle")
|
|
481
|
+
# serve(bundle_path, transport) -> None
|
|
482
|
+
serve = _lazy_attr("okfsmith.mcp_server", "serve")
|
|
483
|
+
serve(bundle, transport)
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
"""okfsmith.core — bundle I/O, frontmatter, index/log, and OKF v0.2 spec constants.
|
|
2
|
+
|
|
3
|
+
This is the API contract the rest of the team codes against:
|
|
4
|
+
|
|
5
|
+
- :mod:`okfsmith.core.spec` — OKF v0.2 constants (reserved files, required
|
|
6
|
+
key, statuses, trust-tier derivation, actor helpers, timestamps).
|
|
7
|
+
- :mod:`okfsmith.core.frontmatter` — YAML frontmatter parse/serialize
|
|
8
|
+
(unknown keys preserved, key order preserved).
|
|
9
|
+
- :mod:`okfsmith.core.bundle` — :class:`Bundle` / :class:`Concept`: load and
|
|
10
|
+
write concept documents on disk.
|
|
11
|
+
- :mod:`okfsmith.core.indexlog` — generate ``index.md`` and append to
|
|
12
|
+
``log.md``.
|
|
13
|
+
|
|
14
|
+
No network calls anywhere in core; stdlib + declared dependencies only.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from okfsmith.core import frontmatter, indexlog, spec
|
|
18
|
+
from okfsmith.core.bundle import Bundle, Concept
|
|
19
|
+
|
|
20
|
+
__all__ = ["Bundle", "Concept", "frontmatter", "indexlog", "spec"]
|
okfsmith/core/bundle.py
ADDED
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""Bundle: the on-disk OKF knowledge bundle (concepts + index.md + log.md).
|
|
2
|
+
|
|
3
|
+
A bundle is a directory tree of ``*.md`` concept documents. Reserved files
|
|
4
|
+
(``index.md``, ``log.md`` — see :mod:`okfsmith.core.spec`) are never concepts.
|
|
5
|
+
A concept's id is its path relative to the bundle root, minus the ``.md``
|
|
6
|
+
suffix, with forward slashes (e.g. ``finance/revenue``).
|
|
7
|
+
|
|
8
|
+
No network calls; stdlib only.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import re
|
|
14
|
+
from dataclasses import dataclass, field
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Iterator
|
|
17
|
+
|
|
18
|
+
from okfsmith.core import frontmatter as _fm
|
|
19
|
+
from okfsmith.core.spec import RESERVED_FILES
|
|
20
|
+
|
|
21
|
+
_SUFFIX = ".md"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def slugify(value: str) -> str:
|
|
25
|
+
"""Slugify *value*: lowercase, runs of non-alphanumerics become ``-``.
|
|
26
|
+
|
|
27
|
+
>>> slugify("Hello, World!")
|
|
28
|
+
'hello-world'
|
|
29
|
+
"""
|
|
30
|
+
slug = re.sub(r"[^a-z0-9]+", "-", value.lower()).strip("-")
|
|
31
|
+
return slug or "untitled"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def concept_path_for(root: Path, concept_id: str) -> Path:
|
|
35
|
+
"""Map a concept id to its on-disk path, slugifying each ``/`` segment.
|
|
36
|
+
|
|
37
|
+
``"Notes/Hello World"`` under ``root`` → ``root/notes/hello-world.md``.
|
|
38
|
+
"""
|
|
39
|
+
parts = [slugify(part) for part in concept_id.replace("\\", "/").split("/")]
|
|
40
|
+
return root.joinpath(*parts).with_suffix(_SUFFIX)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@dataclass
|
|
44
|
+
class Concept:
|
|
45
|
+
"""A single OKF concept document."""
|
|
46
|
+
|
|
47
|
+
id: str
|
|
48
|
+
"""Concept id: path relative to the bundle root, minus ``.md``, forward slashes."""
|
|
49
|
+
|
|
50
|
+
path: Path
|
|
51
|
+
"""Absolute path of the ``.md`` file on disk."""
|
|
52
|
+
|
|
53
|
+
frontmatter: dict = field(default_factory=dict)
|
|
54
|
+
"""Parsed YAML frontmatter; unknown keys preserved as-is."""
|
|
55
|
+
|
|
56
|
+
body: str = ""
|
|
57
|
+
"""Markdown body after the frontmatter block."""
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class Bundle:
|
|
61
|
+
"""An OKF knowledge bundle rooted at a directory on disk."""
|
|
62
|
+
|
|
63
|
+
def __init__(self, root: str | Path) -> None:
|
|
64
|
+
"""Create a bundle handle for *root* (nothing is read yet)."""
|
|
65
|
+
self.root = Path(root).resolve()
|
|
66
|
+
self._concepts: dict[str, Concept] = {}
|
|
67
|
+
self.index_text: str | None = None
|
|
68
|
+
"""Raw text of the root ``index.md``, if present when loaded."""
|
|
69
|
+
self.log_text: str | None = None
|
|
70
|
+
"""Raw text of the root ``log.md``, if present when loaded."""
|
|
71
|
+
|
|
72
|
+
@classmethod
|
|
73
|
+
def load(cls, root: str | Path) -> "Bundle":
|
|
74
|
+
"""Walk *root* and parse every ``*.md`` file into a :class:`Concept`.
|
|
75
|
+
|
|
76
|
+
Files named ``index.md`` / ``log.md`` are skipped as concepts; the
|
|
77
|
+
root ``index.md`` / ``log.md`` are read into ``index_text`` /
|
|
78
|
+
``log_text``. Missing reserved files are fine (``None``).
|
|
79
|
+
"""
|
|
80
|
+
bundle = cls(root)
|
|
81
|
+
bundle.root.mkdir(parents=True, exist_ok=True)
|
|
82
|
+
for md in sorted(bundle.root.rglob(f"*{_SUFFIX}")):
|
|
83
|
+
if md.name in RESERVED_FILES:
|
|
84
|
+
continue
|
|
85
|
+
rel = md.relative_to(bundle.root)
|
|
86
|
+
concept_id = rel.with_suffix("").as_posix()
|
|
87
|
+
data, body = _fm.parse_frontmatter(md.read_text(encoding="utf-8"))
|
|
88
|
+
bundle._concepts[concept_id] = Concept(
|
|
89
|
+
id=concept_id, path=md, frontmatter=data, body=body
|
|
90
|
+
)
|
|
91
|
+
for filename, attr in (("index.md", "index_text"), ("log.md", "log_text")):
|
|
92
|
+
candidate = bundle.root / filename
|
|
93
|
+
if candidate.is_file():
|
|
94
|
+
setattr(bundle, attr, candidate.read_text(encoding="utf-8"))
|
|
95
|
+
return bundle
|
|
96
|
+
|
|
97
|
+
def write_concept(self, concept_id: str, frontmatter: dict, body: str) -> Concept:
|
|
98
|
+
"""Write a concept document to disk and register it in this bundle.
|
|
99
|
+
|
|
100
|
+
The id is slugified segment-by-segment to derive the file path
|
|
101
|
+
(parent directories are created). The returned :class:`Concept`'s
|
|
102
|
+
``id`` is the slugified id, so it round-trips through :meth:`load`.
|
|
103
|
+
Unknown frontmatter keys are written back untouched.
|
|
104
|
+
"""
|
|
105
|
+
self.root.mkdir(parents=True, exist_ok=True)
|
|
106
|
+
path = concept_path_for(self.root, concept_id)
|
|
107
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
108
|
+
path.write_text(_fm.serialize_frontmatter(frontmatter, body), encoding="utf-8")
|
|
109
|
+
stored_id = path.relative_to(self.root).with_suffix("").as_posix()
|
|
110
|
+
concept = Concept(
|
|
111
|
+
id=stored_id, path=path, frontmatter=dict(frontmatter), body=body
|
|
112
|
+
)
|
|
113
|
+
self._concepts[stored_id] = concept
|
|
114
|
+
return concept
|
|
115
|
+
|
|
116
|
+
def iter_concepts(self) -> Iterator[Concept]:
|
|
117
|
+
"""Yield all concepts, sorted by id for deterministic output."""
|
|
118
|
+
for concept_id in sorted(self._concepts):
|
|
119
|
+
yield self._concepts[concept_id]
|
|
120
|
+
|
|
121
|
+
def get(self, concept_id: str) -> Concept | None:
|
|
122
|
+
"""Return the concept with *concept_id*, or ``None`` if absent."""
|
|
123
|
+
return self._concepts.get(concept_id)
|