okfgraph 0.2.4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- okfgraph/__init__.py +7 -0
- okfgraph/cli.py +1364 -0
- okfgraph/components/__init__.py +61 -0
- okfgraph/components/converters.py +128 -0
- okfgraph/components/delta.py +271 -0
- okfgraph/components/diff.py +172 -0
- okfgraph/components/doctor.py +216 -0
- okfgraph/components/embedding.py +323 -0
- okfgraph/components/export.py +354 -0
- okfgraph/components/image_assets.py +319 -0
- okfgraph/components/import_.py +1110 -0
- okfgraph/components/ingest.py +495 -0
- okfgraph/components/links.py +133 -0
- okfgraph/components/lint.py +118 -0
- okfgraph/components/purge.py +342 -0
- okfgraph/components/ranking.py +168 -0
- okfgraph/components/schema.py +443 -0
- okfgraph/components/search.py +1006 -0
- okfgraph/config.py +393 -0
- okfgraph/images.py +435 -0
- okfgraph/mcp_server.py +461 -0
- okfgraph/models.py +128 -0
- okfgraph/router.py +485 -0
- okfgraph/security.py +269 -0
- okfgraph/tools.py +460 -0
- okfgraph-0.2.4.dist-info/METADATA +25 -0
- okfgraph-0.2.4.dist-info/RECORD +31 -0
- okfgraph-0.2.4.dist-info/WHEEL +5 -0
- okfgraph-0.2.4.dist-info/entry_points.txt +3 -0
- okfgraph-0.2.4.dist-info/licenses/LICENSE +6 -0
- okfgraph-0.2.4.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,495 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import base64
|
|
4
|
+
import hashlib
|
|
5
|
+
import heapq
|
|
6
|
+
import json
|
|
7
|
+
import logging
|
|
8
|
+
import math
|
|
9
|
+
import os
|
|
10
|
+
import re
|
|
11
|
+
import time
|
|
12
|
+
import uuid
|
|
13
|
+
from contextlib import contextmanager
|
|
14
|
+
from datetime import datetime, timezone, timedelta
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Any, Callable, Dict, List, Optional, Tuple, Union, Set
|
|
17
|
+
from urllib.parse import urlparse
|
|
18
|
+
|
|
19
|
+
import mordant
|
|
20
|
+
import numpy as np
|
|
21
|
+
import yaml
|
|
22
|
+
import frontmatter
|
|
23
|
+
from okfgraph.models import ChunkModel, ConceptModel, normalize_tags
|
|
24
|
+
|
|
25
|
+
logger = logging.getLogger(__name__)
|
|
26
|
+
|
|
27
|
+
class IngestManager:
|
|
28
|
+
def __init__(self, _write_lock_ctx, bundle_root, device, import_mgr, delta_mgr,
|
|
29
|
+
converter=None):
|
|
30
|
+
self._write_lock_ctx = _write_lock_ctx
|
|
31
|
+
self.bundle_root = bundle_root
|
|
32
|
+
self.device = device
|
|
33
|
+
self.import_mgr = import_mgr
|
|
34
|
+
self.delta_mgr = delta_mgr
|
|
35
|
+
# DocumentConverter (see okfgraph.components.converters). None =
|
|
36
|
+
# default BobineConverter, built lazily so router construction
|
|
37
|
+
# never requires bobine — only actual conversion does.
|
|
38
|
+
self._converter = converter
|
|
39
|
+
|
|
40
|
+
def _ingest_md_inner(
|
|
41
|
+
self,
|
|
42
|
+
md_path: str | Path,
|
|
43
|
+
concept_id: str | None,
|
|
44
|
+
title: str | None,
|
|
45
|
+
description: str | None,
|
|
46
|
+
tags: list[str] | None,
|
|
47
|
+
mode: str,
|
|
48
|
+
) -> Dict[str, Any]:
|
|
49
|
+
"""Inner implementation of ingest_md (called under write lock)."""
|
|
50
|
+
md_path = Path(md_path)
|
|
51
|
+
if not md_path.exists():
|
|
52
|
+
raise FileNotFoundError(f"Markdown file not found: {md_path}")
|
|
53
|
+
|
|
54
|
+
# Lint the file (may auto-fix in-place)
|
|
55
|
+
lint_result = self._lint_converted_md(md_path, auto_fix=True)
|
|
56
|
+
if lint_result["fixed"]:
|
|
57
|
+
md_path.write_text(lint_result["content"], encoding="utf-8")
|
|
58
|
+
|
|
59
|
+
# Parse frontmatter
|
|
60
|
+
post = frontmatter.load(md_path)
|
|
61
|
+
fm = dict(post.metadata)
|
|
62
|
+
|
|
63
|
+
# Determine metadata
|
|
64
|
+
cid = concept_id or md_path.stem.replace(" ", "_").lower()
|
|
65
|
+
t = title or fm.get("title") or md_path.stem
|
|
66
|
+
desc = description or fm.get("description") or fm.get("summary") or ""
|
|
67
|
+
file_tags = normalize_tags(fm.get("tags", []))
|
|
68
|
+
all_tags = list(set((tags or []) + file_tags))
|
|
69
|
+
|
|
70
|
+
# Build ConceptModel
|
|
71
|
+
concept = ConceptModel.model_validate({
|
|
72
|
+
"id": cid,
|
|
73
|
+
"title": t,
|
|
74
|
+
"description": desc,
|
|
75
|
+
"body": post.content,
|
|
76
|
+
"type": fm.get("type", "note"),
|
|
77
|
+
"tags": all_tags,
|
|
78
|
+
})
|
|
79
|
+
|
|
80
|
+
# Import via shared single-concept pipeline
|
|
81
|
+
result = self.import_mgr._import_single_concept(concept, post.content, mode)
|
|
82
|
+
|
|
83
|
+
return {
|
|
84
|
+
"concept_id": result["concept_id"],
|
|
85
|
+
"title": result["title"],
|
|
86
|
+
"description": result["description"],
|
|
87
|
+
"tags": result["tags"],
|
|
88
|
+
"chunk_count": result["chunk_count"],
|
|
89
|
+
"image_count": result["image_count"],
|
|
90
|
+
"lint_issues": {
|
|
91
|
+
"fixed_count": lint_result["fixed_count"],
|
|
92
|
+
"unfixable_count": len(lint_result["unfixable"]),
|
|
93
|
+
"error_count": len(lint_result["errors"]),
|
|
94
|
+
},
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _ingest_thoughts_inner(
|
|
99
|
+
self,
|
|
100
|
+
thoughts: str,
|
|
101
|
+
topic: str,
|
|
102
|
+
concept_id: str | None,
|
|
103
|
+
tags: list[str] | None,
|
|
104
|
+
) -> Dict[str, Any]:
|
|
105
|
+
"""Inner implementation of ingest_thoughts (called under write lock)."""
|
|
106
|
+
import uuid
|
|
107
|
+
|
|
108
|
+
# Generate concept_id from topic if not provided
|
|
109
|
+
if not concept_id:
|
|
110
|
+
ts = datetime.now().strftime("%Y%m%d%H%M%S")
|
|
111
|
+
slug = topic.lower().replace(" ", "_")[:30]
|
|
112
|
+
concept_id = f"thought_{slug}_{ts}_{str(uuid.uuid4())[:6]}"
|
|
113
|
+
|
|
114
|
+
# Build OKF-compliant markdown
|
|
115
|
+
header_lines = [
|
|
116
|
+
"---",
|
|
117
|
+
f'title: "Thought: {topic}"',
|
|
118
|
+
"type: thought",
|
|
119
|
+
"thought_type: reasoning",
|
|
120
|
+
f"topic: {topic}",
|
|
121
|
+
f"tags: [thought, reasoning, {topic}]",
|
|
122
|
+
f"created: {datetime.now().isoformat()}",
|
|
123
|
+
"---",
|
|
124
|
+
"",
|
|
125
|
+
thoughts,
|
|
126
|
+
]
|
|
127
|
+
markdown = "\n".join(header_lines)
|
|
128
|
+
|
|
129
|
+
# Defensive lint: the LLM-provided thoughts text may contain
|
|
130
|
+
# malformed markdown (trailing spaces, blank lines, etc.).
|
|
131
|
+
lint_result = self._lint_converted_md_str(markdown, auto_fix=True)
|
|
132
|
+
if lint_result["fixed"]:
|
|
133
|
+
markdown = lint_result["content"]
|
|
134
|
+
logger.info(
|
|
135
|
+
"ingest_thoughts: linted %s, fixed %d issues",
|
|
136
|
+
concept_id,
|
|
137
|
+
lint_result["fixed_count"],
|
|
138
|
+
)
|
|
139
|
+
if lint_result["errors"]:
|
|
140
|
+
logger.warning(
|
|
141
|
+
"ingest_thoughts: %s has %d structural errors",
|
|
142
|
+
concept_id,
|
|
143
|
+
len(lint_result["errors"]),
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
# Apply tags
|
|
147
|
+
all_tags = list(set(["thought", "reasoning", topic] + (tags or [])))
|
|
148
|
+
|
|
149
|
+
# Build ConceptModel
|
|
150
|
+
concept = ConceptModel.model_validate({
|
|
151
|
+
"id": concept_id,
|
|
152
|
+
"title": f"Thought: {topic}",
|
|
153
|
+
"description": f"Reasoning about {topic}",
|
|
154
|
+
"body": markdown,
|
|
155
|
+
"type": "thought",
|
|
156
|
+
"tags": all_tags,
|
|
157
|
+
})
|
|
158
|
+
|
|
159
|
+
# Import via shared single-concept pipeline
|
|
160
|
+
result = self.import_mgr._import_single_concept(concept, markdown, "text")
|
|
161
|
+
|
|
162
|
+
return {
|
|
163
|
+
"concept_id": result["concept_id"],
|
|
164
|
+
"topic": topic,
|
|
165
|
+
"tags": result["tags"],
|
|
166
|
+
"chunk_count": result["chunk_count"],
|
|
167
|
+
"markdown": markdown,
|
|
168
|
+
"lint_issues": lint_result,
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
def _lint_converted_md(
|
|
172
|
+
self,
|
|
173
|
+
md_path: Path,
|
|
174
|
+
*,
|
|
175
|
+
auto_fix: bool = True,
|
|
176
|
+
) -> Dict[str, Any]:
|
|
177
|
+
"""Lint a markdown file and optionally auto-fix fixable issues.
|
|
178
|
+
|
|
179
|
+
Returns a dict with:
|
|
180
|
+
- "content": the (possibly fixed) markdown content (str)
|
|
181
|
+
- "fixed": whether content was modified (bool)
|
|
182
|
+
- "fixed_count": number of auto-fixed issues (int)
|
|
183
|
+
- "unfixable": list of unfixable diagnostics (list)
|
|
184
|
+
- "errors": list of error-level diagnostics (list)
|
|
185
|
+
"""
|
|
186
|
+
content = md_path.read_text(encoding="utf-8")
|
|
187
|
+
diagnostics = mordant.lint(content, gfm_opts=mordant.GfmOptions.all())
|
|
188
|
+
|
|
189
|
+
if not diagnostics:
|
|
190
|
+
return {
|
|
191
|
+
"content": content,
|
|
192
|
+
"fixed": False,
|
|
193
|
+
"fixed_count": 0,
|
|
194
|
+
"unfixable": [],
|
|
195
|
+
"errors": [],
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
# Categorize diagnostics
|
|
199
|
+
fixable_rules = {"MD009", "MD012", "MD047"} # whitespace/formatting
|
|
200
|
+
error_rules = {"MD001", "MD031", "MD033"} # structural errors
|
|
201
|
+
unfixable = [d for d in diagnostics if d.rule not in fixable_rules]
|
|
202
|
+
errors = [d for d in diagnostics if d.rule in error_rules]
|
|
203
|
+
|
|
204
|
+
fixed_content = content
|
|
205
|
+
fixed_count = 0
|
|
206
|
+
|
|
207
|
+
if auto_fix:
|
|
208
|
+
fixable = [d for d in diagnostics if d.rule in fixable_rules]
|
|
209
|
+
if fixable:
|
|
210
|
+
result = mordant.fix(content, gfm_opts=mordant.GfmOptions.all())
|
|
211
|
+
if result.fixed:
|
|
212
|
+
fixed_content = result.output
|
|
213
|
+
fixed_count = len(result.fixed)
|
|
214
|
+
logger.info(
|
|
215
|
+
"auto-fixed %d issues in %s",
|
|
216
|
+
fixed_count,
|
|
217
|
+
md_path.name,
|
|
218
|
+
)
|
|
219
|
+
|
|
220
|
+
if unfixable:
|
|
221
|
+
logger.warning(
|
|
222
|
+
"%d unfixable issues in %s: %s",
|
|
223
|
+
len(unfixable),
|
|
224
|
+
md_path.name,
|
|
225
|
+
", ".join(f"{d.rule} (line {d.line})" for d in unfixable[:5]),
|
|
226
|
+
)
|
|
227
|
+
|
|
228
|
+
if errors:
|
|
229
|
+
logger.warning(
|
|
230
|
+
"%d structural errors in %s — import may produce unexpected results: %s",
|
|
231
|
+
len(errors),
|
|
232
|
+
md_path.name,
|
|
233
|
+
", ".join(f"{d.rule} (line {d.line})" for d in errors[:3]),
|
|
234
|
+
)
|
|
235
|
+
|
|
236
|
+
return {
|
|
237
|
+
"content": fixed_content,
|
|
238
|
+
"fixed": fixed_count > 0,
|
|
239
|
+
"fixed_count": fixed_count,
|
|
240
|
+
"unfixable": unfixable,
|
|
241
|
+
"errors": errors,
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _lint_converted_md_str(
|
|
246
|
+
self,
|
|
247
|
+
content: str,
|
|
248
|
+
*,
|
|
249
|
+
auto_fix: bool = True,
|
|
250
|
+
) -> Dict[str, Any]:
|
|
251
|
+
"""Lint markdown content in-memory (no file I/O).
|
|
252
|
+
|
|
253
|
+
Returns a dict with the same keys as ``_lint_converted_md``.
|
|
254
|
+
"""
|
|
255
|
+
diagnostics = mordant.lint(content, gfm_opts=mordant.GfmOptions.all())
|
|
256
|
+
|
|
257
|
+
if not diagnostics:
|
|
258
|
+
return {
|
|
259
|
+
"content": content,
|
|
260
|
+
"fixed": False,
|
|
261
|
+
"fixed_count": 0,
|
|
262
|
+
"unfixable": [],
|
|
263
|
+
"errors": [],
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
fixable_rules = {"MD009", "MD012", "MD047"}
|
|
267
|
+
error_rules = {"MD001", "MD031", "MD033"}
|
|
268
|
+
unfixable = [d for d in diagnostics if d.rule not in fixable_rules]
|
|
269
|
+
errors = [d for d in diagnostics if d.rule in error_rules]
|
|
270
|
+
|
|
271
|
+
fixed_content = content
|
|
272
|
+
fixed_count = 0
|
|
273
|
+
|
|
274
|
+
if auto_fix:
|
|
275
|
+
fixable = [d for d in diagnostics if d.rule in fixable_rules]
|
|
276
|
+
if fixable:
|
|
277
|
+
result = mordant.fix(content, gfm_opts=mordant.GfmOptions.all())
|
|
278
|
+
if result.fixed:
|
|
279
|
+
fixed_content = result.output
|
|
280
|
+
fixed_count = len(result.fixed)
|
|
281
|
+
|
|
282
|
+
return {
|
|
283
|
+
"content": fixed_content,
|
|
284
|
+
"fixed": fixed_count > 0,
|
|
285
|
+
"fixed_count": fixed_count,
|
|
286
|
+
"unfixable": [d.rule for d in unfixable],
|
|
287
|
+
"errors": [d.rule for d in errors],
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def ingest_md(
|
|
292
|
+
self,
|
|
293
|
+
md_path: str | Path,
|
|
294
|
+
*,
|
|
295
|
+
concept_id: str | None = None,
|
|
296
|
+
title: str | None = None,
|
|
297
|
+
description: str | None = None,
|
|
298
|
+
tags: list[str] | None = None,
|
|
299
|
+
mode: str = "text",
|
|
300
|
+
) -> Dict[str, Any]:
|
|
301
|
+
"""Import a single markdown file into the knowledge graph.
|
|
302
|
+
|
|
303
|
+
This is the programmatic counterpart to ``import_bundle()`` but
|
|
304
|
+
operates on a single file with explicit metadata control.
|
|
305
|
+
|
|
306
|
+
The file is linted with mordant before import. Fixable issues
|
|
307
|
+
(MD009, MD012, MD047) are auto-corrected. Unfixable issues
|
|
308
|
+
are logged as warnings but do not block import.
|
|
309
|
+
|
|
310
|
+
Args:
|
|
311
|
+
md_path: Path to the markdown file to import.
|
|
312
|
+
concept_id: Optional explicit concept ID. If None, generated from filename.
|
|
313
|
+
title: Optional title override (defaults to frontmatter or filename).
|
|
314
|
+
description: Optional description override (defaults to frontmatter).
|
|
315
|
+
tags: Optional tags to apply to the concept.
|
|
316
|
+
mode: Image ingestion mode (text | optional | omni).
|
|
317
|
+
|
|
318
|
+
Returns:
|
|
319
|
+
Dict with keys:
|
|
320
|
+
- "concept_id": The imported concept ID
|
|
321
|
+
- "title": Title used
|
|
322
|
+
- "description": Description used
|
|
323
|
+
- "tags": Applied tags
|
|
324
|
+
- "chunk_count": Number of chunks created (if chunking enabled)
|
|
325
|
+
- "image_count": Number of images ingested
|
|
326
|
+
- "lint_issues": Lint result dict (fixed_count, unfixable, errors)
|
|
327
|
+
"""
|
|
328
|
+
# Acquire write lock (Gap #7b)
|
|
329
|
+
with self._write_lock_ctx():
|
|
330
|
+
return self._ingest_md_inner(md_path, concept_id, title, description, tags, mode)
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
def _resolve_converter(self, converter):
|
|
334
|
+
"""Per-call override → manager default → lazy BobineConverter."""
|
|
335
|
+
if converter is not None:
|
|
336
|
+
return converter
|
|
337
|
+
if self._converter is None:
|
|
338
|
+
from okfgraph.components.converters import BobineConverter
|
|
339
|
+
self._converter = BobineConverter()
|
|
340
|
+
return self._converter
|
|
341
|
+
|
|
342
|
+
def ingest_pdf(
|
|
343
|
+
self,
|
|
344
|
+
pdf_path: str | Path,
|
|
345
|
+
*,
|
|
346
|
+
auto_import: bool = True,
|
|
347
|
+
output_dir: str | Path | None = None,
|
|
348
|
+
mode: str = "text",
|
|
349
|
+
batch_size: int = 32,
|
|
350
|
+
purge_deleted: bool = False,
|
|
351
|
+
on_page: Callable[[int, int], None] | None = None,
|
|
352
|
+
converter=None,
|
|
353
|
+
) -> Dict[str, Any]:
|
|
354
|
+
"""Convert a PDF to markdown and optionally import into the graph.
|
|
355
|
+
|
|
356
|
+
Conversion is delegated to a DocumentConverter (see
|
|
357
|
+
``okfgraph.components.converters``) — bobine by default, swappable
|
|
358
|
+
for any other pipeline.
|
|
359
|
+
|
|
360
|
+
This is the programmatic counterpart to the ``okf ingest`` CLI command.
|
|
361
|
+
It converts the PDF, then optionally imports the resulting markdown
|
|
362
|
+
into the knowledge graph via ``import_bundle()``.
|
|
363
|
+
|
|
364
|
+
Args:
|
|
365
|
+
pdf_path: Path to the PDF file.
|
|
366
|
+
auto_import: If True, import the converted markdown into the graph.
|
|
367
|
+
If False, write to disk only.
|
|
368
|
+
output_dir: Output directory for the markdown (used when
|
|
369
|
+
auto_import=False). Defaults to the PDF's parent directory.
|
|
370
|
+
mode: Image ingestion mode for auto-import — "text", "optional",
|
|
371
|
+
or "omni". Only used when auto_import=True.
|
|
372
|
+
batch_size: Batch size for encoding during auto-import.
|
|
373
|
+
purge_deleted: If True, purge deleted concepts during auto-import.
|
|
374
|
+
on_page: Optional callback(page_index, page_total) for progress.
|
|
375
|
+
converter: DocumentConverter to use for this call. Defaults to
|
|
376
|
+
the manager's converter (bobine unless overridden).
|
|
377
|
+
|
|
378
|
+
Returns:
|
|
379
|
+
A dict with keys:
|
|
380
|
+
- "md_path": Path to the converted markdown file. Transient
|
|
381
|
+
when auto_import=True (conversion runs in a temp dir that is
|
|
382
|
+
removed after import — the content lives in the graph).
|
|
383
|
+
- "concept_ids": List of imported concept IDs (only when auto_import=True)
|
|
384
|
+
- "image_dir": Path to the staged images directory (always present)
|
|
385
|
+
- "page_count": Number of pages in the PDF
|
|
386
|
+
|
|
387
|
+
Raises:
|
|
388
|
+
RuntimeError: If the default converter needs bobine and it is
|
|
389
|
+
not installed.
|
|
390
|
+
"""
|
|
391
|
+
from tempfile import TemporaryDirectory
|
|
392
|
+
|
|
393
|
+
pdf_path = Path(pdf_path)
|
|
394
|
+
if not pdf_path.exists():
|
|
395
|
+
raise FileNotFoundError(f"PDF not found: {pdf_path}")
|
|
396
|
+
|
|
397
|
+
converter = self._resolve_converter(converter)
|
|
398
|
+
|
|
399
|
+
if auto_import:
|
|
400
|
+
with TemporaryDirectory(prefix="okf_ingest_") as tmp:
|
|
401
|
+
work_dir = Path(tmp)
|
|
402
|
+
logger.info("converting %s → %s", pdf_path, work_dir)
|
|
403
|
+
doc = converter.convert(pdf_path, work_dir, on_page=on_page)
|
|
404
|
+
md_path = Path(doc.md_path)
|
|
405
|
+
lint_result = self._lint_converted_md(md_path, auto_fix=True)
|
|
406
|
+
if lint_result["fixed"]:
|
|
407
|
+
md_path.write_text(lint_result["content"], encoding="utf-8")
|
|
408
|
+
if lint_result["errors"]:
|
|
409
|
+
logger.warning(
|
|
410
|
+
"PDF output has %d structural errors — proceeding anyway",
|
|
411
|
+
len(lint_result["errors"]),
|
|
412
|
+
)
|
|
413
|
+
ids = self._import_work_dir(
|
|
414
|
+
work_dir, batch_size, mode, purge_deleted, pdf_path
|
|
415
|
+
)
|
|
416
|
+
return {
|
|
417
|
+
"md_path": str(md_path),
|
|
418
|
+
"concept_ids": ids,
|
|
419
|
+
"image_dir": str(doc.image_dir),
|
|
420
|
+
"page_count": doc.page_count,
|
|
421
|
+
}
|
|
422
|
+
output_dir = Path(output_dir) if output_dir else pdf_path.parent
|
|
423
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
424
|
+
logger.info("converting %s → %s", pdf_path, output_dir)
|
|
425
|
+
doc = converter.convert(pdf_path, output_dir, on_page=on_page)
|
|
426
|
+
md_path = Path(doc.md_path)
|
|
427
|
+
lint_result = self._lint_converted_md(md_path, auto_fix=True)
|
|
428
|
+
if lint_result["fixed"]:
|
|
429
|
+
md_path.write_text(lint_result["content"], encoding="utf-8")
|
|
430
|
+
logger.info("written %s", md_path)
|
|
431
|
+
return {
|
|
432
|
+
"md_path": str(md_path),
|
|
433
|
+
"concept_ids": [],
|
|
434
|
+
"image_dir": str(doc.image_dir),
|
|
435
|
+
"page_count": doc.page_count,
|
|
436
|
+
}
|
|
437
|
+
|
|
438
|
+
def _import_work_dir(self, work_dir, batch_size, mode, purge_deleted, pdf_path):
|
|
439
|
+
"""Import a converted-PDF work dir, keeping bundle_root overrides in sync."""
|
|
440
|
+
old_bundle_root = self.bundle_root
|
|
441
|
+
self.bundle_root = work_dir
|
|
442
|
+
# Keep the injected DeltaDetector and ImportManager in sync:
|
|
443
|
+
# each stores its own bundle_root copy, and import_bundle /
|
|
444
|
+
# _changed_directories rely on it (Phase 3 refactor).
|
|
445
|
+
self.delta_mgr.bundle_root = work_dir
|
|
446
|
+
self.import_mgr.bundle_root = work_dir
|
|
447
|
+
try:
|
|
448
|
+
ids = self.import_mgr.import_bundle(
|
|
449
|
+
work_dir,
|
|
450
|
+
batch_size=batch_size,
|
|
451
|
+
mode=mode,
|
|
452
|
+
purge_deleted=purge_deleted,
|
|
453
|
+
)
|
|
454
|
+
finally:
|
|
455
|
+
self.bundle_root = old_bundle_root
|
|
456
|
+
self.delta_mgr.bundle_root = old_bundle_root
|
|
457
|
+
self.import_mgr.bundle_root = old_bundle_root
|
|
458
|
+
logger.info("imported %d concept(s) from %s", len(ids), pdf_path)
|
|
459
|
+
return ids
|
|
460
|
+
|
|
461
|
+
def ingest_thoughts(
|
|
462
|
+
self,
|
|
463
|
+
thoughts: str,
|
|
464
|
+
*,
|
|
465
|
+
topic: str,
|
|
466
|
+
concept_id: str | None = None,
|
|
467
|
+
tags: list[str] | None = None,
|
|
468
|
+
) -> Dict[str, Any]:
|
|
469
|
+
"""Store LLM reasoning/thinking as a searchable concept.
|
|
470
|
+
|
|
471
|
+
Wraps the raw reasoning text in OKF-compliant markdown with metadata
|
|
472
|
+
(type=thought, thought_type=reasoning, topic) so it can be searched,
|
|
473
|
+
traversed, and used as context for other queries.
|
|
474
|
+
|
|
475
|
+
The markdown is linted with mordant before import. Fixable issues
|
|
476
|
+
are auto-corrected.
|
|
477
|
+
|
|
478
|
+
Args:
|
|
479
|
+
thoughts: The raw reasoning text from the LLM.
|
|
480
|
+
topic: High-level topic or domain for the reasoning.
|
|
481
|
+
concept_id: Optional explicit concept ID. If None, generated from topic.
|
|
482
|
+
tags: Optional additional tags.
|
|
483
|
+
|
|
484
|
+
Returns:
|
|
485
|
+
Dict with keys:
|
|
486
|
+
- "concept_id": The created concept ID
|
|
487
|
+
- "topic": Topic used
|
|
488
|
+
- "tags": Applied tags
|
|
489
|
+
- "chunk_count": Number of chunks created
|
|
490
|
+
- "markdown": The generated markdown content
|
|
491
|
+
"""
|
|
492
|
+
# Acquire write lock (Gap #7b)
|
|
493
|
+
with self._write_lock_ctx():
|
|
494
|
+
return self._ingest_thoughts_inner(thoughts, topic, concept_id, tags)
|
|
495
|
+
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
"""Shared link primitives: extraction + Obsidian-style name resolution.
|
|
2
|
+
|
|
3
|
+
All three import paths (bundle batch, single concept, single upsert) and the
|
|
4
|
+
structural diff resolve links through these pure helpers so path-links and
|
|
5
|
+
``[[wikilinks]]`` behave identically everywhere. No database access here —
|
|
6
|
+
callers supply the known-id set / name index.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import re
|
|
12
|
+
from typing import Any, Dict, Iterable, List, Mapping, Optional, Set, Tuple
|
|
13
|
+
|
|
14
|
+
MD_LINK_RE = re.compile(r"\[.*?\]\((.*?\.md)\)")
|
|
15
|
+
WIKI_RE = re.compile(r"\[\[(.*?)\]\]")
|
|
16
|
+
SCHEME_RE = re.compile(r"^[a-zA-Z][a-zA-Z0-9+.-]*:")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def extract_md_links(body: str) -> List[str]:
|
|
20
|
+
"""Raw targets of ``[text](target.md)`` links (external URLs included)."""
|
|
21
|
+
return MD_LINK_RE.findall(body or "")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def extract_wikilinks(body: str) -> List[str]:
|
|
25
|
+
"""Raw targets of ``[[target]]`` / ``[[target|display]]`` (display stripped)."""
|
|
26
|
+
return [
|
|
27
|
+
m.split("|", 1)[0].strip()
|
|
28
|
+
for m in WIKI_RE.findall(body or "")
|
|
29
|
+
]
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def is_external(raw: str) -> bool:
|
|
33
|
+
"""True for ``scheme:...`` targets (http, mailto, okf-asset, ...)."""
|
|
34
|
+
return bool(SCHEME_RE.match((raw.split("#", 1)[0]).strip()))
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def normalize_path_link(raw: str) -> str:
|
|
38
|
+
"""Map a markdown link target to a concept id (path without extension)."""
|
|
39
|
+
return raw.lstrip("./").replace("\\", "/").replace(".md", "").lstrip("/")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _as_list(value: Any) -> List[str]:
|
|
43
|
+
if value is None:
|
|
44
|
+
return []
|
|
45
|
+
if isinstance(value, str):
|
|
46
|
+
return [value]
|
|
47
|
+
try:
|
|
48
|
+
return [str(v) for v in value]
|
|
49
|
+
except TypeError: # pragma: no cover - defensive
|
|
50
|
+
return [str(value)]
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def build_name_index(
|
|
54
|
+
concepts: Iterable[Mapping[str, Any]],
|
|
55
|
+
) -> Tuple[Dict[str, Dict[str, str]], Set[str]]:
|
|
56
|
+
"""Build the wikilink name index.
|
|
57
|
+
|
|
58
|
+
Each concept mapping needs ``id`` plus any of ``uid`` (frontmatter
|
|
59
|
+
``id:``, preserved as ``uid`` on the node), ``aliases``/``alias``,
|
|
60
|
+
``title``. Keys are lowercased; precedence at resolution is
|
|
61
|
+
uid → alias → title → filename-stem.
|
|
62
|
+
|
|
63
|
+
Returns ``(maps, ambiguous)`` where ``maps`` is
|
|
64
|
+
``{"uid": {...}, "alias": {...}, "title": {...}, "stem": {...}}`` and
|
|
65
|
+
``ambiguous`` holds every key claimed by more than one concept.
|
|
66
|
+
Ambiguous names never resolve (deterministic miss, never a guess).
|
|
67
|
+
"""
|
|
68
|
+
maps: Dict[str, Dict[str, str]] = {
|
|
69
|
+
"uid": {}, "alias": {}, "title": {}, "stem": {},
|
|
70
|
+
}
|
|
71
|
+
claimed: Dict[str, Dict[str, str]] = {
|
|
72
|
+
"uid": {}, "alias": {}, "title": {}, "stem": {},
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
def _add(kind: str, key: str, cid: str) -> None:
|
|
76
|
+
key = (key or "").strip().lower()
|
|
77
|
+
if not key:
|
|
78
|
+
return
|
|
79
|
+
if key in maps[kind] and maps[kind][key] != cid:
|
|
80
|
+
claimed[kind][key] = cid
|
|
81
|
+
else:
|
|
82
|
+
maps[kind].setdefault(key, cid)
|
|
83
|
+
|
|
84
|
+
for c in concepts:
|
|
85
|
+
cid = c.get("id", "")
|
|
86
|
+
if not cid:
|
|
87
|
+
continue
|
|
88
|
+
uid = c.get("uid")
|
|
89
|
+
if uid is not None:
|
|
90
|
+
_add("uid", str(uid), cid)
|
|
91
|
+
for a in _as_list(c.get("aliases")) + _as_list(c.get("alias")):
|
|
92
|
+
_add("alias", a, cid)
|
|
93
|
+
title = c.get("title")
|
|
94
|
+
if title:
|
|
95
|
+
_add("title", str(title), cid)
|
|
96
|
+
stem = cid.rsplit("/", 1)[-1]
|
|
97
|
+
_add("stem", stem, cid)
|
|
98
|
+
|
|
99
|
+
ambiguous: Set[str] = set()
|
|
100
|
+
for kind in maps:
|
|
101
|
+
for key in claimed[kind]:
|
|
102
|
+
maps[kind].pop(key, None)
|
|
103
|
+
ambiguous.add(f"{kind}:{key}")
|
|
104
|
+
return maps, ambiguous
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def resolve_wiki(
|
|
108
|
+
reference: str,
|
|
109
|
+
maps: Mapping[str, Mapping[str, str]],
|
|
110
|
+
known_ids: Set[str],
|
|
111
|
+
) -> Optional[str]:
|
|
112
|
+
"""Resolve a wikilink reference to a concept id, or None.
|
|
113
|
+
|
|
114
|
+
Order: exact concept id → id-minus-``.md`` → uid → alias → title → stem
|
|
115
|
+
(all case-insensitive except the exact-id probe). Fragment (``#sec``) and
|
|
116
|
+
surrounding whitespace are ignored.
|
|
117
|
+
"""
|
|
118
|
+
ref = (reference or "").split("#", 1)[0].strip()
|
|
119
|
+
if not ref:
|
|
120
|
+
return None
|
|
121
|
+
if ref in known_ids:
|
|
122
|
+
return ref
|
|
123
|
+
cand = ref[:-3] if ref.lower().endswith(".md") else ref
|
|
124
|
+
if cand in known_ids:
|
|
125
|
+
return cand
|
|
126
|
+
low = ref.lower()
|
|
127
|
+
if low.endswith(".md"):
|
|
128
|
+
low = low[:-3]
|
|
129
|
+
for kind in ("uid", "alias", "title", "stem"):
|
|
130
|
+
hit = maps.get(kind, {}).get(low)
|
|
131
|
+
if hit is not None:
|
|
132
|
+
return hit
|
|
133
|
+
return None
|