okfgraph 0.2.4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- okfgraph/__init__.py +7 -0
- okfgraph/cli.py +1364 -0
- okfgraph/components/__init__.py +61 -0
- okfgraph/components/converters.py +128 -0
- okfgraph/components/delta.py +271 -0
- okfgraph/components/diff.py +172 -0
- okfgraph/components/doctor.py +216 -0
- okfgraph/components/embedding.py +323 -0
- okfgraph/components/export.py +354 -0
- okfgraph/components/image_assets.py +319 -0
- okfgraph/components/import_.py +1110 -0
- okfgraph/components/ingest.py +495 -0
- okfgraph/components/links.py +133 -0
- okfgraph/components/lint.py +118 -0
- okfgraph/components/purge.py +342 -0
- okfgraph/components/ranking.py +168 -0
- okfgraph/components/schema.py +443 -0
- okfgraph/components/search.py +1006 -0
- okfgraph/config.py +393 -0
- okfgraph/images.py +435 -0
- okfgraph/mcp_server.py +461 -0
- okfgraph/models.py +128 -0
- okfgraph/router.py +485 -0
- okfgraph/security.py +269 -0
- okfgraph/tools.py +460 -0
- okfgraph-0.2.4.dist-info/METADATA +25 -0
- okfgraph-0.2.4.dist-info/RECORD +31 -0
- okfgraph-0.2.4.dist-info/WHEEL +5 -0
- okfgraph-0.2.4.dist-info/entry_points.txt +3 -0
- okfgraph-0.2.4.dist-info/licenses/LICENSE +6 -0
- okfgraph-0.2.4.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
"""Pre-import bundle gate: frontmatter + link validation, no DB, no model.
|
|
2
|
+
|
|
3
|
+
``lint_bundle()`` checks the file side (what ``import --all`` is *about*
|
|
4
|
+
to ingest); doctor checks the row side (what is *already* indexed).
|
|
5
|
+
Same ``links.py`` pure helpers, same resolution rules as import —
|
|
6
|
+
a lint-clean bundle must import with zero ``broken_link`` findings
|
|
7
|
+
(locked by the consistency test in ``tests/test_lint.py``).
|
|
8
|
+
|
|
9
|
+
Report shape (JSON-stable, all lists sorted)::
|
|
10
|
+
{"dir": str, "files": int,
|
|
11
|
+
"errors": [{"file", "rule", "message", ...}],
|
|
12
|
+
"warnings": [{"file", "rule", "message"}],
|
|
13
|
+
"clean": bool} # True when errors is empty (warnings allowed)
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import logging
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import Any, Dict, List, Tuple
|
|
21
|
+
|
|
22
|
+
import frontmatter
|
|
23
|
+
|
|
24
|
+
from okfgraph.components.import_ import is_concept_file, parse_source_file
|
|
25
|
+
from okfgraph.components.links import (
|
|
26
|
+
build_name_index,
|
|
27
|
+
extract_md_links,
|
|
28
|
+
extract_wikilinks,
|
|
29
|
+
is_external,
|
|
30
|
+
normalize_path_link,
|
|
31
|
+
resolve_wiki,
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
logger = logging.getLogger(__name__)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _err(file: str, rule: str, message: str, **extra: Any) -> Dict[str, Any]:
|
|
38
|
+
item: Dict[str, Any] = {"file": file, "rule": rule, "message": message}
|
|
39
|
+
item.update(extra)
|
|
40
|
+
return item
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def lint_bundle(bundle_dir: str | Path) -> Dict[str, Any]:
|
|
44
|
+
"""Validate a bundle directory without touching any database or model."""
|
|
45
|
+
root = Path(bundle_dir)
|
|
46
|
+
files = sorted(fp for fp in root.rglob("*") if is_concept_file(fp))
|
|
47
|
+
|
|
48
|
+
errors: List[Dict[str, Any]] = []
|
|
49
|
+
warnings: List[Dict[str, Any]] = []
|
|
50
|
+
parsed: List[Tuple[str, Any, str, str]] = [] # (cid, concept, body, rel)
|
|
51
|
+
|
|
52
|
+
for fp in files:
|
|
53
|
+
rel = str(fp.relative_to(root)).replace("\\", "/")
|
|
54
|
+
try:
|
|
55
|
+
concept, body, cid = parse_source_file(fp, root)
|
|
56
|
+
except Exception as e: # noqa: BLE001 — every failure mode is a finding
|
|
57
|
+
errors.append(_err(rel, "parse", f"{type(e).__name__}: {e}"))
|
|
58
|
+
continue
|
|
59
|
+
try:
|
|
60
|
+
raw_fm = dict(frontmatter.load(fp).metadata)
|
|
61
|
+
except Exception as e: # noqa: BLE001 — same finding class as above
|
|
62
|
+
errors.append(_err(rel, "parse", f"{type(e).__name__}: {e}"))
|
|
63
|
+
continue
|
|
64
|
+
# Import synthesizes both (note / stem), so these warn — erroring
|
|
65
|
+
# would contradict import behaviour (deliberate deviation from
|
|
66
|
+
# google-okf, whose pipeline has no synthesis step).
|
|
67
|
+
if not raw_fm.get("type"):
|
|
68
|
+
warnings.append(_err(
|
|
69
|
+
rel, "missing_type", "no 'type:' frontmatter; will import as 'note'"))
|
|
70
|
+
if not raw_fm.get("title"):
|
|
71
|
+
warnings.append(_err(
|
|
72
|
+
rel, "missing_title", "no 'title:' frontmatter; will use filename stem"))
|
|
73
|
+
parsed.append((cid, concept, body, rel))
|
|
74
|
+
|
|
75
|
+
# Resolution context mirrors ImportManager: known ids + name index.
|
|
76
|
+
known_ids = {cid for cid, _, _, _ in parsed}
|
|
77
|
+
index_entries = []
|
|
78
|
+
for cid, concept, _, _ in parsed:
|
|
79
|
+
extra = concept.model_extra or {}
|
|
80
|
+
index_entries.append({
|
|
81
|
+
"id": cid,
|
|
82
|
+
"title": concept.title,
|
|
83
|
+
"uid": extra.get("uid"),
|
|
84
|
+
"aliases": extra.get("aliases", extra.get("alias")),
|
|
85
|
+
})
|
|
86
|
+
maps, _ambiguous = build_name_index(index_entries)
|
|
87
|
+
|
|
88
|
+
for cid, _concept, body, rel in parsed:
|
|
89
|
+
# Same extraction + skip rules as _resolve_body_links: only
|
|
90
|
+
# *.md-anchored targets match; externals never resolve-or-fail.
|
|
91
|
+
for raw in extract_md_links(body):
|
|
92
|
+
if is_external(raw):
|
|
93
|
+
continue
|
|
94
|
+
target = normalize_path_link(raw.split("#", 1)[0])
|
|
95
|
+
if target not in known_ids:
|
|
96
|
+
errors.append(_err(
|
|
97
|
+
rel, "dangling_link",
|
|
98
|
+
f"would import as BrokenLink (target has no concept)",
|
|
99
|
+
link=raw, target=target or "(empty)"))
|
|
100
|
+
for raw in extract_wikilinks(body):
|
|
101
|
+
if not raw or is_external(raw):
|
|
102
|
+
continue
|
|
103
|
+
if resolve_wiki(raw, maps, known_ids) is None:
|
|
104
|
+
errors.append(_err(
|
|
105
|
+
rel, "dangling_wikilink",
|
|
106
|
+
"would import as BrokenLink (name resolves to nothing; "
|
|
107
|
+
"ambiguous names never resolve)",
|
|
108
|
+
link=raw))
|
|
109
|
+
|
|
110
|
+
errors.sort(key=lambda e: (e["file"], e["rule"], e["message"]))
|
|
111
|
+
warnings.sort(key=lambda w: (w["file"], w["rule"], w["message"]))
|
|
112
|
+
return {
|
|
113
|
+
"dir": str(root),
|
|
114
|
+
"files": len(files),
|
|
115
|
+
"errors": errors,
|
|
116
|
+
"warnings": warnings,
|
|
117
|
+
"clean": not errors,
|
|
118
|
+
}
|
|
@@ -0,0 +1,342 @@
|
|
|
1
|
+
"""PurgeManager — hard purge, soft-delete, and recovery-window management.
|
|
2
|
+
|
|
3
|
+
Extracted from OKFRouter (purge + soft-delete-with-recovery sections).
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
import logging
|
|
8
|
+
from datetime import datetime, timezone, timedelta
|
|
9
|
+
from typing import Any, Callable, Dict, List, Optional, Tuple
|
|
10
|
+
|
|
11
|
+
from okfgraph.models import ChunkModel, ConceptModel
|
|
12
|
+
|
|
13
|
+
logger = logging.getLogger(__name__)
|
|
14
|
+
|
|
15
|
+
class PurgeManager:
|
|
16
|
+
def __init__(self, conn, write_lock_ctx: Callable):
|
|
17
|
+
self.conn = conn
|
|
18
|
+
self._write_lock_ctx = write_lock_ctx
|
|
19
|
+
|
|
20
|
+
SOFT_DELETE_WINDOW = 24 * 60 * 60
|
|
21
|
+
|
|
22
|
+
def _purge_concept(self, concept_id: str) -> bool:
|
|
23
|
+
"""Delete a concept and all its dependents from the graph.
|
|
24
|
+
|
|
25
|
+
Removes:
|
|
26
|
+
- The Concept node (and its embedding)
|
|
27
|
+
- All Chunk nodes linked via PART_OF
|
|
28
|
+
- All LINKS_TO relationships (incoming and outgoing)
|
|
29
|
+
- All INCLUDES_ASSET relationships
|
|
30
|
+
- ImageAsset nodes that have **zero remaining** INCLUDES_ASSET
|
|
31
|
+
edges (orphan check — shared assets are preserved)
|
|
32
|
+
- BrokenLink entries where this concept was source or target
|
|
33
|
+
- FileHash entry for the concept's file path
|
|
34
|
+
|
|
35
|
+
Args:
|
|
36
|
+
concept_id: The ID of the concept to purge.
|
|
37
|
+
|
|
38
|
+
Returns:
|
|
39
|
+
True if a concept was found and purged, False if not found.
|
|
40
|
+
"""
|
|
41
|
+
# Check if the concept exists
|
|
42
|
+
rows = self.conn.execute(
|
|
43
|
+
"MATCH (c:Concept {id: $id}) RETURN count(c) AS cnt",
|
|
44
|
+
{"id": concept_id},
|
|
45
|
+
).rows_as_dict().get_all()
|
|
46
|
+
if not rows or rows[0]["cnt"] == 0:
|
|
47
|
+
logger.debug("purge: concept %s not found, skipping", concept_id)
|
|
48
|
+
return False
|
|
49
|
+
|
|
50
|
+
self.conn.execute("BEGIN TRANSACTION")
|
|
51
|
+
try:
|
|
52
|
+
# 1. Find and sever INCLUDES_ASSET edges, collecting asset IDs.
|
|
53
|
+
asset_rows = self.conn.execute(
|
|
54
|
+
"""
|
|
55
|
+
MATCH (c:Concept {id: $id})-[:INCLUDES_ASSET]->(i:ImageAsset)
|
|
56
|
+
RETURN i.id AS aid
|
|
57
|
+
""",
|
|
58
|
+
{"id": concept_id},
|
|
59
|
+
).rows_as_dict().get_all()
|
|
60
|
+
asset_ids = [r["aid"] for r in asset_rows]
|
|
61
|
+
|
|
62
|
+
# 2. Delete all Chunk nodes for this concept (DETACH DELETE on
|
|
63
|
+
# Concept doesn't cascade to Chunk — they are separate nodes).
|
|
64
|
+
self.conn.execute(
|
|
65
|
+
"""
|
|
66
|
+
MATCH (ch:Chunk {parent_doc_id: $id})
|
|
67
|
+
DETACH DELETE ch
|
|
68
|
+
""",
|
|
69
|
+
{"id": concept_id},
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
# 3. DETACH DELETE the Concept (cascades LINKS_TO, CONTAINS,
|
|
73
|
+
# INCLUDES_ASSET edges).
|
|
74
|
+
self.conn.execute(
|
|
75
|
+
"MATCH (c:Concept {id: $id}) DETACH DELETE c",
|
|
76
|
+
{"id": concept_id},
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
# 4. Clean up orphaned ImageAssets — only delete if no other
|
|
80
|
+
# Concept still references them. This handles the case where
|
|
81
|
+
# two files share the same okf-asset://<id> URI.
|
|
82
|
+
for aid in asset_ids:
|
|
83
|
+
ref_count = self.conn.execute(
|
|
84
|
+
"""
|
|
85
|
+
MATCH (i:ImageAsset {id: $aid})
|
|
86
|
+
OPTIONAL MATCH (i)<-[:INCLUDES_ASSET]-(other:Concept)
|
|
87
|
+
RETURN count(other) AS refs
|
|
88
|
+
""",
|
|
89
|
+
{"aid": aid},
|
|
90
|
+
).rows_as_dict().get_all()
|
|
91
|
+
if ref_count and ref_count[0]["refs"] == 0:
|
|
92
|
+
self.conn.execute(
|
|
93
|
+
"MATCH (i:ImageAsset {id: $aid}) DELETE i",
|
|
94
|
+
{"aid": aid},
|
|
95
|
+
)
|
|
96
|
+
logger.debug("purge: deleted orphaned asset %s", aid)
|
|
97
|
+
|
|
98
|
+
# 5. Clean up BrokenLink entries where this concept was source
|
|
99
|
+
# or target.
|
|
100
|
+
self.conn.execute(
|
|
101
|
+
"""
|
|
102
|
+
MATCH (bl:BrokenLink)
|
|
103
|
+
WHERE bl.source_id = $id OR bl.target_id = $id
|
|
104
|
+
DELETE bl
|
|
105
|
+
""",
|
|
106
|
+
{"id": concept_id},
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
# 6. Remove FileHash entry for this concept.
|
|
110
|
+
self.conn.execute(
|
|
111
|
+
"""
|
|
112
|
+
MATCH (f:FileHash {concept_id: $id})
|
|
113
|
+
DELETE f
|
|
114
|
+
""",
|
|
115
|
+
{"id": concept_id},
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
self.conn.execute("COMMIT")
|
|
119
|
+
logger.info("purge: deleted concept %s", concept_id)
|
|
120
|
+
return True
|
|
121
|
+
|
|
122
|
+
except Exception:
|
|
123
|
+
try:
|
|
124
|
+
self.conn.execute("ROLLBACK")
|
|
125
|
+
except Exception:
|
|
126
|
+
pass
|
|
127
|
+
raise
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _soft_delete_concept(self, concept_id: str) -> bool:
|
|
131
|
+
"""Soft-delete a concept by moving it to the DeletedConcept table.
|
|
132
|
+
|
|
133
|
+
The concept is preserved with a timestamp so it can be recovered
|
|
134
|
+
within the recovery window. After the window expires, the concept
|
|
135
|
+
is permanently deleted.
|
|
136
|
+
|
|
137
|
+
Args:
|
|
138
|
+
concept_id: The ID of the concept to soft-delete.
|
|
139
|
+
|
|
140
|
+
Returns:
|
|
141
|
+
True if a concept was found and soft-deleted, False if not found.
|
|
142
|
+
"""
|
|
143
|
+
# Acquire write lock (Gap #7b)
|
|
144
|
+
with self._write_lock_ctx():
|
|
145
|
+
return self._soft_delete_concept_inner(concept_id)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _soft_delete_concept_inner(self, concept_id: str) -> bool:
|
|
149
|
+
"""Inner implementation (called under write lock)."""
|
|
150
|
+
# Check if the concept exists
|
|
151
|
+
rows = self.conn.execute(
|
|
152
|
+
"MATCH (c:Concept {id: $id}) RETURN count(c) AS cnt",
|
|
153
|
+
{"id": concept_id},
|
|
154
|
+
).rows_as_dict().get_all()
|
|
155
|
+
if not rows or rows[0]["cnt"] == 0:
|
|
156
|
+
logger.debug("soft_delete: concept %s not found, skipping", concept_id)
|
|
157
|
+
return False
|
|
158
|
+
|
|
159
|
+
# Fetch concept data for preservation
|
|
160
|
+
concept_rows = self.conn.execute(
|
|
161
|
+
"""
|
|
162
|
+
MATCH (c:Concept {id: $id})
|
|
163
|
+
RETURN c.title AS title, c.body AS body, c.type AS ctype, c.tags AS tags
|
|
164
|
+
""",
|
|
165
|
+
{"id": concept_id},
|
|
166
|
+
).rows_as_dict().get_all()
|
|
167
|
+
if not concept_rows:
|
|
168
|
+
return False
|
|
169
|
+
|
|
170
|
+
concept_data = concept_rows[0]
|
|
171
|
+
deleted_at = datetime.now().isoformat()
|
|
172
|
+
|
|
173
|
+
# Store in DeletedConcept table
|
|
174
|
+
self.conn.execute(
|
|
175
|
+
"""
|
|
176
|
+
INSERT INTO DeletedConcept (id, original_id, title, body, deleted_at, type, tags)
|
|
177
|
+
VALUES ($id, $oid, $title, $body, $deleted_at, $type, $tags)
|
|
178
|
+
""",
|
|
179
|
+
{
|
|
180
|
+
"id": f"deleted_{concept_id}_{int(time.time())}",
|
|
181
|
+
"oid": concept_id,
|
|
182
|
+
"title": concept_data.get("title", ""),
|
|
183
|
+
"body": concept_data.get("body", ""),
|
|
184
|
+
"deleted_at": deleted_at,
|
|
185
|
+
"type": concept_data.get("ctype", ""),
|
|
186
|
+
"tags": json.dumps(concept_data.get("tags", [])),
|
|
187
|
+
},
|
|
188
|
+
)
|
|
189
|
+
|
|
190
|
+
# Now perform the hard delete
|
|
191
|
+
self._purge_concept(concept_id)
|
|
192
|
+
|
|
193
|
+
logger.info(
|
|
194
|
+
"soft_delete: concept %s moved to DeletedConcept (recovery until %s)",
|
|
195
|
+
concept_id,
|
|
196
|
+
(datetime.now() + __import__("datetime").timedelta(seconds=self.SOFT_DELETE_WINDOW)).isoformat(),
|
|
197
|
+
)
|
|
198
|
+
return True
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _recover_concept(self, concept_id: str) -> bool:
|
|
202
|
+
"""Recover a soft-deleted concept from the DeletedConcept table.
|
|
203
|
+
|
|
204
|
+
Args:
|
|
205
|
+
concept_id: The original ID of the concept to recover.
|
|
206
|
+
|
|
207
|
+
Returns:
|
|
208
|
+
True if the concept was recovered, False if not found.
|
|
209
|
+
"""
|
|
210
|
+
# Acquire write lock (Gap #7b)
|
|
211
|
+
with self._write_lock_ctx():
|
|
212
|
+
return self._recover_concept_inner(concept_id)
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def _recover_concept_inner(self, concept_id: str) -> bool:
|
|
216
|
+
"""Inner implementation (called under write lock)."""
|
|
217
|
+
# Check if the concept exists in DeletedConcept
|
|
218
|
+
rows = self.conn.execute(
|
|
219
|
+
"""
|
|
220
|
+
MATCH (d:DeletedConcept {original_id: $id})
|
|
221
|
+
RETURN d.title AS title, d.body AS body, d.type AS ctype, d.tags AS tags, d.deleted_at AS deleted_at
|
|
222
|
+
""",
|
|
223
|
+
{"id": concept_id},
|
|
224
|
+
).rows_as_dict().get_all()
|
|
225
|
+
|
|
226
|
+
if not rows:
|
|
227
|
+
logger.debug("recover: concept %s not found in DeletedConcept", concept_id)
|
|
228
|
+
return False
|
|
229
|
+
|
|
230
|
+
deleted_at = datetime.fromisoformat(rows[0]["deleted_at"])
|
|
231
|
+
now = datetime.now()
|
|
232
|
+
if (now - deleted_at).total_seconds() > self.SOFT_DELETE_WINDOW:
|
|
233
|
+
logger.warning(
|
|
234
|
+
"recover: concept %s deleted at %s is past the recovery window (%ds)",
|
|
235
|
+
concept_id,
|
|
236
|
+
deleted_at.isoformat(),
|
|
237
|
+
self.SOFT_DELETE_WINDOW,
|
|
238
|
+
)
|
|
239
|
+
return False
|
|
240
|
+
|
|
241
|
+
# Restore the concept
|
|
242
|
+
concept_data = rows[0]
|
|
243
|
+
tags = json.loads(concept_data["tags"]) if isinstance(concept_data["tags"], str) else concept_data["tags"]
|
|
244
|
+
|
|
245
|
+
self.conn.execute(
|
|
246
|
+
"""
|
|
247
|
+
INSERT INTO Concept (id, title, body, type, tags)
|
|
248
|
+
VALUES ($id, $title, $body, $type, $tags)
|
|
249
|
+
""",
|
|
250
|
+
{
|
|
251
|
+
"id": concept_id,
|
|
252
|
+
"title": concept_data["title"],
|
|
253
|
+
"body": concept_data["body"],
|
|
254
|
+
"type": concept_data["ctype"],
|
|
255
|
+
"tags": tags,
|
|
256
|
+
},
|
|
257
|
+
)
|
|
258
|
+
|
|
259
|
+
# Remove from DeletedConcept
|
|
260
|
+
self.conn.execute(
|
|
261
|
+
"MATCH (d:DeletedConcept {original_id: $id}) DELETE d",
|
|
262
|
+
{"id": concept_id},
|
|
263
|
+
)
|
|
264
|
+
|
|
265
|
+
logger.info("recover: concept %s restored from DeletedConcept", concept_id)
|
|
266
|
+
return True
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def list_deleted_concepts(self) -> List[Dict[str, Any]]:
|
|
270
|
+
"""List all soft-deleted concepts with recovery status.
|
|
271
|
+
|
|
272
|
+
Returns:
|
|
273
|
+
List of dicts with concept_id, title, deleted_at, and recoverable status.
|
|
274
|
+
"""
|
|
275
|
+
rows = self.conn.execute(
|
|
276
|
+
"""
|
|
277
|
+
MATCH (d:DeletedConcept)
|
|
278
|
+
RETURN d.original_id AS id, d.title AS title, d.deleted_at AS deleted_at, d.type AS type
|
|
279
|
+
ORDER BY d.deleted_at DESC
|
|
280
|
+
"""
|
|
281
|
+
).rows_as_dict().get_all()
|
|
282
|
+
|
|
283
|
+
now = datetime.now()
|
|
284
|
+
results = []
|
|
285
|
+
for row in rows:
|
|
286
|
+
deleted_at = datetime.fromisoformat(row["deleted_at"])
|
|
287
|
+
age_seconds = (now - deleted_at).total_seconds()
|
|
288
|
+
results.append({
|
|
289
|
+
"concept_id": row["id"],
|
|
290
|
+
"title": row["title"],
|
|
291
|
+
"type": row["type"],
|
|
292
|
+
"deleted_at": row["deleted_at"],
|
|
293
|
+
"age_seconds": age_seconds,
|
|
294
|
+
"recoverable": age_seconds <= self.SOFT_DELETE_WINDOW,
|
|
295
|
+
})
|
|
296
|
+
return results
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def purge_deleted_concepts(self, older_than: Optional[int] = None) -> int:
|
|
300
|
+
"""Permanently delete soft-deleted concepts past the recovery window.
|
|
301
|
+
|
|
302
|
+
Args:
|
|
303
|
+
older_than: Optional override for the recovery window (seconds).
|
|
304
|
+
Defaults to SOFT_DELETE_WINDOW.
|
|
305
|
+
|
|
306
|
+
Returns:
|
|
307
|
+
Number of concepts permanently deleted.
|
|
308
|
+
"""
|
|
309
|
+
# Acquire write lock (Gap #7b)
|
|
310
|
+
with self._write_lock_ctx():
|
|
311
|
+
return self._purge_deleted_concepts_inner(older_than)
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
def _purge_deleted_concepts_inner(self, older_than: Optional[int]) -> int:
|
|
315
|
+
"""Inner implementation (called under write lock)."""
|
|
316
|
+
threshold = older_than if older_than is not None else self.SOFT_DELETE_WINDOW
|
|
317
|
+
cutoff = datetime.now() - __import__("datetime").timedelta(seconds=threshold)
|
|
318
|
+
|
|
319
|
+
# Find expired entries
|
|
320
|
+
rows = self.conn.execute(
|
|
321
|
+
"""
|
|
322
|
+
MATCH (d:DeletedConcept)
|
|
323
|
+
WHERE d.deleted_at < $cutoff
|
|
324
|
+
RETURN d.original_id AS id
|
|
325
|
+
""",
|
|
326
|
+
{"cutoff": cutoff.isoformat()},
|
|
327
|
+
).rows_as_dict().get_all()
|
|
328
|
+
|
|
329
|
+
count = 0
|
|
330
|
+
for row in rows:
|
|
331
|
+
# Permanently delete (hard purge)
|
|
332
|
+
if self._purge_concept(row["id"]):
|
|
333
|
+
count += 1
|
|
334
|
+
# Remove from DeletedConcept
|
|
335
|
+
self.conn.execute(
|
|
336
|
+
"MATCH (d:DeletedConcept {original_id: $id}) DELETE d",
|
|
337
|
+
{"id": row["id"]},
|
|
338
|
+
)
|
|
339
|
+
|
|
340
|
+
logger.info("purge_deleted: permanently deleted %d expired concept(s)", count)
|
|
341
|
+
return count
|
|
342
|
+
|
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
"""Model-free retrieval: lexical seeds + exact Personalized PageRank.
|
|
2
|
+
|
|
3
|
+
Mirrors the okf-ingest ``seeds`` → multi-seed PPR cascade: a query is scored
|
|
4
|
+
lexically against concept metadata, the top seeds weight an exact
|
|
5
|
+
power-iteration PPR over the resolved ``LINKS_TO`` graph, and the ranked
|
|
6
|
+
concepts come back — with no embedding model loaded. Over a sparse graph this
|
|
7
|
+
degrades gracefully to seed order (== the lexical baseline).
|
|
8
|
+
|
|
9
|
+
Determinism rules (all load-bearing for the golden fixtures):
|
|
10
|
+
- seed ties break by concept id ascending;
|
|
11
|
+
- edges iterate in sorted ``(src, dst)`` index order;
|
|
12
|
+
- the L1 convergence sum accumulates in ascending node order;
|
|
13
|
+
- scores round half-even to 10 decimals (Python ``round``).
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import re
|
|
19
|
+
from typing import Any, Dict, List, Mapping, Optional, Sequence, Set, Tuple
|
|
20
|
+
|
|
21
|
+
TOKEN_RE = re.compile(r"[a-z0-9]+")
|
|
22
|
+
|
|
23
|
+
#: Fixed stopword list — keep in sync with tests/fixtures, never grow silently.
|
|
24
|
+
STOPWORDS = frozenset({
|
|
25
|
+
"the", "and", "for", "are", "was", "were", "with", "that", "this",
|
|
26
|
+
"from", "how", "what", "when", "where", "which", "does", "did",
|
|
27
|
+
"can", "could", "should", "would", "will", "has", "have", "had",
|
|
28
|
+
"not", "its", "our", "your", "their", "about", "into", "over",
|
|
29
|
+
"under", "why", "who", "whom",
|
|
30
|
+
})
|
|
31
|
+
|
|
32
|
+
TITLE_HIT = 3
|
|
33
|
+
META_HIT = 2 # description or tags
|
|
34
|
+
BODY_HIT = 1
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def query_tokens(query: str) -> List[str]:
|
|
38
|
+
"""Lowercase alphanumeric tokens, len >= 3, minus stopwords (sorted)."""
|
|
39
|
+
return sorted({
|
|
40
|
+
m for m in TOKEN_RE.findall((query or "").lower())
|
|
41
|
+
if len(m) >= 3 and m not in STOPWORDS
|
|
42
|
+
})
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def seeds(
|
|
46
|
+
concepts: Sequence[Mapping[str, Any]],
|
|
47
|
+
query: str,
|
|
48
|
+
k: int = 20,
|
|
49
|
+
) -> List[Tuple[str, float]]:
|
|
50
|
+
"""Lexically score concepts against the query.
|
|
51
|
+
|
|
52
|
+
``+3`` per distinct token in the title, ``+2`` in description/tags,
|
|
53
|
+
``+1`` in the body. Returns ``(concept_id, score)`` sorted by score
|
|
54
|
+
desc, id asc, truncated at ``k``. Pure function — no DB, no model.
|
|
55
|
+
"""
|
|
56
|
+
toks = query_tokens(query)
|
|
57
|
+
if not toks:
|
|
58
|
+
return []
|
|
59
|
+
out: List[Tuple[str, float]] = []
|
|
60
|
+
for c in concepts:
|
|
61
|
+
cid = c.get("id", "")
|
|
62
|
+
if not cid:
|
|
63
|
+
continue
|
|
64
|
+
title = str(c.get("title") or "").lower()
|
|
65
|
+
desc = str(c.get("description") or "").lower()
|
|
66
|
+
tags = c.get("tags") or []
|
|
67
|
+
tags_text = " ".join(str(t) for t in tags).lower() if isinstance(tags, list) else str(tags).lower()
|
|
68
|
+
body = str(c.get("body") or "").lower()
|
|
69
|
+
score = 0
|
|
70
|
+
for t in toks:
|
|
71
|
+
if t in title:
|
|
72
|
+
score += TITLE_HIT
|
|
73
|
+
if t in desc or t in tags_text:
|
|
74
|
+
score += META_HIT
|
|
75
|
+
if t in body:
|
|
76
|
+
score += BODY_HIT
|
|
77
|
+
if score > 0:
|
|
78
|
+
out.append((cid, float(score)))
|
|
79
|
+
out.sort(key=lambda kv: (-kv[1], kv[0]))
|
|
80
|
+
return out[:k]
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def ppr(
|
|
84
|
+
nodes: Sequence[str],
|
|
85
|
+
directed_edges: Sequence[Tuple[str, str]],
|
|
86
|
+
starts: Sequence[str],
|
|
87
|
+
weights: Optional[Sequence[float]] = None,
|
|
88
|
+
damping: float = 0.85,
|
|
89
|
+
tol: float = 1e-12,
|
|
90
|
+
max_iter: int = 200,
|
|
91
|
+
k: Optional[int] = None,
|
|
92
|
+
) -> List[Tuple[str, float]]:
|
|
93
|
+
"""Exact Personalized PageRank over the undirected link graph.
|
|
94
|
+
|
|
95
|
+
``nodes`` need not be sorted (sorted internally); ``directed_edges`` are
|
|
96
|
+
symmetrized, self-loops dropped. Dangling mass returns to the seed
|
|
97
|
+
distribution. Unknown start ids raise ``KeyError``.
|
|
98
|
+
"""
|
|
99
|
+
ids = sorted(set(nodes))
|
|
100
|
+
idx = {nid: i for i, nid in enumerate(ids)}
|
|
101
|
+
missing = [s for s in starts if s not in idx]
|
|
102
|
+
if missing:
|
|
103
|
+
raise KeyError(f"start concept not found: {', '.join(missing)}")
|
|
104
|
+
w = list(weights) if weights is not None else [1.0] * len(starts)
|
|
105
|
+
if len(w) != len(starts) or sum(w) <= 0 or any(x < 0 for x in w):
|
|
106
|
+
raise ValueError("weights must be non-negative, same length as starts, positive sum")
|
|
107
|
+
|
|
108
|
+
n = len(ids)
|
|
109
|
+
edges: Set[Tuple[int, int]] = set()
|
|
110
|
+
for s, d in directed_edges:
|
|
111
|
+
if s == d or s not in idx or d not in idx:
|
|
112
|
+
continue
|
|
113
|
+
edges.add((idx[s], idx[d]))
|
|
114
|
+
edges.add((idx[d], idx[s]))
|
|
115
|
+
ordered = sorted(edges)
|
|
116
|
+
deg = [0] * n
|
|
117
|
+
for s, _ in ordered:
|
|
118
|
+
deg[s] += 1
|
|
119
|
+
|
|
120
|
+
seed = [0.0] * n
|
|
121
|
+
for s, x in zip(starts, w):
|
|
122
|
+
seed[idx[s]] += x
|
|
123
|
+
total = sum(seed)
|
|
124
|
+
seed = [x / total for x in seed]
|
|
125
|
+
|
|
126
|
+
p = list(seed)
|
|
127
|
+
for _ in range(max_iter):
|
|
128
|
+
contrib = [0.0] * n
|
|
129
|
+
for s, d in ordered: # fixed order -> deterministic fp
|
|
130
|
+
if p[s] != 0.0:
|
|
131
|
+
contrib[d] += p[s] / deg[s]
|
|
132
|
+
dangling = sum(p[i] for i in range(n) if deg[i] == 0)
|
|
133
|
+
nxt = [
|
|
134
|
+
(1.0 - damping) * seed[i] + damping * (contrib[i] + dangling * seed[i])
|
|
135
|
+
for i in range(n)
|
|
136
|
+
]
|
|
137
|
+
delta = sum(abs(nxt[i] - p[i]) for i in range(n)) # ascending order
|
|
138
|
+
p = nxt
|
|
139
|
+
if delta < tol:
|
|
140
|
+
break
|
|
141
|
+
|
|
142
|
+
rows = [
|
|
143
|
+
(nid, round(p[i], 10))
|
|
144
|
+
for i, nid in enumerate(ids)
|
|
145
|
+
if round(p[i], 10) > 0.0
|
|
146
|
+
]
|
|
147
|
+
rows.sort(key=lambda kv: (-kv[1], kv[0]))
|
|
148
|
+
return rows[:k] if k is not None else rows
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def seed_ranked_ppr(
|
|
152
|
+
concepts: Sequence[Mapping[str, Any]],
|
|
153
|
+
directed_edges: Sequence[Tuple[str, str]],
|
|
154
|
+
query: str,
|
|
155
|
+
seed_k: int = 20,
|
|
156
|
+
k: Optional[int] = None,
|
|
157
|
+
) -> List[Tuple[str, float]]:
|
|
158
|
+
"""One-call cascade: lexical seeds (as weights) → multi-seed PPR."""
|
|
159
|
+
found = seeds(concepts, query, k=seed_k)
|
|
160
|
+
if not found:
|
|
161
|
+
return []
|
|
162
|
+
ids = [cid for cid, _ in found]
|
|
163
|
+
weights = [s for _, s in found]
|
|
164
|
+
known = {c.get("id") for c in concepts}
|
|
165
|
+
return ppr(
|
|
166
|
+
[c.get("id", "") for c in concepts if c.get("id") in known],
|
|
167
|
+
directed_edges, ids, weights, k=k,
|
|
168
|
+
)
|