treesearchlib 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- treesearch/__init__.py +53 -0
- treesearch/__main__.py +6 -0
- treesearch/_bin/pst-extract.exe +0 -0
- treesearch/cli.py +554 -0
- treesearch/config.py +206 -0
- treesearch/fts.py +2293 -0
- treesearch/heuristics.py +425 -0
- treesearch/indexer.py +2038 -0
- treesearch/parsers/__init__.py +62 -0
- treesearch/parsers/anydoc_parser.py +193 -0
- treesearch/parsers/ast_parser.py +136 -0
- treesearch/parsers/docx_parser.py +304 -0
- treesearch/parsers/email_html_md.py +60 -0
- treesearch/parsers/excel_parser.py +218 -0
- treesearch/parsers/html_parser.py +172 -0
- treesearch/parsers/image_metadata.py +345 -0
- treesearch/parsers/image_parser.py +59 -0
- treesearch/parsers/image_store.py +182 -0
- treesearch/parsers/markitdown_parser.py +258 -0
- treesearch/parsers/mhtml_parser.py +108 -0
- treesearch/parsers/pdf_parser.py +409 -0
- treesearch/parsers/pst_attachment_store.py +156 -0
- treesearch/parsers/pst_parser.py +733 -0
- treesearch/parsers/registry.py +405 -0
- treesearch/parsers/treesitter_parser.py +433 -0
- treesearch/pathutil.py +227 -0
- treesearch/py.typed +0 -0
- treesearch/ripgrep.py +159 -0
- treesearch/search.py +935 -0
- treesearch/tokenizer.py +176 -0
- treesearch/tree.py +393 -0
- treesearch/tree_searcher.py +1006 -0
- treesearch/treesearch.py +574 -0
- treesearch/watch.py +305 -0
- treesearchlib-1.1.0.dist-info/METADATA +124 -0
- treesearchlib-1.1.0.dist-info/RECORD +39 -0
- treesearchlib-1.1.0.dist-info/WHEEL +5 -0
- treesearchlib-1.1.0.dist-info/entry_points.txt +2 -0
- treesearchlib-1.1.0.dist-info/top_level.txt +1 -0
treesearch/fts.py
ADDED
|
@@ -0,0 +1,2293 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""
|
|
3
|
+
@author:XuMing(xuming624@qq.com)
|
|
4
|
+
@description: SQLite FTS5 full-text search engine for tree-structured documents.
|
|
5
|
+
|
|
6
|
+
Single-file storage: tree structures, FTS5 indexes, and incremental metadata
|
|
7
|
+
are all stored in one SQLite database (.db file).
|
|
8
|
+
|
|
9
|
+
Architecture: "SQLite FTS5 + Producer-Consumer"
|
|
10
|
+
- Deferred indexing via WAL mode solves real-time freshness
|
|
11
|
+
- Local SQL execution handles aggregation needs
|
|
12
|
+
- FTS5 inverted index guarantees retrieval performance
|
|
13
|
+
- Tree structure persistence in `documents.structure_json` column
|
|
14
|
+
|
|
15
|
+
Key features:
|
|
16
|
+
- WAL mode for concurrent read/write
|
|
17
|
+
- Lazy indexing: nodes are inserted on demand, not precomputed
|
|
18
|
+
- MD-aware schema: front_matter, title, summary, body, code_blocks
|
|
19
|
+
- CJK-aware tokenizer: jieba segmentation only when Chinese text is detected
|
|
20
|
+
- Implements PreFilter protocol for seamless integration with search()
|
|
21
|
+
- Hierarchical field boosting via FTS5 column weighting
|
|
22
|
+
"""
|
|
23
|
+
import hashlib
|
|
24
|
+
import json
|
|
25
|
+
import logging
|
|
26
|
+
import os
|
|
27
|
+
import re
|
|
28
|
+
import sqlite3
|
|
29
|
+
from typing import Optional
|
|
30
|
+
|
|
31
|
+
logger = logging.getLogger(__name__)
|
|
32
|
+
|
|
33
|
+
# FTS5 column weights: title > body > summary > code
|
|
34
|
+
# Used in bm25() ranking function
|
|
35
|
+
_DEFAULT_WEIGHTS = {
|
|
36
|
+
"title": 5.0,
|
|
37
|
+
"summary": 2.0,
|
|
38
|
+
"body": 10.0,
|
|
39
|
+
"code_blocks": 1.0,
|
|
40
|
+
"front_matter": 2.0,
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
# ---------------------------------------------------------------------------
|
|
44
|
+
# FTS5 availability detection
|
|
45
|
+
# ---------------------------------------------------------------------------
|
|
46
|
+
|
|
47
|
+
_FTS5_AVAILABLE: Optional[bool] = None
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class _NullContext:
|
|
51
|
+
"""No-op transaction context for callers managing commits externally."""
|
|
52
|
+
def __enter__(self):
|
|
53
|
+
return self
|
|
54
|
+
def __exit__(self, *exc):
|
|
55
|
+
return False
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _check_fts5() -> bool:
|
|
59
|
+
"""Check whether the current SQLite build includes the FTS5 extension."""
|
|
60
|
+
global _FTS5_AVAILABLE
|
|
61
|
+
if _FTS5_AVAILABLE is not None:
|
|
62
|
+
return _FTS5_AVAILABLE
|
|
63
|
+
try:
|
|
64
|
+
conn = sqlite3.connect(":memory:")
|
|
65
|
+
conn.execute("CREATE VIRTUAL TABLE _fts5_test USING fts5(x)")
|
|
66
|
+
conn.execute("DROP TABLE _fts5_test")
|
|
67
|
+
conn.close()
|
|
68
|
+
_FTS5_AVAILABLE = True
|
|
69
|
+
except sqlite3.OperationalError:
|
|
70
|
+
_FTS5_AVAILABLE = False
|
|
71
|
+
logger.warning(
|
|
72
|
+
"SQLite FTS5 extension is not available (SQLite version: %s). "
|
|
73
|
+
"Falling back to plain-text LIKE search. Performance and ranking "
|
|
74
|
+
"quality will be reduced. To fix: upgrade SQLite or install a "
|
|
75
|
+
"Python build that includes FTS5.",
|
|
76
|
+
sqlite3.sqlite_version,
|
|
77
|
+
)
|
|
78
|
+
return _FTS5_AVAILABLE
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _sqlite_regexp(pattern: str, string: str) -> bool:
|
|
82
|
+
"""SQLite REGEXP callback — registered via create_function."""
|
|
83
|
+
if string is None or pattern is None:
|
|
84
|
+
return False
|
|
85
|
+
try:
|
|
86
|
+
return bool(re.search(pattern, string))
|
|
87
|
+
except re.error:
|
|
88
|
+
return False
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
# ---------------------------------------------------------------------------
|
|
92
|
+
# Markdown structure parser
|
|
93
|
+
# ---------------------------------------------------------------------------
|
|
94
|
+
|
|
95
|
+
_RE_FRONT_MATTER = re.compile(r"^---\s*\n(.*?\n)---\s*\n", re.DOTALL)
|
|
96
|
+
_RE_CODE_BLOCK = re.compile(r"```[\w]*\n(.*?)```", re.DOTALL)
|
|
97
|
+
_RE_HEADING_LINE = re.compile(r"^#{1,6}\s+")
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def parse_md_node_text(text: str) -> dict:
|
|
101
|
+
"""Parse a node's text into MD-aware structured fields.
|
|
102
|
+
|
|
103
|
+
Returns:
|
|
104
|
+
{
|
|
105
|
+
"front_matter": str, # YAML front matter (if present)
|
|
106
|
+
"body": str, # main text (headings, paragraphs)
|
|
107
|
+
"code_blocks": str, # concatenated code blocks
|
|
108
|
+
}
|
|
109
|
+
"""
|
|
110
|
+
if not text:
|
|
111
|
+
return {"front_matter": "", "body": "", "code_blocks": ""}
|
|
112
|
+
|
|
113
|
+
front_matter = ""
|
|
114
|
+
remaining = text
|
|
115
|
+
|
|
116
|
+
# Extract front matter
|
|
117
|
+
fm_match = _RE_FRONT_MATTER.match(text)
|
|
118
|
+
if fm_match:
|
|
119
|
+
front_matter = fm_match.group(1).strip()
|
|
120
|
+
remaining = text[fm_match.end():]
|
|
121
|
+
|
|
122
|
+
# Extract code blocks
|
|
123
|
+
code_parts = []
|
|
124
|
+
def _replace_code(m):
|
|
125
|
+
code_parts.append(m.group(1).strip())
|
|
126
|
+
return "" # remove from body
|
|
127
|
+
body = _RE_CODE_BLOCK.sub(_replace_code, remaining)
|
|
128
|
+
code_blocks = "\n".join(code_parts)
|
|
129
|
+
|
|
130
|
+
# Clean body: collapse blank lines
|
|
131
|
+
body = re.sub(r"\n{3,}", "\n\n", body).strip()
|
|
132
|
+
|
|
133
|
+
return {
|
|
134
|
+
"front_matter": front_matter,
|
|
135
|
+
"body": body,
|
|
136
|
+
"code_blocks": code_blocks,
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
# ---------------------------------------------------------------------------
|
|
141
|
+
# Tokenizer for FTS5 (Chinese/English)
|
|
142
|
+
# ---------------------------------------------------------------------------
|
|
143
|
+
|
|
144
|
+
from functools import lru_cache
|
|
145
|
+
|
|
146
|
+
from .tokenizer import _RE_HAS_CJK
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
@lru_cache(maxsize=4096)
|
|
150
|
+
def _tokenize_for_fts(text: str) -> str:
|
|
151
|
+
"""Tokenize text for FTS5 indexing. Space-separated tokens.
|
|
152
|
+
|
|
153
|
+
Only uses jieba segmentation when Chinese (CJK) characters are detected.
|
|
154
|
+
For pure English/non-CJK text, relies on FTS5's built-in unicode61 tokenizer
|
|
155
|
+
(no jieba overhead).
|
|
156
|
+
|
|
157
|
+
Results are LRU-cached to avoid repeated jieba overhead for the same text.
|
|
158
|
+
"""
|
|
159
|
+
if not text or not text.strip():
|
|
160
|
+
return ""
|
|
161
|
+
if _RE_HAS_CJK.search(text):
|
|
162
|
+
from .tokenizer import tokenize
|
|
163
|
+
tokens = tokenize(text)
|
|
164
|
+
return " ".join(tokens)
|
|
165
|
+
# English / non-CJK: return as-is, FTS5 unicode61 handles tokenization
|
|
166
|
+
return text
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
# FTS5 operators that should NOT be tokenized
|
|
170
|
+
_FTS5_OPERATORS = {"AND", "OR", "NOT", "NEAR"}
|
|
171
|
+
|
|
172
|
+
# Characters that should be stripped from FTS5 query tokens.
|
|
173
|
+
# Keep only word characters (letters, digits, underscore) and CJK ranges.
|
|
174
|
+
_RE_FTS5_SPECIAL = re.compile(r'[^\w\u4e00-\u9fff\u3400-\u4dbf]')
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _tokenize_fts_expression(expr: str) -> str:
|
|
178
|
+
"""Tokenize terms in an FTS5 expression while preserving operators.
|
|
179
|
+
|
|
180
|
+
Raw FTS5 expressions like ``"machine AND learning"`` must have their
|
|
181
|
+
terms tokenized to match the indexed content, but FTS5
|
|
182
|
+
operators (AND, OR, NOT, NEAR) must remain untouched.
|
|
183
|
+
|
|
184
|
+
Handles:
|
|
185
|
+
- FTS5 keyword operators: AND, OR, NOT, NEAR
|
|
186
|
+
- Unary NOT prefix: ``-term`` → ``-tokenized``
|
|
187
|
+
- Quoted phrases: ``"exact phrase"`` → ``"tokenized phrase"``
|
|
188
|
+
|
|
189
|
+
Only applies jieba segmentation when CJK characters are detected.
|
|
190
|
+
"""
|
|
191
|
+
# Split while preserving quoted segments as single units
|
|
192
|
+
tokens: list[str] = []
|
|
193
|
+
i = 0
|
|
194
|
+
while i < len(expr):
|
|
195
|
+
if expr[i] == '"':
|
|
196
|
+
end = expr.find('"', i + 1)
|
|
197
|
+
if end == -1:
|
|
198
|
+
end = len(expr)
|
|
199
|
+
tokens.append(expr[i:end + 1])
|
|
200
|
+
i = end + 1
|
|
201
|
+
elif expr[i].isspace():
|
|
202
|
+
i += 1
|
|
203
|
+
else:
|
|
204
|
+
end = i
|
|
205
|
+
while end < len(expr) and not expr[end].isspace() and expr[end] != '"':
|
|
206
|
+
end += 1
|
|
207
|
+
tokens.append(expr[i:end])
|
|
208
|
+
i = end
|
|
209
|
+
|
|
210
|
+
result = []
|
|
211
|
+
for part in tokens:
|
|
212
|
+
upper = part.upper()
|
|
213
|
+
if upper in _FTS5_OPERATORS:
|
|
214
|
+
result.append(upper)
|
|
215
|
+
elif part.startswith('"') and part.endswith('"'):
|
|
216
|
+
# Phrase query: "exact phrase" → tokenize inner content, keep quotes
|
|
217
|
+
inner = part[1:-1]
|
|
218
|
+
tokenized = _tokenize_for_fts(inner)
|
|
219
|
+
if tokenized.strip():
|
|
220
|
+
result.append(f'"{tokenized.strip()}"')
|
|
221
|
+
elif part.startswith('-'):
|
|
222
|
+
# Unary NOT: -term → -tokenized
|
|
223
|
+
tokenized = _tokenize_for_fts(part[1:])
|
|
224
|
+
if tokenized.strip():
|
|
225
|
+
result.append(f"-{tokenized.strip()}")
|
|
226
|
+
else:
|
|
227
|
+
tokenized = _tokenize_for_fts(part)
|
|
228
|
+
if tokenized.strip():
|
|
229
|
+
result.append(tokenized.strip())
|
|
230
|
+
return " ".join(result)
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
# ---------------------------------------------------------------------------
|
|
234
|
+
# FTS5 Index Engine
|
|
235
|
+
# ---------------------------------------------------------------------------
|
|
236
|
+
|
|
237
|
+
class FTS5Index:
|
|
238
|
+
"""SQLite FTS5 full-text search index for tree-structured documents.
|
|
239
|
+
|
|
240
|
+
Features:
|
|
241
|
+
- WAL journal mode for concurrent read/write
|
|
242
|
+
- MD-aware columns: title, summary, body, code_blocks, front_matter
|
|
243
|
+
- Hierarchical column weighting via bm25() rank function
|
|
244
|
+
- Deferred indexing: call index_document() when ready
|
|
245
|
+
- Implements PreFilter protocol: score_nodes(query, doc_id)
|
|
246
|
+
- Supports FTS5 query syntax: AND, OR, NOT, NEAR, phrase "..."
|
|
247
|
+
|
|
248
|
+
In-memory mode (``db_path=None``):
|
|
249
|
+
All indexes are kept in SQLite ``:memory:`` — no file is written to disk.
|
|
250
|
+
Performance is excellent even with thousands of documents (5,000 docs < 10ms).
|
|
251
|
+
Indexes are lost when the process exits or the instance is closed.
|
|
252
|
+
"""
|
|
253
|
+
|
|
254
|
+
def __init__(
|
|
255
|
+
self,
|
|
256
|
+
db_path: Optional[str] = None,
|
|
257
|
+
weights: Optional[dict] = None,
|
|
258
|
+
tokenize_log_path: Optional[str] = None,
|
|
259
|
+
):
|
|
260
|
+
"""
|
|
261
|
+
Args:
|
|
262
|
+
db_path: Path to SQLite database file. ``None`` for in-memory mode
|
|
263
|
+
(no file written to disk). Default: ``None``.
|
|
264
|
+
weights: column weight overrides for bm25() ranking.
|
|
265
|
+
tokenize_log_path: Optional path to write per-document tokenize logs.
|
|
266
|
+
When set, each call to ``index_document()`` appends deduplicated
|
|
267
|
+
token lists to this file.
|
|
268
|
+
"""
|
|
269
|
+
self._db_path = db_path or ":memory:"
|
|
270
|
+
self._weights = {**_DEFAULT_WEIGHTS, **(weights or {})}
|
|
271
|
+
self._conn: Optional[sqlite3.Connection] = None
|
|
272
|
+
# Populated by index_document(); inspected by build_index for stats.
|
|
273
|
+
self.last_node_diff: dict[str, int] = {"added": 0, "changed": 0, "removed": 0, "kept": 0}
|
|
274
|
+
self._tokenize_log_path = tokenize_log_path
|
|
275
|
+
self._tokenize_log_file = None # lazy-opened in _log_tokenize()
|
|
276
|
+
try:
|
|
277
|
+
self._init_db()
|
|
278
|
+
except Exception:
|
|
279
|
+
self.close()
|
|
280
|
+
raise
|
|
281
|
+
|
|
282
|
+
def _init_db(self) -> None:
|
|
283
|
+
"""Initialize SQLite database with FTS5 virtual table (or fallback plain table)."""
|
|
284
|
+
if self._db_path != ":memory:":
|
|
285
|
+
os.makedirs(os.path.dirname(os.path.abspath(self._db_path)), exist_ok=True)
|
|
286
|
+
|
|
287
|
+
self._conn = sqlite3.connect(self._db_path, check_same_thread=False)
|
|
288
|
+
self._conn.execute("PRAGMA journal_mode=WAL")
|
|
289
|
+
self._conn.execute("PRAGMA synchronous=NORMAL")
|
|
290
|
+
|
|
291
|
+
# Register REGEXP function for like_search(use_regex=True)
|
|
292
|
+
self._conn.create_function("REGEXP", 2, _sqlite_regexp)
|
|
293
|
+
|
|
294
|
+
self._use_fts5 = _check_fts5()
|
|
295
|
+
|
|
296
|
+
# Metadata table for nodes (structured fields for filtering)
|
|
297
|
+
self._conn.execute("""
|
|
298
|
+
CREATE TABLE IF NOT EXISTS nodes (
|
|
299
|
+
node_id TEXT NOT NULL,
|
|
300
|
+
doc_id TEXT NOT NULL,
|
|
301
|
+
title TEXT DEFAULT '',
|
|
302
|
+
summary TEXT DEFAULT '',
|
|
303
|
+
depth INTEGER DEFAULT 0,
|
|
304
|
+
line_start INTEGER,
|
|
305
|
+
line_end INTEGER,
|
|
306
|
+
parent_node_id TEXT,
|
|
307
|
+
content_hash TEXT,
|
|
308
|
+
PRIMARY KEY (doc_id, node_id)
|
|
309
|
+
)
|
|
310
|
+
""")
|
|
311
|
+
|
|
312
|
+
if self._use_fts5:
|
|
313
|
+
# FTS5 virtual table with content sync
|
|
314
|
+
# tokenize='unicode61' handles basic multi-language, but we pre-tokenize
|
|
315
|
+
# Chinese text with jieba and store space-separated tokens
|
|
316
|
+
self._conn.execute("""
|
|
317
|
+
CREATE VIRTUAL TABLE IF NOT EXISTS fts_nodes USING fts5(
|
|
318
|
+
node_id UNINDEXED,
|
|
319
|
+
doc_id UNINDEXED,
|
|
320
|
+
title,
|
|
321
|
+
summary,
|
|
322
|
+
body,
|
|
323
|
+
code_blocks,
|
|
324
|
+
front_matter,
|
|
325
|
+
tokenize='unicode61 remove_diacritics 2'
|
|
326
|
+
)
|
|
327
|
+
""")
|
|
328
|
+
else:
|
|
329
|
+
# Fallback: plain table with same columns for LIKE-based search
|
|
330
|
+
self._conn.execute("""
|
|
331
|
+
CREATE TABLE IF NOT EXISTS fts_nodes (
|
|
332
|
+
rowid INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
333
|
+
node_id TEXT NOT NULL,
|
|
334
|
+
doc_id TEXT NOT NULL,
|
|
335
|
+
title TEXT DEFAULT '',
|
|
336
|
+
summary TEXT DEFAULT '',
|
|
337
|
+
body TEXT DEFAULT '',
|
|
338
|
+
code_blocks TEXT DEFAULT '',
|
|
339
|
+
front_matter TEXT DEFAULT ''
|
|
340
|
+
)
|
|
341
|
+
""")
|
|
342
|
+
self._conn.execute(
|
|
343
|
+
"CREATE INDEX IF NOT EXISTS idx_fts_nodes_doc_id ON fts_nodes (doc_id)"
|
|
344
|
+
)
|
|
345
|
+
|
|
346
|
+
# Document metadata table (also stores tree structure for persistence)
|
|
347
|
+
self._conn.execute("""
|
|
348
|
+
CREATE TABLE IF NOT EXISTS documents (
|
|
349
|
+
doc_id TEXT PRIMARY KEY,
|
|
350
|
+
doc_name TEXT DEFAULT '',
|
|
351
|
+
doc_description TEXT DEFAULT '',
|
|
352
|
+
source_path TEXT DEFAULT '',
|
|
353
|
+
source_type TEXT DEFAULT '',
|
|
354
|
+
structure_json TEXT DEFAULT '',
|
|
355
|
+
node_count INTEGER DEFAULT 0,
|
|
356
|
+
index_hash TEXT
|
|
357
|
+
)
|
|
358
|
+
""")
|
|
359
|
+
|
|
360
|
+
# Incremental index metadata (replaces _index_meta.json)
|
|
361
|
+
self._conn.execute("""
|
|
362
|
+
CREATE TABLE IF NOT EXISTS index_meta (
|
|
363
|
+
source_path TEXT PRIMARY KEY,
|
|
364
|
+
file_hash TEXT NOT NULL
|
|
365
|
+
)
|
|
366
|
+
""")
|
|
367
|
+
|
|
368
|
+
# Failed files tracking (auto-skip after consecutive parse failures)
|
|
369
|
+
self._conn.execute("""
|
|
370
|
+
CREATE TABLE IF NOT EXISTS failed_files (
|
|
371
|
+
source_path TEXT PRIMARY KEY,
|
|
372
|
+
fail_count INTEGER NOT NULL DEFAULT 1,
|
|
373
|
+
last_error TEXT DEFAULT '',
|
|
374
|
+
last_fail_time REAL NOT NULL,
|
|
375
|
+
file_hash TEXT DEFAULT ''
|
|
376
|
+
)
|
|
377
|
+
""")
|
|
378
|
+
|
|
379
|
+
# Vision parse queue (image files: placeholder first, background vision
|
|
380
|
+
# worker consumes serially and replaces the placeholder in-place).
|
|
381
|
+
# status: pending | processing | done | failed
|
|
382
|
+
self._conn.execute("""
|
|
383
|
+
CREATE TABLE IF NOT EXISTS vision_queue (
|
|
384
|
+
source_path TEXT PRIMARY KEY,
|
|
385
|
+
rel_path TEXT DEFAULT '',
|
|
386
|
+
status TEXT NOT NULL DEFAULT 'pending',
|
|
387
|
+
attempts INTEGER NOT NULL DEFAULT 0,
|
|
388
|
+
model TEXT DEFAULT '',
|
|
389
|
+
last_error TEXT DEFAULT '',
|
|
390
|
+
updated_at REAL NOT NULL
|
|
391
|
+
)
|
|
392
|
+
""")
|
|
393
|
+
|
|
394
|
+
# PST 邮件元数据(ADR-0005):每封派生邮件文档一行,供物理 PST 的
|
|
395
|
+
# 邮件列表分页查询(主题/发件人/日期/文件夹 + 附件下载清单)。
|
|
396
|
+
self._conn.execute("""
|
|
397
|
+
CREATE TABLE IF NOT EXISTS pst_email_meta (
|
|
398
|
+
doc_id TEXT PRIMARY KEY,
|
|
399
|
+
pst_path TEXT NOT NULL,
|
|
400
|
+
entry_id TEXT NOT NULL,
|
|
401
|
+
subject TEXT DEFAULT '',
|
|
402
|
+
sender TEXT DEFAULT '',
|
|
403
|
+
date TEXT DEFAULT '',
|
|
404
|
+
folder TEXT DEFAULT '',
|
|
405
|
+
attachments_json TEXT DEFAULT '[]'
|
|
406
|
+
)
|
|
407
|
+
""")
|
|
408
|
+
self._conn.execute(
|
|
409
|
+
"CREATE INDEX IF NOT EXISTS idx_pst_email_meta_pst ON pst_email_meta (pst_path)"
|
|
410
|
+
)
|
|
411
|
+
|
|
412
|
+
# Performance indexes for large-scale document sets (10k+ docs)
|
|
413
|
+
self._conn.execute(
|
|
414
|
+
"CREATE INDEX IF NOT EXISTS idx_nodes_doc_id ON nodes (doc_id)"
|
|
415
|
+
)
|
|
416
|
+
self._conn.execute(
|
|
417
|
+
"CREATE INDEX IF NOT EXISTS idx_documents_source_path ON documents (source_path)"
|
|
418
|
+
)
|
|
419
|
+
|
|
420
|
+
self._conn.commit()
|
|
421
|
+
|
|
422
|
+
@property
|
|
423
|
+
def db_path(self) -> str:
|
|
424
|
+
return self._db_path
|
|
425
|
+
|
|
426
|
+
def close(self) -> None:
|
|
427
|
+
"""Close database connection and tokenize log file."""
|
|
428
|
+
if self._tokenize_log_file:
|
|
429
|
+
self._tokenize_log_file.close()
|
|
430
|
+
self._tokenize_log_file = None
|
|
431
|
+
if self._conn:
|
|
432
|
+
self._conn.close()
|
|
433
|
+
self._conn = None
|
|
434
|
+
|
|
435
|
+
def _log_tokenize(self, doc_id: str, tokens: set[str]) -> None:
|
|
436
|
+
"""Append deduplicated token list for *doc_id* to the tokenize log."""
|
|
437
|
+
if not self._tokenize_log_path:
|
|
438
|
+
return
|
|
439
|
+
if self._tokenize_log_file is None:
|
|
440
|
+
self._tokenize_log_file = open(
|
|
441
|
+
self._tokenize_log_path, 'a', encoding='utf-8',
|
|
442
|
+
)
|
|
443
|
+
sorted_tokens = sorted(tokens)
|
|
444
|
+
self._tokenize_log_file.write(
|
|
445
|
+
f"--- {doc_id} ({len(sorted_tokens)} tokens) ---\n"
|
|
446
|
+
)
|
|
447
|
+
self._tokenize_log_file.write(" ".join(sorted_tokens) + "\n\n")
|
|
448
|
+
self._tokenize_log_file.flush()
|
|
449
|
+
|
|
450
|
+
def __del__(self):
|
|
451
|
+
try:
|
|
452
|
+
self.close()
|
|
453
|
+
except Exception:
|
|
454
|
+
pass
|
|
455
|
+
|
|
456
|
+
# -------------------------------------------------------------------
|
|
457
|
+
# Failed files tracking
|
|
458
|
+
# -------------------------------------------------------------------
|
|
459
|
+
|
|
460
|
+
def get_all_failed_files(self) -> dict[str, tuple[int, str, str]]:
|
|
461
|
+
"""Batch load all failed file records.
|
|
462
|
+
|
|
463
|
+
Returns:
|
|
464
|
+
``{source_path: (fail_count, file_hash, last_error)}``
|
|
465
|
+
"""
|
|
466
|
+
rows = self._conn.execute(
|
|
467
|
+
"SELECT source_path, fail_count, file_hash, last_error FROM failed_files"
|
|
468
|
+
).fetchall()
|
|
469
|
+
return {row[0]: (row[1], row[2], row[3] or "") for row in rows}
|
|
470
|
+
|
|
471
|
+
def upsert_failed_file(self, path: str, error_msg: str, file_hash: str = "") -> None:
|
|
472
|
+
"""Insert or increment fail count for a file that failed to parse."""
|
|
473
|
+
import time as _time
|
|
474
|
+
now = _time.time()
|
|
475
|
+
self._conn.execute(
|
|
476
|
+
"""INSERT INTO failed_files (source_path, fail_count, last_error, last_fail_time, file_hash)
|
|
477
|
+
VALUES (?, 1, ?, ?, ?)
|
|
478
|
+
ON CONFLICT(source_path) DO UPDATE SET
|
|
479
|
+
fail_count = fail_count + 1,
|
|
480
|
+
last_error = excluded.last_error,
|
|
481
|
+
last_fail_time = excluded.last_fail_time,
|
|
482
|
+
file_hash = excluded.file_hash
|
|
483
|
+
""",
|
|
484
|
+
(path, error_msg, now, file_hash),
|
|
485
|
+
)
|
|
486
|
+
|
|
487
|
+
def clear_failed_file(self, path: str) -> None:
|
|
488
|
+
"""Remove a file's failed record after it is successfully indexed."""
|
|
489
|
+
self._conn.execute(
|
|
490
|
+
"DELETE FROM failed_files WHERE source_path = ?", (path,)
|
|
491
|
+
)
|
|
492
|
+
|
|
493
|
+
def clear_all_failed_files(self) -> None:
|
|
494
|
+
"""Remove all failed file records (used during full rebuild)."""
|
|
495
|
+
with self._conn:
|
|
496
|
+
self._conn.execute("DELETE FROM failed_files")
|
|
497
|
+
|
|
498
|
+
def get_failed_files_summary(self) -> list[tuple[str, int, str]]:
|
|
499
|
+
"""Return failed files list for display.
|
|
500
|
+
|
|
501
|
+
Returns:
|
|
502
|
+
list of ``(source_path, fail_count, last_error)``
|
|
503
|
+
"""
|
|
504
|
+
rows = self._conn.execute(
|
|
505
|
+
"SELECT source_path, fail_count, last_error FROM failed_files ORDER BY fail_count DESC"
|
|
506
|
+
).fetchall()
|
|
507
|
+
return [(row[0], row[1], row[2]) for row in rows]
|
|
508
|
+
|
|
509
|
+
# -------------------------------------------------------------------
|
|
510
|
+
# Vision parse queue (image files; consumed by host-side vision worker)
|
|
511
|
+
# -------------------------------------------------------------------
|
|
512
|
+
|
|
513
|
+
def vision_enqueue(self, source_path: str, rel_path: str = "") -> None:
|
|
514
|
+
"""登记一个待视觉解析的图像文件(重复登记=文件变更,重置为 pending)。"""
|
|
515
|
+
import time as _time
|
|
516
|
+
self._conn.execute(
|
|
517
|
+
"""INSERT INTO vision_queue (source_path, rel_path, status, attempts, model, last_error, updated_at)
|
|
518
|
+
VALUES (?, ?, 'pending', 0, '', '', ?)
|
|
519
|
+
ON CONFLICT(source_path) DO UPDATE SET
|
|
520
|
+
rel_path = excluded.rel_path,
|
|
521
|
+
status = 'pending',
|
|
522
|
+
attempts = 0,
|
|
523
|
+
last_error = '',
|
|
524
|
+
updated_at = excluded.updated_at
|
|
525
|
+
""",
|
|
526
|
+
(source_path, rel_path, _time.time()),
|
|
527
|
+
)
|
|
528
|
+
|
|
529
|
+
def vision_next_pending(self) -> Optional[dict]:
|
|
530
|
+
"""取出下一个待解析项并置为 processing(原子抢占)。无则返回 None。"""
|
|
531
|
+
import time as _time
|
|
532
|
+
row = self._conn.execute(
|
|
533
|
+
"SELECT source_path, rel_path, attempts FROM vision_queue "
|
|
534
|
+
"WHERE status = 'pending' ORDER BY updated_at LIMIT 1"
|
|
535
|
+
).fetchone()
|
|
536
|
+
if row is None:
|
|
537
|
+
return None
|
|
538
|
+
cur = self._conn.execute(
|
|
539
|
+
"UPDATE vision_queue SET status = 'processing', updated_at = ? "
|
|
540
|
+
"WHERE source_path = ? AND status = 'pending'",
|
|
541
|
+
(_time.time(), row[0]),
|
|
542
|
+
)
|
|
543
|
+
self._conn.commit()
|
|
544
|
+
if cur.rowcount == 0:
|
|
545
|
+
return None # 被其他消费者抢走
|
|
546
|
+
return {"source_path": row[0], "rel_path": row[1], "attempts": row[2]}
|
|
547
|
+
|
|
548
|
+
def vision_mark_done(self, source_path: str, model: str) -> bool:
|
|
549
|
+
"""标记解析完成。仅当仍处于 processing 时生效(防止与 force 重建竞态)。"""
|
|
550
|
+
import time as _time
|
|
551
|
+
cur = self._conn.execute(
|
|
552
|
+
"UPDATE vision_queue SET status = 'done', model = ?, last_error = '', updated_at = ? "
|
|
553
|
+
"WHERE source_path = ? AND status = 'processing'",
|
|
554
|
+
(model, _time.time(), source_path),
|
|
555
|
+
)
|
|
556
|
+
self._conn.commit()
|
|
557
|
+
return cur.rowcount > 0
|
|
558
|
+
|
|
559
|
+
def vision_mark_failed(self, source_path: str, error: str, *, final: bool) -> None:
|
|
560
|
+
"""记录一次失败;final=True(达到连败上限)时置 failed,否则回 pending 待重试。"""
|
|
561
|
+
import time as _time
|
|
562
|
+
self._conn.execute(
|
|
563
|
+
"UPDATE vision_queue SET status = ?, attempts = attempts + 1, last_error = ?, updated_at = ? "
|
|
564
|
+
"WHERE source_path = ?",
|
|
565
|
+
("failed" if final else "pending", error[:500], _time.time(), source_path),
|
|
566
|
+
)
|
|
567
|
+
self._conn.commit()
|
|
568
|
+
|
|
569
|
+
def vision_remove(self, source_path: str) -> None:
|
|
570
|
+
"""从队列移除(源文件被 prune / 删除时调用)。"""
|
|
571
|
+
self._conn.execute(
|
|
572
|
+
"DELETE FROM vision_queue WHERE source_path = ?", (source_path,)
|
|
573
|
+
)
|
|
574
|
+
|
|
575
|
+
def vision_clear(self) -> None:
|
|
576
|
+
"""清空队列(force 全量重建时调用;重建过程会重新登记)。"""
|
|
577
|
+
with self._conn:
|
|
578
|
+
self._conn.execute("DELETE FROM vision_queue")
|
|
579
|
+
|
|
580
|
+
def vision_reset_stale_processing(self) -> int:
|
|
581
|
+
"""启动时把残留的 processing(上次进程崩溃)重置回 pending。"""
|
|
582
|
+
with self._conn:
|
|
583
|
+
cur = self._conn.execute(
|
|
584
|
+
"UPDATE vision_queue SET status = 'pending' WHERE status = 'processing'"
|
|
585
|
+
)
|
|
586
|
+
return cur.rowcount
|
|
587
|
+
|
|
588
|
+
def vision_requeue_model_changed(self, current_tag: str) -> int:
|
|
589
|
+
"""模型/prompt/协议版本变化时,把已完成或失败的项重新置为 pending 以便重解析。
|
|
590
|
+
|
|
591
|
+
失败项也纳入:改协议(如 OpenAI-compat → anthropic)后,旧协议下失败的
|
|
592
|
+
图像应在启动时自动重试,而不是永远卡在 failed。
|
|
593
|
+
"""
|
|
594
|
+
import time as _time
|
|
595
|
+
with self._conn:
|
|
596
|
+
cur = self._conn.execute(
|
|
597
|
+
"UPDATE vision_queue SET status = 'pending', attempts = 0, last_error = '', updated_at = ? "
|
|
598
|
+
"WHERE status IN ('done', 'failed') AND model != ? AND model != 'manual'",
|
|
599
|
+
(_time.time(), current_tag),
|
|
600
|
+
)
|
|
601
|
+
return cur.rowcount
|
|
602
|
+
|
|
603
|
+
def vision_counts(self) -> dict[str, int]:
|
|
604
|
+
"""各状态计数,供状态展示。如 ``{"pending": 3, "done": 12}``。"""
|
|
605
|
+
rows = self._conn.execute(
|
|
606
|
+
"SELECT status, COUNT(*) FROM vision_queue GROUP BY status"
|
|
607
|
+
).fetchall()
|
|
608
|
+
return {row[0]: row[1] for row in rows}
|
|
609
|
+
|
|
610
|
+
# -------------------------------------------------------------------
|
|
611
|
+
# Indexing (Producer side)
|
|
612
|
+
# -------------------------------------------------------------------
|
|
613
|
+
|
|
614
|
+
def index_document(self, document, force: bool = False, auto_commit: bool = True,
|
|
615
|
+
file_hash: Optional[str] = None) -> int:
|
|
616
|
+
"""Index all nodes from a Document into FTS5.
|
|
617
|
+
|
|
618
|
+
Performs **node-level incremental diff**: only nodes whose content
|
|
619
|
+
actually changed (or appeared/disappeared) are removed/inserted into
|
|
620
|
+
the FTS5 index. The full document tree is still re-serialized into the
|
|
621
|
+
``documents`` table so search continues to read the latest structure.
|
|
622
|
+
|
|
623
|
+
Stable ``node_id``s (assigned by :func:`treesearch.tree.assign_node_ids`)
|
|
624
|
+
are required for correct diffing — same logical position must yield the
|
|
625
|
+
same id across runs.
|
|
626
|
+
|
|
627
|
+
Side-effects on ``last_node_diff`` so callers (e.g. build_index) can
|
|
628
|
+
report diff stats.
|
|
629
|
+
|
|
630
|
+
Args:
|
|
631
|
+
document: Document object with structure tree.
|
|
632
|
+
force: re-index every node even if hashes match.
|
|
633
|
+
auto_commit: if False, skip commit (caller is responsible).
|
|
634
|
+
file_hash: optional file fingerprint to write to ``index_meta``
|
|
635
|
+
in the same transaction (avoids the "FTS written but
|
|
636
|
+
fingerprint missing → next run rebuilds" failure mode).
|
|
637
|
+
|
|
638
|
+
Returns:
|
|
639
|
+
number of nodes (re-)indexed in this call.
|
|
640
|
+
"""
|
|
641
|
+
# Compute content hash for incremental check
|
|
642
|
+
content_str = json.dumps(document.structure, ensure_ascii=False, sort_keys=True)
|
|
643
|
+
content_hash = hashlib.md5(content_str.encode()).hexdigest()
|
|
644
|
+
|
|
645
|
+
# Check if already indexed (whole-document fast-path)
|
|
646
|
+
if not force:
|
|
647
|
+
row = self._conn.execute(
|
|
648
|
+
"SELECT index_hash FROM documents WHERE doc_id = ?",
|
|
649
|
+
(document.doc_id,),
|
|
650
|
+
).fetchone()
|
|
651
|
+
if row and row[0] == content_hash:
|
|
652
|
+
# logger.debug("Document %s already indexed (hash match), skipping", document.doc_id)
|
|
653
|
+
self.last_node_diff = {"added": 0, "changed": 0, "removed": 0, "kept": 0}
|
|
654
|
+
# Still write index_meta if requested — we may have been called
|
|
655
|
+
# because the file fingerprint changed but the structure didn't.
|
|
656
|
+
if file_hash and document.metadata.get("source_path"):
|
|
657
|
+
self._conn.execute(
|
|
658
|
+
"INSERT OR REPLACE INTO index_meta (source_path, file_hash) VALUES (?, ?)",
|
|
659
|
+
(document.metadata["source_path"], file_hash),
|
|
660
|
+
)
|
|
661
|
+
if auto_commit:
|
|
662
|
+
self._conn.commit()
|
|
663
|
+
return 0
|
|
664
|
+
|
|
665
|
+
# ---- Compute node-level diff ----
|
|
666
|
+
from .tree import flatten_tree, build_tree_maps
|
|
667
|
+
_, parent_map, depth_map = build_tree_maps(document.structure)
|
|
668
|
+
|
|
669
|
+
all_nodes = [n for n in flatten_tree(document.structure) if n.get("node_id")]
|
|
670
|
+
|
|
671
|
+
new_hashes: dict[str, str] = {}
|
|
672
|
+
for node in all_nodes:
|
|
673
|
+
text = node.get("text", "")
|
|
674
|
+
new_hashes[node["node_id"]] = hashlib.md5(text.encode()).hexdigest()[:16]
|
|
675
|
+
|
|
676
|
+
if force:
|
|
677
|
+
old_hashes: dict[str, str] = {}
|
|
678
|
+
else:
|
|
679
|
+
old_hashes = {
|
|
680
|
+
nid: chash for (nid, chash) in self._conn.execute(
|
|
681
|
+
"SELECT node_id, content_hash FROM nodes WHERE doc_id = ?",
|
|
682
|
+
(document.doc_id,),
|
|
683
|
+
).fetchall()
|
|
684
|
+
}
|
|
685
|
+
|
|
686
|
+
new_ids = set(new_hashes.keys())
|
|
687
|
+
old_ids = set(old_hashes.keys())
|
|
688
|
+
added = new_ids - old_ids
|
|
689
|
+
removed = old_ids - new_ids
|
|
690
|
+
changed = {nid for nid in (new_ids & old_ids) if new_hashes[nid] != old_hashes[nid]}
|
|
691
|
+
kept = (new_ids & old_ids) - changed
|
|
692
|
+
|
|
693
|
+
to_write = added | changed
|
|
694
|
+
diff_stats = {
|
|
695
|
+
"added": len(added),
|
|
696
|
+
"changed": len(changed),
|
|
697
|
+
"removed": len(removed),
|
|
698
|
+
"kept": len(kept),
|
|
699
|
+
}
|
|
700
|
+
self.last_node_diff = diff_stats
|
|
701
|
+
|
|
702
|
+
# ---- Stage all rows ahead of the transaction ----
|
|
703
|
+
node_rows: list[tuple] = []
|
|
704
|
+
fts_rows: list[tuple] = []
|
|
705
|
+
doc_tokens: set[str] = set()
|
|
706
|
+
for node in all_nodes:
|
|
707
|
+
nid = node["node_id"]
|
|
708
|
+
if nid not in to_write:
|
|
709
|
+
continue
|
|
710
|
+
title = node.get("title", "")
|
|
711
|
+
summary = node.get("summary", node.get("prefix_summary", ""))
|
|
712
|
+
text = node.get("text", "")
|
|
713
|
+
depth = depth_map.get(nid, 0)
|
|
714
|
+
node_rows.append((
|
|
715
|
+
nid, document.doc_id, title, summary, depth,
|
|
716
|
+
node.get("line_start"), node.get("line_end"),
|
|
717
|
+
parent_map.get(nid), new_hashes[nid],
|
|
718
|
+
))
|
|
719
|
+
|
|
720
|
+
parsed = parse_md_node_text(text)
|
|
721
|
+
tok_title = _tokenize_for_fts(title)
|
|
722
|
+
tok_summary = _tokenize_for_fts(summary)
|
|
723
|
+
tok_body = _tokenize_for_fts(parsed["body"])
|
|
724
|
+
tok_code = _tokenize_for_fts(parsed["code_blocks"])
|
|
725
|
+
tok_fm = _tokenize_for_fts(parsed["front_matter"])
|
|
726
|
+
|
|
727
|
+
# 将文件名和路径分词后注入 front_matter,使文件名/路径可被 FTS5 搜索到
|
|
728
|
+
tok_doc_name = _tokenize_for_fts(document.doc_name)
|
|
729
|
+
if tok_doc_name and tok_doc_name not in tok_fm:
|
|
730
|
+
tok_fm = tok_doc_name + " " + tok_fm
|
|
731
|
+
source_path = document.metadata.get("source_path", "")
|
|
732
|
+
tok_path = _tokenize_for_fts(source_path)
|
|
733
|
+
if tok_path and tok_path not in tok_fm:
|
|
734
|
+
tok_fm = tok_path + " " + tok_fm
|
|
735
|
+
fts_rows.append((
|
|
736
|
+
nid, document.doc_id,
|
|
737
|
+
tok_title, tok_summary, tok_body, tok_code, tok_fm,
|
|
738
|
+
))
|
|
739
|
+
for tok_str in (tok_title, tok_summary, tok_body, tok_code, tok_fm):
|
|
740
|
+
doc_tokens.update(tok_str.split())
|
|
741
|
+
|
|
742
|
+
self._log_tokenize(document.doc_id, doc_tokens)
|
|
743
|
+
|
|
744
|
+
# ---- Single atomic transaction ----
|
|
745
|
+
# Wraps deletes + inserts + document metadata + index_meta so a crash
|
|
746
|
+
# in the middle either rolls back fully or commits everything.
|
|
747
|
+
structure_json = json.dumps(document.structure, ensure_ascii=False)
|
|
748
|
+
|
|
749
|
+
if auto_commit:
|
|
750
|
+
txn_ctx = self._conn # 'with conn' = implicit transaction
|
|
751
|
+
else:
|
|
752
|
+
txn_ctx = _NullContext()
|
|
753
|
+
|
|
754
|
+
with txn_ctx:
|
|
755
|
+
# Targeted deletes for removed + changed (NOT a wholesale wipe).
|
|
756
|
+
del_ids = removed | changed | (added if force else set())
|
|
757
|
+
if del_ids:
|
|
758
|
+
if self._use_fts5:
|
|
759
|
+
placeholders = ",".join("?" for _ in del_ids)
|
|
760
|
+
old_rowids = self._conn.execute(
|
|
761
|
+
f"SELECT rowid FROM fts_nodes WHERE doc_id = ? AND node_id IN ({placeholders})",
|
|
762
|
+
(document.doc_id, *del_ids),
|
|
763
|
+
).fetchall()
|
|
764
|
+
if old_rowids:
|
|
765
|
+
ph2 = ",".join("?" for _ in old_rowids)
|
|
766
|
+
self._conn.execute(
|
|
767
|
+
f"DELETE FROM fts_nodes WHERE rowid IN ({ph2})",
|
|
768
|
+
[r[0] for r in old_rowids],
|
|
769
|
+
)
|
|
770
|
+
else:
|
|
771
|
+
placeholders = ",".join("?" for _ in del_ids)
|
|
772
|
+
self._conn.execute(
|
|
773
|
+
f"DELETE FROM fts_nodes WHERE doc_id = ? AND node_id IN ({placeholders})",
|
|
774
|
+
(document.doc_id, *del_ids),
|
|
775
|
+
)
|
|
776
|
+
placeholders = ",".join("?" for _ in del_ids)
|
|
777
|
+
self._conn.execute(
|
|
778
|
+
f"DELETE FROM nodes WHERE doc_id = ? AND node_id IN ({placeholders})",
|
|
779
|
+
(document.doc_id, *del_ids),
|
|
780
|
+
)
|
|
781
|
+
|
|
782
|
+
if node_rows:
|
|
783
|
+
self._conn.executemany(
|
|
784
|
+
"""INSERT INTO nodes
|
|
785
|
+
(node_id, doc_id, title, summary, depth, line_start, line_end, parent_node_id, content_hash)
|
|
786
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)""",
|
|
787
|
+
node_rows,
|
|
788
|
+
)
|
|
789
|
+
if fts_rows:
|
|
790
|
+
self._conn.executemany(
|
|
791
|
+
"""INSERT INTO fts_nodes
|
|
792
|
+
(node_id, doc_id, title, summary, body, code_blocks, front_matter)
|
|
793
|
+
VALUES (?, ?, ?, ?, ?, ?, ?)""",
|
|
794
|
+
fts_rows,
|
|
795
|
+
)
|
|
796
|
+
|
|
797
|
+
self._conn.execute(
|
|
798
|
+
"""INSERT OR REPLACE INTO documents
|
|
799
|
+
(doc_id, doc_name, doc_description, source_path, source_type,
|
|
800
|
+
structure_json, node_count, index_hash)
|
|
801
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, ?)""",
|
|
802
|
+
(
|
|
803
|
+
document.doc_id, document.doc_name, document.doc_description,
|
|
804
|
+
document.metadata.get("source_path", ""),
|
|
805
|
+
document.source_type,
|
|
806
|
+
structure_json,
|
|
807
|
+
len(all_nodes), content_hash,
|
|
808
|
+
),
|
|
809
|
+
)
|
|
810
|
+
|
|
811
|
+
if file_hash and document.metadata.get("source_path"):
|
|
812
|
+
self._conn.execute(
|
|
813
|
+
"INSERT OR REPLACE INTO index_meta (source_path, file_hash) VALUES (?, ?)",
|
|
814
|
+
(document.metadata["source_path"], file_hash),
|
|
815
|
+
)
|
|
816
|
+
|
|
817
|
+
logger.debug(
|
|
818
|
+
"FTS5 reindexed %s: +%d ~%d -%d (kept %d)",
|
|
819
|
+
document.doc_id, diff_stats["added"], diff_stats["changed"],
|
|
820
|
+
diff_stats["removed"], diff_stats["kept"],
|
|
821
|
+
)
|
|
822
|
+
return len(node_rows)
|
|
823
|
+
|
|
824
|
+
def commit(self) -> None:
|
|
825
|
+
"""Manually commit pending changes to the database."""
|
|
826
|
+
self._conn.commit()
|
|
827
|
+
|
|
828
|
+
def index_documents(self, documents: list, force: bool = False) -> int:
|
|
829
|
+
"""Batch index multiple documents.
|
|
830
|
+
|
|
831
|
+
Returns:
|
|
832
|
+
total number of nodes indexed
|
|
833
|
+
"""
|
|
834
|
+
total = 0
|
|
835
|
+
for doc in documents:
|
|
836
|
+
total += self.index_document(doc, force=force)
|
|
837
|
+
return total
|
|
838
|
+
|
|
839
|
+
# -------------------------------------------------------------------
|
|
840
|
+
# Search (Consumer side)
|
|
841
|
+
# -------------------------------------------------------------------
|
|
842
|
+
|
|
843
|
+
def _build_match_expr(self, query: str, fts_expression: Optional[str] = None) -> Optional[str]:
|
|
844
|
+
"""Build FTS5 MATCH expression from query (cached tokenization).
|
|
845
|
+
|
|
846
|
+
Returns None if no valid tokens could be extracted.
|
|
847
|
+
"""
|
|
848
|
+
if fts_expression:
|
|
849
|
+
return _tokenize_fts_expression(fts_expression)
|
|
850
|
+
|
|
851
|
+
tokens = _tokenize_for_fts(query)
|
|
852
|
+
if not tokens.strip():
|
|
853
|
+
return None
|
|
854
|
+
words = tokens.split()
|
|
855
|
+
clean_words = []
|
|
856
|
+
for w in words:
|
|
857
|
+
cleaned = _RE_FTS5_SPECIAL.sub("", w).strip()
|
|
858
|
+
if cleaned and cleaned.upper() not in _FTS5_OPERATORS:
|
|
859
|
+
clean_words.append(cleaned)
|
|
860
|
+
if not clean_words:
|
|
861
|
+
return None
|
|
862
|
+
if len(clean_words) > 1:
|
|
863
|
+
return " OR ".join(clean_words)
|
|
864
|
+
return clean_words[0]
|
|
865
|
+
|
|
866
|
+
def search(
|
|
867
|
+
self,
|
|
868
|
+
query: str,
|
|
869
|
+
doc_id: Optional[str] = None,
|
|
870
|
+
top_k: int = 20,
|
|
871
|
+
fts_expression: Optional[str] = None,
|
|
872
|
+
_precomputed_match_expr: Optional[str] = None,
|
|
873
|
+
) -> list[dict]:
|
|
874
|
+
"""Search nodes using FTS5 BM25 ranking (or LIKE fallback).
|
|
875
|
+
|
|
876
|
+
Args:
|
|
877
|
+
query: natural language query (will be tokenized)
|
|
878
|
+
doc_id: optional filter by document
|
|
879
|
+
top_k: max results
|
|
880
|
+
fts_expression: raw FTS5 query expression (overrides query tokenization).
|
|
881
|
+
Supports AND, OR, NOT, NEAR, phrases.
|
|
882
|
+
_precomputed_match_expr: internal — skip tokenization if already computed.
|
|
883
|
+
|
|
884
|
+
Returns:
|
|
885
|
+
list of {node_id, doc_id, title, summary, fts_score, depth}
|
|
886
|
+
"""
|
|
887
|
+
if not self._use_fts5:
|
|
888
|
+
return self._search_like(query, doc_id=doc_id, top_k=top_k)
|
|
889
|
+
|
|
890
|
+
if _precomputed_match_expr is not None:
|
|
891
|
+
match_expr = _precomputed_match_expr
|
|
892
|
+
else:
|
|
893
|
+
match_expr = self._build_match_expr(query, fts_expression)
|
|
894
|
+
if match_expr is None:
|
|
895
|
+
return []
|
|
896
|
+
|
|
897
|
+
# Phase 1: phrase boosting for multi-word queries
|
|
898
|
+
# Run a separate phrase match query and record which nodes get a boost
|
|
899
|
+
phrase_boost_nids: set[str] = set()
|
|
900
|
+
if not fts_expression and _precomputed_match_expr is None and len(query.split()) >= 2:
|
|
901
|
+
# Build phrase expression from original (unstemmed) query words
|
|
902
|
+
raw_words = [w.lower().strip() for w in re.split(r'\W+', query) if w.strip() and len(w.strip()) > 2]
|
|
903
|
+
if len(raw_words) >= 2:
|
|
904
|
+
# Try phrase match: "word1 word2 ..."
|
|
905
|
+
phrase_expr = '"' + ' '.join(raw_words) + '"'
|
|
906
|
+
try:
|
|
907
|
+
if doc_id:
|
|
908
|
+
phrase_rows = self._conn.execute(
|
|
909
|
+
f"SELECT f.node_id FROM fts_nodes f WHERE fts_nodes MATCH ? AND f.doc_id = ? LIMIT 50",
|
|
910
|
+
(phrase_expr, doc_id),
|
|
911
|
+
).fetchall()
|
|
912
|
+
else:
|
|
913
|
+
phrase_rows = self._conn.execute(
|
|
914
|
+
f"SELECT f.node_id FROM fts_nodes f WHERE fts_nodes MATCH ? LIMIT 50",
|
|
915
|
+
(phrase_expr,),
|
|
916
|
+
).fetchall()
|
|
917
|
+
phrase_boost_nids = {r[0] for r in phrase_rows}
|
|
918
|
+
except sqlite3.OperationalError:
|
|
919
|
+
pass # phrase query syntax error, skip boost
|
|
920
|
+
|
|
921
|
+
# Build SQL with column weights for bm25()
|
|
922
|
+
# bm25(fts_nodes, w1, w2, w3, w4, w5) where weights correspond to:
|
|
923
|
+
# node_id(UNINDEXED), doc_id(UNINDEXED), title, summary, body, code_blocks, front_matter
|
|
924
|
+
w = self._weights
|
|
925
|
+
weight_args = f"{w['title']}, {w['summary']}, {w['body']}, {w['code_blocks']}, {w['front_matter']}"
|
|
926
|
+
|
|
927
|
+
# Query FTS5 directly without JOIN to nodes table, because fts_nodes
|
|
928
|
+
# stores chunk node_ids (e.g. "0_chunk0") that don't match the original
|
|
929
|
+
# node_ids in nodes table (e.g. "0"). Metadata is looked up separately.
|
|
930
|
+
if doc_id:
|
|
931
|
+
sql = f"""
|
|
932
|
+
SELECT f.node_id, f.doc_id, f.title, f.summary,
|
|
933
|
+
bm25(fts_nodes, {weight_args}) AS rank_score
|
|
934
|
+
FROM fts_nodes f
|
|
935
|
+
WHERE fts_nodes MATCH ?
|
|
936
|
+
AND f.doc_id = ?
|
|
937
|
+
ORDER BY rank_score
|
|
938
|
+
LIMIT ?
|
|
939
|
+
"""
|
|
940
|
+
params = (match_expr, doc_id, top_k)
|
|
941
|
+
else:
|
|
942
|
+
sql = f"""
|
|
943
|
+
SELECT f.node_id, f.doc_id, f.title, f.summary,
|
|
944
|
+
bm25(fts_nodes, {weight_args}) AS rank_score
|
|
945
|
+
FROM fts_nodes f
|
|
946
|
+
WHERE fts_nodes MATCH ?
|
|
947
|
+
ORDER BY rank_score
|
|
948
|
+
LIMIT ?
|
|
949
|
+
"""
|
|
950
|
+
params = (match_expr, top_k)
|
|
951
|
+
|
|
952
|
+
try:
|
|
953
|
+
rows = self._conn.execute(sql, params).fetchall()
|
|
954
|
+
except sqlite3.OperationalError as e:
|
|
955
|
+
logger.warning("FTS5 query error: %s, query=%r", e, match_expr)
|
|
956
|
+
rows = []
|
|
957
|
+
|
|
958
|
+
# Pre-fetch node metadata (depth, title, summary) for deduped node_ids
|
|
959
|
+
unique_nids_in_result = {r[0] for r in rows}
|
|
960
|
+
node_meta: dict[tuple[str, str], dict] = {}
|
|
961
|
+
if unique_nids_in_result:
|
|
962
|
+
# Batch lookup from nodes table
|
|
963
|
+
for raw_nid in unique_nids_in_result:
|
|
964
|
+
for r in rows:
|
|
965
|
+
if r[0] == raw_nid:
|
|
966
|
+
did = r[1]
|
|
967
|
+
break
|
|
968
|
+
else:
|
|
969
|
+
continue
|
|
970
|
+
meta_row = self._conn.execute(
|
|
971
|
+
"SELECT title, summary, depth FROM nodes WHERE node_id = ? AND doc_id = ?",
|
|
972
|
+
(raw_nid, did),
|
|
973
|
+
).fetchone()
|
|
974
|
+
if meta_row:
|
|
975
|
+
node_meta[(raw_nid, did)] = {"title": meta_row[0], "summary": meta_row[1], "depth": meta_row[2]}
|
|
976
|
+
|
|
977
|
+
results = []
|
|
978
|
+
seen_nids: dict[str, int] = {} # track dedup by node_id
|
|
979
|
+
for row in rows:
|
|
980
|
+
# bm25() returns negative values (lower = more relevant)
|
|
981
|
+
fts_score = -row[4] if row[4] else 0.0
|
|
982
|
+
# Apply phrase boost: nodes matching exact phrase get 50% score bonus
|
|
983
|
+
if row[0] in phrase_boost_nids:
|
|
984
|
+
fts_score *= 1.5
|
|
985
|
+
nid = row[0]
|
|
986
|
+
if nid in seen_nids:
|
|
987
|
+
# Keep the higher score for the same node
|
|
988
|
+
idx = seen_nids[nid]
|
|
989
|
+
if fts_score > results[idx]["fts_score"]:
|
|
990
|
+
results[idx]["fts_score"] = round(fts_score, 6)
|
|
991
|
+
continue
|
|
992
|
+
seen_nids[nid] = len(results)
|
|
993
|
+
meta = node_meta.get((nid, row[1]))
|
|
994
|
+
results.append({
|
|
995
|
+
"node_id": nid,
|
|
996
|
+
"doc_id": row[1],
|
|
997
|
+
"title": meta["title"] if meta else row[2],
|
|
998
|
+
"summary": meta["summary"] if meta else row[3],
|
|
999
|
+
"depth": meta["depth"] if meta else 0,
|
|
1000
|
+
"fts_score": round(fts_score, 6),
|
|
1001
|
+
})
|
|
1002
|
+
|
|
1003
|
+
# Re-sort after phrase boosting
|
|
1004
|
+
if phrase_boost_nids:
|
|
1005
|
+
results.sort(key=lambda x: -x["fts_score"])
|
|
1006
|
+
|
|
1007
|
+
return results
|
|
1008
|
+
|
|
1009
|
+
def _search_like(
|
|
1010
|
+
self,
|
|
1011
|
+
query: str,
|
|
1012
|
+
doc_id: Optional[str] = None,
|
|
1013
|
+
top_k: int = 20,
|
|
1014
|
+
) -> list[dict]:
|
|
1015
|
+
"""Fallback search using LIKE when FTS5 is unavailable.
|
|
1016
|
+
|
|
1017
|
+
Splits query into keywords and scores nodes by weighted keyword hits
|
|
1018
|
+
across (title, summary, body, code_blocks, front_matter).
|
|
1019
|
+
"""
|
|
1020
|
+
tokens = _tokenize_for_fts(query)
|
|
1021
|
+
if not tokens.strip():
|
|
1022
|
+
return []
|
|
1023
|
+
keywords = [kw.strip().lower() for kw in tokens.split() if kw.strip()]
|
|
1024
|
+
if not keywords:
|
|
1025
|
+
return []
|
|
1026
|
+
|
|
1027
|
+
w = self._weights
|
|
1028
|
+
|
|
1029
|
+
# Pre-fetch all node metadata to avoid N+1 queries
|
|
1030
|
+
if doc_id:
|
|
1031
|
+
meta_rows = self._conn.execute(
|
|
1032
|
+
"SELECT node_id, doc_id, title, summary, depth FROM nodes WHERE doc_id = ?",
|
|
1033
|
+
(doc_id,),
|
|
1034
|
+
).fetchall()
|
|
1035
|
+
else:
|
|
1036
|
+
meta_rows = self._conn.execute(
|
|
1037
|
+
"SELECT node_id, doc_id, title, summary, depth FROM nodes"
|
|
1038
|
+
).fetchall()
|
|
1039
|
+
meta_map = {(r[0], r[1]): {"title": r[2], "summary": r[3], "depth": r[4]} for r in meta_rows}
|
|
1040
|
+
|
|
1041
|
+
if doc_id:
|
|
1042
|
+
rows = self._conn.execute(
|
|
1043
|
+
"""SELECT node_id, doc_id, title, summary, body, code_blocks, front_matter
|
|
1044
|
+
FROM fts_nodes WHERE doc_id = ?""",
|
|
1045
|
+
(doc_id,),
|
|
1046
|
+
).fetchall()
|
|
1047
|
+
else:
|
|
1048
|
+
rows = self._conn.execute(
|
|
1049
|
+
"SELECT node_id, doc_id, title, summary, body, code_blocks, front_matter FROM fts_nodes"
|
|
1050
|
+
).fetchall()
|
|
1051
|
+
|
|
1052
|
+
scored: list[tuple[float, dict]] = []
|
|
1053
|
+
seen_original_nids: dict[str, int] = {}
|
|
1054
|
+
for row in rows:
|
|
1055
|
+
nid, did, title, summary, body, code_blocks, front_matter = row
|
|
1056
|
+
score = 0.0
|
|
1057
|
+
fields = [
|
|
1058
|
+
(title or "", w["title"]),
|
|
1059
|
+
(summary or "", w["summary"]),
|
|
1060
|
+
(body or "", w["body"]),
|
|
1061
|
+
(code_blocks or "", w["code_blocks"]),
|
|
1062
|
+
(front_matter or "", w["front_matter"]),
|
|
1063
|
+
]
|
|
1064
|
+
for kw in keywords:
|
|
1065
|
+
for text, weight in fields:
|
|
1066
|
+
if kw in text.lower():
|
|
1067
|
+
score += weight
|
|
1068
|
+
if score > 0:
|
|
1069
|
+
if nid in seen_original_nids:
|
|
1070
|
+
idx = seen_original_nids[nid]
|
|
1071
|
+
if score > scored[idx][0]:
|
|
1072
|
+
meta = meta_map.get((nid, did))
|
|
1073
|
+
scored[idx] = (score, {
|
|
1074
|
+
"node_id": nid,
|
|
1075
|
+
"doc_id": did,
|
|
1076
|
+
"title": meta["title"] if meta else title,
|
|
1077
|
+
"summary": meta["summary"] if meta else summary,
|
|
1078
|
+
"depth": meta["depth"] if meta else 0,
|
|
1079
|
+
"fts_score": round(score, 6),
|
|
1080
|
+
})
|
|
1081
|
+
continue
|
|
1082
|
+
seen_original_nids[nid] = len(scored)
|
|
1083
|
+
meta = meta_map.get((nid, did))
|
|
1084
|
+
scored.append((score, {
|
|
1085
|
+
"node_id": nid,
|
|
1086
|
+
"doc_id": did,
|
|
1087
|
+
"title": meta["title"] if meta else title,
|
|
1088
|
+
"summary": meta["summary"] if meta else summary,
|
|
1089
|
+
"depth": meta["depth"] if meta else 0,
|
|
1090
|
+
"fts_score": round(score, 6),
|
|
1091
|
+
}))
|
|
1092
|
+
|
|
1093
|
+
scored.sort(key=lambda x: -x[0])
|
|
1094
|
+
return [item[1] for item in scored[:top_k]]
|
|
1095
|
+
|
|
1096
|
+
def like_search(self, query: str, top_k: int = 20, use_regex: bool = False) -> list[dict]:
|
|
1097
|
+
"""Substring or regex search on nodes.summary/title (original text, not tokenized).
|
|
1098
|
+
|
|
1099
|
+
Fallback when FTS5 tokenization causes query mismatches.
|
|
1100
|
+
Searches the ``nodes`` table which stores original (un-tokenized) text,
|
|
1101
|
+
so phrases that were split by jieba can still be matched.
|
|
1102
|
+
|
|
1103
|
+
Args:
|
|
1104
|
+
query: search string or regex pattern (when use_regex=True).
|
|
1105
|
+
top_k: max results.
|
|
1106
|
+
use_regex: if True, treat query as a regular expression (SQLite REGEXP).
|
|
1107
|
+
|
|
1108
|
+
If ``nodes`` yields no matches, searches ``documents.structure_json``
|
|
1109
|
+
which contains the full original document structure.
|
|
1110
|
+
|
|
1111
|
+
Returns same format as search(): list of {node_id, doc_id, title, summary, fts_score, depth}.
|
|
1112
|
+
"""
|
|
1113
|
+
results: list[dict] = []
|
|
1114
|
+
seen: set[str] = set()
|
|
1115
|
+
|
|
1116
|
+
# Build match expression based on mode
|
|
1117
|
+
if use_regex:
|
|
1118
|
+
compiled = re.compile(query, re.IGNORECASE)
|
|
1119
|
+
where_clause = "n.summary REGEXP ? OR n.title REGEXP ?"
|
|
1120
|
+
params = (query, query)
|
|
1121
|
+
else:
|
|
1122
|
+
pattern = f"%{query}%"
|
|
1123
|
+
where_clause = "n.summary LIKE ? OR n.title LIKE ?"
|
|
1124
|
+
params = (pattern, pattern)
|
|
1125
|
+
|
|
1126
|
+
# Phase 1: Search nodes.summary and nodes.title (original text)
|
|
1127
|
+
rows = self._conn.execute(
|
|
1128
|
+
f"""SELECT n.node_id, n.doc_id, n.title, n.summary, n.depth,
|
|
1129
|
+
d.doc_name, d.source_path
|
|
1130
|
+
FROM nodes n
|
|
1131
|
+
JOIN documents d ON n.doc_id = d.doc_id
|
|
1132
|
+
WHERE {where_clause}
|
|
1133
|
+
ORDER BY n.depth ASC""",
|
|
1134
|
+
params,
|
|
1135
|
+
).fetchall()
|
|
1136
|
+
|
|
1137
|
+
for row in rows:
|
|
1138
|
+
nid, did = row[0], row[1]
|
|
1139
|
+
key = f"{did}/{nid}"
|
|
1140
|
+
if key in seen:
|
|
1141
|
+
continue
|
|
1142
|
+
seen.add(key)
|
|
1143
|
+
results.append({
|
|
1144
|
+
"node_id": nid,
|
|
1145
|
+
"doc_id": did,
|
|
1146
|
+
"title": row[2] or "",
|
|
1147
|
+
"summary": row[3] or "",
|
|
1148
|
+
"depth": row[4] or 0,
|
|
1149
|
+
"fts_score": 1.0,
|
|
1150
|
+
})
|
|
1151
|
+
|
|
1152
|
+
if results:
|
|
1153
|
+
return results[:top_k]
|
|
1154
|
+
|
|
1155
|
+
# Phase 2: Search documents.structure_json (full original text)
|
|
1156
|
+
if use_regex:
|
|
1157
|
+
match_fn = compiled.search
|
|
1158
|
+
else:
|
|
1159
|
+
query_lower = query.lower()
|
|
1160
|
+
match_fn = lambda t: query_lower in t.lower() if t else False
|
|
1161
|
+
|
|
1162
|
+
try:
|
|
1163
|
+
if use_regex:
|
|
1164
|
+
doc_rows = self._conn.execute(
|
|
1165
|
+
"SELECT doc_id, structure_json FROM documents WHERE structure_json REGEXP ?",
|
|
1166
|
+
(query,),
|
|
1167
|
+
).fetchall()
|
|
1168
|
+
else:
|
|
1169
|
+
doc_rows = self._conn.execute(
|
|
1170
|
+
"SELECT doc_id, structure_json FROM documents WHERE structure_json LIKE ?",
|
|
1171
|
+
(pattern,),
|
|
1172
|
+
).fetchall()
|
|
1173
|
+
except Exception:
|
|
1174
|
+
return []
|
|
1175
|
+
|
|
1176
|
+
for doc_id, structure_json in doc_rows:
|
|
1177
|
+
if not structure_json:
|
|
1178
|
+
continue
|
|
1179
|
+
try:
|
|
1180
|
+
nodes_data = json.loads(structure_json)
|
|
1181
|
+
except (json.JSONDecodeError, TypeError):
|
|
1182
|
+
continue
|
|
1183
|
+
if not isinstance(nodes_data, list):
|
|
1184
|
+
continue
|
|
1185
|
+
from .tree import flatten_tree
|
|
1186
|
+
for node in flatten_tree(nodes_data):
|
|
1187
|
+
text = node.get("text", "") or ""
|
|
1188
|
+
title = node.get("title", "") or ""
|
|
1189
|
+
if match_fn(text) or match_fn(title):
|
|
1190
|
+
nid = node.get("node_id", "")
|
|
1191
|
+
key = f"{doc_id}/{nid}"
|
|
1192
|
+
if key in seen:
|
|
1193
|
+
continue
|
|
1194
|
+
seen.add(key)
|
|
1195
|
+
# 提取包含匹配关键词的片段
|
|
1196
|
+
snippet = _extract_match_snippet(text, query, use_regex, size=300)
|
|
1197
|
+
results.append({
|
|
1198
|
+
"node_id": nid,
|
|
1199
|
+
"doc_id": doc_id,
|
|
1200
|
+
"title": title,
|
|
1201
|
+
"summary": snippet,
|
|
1202
|
+
"depth": node.get("depth", 0),
|
|
1203
|
+
"fts_score": 0.5,
|
|
1204
|
+
})
|
|
1205
|
+
|
|
1206
|
+
return results[:top_k]
|
|
1207
|
+
|
|
1208
|
+
def search_with_aggregation(
|
|
1209
|
+
self,
|
|
1210
|
+
query: str,
|
|
1211
|
+
group_by_doc: bool = True,
|
|
1212
|
+
top_k: int = 20,
|
|
1213
|
+
fts_expression: Optional[str] = None,
|
|
1214
|
+
) -> list[dict]:
|
|
1215
|
+
"""Search with SQL aggregation capabilities.
|
|
1216
|
+
|
|
1217
|
+
Returns per-document aggregated results: total hits, max score, avg score.
|
|
1218
|
+
"""
|
|
1219
|
+
if not group_by_doc:
|
|
1220
|
+
return self.search(query, top_k=top_k, fts_expression=fts_expression)
|
|
1221
|
+
|
|
1222
|
+
# Two-step: first get all matched nodes with scores, then aggregate in Python
|
|
1223
|
+
results = self.search(query, top_k=200, fts_expression=fts_expression)
|
|
1224
|
+
if not results:
|
|
1225
|
+
return []
|
|
1226
|
+
|
|
1227
|
+
doc_agg: dict[str, dict] = {}
|
|
1228
|
+
for r in results:
|
|
1229
|
+
did = r["doc_id"]
|
|
1230
|
+
if did not in doc_agg:
|
|
1231
|
+
# Fetch doc_name
|
|
1232
|
+
row = self._conn.execute(
|
|
1233
|
+
"SELECT doc_name FROM documents WHERE doc_id = ?", (did,)
|
|
1234
|
+
).fetchone()
|
|
1235
|
+
doc_agg[did] = {
|
|
1236
|
+
"doc_id": did,
|
|
1237
|
+
"doc_name": row[0] if row else "",
|
|
1238
|
+
"hit_count": 0,
|
|
1239
|
+
"best_score": 0.0,
|
|
1240
|
+
"total_score": 0.0,
|
|
1241
|
+
}
|
|
1242
|
+
doc_agg[did]["hit_count"] += 1
|
|
1243
|
+
doc_agg[did]["best_score"] = max(doc_agg[did]["best_score"], r["fts_score"])
|
|
1244
|
+
doc_agg[did]["total_score"] += r["fts_score"]
|
|
1245
|
+
|
|
1246
|
+
agg_results = []
|
|
1247
|
+
for agg in doc_agg.values():
|
|
1248
|
+
agg["avg_score"] = round(agg["total_score"] / agg["hit_count"], 6)
|
|
1249
|
+
agg["best_score"] = round(agg["best_score"], 6)
|
|
1250
|
+
del agg["total_score"]
|
|
1251
|
+
agg_results.append(agg)
|
|
1252
|
+
|
|
1253
|
+
agg_results.sort(key=lambda x: -x["best_score"])
|
|
1254
|
+
return agg_results[:top_k]
|
|
1255
|
+
|
|
1256
|
+
def score_nodes(self, query: str, doc_id: str, ancestor_decay: float = 0.6) -> dict[str, float]:
|
|
1257
|
+
"""PreFilter protocol: return {node_id: score} for search() integration.
|
|
1258
|
+
|
|
1259
|
+
This allows FTS5Index to be used as a drop-in PreFilter in the search pipeline.
|
|
1260
|
+
|
|
1261
|
+
Includes:
|
|
1262
|
+
- Ancestor score propagation: parent nodes inherit child scores
|
|
1263
|
+
|
|
1264
|
+
For scoring multiple documents at once, prefer score_nodes_batch() which uses
|
|
1265
|
+
a single SQL query instead of one query per document.
|
|
1266
|
+
"""
|
|
1267
|
+
result = self.score_nodes_batch(query, doc_ids=[doc_id], ancestor_decay=ancestor_decay)
|
|
1268
|
+
return result.get(doc_id, {})
|
|
1269
|
+
|
|
1270
|
+
def score_nodes_batch(
|
|
1271
|
+
self,
|
|
1272
|
+
query: str,
|
|
1273
|
+
doc_ids: list[str] | None = None,
|
|
1274
|
+
ancestor_decay: float = 0.6,
|
|
1275
|
+
fts_expression: Optional[str] = None,
|
|
1276
|
+
) -> dict[str, dict[str, float]]:
|
|
1277
|
+
"""Batch version of score_nodes: score all documents in a single SQL query.
|
|
1278
|
+
|
|
1279
|
+
Returns {doc_id: {node_id: score}} for all matched documents.
|
|
1280
|
+
|
|
1281
|
+
This is the fast path for tree search, replacing the N_docs loop:
|
|
1282
|
+
# Before (slow): N SQL queries
|
|
1283
|
+
for doc in documents:
|
|
1284
|
+
scores = fts_index.score_nodes(query, doc.doc_id)
|
|
1285
|
+
|
|
1286
|
+
# After (fast): 1 SQL query
|
|
1287
|
+
all_scores = fts_index.score_nodes_batch(query, [d.doc_id for d in documents])
|
|
1288
|
+
|
|
1289
|
+
Args:
|
|
1290
|
+
query: natural language query (tokenized internally)
|
|
1291
|
+
doc_ids: optional filter to specific documents
|
|
1292
|
+
ancestor_decay: propagation weight from child scores to parent nodes (0 = off)
|
|
1293
|
+
fts_expression: raw FTS5 MATCH expression that *overrides* automatic query
|
|
1294
|
+
tokenization. Supports full FTS5 syntax including prefix matching (*),
|
|
1295
|
+
AND/OR/NOT, NEAR(), and column filters.
|
|
1296
|
+
|
|
1297
|
+
Examples::
|
|
1298
|
+
|
|
1299
|
+
# Prefix match: "fts" matches fts, fts5, ftsearch, ...
|
|
1300
|
+
fts_expression="fts*"
|
|
1301
|
+
|
|
1302
|
+
# Multi-term prefix OR
|
|
1303
|
+
fts_expression="fts* OR python*"
|
|
1304
|
+
|
|
1305
|
+
# Exact phrase
|
|
1306
|
+
fts_expression='"machine learning"'
|
|
1307
|
+
|
|
1308
|
+
# Column-scoped search
|
|
1309
|
+
fts_expression="title : config*"
|
|
1310
|
+
|
|
1311
|
+
# Build with helper
|
|
1312
|
+
expr = FTS5Index.build_fts_expression(["fts", "python"],
|
|
1313
|
+
prefix=True, operator="OR")
|
|
1314
|
+
"""
|
|
1315
|
+
match_expr = self._build_match_expr(query, fts_expression)
|
|
1316
|
+
if match_expr is None:
|
|
1317
|
+
return {}
|
|
1318
|
+
|
|
1319
|
+
w = self._weights
|
|
1320
|
+
weight_args = f"{w['title']}, {w['summary']}, {w['body']}, {w['code_blocks']}, {w['front_matter']}"
|
|
1321
|
+
|
|
1322
|
+
# Single SQL query across all requested doc_ids (or entire index)
|
|
1323
|
+
if doc_ids:
|
|
1324
|
+
placeholders = ",".join("?" * len(doc_ids))
|
|
1325
|
+
sql = f"""
|
|
1326
|
+
SELECT f.node_id, f.doc_id,
|
|
1327
|
+
bm25(fts_nodes, {weight_args}) AS rank_score
|
|
1328
|
+
FROM fts_nodes f
|
|
1329
|
+
WHERE fts_nodes MATCH ?
|
|
1330
|
+
AND f.doc_id IN ({placeholders})
|
|
1331
|
+
ORDER BY rank_score
|
|
1332
|
+
LIMIT 5000
|
|
1333
|
+
"""
|
|
1334
|
+
params = (match_expr, *doc_ids)
|
|
1335
|
+
else:
|
|
1336
|
+
sql = f"""
|
|
1337
|
+
SELECT f.node_id, f.doc_id,
|
|
1338
|
+
bm25(fts_nodes, {weight_args}) AS rank_score
|
|
1339
|
+
FROM fts_nodes f
|
|
1340
|
+
WHERE fts_nodes MATCH ?
|
|
1341
|
+
ORDER BY rank_score
|
|
1342
|
+
LIMIT 5000
|
|
1343
|
+
"""
|
|
1344
|
+
params = (match_expr,)
|
|
1345
|
+
|
|
1346
|
+
try:
|
|
1347
|
+
rows = self._conn.execute(sql, params).fetchall()
|
|
1348
|
+
except Exception:
|
|
1349
|
+
return {}
|
|
1350
|
+
|
|
1351
|
+
if not rows:
|
|
1352
|
+
return {}
|
|
1353
|
+
|
|
1354
|
+
# Group raw scores by doc_id
|
|
1355
|
+
per_doc_raw: dict[str, dict[str, float]] = {}
|
|
1356
|
+
for node_id, doc_id, rank_score in rows:
|
|
1357
|
+
fts_score = -rank_score if rank_score else 0.0
|
|
1358
|
+
per_doc_raw.setdefault(doc_id, {})
|
|
1359
|
+
old = per_doc_raw[doc_id].get(node_id, 0.0)
|
|
1360
|
+
per_doc_raw[doc_id][node_id] = max(old, fts_score)
|
|
1361
|
+
|
|
1362
|
+
# Per-doc: normalize + ancestor propagation in one pass
|
|
1363
|
+
result: dict[str, dict[str, float]] = {}
|
|
1364
|
+
doc_children_map: dict[str, dict[str, list[str]]] = {}
|
|
1365
|
+
if ancestor_decay > 0 and per_doc_raw:
|
|
1366
|
+
# Single query to fetch parent maps for all affected docs
|
|
1367
|
+
affected_docs = list(per_doc_raw.keys())
|
|
1368
|
+
ph = ",".join("?" * len(affected_docs))
|
|
1369
|
+
parent_rows = self._conn.execute(
|
|
1370
|
+
f"SELECT doc_id, node_id, parent_node_id FROM nodes WHERE doc_id IN ({ph})",
|
|
1371
|
+
affected_docs,
|
|
1372
|
+
).fetchall()
|
|
1373
|
+
# Build per-doc children maps for bottom-up propagation
|
|
1374
|
+
for d_id, nid, pid in parent_rows:
|
|
1375
|
+
if pid:
|
|
1376
|
+
doc_children_map.setdefault(d_id, {}).setdefault(pid, []).append(nid)
|
|
1377
|
+
|
|
1378
|
+
for doc_id, raw_scores in per_doc_raw.items():
|
|
1379
|
+
# Normalize to [0, 1]
|
|
1380
|
+
max_s = max(raw_scores.values()) if raw_scores else 1.0
|
|
1381
|
+
if max_s <= 0:
|
|
1382
|
+
max_s = 1.0
|
|
1383
|
+
scores = {nid: s / max_s for nid, s in raw_scores.items()}
|
|
1384
|
+
|
|
1385
|
+
# Ancestor propagation
|
|
1386
|
+
if ancestor_decay > 0:
|
|
1387
|
+
children_map = doc_children_map.get(doc_id, {})
|
|
1388
|
+
for pid, cids in children_map.items():
|
|
1389
|
+
child_scores = [scores.get(c, 0.0) for c in cids]
|
|
1390
|
+
if not child_scores:
|
|
1391
|
+
continue
|
|
1392
|
+
bonus = ancestor_decay * max(child_scores)
|
|
1393
|
+
scores[pid] = scores.get(pid, 0.0) + bonus
|
|
1394
|
+
|
|
1395
|
+
# Re-normalize after propagation
|
|
1396
|
+
final_max = max(scores.values()) if scores else 1.0
|
|
1397
|
+
if final_max > 1.0:
|
|
1398
|
+
scores = {nid: s / final_max for nid, s in scores.items()}
|
|
1399
|
+
|
|
1400
|
+
result[doc_id] = {nid: round(s, 6) for nid, s in scores.items()}
|
|
1401
|
+
|
|
1402
|
+
return result
|
|
1403
|
+
|
|
1404
|
+
def ranked_node_ids(
|
|
1405
|
+
self,
|
|
1406
|
+
query: str,
|
|
1407
|
+
doc_ids: list[str] | None = None,
|
|
1408
|
+
top_k: int = 10,
|
|
1409
|
+
fts_expression: Optional[str] = None,
|
|
1410
|
+
) -> list[str]:
|
|
1411
|
+
"""Convenience: return top-k node IDs ranked by FTS5 score.
|
|
1412
|
+
|
|
1413
|
+
Wraps score_nodes_batch → flatten → sort → top-k in one call.
|
|
1414
|
+
Useful for flat FTS5 evaluation without manual dict wrangling.
|
|
1415
|
+
|
|
1416
|
+
Args:
|
|
1417
|
+
query: search query
|
|
1418
|
+
doc_ids: optional filter to specific documents
|
|
1419
|
+
top_k: max results
|
|
1420
|
+
fts_expression: raw FTS5 MATCH expression (overrides query tokenization).
|
|
1421
|
+
Use for prefix matching, boolean operators, NEAR(), etc.
|
|
1422
|
+
Example: ``fts_expression="config*"`` matches config, configuration, ...
|
|
1423
|
+
|
|
1424
|
+
Returns:
|
|
1425
|
+
list of node_id strings, highest score first
|
|
1426
|
+
"""
|
|
1427
|
+
batch = self.score_nodes_batch(query, doc_ids=doc_ids, fts_expression=fts_expression)
|
|
1428
|
+
all_scored: list[tuple[str, float]] = []
|
|
1429
|
+
for nscores in batch.values():
|
|
1430
|
+
all_scored.extend(nscores.items())
|
|
1431
|
+
all_scored.sort(key=lambda x: -x[1])
|
|
1432
|
+
return [nid for nid, _ in all_scored[:top_k]]
|
|
1433
|
+
|
|
1434
|
+
# -------------------------------------------------------------------
|
|
1435
|
+
# Document persistence (tree structure storage)
|
|
1436
|
+
# -------------------------------------------------------------------
|
|
1437
|
+
|
|
1438
|
+
def save_document(self, document, auto_commit: bool = True) -> None:
|
|
1439
|
+
"""Save/update a Document's tree structure into the DB.
|
|
1440
|
+
|
|
1441
|
+
This persists the tree structure so that JSON files are no longer needed.
|
|
1442
|
+
FTS indexing is NOT performed here — call index_document() separately.
|
|
1443
|
+
|
|
1444
|
+
Args:
|
|
1445
|
+
document: Document object with structure tree
|
|
1446
|
+
auto_commit: if False, skip commit (caller is responsible for committing)
|
|
1447
|
+
"""
|
|
1448
|
+
from .tree import flatten_tree
|
|
1449
|
+
structure_json = json.dumps(document.structure, ensure_ascii=False)
|
|
1450
|
+
content_hash = hashlib.md5(structure_json.encode()).hexdigest()
|
|
1451
|
+
self._conn.execute(
|
|
1452
|
+
"""INSERT OR REPLACE INTO documents
|
|
1453
|
+
(doc_id, doc_name, doc_description, source_path, source_type, structure_json, node_count, index_hash)
|
|
1454
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, ?)""",
|
|
1455
|
+
(
|
|
1456
|
+
document.doc_id, document.doc_name, document.doc_description,
|
|
1457
|
+
document.metadata.get("source_path", ""),
|
|
1458
|
+
document.source_type,
|
|
1459
|
+
structure_json,
|
|
1460
|
+
len(flatten_tree(document.structure)),
|
|
1461
|
+
content_hash,
|
|
1462
|
+
),
|
|
1463
|
+
)
|
|
1464
|
+
if auto_commit:
|
|
1465
|
+
self._conn.commit()
|
|
1466
|
+
|
|
1467
|
+
def load_document(self, doc_id: str):
|
|
1468
|
+
"""Load a single Document from the DB by doc_id.
|
|
1469
|
+
|
|
1470
|
+
Returns:
|
|
1471
|
+
Document object, or None if not found.
|
|
1472
|
+
"""
|
|
1473
|
+
from .tree import Document
|
|
1474
|
+
row = self._conn.execute(
|
|
1475
|
+
"SELECT doc_id, doc_name, doc_description, source_path, source_type, structure_json FROM documents WHERE doc_id = ?",
|
|
1476
|
+
(doc_id,),
|
|
1477
|
+
).fetchone()
|
|
1478
|
+
if not row:
|
|
1479
|
+
return None
|
|
1480
|
+
structure = json.loads(row[5]) if row[5] else []
|
|
1481
|
+
return Document(
|
|
1482
|
+
doc_id=row[0],
|
|
1483
|
+
doc_name=row[1],
|
|
1484
|
+
structure=structure,
|
|
1485
|
+
doc_description=row[2] or "",
|
|
1486
|
+
metadata={"source_path": row[3] or ""},
|
|
1487
|
+
source_type=row[4] or "",
|
|
1488
|
+
)
|
|
1489
|
+
|
|
1490
|
+
def load_document_by_source_path(self, source_path: str) -> Optional["Document"]:
|
|
1491
|
+
"""按 source_path 加载单个 Document(含 structure_json 反序列化)。
|
|
1492
|
+
|
|
1493
|
+
Args:
|
|
1494
|
+
source_path: 文件绝对路径,必须与索引时存入 documents.source_path 完全一致。
|
|
1495
|
+
|
|
1496
|
+
Returns:
|
|
1497
|
+
Document 对象;无匹配返回 None。
|
|
1498
|
+
"""
|
|
1499
|
+
from .tree import Document # 局部导入避免循环依赖
|
|
1500
|
+
|
|
1501
|
+
row = self._conn.execute(
|
|
1502
|
+
"SELECT doc_id, doc_name, doc_description, source_path, source_type, structure_json "
|
|
1503
|
+
"FROM documents WHERE source_path = ? LIMIT 1",
|
|
1504
|
+
(source_path,),
|
|
1505
|
+
).fetchone()
|
|
1506
|
+
if not row:
|
|
1507
|
+
return None
|
|
1508
|
+
doc_id, doc_name, doc_description, sp, source_type, structure_json = row
|
|
1509
|
+
structure = json.loads(structure_json) if structure_json else []
|
|
1510
|
+
return Document(
|
|
1511
|
+
doc_id=doc_id,
|
|
1512
|
+
doc_name=doc_name or "",
|
|
1513
|
+
doc_description=doc_description or "",
|
|
1514
|
+
structure=structure,
|
|
1515
|
+
source_type=source_type or "",
|
|
1516
|
+
metadata={"source_path": sp or ""},
|
|
1517
|
+
)
|
|
1518
|
+
|
|
1519
|
+
def load_all_documents(self) -> list:
|
|
1520
|
+
"""Load all Documents stored in the DB.
|
|
1521
|
+
|
|
1522
|
+
Returns:
|
|
1523
|
+
List of Document objects.
|
|
1524
|
+
"""
|
|
1525
|
+
from .tree import Document
|
|
1526
|
+
rows = self._conn.execute(
|
|
1527
|
+
"SELECT doc_id, doc_name, doc_description, source_path, source_type, structure_json FROM documents ORDER BY doc_id"
|
|
1528
|
+
).fetchall()
|
|
1529
|
+
documents = []
|
|
1530
|
+
for row in rows:
|
|
1531
|
+
structure = json.loads(row[5]) if row[5] else []
|
|
1532
|
+
documents.append(Document(
|
|
1533
|
+
doc_id=row[0],
|
|
1534
|
+
doc_name=row[1],
|
|
1535
|
+
structure=structure,
|
|
1536
|
+
doc_description=row[2] or "",
|
|
1537
|
+
metadata={"source_path": row[3] or ""},
|
|
1538
|
+
source_type=row[4] or "",
|
|
1539
|
+
))
|
|
1540
|
+
return documents
|
|
1541
|
+
|
|
1542
|
+
def delete_document(self, doc_id: str) -> bool:
|
|
1543
|
+
"""Delete a document and all its indexed data from the DB atomically.
|
|
1544
|
+
|
|
1545
|
+
Clears all four storage locations in a single transaction:
|
|
1546
|
+
- ``fts_nodes`` (FTS5 inverted index entries)
|
|
1547
|
+
- ``nodes`` (structured node metadata)
|
|
1548
|
+
- ``documents`` (tree structure + document metadata)
|
|
1549
|
+
- ``index_meta`` (incremental indexing fingerprint)
|
|
1550
|
+
|
|
1551
|
+
Clearing ``index_meta`` is critical: without it the incremental indexing
|
|
1552
|
+
logic would see the file as already-processed and silently skip re-indexing
|
|
1553
|
+
it after a subsequent ``index()`` call.
|
|
1554
|
+
|
|
1555
|
+
Args:
|
|
1556
|
+
doc_id: document identifier to delete.
|
|
1557
|
+
|
|
1558
|
+
Returns:
|
|
1559
|
+
``True`` if the document existed and was deleted, ``False`` if it was
|
|
1560
|
+
not found (operation is idempotent — no exception is raised).
|
|
1561
|
+
"""
|
|
1562
|
+
# Check existence before entering the transaction so we can return a
|
|
1563
|
+
# meaningful bool without relying on changes_count() across all tables.
|
|
1564
|
+
row = self._conn.execute(
|
|
1565
|
+
"SELECT source_path FROM documents WHERE doc_id = ?", (doc_id,)
|
|
1566
|
+
).fetchone()
|
|
1567
|
+
|
|
1568
|
+
if row is None:
|
|
1569
|
+
logger.warning("delete_document: doc_id=%r not found, nothing deleted", doc_id)
|
|
1570
|
+
return False
|
|
1571
|
+
|
|
1572
|
+
source_path = row[0] or ""
|
|
1573
|
+
|
|
1574
|
+
try:
|
|
1575
|
+
# Single atomic transaction — all-or-nothing.
|
|
1576
|
+
with self._conn:
|
|
1577
|
+
# 1. FTS5 virtual table: must delete by rowid (UNINDEXED columns
|
|
1578
|
+
# cannot be used in a WHERE clause for DELETE on FTS5 tables).
|
|
1579
|
+
if self._use_fts5:
|
|
1580
|
+
old_rowids = self._conn.execute(
|
|
1581
|
+
"SELECT rowid FROM fts_nodes WHERE doc_id = ?", (doc_id,)
|
|
1582
|
+
).fetchall()
|
|
1583
|
+
if old_rowids:
|
|
1584
|
+
placeholders = ",".join("?" for _ in old_rowids)
|
|
1585
|
+
self._conn.execute(
|
|
1586
|
+
f"DELETE FROM fts_nodes WHERE rowid IN ({placeholders})",
|
|
1587
|
+
[r[0] for r in old_rowids],
|
|
1588
|
+
)
|
|
1589
|
+
else:
|
|
1590
|
+
self._conn.execute("DELETE FROM fts_nodes WHERE doc_id = ?", (doc_id,))
|
|
1591
|
+
|
|
1592
|
+
# 2. Structured node metadata.
|
|
1593
|
+
self._conn.execute("DELETE FROM nodes WHERE doc_id = ?", (doc_id,))
|
|
1594
|
+
|
|
1595
|
+
# 3. Document record (tree structure + metadata).
|
|
1596
|
+
self._conn.execute("DELETE FROM documents WHERE doc_id = ?", (doc_id,))
|
|
1597
|
+
|
|
1598
|
+
# 4. Incremental index fingerprint — CRITICAL: must be cleared so
|
|
1599
|
+
# that the next index() call re-processes this file rather than
|
|
1600
|
+
# skipping it as "already indexed".
|
|
1601
|
+
if source_path:
|
|
1602
|
+
self._conn.execute(
|
|
1603
|
+
"DELETE FROM index_meta WHERE source_path = ?", (source_path,)
|
|
1604
|
+
)
|
|
1605
|
+
|
|
1606
|
+
except sqlite3.DatabaseError as e:
|
|
1607
|
+
logger.error(
|
|
1608
|
+
"delete_document failed for doc_id=%r (source_path=%r): %s",
|
|
1609
|
+
doc_id, source_path, e,
|
|
1610
|
+
)
|
|
1611
|
+
raise
|
|
1612
|
+
|
|
1613
|
+
logger.info("Deleted document doc_id=%r (source_path=%r)", doc_id, source_path)
|
|
1614
|
+
return True
|
|
1615
|
+
|
|
1616
|
+
def remove_document(self, doc_id: str) -> None:
|
|
1617
|
+
"""Remove a document from the DB.
|
|
1618
|
+
|
|
1619
|
+
.. deprecated::
|
|
1620
|
+
Use :meth:`delete_document` instead. ``delete_document`` fixes a
|
|
1621
|
+
bug where ``index_meta`` was not cleared, uses an atomic transaction,
|
|
1622
|
+
and returns a boolean indicating whether the document existed.
|
|
1623
|
+
"""
|
|
1624
|
+
self.delete_document(doc_id)
|
|
1625
|
+
|
|
1626
|
+
def find_doc_by_fingerprint(self, file_hash: str, exclude_paths: Optional[set] = None) -> Optional[str]:
|
|
1627
|
+
"""Find a doc whose ``index_meta.file_hash`` matches ``file_hash``.
|
|
1628
|
+
|
|
1629
|
+
Used by the indexer to detect file moves/renames: if the same fingerprint
|
|
1630
|
+
already exists under a different source_path, we can update the path
|
|
1631
|
+
instead of re-parsing+re-indexing the file.
|
|
1632
|
+
|
|
1633
|
+
Args:
|
|
1634
|
+
file_hash: target fingerprint string.
|
|
1635
|
+
exclude_paths: source_paths to skip (e.g. paths still on disk).
|
|
1636
|
+
|
|
1637
|
+
Returns:
|
|
1638
|
+
doc_id of the matching document, or None if no candidate.
|
|
1639
|
+
"""
|
|
1640
|
+
rows = self._conn.execute(
|
|
1641
|
+
"SELECT source_path FROM index_meta WHERE file_hash = ?",
|
|
1642
|
+
(file_hash,),
|
|
1643
|
+
).fetchall()
|
|
1644
|
+
for (sp,) in rows:
|
|
1645
|
+
if exclude_paths and sp in exclude_paths:
|
|
1646
|
+
continue
|
|
1647
|
+
doc_id = self.get_doc_id_by_source_path(sp)
|
|
1648
|
+
if doc_id is not None:
|
|
1649
|
+
return doc_id
|
|
1650
|
+
return None
|
|
1651
|
+
|
|
1652
|
+
def update_source_path(self, doc_id: str, new_source_path: str) -> None:
|
|
1653
|
+
"""Atomically remap a document's source_path (move/rename support)."""
|
|
1654
|
+
with self._conn:
|
|
1655
|
+
old_row = self._conn.execute(
|
|
1656
|
+
"SELECT source_path FROM documents WHERE doc_id = ?", (doc_id,)
|
|
1657
|
+
).fetchone()
|
|
1658
|
+
self._conn.execute(
|
|
1659
|
+
"UPDATE documents SET source_path = ? WHERE doc_id = ?",
|
|
1660
|
+
(new_source_path, doc_id),
|
|
1661
|
+
)
|
|
1662
|
+
if old_row and old_row[0]:
|
|
1663
|
+
self._conn.execute(
|
|
1664
|
+
"DELETE FROM index_meta WHERE source_path = ?", (old_row[0],)
|
|
1665
|
+
)
|
|
1666
|
+
|
|
1667
|
+
def rename_document(
|
|
1668
|
+
self,
|
|
1669
|
+
old_doc_id: str,
|
|
1670
|
+
new_doc_id: str,
|
|
1671
|
+
new_doc_name: str,
|
|
1672
|
+
new_source_path: str,
|
|
1673
|
+
) -> bool:
|
|
1674
|
+
"""Rename a doc in place across nodes / fts_nodes / documents / index_meta.
|
|
1675
|
+
|
|
1676
|
+
Used by `build_index`'s move-detection pre-pass when a file with an
|
|
1677
|
+
unchanged content fingerprint shows up under a new path. Keeps the
|
|
1678
|
+
doc identity (`doc_id` and `doc_name`) consistent with the new file
|
|
1679
|
+
basename so callers don't see stale names after a rename.
|
|
1680
|
+
|
|
1681
|
+
Returns ``False`` (and writes nothing) when:
|
|
1682
|
+
- the original ``old_doc_id`` no longer exists, or
|
|
1683
|
+
- ``new_doc_id`` is already taken by a *different* document — in
|
|
1684
|
+
that case the caller should fall back to a full re-index.
|
|
1685
|
+
"""
|
|
1686
|
+
old_row = self._conn.execute(
|
|
1687
|
+
"SELECT doc_id, source_path FROM documents WHERE doc_id = ?",
|
|
1688
|
+
(old_doc_id,),
|
|
1689
|
+
).fetchone()
|
|
1690
|
+
if old_row is None:
|
|
1691
|
+
return False
|
|
1692
|
+
old_source_path = old_row[1] or ""
|
|
1693
|
+
|
|
1694
|
+
if new_doc_id != old_doc_id:
|
|
1695
|
+
clash = self._conn.execute(
|
|
1696
|
+
"SELECT 1 FROM documents WHERE doc_id = ?", (new_doc_id,)
|
|
1697
|
+
).fetchone()
|
|
1698
|
+
if clash is not None:
|
|
1699
|
+
return False
|
|
1700
|
+
|
|
1701
|
+
with self._conn:
|
|
1702
|
+
if new_doc_id != old_doc_id:
|
|
1703
|
+
self._conn.execute(
|
|
1704
|
+
"UPDATE nodes SET doc_id = ? WHERE doc_id = ?",
|
|
1705
|
+
(new_doc_id, old_doc_id),
|
|
1706
|
+
)
|
|
1707
|
+
self._conn.execute(
|
|
1708
|
+
"UPDATE fts_nodes SET doc_id = ? WHERE doc_id = ?",
|
|
1709
|
+
(new_doc_id, old_doc_id),
|
|
1710
|
+
)
|
|
1711
|
+
self._conn.execute(
|
|
1712
|
+
"UPDATE documents SET doc_id = ?, doc_name = ?, source_path = ? "
|
|
1713
|
+
"WHERE doc_id = ?",
|
|
1714
|
+
(new_doc_id, new_doc_name, new_source_path, old_doc_id),
|
|
1715
|
+
)
|
|
1716
|
+
else:
|
|
1717
|
+
self._conn.execute(
|
|
1718
|
+
"UPDATE documents SET doc_name = ?, source_path = ? WHERE doc_id = ?",
|
|
1719
|
+
(new_doc_name, new_source_path, old_doc_id),
|
|
1720
|
+
)
|
|
1721
|
+
if old_source_path and old_source_path != new_source_path:
|
|
1722
|
+
self._conn.execute(
|
|
1723
|
+
"DELETE FROM index_meta WHERE source_path = ?", (old_source_path,)
|
|
1724
|
+
)
|
|
1725
|
+
|
|
1726
|
+
return True
|
|
1727
|
+
|
|
1728
|
+
def delete_documents(self, doc_ids: list[str]) -> int:
|
|
1729
|
+
"""Batch-delete multiple documents in a single transaction.
|
|
1730
|
+
|
|
1731
|
+
Significantly faster than calling ``delete_document`` in a loop because
|
|
1732
|
+
every table is hit once with an ``IN (...)`` clause.
|
|
1733
|
+
|
|
1734
|
+
Returns the number of documents that actually existed and were removed.
|
|
1735
|
+
"""
|
|
1736
|
+
if not doc_ids:
|
|
1737
|
+
return 0
|
|
1738
|
+
|
|
1739
|
+
placeholders = ",".join("?" for _ in doc_ids)
|
|
1740
|
+
existing_rows = self._conn.execute(
|
|
1741
|
+
f"SELECT doc_id, source_path FROM documents WHERE doc_id IN ({placeholders})",
|
|
1742
|
+
doc_ids,
|
|
1743
|
+
).fetchall()
|
|
1744
|
+
if not existing_rows:
|
|
1745
|
+
return 0
|
|
1746
|
+
|
|
1747
|
+
existing_ids = [r[0] for r in existing_rows]
|
|
1748
|
+
existing_paths = [r[1] for r in existing_rows if r[1]]
|
|
1749
|
+
ph_e = ",".join("?" for _ in existing_ids)
|
|
1750
|
+
|
|
1751
|
+
with self._conn:
|
|
1752
|
+
if self._use_fts5:
|
|
1753
|
+
old_rowids = self._conn.execute(
|
|
1754
|
+
f"SELECT rowid FROM fts_nodes WHERE doc_id IN ({ph_e})",
|
|
1755
|
+
existing_ids,
|
|
1756
|
+
).fetchall()
|
|
1757
|
+
if old_rowids:
|
|
1758
|
+
# Batch delete in chunks to avoid SQLite's variable limit (~999)
|
|
1759
|
+
_MAX_VARS = 500
|
|
1760
|
+
rowid_list = [r[0] for r in old_rowids]
|
|
1761
|
+
for i in range(0, len(rowid_list), _MAX_VARS):
|
|
1762
|
+
chunk = rowid_list[i:i + _MAX_VARS]
|
|
1763
|
+
ph2 = ",".join("?" for _ in chunk)
|
|
1764
|
+
self._conn.execute(
|
|
1765
|
+
f"DELETE FROM fts_nodes WHERE rowid IN ({ph2})",
|
|
1766
|
+
chunk,
|
|
1767
|
+
)
|
|
1768
|
+
else:
|
|
1769
|
+
self._conn.execute(
|
|
1770
|
+
f"DELETE FROM fts_nodes WHERE doc_id IN ({ph_e})", existing_ids
|
|
1771
|
+
)
|
|
1772
|
+
self._conn.execute(
|
|
1773
|
+
f"DELETE FROM nodes WHERE doc_id IN ({ph_e})", existing_ids
|
|
1774
|
+
)
|
|
1775
|
+
self._conn.execute(
|
|
1776
|
+
f"DELETE FROM documents WHERE doc_id IN ({ph_e})", existing_ids
|
|
1777
|
+
)
|
|
1778
|
+
self._conn.execute(
|
|
1779
|
+
f"DELETE FROM pst_email_meta WHERE doc_id IN ({ph_e})", existing_ids
|
|
1780
|
+
)
|
|
1781
|
+
if existing_paths:
|
|
1782
|
+
ph_p = ",".join("?" for _ in existing_paths)
|
|
1783
|
+
self._conn.execute(
|
|
1784
|
+
f"DELETE FROM index_meta WHERE source_path IN ({ph_p})",
|
|
1785
|
+
existing_paths,
|
|
1786
|
+
)
|
|
1787
|
+
|
|
1788
|
+
logger.info("Batch-deleted %d document(s)", len(existing_ids))
|
|
1789
|
+
return len(existing_ids)
|
|
1790
|
+
|
|
1791
|
+
def get_doc_id_by_source_path(self, source_path: str) -> Optional[str]:
|
|
1792
|
+
"""Look up a doc_id from a source file path.
|
|
1793
|
+
|
|
1794
|
+
Useful when callers know the file path but not the internal doc_id.
|
|
1795
|
+
|
|
1796
|
+
Args:
|
|
1797
|
+
source_path: absolute path of the source file.
|
|
1798
|
+
|
|
1799
|
+
Returns:
|
|
1800
|
+
The ``doc_id`` string, or ``None`` if no document matches.
|
|
1801
|
+
"""
|
|
1802
|
+
row = self._conn.execute(
|
|
1803
|
+
"SELECT doc_id FROM documents WHERE source_path = ?", (source_path,)
|
|
1804
|
+
).fetchone()
|
|
1805
|
+
return row[0] if row else None
|
|
1806
|
+
|
|
1807
|
+
def get_doc_ids_by_source_prefix(self, prefix: str) -> list[str]:
|
|
1808
|
+
"""Return doc_ids whose source_path starts with ``prefix``.
|
|
1809
|
+
|
|
1810
|
+
Used for multi-document sources (e.g. PST archives whose email docs
|
|
1811
|
+
get derived paths ``<file>#<entry_id>``) — cascade delete / replace.
|
|
1812
|
+
"""
|
|
1813
|
+
rows = self._conn.execute(
|
|
1814
|
+
"SELECT doc_id FROM documents WHERE instr(source_path, ?) = 1",
|
|
1815
|
+
(prefix,),
|
|
1816
|
+
).fetchall()
|
|
1817
|
+
return [r[0] for r in rows]
|
|
1818
|
+
|
|
1819
|
+
def has_docs_with_source_prefix(self, prefix: str) -> bool:
|
|
1820
|
+
"""Whether any document's source_path starts with ``prefix``."""
|
|
1821
|
+
row = self._conn.execute(
|
|
1822
|
+
"SELECT 1 FROM documents WHERE instr(source_path, ?) = 1 LIMIT 1",
|
|
1823
|
+
(prefix,),
|
|
1824
|
+
).fetchone()
|
|
1825
|
+
return row is not None
|
|
1826
|
+
|
|
1827
|
+
def count_docs_with_source_prefix(self, prefix: str) -> int:
|
|
1828
|
+
"""Count documents whose source_path starts with ``prefix``."""
|
|
1829
|
+
row = self._conn.execute(
|
|
1830
|
+
"SELECT COUNT(*) FROM documents WHERE instr(source_path, ?) = 1",
|
|
1831
|
+
(prefix,),
|
|
1832
|
+
).fetchone()
|
|
1833
|
+
return row[0] if row else 0
|
|
1834
|
+
|
|
1835
|
+
def list_doc_names_by_source_prefix(self, prefix: str, limit: int) -> list[str]:
|
|
1836
|
+
"""List doc_names whose source_path starts with ``prefix`` (ordered, capped)."""
|
|
1837
|
+
rows = self._conn.execute(
|
|
1838
|
+
"SELECT doc_name FROM documents WHERE instr(source_path, ?) = 1 "
|
|
1839
|
+
"ORDER BY doc_name LIMIT ?",
|
|
1840
|
+
(prefix, limit),
|
|
1841
|
+
).fetchall()
|
|
1842
|
+
return [r[0] for r in rows]
|
|
1843
|
+
|
|
1844
|
+
# -------------------------------------------------------------------
|
|
1845
|
+
# PST 邮件元数据(ADR-0005:邮件列表分页 + 附件下载清单)
|
|
1846
|
+
# -------------------------------------------------------------------
|
|
1847
|
+
|
|
1848
|
+
def upsert_email_meta(self, doc_id: str, pst_path: str, meta: dict) -> None:
|
|
1849
|
+
"""写入一封邮件的元数据(随派生文档同事务批次提交)。
|
|
1850
|
+
|
|
1851
|
+
Args:
|
|
1852
|
+
doc_id: 派生文档 id(``<pst文档id>__<entry_id>``)
|
|
1853
|
+
pst_path: 物理 PST 绝对路径(列表查询的分组键)
|
|
1854
|
+
meta: pst_parser 产出的 email_meta dict
|
|
1855
|
+
"""
|
|
1856
|
+
import json as _json
|
|
1857
|
+
|
|
1858
|
+
self._conn.execute(
|
|
1859
|
+
"""INSERT OR REPLACE INTO pst_email_meta
|
|
1860
|
+
(doc_id, pst_path, entry_id, subject, sender, date, folder, attachments_json)
|
|
1861
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, ?)""",
|
|
1862
|
+
(
|
|
1863
|
+
doc_id,
|
|
1864
|
+
pst_path,
|
|
1865
|
+
str(meta.get("entry_id", "")),
|
|
1866
|
+
meta.get("subject", ""),
|
|
1867
|
+
meta.get("sender", ""),
|
|
1868
|
+
meta.get("date", ""),
|
|
1869
|
+
meta.get("folder", ""),
|
|
1870
|
+
_json.dumps(meta.get("attachments") or [], ensure_ascii=False),
|
|
1871
|
+
),
|
|
1872
|
+
)
|
|
1873
|
+
|
|
1874
|
+
def list_email_meta(
|
|
1875
|
+
self, pst_path: str, offset: int = 0, limit: int = 50
|
|
1876
|
+
) -> tuple[int, list[dict]]:
|
|
1877
|
+
"""分页列出一个 PST 的邮件元数据(日期倒序,无日期排尾)。
|
|
1878
|
+
|
|
1879
|
+
Returns:
|
|
1880
|
+
(total, rows):rows 含 entry_id/subject/sender/date/folder/doc_id。
|
|
1881
|
+
"""
|
|
1882
|
+
total = self._conn.execute(
|
|
1883
|
+
"SELECT COUNT(*) FROM pst_email_meta WHERE pst_path = ?", (pst_path,)
|
|
1884
|
+
).fetchone()[0]
|
|
1885
|
+
rows = self._conn.execute(
|
|
1886
|
+
"""SELECT entry_id, subject, sender, date, folder, doc_id
|
|
1887
|
+
FROM pst_email_meta WHERE pst_path = ?
|
|
1888
|
+
ORDER BY CASE WHEN date = '' THEN 1 ELSE 0 END, date DESC, entry_id DESC
|
|
1889
|
+
LIMIT ? OFFSET ?""",
|
|
1890
|
+
(pst_path, limit, offset),
|
|
1891
|
+
).fetchall()
|
|
1892
|
+
return total, [
|
|
1893
|
+
{
|
|
1894
|
+
"entry_id": r[0],
|
|
1895
|
+
"subject": r[1],
|
|
1896
|
+
"sender": r[2],
|
|
1897
|
+
"date": r[3],
|
|
1898
|
+
"folder": r[4],
|
|
1899
|
+
"doc_id": r[5],
|
|
1900
|
+
}
|
|
1901
|
+
for r in rows
|
|
1902
|
+
]
|
|
1903
|
+
|
|
1904
|
+
def get_email_attachments(self, doc_id: str) -> list[dict] | None:
|
|
1905
|
+
"""按派生 doc_id 取附件清单([{name,size,stored,filename}]);无记录返回 None。"""
|
|
1906
|
+
row = self._conn.execute(
|
|
1907
|
+
"SELECT attachments_json FROM pst_email_meta WHERE doc_id = ?", (doc_id,)
|
|
1908
|
+
).fetchone()
|
|
1909
|
+
return self._attachments_from_row(row)
|
|
1910
|
+
|
|
1911
|
+
def get_email_attachments_by_entry(
|
|
1912
|
+
self, pst_path: str, entry_id: str
|
|
1913
|
+
) -> list[dict] | None:
|
|
1914
|
+
"""按 (PST 绝对路径, entry_id) 取附件清单(web 层不知道 doc_id 时用)。"""
|
|
1915
|
+
row = self._conn.execute(
|
|
1916
|
+
"SELECT attachments_json FROM pst_email_meta WHERE pst_path = ? AND entry_id = ?",
|
|
1917
|
+
(pst_path, str(entry_id)),
|
|
1918
|
+
).fetchone()
|
|
1919
|
+
return self._attachments_from_row(row)
|
|
1920
|
+
|
|
1921
|
+
@staticmethod
|
|
1922
|
+
def _attachments_from_row(row) -> list[dict] | None:
|
|
1923
|
+
import json as _json
|
|
1924
|
+
|
|
1925
|
+
if row is None:
|
|
1926
|
+
return None
|
|
1927
|
+
try:
|
|
1928
|
+
return _json.loads(row[0] or "[]")
|
|
1929
|
+
except _json.JSONDecodeError:
|
|
1930
|
+
return []
|
|
1931
|
+
|
|
1932
|
+
# -------------------------------------------------------------------
|
|
1933
|
+
# Index metadata (replaces _index_meta.json)
|
|
1934
|
+
# -------------------------------------------------------------------
|
|
1935
|
+
|
|
1936
|
+
def get_index_meta(self, source_path: str) -> Optional[str]:
|
|
1937
|
+
"""Get the stored file hash for a source path.
|
|
1938
|
+
|
|
1939
|
+
Returns:
|
|
1940
|
+
File hash string, or None if not tracked.
|
|
1941
|
+
"""
|
|
1942
|
+
row = self._conn.execute(
|
|
1943
|
+
"SELECT file_hash FROM index_meta WHERE source_path = ?",
|
|
1944
|
+
(source_path,),
|
|
1945
|
+
).fetchone()
|
|
1946
|
+
return row[0] if row else None
|
|
1947
|
+
|
|
1948
|
+
def set_index_meta(self, source_path: str, file_hash: str) -> None:
|
|
1949
|
+
"""Store/update the file hash for a source path."""
|
|
1950
|
+
self._conn.execute(
|
|
1951
|
+
"INSERT OR REPLACE INTO index_meta (source_path, file_hash) VALUES (?, ?)",
|
|
1952
|
+
(source_path, file_hash),
|
|
1953
|
+
)
|
|
1954
|
+
self._conn.commit()
|
|
1955
|
+
|
|
1956
|
+
def set_index_meta_batch(self, meta: dict[str, str]) -> None:
|
|
1957
|
+
"""Batch store/update file hashes. Single transaction for performance."""
|
|
1958
|
+
self._conn.executemany(
|
|
1959
|
+
"INSERT OR REPLACE INTO index_meta (source_path, file_hash) VALUES (?, ?)",
|
|
1960
|
+
list(meta.items()),
|
|
1961
|
+
)
|
|
1962
|
+
self._conn.commit()
|
|
1963
|
+
|
|
1964
|
+
def get_all_index_meta(self) -> dict[str, str]:
|
|
1965
|
+
"""Get all stored file hashes.
|
|
1966
|
+
|
|
1967
|
+
Returns:
|
|
1968
|
+
Dict mapping source_path -> file_hash.
|
|
1969
|
+
"""
|
|
1970
|
+
rows = self._conn.execute("SELECT source_path, file_hash FROM index_meta").fetchall()
|
|
1971
|
+
return {r[0]: r[1] for r in rows}
|
|
1972
|
+
|
|
1973
|
+
# -------------------------------------------------------------------
|
|
1974
|
+
# FTS5 query expression builder
|
|
1975
|
+
# -------------------------------------------------------------------
|
|
1976
|
+
|
|
1977
|
+
@staticmethod
|
|
1978
|
+
def build_fts_expression(
|
|
1979
|
+
keywords: list[str],
|
|
1980
|
+
operator: str = "OR",
|
|
1981
|
+
column: Optional[str] = None,
|
|
1982
|
+
near_distance: Optional[int] = None,
|
|
1983
|
+
prefix: bool = False,
|
|
1984
|
+
) -> str:
|
|
1985
|
+
"""Build FTS5 match expression from keyword list.
|
|
1986
|
+
|
|
1987
|
+
Args:
|
|
1988
|
+
keywords: list of search terms
|
|
1989
|
+
operator: "AND" | "OR" | "NOT" (first keyword AND NOT others)
|
|
1990
|
+
column: optional column filter (e.g. "title", "body")
|
|
1991
|
+
near_distance: if set, uses NEAR(kw1 kw2, N) syntax
|
|
1992
|
+
prefix: if True, append ``*`` to each term for prefix matching.
|
|
1993
|
+
Prefix matching allows partial token matches: "conf*" matches
|
|
1994
|
+
"config", "configuration", "confirm", etc. Useful for
|
|
1995
|
+
autocomplete-style search or when query tokens may be substrings
|
|
1996
|
+
of indexed terms (e.g. "fts*" matches both "fts" and "fts5").
|
|
1997
|
+
|
|
1998
|
+
Returns:
|
|
1999
|
+
FTS5 match expression string
|
|
2000
|
+
|
|
2001
|
+
Examples::
|
|
2002
|
+
|
|
2003
|
+
build_fts_expression(["python", "async"], "AND")
|
|
2004
|
+
-> "python AND async"
|
|
2005
|
+
|
|
2006
|
+
build_fts_expression(["fts"], prefix=True)
|
|
2007
|
+
-> "fts*" # matches fts, fts5, ftsearch, ...
|
|
2008
|
+
|
|
2009
|
+
build_fts_expression(["fts", "python"], prefix=True, operator="OR")
|
|
2010
|
+
-> "fts* OR python*"
|
|
2011
|
+
|
|
2012
|
+
build_fts_expression(["machine", "learning"], column="title")
|
|
2013
|
+
-> "title : (machine OR learning)"
|
|
2014
|
+
|
|
2015
|
+
build_fts_expression(["deep", "learning"], near_distance=5)
|
|
2016
|
+
-> 'NEAR(deep learning, 5)'
|
|
2017
|
+
"""
|
|
2018
|
+
if not keywords:
|
|
2019
|
+
return ""
|
|
2020
|
+
|
|
2021
|
+
# Escape FTS5 special characters in keywords
|
|
2022
|
+
safe_kws = []
|
|
2023
|
+
for kw in keywords:
|
|
2024
|
+
cleaned = _RE_FTS5_SPECIAL.sub("", kw.strip())
|
|
2025
|
+
if cleaned:
|
|
2026
|
+
# Tokenize for CJK
|
|
2027
|
+
tokenized = _tokenize_for_fts(cleaned)
|
|
2028
|
+
if tokenized.strip():
|
|
2029
|
+
term = tokenized.strip()
|
|
2030
|
+
if prefix:
|
|
2031
|
+
term = term + "*"
|
|
2032
|
+
safe_kws.append(term)
|
|
2033
|
+
|
|
2034
|
+
if not safe_kws:
|
|
2035
|
+
return ""
|
|
2036
|
+
|
|
2037
|
+
if near_distance is not None and len(safe_kws) >= 2:
|
|
2038
|
+
all_tokens = " ".join(safe_kws)
|
|
2039
|
+
expr = f"NEAR({all_tokens}, {near_distance})"
|
|
2040
|
+
elif operator == "NOT" and len(safe_kws) >= 2:
|
|
2041
|
+
expr = f"{safe_kws[0]} NOT {' NOT '.join(safe_kws[1:])}"
|
|
2042
|
+
else:
|
|
2043
|
+
expr = f" {operator} ".join(safe_kws)
|
|
2044
|
+
|
|
2045
|
+
if column:
|
|
2046
|
+
expr = f"{column} : ({expr})"
|
|
2047
|
+
|
|
2048
|
+
return expr
|
|
2049
|
+
|
|
2050
|
+
# -------------------------------------------------------------------
|
|
2051
|
+
# Maintenance
|
|
2052
|
+
# -------------------------------------------------------------------
|
|
2053
|
+
|
|
2054
|
+
def optimize(self) -> None:
|
|
2055
|
+
"""Run FTS5 merge optimization for better query performance."""
|
|
2056
|
+
if not self._use_fts5:
|
|
2057
|
+
logger.debug("FTS5 not available, skipping optimize")
|
|
2058
|
+
return
|
|
2059
|
+
try:
|
|
2060
|
+
self._conn.execute("INSERT INTO fts_nodes(fts_nodes) VALUES('optimize')")
|
|
2061
|
+
self._conn.commit()
|
|
2062
|
+
logger.info("FTS5 index optimized")
|
|
2063
|
+
except sqlite3.OperationalError as e:
|
|
2064
|
+
logger.warning("FTS5 optimize failed: %s", e)
|
|
2065
|
+
|
|
2066
|
+
def rebuild(self) -> None:
|
|
2067
|
+
"""Rebuild FTS5 index from scratch."""
|
|
2068
|
+
if not self._use_fts5:
|
|
2069
|
+
logger.debug("FTS5 not available, skipping rebuild")
|
|
2070
|
+
return
|
|
2071
|
+
try:
|
|
2072
|
+
self._conn.execute("INSERT INTO fts_nodes(fts_nodes) VALUES('rebuild')")
|
|
2073
|
+
self._conn.commit()
|
|
2074
|
+
logger.info("FTS5 index rebuilt")
|
|
2075
|
+
except sqlite3.OperationalError as e:
|
|
2076
|
+
logger.warning("FTS5 rebuild failed: %s", e)
|
|
2077
|
+
|
|
2078
|
+
def get_stats(self) -> dict:
|
|
2079
|
+
"""Get index statistics."""
|
|
2080
|
+
doc_count = self._conn.execute("SELECT COUNT(*) FROM documents").fetchone()[0]
|
|
2081
|
+
node_count = self._conn.execute("SELECT COUNT(*) FROM nodes").fetchone()[0]
|
|
2082
|
+
return {
|
|
2083
|
+
"db_path": self._db_path,
|
|
2084
|
+
"document_count": doc_count,
|
|
2085
|
+
"node_count": node_count,
|
|
2086
|
+
}
|
|
2087
|
+
|
|
2088
|
+
def clear(self) -> None:
|
|
2089
|
+
"""Clear all indexed data."""
|
|
2090
|
+
self._conn.execute("DELETE FROM fts_nodes")
|
|
2091
|
+
self._conn.execute("DELETE FROM nodes")
|
|
2092
|
+
self._conn.execute("DELETE FROM documents")
|
|
2093
|
+
self._conn.execute("DELETE FROM index_meta")
|
|
2094
|
+
self._conn.commit()
|
|
2095
|
+
|
|
2096
|
+
def is_document_indexed(self, doc_id: str) -> bool:
|
|
2097
|
+
"""Check if a document is already indexed."""
|
|
2098
|
+
row = self._conn.execute(
|
|
2099
|
+
"SELECT 1 FROM documents WHERE doc_id = ?", (doc_id,)
|
|
2100
|
+
).fetchone()
|
|
2101
|
+
return row is not None
|
|
2102
|
+
|
|
2103
|
+
def wal_checkpoint(self, mode: str = "TRUNCATE") -> None:
|
|
2104
|
+
"""Force a WAL checkpoint to fold the ``-wal`` sidecar back into the DB.
|
|
2105
|
+
|
|
2106
|
+
Useful at the end of a long incremental indexing run so the ``-wal``
|
|
2107
|
+
file does not stay multi-GB after many small commits. ``TRUNCATE``
|
|
2108
|
+
is the strongest variant (always safe — only no-ops when readers hold
|
|
2109
|
+
the WAL open).
|
|
2110
|
+
"""
|
|
2111
|
+
try:
|
|
2112
|
+
self._conn.execute(f"PRAGMA wal_checkpoint({mode})")
|
|
2113
|
+
except sqlite3.OperationalError as e:
|
|
2114
|
+
logger.debug("WAL checkpoint(%s) failed: %s", mode, e)
|
|
2115
|
+
|
|
2116
|
+
def verify_index(self) -> dict:
|
|
2117
|
+
"""Cross-table consistency check.
|
|
2118
|
+
|
|
2119
|
+
Detects orphan rows that violate the four-table invariant:
|
|
2120
|
+
- ``nodes`` rows whose doc_id has no entry in ``documents``.
|
|
2121
|
+
- ``fts_nodes`` rows whose doc_id has no entry in ``documents``.
|
|
2122
|
+
- ``index_meta`` rows whose source_path has no entry in ``documents``.
|
|
2123
|
+
- ``documents`` rows with non-existent on-disk source_path.
|
|
2124
|
+
|
|
2125
|
+
Returns a report dict with ``healthy: bool`` and lists of problem ids.
|
|
2126
|
+
Use :meth:`repair_index` to clean orphans.
|
|
2127
|
+
"""
|
|
2128
|
+
report: dict = {
|
|
2129
|
+
"healthy": True,
|
|
2130
|
+
"orphan_node_doc_ids": [],
|
|
2131
|
+
"orphan_fts_doc_ids": [],
|
|
2132
|
+
"orphan_meta_paths": [],
|
|
2133
|
+
"missing_source_paths": [],
|
|
2134
|
+
}
|
|
2135
|
+
cur = self._conn.execute(
|
|
2136
|
+
"SELECT DISTINCT n.doc_id FROM nodes n "
|
|
2137
|
+
"LEFT JOIN documents d ON n.doc_id = d.doc_id WHERE d.doc_id IS NULL"
|
|
2138
|
+
).fetchall()
|
|
2139
|
+
report["orphan_node_doc_ids"] = [r[0] for r in cur]
|
|
2140
|
+
cur = self._conn.execute(
|
|
2141
|
+
"SELECT DISTINCT f.doc_id FROM fts_nodes f "
|
|
2142
|
+
"LEFT JOIN documents d ON f.doc_id = d.doc_id WHERE d.doc_id IS NULL"
|
|
2143
|
+
).fetchall()
|
|
2144
|
+
report["orphan_fts_doc_ids"] = [r[0] for r in cur]
|
|
2145
|
+
cur = self._conn.execute(
|
|
2146
|
+
"SELECT m.source_path FROM index_meta m "
|
|
2147
|
+
"LEFT JOIN documents d ON m.source_path = d.source_path WHERE d.doc_id IS NULL"
|
|
2148
|
+
).fetchall()
|
|
2149
|
+
report["orphan_meta_paths"] = [r[0] for r in cur]
|
|
2150
|
+
cur = self._conn.execute(
|
|
2151
|
+
"SELECT doc_id, source_path FROM documents WHERE source_path != ''"
|
|
2152
|
+
).fetchall()
|
|
2153
|
+
report["missing_source_paths"] = [
|
|
2154
|
+
(doc_id, sp) for (doc_id, sp) in cur if not os.path.isfile(sp)
|
|
2155
|
+
]
|
|
2156
|
+
report["healthy"] = not any(
|
|
2157
|
+
report[k] for k in
|
|
2158
|
+
("orphan_node_doc_ids", "orphan_fts_doc_ids",
|
|
2159
|
+
"orphan_meta_paths", "missing_source_paths")
|
|
2160
|
+
)
|
|
2161
|
+
return report
|
|
2162
|
+
|
|
2163
|
+
def repair_index(self, drop_missing_files: bool = False) -> dict:
|
|
2164
|
+
"""Remove orphan rows surfaced by :meth:`verify_index`.
|
|
2165
|
+
|
|
2166
|
+
Args:
|
|
2167
|
+
drop_missing_files: also delete documents whose source file no
|
|
2168
|
+
longer exists on disk (default False — those may be intentional
|
|
2169
|
+
in-memory loads or inaccessible mounts).
|
|
2170
|
+
|
|
2171
|
+
Returns:
|
|
2172
|
+
Dict with counts of removed orphan rows.
|
|
2173
|
+
"""
|
|
2174
|
+
report = self.verify_index()
|
|
2175
|
+
removed = {"orphan_nodes": 0, "orphan_fts": 0, "orphan_meta": 0, "missing_files": 0}
|
|
2176
|
+
|
|
2177
|
+
with self._conn:
|
|
2178
|
+
if report["orphan_node_doc_ids"]:
|
|
2179
|
+
ph = ",".join("?" for _ in report["orphan_node_doc_ids"])
|
|
2180
|
+
cur = self._conn.execute(
|
|
2181
|
+
f"DELETE FROM nodes WHERE doc_id IN ({ph})",
|
|
2182
|
+
report["orphan_node_doc_ids"],
|
|
2183
|
+
)
|
|
2184
|
+
removed["orphan_nodes"] = cur.rowcount
|
|
2185
|
+
if report["orphan_fts_doc_ids"]:
|
|
2186
|
+
ph = ",".join("?" for _ in report["orphan_fts_doc_ids"])
|
|
2187
|
+
if self._use_fts5:
|
|
2188
|
+
rowids = self._conn.execute(
|
|
2189
|
+
f"SELECT rowid FROM fts_nodes WHERE doc_id IN ({ph})",
|
|
2190
|
+
report["orphan_fts_doc_ids"],
|
|
2191
|
+
).fetchall()
|
|
2192
|
+
if rowids:
|
|
2193
|
+
ph2 = ",".join("?" for _ in rowids)
|
|
2194
|
+
cur = self._conn.execute(
|
|
2195
|
+
f"DELETE FROM fts_nodes WHERE rowid IN ({ph2})",
|
|
2196
|
+
[r[0] for r in rowids],
|
|
2197
|
+
)
|
|
2198
|
+
removed["orphan_fts"] = cur.rowcount
|
|
2199
|
+
else:
|
|
2200
|
+
cur = self._conn.execute(
|
|
2201
|
+
f"DELETE FROM fts_nodes WHERE doc_id IN ({ph})",
|
|
2202
|
+
report["orphan_fts_doc_ids"],
|
|
2203
|
+
)
|
|
2204
|
+
removed["orphan_fts"] = cur.rowcount
|
|
2205
|
+
if report["orphan_meta_paths"]:
|
|
2206
|
+
ph = ",".join("?" for _ in report["orphan_meta_paths"])
|
|
2207
|
+
cur = self._conn.execute(
|
|
2208
|
+
f"DELETE FROM index_meta WHERE source_path IN ({ph})",
|
|
2209
|
+
report["orphan_meta_paths"],
|
|
2210
|
+
)
|
|
2211
|
+
removed["orphan_meta"] = cur.rowcount
|
|
2212
|
+
|
|
2213
|
+
if drop_missing_files and report["missing_source_paths"]:
|
|
2214
|
+
doc_ids = [doc_id for (doc_id, _) in report["missing_source_paths"]]
|
|
2215
|
+
removed["missing_files"] = self.delete_documents(doc_ids)
|
|
2216
|
+
|
|
2217
|
+
return removed
|
|
2218
|
+
|
|
2219
|
+
def get_unindexed_doc_ids(self, doc_ids: list[str]) -> set[str]:
|
|
2220
|
+
"""Return the subset of doc_ids that are NOT yet indexed.
|
|
2221
|
+
|
|
2222
|
+
Uses a single SQL query instead of per-document checks.
|
|
2223
|
+
"""
|
|
2224
|
+
if not doc_ids:
|
|
2225
|
+
return set()
|
|
2226
|
+
placeholders = ",".join("?" for _ in doc_ids)
|
|
2227
|
+
rows = self._conn.execute(
|
|
2228
|
+
f"SELECT doc_id FROM documents WHERE doc_id IN ({placeholders})",
|
|
2229
|
+
doc_ids,
|
|
2230
|
+
).fetchall()
|
|
2231
|
+
indexed = {r[0] for r in rows}
|
|
2232
|
+
return set(doc_ids) - indexed
|
|
2233
|
+
|
|
2234
|
+
|
|
2235
|
+
def _extract_match_snippet(text: str, query: str, use_regex: bool, size: int = 300) -> str:
|
|
2236
|
+
"""Extract a snippet of *size* chars centered around the first match."""
|
|
2237
|
+
if len(text) <= size:
|
|
2238
|
+
return text
|
|
2239
|
+
pos = -1
|
|
2240
|
+
if use_regex:
|
|
2241
|
+
m = re.search(query, text, re.IGNORECASE)
|
|
2242
|
+
if m:
|
|
2243
|
+
pos = m.start()
|
|
2244
|
+
else:
|
|
2245
|
+
pos = text.lower().find(query.lower())
|
|
2246
|
+
if pos < 0:
|
|
2247
|
+
pos = 0
|
|
2248
|
+
half = size // 2
|
|
2249
|
+
start = max(0, pos - half)
|
|
2250
|
+
end = min(len(text), start + size)
|
|
2251
|
+
start = max(0, end - size)
|
|
2252
|
+
return text[start:end]
|
|
2253
|
+
|
|
2254
|
+
|
|
2255
|
+
# ---------------------------------------------------------------------------
|
|
2256
|
+
# Global FTS5 index singleton
|
|
2257
|
+
# ---------------------------------------------------------------------------
|
|
2258
|
+
|
|
2259
|
+
_global_fts: Optional[FTS5Index] = None
|
|
2260
|
+
|
|
2261
|
+
|
|
2262
|
+
def get_fts_index(db_path: Optional[str] = None, weights: Optional[dict] = None) -> FTS5Index:
|
|
2263
|
+
"""Get or create the global FTS5 index.
|
|
2264
|
+
|
|
2265
|
+
Args:
|
|
2266
|
+
db_path: database path. If None, uses in-memory database.
|
|
2267
|
+
Pass a file path for persistent indexing across sessions.
|
|
2268
|
+
weights: column weight overrides for bm25() ranking.
|
|
2269
|
+
"""
|
|
2270
|
+
global _global_fts
|
|
2271
|
+
if _global_fts is not None:
|
|
2272
|
+
# If db_path changed, re-create the singleton
|
|
2273
|
+
requested = db_path or ":memory:"
|
|
2274
|
+
if _global_fts.db_path != requested:
|
|
2275
|
+
_global_fts.close()
|
|
2276
|
+
_global_fts = None
|
|
2277
|
+
if _global_fts is None:
|
|
2278
|
+
_global_fts = FTS5Index(db_path=db_path, weights=weights)
|
|
2279
|
+
return _global_fts
|
|
2280
|
+
|
|
2281
|
+
|
|
2282
|
+
def set_fts_index(index: FTS5Index) -> None:
|
|
2283
|
+
"""Set the global FTS5 index instance."""
|
|
2284
|
+
global _global_fts
|
|
2285
|
+
_global_fts = index
|
|
2286
|
+
|
|
2287
|
+
|
|
2288
|
+
def reset_fts_index() -> None:
|
|
2289
|
+
"""Close and reset the global FTS5 index."""
|
|
2290
|
+
global _global_fts
|
|
2291
|
+
if _global_fts is not None:
|
|
2292
|
+
_global_fts.close()
|
|
2293
|
+
_global_fts = None
|