treesearchlib 1.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. treesearch/__init__.py +53 -0
  2. treesearch/__main__.py +6 -0
  3. treesearch/_bin/pst-extract.exe +0 -0
  4. treesearch/cli.py +554 -0
  5. treesearch/config.py +206 -0
  6. treesearch/fts.py +2293 -0
  7. treesearch/heuristics.py +425 -0
  8. treesearch/indexer.py +2038 -0
  9. treesearch/parsers/__init__.py +62 -0
  10. treesearch/parsers/anydoc_parser.py +193 -0
  11. treesearch/parsers/ast_parser.py +136 -0
  12. treesearch/parsers/docx_parser.py +304 -0
  13. treesearch/parsers/email_html_md.py +60 -0
  14. treesearch/parsers/excel_parser.py +218 -0
  15. treesearch/parsers/html_parser.py +172 -0
  16. treesearch/parsers/image_metadata.py +345 -0
  17. treesearch/parsers/image_parser.py +59 -0
  18. treesearch/parsers/image_store.py +182 -0
  19. treesearch/parsers/markitdown_parser.py +258 -0
  20. treesearch/parsers/mhtml_parser.py +108 -0
  21. treesearch/parsers/pdf_parser.py +409 -0
  22. treesearch/parsers/pst_attachment_store.py +156 -0
  23. treesearch/parsers/pst_parser.py +733 -0
  24. treesearch/parsers/registry.py +405 -0
  25. treesearch/parsers/treesitter_parser.py +433 -0
  26. treesearch/pathutil.py +227 -0
  27. treesearch/py.typed +0 -0
  28. treesearch/ripgrep.py +159 -0
  29. treesearch/search.py +935 -0
  30. treesearch/tokenizer.py +176 -0
  31. treesearch/tree.py +393 -0
  32. treesearch/tree_searcher.py +1006 -0
  33. treesearch/treesearch.py +574 -0
  34. treesearch/watch.py +305 -0
  35. treesearchlib-1.1.0.dist-info/METADATA +124 -0
  36. treesearchlib-1.1.0.dist-info/RECORD +39 -0
  37. treesearchlib-1.1.0.dist-info/WHEEL +5 -0
  38. treesearchlib-1.1.0.dist-info/entry_points.txt +2 -0
  39. treesearchlib-1.1.0.dist-info/top_level.txt +1 -0
treesearch/fts.py ADDED
@@ -0,0 +1,2293 @@
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ @author:XuMing(xuming624@qq.com)
4
+ @description: SQLite FTS5 full-text search engine for tree-structured documents.
5
+
6
+ Single-file storage: tree structures, FTS5 indexes, and incremental metadata
7
+ are all stored in one SQLite database (.db file).
8
+
9
+ Architecture: "SQLite FTS5 + Producer-Consumer"
10
+ - Deferred indexing via WAL mode solves real-time freshness
11
+ - Local SQL execution handles aggregation needs
12
+ - FTS5 inverted index guarantees retrieval performance
13
+ - Tree structure persistence in `documents.structure_json` column
14
+
15
+ Key features:
16
+ - WAL mode for concurrent read/write
17
+ - Lazy indexing: nodes are inserted on demand, not precomputed
18
+ - MD-aware schema: front_matter, title, summary, body, code_blocks
19
+ - CJK-aware tokenizer: jieba segmentation only when Chinese text is detected
20
+ - Implements PreFilter protocol for seamless integration with search()
21
+ - Hierarchical field boosting via FTS5 column weighting
22
+ """
23
+ import hashlib
24
+ import json
25
+ import logging
26
+ import os
27
+ import re
28
+ import sqlite3
29
+ from typing import Optional
30
+
31
+ logger = logging.getLogger(__name__)
32
+
33
+ # FTS5 column weights: title > body > summary > code
34
+ # Used in bm25() ranking function
35
+ _DEFAULT_WEIGHTS = {
36
+ "title": 5.0,
37
+ "summary": 2.0,
38
+ "body": 10.0,
39
+ "code_blocks": 1.0,
40
+ "front_matter": 2.0,
41
+ }
42
+
43
+ # ---------------------------------------------------------------------------
44
+ # FTS5 availability detection
45
+ # ---------------------------------------------------------------------------
46
+
47
+ _FTS5_AVAILABLE: Optional[bool] = None
48
+
49
+
50
+ class _NullContext:
51
+ """No-op transaction context for callers managing commits externally."""
52
+ def __enter__(self):
53
+ return self
54
+ def __exit__(self, *exc):
55
+ return False
56
+
57
+
58
+ def _check_fts5() -> bool:
59
+ """Check whether the current SQLite build includes the FTS5 extension."""
60
+ global _FTS5_AVAILABLE
61
+ if _FTS5_AVAILABLE is not None:
62
+ return _FTS5_AVAILABLE
63
+ try:
64
+ conn = sqlite3.connect(":memory:")
65
+ conn.execute("CREATE VIRTUAL TABLE _fts5_test USING fts5(x)")
66
+ conn.execute("DROP TABLE _fts5_test")
67
+ conn.close()
68
+ _FTS5_AVAILABLE = True
69
+ except sqlite3.OperationalError:
70
+ _FTS5_AVAILABLE = False
71
+ logger.warning(
72
+ "SQLite FTS5 extension is not available (SQLite version: %s). "
73
+ "Falling back to plain-text LIKE search. Performance and ranking "
74
+ "quality will be reduced. To fix: upgrade SQLite or install a "
75
+ "Python build that includes FTS5.",
76
+ sqlite3.sqlite_version,
77
+ )
78
+ return _FTS5_AVAILABLE
79
+
80
+
81
+ def _sqlite_regexp(pattern: str, string: str) -> bool:
82
+ """SQLite REGEXP callback — registered via create_function."""
83
+ if string is None or pattern is None:
84
+ return False
85
+ try:
86
+ return bool(re.search(pattern, string))
87
+ except re.error:
88
+ return False
89
+
90
+
91
+ # ---------------------------------------------------------------------------
92
+ # Markdown structure parser
93
+ # ---------------------------------------------------------------------------
94
+
95
+ _RE_FRONT_MATTER = re.compile(r"^---\s*\n(.*?\n)---\s*\n", re.DOTALL)
96
+ _RE_CODE_BLOCK = re.compile(r"```[\w]*\n(.*?)```", re.DOTALL)
97
+ _RE_HEADING_LINE = re.compile(r"^#{1,6}\s+")
98
+
99
+
100
+ def parse_md_node_text(text: str) -> dict:
101
+ """Parse a node's text into MD-aware structured fields.
102
+
103
+ Returns:
104
+ {
105
+ "front_matter": str, # YAML front matter (if present)
106
+ "body": str, # main text (headings, paragraphs)
107
+ "code_blocks": str, # concatenated code blocks
108
+ }
109
+ """
110
+ if not text:
111
+ return {"front_matter": "", "body": "", "code_blocks": ""}
112
+
113
+ front_matter = ""
114
+ remaining = text
115
+
116
+ # Extract front matter
117
+ fm_match = _RE_FRONT_MATTER.match(text)
118
+ if fm_match:
119
+ front_matter = fm_match.group(1).strip()
120
+ remaining = text[fm_match.end():]
121
+
122
+ # Extract code blocks
123
+ code_parts = []
124
+ def _replace_code(m):
125
+ code_parts.append(m.group(1).strip())
126
+ return "" # remove from body
127
+ body = _RE_CODE_BLOCK.sub(_replace_code, remaining)
128
+ code_blocks = "\n".join(code_parts)
129
+
130
+ # Clean body: collapse blank lines
131
+ body = re.sub(r"\n{3,}", "\n\n", body).strip()
132
+
133
+ return {
134
+ "front_matter": front_matter,
135
+ "body": body,
136
+ "code_blocks": code_blocks,
137
+ }
138
+
139
+
140
+ # ---------------------------------------------------------------------------
141
+ # Tokenizer for FTS5 (Chinese/English)
142
+ # ---------------------------------------------------------------------------
143
+
144
+ from functools import lru_cache
145
+
146
+ from .tokenizer import _RE_HAS_CJK
147
+
148
+
149
+ @lru_cache(maxsize=4096)
150
+ def _tokenize_for_fts(text: str) -> str:
151
+ """Tokenize text for FTS5 indexing. Space-separated tokens.
152
+
153
+ Only uses jieba segmentation when Chinese (CJK) characters are detected.
154
+ For pure English/non-CJK text, relies on FTS5's built-in unicode61 tokenizer
155
+ (no jieba overhead).
156
+
157
+ Results are LRU-cached to avoid repeated jieba overhead for the same text.
158
+ """
159
+ if not text or not text.strip():
160
+ return ""
161
+ if _RE_HAS_CJK.search(text):
162
+ from .tokenizer import tokenize
163
+ tokens = tokenize(text)
164
+ return " ".join(tokens)
165
+ # English / non-CJK: return as-is, FTS5 unicode61 handles tokenization
166
+ return text
167
+
168
+
169
+ # FTS5 operators that should NOT be tokenized
170
+ _FTS5_OPERATORS = {"AND", "OR", "NOT", "NEAR"}
171
+
172
+ # Characters that should be stripped from FTS5 query tokens.
173
+ # Keep only word characters (letters, digits, underscore) and CJK ranges.
174
+ _RE_FTS5_SPECIAL = re.compile(r'[^\w\u4e00-\u9fff\u3400-\u4dbf]')
175
+
176
+
177
+ def _tokenize_fts_expression(expr: str) -> str:
178
+ """Tokenize terms in an FTS5 expression while preserving operators.
179
+
180
+ Raw FTS5 expressions like ``"machine AND learning"`` must have their
181
+ terms tokenized to match the indexed content, but FTS5
182
+ operators (AND, OR, NOT, NEAR) must remain untouched.
183
+
184
+ Handles:
185
+ - FTS5 keyword operators: AND, OR, NOT, NEAR
186
+ - Unary NOT prefix: ``-term`` → ``-tokenized``
187
+ - Quoted phrases: ``"exact phrase"`` → ``"tokenized phrase"``
188
+
189
+ Only applies jieba segmentation when CJK characters are detected.
190
+ """
191
+ # Split while preserving quoted segments as single units
192
+ tokens: list[str] = []
193
+ i = 0
194
+ while i < len(expr):
195
+ if expr[i] == '"':
196
+ end = expr.find('"', i + 1)
197
+ if end == -1:
198
+ end = len(expr)
199
+ tokens.append(expr[i:end + 1])
200
+ i = end + 1
201
+ elif expr[i].isspace():
202
+ i += 1
203
+ else:
204
+ end = i
205
+ while end < len(expr) and not expr[end].isspace() and expr[end] != '"':
206
+ end += 1
207
+ tokens.append(expr[i:end])
208
+ i = end
209
+
210
+ result = []
211
+ for part in tokens:
212
+ upper = part.upper()
213
+ if upper in _FTS5_OPERATORS:
214
+ result.append(upper)
215
+ elif part.startswith('"') and part.endswith('"'):
216
+ # Phrase query: "exact phrase" → tokenize inner content, keep quotes
217
+ inner = part[1:-1]
218
+ tokenized = _tokenize_for_fts(inner)
219
+ if tokenized.strip():
220
+ result.append(f'"{tokenized.strip()}"')
221
+ elif part.startswith('-'):
222
+ # Unary NOT: -term → -tokenized
223
+ tokenized = _tokenize_for_fts(part[1:])
224
+ if tokenized.strip():
225
+ result.append(f"-{tokenized.strip()}")
226
+ else:
227
+ tokenized = _tokenize_for_fts(part)
228
+ if tokenized.strip():
229
+ result.append(tokenized.strip())
230
+ return " ".join(result)
231
+
232
+
233
+ # ---------------------------------------------------------------------------
234
+ # FTS5 Index Engine
235
+ # ---------------------------------------------------------------------------
236
+
237
+ class FTS5Index:
238
+ """SQLite FTS5 full-text search index for tree-structured documents.
239
+
240
+ Features:
241
+ - WAL journal mode for concurrent read/write
242
+ - MD-aware columns: title, summary, body, code_blocks, front_matter
243
+ - Hierarchical column weighting via bm25() rank function
244
+ - Deferred indexing: call index_document() when ready
245
+ - Implements PreFilter protocol: score_nodes(query, doc_id)
246
+ - Supports FTS5 query syntax: AND, OR, NOT, NEAR, phrase "..."
247
+
248
+ In-memory mode (``db_path=None``):
249
+ All indexes are kept in SQLite ``:memory:`` — no file is written to disk.
250
+ Performance is excellent even with thousands of documents (5,000 docs < 10ms).
251
+ Indexes are lost when the process exits or the instance is closed.
252
+ """
253
+
254
+ def __init__(
255
+ self,
256
+ db_path: Optional[str] = None,
257
+ weights: Optional[dict] = None,
258
+ tokenize_log_path: Optional[str] = None,
259
+ ):
260
+ """
261
+ Args:
262
+ db_path: Path to SQLite database file. ``None`` for in-memory mode
263
+ (no file written to disk). Default: ``None``.
264
+ weights: column weight overrides for bm25() ranking.
265
+ tokenize_log_path: Optional path to write per-document tokenize logs.
266
+ When set, each call to ``index_document()`` appends deduplicated
267
+ token lists to this file.
268
+ """
269
+ self._db_path = db_path or ":memory:"
270
+ self._weights = {**_DEFAULT_WEIGHTS, **(weights or {})}
271
+ self._conn: Optional[sqlite3.Connection] = None
272
+ # Populated by index_document(); inspected by build_index for stats.
273
+ self.last_node_diff: dict[str, int] = {"added": 0, "changed": 0, "removed": 0, "kept": 0}
274
+ self._tokenize_log_path = tokenize_log_path
275
+ self._tokenize_log_file = None # lazy-opened in _log_tokenize()
276
+ try:
277
+ self._init_db()
278
+ except Exception:
279
+ self.close()
280
+ raise
281
+
282
+ def _init_db(self) -> None:
283
+ """Initialize SQLite database with FTS5 virtual table (or fallback plain table)."""
284
+ if self._db_path != ":memory:":
285
+ os.makedirs(os.path.dirname(os.path.abspath(self._db_path)), exist_ok=True)
286
+
287
+ self._conn = sqlite3.connect(self._db_path, check_same_thread=False)
288
+ self._conn.execute("PRAGMA journal_mode=WAL")
289
+ self._conn.execute("PRAGMA synchronous=NORMAL")
290
+
291
+ # Register REGEXP function for like_search(use_regex=True)
292
+ self._conn.create_function("REGEXP", 2, _sqlite_regexp)
293
+
294
+ self._use_fts5 = _check_fts5()
295
+
296
+ # Metadata table for nodes (structured fields for filtering)
297
+ self._conn.execute("""
298
+ CREATE TABLE IF NOT EXISTS nodes (
299
+ node_id TEXT NOT NULL,
300
+ doc_id TEXT NOT NULL,
301
+ title TEXT DEFAULT '',
302
+ summary TEXT DEFAULT '',
303
+ depth INTEGER DEFAULT 0,
304
+ line_start INTEGER,
305
+ line_end INTEGER,
306
+ parent_node_id TEXT,
307
+ content_hash TEXT,
308
+ PRIMARY KEY (doc_id, node_id)
309
+ )
310
+ """)
311
+
312
+ if self._use_fts5:
313
+ # FTS5 virtual table with content sync
314
+ # tokenize='unicode61' handles basic multi-language, but we pre-tokenize
315
+ # Chinese text with jieba and store space-separated tokens
316
+ self._conn.execute("""
317
+ CREATE VIRTUAL TABLE IF NOT EXISTS fts_nodes USING fts5(
318
+ node_id UNINDEXED,
319
+ doc_id UNINDEXED,
320
+ title,
321
+ summary,
322
+ body,
323
+ code_blocks,
324
+ front_matter,
325
+ tokenize='unicode61 remove_diacritics 2'
326
+ )
327
+ """)
328
+ else:
329
+ # Fallback: plain table with same columns for LIKE-based search
330
+ self._conn.execute("""
331
+ CREATE TABLE IF NOT EXISTS fts_nodes (
332
+ rowid INTEGER PRIMARY KEY AUTOINCREMENT,
333
+ node_id TEXT NOT NULL,
334
+ doc_id TEXT NOT NULL,
335
+ title TEXT DEFAULT '',
336
+ summary TEXT DEFAULT '',
337
+ body TEXT DEFAULT '',
338
+ code_blocks TEXT DEFAULT '',
339
+ front_matter TEXT DEFAULT ''
340
+ )
341
+ """)
342
+ self._conn.execute(
343
+ "CREATE INDEX IF NOT EXISTS idx_fts_nodes_doc_id ON fts_nodes (doc_id)"
344
+ )
345
+
346
+ # Document metadata table (also stores tree structure for persistence)
347
+ self._conn.execute("""
348
+ CREATE TABLE IF NOT EXISTS documents (
349
+ doc_id TEXT PRIMARY KEY,
350
+ doc_name TEXT DEFAULT '',
351
+ doc_description TEXT DEFAULT '',
352
+ source_path TEXT DEFAULT '',
353
+ source_type TEXT DEFAULT '',
354
+ structure_json TEXT DEFAULT '',
355
+ node_count INTEGER DEFAULT 0,
356
+ index_hash TEXT
357
+ )
358
+ """)
359
+
360
+ # Incremental index metadata (replaces _index_meta.json)
361
+ self._conn.execute("""
362
+ CREATE TABLE IF NOT EXISTS index_meta (
363
+ source_path TEXT PRIMARY KEY,
364
+ file_hash TEXT NOT NULL
365
+ )
366
+ """)
367
+
368
+ # Failed files tracking (auto-skip after consecutive parse failures)
369
+ self._conn.execute("""
370
+ CREATE TABLE IF NOT EXISTS failed_files (
371
+ source_path TEXT PRIMARY KEY,
372
+ fail_count INTEGER NOT NULL DEFAULT 1,
373
+ last_error TEXT DEFAULT '',
374
+ last_fail_time REAL NOT NULL,
375
+ file_hash TEXT DEFAULT ''
376
+ )
377
+ """)
378
+
379
+ # Vision parse queue (image files: placeholder first, background vision
380
+ # worker consumes serially and replaces the placeholder in-place).
381
+ # status: pending | processing | done | failed
382
+ self._conn.execute("""
383
+ CREATE TABLE IF NOT EXISTS vision_queue (
384
+ source_path TEXT PRIMARY KEY,
385
+ rel_path TEXT DEFAULT '',
386
+ status TEXT NOT NULL DEFAULT 'pending',
387
+ attempts INTEGER NOT NULL DEFAULT 0,
388
+ model TEXT DEFAULT '',
389
+ last_error TEXT DEFAULT '',
390
+ updated_at REAL NOT NULL
391
+ )
392
+ """)
393
+
394
+ # PST 邮件元数据(ADR-0005):每封派生邮件文档一行,供物理 PST 的
395
+ # 邮件列表分页查询(主题/发件人/日期/文件夹 + 附件下载清单)。
396
+ self._conn.execute("""
397
+ CREATE TABLE IF NOT EXISTS pst_email_meta (
398
+ doc_id TEXT PRIMARY KEY,
399
+ pst_path TEXT NOT NULL,
400
+ entry_id TEXT NOT NULL,
401
+ subject TEXT DEFAULT '',
402
+ sender TEXT DEFAULT '',
403
+ date TEXT DEFAULT '',
404
+ folder TEXT DEFAULT '',
405
+ attachments_json TEXT DEFAULT '[]'
406
+ )
407
+ """)
408
+ self._conn.execute(
409
+ "CREATE INDEX IF NOT EXISTS idx_pst_email_meta_pst ON pst_email_meta (pst_path)"
410
+ )
411
+
412
+ # Performance indexes for large-scale document sets (10k+ docs)
413
+ self._conn.execute(
414
+ "CREATE INDEX IF NOT EXISTS idx_nodes_doc_id ON nodes (doc_id)"
415
+ )
416
+ self._conn.execute(
417
+ "CREATE INDEX IF NOT EXISTS idx_documents_source_path ON documents (source_path)"
418
+ )
419
+
420
+ self._conn.commit()
421
+
422
+ @property
423
+ def db_path(self) -> str:
424
+ return self._db_path
425
+
426
+ def close(self) -> None:
427
+ """Close database connection and tokenize log file."""
428
+ if self._tokenize_log_file:
429
+ self._tokenize_log_file.close()
430
+ self._tokenize_log_file = None
431
+ if self._conn:
432
+ self._conn.close()
433
+ self._conn = None
434
+
435
+ def _log_tokenize(self, doc_id: str, tokens: set[str]) -> None:
436
+ """Append deduplicated token list for *doc_id* to the tokenize log."""
437
+ if not self._tokenize_log_path:
438
+ return
439
+ if self._tokenize_log_file is None:
440
+ self._tokenize_log_file = open(
441
+ self._tokenize_log_path, 'a', encoding='utf-8',
442
+ )
443
+ sorted_tokens = sorted(tokens)
444
+ self._tokenize_log_file.write(
445
+ f"--- {doc_id} ({len(sorted_tokens)} tokens) ---\n"
446
+ )
447
+ self._tokenize_log_file.write(" ".join(sorted_tokens) + "\n\n")
448
+ self._tokenize_log_file.flush()
449
+
450
+ def __del__(self):
451
+ try:
452
+ self.close()
453
+ except Exception:
454
+ pass
455
+
456
+ # -------------------------------------------------------------------
457
+ # Failed files tracking
458
+ # -------------------------------------------------------------------
459
+
460
+ def get_all_failed_files(self) -> dict[str, tuple[int, str, str]]:
461
+ """Batch load all failed file records.
462
+
463
+ Returns:
464
+ ``{source_path: (fail_count, file_hash, last_error)}``
465
+ """
466
+ rows = self._conn.execute(
467
+ "SELECT source_path, fail_count, file_hash, last_error FROM failed_files"
468
+ ).fetchall()
469
+ return {row[0]: (row[1], row[2], row[3] or "") for row in rows}
470
+
471
+ def upsert_failed_file(self, path: str, error_msg: str, file_hash: str = "") -> None:
472
+ """Insert or increment fail count for a file that failed to parse."""
473
+ import time as _time
474
+ now = _time.time()
475
+ self._conn.execute(
476
+ """INSERT INTO failed_files (source_path, fail_count, last_error, last_fail_time, file_hash)
477
+ VALUES (?, 1, ?, ?, ?)
478
+ ON CONFLICT(source_path) DO UPDATE SET
479
+ fail_count = fail_count + 1,
480
+ last_error = excluded.last_error,
481
+ last_fail_time = excluded.last_fail_time,
482
+ file_hash = excluded.file_hash
483
+ """,
484
+ (path, error_msg, now, file_hash),
485
+ )
486
+
487
+ def clear_failed_file(self, path: str) -> None:
488
+ """Remove a file's failed record after it is successfully indexed."""
489
+ self._conn.execute(
490
+ "DELETE FROM failed_files WHERE source_path = ?", (path,)
491
+ )
492
+
493
+ def clear_all_failed_files(self) -> None:
494
+ """Remove all failed file records (used during full rebuild)."""
495
+ with self._conn:
496
+ self._conn.execute("DELETE FROM failed_files")
497
+
498
+ def get_failed_files_summary(self) -> list[tuple[str, int, str]]:
499
+ """Return failed files list for display.
500
+
501
+ Returns:
502
+ list of ``(source_path, fail_count, last_error)``
503
+ """
504
+ rows = self._conn.execute(
505
+ "SELECT source_path, fail_count, last_error FROM failed_files ORDER BY fail_count DESC"
506
+ ).fetchall()
507
+ return [(row[0], row[1], row[2]) for row in rows]
508
+
509
+ # -------------------------------------------------------------------
510
+ # Vision parse queue (image files; consumed by host-side vision worker)
511
+ # -------------------------------------------------------------------
512
+
513
+ def vision_enqueue(self, source_path: str, rel_path: str = "") -> None:
514
+ """登记一个待视觉解析的图像文件(重复登记=文件变更,重置为 pending)。"""
515
+ import time as _time
516
+ self._conn.execute(
517
+ """INSERT INTO vision_queue (source_path, rel_path, status, attempts, model, last_error, updated_at)
518
+ VALUES (?, ?, 'pending', 0, '', '', ?)
519
+ ON CONFLICT(source_path) DO UPDATE SET
520
+ rel_path = excluded.rel_path,
521
+ status = 'pending',
522
+ attempts = 0,
523
+ last_error = '',
524
+ updated_at = excluded.updated_at
525
+ """,
526
+ (source_path, rel_path, _time.time()),
527
+ )
528
+
529
+ def vision_next_pending(self) -> Optional[dict]:
530
+ """取出下一个待解析项并置为 processing(原子抢占)。无则返回 None。"""
531
+ import time as _time
532
+ row = self._conn.execute(
533
+ "SELECT source_path, rel_path, attempts FROM vision_queue "
534
+ "WHERE status = 'pending' ORDER BY updated_at LIMIT 1"
535
+ ).fetchone()
536
+ if row is None:
537
+ return None
538
+ cur = self._conn.execute(
539
+ "UPDATE vision_queue SET status = 'processing', updated_at = ? "
540
+ "WHERE source_path = ? AND status = 'pending'",
541
+ (_time.time(), row[0]),
542
+ )
543
+ self._conn.commit()
544
+ if cur.rowcount == 0:
545
+ return None # 被其他消费者抢走
546
+ return {"source_path": row[0], "rel_path": row[1], "attempts": row[2]}
547
+
548
+ def vision_mark_done(self, source_path: str, model: str) -> bool:
549
+ """标记解析完成。仅当仍处于 processing 时生效(防止与 force 重建竞态)。"""
550
+ import time as _time
551
+ cur = self._conn.execute(
552
+ "UPDATE vision_queue SET status = 'done', model = ?, last_error = '', updated_at = ? "
553
+ "WHERE source_path = ? AND status = 'processing'",
554
+ (model, _time.time(), source_path),
555
+ )
556
+ self._conn.commit()
557
+ return cur.rowcount > 0
558
+
559
+ def vision_mark_failed(self, source_path: str, error: str, *, final: bool) -> None:
560
+ """记录一次失败;final=True(达到连败上限)时置 failed,否则回 pending 待重试。"""
561
+ import time as _time
562
+ self._conn.execute(
563
+ "UPDATE vision_queue SET status = ?, attempts = attempts + 1, last_error = ?, updated_at = ? "
564
+ "WHERE source_path = ?",
565
+ ("failed" if final else "pending", error[:500], _time.time(), source_path),
566
+ )
567
+ self._conn.commit()
568
+
569
+ def vision_remove(self, source_path: str) -> None:
570
+ """从队列移除(源文件被 prune / 删除时调用)。"""
571
+ self._conn.execute(
572
+ "DELETE FROM vision_queue WHERE source_path = ?", (source_path,)
573
+ )
574
+
575
+ def vision_clear(self) -> None:
576
+ """清空队列(force 全量重建时调用;重建过程会重新登记)。"""
577
+ with self._conn:
578
+ self._conn.execute("DELETE FROM vision_queue")
579
+
580
+ def vision_reset_stale_processing(self) -> int:
581
+ """启动时把残留的 processing(上次进程崩溃)重置回 pending。"""
582
+ with self._conn:
583
+ cur = self._conn.execute(
584
+ "UPDATE vision_queue SET status = 'pending' WHERE status = 'processing'"
585
+ )
586
+ return cur.rowcount
587
+
588
+ def vision_requeue_model_changed(self, current_tag: str) -> int:
589
+ """模型/prompt/协议版本变化时,把已完成或失败的项重新置为 pending 以便重解析。
590
+
591
+ 失败项也纳入:改协议(如 OpenAI-compat → anthropic)后,旧协议下失败的
592
+ 图像应在启动时自动重试,而不是永远卡在 failed。
593
+ """
594
+ import time as _time
595
+ with self._conn:
596
+ cur = self._conn.execute(
597
+ "UPDATE vision_queue SET status = 'pending', attempts = 0, last_error = '', updated_at = ? "
598
+ "WHERE status IN ('done', 'failed') AND model != ? AND model != 'manual'",
599
+ (_time.time(), current_tag),
600
+ )
601
+ return cur.rowcount
602
+
603
+ def vision_counts(self) -> dict[str, int]:
604
+ """各状态计数,供状态展示。如 ``{"pending": 3, "done": 12}``。"""
605
+ rows = self._conn.execute(
606
+ "SELECT status, COUNT(*) FROM vision_queue GROUP BY status"
607
+ ).fetchall()
608
+ return {row[0]: row[1] for row in rows}
609
+
610
+ # -------------------------------------------------------------------
611
+ # Indexing (Producer side)
612
+ # -------------------------------------------------------------------
613
+
614
+ def index_document(self, document, force: bool = False, auto_commit: bool = True,
615
+ file_hash: Optional[str] = None) -> int:
616
+ """Index all nodes from a Document into FTS5.
617
+
618
+ Performs **node-level incremental diff**: only nodes whose content
619
+ actually changed (or appeared/disappeared) are removed/inserted into
620
+ the FTS5 index. The full document tree is still re-serialized into the
621
+ ``documents`` table so search continues to read the latest structure.
622
+
623
+ Stable ``node_id``s (assigned by :func:`treesearch.tree.assign_node_ids`)
624
+ are required for correct diffing — same logical position must yield the
625
+ same id across runs.
626
+
627
+ Side-effects on ``last_node_diff`` so callers (e.g. build_index) can
628
+ report diff stats.
629
+
630
+ Args:
631
+ document: Document object with structure tree.
632
+ force: re-index every node even if hashes match.
633
+ auto_commit: if False, skip commit (caller is responsible).
634
+ file_hash: optional file fingerprint to write to ``index_meta``
635
+ in the same transaction (avoids the "FTS written but
636
+ fingerprint missing → next run rebuilds" failure mode).
637
+
638
+ Returns:
639
+ number of nodes (re-)indexed in this call.
640
+ """
641
+ # Compute content hash for incremental check
642
+ content_str = json.dumps(document.structure, ensure_ascii=False, sort_keys=True)
643
+ content_hash = hashlib.md5(content_str.encode()).hexdigest()
644
+
645
+ # Check if already indexed (whole-document fast-path)
646
+ if not force:
647
+ row = self._conn.execute(
648
+ "SELECT index_hash FROM documents WHERE doc_id = ?",
649
+ (document.doc_id,),
650
+ ).fetchone()
651
+ if row and row[0] == content_hash:
652
+ # logger.debug("Document %s already indexed (hash match), skipping", document.doc_id)
653
+ self.last_node_diff = {"added": 0, "changed": 0, "removed": 0, "kept": 0}
654
+ # Still write index_meta if requested — we may have been called
655
+ # because the file fingerprint changed but the structure didn't.
656
+ if file_hash and document.metadata.get("source_path"):
657
+ self._conn.execute(
658
+ "INSERT OR REPLACE INTO index_meta (source_path, file_hash) VALUES (?, ?)",
659
+ (document.metadata["source_path"], file_hash),
660
+ )
661
+ if auto_commit:
662
+ self._conn.commit()
663
+ return 0
664
+
665
+ # ---- Compute node-level diff ----
666
+ from .tree import flatten_tree, build_tree_maps
667
+ _, parent_map, depth_map = build_tree_maps(document.structure)
668
+
669
+ all_nodes = [n for n in flatten_tree(document.structure) if n.get("node_id")]
670
+
671
+ new_hashes: dict[str, str] = {}
672
+ for node in all_nodes:
673
+ text = node.get("text", "")
674
+ new_hashes[node["node_id"]] = hashlib.md5(text.encode()).hexdigest()[:16]
675
+
676
+ if force:
677
+ old_hashes: dict[str, str] = {}
678
+ else:
679
+ old_hashes = {
680
+ nid: chash for (nid, chash) in self._conn.execute(
681
+ "SELECT node_id, content_hash FROM nodes WHERE doc_id = ?",
682
+ (document.doc_id,),
683
+ ).fetchall()
684
+ }
685
+
686
+ new_ids = set(new_hashes.keys())
687
+ old_ids = set(old_hashes.keys())
688
+ added = new_ids - old_ids
689
+ removed = old_ids - new_ids
690
+ changed = {nid for nid in (new_ids & old_ids) if new_hashes[nid] != old_hashes[nid]}
691
+ kept = (new_ids & old_ids) - changed
692
+
693
+ to_write = added | changed
694
+ diff_stats = {
695
+ "added": len(added),
696
+ "changed": len(changed),
697
+ "removed": len(removed),
698
+ "kept": len(kept),
699
+ }
700
+ self.last_node_diff = diff_stats
701
+
702
+ # ---- Stage all rows ahead of the transaction ----
703
+ node_rows: list[tuple] = []
704
+ fts_rows: list[tuple] = []
705
+ doc_tokens: set[str] = set()
706
+ for node in all_nodes:
707
+ nid = node["node_id"]
708
+ if nid not in to_write:
709
+ continue
710
+ title = node.get("title", "")
711
+ summary = node.get("summary", node.get("prefix_summary", ""))
712
+ text = node.get("text", "")
713
+ depth = depth_map.get(nid, 0)
714
+ node_rows.append((
715
+ nid, document.doc_id, title, summary, depth,
716
+ node.get("line_start"), node.get("line_end"),
717
+ parent_map.get(nid), new_hashes[nid],
718
+ ))
719
+
720
+ parsed = parse_md_node_text(text)
721
+ tok_title = _tokenize_for_fts(title)
722
+ tok_summary = _tokenize_for_fts(summary)
723
+ tok_body = _tokenize_for_fts(parsed["body"])
724
+ tok_code = _tokenize_for_fts(parsed["code_blocks"])
725
+ tok_fm = _tokenize_for_fts(parsed["front_matter"])
726
+
727
+ # 将文件名和路径分词后注入 front_matter,使文件名/路径可被 FTS5 搜索到
728
+ tok_doc_name = _tokenize_for_fts(document.doc_name)
729
+ if tok_doc_name and tok_doc_name not in tok_fm:
730
+ tok_fm = tok_doc_name + " " + tok_fm
731
+ source_path = document.metadata.get("source_path", "")
732
+ tok_path = _tokenize_for_fts(source_path)
733
+ if tok_path and tok_path not in tok_fm:
734
+ tok_fm = tok_path + " " + tok_fm
735
+ fts_rows.append((
736
+ nid, document.doc_id,
737
+ tok_title, tok_summary, tok_body, tok_code, tok_fm,
738
+ ))
739
+ for tok_str in (tok_title, tok_summary, tok_body, tok_code, tok_fm):
740
+ doc_tokens.update(tok_str.split())
741
+
742
+ self._log_tokenize(document.doc_id, doc_tokens)
743
+
744
+ # ---- Single atomic transaction ----
745
+ # Wraps deletes + inserts + document metadata + index_meta so a crash
746
+ # in the middle either rolls back fully or commits everything.
747
+ structure_json = json.dumps(document.structure, ensure_ascii=False)
748
+
749
+ if auto_commit:
750
+ txn_ctx = self._conn # 'with conn' = implicit transaction
751
+ else:
752
+ txn_ctx = _NullContext()
753
+
754
+ with txn_ctx:
755
+ # Targeted deletes for removed + changed (NOT a wholesale wipe).
756
+ del_ids = removed | changed | (added if force else set())
757
+ if del_ids:
758
+ if self._use_fts5:
759
+ placeholders = ",".join("?" for _ in del_ids)
760
+ old_rowids = self._conn.execute(
761
+ f"SELECT rowid FROM fts_nodes WHERE doc_id = ? AND node_id IN ({placeholders})",
762
+ (document.doc_id, *del_ids),
763
+ ).fetchall()
764
+ if old_rowids:
765
+ ph2 = ",".join("?" for _ in old_rowids)
766
+ self._conn.execute(
767
+ f"DELETE FROM fts_nodes WHERE rowid IN ({ph2})",
768
+ [r[0] for r in old_rowids],
769
+ )
770
+ else:
771
+ placeholders = ",".join("?" for _ in del_ids)
772
+ self._conn.execute(
773
+ f"DELETE FROM fts_nodes WHERE doc_id = ? AND node_id IN ({placeholders})",
774
+ (document.doc_id, *del_ids),
775
+ )
776
+ placeholders = ",".join("?" for _ in del_ids)
777
+ self._conn.execute(
778
+ f"DELETE FROM nodes WHERE doc_id = ? AND node_id IN ({placeholders})",
779
+ (document.doc_id, *del_ids),
780
+ )
781
+
782
+ if node_rows:
783
+ self._conn.executemany(
784
+ """INSERT INTO nodes
785
+ (node_id, doc_id, title, summary, depth, line_start, line_end, parent_node_id, content_hash)
786
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)""",
787
+ node_rows,
788
+ )
789
+ if fts_rows:
790
+ self._conn.executemany(
791
+ """INSERT INTO fts_nodes
792
+ (node_id, doc_id, title, summary, body, code_blocks, front_matter)
793
+ VALUES (?, ?, ?, ?, ?, ?, ?)""",
794
+ fts_rows,
795
+ )
796
+
797
+ self._conn.execute(
798
+ """INSERT OR REPLACE INTO documents
799
+ (doc_id, doc_name, doc_description, source_path, source_type,
800
+ structure_json, node_count, index_hash)
801
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?)""",
802
+ (
803
+ document.doc_id, document.doc_name, document.doc_description,
804
+ document.metadata.get("source_path", ""),
805
+ document.source_type,
806
+ structure_json,
807
+ len(all_nodes), content_hash,
808
+ ),
809
+ )
810
+
811
+ if file_hash and document.metadata.get("source_path"):
812
+ self._conn.execute(
813
+ "INSERT OR REPLACE INTO index_meta (source_path, file_hash) VALUES (?, ?)",
814
+ (document.metadata["source_path"], file_hash),
815
+ )
816
+
817
+ logger.debug(
818
+ "FTS5 reindexed %s: +%d ~%d -%d (kept %d)",
819
+ document.doc_id, diff_stats["added"], diff_stats["changed"],
820
+ diff_stats["removed"], diff_stats["kept"],
821
+ )
822
+ return len(node_rows)
823
+
824
+ def commit(self) -> None:
825
+ """Manually commit pending changes to the database."""
826
+ self._conn.commit()
827
+
828
+ def index_documents(self, documents: list, force: bool = False) -> int:
829
+ """Batch index multiple documents.
830
+
831
+ Returns:
832
+ total number of nodes indexed
833
+ """
834
+ total = 0
835
+ for doc in documents:
836
+ total += self.index_document(doc, force=force)
837
+ return total
838
+
839
+ # -------------------------------------------------------------------
840
+ # Search (Consumer side)
841
+ # -------------------------------------------------------------------
842
+
843
+ def _build_match_expr(self, query: str, fts_expression: Optional[str] = None) -> Optional[str]:
844
+ """Build FTS5 MATCH expression from query (cached tokenization).
845
+
846
+ Returns None if no valid tokens could be extracted.
847
+ """
848
+ if fts_expression:
849
+ return _tokenize_fts_expression(fts_expression)
850
+
851
+ tokens = _tokenize_for_fts(query)
852
+ if not tokens.strip():
853
+ return None
854
+ words = tokens.split()
855
+ clean_words = []
856
+ for w in words:
857
+ cleaned = _RE_FTS5_SPECIAL.sub("", w).strip()
858
+ if cleaned and cleaned.upper() not in _FTS5_OPERATORS:
859
+ clean_words.append(cleaned)
860
+ if not clean_words:
861
+ return None
862
+ if len(clean_words) > 1:
863
+ return " OR ".join(clean_words)
864
+ return clean_words[0]
865
+
866
+ def search(
867
+ self,
868
+ query: str,
869
+ doc_id: Optional[str] = None,
870
+ top_k: int = 20,
871
+ fts_expression: Optional[str] = None,
872
+ _precomputed_match_expr: Optional[str] = None,
873
+ ) -> list[dict]:
874
+ """Search nodes using FTS5 BM25 ranking (or LIKE fallback).
875
+
876
+ Args:
877
+ query: natural language query (will be tokenized)
878
+ doc_id: optional filter by document
879
+ top_k: max results
880
+ fts_expression: raw FTS5 query expression (overrides query tokenization).
881
+ Supports AND, OR, NOT, NEAR, phrases.
882
+ _precomputed_match_expr: internal — skip tokenization if already computed.
883
+
884
+ Returns:
885
+ list of {node_id, doc_id, title, summary, fts_score, depth}
886
+ """
887
+ if not self._use_fts5:
888
+ return self._search_like(query, doc_id=doc_id, top_k=top_k)
889
+
890
+ if _precomputed_match_expr is not None:
891
+ match_expr = _precomputed_match_expr
892
+ else:
893
+ match_expr = self._build_match_expr(query, fts_expression)
894
+ if match_expr is None:
895
+ return []
896
+
897
+ # Phase 1: phrase boosting for multi-word queries
898
+ # Run a separate phrase match query and record which nodes get a boost
899
+ phrase_boost_nids: set[str] = set()
900
+ if not fts_expression and _precomputed_match_expr is None and len(query.split()) >= 2:
901
+ # Build phrase expression from original (unstemmed) query words
902
+ raw_words = [w.lower().strip() for w in re.split(r'\W+', query) if w.strip() and len(w.strip()) > 2]
903
+ if len(raw_words) >= 2:
904
+ # Try phrase match: "word1 word2 ..."
905
+ phrase_expr = '"' + ' '.join(raw_words) + '"'
906
+ try:
907
+ if doc_id:
908
+ phrase_rows = self._conn.execute(
909
+ f"SELECT f.node_id FROM fts_nodes f WHERE fts_nodes MATCH ? AND f.doc_id = ? LIMIT 50",
910
+ (phrase_expr, doc_id),
911
+ ).fetchall()
912
+ else:
913
+ phrase_rows = self._conn.execute(
914
+ f"SELECT f.node_id FROM fts_nodes f WHERE fts_nodes MATCH ? LIMIT 50",
915
+ (phrase_expr,),
916
+ ).fetchall()
917
+ phrase_boost_nids = {r[0] for r in phrase_rows}
918
+ except sqlite3.OperationalError:
919
+ pass # phrase query syntax error, skip boost
920
+
921
+ # Build SQL with column weights for bm25()
922
+ # bm25(fts_nodes, w1, w2, w3, w4, w5) where weights correspond to:
923
+ # node_id(UNINDEXED), doc_id(UNINDEXED), title, summary, body, code_blocks, front_matter
924
+ w = self._weights
925
+ weight_args = f"{w['title']}, {w['summary']}, {w['body']}, {w['code_blocks']}, {w['front_matter']}"
926
+
927
+ # Query FTS5 directly without JOIN to nodes table, because fts_nodes
928
+ # stores chunk node_ids (e.g. "0_chunk0") that don't match the original
929
+ # node_ids in nodes table (e.g. "0"). Metadata is looked up separately.
930
+ if doc_id:
931
+ sql = f"""
932
+ SELECT f.node_id, f.doc_id, f.title, f.summary,
933
+ bm25(fts_nodes, {weight_args}) AS rank_score
934
+ FROM fts_nodes f
935
+ WHERE fts_nodes MATCH ?
936
+ AND f.doc_id = ?
937
+ ORDER BY rank_score
938
+ LIMIT ?
939
+ """
940
+ params = (match_expr, doc_id, top_k)
941
+ else:
942
+ sql = f"""
943
+ SELECT f.node_id, f.doc_id, f.title, f.summary,
944
+ bm25(fts_nodes, {weight_args}) AS rank_score
945
+ FROM fts_nodes f
946
+ WHERE fts_nodes MATCH ?
947
+ ORDER BY rank_score
948
+ LIMIT ?
949
+ """
950
+ params = (match_expr, top_k)
951
+
952
+ try:
953
+ rows = self._conn.execute(sql, params).fetchall()
954
+ except sqlite3.OperationalError as e:
955
+ logger.warning("FTS5 query error: %s, query=%r", e, match_expr)
956
+ rows = []
957
+
958
+ # Pre-fetch node metadata (depth, title, summary) for deduped node_ids
959
+ unique_nids_in_result = {r[0] for r in rows}
960
+ node_meta: dict[tuple[str, str], dict] = {}
961
+ if unique_nids_in_result:
962
+ # Batch lookup from nodes table
963
+ for raw_nid in unique_nids_in_result:
964
+ for r in rows:
965
+ if r[0] == raw_nid:
966
+ did = r[1]
967
+ break
968
+ else:
969
+ continue
970
+ meta_row = self._conn.execute(
971
+ "SELECT title, summary, depth FROM nodes WHERE node_id = ? AND doc_id = ?",
972
+ (raw_nid, did),
973
+ ).fetchone()
974
+ if meta_row:
975
+ node_meta[(raw_nid, did)] = {"title": meta_row[0], "summary": meta_row[1], "depth": meta_row[2]}
976
+
977
+ results = []
978
+ seen_nids: dict[str, int] = {} # track dedup by node_id
979
+ for row in rows:
980
+ # bm25() returns negative values (lower = more relevant)
981
+ fts_score = -row[4] if row[4] else 0.0
982
+ # Apply phrase boost: nodes matching exact phrase get 50% score bonus
983
+ if row[0] in phrase_boost_nids:
984
+ fts_score *= 1.5
985
+ nid = row[0]
986
+ if nid in seen_nids:
987
+ # Keep the higher score for the same node
988
+ idx = seen_nids[nid]
989
+ if fts_score > results[idx]["fts_score"]:
990
+ results[idx]["fts_score"] = round(fts_score, 6)
991
+ continue
992
+ seen_nids[nid] = len(results)
993
+ meta = node_meta.get((nid, row[1]))
994
+ results.append({
995
+ "node_id": nid,
996
+ "doc_id": row[1],
997
+ "title": meta["title"] if meta else row[2],
998
+ "summary": meta["summary"] if meta else row[3],
999
+ "depth": meta["depth"] if meta else 0,
1000
+ "fts_score": round(fts_score, 6),
1001
+ })
1002
+
1003
+ # Re-sort after phrase boosting
1004
+ if phrase_boost_nids:
1005
+ results.sort(key=lambda x: -x["fts_score"])
1006
+
1007
+ return results
1008
+
1009
+ def _search_like(
1010
+ self,
1011
+ query: str,
1012
+ doc_id: Optional[str] = None,
1013
+ top_k: int = 20,
1014
+ ) -> list[dict]:
1015
+ """Fallback search using LIKE when FTS5 is unavailable.
1016
+
1017
+ Splits query into keywords and scores nodes by weighted keyword hits
1018
+ across (title, summary, body, code_blocks, front_matter).
1019
+ """
1020
+ tokens = _tokenize_for_fts(query)
1021
+ if not tokens.strip():
1022
+ return []
1023
+ keywords = [kw.strip().lower() for kw in tokens.split() if kw.strip()]
1024
+ if not keywords:
1025
+ return []
1026
+
1027
+ w = self._weights
1028
+
1029
+ # Pre-fetch all node metadata to avoid N+1 queries
1030
+ if doc_id:
1031
+ meta_rows = self._conn.execute(
1032
+ "SELECT node_id, doc_id, title, summary, depth FROM nodes WHERE doc_id = ?",
1033
+ (doc_id,),
1034
+ ).fetchall()
1035
+ else:
1036
+ meta_rows = self._conn.execute(
1037
+ "SELECT node_id, doc_id, title, summary, depth FROM nodes"
1038
+ ).fetchall()
1039
+ meta_map = {(r[0], r[1]): {"title": r[2], "summary": r[3], "depth": r[4]} for r in meta_rows}
1040
+
1041
+ if doc_id:
1042
+ rows = self._conn.execute(
1043
+ """SELECT node_id, doc_id, title, summary, body, code_blocks, front_matter
1044
+ FROM fts_nodes WHERE doc_id = ?""",
1045
+ (doc_id,),
1046
+ ).fetchall()
1047
+ else:
1048
+ rows = self._conn.execute(
1049
+ "SELECT node_id, doc_id, title, summary, body, code_blocks, front_matter FROM fts_nodes"
1050
+ ).fetchall()
1051
+
1052
+ scored: list[tuple[float, dict]] = []
1053
+ seen_original_nids: dict[str, int] = {}
1054
+ for row in rows:
1055
+ nid, did, title, summary, body, code_blocks, front_matter = row
1056
+ score = 0.0
1057
+ fields = [
1058
+ (title or "", w["title"]),
1059
+ (summary or "", w["summary"]),
1060
+ (body or "", w["body"]),
1061
+ (code_blocks or "", w["code_blocks"]),
1062
+ (front_matter or "", w["front_matter"]),
1063
+ ]
1064
+ for kw in keywords:
1065
+ for text, weight in fields:
1066
+ if kw in text.lower():
1067
+ score += weight
1068
+ if score > 0:
1069
+ if nid in seen_original_nids:
1070
+ idx = seen_original_nids[nid]
1071
+ if score > scored[idx][0]:
1072
+ meta = meta_map.get((nid, did))
1073
+ scored[idx] = (score, {
1074
+ "node_id": nid,
1075
+ "doc_id": did,
1076
+ "title": meta["title"] if meta else title,
1077
+ "summary": meta["summary"] if meta else summary,
1078
+ "depth": meta["depth"] if meta else 0,
1079
+ "fts_score": round(score, 6),
1080
+ })
1081
+ continue
1082
+ seen_original_nids[nid] = len(scored)
1083
+ meta = meta_map.get((nid, did))
1084
+ scored.append((score, {
1085
+ "node_id": nid,
1086
+ "doc_id": did,
1087
+ "title": meta["title"] if meta else title,
1088
+ "summary": meta["summary"] if meta else summary,
1089
+ "depth": meta["depth"] if meta else 0,
1090
+ "fts_score": round(score, 6),
1091
+ }))
1092
+
1093
+ scored.sort(key=lambda x: -x[0])
1094
+ return [item[1] for item in scored[:top_k]]
1095
+
1096
+ def like_search(self, query: str, top_k: int = 20, use_regex: bool = False) -> list[dict]:
1097
+ """Substring or regex search on nodes.summary/title (original text, not tokenized).
1098
+
1099
+ Fallback when FTS5 tokenization causes query mismatches.
1100
+ Searches the ``nodes`` table which stores original (un-tokenized) text,
1101
+ so phrases that were split by jieba can still be matched.
1102
+
1103
+ Args:
1104
+ query: search string or regex pattern (when use_regex=True).
1105
+ top_k: max results.
1106
+ use_regex: if True, treat query as a regular expression (SQLite REGEXP).
1107
+
1108
+ If ``nodes`` yields no matches, searches ``documents.structure_json``
1109
+ which contains the full original document structure.
1110
+
1111
+ Returns same format as search(): list of {node_id, doc_id, title, summary, fts_score, depth}.
1112
+ """
1113
+ results: list[dict] = []
1114
+ seen: set[str] = set()
1115
+
1116
+ # Build match expression based on mode
1117
+ if use_regex:
1118
+ compiled = re.compile(query, re.IGNORECASE)
1119
+ where_clause = "n.summary REGEXP ? OR n.title REGEXP ?"
1120
+ params = (query, query)
1121
+ else:
1122
+ pattern = f"%{query}%"
1123
+ where_clause = "n.summary LIKE ? OR n.title LIKE ?"
1124
+ params = (pattern, pattern)
1125
+
1126
+ # Phase 1: Search nodes.summary and nodes.title (original text)
1127
+ rows = self._conn.execute(
1128
+ f"""SELECT n.node_id, n.doc_id, n.title, n.summary, n.depth,
1129
+ d.doc_name, d.source_path
1130
+ FROM nodes n
1131
+ JOIN documents d ON n.doc_id = d.doc_id
1132
+ WHERE {where_clause}
1133
+ ORDER BY n.depth ASC""",
1134
+ params,
1135
+ ).fetchall()
1136
+
1137
+ for row in rows:
1138
+ nid, did = row[0], row[1]
1139
+ key = f"{did}/{nid}"
1140
+ if key in seen:
1141
+ continue
1142
+ seen.add(key)
1143
+ results.append({
1144
+ "node_id": nid,
1145
+ "doc_id": did,
1146
+ "title": row[2] or "",
1147
+ "summary": row[3] or "",
1148
+ "depth": row[4] or 0,
1149
+ "fts_score": 1.0,
1150
+ })
1151
+
1152
+ if results:
1153
+ return results[:top_k]
1154
+
1155
+ # Phase 2: Search documents.structure_json (full original text)
1156
+ if use_regex:
1157
+ match_fn = compiled.search
1158
+ else:
1159
+ query_lower = query.lower()
1160
+ match_fn = lambda t: query_lower in t.lower() if t else False
1161
+
1162
+ try:
1163
+ if use_regex:
1164
+ doc_rows = self._conn.execute(
1165
+ "SELECT doc_id, structure_json FROM documents WHERE structure_json REGEXP ?",
1166
+ (query,),
1167
+ ).fetchall()
1168
+ else:
1169
+ doc_rows = self._conn.execute(
1170
+ "SELECT doc_id, structure_json FROM documents WHERE structure_json LIKE ?",
1171
+ (pattern,),
1172
+ ).fetchall()
1173
+ except Exception:
1174
+ return []
1175
+
1176
+ for doc_id, structure_json in doc_rows:
1177
+ if not structure_json:
1178
+ continue
1179
+ try:
1180
+ nodes_data = json.loads(structure_json)
1181
+ except (json.JSONDecodeError, TypeError):
1182
+ continue
1183
+ if not isinstance(nodes_data, list):
1184
+ continue
1185
+ from .tree import flatten_tree
1186
+ for node in flatten_tree(nodes_data):
1187
+ text = node.get("text", "") or ""
1188
+ title = node.get("title", "") or ""
1189
+ if match_fn(text) or match_fn(title):
1190
+ nid = node.get("node_id", "")
1191
+ key = f"{doc_id}/{nid}"
1192
+ if key in seen:
1193
+ continue
1194
+ seen.add(key)
1195
+ # 提取包含匹配关键词的片段
1196
+ snippet = _extract_match_snippet(text, query, use_regex, size=300)
1197
+ results.append({
1198
+ "node_id": nid,
1199
+ "doc_id": doc_id,
1200
+ "title": title,
1201
+ "summary": snippet,
1202
+ "depth": node.get("depth", 0),
1203
+ "fts_score": 0.5,
1204
+ })
1205
+
1206
+ return results[:top_k]
1207
+
1208
+ def search_with_aggregation(
1209
+ self,
1210
+ query: str,
1211
+ group_by_doc: bool = True,
1212
+ top_k: int = 20,
1213
+ fts_expression: Optional[str] = None,
1214
+ ) -> list[dict]:
1215
+ """Search with SQL aggregation capabilities.
1216
+
1217
+ Returns per-document aggregated results: total hits, max score, avg score.
1218
+ """
1219
+ if not group_by_doc:
1220
+ return self.search(query, top_k=top_k, fts_expression=fts_expression)
1221
+
1222
+ # Two-step: first get all matched nodes with scores, then aggregate in Python
1223
+ results = self.search(query, top_k=200, fts_expression=fts_expression)
1224
+ if not results:
1225
+ return []
1226
+
1227
+ doc_agg: dict[str, dict] = {}
1228
+ for r in results:
1229
+ did = r["doc_id"]
1230
+ if did not in doc_agg:
1231
+ # Fetch doc_name
1232
+ row = self._conn.execute(
1233
+ "SELECT doc_name FROM documents WHERE doc_id = ?", (did,)
1234
+ ).fetchone()
1235
+ doc_agg[did] = {
1236
+ "doc_id": did,
1237
+ "doc_name": row[0] if row else "",
1238
+ "hit_count": 0,
1239
+ "best_score": 0.0,
1240
+ "total_score": 0.0,
1241
+ }
1242
+ doc_agg[did]["hit_count"] += 1
1243
+ doc_agg[did]["best_score"] = max(doc_agg[did]["best_score"], r["fts_score"])
1244
+ doc_agg[did]["total_score"] += r["fts_score"]
1245
+
1246
+ agg_results = []
1247
+ for agg in doc_agg.values():
1248
+ agg["avg_score"] = round(agg["total_score"] / agg["hit_count"], 6)
1249
+ agg["best_score"] = round(agg["best_score"], 6)
1250
+ del agg["total_score"]
1251
+ agg_results.append(agg)
1252
+
1253
+ agg_results.sort(key=lambda x: -x["best_score"])
1254
+ return agg_results[:top_k]
1255
+
1256
+ def score_nodes(self, query: str, doc_id: str, ancestor_decay: float = 0.6) -> dict[str, float]:
1257
+ """PreFilter protocol: return {node_id: score} for search() integration.
1258
+
1259
+ This allows FTS5Index to be used as a drop-in PreFilter in the search pipeline.
1260
+
1261
+ Includes:
1262
+ - Ancestor score propagation: parent nodes inherit child scores
1263
+
1264
+ For scoring multiple documents at once, prefer score_nodes_batch() which uses
1265
+ a single SQL query instead of one query per document.
1266
+ """
1267
+ result = self.score_nodes_batch(query, doc_ids=[doc_id], ancestor_decay=ancestor_decay)
1268
+ return result.get(doc_id, {})
1269
+
1270
+ def score_nodes_batch(
1271
+ self,
1272
+ query: str,
1273
+ doc_ids: list[str] | None = None,
1274
+ ancestor_decay: float = 0.6,
1275
+ fts_expression: Optional[str] = None,
1276
+ ) -> dict[str, dict[str, float]]:
1277
+ """Batch version of score_nodes: score all documents in a single SQL query.
1278
+
1279
+ Returns {doc_id: {node_id: score}} for all matched documents.
1280
+
1281
+ This is the fast path for tree search, replacing the N_docs loop:
1282
+ # Before (slow): N SQL queries
1283
+ for doc in documents:
1284
+ scores = fts_index.score_nodes(query, doc.doc_id)
1285
+
1286
+ # After (fast): 1 SQL query
1287
+ all_scores = fts_index.score_nodes_batch(query, [d.doc_id for d in documents])
1288
+
1289
+ Args:
1290
+ query: natural language query (tokenized internally)
1291
+ doc_ids: optional filter to specific documents
1292
+ ancestor_decay: propagation weight from child scores to parent nodes (0 = off)
1293
+ fts_expression: raw FTS5 MATCH expression that *overrides* automatic query
1294
+ tokenization. Supports full FTS5 syntax including prefix matching (*),
1295
+ AND/OR/NOT, NEAR(), and column filters.
1296
+
1297
+ Examples::
1298
+
1299
+ # Prefix match: "fts" matches fts, fts5, ftsearch, ...
1300
+ fts_expression="fts*"
1301
+
1302
+ # Multi-term prefix OR
1303
+ fts_expression="fts* OR python*"
1304
+
1305
+ # Exact phrase
1306
+ fts_expression='"machine learning"'
1307
+
1308
+ # Column-scoped search
1309
+ fts_expression="title : config*"
1310
+
1311
+ # Build with helper
1312
+ expr = FTS5Index.build_fts_expression(["fts", "python"],
1313
+ prefix=True, operator="OR")
1314
+ """
1315
+ match_expr = self._build_match_expr(query, fts_expression)
1316
+ if match_expr is None:
1317
+ return {}
1318
+
1319
+ w = self._weights
1320
+ weight_args = f"{w['title']}, {w['summary']}, {w['body']}, {w['code_blocks']}, {w['front_matter']}"
1321
+
1322
+ # Single SQL query across all requested doc_ids (or entire index)
1323
+ if doc_ids:
1324
+ placeholders = ",".join("?" * len(doc_ids))
1325
+ sql = f"""
1326
+ SELECT f.node_id, f.doc_id,
1327
+ bm25(fts_nodes, {weight_args}) AS rank_score
1328
+ FROM fts_nodes f
1329
+ WHERE fts_nodes MATCH ?
1330
+ AND f.doc_id IN ({placeholders})
1331
+ ORDER BY rank_score
1332
+ LIMIT 5000
1333
+ """
1334
+ params = (match_expr, *doc_ids)
1335
+ else:
1336
+ sql = f"""
1337
+ SELECT f.node_id, f.doc_id,
1338
+ bm25(fts_nodes, {weight_args}) AS rank_score
1339
+ FROM fts_nodes f
1340
+ WHERE fts_nodes MATCH ?
1341
+ ORDER BY rank_score
1342
+ LIMIT 5000
1343
+ """
1344
+ params = (match_expr,)
1345
+
1346
+ try:
1347
+ rows = self._conn.execute(sql, params).fetchall()
1348
+ except Exception:
1349
+ return {}
1350
+
1351
+ if not rows:
1352
+ return {}
1353
+
1354
+ # Group raw scores by doc_id
1355
+ per_doc_raw: dict[str, dict[str, float]] = {}
1356
+ for node_id, doc_id, rank_score in rows:
1357
+ fts_score = -rank_score if rank_score else 0.0
1358
+ per_doc_raw.setdefault(doc_id, {})
1359
+ old = per_doc_raw[doc_id].get(node_id, 0.0)
1360
+ per_doc_raw[doc_id][node_id] = max(old, fts_score)
1361
+
1362
+ # Per-doc: normalize + ancestor propagation in one pass
1363
+ result: dict[str, dict[str, float]] = {}
1364
+ doc_children_map: dict[str, dict[str, list[str]]] = {}
1365
+ if ancestor_decay > 0 and per_doc_raw:
1366
+ # Single query to fetch parent maps for all affected docs
1367
+ affected_docs = list(per_doc_raw.keys())
1368
+ ph = ",".join("?" * len(affected_docs))
1369
+ parent_rows = self._conn.execute(
1370
+ f"SELECT doc_id, node_id, parent_node_id FROM nodes WHERE doc_id IN ({ph})",
1371
+ affected_docs,
1372
+ ).fetchall()
1373
+ # Build per-doc children maps for bottom-up propagation
1374
+ for d_id, nid, pid in parent_rows:
1375
+ if pid:
1376
+ doc_children_map.setdefault(d_id, {}).setdefault(pid, []).append(nid)
1377
+
1378
+ for doc_id, raw_scores in per_doc_raw.items():
1379
+ # Normalize to [0, 1]
1380
+ max_s = max(raw_scores.values()) if raw_scores else 1.0
1381
+ if max_s <= 0:
1382
+ max_s = 1.0
1383
+ scores = {nid: s / max_s for nid, s in raw_scores.items()}
1384
+
1385
+ # Ancestor propagation
1386
+ if ancestor_decay > 0:
1387
+ children_map = doc_children_map.get(doc_id, {})
1388
+ for pid, cids in children_map.items():
1389
+ child_scores = [scores.get(c, 0.0) for c in cids]
1390
+ if not child_scores:
1391
+ continue
1392
+ bonus = ancestor_decay * max(child_scores)
1393
+ scores[pid] = scores.get(pid, 0.0) + bonus
1394
+
1395
+ # Re-normalize after propagation
1396
+ final_max = max(scores.values()) if scores else 1.0
1397
+ if final_max > 1.0:
1398
+ scores = {nid: s / final_max for nid, s in scores.items()}
1399
+
1400
+ result[doc_id] = {nid: round(s, 6) for nid, s in scores.items()}
1401
+
1402
+ return result
1403
+
1404
+ def ranked_node_ids(
1405
+ self,
1406
+ query: str,
1407
+ doc_ids: list[str] | None = None,
1408
+ top_k: int = 10,
1409
+ fts_expression: Optional[str] = None,
1410
+ ) -> list[str]:
1411
+ """Convenience: return top-k node IDs ranked by FTS5 score.
1412
+
1413
+ Wraps score_nodes_batch → flatten → sort → top-k in one call.
1414
+ Useful for flat FTS5 evaluation without manual dict wrangling.
1415
+
1416
+ Args:
1417
+ query: search query
1418
+ doc_ids: optional filter to specific documents
1419
+ top_k: max results
1420
+ fts_expression: raw FTS5 MATCH expression (overrides query tokenization).
1421
+ Use for prefix matching, boolean operators, NEAR(), etc.
1422
+ Example: ``fts_expression="config*"`` matches config, configuration, ...
1423
+
1424
+ Returns:
1425
+ list of node_id strings, highest score first
1426
+ """
1427
+ batch = self.score_nodes_batch(query, doc_ids=doc_ids, fts_expression=fts_expression)
1428
+ all_scored: list[tuple[str, float]] = []
1429
+ for nscores in batch.values():
1430
+ all_scored.extend(nscores.items())
1431
+ all_scored.sort(key=lambda x: -x[1])
1432
+ return [nid for nid, _ in all_scored[:top_k]]
1433
+
1434
+ # -------------------------------------------------------------------
1435
+ # Document persistence (tree structure storage)
1436
+ # -------------------------------------------------------------------
1437
+
1438
+ def save_document(self, document, auto_commit: bool = True) -> None:
1439
+ """Save/update a Document's tree structure into the DB.
1440
+
1441
+ This persists the tree structure so that JSON files are no longer needed.
1442
+ FTS indexing is NOT performed here — call index_document() separately.
1443
+
1444
+ Args:
1445
+ document: Document object with structure tree
1446
+ auto_commit: if False, skip commit (caller is responsible for committing)
1447
+ """
1448
+ from .tree import flatten_tree
1449
+ structure_json = json.dumps(document.structure, ensure_ascii=False)
1450
+ content_hash = hashlib.md5(structure_json.encode()).hexdigest()
1451
+ self._conn.execute(
1452
+ """INSERT OR REPLACE INTO documents
1453
+ (doc_id, doc_name, doc_description, source_path, source_type, structure_json, node_count, index_hash)
1454
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?)""",
1455
+ (
1456
+ document.doc_id, document.doc_name, document.doc_description,
1457
+ document.metadata.get("source_path", ""),
1458
+ document.source_type,
1459
+ structure_json,
1460
+ len(flatten_tree(document.structure)),
1461
+ content_hash,
1462
+ ),
1463
+ )
1464
+ if auto_commit:
1465
+ self._conn.commit()
1466
+
1467
+ def load_document(self, doc_id: str):
1468
+ """Load a single Document from the DB by doc_id.
1469
+
1470
+ Returns:
1471
+ Document object, or None if not found.
1472
+ """
1473
+ from .tree import Document
1474
+ row = self._conn.execute(
1475
+ "SELECT doc_id, doc_name, doc_description, source_path, source_type, structure_json FROM documents WHERE doc_id = ?",
1476
+ (doc_id,),
1477
+ ).fetchone()
1478
+ if not row:
1479
+ return None
1480
+ structure = json.loads(row[5]) if row[5] else []
1481
+ return Document(
1482
+ doc_id=row[0],
1483
+ doc_name=row[1],
1484
+ structure=structure,
1485
+ doc_description=row[2] or "",
1486
+ metadata={"source_path": row[3] or ""},
1487
+ source_type=row[4] or "",
1488
+ )
1489
+
1490
+ def load_document_by_source_path(self, source_path: str) -> Optional["Document"]:
1491
+ """按 source_path 加载单个 Document(含 structure_json 反序列化)。
1492
+
1493
+ Args:
1494
+ source_path: 文件绝对路径,必须与索引时存入 documents.source_path 完全一致。
1495
+
1496
+ Returns:
1497
+ Document 对象;无匹配返回 None。
1498
+ """
1499
+ from .tree import Document # 局部导入避免循环依赖
1500
+
1501
+ row = self._conn.execute(
1502
+ "SELECT doc_id, doc_name, doc_description, source_path, source_type, structure_json "
1503
+ "FROM documents WHERE source_path = ? LIMIT 1",
1504
+ (source_path,),
1505
+ ).fetchone()
1506
+ if not row:
1507
+ return None
1508
+ doc_id, doc_name, doc_description, sp, source_type, structure_json = row
1509
+ structure = json.loads(structure_json) if structure_json else []
1510
+ return Document(
1511
+ doc_id=doc_id,
1512
+ doc_name=doc_name or "",
1513
+ doc_description=doc_description or "",
1514
+ structure=structure,
1515
+ source_type=source_type or "",
1516
+ metadata={"source_path": sp or ""},
1517
+ )
1518
+
1519
+ def load_all_documents(self) -> list:
1520
+ """Load all Documents stored in the DB.
1521
+
1522
+ Returns:
1523
+ List of Document objects.
1524
+ """
1525
+ from .tree import Document
1526
+ rows = self._conn.execute(
1527
+ "SELECT doc_id, doc_name, doc_description, source_path, source_type, structure_json FROM documents ORDER BY doc_id"
1528
+ ).fetchall()
1529
+ documents = []
1530
+ for row in rows:
1531
+ structure = json.loads(row[5]) if row[5] else []
1532
+ documents.append(Document(
1533
+ doc_id=row[0],
1534
+ doc_name=row[1],
1535
+ structure=structure,
1536
+ doc_description=row[2] or "",
1537
+ metadata={"source_path": row[3] or ""},
1538
+ source_type=row[4] or "",
1539
+ ))
1540
+ return documents
1541
+
1542
+ def delete_document(self, doc_id: str) -> bool:
1543
+ """Delete a document and all its indexed data from the DB atomically.
1544
+
1545
+ Clears all four storage locations in a single transaction:
1546
+ - ``fts_nodes`` (FTS5 inverted index entries)
1547
+ - ``nodes`` (structured node metadata)
1548
+ - ``documents`` (tree structure + document metadata)
1549
+ - ``index_meta`` (incremental indexing fingerprint)
1550
+
1551
+ Clearing ``index_meta`` is critical: without it the incremental indexing
1552
+ logic would see the file as already-processed and silently skip re-indexing
1553
+ it after a subsequent ``index()`` call.
1554
+
1555
+ Args:
1556
+ doc_id: document identifier to delete.
1557
+
1558
+ Returns:
1559
+ ``True`` if the document existed and was deleted, ``False`` if it was
1560
+ not found (operation is idempotent — no exception is raised).
1561
+ """
1562
+ # Check existence before entering the transaction so we can return a
1563
+ # meaningful bool without relying on changes_count() across all tables.
1564
+ row = self._conn.execute(
1565
+ "SELECT source_path FROM documents WHERE doc_id = ?", (doc_id,)
1566
+ ).fetchone()
1567
+
1568
+ if row is None:
1569
+ logger.warning("delete_document: doc_id=%r not found, nothing deleted", doc_id)
1570
+ return False
1571
+
1572
+ source_path = row[0] or ""
1573
+
1574
+ try:
1575
+ # Single atomic transaction — all-or-nothing.
1576
+ with self._conn:
1577
+ # 1. FTS5 virtual table: must delete by rowid (UNINDEXED columns
1578
+ # cannot be used in a WHERE clause for DELETE on FTS5 tables).
1579
+ if self._use_fts5:
1580
+ old_rowids = self._conn.execute(
1581
+ "SELECT rowid FROM fts_nodes WHERE doc_id = ?", (doc_id,)
1582
+ ).fetchall()
1583
+ if old_rowids:
1584
+ placeholders = ",".join("?" for _ in old_rowids)
1585
+ self._conn.execute(
1586
+ f"DELETE FROM fts_nodes WHERE rowid IN ({placeholders})",
1587
+ [r[0] for r in old_rowids],
1588
+ )
1589
+ else:
1590
+ self._conn.execute("DELETE FROM fts_nodes WHERE doc_id = ?", (doc_id,))
1591
+
1592
+ # 2. Structured node metadata.
1593
+ self._conn.execute("DELETE FROM nodes WHERE doc_id = ?", (doc_id,))
1594
+
1595
+ # 3. Document record (tree structure + metadata).
1596
+ self._conn.execute("DELETE FROM documents WHERE doc_id = ?", (doc_id,))
1597
+
1598
+ # 4. Incremental index fingerprint — CRITICAL: must be cleared so
1599
+ # that the next index() call re-processes this file rather than
1600
+ # skipping it as "already indexed".
1601
+ if source_path:
1602
+ self._conn.execute(
1603
+ "DELETE FROM index_meta WHERE source_path = ?", (source_path,)
1604
+ )
1605
+
1606
+ except sqlite3.DatabaseError as e:
1607
+ logger.error(
1608
+ "delete_document failed for doc_id=%r (source_path=%r): %s",
1609
+ doc_id, source_path, e,
1610
+ )
1611
+ raise
1612
+
1613
+ logger.info("Deleted document doc_id=%r (source_path=%r)", doc_id, source_path)
1614
+ return True
1615
+
1616
+ def remove_document(self, doc_id: str) -> None:
1617
+ """Remove a document from the DB.
1618
+
1619
+ .. deprecated::
1620
+ Use :meth:`delete_document` instead. ``delete_document`` fixes a
1621
+ bug where ``index_meta`` was not cleared, uses an atomic transaction,
1622
+ and returns a boolean indicating whether the document existed.
1623
+ """
1624
+ self.delete_document(doc_id)
1625
+
1626
+ def find_doc_by_fingerprint(self, file_hash: str, exclude_paths: Optional[set] = None) -> Optional[str]:
1627
+ """Find a doc whose ``index_meta.file_hash`` matches ``file_hash``.
1628
+
1629
+ Used by the indexer to detect file moves/renames: if the same fingerprint
1630
+ already exists under a different source_path, we can update the path
1631
+ instead of re-parsing+re-indexing the file.
1632
+
1633
+ Args:
1634
+ file_hash: target fingerprint string.
1635
+ exclude_paths: source_paths to skip (e.g. paths still on disk).
1636
+
1637
+ Returns:
1638
+ doc_id of the matching document, or None if no candidate.
1639
+ """
1640
+ rows = self._conn.execute(
1641
+ "SELECT source_path FROM index_meta WHERE file_hash = ?",
1642
+ (file_hash,),
1643
+ ).fetchall()
1644
+ for (sp,) in rows:
1645
+ if exclude_paths and sp in exclude_paths:
1646
+ continue
1647
+ doc_id = self.get_doc_id_by_source_path(sp)
1648
+ if doc_id is not None:
1649
+ return doc_id
1650
+ return None
1651
+
1652
+ def update_source_path(self, doc_id: str, new_source_path: str) -> None:
1653
+ """Atomically remap a document's source_path (move/rename support)."""
1654
+ with self._conn:
1655
+ old_row = self._conn.execute(
1656
+ "SELECT source_path FROM documents WHERE doc_id = ?", (doc_id,)
1657
+ ).fetchone()
1658
+ self._conn.execute(
1659
+ "UPDATE documents SET source_path = ? WHERE doc_id = ?",
1660
+ (new_source_path, doc_id),
1661
+ )
1662
+ if old_row and old_row[0]:
1663
+ self._conn.execute(
1664
+ "DELETE FROM index_meta WHERE source_path = ?", (old_row[0],)
1665
+ )
1666
+
1667
+ def rename_document(
1668
+ self,
1669
+ old_doc_id: str,
1670
+ new_doc_id: str,
1671
+ new_doc_name: str,
1672
+ new_source_path: str,
1673
+ ) -> bool:
1674
+ """Rename a doc in place across nodes / fts_nodes / documents / index_meta.
1675
+
1676
+ Used by `build_index`'s move-detection pre-pass when a file with an
1677
+ unchanged content fingerprint shows up under a new path. Keeps the
1678
+ doc identity (`doc_id` and `doc_name`) consistent with the new file
1679
+ basename so callers don't see stale names after a rename.
1680
+
1681
+ Returns ``False`` (and writes nothing) when:
1682
+ - the original ``old_doc_id`` no longer exists, or
1683
+ - ``new_doc_id`` is already taken by a *different* document — in
1684
+ that case the caller should fall back to a full re-index.
1685
+ """
1686
+ old_row = self._conn.execute(
1687
+ "SELECT doc_id, source_path FROM documents WHERE doc_id = ?",
1688
+ (old_doc_id,),
1689
+ ).fetchone()
1690
+ if old_row is None:
1691
+ return False
1692
+ old_source_path = old_row[1] or ""
1693
+
1694
+ if new_doc_id != old_doc_id:
1695
+ clash = self._conn.execute(
1696
+ "SELECT 1 FROM documents WHERE doc_id = ?", (new_doc_id,)
1697
+ ).fetchone()
1698
+ if clash is not None:
1699
+ return False
1700
+
1701
+ with self._conn:
1702
+ if new_doc_id != old_doc_id:
1703
+ self._conn.execute(
1704
+ "UPDATE nodes SET doc_id = ? WHERE doc_id = ?",
1705
+ (new_doc_id, old_doc_id),
1706
+ )
1707
+ self._conn.execute(
1708
+ "UPDATE fts_nodes SET doc_id = ? WHERE doc_id = ?",
1709
+ (new_doc_id, old_doc_id),
1710
+ )
1711
+ self._conn.execute(
1712
+ "UPDATE documents SET doc_id = ?, doc_name = ?, source_path = ? "
1713
+ "WHERE doc_id = ?",
1714
+ (new_doc_id, new_doc_name, new_source_path, old_doc_id),
1715
+ )
1716
+ else:
1717
+ self._conn.execute(
1718
+ "UPDATE documents SET doc_name = ?, source_path = ? WHERE doc_id = ?",
1719
+ (new_doc_name, new_source_path, old_doc_id),
1720
+ )
1721
+ if old_source_path and old_source_path != new_source_path:
1722
+ self._conn.execute(
1723
+ "DELETE FROM index_meta WHERE source_path = ?", (old_source_path,)
1724
+ )
1725
+
1726
+ return True
1727
+
1728
+ def delete_documents(self, doc_ids: list[str]) -> int:
1729
+ """Batch-delete multiple documents in a single transaction.
1730
+
1731
+ Significantly faster than calling ``delete_document`` in a loop because
1732
+ every table is hit once with an ``IN (...)`` clause.
1733
+
1734
+ Returns the number of documents that actually existed and were removed.
1735
+ """
1736
+ if not doc_ids:
1737
+ return 0
1738
+
1739
+ placeholders = ",".join("?" for _ in doc_ids)
1740
+ existing_rows = self._conn.execute(
1741
+ f"SELECT doc_id, source_path FROM documents WHERE doc_id IN ({placeholders})",
1742
+ doc_ids,
1743
+ ).fetchall()
1744
+ if not existing_rows:
1745
+ return 0
1746
+
1747
+ existing_ids = [r[0] for r in existing_rows]
1748
+ existing_paths = [r[1] for r in existing_rows if r[1]]
1749
+ ph_e = ",".join("?" for _ in existing_ids)
1750
+
1751
+ with self._conn:
1752
+ if self._use_fts5:
1753
+ old_rowids = self._conn.execute(
1754
+ f"SELECT rowid FROM fts_nodes WHERE doc_id IN ({ph_e})",
1755
+ existing_ids,
1756
+ ).fetchall()
1757
+ if old_rowids:
1758
+ # Batch delete in chunks to avoid SQLite's variable limit (~999)
1759
+ _MAX_VARS = 500
1760
+ rowid_list = [r[0] for r in old_rowids]
1761
+ for i in range(0, len(rowid_list), _MAX_VARS):
1762
+ chunk = rowid_list[i:i + _MAX_VARS]
1763
+ ph2 = ",".join("?" for _ in chunk)
1764
+ self._conn.execute(
1765
+ f"DELETE FROM fts_nodes WHERE rowid IN ({ph2})",
1766
+ chunk,
1767
+ )
1768
+ else:
1769
+ self._conn.execute(
1770
+ f"DELETE FROM fts_nodes WHERE doc_id IN ({ph_e})", existing_ids
1771
+ )
1772
+ self._conn.execute(
1773
+ f"DELETE FROM nodes WHERE doc_id IN ({ph_e})", existing_ids
1774
+ )
1775
+ self._conn.execute(
1776
+ f"DELETE FROM documents WHERE doc_id IN ({ph_e})", existing_ids
1777
+ )
1778
+ self._conn.execute(
1779
+ f"DELETE FROM pst_email_meta WHERE doc_id IN ({ph_e})", existing_ids
1780
+ )
1781
+ if existing_paths:
1782
+ ph_p = ",".join("?" for _ in existing_paths)
1783
+ self._conn.execute(
1784
+ f"DELETE FROM index_meta WHERE source_path IN ({ph_p})",
1785
+ existing_paths,
1786
+ )
1787
+
1788
+ logger.info("Batch-deleted %d document(s)", len(existing_ids))
1789
+ return len(existing_ids)
1790
+
1791
+ def get_doc_id_by_source_path(self, source_path: str) -> Optional[str]:
1792
+ """Look up a doc_id from a source file path.
1793
+
1794
+ Useful when callers know the file path but not the internal doc_id.
1795
+
1796
+ Args:
1797
+ source_path: absolute path of the source file.
1798
+
1799
+ Returns:
1800
+ The ``doc_id`` string, or ``None`` if no document matches.
1801
+ """
1802
+ row = self._conn.execute(
1803
+ "SELECT doc_id FROM documents WHERE source_path = ?", (source_path,)
1804
+ ).fetchone()
1805
+ return row[0] if row else None
1806
+
1807
+ def get_doc_ids_by_source_prefix(self, prefix: str) -> list[str]:
1808
+ """Return doc_ids whose source_path starts with ``prefix``.
1809
+
1810
+ Used for multi-document sources (e.g. PST archives whose email docs
1811
+ get derived paths ``<file>#<entry_id>``) — cascade delete / replace.
1812
+ """
1813
+ rows = self._conn.execute(
1814
+ "SELECT doc_id FROM documents WHERE instr(source_path, ?) = 1",
1815
+ (prefix,),
1816
+ ).fetchall()
1817
+ return [r[0] for r in rows]
1818
+
1819
+ def has_docs_with_source_prefix(self, prefix: str) -> bool:
1820
+ """Whether any document's source_path starts with ``prefix``."""
1821
+ row = self._conn.execute(
1822
+ "SELECT 1 FROM documents WHERE instr(source_path, ?) = 1 LIMIT 1",
1823
+ (prefix,),
1824
+ ).fetchone()
1825
+ return row is not None
1826
+
1827
+ def count_docs_with_source_prefix(self, prefix: str) -> int:
1828
+ """Count documents whose source_path starts with ``prefix``."""
1829
+ row = self._conn.execute(
1830
+ "SELECT COUNT(*) FROM documents WHERE instr(source_path, ?) = 1",
1831
+ (prefix,),
1832
+ ).fetchone()
1833
+ return row[0] if row else 0
1834
+
1835
+ def list_doc_names_by_source_prefix(self, prefix: str, limit: int) -> list[str]:
1836
+ """List doc_names whose source_path starts with ``prefix`` (ordered, capped)."""
1837
+ rows = self._conn.execute(
1838
+ "SELECT doc_name FROM documents WHERE instr(source_path, ?) = 1 "
1839
+ "ORDER BY doc_name LIMIT ?",
1840
+ (prefix, limit),
1841
+ ).fetchall()
1842
+ return [r[0] for r in rows]
1843
+
1844
+ # -------------------------------------------------------------------
1845
+ # PST 邮件元数据(ADR-0005:邮件列表分页 + 附件下载清单)
1846
+ # -------------------------------------------------------------------
1847
+
1848
+ def upsert_email_meta(self, doc_id: str, pst_path: str, meta: dict) -> None:
1849
+ """写入一封邮件的元数据(随派生文档同事务批次提交)。
1850
+
1851
+ Args:
1852
+ doc_id: 派生文档 id(``<pst文档id>__<entry_id>``)
1853
+ pst_path: 物理 PST 绝对路径(列表查询的分组键)
1854
+ meta: pst_parser 产出的 email_meta dict
1855
+ """
1856
+ import json as _json
1857
+
1858
+ self._conn.execute(
1859
+ """INSERT OR REPLACE INTO pst_email_meta
1860
+ (doc_id, pst_path, entry_id, subject, sender, date, folder, attachments_json)
1861
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?)""",
1862
+ (
1863
+ doc_id,
1864
+ pst_path,
1865
+ str(meta.get("entry_id", "")),
1866
+ meta.get("subject", ""),
1867
+ meta.get("sender", ""),
1868
+ meta.get("date", ""),
1869
+ meta.get("folder", ""),
1870
+ _json.dumps(meta.get("attachments") or [], ensure_ascii=False),
1871
+ ),
1872
+ )
1873
+
1874
+ def list_email_meta(
1875
+ self, pst_path: str, offset: int = 0, limit: int = 50
1876
+ ) -> tuple[int, list[dict]]:
1877
+ """分页列出一个 PST 的邮件元数据(日期倒序,无日期排尾)。
1878
+
1879
+ Returns:
1880
+ (total, rows):rows 含 entry_id/subject/sender/date/folder/doc_id。
1881
+ """
1882
+ total = self._conn.execute(
1883
+ "SELECT COUNT(*) FROM pst_email_meta WHERE pst_path = ?", (pst_path,)
1884
+ ).fetchone()[0]
1885
+ rows = self._conn.execute(
1886
+ """SELECT entry_id, subject, sender, date, folder, doc_id
1887
+ FROM pst_email_meta WHERE pst_path = ?
1888
+ ORDER BY CASE WHEN date = '' THEN 1 ELSE 0 END, date DESC, entry_id DESC
1889
+ LIMIT ? OFFSET ?""",
1890
+ (pst_path, limit, offset),
1891
+ ).fetchall()
1892
+ return total, [
1893
+ {
1894
+ "entry_id": r[0],
1895
+ "subject": r[1],
1896
+ "sender": r[2],
1897
+ "date": r[3],
1898
+ "folder": r[4],
1899
+ "doc_id": r[5],
1900
+ }
1901
+ for r in rows
1902
+ ]
1903
+
1904
+ def get_email_attachments(self, doc_id: str) -> list[dict] | None:
1905
+ """按派生 doc_id 取附件清单([{name,size,stored,filename}]);无记录返回 None。"""
1906
+ row = self._conn.execute(
1907
+ "SELECT attachments_json FROM pst_email_meta WHERE doc_id = ?", (doc_id,)
1908
+ ).fetchone()
1909
+ return self._attachments_from_row(row)
1910
+
1911
+ def get_email_attachments_by_entry(
1912
+ self, pst_path: str, entry_id: str
1913
+ ) -> list[dict] | None:
1914
+ """按 (PST 绝对路径, entry_id) 取附件清单(web 层不知道 doc_id 时用)。"""
1915
+ row = self._conn.execute(
1916
+ "SELECT attachments_json FROM pst_email_meta WHERE pst_path = ? AND entry_id = ?",
1917
+ (pst_path, str(entry_id)),
1918
+ ).fetchone()
1919
+ return self._attachments_from_row(row)
1920
+
1921
+ @staticmethod
1922
+ def _attachments_from_row(row) -> list[dict] | None:
1923
+ import json as _json
1924
+
1925
+ if row is None:
1926
+ return None
1927
+ try:
1928
+ return _json.loads(row[0] or "[]")
1929
+ except _json.JSONDecodeError:
1930
+ return []
1931
+
1932
+ # -------------------------------------------------------------------
1933
+ # Index metadata (replaces _index_meta.json)
1934
+ # -------------------------------------------------------------------
1935
+
1936
+ def get_index_meta(self, source_path: str) -> Optional[str]:
1937
+ """Get the stored file hash for a source path.
1938
+
1939
+ Returns:
1940
+ File hash string, or None if not tracked.
1941
+ """
1942
+ row = self._conn.execute(
1943
+ "SELECT file_hash FROM index_meta WHERE source_path = ?",
1944
+ (source_path,),
1945
+ ).fetchone()
1946
+ return row[0] if row else None
1947
+
1948
+ def set_index_meta(self, source_path: str, file_hash: str) -> None:
1949
+ """Store/update the file hash for a source path."""
1950
+ self._conn.execute(
1951
+ "INSERT OR REPLACE INTO index_meta (source_path, file_hash) VALUES (?, ?)",
1952
+ (source_path, file_hash),
1953
+ )
1954
+ self._conn.commit()
1955
+
1956
+ def set_index_meta_batch(self, meta: dict[str, str]) -> None:
1957
+ """Batch store/update file hashes. Single transaction for performance."""
1958
+ self._conn.executemany(
1959
+ "INSERT OR REPLACE INTO index_meta (source_path, file_hash) VALUES (?, ?)",
1960
+ list(meta.items()),
1961
+ )
1962
+ self._conn.commit()
1963
+
1964
+ def get_all_index_meta(self) -> dict[str, str]:
1965
+ """Get all stored file hashes.
1966
+
1967
+ Returns:
1968
+ Dict mapping source_path -> file_hash.
1969
+ """
1970
+ rows = self._conn.execute("SELECT source_path, file_hash FROM index_meta").fetchall()
1971
+ return {r[0]: r[1] for r in rows}
1972
+
1973
+ # -------------------------------------------------------------------
1974
+ # FTS5 query expression builder
1975
+ # -------------------------------------------------------------------
1976
+
1977
+ @staticmethod
1978
+ def build_fts_expression(
1979
+ keywords: list[str],
1980
+ operator: str = "OR",
1981
+ column: Optional[str] = None,
1982
+ near_distance: Optional[int] = None,
1983
+ prefix: bool = False,
1984
+ ) -> str:
1985
+ """Build FTS5 match expression from keyword list.
1986
+
1987
+ Args:
1988
+ keywords: list of search terms
1989
+ operator: "AND" | "OR" | "NOT" (first keyword AND NOT others)
1990
+ column: optional column filter (e.g. "title", "body")
1991
+ near_distance: if set, uses NEAR(kw1 kw2, N) syntax
1992
+ prefix: if True, append ``*`` to each term for prefix matching.
1993
+ Prefix matching allows partial token matches: "conf*" matches
1994
+ "config", "configuration", "confirm", etc. Useful for
1995
+ autocomplete-style search or when query tokens may be substrings
1996
+ of indexed terms (e.g. "fts*" matches both "fts" and "fts5").
1997
+
1998
+ Returns:
1999
+ FTS5 match expression string
2000
+
2001
+ Examples::
2002
+
2003
+ build_fts_expression(["python", "async"], "AND")
2004
+ -> "python AND async"
2005
+
2006
+ build_fts_expression(["fts"], prefix=True)
2007
+ -> "fts*" # matches fts, fts5, ftsearch, ...
2008
+
2009
+ build_fts_expression(["fts", "python"], prefix=True, operator="OR")
2010
+ -> "fts* OR python*"
2011
+
2012
+ build_fts_expression(["machine", "learning"], column="title")
2013
+ -> "title : (machine OR learning)"
2014
+
2015
+ build_fts_expression(["deep", "learning"], near_distance=5)
2016
+ -> 'NEAR(deep learning, 5)'
2017
+ """
2018
+ if not keywords:
2019
+ return ""
2020
+
2021
+ # Escape FTS5 special characters in keywords
2022
+ safe_kws = []
2023
+ for kw in keywords:
2024
+ cleaned = _RE_FTS5_SPECIAL.sub("", kw.strip())
2025
+ if cleaned:
2026
+ # Tokenize for CJK
2027
+ tokenized = _tokenize_for_fts(cleaned)
2028
+ if tokenized.strip():
2029
+ term = tokenized.strip()
2030
+ if prefix:
2031
+ term = term + "*"
2032
+ safe_kws.append(term)
2033
+
2034
+ if not safe_kws:
2035
+ return ""
2036
+
2037
+ if near_distance is not None and len(safe_kws) >= 2:
2038
+ all_tokens = " ".join(safe_kws)
2039
+ expr = f"NEAR({all_tokens}, {near_distance})"
2040
+ elif operator == "NOT" and len(safe_kws) >= 2:
2041
+ expr = f"{safe_kws[0]} NOT {' NOT '.join(safe_kws[1:])}"
2042
+ else:
2043
+ expr = f" {operator} ".join(safe_kws)
2044
+
2045
+ if column:
2046
+ expr = f"{column} : ({expr})"
2047
+
2048
+ return expr
2049
+
2050
+ # -------------------------------------------------------------------
2051
+ # Maintenance
2052
+ # -------------------------------------------------------------------
2053
+
2054
+ def optimize(self) -> None:
2055
+ """Run FTS5 merge optimization for better query performance."""
2056
+ if not self._use_fts5:
2057
+ logger.debug("FTS5 not available, skipping optimize")
2058
+ return
2059
+ try:
2060
+ self._conn.execute("INSERT INTO fts_nodes(fts_nodes) VALUES('optimize')")
2061
+ self._conn.commit()
2062
+ logger.info("FTS5 index optimized")
2063
+ except sqlite3.OperationalError as e:
2064
+ logger.warning("FTS5 optimize failed: %s", e)
2065
+
2066
+ def rebuild(self) -> None:
2067
+ """Rebuild FTS5 index from scratch."""
2068
+ if not self._use_fts5:
2069
+ logger.debug("FTS5 not available, skipping rebuild")
2070
+ return
2071
+ try:
2072
+ self._conn.execute("INSERT INTO fts_nodes(fts_nodes) VALUES('rebuild')")
2073
+ self._conn.commit()
2074
+ logger.info("FTS5 index rebuilt")
2075
+ except sqlite3.OperationalError as e:
2076
+ logger.warning("FTS5 rebuild failed: %s", e)
2077
+
2078
+ def get_stats(self) -> dict:
2079
+ """Get index statistics."""
2080
+ doc_count = self._conn.execute("SELECT COUNT(*) FROM documents").fetchone()[0]
2081
+ node_count = self._conn.execute("SELECT COUNT(*) FROM nodes").fetchone()[0]
2082
+ return {
2083
+ "db_path": self._db_path,
2084
+ "document_count": doc_count,
2085
+ "node_count": node_count,
2086
+ }
2087
+
2088
+ def clear(self) -> None:
2089
+ """Clear all indexed data."""
2090
+ self._conn.execute("DELETE FROM fts_nodes")
2091
+ self._conn.execute("DELETE FROM nodes")
2092
+ self._conn.execute("DELETE FROM documents")
2093
+ self._conn.execute("DELETE FROM index_meta")
2094
+ self._conn.commit()
2095
+
2096
+ def is_document_indexed(self, doc_id: str) -> bool:
2097
+ """Check if a document is already indexed."""
2098
+ row = self._conn.execute(
2099
+ "SELECT 1 FROM documents WHERE doc_id = ?", (doc_id,)
2100
+ ).fetchone()
2101
+ return row is not None
2102
+
2103
+ def wal_checkpoint(self, mode: str = "TRUNCATE") -> None:
2104
+ """Force a WAL checkpoint to fold the ``-wal`` sidecar back into the DB.
2105
+
2106
+ Useful at the end of a long incremental indexing run so the ``-wal``
2107
+ file does not stay multi-GB after many small commits. ``TRUNCATE``
2108
+ is the strongest variant (always safe — only no-ops when readers hold
2109
+ the WAL open).
2110
+ """
2111
+ try:
2112
+ self._conn.execute(f"PRAGMA wal_checkpoint({mode})")
2113
+ except sqlite3.OperationalError as e:
2114
+ logger.debug("WAL checkpoint(%s) failed: %s", mode, e)
2115
+
2116
+ def verify_index(self) -> dict:
2117
+ """Cross-table consistency check.
2118
+
2119
+ Detects orphan rows that violate the four-table invariant:
2120
+ - ``nodes`` rows whose doc_id has no entry in ``documents``.
2121
+ - ``fts_nodes`` rows whose doc_id has no entry in ``documents``.
2122
+ - ``index_meta`` rows whose source_path has no entry in ``documents``.
2123
+ - ``documents`` rows with non-existent on-disk source_path.
2124
+
2125
+ Returns a report dict with ``healthy: bool`` and lists of problem ids.
2126
+ Use :meth:`repair_index` to clean orphans.
2127
+ """
2128
+ report: dict = {
2129
+ "healthy": True,
2130
+ "orphan_node_doc_ids": [],
2131
+ "orphan_fts_doc_ids": [],
2132
+ "orphan_meta_paths": [],
2133
+ "missing_source_paths": [],
2134
+ }
2135
+ cur = self._conn.execute(
2136
+ "SELECT DISTINCT n.doc_id FROM nodes n "
2137
+ "LEFT JOIN documents d ON n.doc_id = d.doc_id WHERE d.doc_id IS NULL"
2138
+ ).fetchall()
2139
+ report["orphan_node_doc_ids"] = [r[0] for r in cur]
2140
+ cur = self._conn.execute(
2141
+ "SELECT DISTINCT f.doc_id FROM fts_nodes f "
2142
+ "LEFT JOIN documents d ON f.doc_id = d.doc_id WHERE d.doc_id IS NULL"
2143
+ ).fetchall()
2144
+ report["orphan_fts_doc_ids"] = [r[0] for r in cur]
2145
+ cur = self._conn.execute(
2146
+ "SELECT m.source_path FROM index_meta m "
2147
+ "LEFT JOIN documents d ON m.source_path = d.source_path WHERE d.doc_id IS NULL"
2148
+ ).fetchall()
2149
+ report["orphan_meta_paths"] = [r[0] for r in cur]
2150
+ cur = self._conn.execute(
2151
+ "SELECT doc_id, source_path FROM documents WHERE source_path != ''"
2152
+ ).fetchall()
2153
+ report["missing_source_paths"] = [
2154
+ (doc_id, sp) for (doc_id, sp) in cur if not os.path.isfile(sp)
2155
+ ]
2156
+ report["healthy"] = not any(
2157
+ report[k] for k in
2158
+ ("orphan_node_doc_ids", "orphan_fts_doc_ids",
2159
+ "orphan_meta_paths", "missing_source_paths")
2160
+ )
2161
+ return report
2162
+
2163
+ def repair_index(self, drop_missing_files: bool = False) -> dict:
2164
+ """Remove orphan rows surfaced by :meth:`verify_index`.
2165
+
2166
+ Args:
2167
+ drop_missing_files: also delete documents whose source file no
2168
+ longer exists on disk (default False — those may be intentional
2169
+ in-memory loads or inaccessible mounts).
2170
+
2171
+ Returns:
2172
+ Dict with counts of removed orphan rows.
2173
+ """
2174
+ report = self.verify_index()
2175
+ removed = {"orphan_nodes": 0, "orphan_fts": 0, "orphan_meta": 0, "missing_files": 0}
2176
+
2177
+ with self._conn:
2178
+ if report["orphan_node_doc_ids"]:
2179
+ ph = ",".join("?" for _ in report["orphan_node_doc_ids"])
2180
+ cur = self._conn.execute(
2181
+ f"DELETE FROM nodes WHERE doc_id IN ({ph})",
2182
+ report["orphan_node_doc_ids"],
2183
+ )
2184
+ removed["orphan_nodes"] = cur.rowcount
2185
+ if report["orphan_fts_doc_ids"]:
2186
+ ph = ",".join("?" for _ in report["orphan_fts_doc_ids"])
2187
+ if self._use_fts5:
2188
+ rowids = self._conn.execute(
2189
+ f"SELECT rowid FROM fts_nodes WHERE doc_id IN ({ph})",
2190
+ report["orphan_fts_doc_ids"],
2191
+ ).fetchall()
2192
+ if rowids:
2193
+ ph2 = ",".join("?" for _ in rowids)
2194
+ cur = self._conn.execute(
2195
+ f"DELETE FROM fts_nodes WHERE rowid IN ({ph2})",
2196
+ [r[0] for r in rowids],
2197
+ )
2198
+ removed["orphan_fts"] = cur.rowcount
2199
+ else:
2200
+ cur = self._conn.execute(
2201
+ f"DELETE FROM fts_nodes WHERE doc_id IN ({ph})",
2202
+ report["orphan_fts_doc_ids"],
2203
+ )
2204
+ removed["orphan_fts"] = cur.rowcount
2205
+ if report["orphan_meta_paths"]:
2206
+ ph = ",".join("?" for _ in report["orphan_meta_paths"])
2207
+ cur = self._conn.execute(
2208
+ f"DELETE FROM index_meta WHERE source_path IN ({ph})",
2209
+ report["orphan_meta_paths"],
2210
+ )
2211
+ removed["orphan_meta"] = cur.rowcount
2212
+
2213
+ if drop_missing_files and report["missing_source_paths"]:
2214
+ doc_ids = [doc_id for (doc_id, _) in report["missing_source_paths"]]
2215
+ removed["missing_files"] = self.delete_documents(doc_ids)
2216
+
2217
+ return removed
2218
+
2219
+ def get_unindexed_doc_ids(self, doc_ids: list[str]) -> set[str]:
2220
+ """Return the subset of doc_ids that are NOT yet indexed.
2221
+
2222
+ Uses a single SQL query instead of per-document checks.
2223
+ """
2224
+ if not doc_ids:
2225
+ return set()
2226
+ placeholders = ",".join("?" for _ in doc_ids)
2227
+ rows = self._conn.execute(
2228
+ f"SELECT doc_id FROM documents WHERE doc_id IN ({placeholders})",
2229
+ doc_ids,
2230
+ ).fetchall()
2231
+ indexed = {r[0] for r in rows}
2232
+ return set(doc_ids) - indexed
2233
+
2234
+
2235
+ def _extract_match_snippet(text: str, query: str, use_regex: bool, size: int = 300) -> str:
2236
+ """Extract a snippet of *size* chars centered around the first match."""
2237
+ if len(text) <= size:
2238
+ return text
2239
+ pos = -1
2240
+ if use_regex:
2241
+ m = re.search(query, text, re.IGNORECASE)
2242
+ if m:
2243
+ pos = m.start()
2244
+ else:
2245
+ pos = text.lower().find(query.lower())
2246
+ if pos < 0:
2247
+ pos = 0
2248
+ half = size // 2
2249
+ start = max(0, pos - half)
2250
+ end = min(len(text), start + size)
2251
+ start = max(0, end - size)
2252
+ return text[start:end]
2253
+
2254
+
2255
+ # ---------------------------------------------------------------------------
2256
+ # Global FTS5 index singleton
2257
+ # ---------------------------------------------------------------------------
2258
+
2259
+ _global_fts: Optional[FTS5Index] = None
2260
+
2261
+
2262
+ def get_fts_index(db_path: Optional[str] = None, weights: Optional[dict] = None) -> FTS5Index:
2263
+ """Get or create the global FTS5 index.
2264
+
2265
+ Args:
2266
+ db_path: database path. If None, uses in-memory database.
2267
+ Pass a file path for persistent indexing across sessions.
2268
+ weights: column weight overrides for bm25() ranking.
2269
+ """
2270
+ global _global_fts
2271
+ if _global_fts is not None:
2272
+ # If db_path changed, re-create the singleton
2273
+ requested = db_path or ":memory:"
2274
+ if _global_fts.db_path != requested:
2275
+ _global_fts.close()
2276
+ _global_fts = None
2277
+ if _global_fts is None:
2278
+ _global_fts = FTS5Index(db_path=db_path, weights=weights)
2279
+ return _global_fts
2280
+
2281
+
2282
+ def set_fts_index(index: FTS5Index) -> None:
2283
+ """Set the global FTS5 index instance."""
2284
+ global _global_fts
2285
+ _global_fts = index
2286
+
2287
+
2288
+ def reset_fts_index() -> None:
2289
+ """Close and reset the global FTS5 index."""
2290
+ global _global_fts
2291
+ if _global_fts is not None:
2292
+ _global_fts.close()
2293
+ _global_fts = None