treesearchlib 1.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. treesearch/__init__.py +53 -0
  2. treesearch/__main__.py +6 -0
  3. treesearch/_bin/pst-extract.exe +0 -0
  4. treesearch/cli.py +554 -0
  5. treesearch/config.py +206 -0
  6. treesearch/fts.py +2293 -0
  7. treesearch/heuristics.py +425 -0
  8. treesearch/indexer.py +2038 -0
  9. treesearch/parsers/__init__.py +62 -0
  10. treesearch/parsers/anydoc_parser.py +193 -0
  11. treesearch/parsers/ast_parser.py +136 -0
  12. treesearch/parsers/docx_parser.py +304 -0
  13. treesearch/parsers/email_html_md.py +60 -0
  14. treesearch/parsers/excel_parser.py +218 -0
  15. treesearch/parsers/html_parser.py +172 -0
  16. treesearch/parsers/image_metadata.py +345 -0
  17. treesearch/parsers/image_parser.py +59 -0
  18. treesearch/parsers/image_store.py +182 -0
  19. treesearch/parsers/markitdown_parser.py +258 -0
  20. treesearch/parsers/mhtml_parser.py +108 -0
  21. treesearch/parsers/pdf_parser.py +409 -0
  22. treesearch/parsers/pst_attachment_store.py +156 -0
  23. treesearch/parsers/pst_parser.py +733 -0
  24. treesearch/parsers/registry.py +405 -0
  25. treesearch/parsers/treesitter_parser.py +433 -0
  26. treesearch/pathutil.py +227 -0
  27. treesearch/py.typed +0 -0
  28. treesearch/ripgrep.py +159 -0
  29. treesearch/search.py +935 -0
  30. treesearch/tokenizer.py +176 -0
  31. treesearch/tree.py +393 -0
  32. treesearch/tree_searcher.py +1006 -0
  33. treesearch/treesearch.py +574 -0
  34. treesearch/watch.py +305 -0
  35. treesearchlib-1.1.0.dist-info/METADATA +124 -0
  36. treesearchlib-1.1.0.dist-info/RECORD +39 -0
  37. treesearchlib-1.1.0.dist-info/WHEEL +5 -0
  38. treesearchlib-1.1.0.dist-info/entry_points.txt +2 -0
  39. treesearchlib-1.1.0.dist-info/top_level.txt +1 -0
treesearch/indexer.py ADDED
@@ -0,0 +1,2038 @@
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ @author:XuMing(xuming624@qq.com)
4
+ @description: Async-first document indexer. Builds tree structure from Markdown or plain text.
5
+
6
+ Supports batch indexing via ``build_index()`` which accepts glob patterns and
7
+ processes multiple files concurrently.
8
+ """
9
+ import asyncio
10
+ import hashlib
11
+ import json
12
+ import logging
13
+ import os
14
+ import re
15
+ import time
16
+ from contextlib import nullcontext
17
+ from dataclasses import dataclass, field
18
+ from pathlib import Path
19
+ from typing import Optional
20
+
21
+ from tqdm import tqdm
22
+
23
+ from .tree import (
24
+ Document, assign_node_ids, flatten_tree, format_structure, remove_fields,
25
+ )
26
+ from .pathutil import resolve_paths, DEFAULT_IGNORE_DIRS, MAX_DIR_FILES, shadow_md_path
27
+
28
+ logger = logging.getLogger(__name__)
29
+
30
+ # PST 解析器输出格式版本盐(折进 PST 文件的指纹):pst_parser 的输出格式
31
+ # 变化(正文转写/元数据/附件落盘,ADR-0005)时 bump,让旧 PST 索引在下次
32
+ # 增量索引自动重建——只影响 PST,不像 INDEX_SCHEMA_VERSION 那样连累全库。
33
+ # 历史:
34
+ # ":pst2" — 2026-07-29 ADR-0005:HTML 正文转写 + 附件全量落盘 + 邮件元数据。
35
+ # ":pst3" — 2026-07-30:邮件头转储正文(Outlook 系统报告类)可读化重排
36
+ # (_reformat_header_dump:解码 encoded-word、折叠超长地址列表)。
37
+ # ":pst4" — 2026-07-30:头转储检测前移到 HTML 转写之前 + 逻辑块解析
38
+ # (body_html 装天书的路径不再漏检)。
39
+ # ":pst5" — 2026-07-30:头转储带完整 RFC822 源码时按 MIME 解析提取内嵌
40
+ # 正文(base64 附件只列名);头行正则兼容空值头(Subject:)。
41
+ PST_PARSER_FINGERPRINT_SALT = ":pst5"
42
+
43
+
44
+ # ============================================================================
45
+ # Shared helpers
46
+ # ============================================================================
47
+
48
+ def _generate_shadow_md(binary_path: str) -> None:
49
+ """Convert a binary file to a hidden Markdown text copy using markitdown.
50
+
51
+ The shadow file (``._<name>.<ext>.md``) lives alongside the source and is
52
+ used by ripgrep fallback search when FTS5 has no results.
53
+
54
+ Skips generation if the shadow file is newer than the source (incremental).
55
+ """
56
+ md_path = shadow_md_path(binary_path)
57
+ # Incremental: skip if shadow is up-to-date
58
+ try:
59
+ if os.path.exists(md_path) and os.path.getmtime(md_path) >= os.path.getmtime(binary_path):
60
+ return
61
+ except OSError:
62
+ pass
63
+
64
+ try:
65
+ from markitdown import MarkItDown
66
+ except ImportError:
67
+ logger.debug("markitdown not installed, skipping shadow MD for %s", binary_path)
68
+ return
69
+
70
+ md = MarkItDown()
71
+ result = md.convert(binary_path)
72
+ text = result.text_content or ""
73
+ if not text.strip():
74
+ logger.debug("markitdown returned empty content for %s", binary_path)
75
+ return
76
+
77
+ try:
78
+ with open(md_path, "w", encoding="utf-8") as f:
79
+ f.write(text)
80
+ logger.debug("Generated shadow MD: %s", md_path)
81
+ except OSError as e:
82
+ logger.warning("Failed to write shadow MD %s: %s", md_path, e)
83
+
84
+
85
+ def _children_indices(node_list: list[dict], parent_idx: int, parent_level: int) -> list[int]:
86
+ """Return indices of all descendants of node_list[parent_idx]."""
87
+ indices = []
88
+ for j in range(parent_idx + 1, len(node_list)):
89
+ if node_list[j]["level"] <= parent_level:
90
+ break
91
+ indices.append(j)
92
+ return indices
93
+
94
+
95
+ # ============================================================================
96
+ # Summary generation (shared by MD and Text)
97
+ # ============================================================================
98
+
99
+ def _summarize_node(node: dict, threshold: int = 600) -> str:
100
+ """Generate a summary for a single node. Short nodes use their own text.
101
+
102
+ Args:
103
+ threshold: character count threshold. Nodes shorter than this use full text as summary.
104
+ For long nodes: head 250 chars + tail 100 chars (captures intro and conclusion).
105
+ """
106
+ text = node.get("text", "")
107
+ if len(text) < threshold:
108
+ return text
109
+ head = text[:250].replace("\n", " ").strip()
110
+ tail = text[-100:].replace("\n", " ").strip()
111
+ return f"{head} ... {tail}"
112
+
113
+
114
+ def generate_summaries(structure, threshold: int = 600):
115
+ """Generate summaries for all nodes in a tree."""
116
+ nodes = flatten_tree(structure)
117
+ summaries = [_summarize_node(n, threshold=threshold) for n in nodes]
118
+
119
+ for node, summary in zip(nodes, summaries):
120
+ if node.get("nodes"):
121
+ node["prefix_summary"] = summary
122
+ else:
123
+ node["summary"] = summary
124
+ return structure
125
+
126
+
127
+ def generate_doc_description(structure) -> str:
128
+ """Generate a document description from its tree structure (no LLM).
129
+
130
+ Extracts top-level titles and first substantial text paragraph.
131
+ """
132
+ nodes = flatten_tree(structure)
133
+ titles = [n.get("title", "") for n in nodes if n.get("title")][:5]
134
+ title_str = " > ".join(titles)
135
+ text = ""
136
+ for n in nodes:
137
+ t = n.get("text", "")
138
+ if t and len(t) > 20:
139
+ text = t[:200]
140
+ break
141
+ return f"{title_str}. {text}" if text else title_str
142
+
143
+
144
+ def _finalize_tree(
145
+ tree,
146
+ doc_name: str,
147
+ source_path: str = "",
148
+ source_type: str = "",
149
+ *,
150
+ if_add_node_id: bool = True,
151
+ if_add_node_summary: bool = True,
152
+ summary_chars_threshold: int = 600,
153
+ if_add_node_text: bool = False,
154
+ if_add_doc_description: bool = False,
155
+ ) -> dict:
156
+ """Common post-processing for all *_to_tree functions.
157
+
158
+ Steps: split_oversized_nodes -> assign_node_ids -> format_structure -> generate_summaries -> doc_description.
159
+ """
160
+ # Split oversized nodes before assigning IDs (so sub-nodes get proper IDs)
161
+ from .config import get_config
162
+ max_node_chars = get_config().max_node_chars
163
+ if max_node_chars:
164
+ tree = _split_oversized_nodes(tree, max_node_chars)
165
+
166
+ if if_add_node_id:
167
+ assign_node_ids(tree)
168
+
169
+ base_order = ["title", "node_id", "summary", "prefix_summary"]
170
+ text_fields = ["text"] if if_add_node_text or if_add_node_summary else []
171
+ tail_fields = ["line_start", "line_end", "nodes"]
172
+ order = base_order + text_fields + tail_fields
173
+
174
+ tree = format_structure(tree, order=order)
175
+
176
+ if if_add_node_summary:
177
+ logger.debug("Generating summaries...")
178
+ tree = generate_summaries(tree, threshold=summary_chars_threshold)
179
+ if not if_add_node_text:
180
+ order_no_text = [f for f in order if f != "text"]
181
+ tree = format_structure(tree, order=order_no_text)
182
+
183
+ result = {"doc_name": doc_name, "structure": tree}
184
+ if source_path:
185
+ result["source_path"] = source_path
186
+ if source_type:
187
+ result["source_type"] = source_type
188
+
189
+ if if_add_doc_description:
190
+ logger.debug("Generating document description...")
191
+ result["doc_description"] = generate_doc_description(tree)
192
+
193
+ return result
194
+
195
+
196
+ # ============================================================================
197
+ # Markdown indexer
198
+ # ============================================================================
199
+
200
+ def _extract_md_headings(content: str) -> tuple[list[dict], list[str]]:
201
+ """Extract heading markers from Markdown content."""
202
+ header_re = re.compile(r"^(#{1,6})\s+(.+)$")
203
+ code_fence = re.compile(r"^```")
204
+ markers = []
205
+ lines = content.split("\n")
206
+ in_code = False
207
+
208
+ for num, line in enumerate(lines, 1):
209
+ stripped = line.strip()
210
+ if code_fence.match(stripped):
211
+ in_code = not in_code
212
+ continue
213
+ if in_code or not stripped:
214
+ continue
215
+ m = header_re.match(stripped)
216
+ if m:
217
+ markers.append({
218
+ "title": m.group(2).strip(),
219
+ "line_num": num,
220
+ "level": len(m.group(1)),
221
+ })
222
+ return markers, lines
223
+
224
+
225
+ def _cut_md_text(markers: list[dict], lines: list[str]) -> list[dict]:
226
+ """Cut text content between headings."""
227
+ nodes = []
228
+ for i, mk in enumerate(markers):
229
+ start = mk["line_num"] - 1
230
+ end = markers[i + 1]["line_num"] - 1 if i + 1 < len(markers) else len(lines)
231
+ nodes.append({
232
+ "title": mk["title"],
233
+ "line_num": mk["line_num"],
234
+ "line_start": mk["line_num"],
235
+ "line_end": end,
236
+ "level": mk["level"],
237
+ "text": "\n".join(lines[start:end]).strip(),
238
+ })
239
+ return nodes
240
+
241
+
242
+ def _update_char_counts(node_list: list[dict]) -> list[dict]:
243
+ """Compute cumulative character counts (self + descendants) for thinning."""
244
+ for i in range(len(node_list) - 1, -1, -1):
245
+ text = node_list[i].get("text", "")
246
+ for ci in _children_indices(node_list, i, node_list[i]["level"]):
247
+ ct = node_list[ci].get("text", "")
248
+ if ct:
249
+ text += "\n" + ct
250
+ node_list[i]["text_char_count"] = len(text)
251
+ return node_list
252
+
253
+
254
+ def _thin_tree(node_list: list[dict], min_chars: int) -> list[dict]:
255
+ """Merge small sub-trees into their parent nodes."""
256
+ to_remove = set()
257
+ for i in range(len(node_list) - 1, -1, -1):
258
+ if i in to_remove:
259
+ continue
260
+ if node_list[i].get("text_char_count", 0) < min_chars:
261
+ children = _children_indices(node_list, i, node_list[i]["level"])
262
+ merged_parts = []
263
+ for ci in sorted(children):
264
+ if ci not in to_remove:
265
+ ct = node_list[ci].get("text", "")
266
+ if ct.strip():
267
+ merged_parts.append(ct)
268
+ to_remove.add(ci)
269
+ if merged_parts:
270
+ base = node_list[i].get("text", "")
271
+ node_list[i]["text"] = base + "\n\n" + "\n\n".join(merged_parts) if base else "\n\n".join(merged_parts)
272
+ node_list[i]["text_char_count"] = len(node_list[i]["text"])
273
+
274
+ for idx in sorted(to_remove, reverse=True):
275
+ node_list.pop(idx)
276
+ return node_list
277
+
278
+
279
+ def _build_tree(node_list: list[dict]) -> list[dict]:
280
+ """Build hierarchical tree from flat node list using a stack algorithm."""
281
+ if not node_list:
282
+ return []
283
+ stack = []
284
+ roots = []
285
+ counter = 1
286
+
287
+ for node in node_list:
288
+ level = node["level"]
289
+ tree_node = {
290
+ "title": node["title"],
291
+ "node_id": str(counter),
292
+ "text": node.get("text", ""),
293
+ "line_start": node.get("line_start", node.get("line_num")),
294
+ "line_end": node.get("line_end"),
295
+ "nodes": [],
296
+ }
297
+ counter += 1
298
+
299
+ while stack and stack[-1][1] >= level:
300
+ stack.pop()
301
+
302
+ if not stack:
303
+ roots.append(tree_node)
304
+ else:
305
+ stack[-1][0]["nodes"].append(tree_node)
306
+
307
+ stack.append((tree_node, level))
308
+ return roots
309
+
310
+
311
+ # ============================================================================
312
+ # Structure-aware node splitting for oversized nodes
313
+ # ============================================================================
314
+
315
+ def _split_text_by_paragraphs(text: str, max_chars: int) -> list[str]:
316
+ """Split text into chunks at paragraph boundaries (double newline).
317
+
318
+ Each chunk stays under max_chars. Falls back to single newline
319
+ boundaries, then hard character cut if paragraphs are still too large.
320
+ """
321
+ if len(text) <= max_chars:
322
+ return [text]
323
+
324
+ # Try splitting by double newline (paragraph boundary)
325
+ chunks = _split_at_boundary(text, max_chars, "\n\n")
326
+ if chunks:
327
+ return chunks
328
+
329
+ # Fallback: split by single newline (line boundary)
330
+ chunks = _split_at_boundary(text, max_chars, "\n")
331
+ if chunks:
332
+ return chunks
333
+
334
+ # Last resort: hard character cut (should rarely happen)
335
+ return [text[i:i + max_chars] for i in range(0, len(text), max_chars)]
336
+
337
+
338
+ def _split_at_boundary(text: str, max_chars: int, separator: str) -> list[str]:
339
+ """Split text into chunks at the given separator, each under max_chars.
340
+
341
+ Returns None if any segment between separators exceeds max_chars
342
+ (caller should try a finer-grained separator).
343
+ """
344
+ segments = text.split(separator)
345
+ chunks = []
346
+ current = []
347
+ current_len = 0
348
+
349
+ for seg in segments:
350
+ seg_with_sep = (separator + seg) if current else seg
351
+ new_len = current_len + len(seg_with_sep)
352
+
353
+ if new_len > max_chars and current:
354
+ # Flush current chunk
355
+ chunks.append(separator.join(current))
356
+ current = [seg]
357
+ current_len = len(seg)
358
+ # If a single segment exceeds max_chars, this separator is too coarse
359
+ if current_len > max_chars:
360
+ return None
361
+ else:
362
+ current.append(seg)
363
+ current_len = new_len if current_len > 0 else len(seg)
364
+
365
+ if current:
366
+ chunks.append(separator.join(current))
367
+
368
+ return chunks if chunks else None
369
+
370
+
371
+ def _split_oversized_nodes(tree: list[dict], max_chars: int) -> list[dict]:
372
+ """Recursively split oversized tree nodes into smaller sub-nodes.
373
+
374
+ For leaf nodes with text exceeding max_chars:
375
+ - Split text at paragraph boundaries (structure-aware)
376
+ - Create child nodes with sequential titles: "Part 1", "Part 2", etc.
377
+ - Parent node retains the original title with empty text
378
+
379
+ For non-leaf nodes: recurse into children first, then check the parent's
380
+ own text (the text before the first child heading).
381
+ """
382
+ if not max_chars:
383
+ return tree
384
+
385
+ result = []
386
+ for node in tree:
387
+ # Recurse into children first
388
+ children = node.get("nodes", [])
389
+ if children:
390
+ node["nodes"] = _split_oversized_nodes(children, max_chars)
391
+
392
+ text = node.get("text", "")
393
+ if len(text) <= max_chars:
394
+ result.append(node)
395
+ continue
396
+
397
+ # Node text exceeds max_chars: split into sub-nodes
398
+ chunks = _split_text_by_paragraphs(text, max_chars)
399
+ if len(chunks) <= 1:
400
+ result.append(node)
401
+ continue
402
+
403
+ title = node.get("title", "")
404
+ line_start = node.get("line_start")
405
+ line_end = node.get("line_end")
406
+ existing_children = node.get("nodes", [])
407
+
408
+ # Create sub-nodes from text chunks
409
+ sub_nodes = []
410
+ for i, chunk in enumerate(chunks):
411
+ sub_node = {
412
+ "title": f"{title} (part {i + 1})",
413
+ "text": chunk,
414
+ "line_start": line_start,
415
+ "line_end": line_end,
416
+ "nodes": [],
417
+ }
418
+ sub_nodes.append(sub_node)
419
+
420
+ # Attach existing children to the last sub-node (they belong to the tail of the text)
421
+ if existing_children:
422
+ sub_nodes[-1]["nodes"] = existing_children
423
+
424
+ # Parent node becomes a container with truncated text
425
+ node["text"] = ""
426
+ node["nodes"] = sub_nodes
427
+ result.append(node)
428
+
429
+ return result
430
+
431
+
432
+ async def md_to_tree(
433
+ md_path: Optional[str] = None,
434
+ md_content: Optional[str] = None,
435
+ *,
436
+ if_thinning: bool = False,
437
+ min_thinning_chars: int = 15000,
438
+ if_add_node_summary: bool = True,
439
+ summary_chars_threshold: int = 600,
440
+ if_add_doc_description: bool = False,
441
+ if_add_node_text: bool = False,
442
+ if_add_node_id: bool = True,
443
+ **kwargs,
444
+ ) -> dict:
445
+ """
446
+ Build a tree index from a Markdown file or string.
447
+
448
+ Returns: {'doc_name': str, 'structure': list, 'doc_description'?: str}
449
+ """
450
+ if md_path and md_content:
451
+ raise ValueError("Specify only one of md_path or md_content")
452
+ if not md_path and not md_content:
453
+ raise ValueError("Must specify md_path or md_content")
454
+
455
+ if md_path:
456
+ with open(md_path, "r", encoding="utf-8", errors="replace") as f:
457
+ md_content = f.read()
458
+ doc_name = os.path.splitext(os.path.basename(md_path))[0]
459
+ else:
460
+ doc_name = "untitled"
461
+
462
+ logger.debug("Extracting headings from markdown...")
463
+ markers, lines = _extract_md_headings(md_content)
464
+ nodes = _cut_md_text(markers, lines)
465
+
466
+ if if_thinning and min_thinning_chars:
467
+ nodes = _update_char_counts(nodes)
468
+ logger.debug("Thinning tree (threshold=%d chars)...", min_thinning_chars)
469
+ nodes = _thin_tree(nodes, min_thinning_chars)
470
+
471
+ logger.debug("Building tree from %d nodes...", len(nodes))
472
+ tree = _build_tree(nodes)
473
+
474
+ # 无标题回退:为纯文本 Markdown 创建一个根节点
475
+ if not tree and md_content.strip():
476
+ total_lines = len(md_content.split("\n"))
477
+ tree = [{
478
+ "title": doc_name,
479
+ "node_id": "0",
480
+ "text": md_content.strip(),
481
+ "line_start": 1,
482
+ "line_end": total_lines,
483
+ "nodes": [],
484
+ }]
485
+
486
+ return _finalize_tree(
487
+ tree, doc_name,
488
+ source_path=os.path.abspath(md_path) if md_path else "",
489
+ if_add_node_id=if_add_node_id,
490
+ if_add_node_summary=if_add_node_summary,
491
+ summary_chars_threshold=summary_chars_threshold,
492
+ if_add_node_text=if_add_node_text,
493
+ if_add_doc_description=if_add_doc_description,
494
+ )
495
+
496
+
497
+ # ============================================================================
498
+ # Plain text indexer
499
+ # ============================================================================
500
+
501
+ # --- Heading detection patterns ---
502
+
503
+ _RE_NUMERIC = re.compile(r"^(?P<prefix>(?:\d+\.)+\d*)\s*(?P<title>.+)$")
504
+ _RE_PAREN_NUM = re.compile(r"^(?:\(?\d+\))\s+(?P<title>.+)$")
505
+ _RE_ROMAN = re.compile(r"^(?P<prefix>[IVXLCDM]+)\.\s+(?P<title>.+)$")
506
+ _RE_LETTER = re.compile(r"^(?P<prefix>[A-Z])[.)]\s+(?P<title>.+)$")
507
+ _RE_CN_SECTION = re.compile(r"^(?:第[一二三四五六七八九十百千万零\d]+[章节篇部])\s*(?P<title>.*)$")
508
+ _RE_CN_NUM = re.compile(r"^(?P<prefix>[一二三四五六七八九十百千万零]+)[、..]\s*(?P<title>.+)$")
509
+ _RE_CN_PAREN = re.compile(r"^[((](?P<prefix>[一二三四五六七八九十百千万零\d]+)[))]\s*(?P<title>.+)$")
510
+ _RE_RST_UNDERLINE = re.compile(r"^[=\-~^+#]{3,}$")
511
+ _RE_ALL_CAPS = re.compile(r"^[A-Z][A-Z\s\-:,&/]{2,}$")
512
+
513
+ _ROMAN_VALID = {
514
+ "I", "II", "III", "IV", "V", "VI", "VII", "VIII", "IX", "X",
515
+ "XI", "XII", "XIII", "XIV", "XV", "XVI", "XVII", "XVIII", "XIX", "XX",
516
+ }
517
+
518
+
519
+ def _is_short(line: str, limit: int = 80) -> bool:
520
+ return 0 < len(line.strip()) < limit
521
+
522
+
523
+ def _has_blank_neighbor(lines: list, idx: int) -> bool:
524
+ prev_blank = (idx == 0) or (not lines[idx - 1].strip())
525
+ next_blank = (idx >= len(lines) - 1) or (not lines[idx + 1].strip())
526
+ return prev_blank or next_blank
527
+
528
+
529
+ def _detect_headings(lines: list[str]) -> list[dict]:
530
+ """Detect headings from raw text lines using pattern matching."""
531
+ headings = []
532
+ in_code = False
533
+
534
+ for idx, raw in enumerate(lines):
535
+ line = raw.strip()
536
+ if line.startswith("```"):
537
+ in_code = not in_code
538
+ continue
539
+ if in_code or not line:
540
+ continue
541
+ num = idx + 1
542
+
543
+ # Chinese chapter/section
544
+ m = _RE_CN_SECTION.match(line)
545
+ if m:
546
+ level = 1 if any(c in line for c in "章篇部") else 2
547
+ headings.append({"title": line, "line_num": num, "level": level})
548
+ continue
549
+
550
+ m = _RE_CN_NUM.match(line)
551
+ if m:
552
+ headings.append({"title": line, "line_num": num, "level": 1})
553
+ continue
554
+
555
+ m = _RE_CN_PAREN.match(line)
556
+ if m:
557
+ headings.append({"title": line, "line_num": num, "level": 2})
558
+ continue
559
+
560
+ # Numeric hierarchical (e.g. "1.2 Introduction", "3.1.1 Methods")
561
+ # Require short line and title starts with a letter to avoid matching
562
+ # math expressions like "0.1 + (-0.6) + 0.9 = 0.4..."
563
+ m = _RE_NUMERIC.match(line)
564
+ if m and _is_short(line) and re.match(r"[A-Za-z\u4e00-\u9fff]", m.group("title")):
565
+ level = len(m.group("prefix").rstrip(".").split("."))
566
+ headings.append({"title": line, "line_num": num, "level": level})
567
+ continue
568
+
569
+ # Parenthesized number
570
+ m = _RE_PAREN_NUM.match(line)
571
+ if m:
572
+ headings.append({"title": line, "line_num": num, "level": 2})
573
+ continue
574
+
575
+ # Roman numeral
576
+ m = _RE_ROMAN.match(line)
577
+ if m and m.group("prefix") in _ROMAN_VALID:
578
+ headings.append({"title": line, "line_num": num, "level": 1})
579
+ continue
580
+
581
+ # Letter heading (e.g. "A. Introduction", "B) Methods")
582
+ # Only match short lines to avoid false positives like "D. All these works..."
583
+ m = _RE_LETTER.match(line)
584
+ if m and _is_short(line, 60):
585
+ headings.append({"title": line, "line_num": num, "level": 2})
586
+ continue
587
+
588
+ # RST underline style
589
+ if idx > 0 and _RE_RST_UNDERLINE.match(line):
590
+ prev = lines[idx - 1].strip()
591
+ if prev and _is_short(prev):
592
+ if not headings or headings[-1]["line_num"] != idx:
593
+ level = {"=": 1, "-": 2, "~": 3, "^": 4}.get(line[0], 2)
594
+ headings.append({"title": prev, "line_num": idx, "level": level})
595
+ continue
596
+
597
+ # ALL CAPS
598
+ if _RE_ALL_CAPS.match(line) and _is_short(line) and _has_blank_neighbor(lines, idx):
599
+ headings.append({"title": line, "line_num": num, "level": 1})
600
+
601
+ return headings
602
+
603
+
604
+ def _preprocess_text(text: str) -> str:
605
+ """Normalize line endings and collapse excessive blank lines."""
606
+ text = text.replace("\r\n", "\n").replace("\r", "\n").replace("\f", "\n")
607
+ return re.sub(r"\n{3,}", "\n\n", text)
608
+
609
+
610
+
611
+ async def text_to_tree(
612
+ text_path: Optional[str] = None,
613
+ text_content: Optional[str] = None,
614
+ *,
615
+ if_thinning: bool = False,
616
+ min_thinning_chars: int = 15000,
617
+ if_add_node_summary: bool = True,
618
+ summary_chars_threshold: int = 600,
619
+ if_add_doc_description: bool = False,
620
+ if_add_node_text: bool = False,
621
+ if_add_node_id: bool = True,
622
+ **kwargs,
623
+ ) -> dict:
624
+ """
625
+ Build a tree index from plain text (pure rule-based, no LLM).
626
+
627
+ Args:
628
+ text_path: path to a .txt file
629
+ text_content: raw text string (alternative to text_path)
630
+ Returns:
631
+ {'doc_name': str, 'structure': list, 'doc_description'?: str}
632
+ """
633
+ if text_path and text_content:
634
+ raise ValueError("Specify only one of text_path or text_content")
635
+ if not text_path and not text_content:
636
+ raise ValueError("Must specify text_path or text_content")
637
+
638
+ if text_path:
639
+ with open(text_path, "r", encoding="utf-8", errors="replace") as f:
640
+ raw = f.read()
641
+ doc_name = os.path.splitext(os.path.basename(text_path))[0]
642
+ else:
643
+ raw = text_content
644
+ doc_name = "untitled"
645
+
646
+ text = _preprocess_text(raw)
647
+ lines = text.split("\n")
648
+ logger.debug("Text loaded: %d lines", len(lines))
649
+
650
+ # Step 1: heading detection (pure rule-based)
651
+ headings = _detect_headings(lines)
652
+ markers = [{"title": h["title"], "line_num": h["line_num"], "level": h["level"]} for h in headings]
653
+ logger.debug("Rule-based detection: %d headings", len(markers))
654
+
655
+ # Fallback: single root node if no headings detected
656
+ if not markers:
657
+ markers = [{"title": doc_name, "line_num": 1, "level": 1}]
658
+
659
+ # Step 2: extract text
660
+ nodes = _cut_md_text(markers, lines)
661
+
662
+ # Step 3: thinning
663
+ if if_thinning and min_thinning_chars:
664
+ nodes = _update_char_counts(nodes)
665
+ logger.debug("Thinning tree (threshold=%d chars)...", min_thinning_chars)
666
+ nodes = _thin_tree(nodes, min_thinning_chars)
667
+
668
+ # Step 4: build tree
669
+ logger.debug("Building tree from %d nodes...", len(nodes))
670
+ tree = _build_tree(nodes)
671
+
672
+ return _finalize_tree(
673
+ tree, doc_name,
674
+ source_path=os.path.abspath(text_path) if text_path else "",
675
+ if_add_node_id=if_add_node_id,
676
+ if_add_node_summary=if_add_node_summary,
677
+ summary_chars_threshold=summary_chars_threshold,
678
+ if_add_node_text=if_add_node_text,
679
+ if_add_doc_description=if_add_doc_description,
680
+ )
681
+
682
+
683
+ # ============================================================================
684
+ # Code file indexer
685
+ # ============================================================================
686
+
687
+ def _detect_code_headings(lines: list[str], ext: str, source: str = "") -> list[dict]:
688
+ """Detect classes and methods from code lines.
689
+
690
+ For ``.py`` files, tries AST-based parsing first (richer signatures),
691
+ falling back to regex if AST fails (e.g. syntax errors).
692
+ """
693
+ # Python: use AST parser for accurate structure extraction
694
+ if ext == ".py" and source:
695
+ from .parsers.ast_parser import parse_python_structure
696
+ headings = parse_python_structure(source)
697
+ if headings:
698
+ return headings
699
+ # AST failed, fall through to regex
700
+
701
+ headings = []
702
+ patterns = []
703
+ if ext == ".py":
704
+ patterns = [
705
+ (re.compile(r"^(class\s+\w+.*)"), 1),
706
+ (re.compile(r"^(\s*def\s+\w+.*)"), 2)
707
+ ]
708
+ elif ext in (".java", ".ts", ".js", ".cpp", ".cc", ".cs", ".php"):
709
+ patterns = [
710
+ (re.compile(r"^(\s*(?:public|private|protected|static|abstract|final\s+)*class\s+\w+.*)"), 1),
711
+ (re.compile(r"^(\s*(?:public|private|protected|static|abstract|final\s+)*interface\s+\w+.*)"), 1),
712
+ (re.compile(r"^(\s*(?:public|private|protected|static|abstract|final\s+)*(?:[\w<>\[\]]+\s+)+\w+\s*\(.*)"), 2),
713
+ (re.compile(r"^(\s*function\s+\w+.*)"), 2)
714
+ ]
715
+ elif ext == ".go":
716
+ patterns = [
717
+ (re.compile(r"^(\s*type\s+\w+\s+struct.*)"), 1),
718
+ (re.compile(r"^(\s*type\s+\w+\s+interface.*)"), 1),
719
+ (re.compile(r"^(\s*func\s+(?:\([^)]+\)\s+)?\w+.*)"), 2)
720
+ ]
721
+ elif ext == ".html":
722
+ patterns = [
723
+ (re.compile(r"^\s*<h1.*>(.*)</h1>"), 1),
724
+ (re.compile(r"^\s*<h2.*>(.*)</h2>"), 2),
725
+ (re.compile(r"^\s*<h3.*>(.*)</h3>"), 3),
726
+ (re.compile(r"^\s*<div.*id=\"(.*)\".*>"), 2),
727
+ (re.compile(r"^\s*<section.*id=\"(.*)\".*>"), 2)
728
+ ]
729
+ elif ext == ".xml":
730
+ patterns = [
731
+ (re.compile(r"^\s*<(\w+).*>\s*$"), 1),
732
+ ]
733
+
734
+ if not patterns:
735
+ return []
736
+
737
+ for idx, raw in enumerate(lines):
738
+ line = raw.rstrip()
739
+ if not line:
740
+ continue
741
+ num = idx + 1
742
+
743
+ for pat, level in patterns:
744
+ m = pat.match(line)
745
+ if m:
746
+ title = m.group(1).strip().rstrip(":{").strip()[:100]
747
+ headings.append({"title": title, "line_num": num, "level": level})
748
+ break
749
+
750
+ return headings
751
+
752
+
753
+ async def code_to_tree(
754
+ code_path: str,
755
+ *,
756
+ if_thinning: bool = False,
757
+ min_thinning_chars: int = 15000,
758
+ if_add_node_summary: bool = True,
759
+ summary_chars_threshold: int = 600,
760
+ if_add_doc_description: bool = False,
761
+ if_add_node_text: bool = False,
762
+ if_add_node_id: bool = True,
763
+ **kwargs,
764
+ ) -> dict:
765
+ """
766
+ Build a tree index from a code file.
767
+
768
+ Returns:
769
+ {'doc_name': str, 'structure': list, 'doc_description'?: str}
770
+ """
771
+ with open(code_path, "r", encoding="utf-8", errors="replace") as f:
772
+ raw = f.read()
773
+ doc_name = os.path.splitext(os.path.basename(code_path))[0]
774
+ ext = os.path.splitext(code_path)[1].lower()
775
+
776
+ text = raw.replace("\r\n", "\n").replace("\r", "\n")
777
+ lines = text.split("\n")
778
+ logger.debug("Code loaded: %d lines", len(lines))
779
+
780
+ headings = _detect_code_headings(lines, ext, source=text)
781
+ markers = [{"title": h["title"], "line_num": h["line_num"], "level": h["level"]} for h in headings]
782
+ logger.debug("Code structure detection: %d methods/classes", len(markers))
783
+
784
+ if not markers:
785
+ markers = [{"title": doc_name, "line_num": 1, "level": 1}]
786
+
787
+ nodes = _cut_md_text(markers, lines)
788
+
789
+ if if_thinning and min_thinning_chars:
790
+ nodes = _update_char_counts(nodes)
791
+ logger.debug("Thinning tree (threshold=%d chars)...", min_thinning_chars)
792
+ nodes = _thin_tree(nodes, min_thinning_chars)
793
+
794
+ logger.debug("Building tree from %d nodes...", len(nodes))
795
+ tree = _build_tree(nodes)
796
+
797
+ return _finalize_tree(
798
+ tree, doc_name,
799
+ source_path=os.path.abspath(code_path),
800
+ if_add_node_id=if_add_node_id,
801
+ if_add_node_summary=if_add_node_summary,
802
+ summary_chars_threshold=summary_chars_threshold,
803
+ if_add_node_text=if_add_node_text,
804
+ if_add_doc_description=if_add_doc_description,
805
+ )
806
+
807
+
808
+ # ============================================================================
809
+ # JSON file indexer
810
+ # ============================================================================
811
+
812
+ def _json_to_nodes(data, prefix: str = "", level: int = 1) -> list[dict]:
813
+ """Recursively convert JSON data into flat node list."""
814
+ nodes = []
815
+ if isinstance(data, dict):
816
+ for key, value in data.items():
817
+ path = f"{prefix}.{key}" if prefix else key
818
+ if isinstance(value, (dict, list)):
819
+ nodes.append({"title": path, "level": level, "text": ""})
820
+ nodes.extend(_json_to_nodes(value, prefix=path, level=level + 1))
821
+ else:
822
+ nodes.append({"title": path, "level": level, "text": f"{key}: {value}"})
823
+ elif isinstance(data, list):
824
+ for i, item in enumerate(data):
825
+ path = f"{prefix}[{i}]"
826
+ if isinstance(item, (dict, list)):
827
+ nodes.append({"title": path, "level": level, "text": ""})
828
+ nodes.extend(_json_to_nodes(item, prefix=path, level=level + 1))
829
+ else:
830
+ nodes.append({"title": path, "level": level, "text": str(item)})
831
+ return nodes
832
+
833
+
834
+ def _load_json_lenient(raw: str):
835
+ """Load JSON tolerantly: handle control characters, comments, and trailing commas."""
836
+ # First try with strict=False to tolerate control characters
837
+ try:
838
+ return json.loads(raw, strict=False)
839
+ except json.JSONDecodeError:
840
+ pass
841
+
842
+ # Strip trailing commas before } or ]
843
+ stripped = re.sub(r',\s*([}\]])', r'\1', raw)
844
+ # Strip // comments, but preserve // inside quoted strings (e.g. URLs)
845
+ stripped = re.sub(r'("(?:[^"\\]|\\.)*")|//[^\n]*', r'\1', stripped)
846
+ return json.loads(stripped, strict=False)
847
+
848
+
849
+ async def json_to_tree(
850
+ json_path: str,
851
+ *,
852
+ if_add_node_summary: bool = True,
853
+ summary_chars_threshold: int = 600,
854
+ if_add_doc_description: bool = False,
855
+ if_add_node_text: bool = False,
856
+ if_add_node_id: bool = True,
857
+ **kwargs,
858
+ ) -> dict:
859
+ """Build a tree index from a JSON file."""
860
+ with open(json_path, "r", encoding="utf-8", errors="replace") as f:
861
+ raw = f.read()
862
+ doc_name = os.path.splitext(os.path.basename(json_path))[0]
863
+
864
+ try:
865
+ data = _load_json_lenient(raw)
866
+ except (json.JSONDecodeError, ValueError):
867
+ # JSON 解析失败,降级为纯文本
868
+ logger.debug("JSON parse failed for %s, falling back to text", json_path)
869
+ result = await text_to_tree(
870
+ text_content=raw,
871
+ if_add_node_summary=if_add_node_summary,
872
+ summary_chars_threshold=summary_chars_threshold,
873
+ if_add_doc_description=if_add_doc_description,
874
+ if_add_node_text=if_add_node_text,
875
+ if_add_node_id=if_add_node_id,
876
+ **kwargs,
877
+ )
878
+ result["doc_name"] = doc_name
879
+ result["source_path"] = os.path.abspath(json_path)
880
+ return result
881
+
882
+ flat_nodes = _json_to_nodes(data)
883
+ if not flat_nodes:
884
+ flat_nodes = [{"title": doc_name, "level": 1, "text": json.dumps(data, ensure_ascii=False)[:500]}]
885
+
886
+ # Assign line_num for _build_tree compatibility
887
+ for i, node in enumerate(flat_nodes):
888
+ node["line_num"] = i + 1
889
+ node["line_start"] = i + 1
890
+ node["line_end"] = i + 1
891
+
892
+ tree = _build_tree(flat_nodes)
893
+
894
+ return _finalize_tree(
895
+ tree, doc_name,
896
+ source_path=os.path.abspath(json_path),
897
+ if_add_node_id=if_add_node_id,
898
+ if_add_node_summary=if_add_node_summary,
899
+ summary_chars_threshold=summary_chars_threshold,
900
+ if_add_node_text=if_add_node_text,
901
+ if_add_doc_description=if_add_doc_description,
902
+ )
903
+
904
+
905
+ # ============================================================================
906
+ # JSONL file indexer
907
+ # ============================================================================
908
+
909
+ def _jsonl_to_nodes(records: list[dict], key_field: str = None) -> list[dict]:
910
+ """Convert a list of JSONL records into flat node list.
911
+
912
+ Each record becomes a level-1 node. If key_field is specified and exists
913
+ in the record, it is used as the title; otherwise uses the record index.
914
+ Nested structures within each record are expanded as child nodes.
915
+ """
916
+ nodes = []
917
+ for i, record in enumerate(records):
918
+ # Determine title for this record
919
+ if key_field and isinstance(record, dict) and key_field in record:
920
+ title = str(record[key_field])
921
+ elif isinstance(record, dict):
922
+ # Auto-detect: use first string-valued field as title
923
+ title = None
924
+ for k, v in record.items():
925
+ if isinstance(v, str) and len(v) < 200:
926
+ title = f"{k}: {v}"
927
+ break
928
+ if title is None:
929
+ title = f"record[{i}]"
930
+ else:
931
+ title = f"record[{i}]"
932
+
933
+ # Build text from record content
934
+ if isinstance(record, dict):
935
+ text_parts = []
936
+ child_nodes = []
937
+ for k, v in record.items():
938
+ if isinstance(v, (dict, list)):
939
+ child_nodes.extend(_json_to_nodes(v, prefix=f"record[{i}].{k}", level=2))
940
+ else:
941
+ text_parts.append(f"{k}: {v}")
942
+ text = "\n".join(text_parts)
943
+ nodes.append({"title": title, "level": 1, "text": text})
944
+ nodes.extend(child_nodes)
945
+ else:
946
+ nodes.append({"title": title, "level": 1, "text": str(record)})
947
+
948
+ return nodes
949
+
950
+
951
+ async def jsonl_to_tree(
952
+ jsonl_path: str,
953
+ *,
954
+ key_field: str = None,
955
+ if_add_node_summary: bool = True,
956
+ summary_chars_threshold: int = 600,
957
+ if_add_doc_description: bool = False,
958
+ if_add_node_text: bool = False,
959
+ if_add_node_id: bool = True,
960
+ **kwargs,
961
+ ) -> dict:
962
+ """Build a tree index from a JSONL file (one JSON object per line).
963
+
964
+ Args:
965
+ jsonl_path: path to the .jsonl file
966
+ key_field: optional field name to use as record title
967
+ """
968
+ records = []
969
+ with open(jsonl_path, "r", encoding="utf-8", errors="replace") as f:
970
+ for line_num, line in enumerate(f, 1):
971
+ line = line.strip()
972
+ if not line:
973
+ continue
974
+ try:
975
+ records.append(json.loads(line))
976
+ except json.JSONDecodeError as e:
977
+ logger.warning("Skipping invalid JSON at line %d in %s: %s", line_num, jsonl_path, e)
978
+
979
+ doc_name = os.path.splitext(os.path.basename(jsonl_path))[0]
980
+ logger.debug("JSONL loaded: %d records from %s", len(records), jsonl_path)
981
+
982
+ flat_nodes = _jsonl_to_nodes(records, key_field=key_field)
983
+ if not flat_nodes:
984
+ flat_nodes = [{"title": doc_name, "level": 1, "text": ""}]
985
+
986
+ for i, node in enumerate(flat_nodes):
987
+ node["line_num"] = i + 1
988
+ node["line_start"] = i + 1
989
+ node["line_end"] = i + 1
990
+
991
+ tree = _build_tree(flat_nodes)
992
+
993
+ return _finalize_tree(
994
+ tree, doc_name,
995
+ source_path=os.path.abspath(jsonl_path),
996
+ if_add_node_id=if_add_node_id,
997
+ if_add_node_summary=if_add_node_summary,
998
+ summary_chars_threshold=summary_chars_threshold,
999
+ if_add_node_text=if_add_node_text,
1000
+ if_add_doc_description=if_add_doc_description,
1001
+ )
1002
+
1003
+
1004
+ # ============================================================================
1005
+ # CSV file indexer
1006
+ # ============================================================================
1007
+
1008
+ async def csv_to_tree(
1009
+ csv_path: str,
1010
+ *,
1011
+ if_add_node_summary: bool = True,
1012
+ summary_chars_threshold: int = 600,
1013
+ if_add_doc_description: bool = False,
1014
+ if_add_node_text: bool = False,
1015
+ if_add_node_id: bool = True,
1016
+ **kwargs,
1017
+ ) -> dict:
1018
+ """Build a tree index from a CSV file. Each row becomes a leaf node under a header node."""
1019
+ import csv as csvmod
1020
+
1021
+ with open(csv_path, "r", encoding="utf-8", errors="replace") as f:
1022
+ reader = csvmod.reader(f)
1023
+ rows = list(reader)
1024
+
1025
+ doc_name = os.path.splitext(os.path.basename(csv_path))[0]
1026
+ if not rows:
1027
+ return {"doc_name": doc_name, "structure": [{"title": doc_name, "node_id": "0001", "nodes": []}]}
1028
+
1029
+ headers = rows[0]
1030
+ flat_nodes = [{"title": doc_name, "level": 1, "text": f"Columns: {', '.join(headers)}", "line_num": 1, "line_start": 1, "line_end": 1}]
1031
+
1032
+ for i, row in enumerate(rows[1:], start=2):
1033
+ row_text = "; ".join(f"{h}: {v}" for h, v in zip(headers, row) if v.strip())
1034
+ title = row_text[:80] if row_text else f"Row {i}"
1035
+ flat_nodes.append({"title": title, "level": 2, "text": row_text, "line_num": i, "line_start": i, "line_end": i})
1036
+
1037
+ tree = _build_tree(flat_nodes)
1038
+
1039
+ return _finalize_tree(
1040
+ tree, doc_name,
1041
+ source_path=os.path.abspath(csv_path),
1042
+ if_add_node_id=if_add_node_id,
1043
+ if_add_node_summary=if_add_node_summary,
1044
+ summary_chars_threshold=summary_chars_threshold,
1045
+ if_add_node_text=if_add_node_text,
1046
+ if_add_doc_description=if_add_doc_description,
1047
+ )
1048
+
1049
+
1050
+ # ============================================================================
1051
+ # Index statistics
1052
+ # ============================================================================
1053
+
1054
+ @dataclass
1055
+ class IndexStats:
1056
+ """Statistics collected during an indexing run.
1057
+
1058
+ Attributes:
1059
+ total_files: Total files discovered (including skipped).
1060
+ indexed_files: Files actually (re-)indexed in this run.
1061
+ skipped_files: Files skipped because they were unchanged.
1062
+ failed_files: Files that failed to parse.
1063
+ total_nodes: Total tree nodes generated across all indexed files.
1064
+ total_time_s: Total wall-clock time for the indexing run.
1065
+ per_type: Breakdown by source_type with counts, node totals, and timings.
1066
+ db_path: Path to the SQLite database file.
1067
+ db_size_bytes: Size of the database file on disk (0 for in-memory).
1068
+ failed_paths: List of file paths that failed to index.
1069
+ node_diff: Aggregate node-level diff across reindexed docs:
1070
+ ``{"added", "changed", "removed", "kept"}``.
1071
+ Useful for verifying incremental behaviour.
1072
+ pruned_paths: Source paths whose documents were removed from the
1073
+ index because the file no longer exists in the indexed scope.
1074
+ """
1075
+ total_files: int = 0
1076
+ indexed_files: int = 0
1077
+ skipped_files: int = 0
1078
+ failed_files: int = 0
1079
+ excluded_files: int = 0
1080
+ total_nodes: int = 0
1081
+ total_time_s: float = 0.0
1082
+ per_type: dict = field(default_factory=dict)
1083
+ db_path: str = ""
1084
+ db_size_bytes: int = 0
1085
+ failed_paths: list = field(default_factory=list)
1086
+ node_diff: dict = field(
1087
+ default_factory=lambda: {"added": 0, "changed": 0, "removed": 0, "kept": 0}
1088
+ )
1089
+ pruned_paths: list = field(default_factory=list)
1090
+
1091
+ def summary(self) -> str:
1092
+ """Return a human-readable summary string."""
1093
+ lines = []
1094
+ lines.append(f"Index Statistics")
1095
+ lines.append(f" Total files discovered: {self.total_files}")
1096
+ lines.append(f" Indexed (new/changed): {self.indexed_files}")
1097
+ lines.append(f" Skipped (unchanged): {self.skipped_files}")
1098
+ if self.failed_files:
1099
+ lines.append(f" Failed: {self.failed_files}")
1100
+ if self.excluded_files:
1101
+ lines.append(f" Exceeded failures: {self.excluded_files}")
1102
+ if self.pruned_paths:
1103
+ lines.append(f" Pruned (orphans): {len(self.pruned_paths)}")
1104
+ lines.append(f" Total nodes generated: {self.total_nodes}")
1105
+ nd = self.node_diff
1106
+ if any(nd.values()):
1107
+ lines.append(
1108
+ f" Node-level diff: +{nd['added']} ~{nd['changed']} "
1109
+ f"-{nd['removed']} (kept {nd['kept']})"
1110
+ )
1111
+ lines.append(f" Total time: {self.total_time_s:.3f}s")
1112
+ if self.db_path:
1113
+ size_str = _format_size(self.db_size_bytes)
1114
+ lines.append(f" Database: {self.db_path} ({size_str})")
1115
+
1116
+ if self.per_type:
1117
+ lines.append(f"")
1118
+ lines.append(f" Per file type:")
1119
+ # Sort by file count descending
1120
+ for stype, info in sorted(self.per_type.items(), key=lambda x: -x[1]["count"]):
1121
+ cnt = info["count"]
1122
+ nodes = info["nodes"]
1123
+ t = info["time_s"]
1124
+ lines.append(f" {stype:12s} {cnt:4d} file(s) {nodes:5d} nodes {t:.3f}s")
1125
+
1126
+ if self.failed_paths:
1127
+ lines.append(f"")
1128
+ lines.append(f" Failed files:")
1129
+ for fp in self.failed_paths[:10]:
1130
+ lines.append(f" - {fp}")
1131
+ if len(self.failed_paths) > 10:
1132
+ lines.append(f" ... and {len(self.failed_paths) - 10} more")
1133
+
1134
+ return "\n".join(lines)
1135
+
1136
+
1137
+ def _format_size(size_bytes: int) -> str:
1138
+ """Format bytes into human-readable string."""
1139
+ if size_bytes < 1024:
1140
+ return f"{size_bytes} B"
1141
+ elif size_bytes < 1024 * 1024:
1142
+ return f"{size_bytes / 1024:.1f} KB"
1143
+ else:
1144
+ return f"{size_bytes / (1024 * 1024):.1f} MB"
1145
+
1146
+
1147
+ # ============================================================================
1148
+ # Batch indexing API
1149
+ # ============================================================================
1150
+
1151
+ class _NullLock:
1152
+ """No-op lock context for in-memory or non-POSIX paths."""
1153
+ def __enter__(self):
1154
+ return self
1155
+ def __exit__(self, *a):
1156
+ return False
1157
+ def release(self) -> None:
1158
+ return None
1159
+
1160
+
1161
+ def _acquire_index_lock(db_path: str):
1162
+ """Acquire an exclusive advisory lock on ``{db_path}.lock``.
1163
+
1164
+ Returns a handle whose ``release()`` method closes the file (which
1165
+ releases the lock). Returns a no-op handle for in-memory or non-existent
1166
+ paths, and on platforms without ``fcntl`` (e.g. Windows).
1167
+ """
1168
+ if not db_path or db_path == ":memory:":
1169
+ return _NullLock()
1170
+
1171
+ # Unix: 使用 fcntl.flock
1172
+ try:
1173
+ import fcntl
1174
+ except ImportError:
1175
+ pass
1176
+ else:
1177
+ lock_path = db_path + ".lock"
1178
+ os.makedirs(os.path.dirname(os.path.abspath(lock_path)), exist_ok=True)
1179
+ f = open(lock_path, "w")
1180
+ try:
1181
+ fcntl.flock(f.fileno(), fcntl.LOCK_EX)
1182
+ except OSError as e:
1183
+ f.close()
1184
+ logger.warning("Failed to acquire index lock %s: %s", lock_path, e)
1185
+ return _NullLock()
1186
+
1187
+ class _Handle:
1188
+ def __init__(self, fh):
1189
+ self._fh = fh
1190
+ def __enter__(self):
1191
+ return self
1192
+ def __exit__(self, *a):
1193
+ self.release()
1194
+ return False
1195
+ def release(self) -> None:
1196
+ if self._fh is not None:
1197
+ try:
1198
+ fcntl.flock(self._fh.fileno(), fcntl.LOCK_UN)
1199
+ finally:
1200
+ self._fh.close()
1201
+ self._fh = None
1202
+
1203
+ return _Handle(f)
1204
+
1205
+ # Windows: 使用 msvcrt.locking
1206
+ try:
1207
+ import msvcrt
1208
+ except ImportError:
1209
+ logger.warning("No file locking available on this platform (no fcntl, no msvcrt)")
1210
+ return _NullLock()
1211
+
1212
+ lock_path = db_path + ".lock"
1213
+ os.makedirs(os.path.dirname(os.path.abspath(lock_path)), exist_ok=True)
1214
+ f = open(lock_path, "w+")
1215
+
1216
+ class _WindowsHandle:
1217
+ def __init__(self, fh):
1218
+ self._fh = fh
1219
+ def __enter__(self):
1220
+ return self
1221
+ def __exit__(self, *a):
1222
+ self.release()
1223
+ return False
1224
+ def release(self) -> None:
1225
+ if self._fh is not None:
1226
+ try:
1227
+ msvcrt.locking(self._fh.fileno(), msvcrt.LK_UNLCK, 0)
1228
+ except (OSError, IOError):
1229
+ pass # Lock auto-released on close on Windows
1230
+ finally:
1231
+ self._fh.close()
1232
+ self._fh = None
1233
+
1234
+ import time as _time
1235
+ # 重试:瞬态锁冲突(同进程内并发的 build_index)不该让本次索引直接放弃并返回 []
1236
+ # (会把 documents 清空,导致 files API 的 indexed 标志全部消失)。
1237
+ last_err = None
1238
+ for _attempt in range(10):
1239
+ try:
1240
+ msvcrt.locking(f.fileno(), msvcrt.LK_NBLCK, 1)
1241
+ return _WindowsHandle(f)
1242
+ except (OSError, IOError) as e:
1243
+ last_err = e
1244
+ _time.sleep(0.2)
1245
+ f.close()
1246
+ logger.warning("Failed to acquire Windows index lock %s after retries: %s", lock_path, last_err)
1247
+ return _NullLock()
1248
+
1249
+
1250
+ def file_hash(fp: str, mode: Optional[str] = None) -> str:
1251
+ """Compute a fingerprint for incremental indexing.
1252
+
1253
+ Format: ``"v{INDEX_SCHEMA_VERSION}:{mode}:{payload}"`` so that:
1254
+ - bumping ``INDEX_SCHEMA_VERSION`` invalidates every stored hash, forcing a
1255
+ clean rebuild after a parser/tokenizer/schema change.
1256
+ - switching ``fingerprint_mode`` between ``stat`` and ``content`` likewise
1257
+ invalidates so users can opt-in safely.
1258
+
1259
+ Modes:
1260
+ - ``stat`` (default): ``(mtime_ns:size)`` — fast, catches normal edits.
1261
+ - ``content``: full md5 for files <``content_fingerprint_size_threshold``;
1262
+ large files are sampled at head/middle/tail with
1263
+ ``content_fingerprint_sample_bytes`` per region. Robust against
1264
+ ``touch``/CI-replay scenarios at the cost of one read per file.
1265
+
1266
+ Returns empty string if the file does not exist.
1267
+ """
1268
+ from .config import INDEX_SCHEMA_VERSION, get_config
1269
+ cfg = get_config()
1270
+ if mode is None:
1271
+ mode = cfg.fingerprint_mode
1272
+
1273
+ try:
1274
+ st = os.stat(fp)
1275
+ except (FileNotFoundError, OSError):
1276
+ return ""
1277
+
1278
+ if mode == "content":
1279
+ threshold = cfg.content_fingerprint_size_threshold
1280
+ sample = cfg.content_fingerprint_sample_bytes
1281
+ h = hashlib.md5()
1282
+ try:
1283
+ with open(fp, "rb") as f:
1284
+ if st.st_size <= threshold:
1285
+ h.update(f.read())
1286
+ else:
1287
+ h.update(f.read(sample))
1288
+ if st.st_size > 2 * sample:
1289
+ f.seek(st.st_size // 2)
1290
+ h.update(f.read(sample))
1291
+ f.seek(max(0, st.st_size - sample))
1292
+ h.update(f.read(sample))
1293
+ except OSError:
1294
+ return ""
1295
+ payload = f"{st.st_size}:{h.hexdigest()}"
1296
+ else:
1297
+ payload = f"{st.st_mtime_ns}:{st.st_size}"
1298
+
1299
+ return f"v{INDEX_SCHEMA_VERSION}:{mode}:{payload}"
1300
+
1301
+
1302
+ def file_hash_with_salts(fp: str, mode: Optional[str] = None) -> str:
1303
+ """``file_hash`` + 解析器格式盐(插在版本号后,保持末尾 size 可解析)。
1304
+
1305
+ PST 解析器输出格式变化(ADR-0005)时 bump PST_PARSER_FINGERPRINT_SALT,
1306
+ 旧 PST 索引自动重建,不像 INDEX_SCHEMA_VERSION 那样连累全库。
1307
+ 增量比较与移动检测都必须用本函数(口径一致才能匹配)。
1308
+
1309
+ 图像文件(jpg/jpeg/png/webp)改用解码像素指纹:写回 EXIF/XMP 元数据不改像素,
1310
+ 故不触发「写回 → hash 变 → 增量重解析」死循环(ADR-0009 / 工单 03)。解码失败
1311
+ 回退到 stat/content。注:口径切换使旧图像 hash 失效 → 首迁移重索引(已知代价)。
1312
+ """
1313
+ ext = os.path.splitext(fp)[1].lower()
1314
+ if ext in (".jpg", ".jpeg", ".png", ".webp"): # 与 image_metadata.INTERPRETED_IMAGE_EXTS 一致
1315
+ try:
1316
+ from .config import INDEX_SCHEMA_VERSION
1317
+ from .parsers.image_metadata import content_fingerprint
1318
+ return f"v{INDEX_SCHEMA_VERSION}:image:{content_fingerprint(fp)}"
1319
+ except Exception:
1320
+ pass # 解码失败回退到 stat/content
1321
+ h = file_hash(fp, mode)
1322
+ if h and fp.lower().endswith(".pst"):
1323
+ ver, sep, rest = h.partition(":")
1324
+ return f"{ver}{sep}{PST_PARSER_FINGERPRINT_SALT.strip(':')}:{rest}"
1325
+ return h
1326
+
1327
+
1328
+ def _extract_size_from_fingerprint(fp_hash: str) -> Optional[str]:
1329
+ """Extract the file size portion from a fingerprint string.
1330
+
1331
+ Supports both stat (``v{VER}:stat:{mtime}:{size}``) and content
1332
+ (``v{VER}:content:{size}:{md5}``) formats.
1333
+ """
1334
+ parts = fp_hash.split(":")
1335
+ if len(parts) < 3:
1336
+ return None
1337
+ candidate = parts[-1]
1338
+ return candidate if candidate.isdigit() else None
1339
+
1340
+
1341
+ def _detect_and_apply_moves(
1342
+ expanded: list[str],
1343
+ all_meta: dict[str, str],
1344
+ fts,
1345
+ fp_to_doc_id: dict[str, str],
1346
+ file_hash_fn,
1347
+ ) -> list[tuple[str, str]]:
1348
+ """Detect file moves/renames and apply them in-place via ``fts.rename_document``.
1349
+
1350
+ Two-pass strategy:
1351
+
1352
+ 1. **Fingerprint match** (exact): compute the file fingerprint and look it up
1353
+ in a reverse index built from stored metadata. This works perfectly when
1354
+ the fingerprint is content-based (MD5), and also catches the rare case
1355
+ where ``stat`` mtime is preserved after a move.
1356
+
1357
+ 2. **Size heuristic** (fallback): for stat-mode fingerprints where mtime
1358
+ changed on move, match by ``(basename, file_size)``. This is reliable
1359
+ because: same basename + same byte size + old path gone = move.
1360
+
1361
+ Mutates ``all_meta`` in-place by popping old entries and inserting new ones.
1362
+ Returns a list of ``(old_path, new_path)`` tuples for logging.
1363
+ """
1364
+ from .config import INDEX_SCHEMA_VERSION
1365
+
1366
+ _current_prefix = f"v{INDEX_SCHEMA_VERSION}:"
1367
+
1368
+ # --- Build reverse index: fingerprint → stored_path ---
1369
+ hash_to_old_path: dict[str, str] = {}
1370
+ for sp, fh in all_meta.items():
1371
+ if fh.startswith(_current_prefix):
1372
+ hash_to_old_path.setdefault(fh, sp)
1373
+
1374
+ moved: list[tuple[str, str]] = []
1375
+ expanded_abs = {os.path.abspath(p) for p in expanded}
1376
+ matched_new = set() # abs_fp of new files already claimed
1377
+
1378
+ # --- Pass 1: fingerprint-based matching ---
1379
+ if hash_to_old_path:
1380
+ for fp in expanded:
1381
+ abs_fp = os.path.abspath(fp)
1382
+ if abs_fp in all_meta:
1383
+ continue
1384
+ fh = file_hash_fn(abs_fp)
1385
+ if not fh:
1386
+ continue
1387
+ old_path = hash_to_old_path.get(fh)
1388
+ if old_path and old_path != abs_fp and not os.path.isfile(old_path):
1389
+ moved_doc_id = fts.get_doc_id_by_source_path(old_path)
1390
+ if moved_doc_id:
1391
+ new_doc_id = fp_to_doc_id.get(fp, os.path.splitext(os.path.basename(abs_fp))[0])
1392
+ if fts.rename_document(moved_doc_id, new_doc_id, new_doc_id, abs_fp):
1393
+ fts.set_index_meta(abs_fp, fh)
1394
+ all_meta.pop(old_path, None)
1395
+ all_meta[abs_fp] = fh
1396
+ hash_to_old_path[fh] = abs_fp
1397
+ matched_new.add(abs_fp)
1398
+ moved.append((old_path, abs_fp))
1399
+ logger.info(
1400
+ "Detected moved file: %s -> %s (doc_id %s -> %s)",
1401
+ old_path, abs_fp, moved_doc_id, new_doc_id,
1402
+ )
1403
+ else:
1404
+ logger.debug(
1405
+ "Move-detection skipped for %s -> %s "
1406
+ "(doc_id collision); will full re-index",
1407
+ old_path, abs_fp,
1408
+ )
1409
+
1410
+ # --- Pass 2: size-based heuristic for stat-mode moves ---
1411
+ # Collect orphan candidates: indexed paths whose files disappeared from disk
1412
+ # and were not already remapped by Pass 1.
1413
+ orphan_by_key: dict[str, list[tuple[str, str]]] = {} # "basename:size" → [(path, hash)]
1414
+ for sp, sh in list(all_meta.items()):
1415
+ if sp in expanded_abs or os.path.isfile(sp):
1416
+ continue
1417
+ size_str = _extract_size_from_fingerprint(sh)
1418
+ if not size_str:
1419
+ continue
1420
+ bn = os.path.basename(sp)
1421
+ orphan_by_key.setdefault(f"{bn}:{size_str}", []).append((sp, sh))
1422
+
1423
+ if orphan_by_key:
1424
+ for fp in expanded:
1425
+ abs_fp = os.path.abspath(fp)
1426
+ if abs_fp in all_meta or abs_fp in matched_new:
1427
+ continue
1428
+ try:
1429
+ st = os.stat(abs_fp)
1430
+ except OSError:
1431
+ continue
1432
+ bn = os.path.basename(abs_fp)
1433
+ key = f"{bn}:{st.st_size}"
1434
+ candidates = orphan_by_key.get(key, [])
1435
+ for old_path, stored_hash in candidates:
1436
+ if old_path not in all_meta:
1437
+ continue # already claimed by another new file
1438
+ moved_doc_id = fts.get_doc_id_by_source_path(old_path)
1439
+ if not moved_doc_id:
1440
+ continue
1441
+ new_doc_id = fp_to_doc_id.get(fp, os.path.splitext(bn)[0])
1442
+ if fts.rename_document(moved_doc_id, new_doc_id, new_doc_id, abs_fp):
1443
+ new_fh = file_hash_fn(abs_fp)
1444
+ fts.set_index_meta(abs_fp, new_fh)
1445
+ all_meta.pop(old_path, None)
1446
+ all_meta[abs_fp] = new_fh
1447
+ moved.append((old_path, abs_fp))
1448
+ logger.info(
1449
+ "Detected moved/renamed file (size heuristic): %s -> %s",
1450
+ old_path, abs_fp,
1451
+ )
1452
+ break
1453
+
1454
+ return moved
1455
+
1456
+
1457
+ async def build_index(
1458
+ paths: list[str],
1459
+ output_dir: str = "./indexes",
1460
+ *,
1461
+ db_path: str = "",
1462
+ if_add_node_summary: Optional[bool] = None,
1463
+ if_add_doc_description: Optional[bool] = None,
1464
+ if_add_node_text: Optional[bool] = None,
1465
+ if_add_node_id: Optional[bool] = None,
1466
+ max_concurrency: Optional[int] = None,
1467
+ force: bool = False,
1468
+ ignore_dirs: frozenset[str] = DEFAULT_IGNORE_DIRS,
1469
+ respect_gitignore: bool = True,
1470
+ max_files: int = MAX_DIR_FILES,
1471
+ prune: Optional[bool] = None,
1472
+ progress_callback: Optional[callable] = None,
1473
+ sub_progress_callback: Optional[callable] = None,
1474
+ **kwargs,
1475
+ ) -> list[Document]:
1476
+ """
1477
+ Build tree indexes for multiple files. Returns list of Document objects ready for search.
1478
+
1479
+ All parameters default to ``get_config()`` values when not explicitly set.
1480
+
1481
+ Args:
1482
+ paths: list of file paths, glob patterns, or directories
1483
+ (e.g. ``["docs/*.md", "paper.txt", "src/"]``)
1484
+ output_dir: directory for the database file (used to derive db_path if db_path is empty)
1485
+ db_path: path to the SQLite database file. If empty, defaults to ``{output_dir}/index.db``.
1486
+ max_concurrency: max concurrent indexing tasks
1487
+ force: force re-index even if file unchanged (default: False)
1488
+ ignore_dirs: directory names to skip during recursive walk
1489
+ respect_gitignore: honour ``.gitignore`` files when walking directories
1490
+ max_files: safety cap on files discovered per directory walk
1491
+ prune: if True, delete documents whose source files are no longer
1492
+ reachable through ``paths`` (orphan cleanup). When ``None``,
1493
+ defaults to True iff at least one entry of ``paths`` is a directory
1494
+ or recursive glob (full-scope reindex), False otherwise.
1495
+ **kwargs: passed through to individual parsers
1496
+
1497
+ Returns:
1498
+ list of Document objects (directly usable with search())
1499
+ """
1500
+ from .config import get_config
1501
+ from .fts import FTS5Index
1502
+ cfg = get_config()
1503
+
1504
+ # Resolve defaults from config
1505
+ if if_add_node_summary is None:
1506
+ if_add_node_summary = cfg.if_add_node_summary
1507
+ if if_add_doc_description is None:
1508
+ if_add_doc_description = cfg.if_add_doc_description
1509
+ if if_add_node_text is None:
1510
+ if_add_node_text = cfg.if_add_node_text
1511
+ if if_add_node_id is None:
1512
+ if_add_node_id = cfg.if_add_node_id
1513
+ if max_concurrency is None:
1514
+ max_concurrency = cfg.max_concurrency
1515
+
1516
+ # Resolve db_path
1517
+ if not db_path:
1518
+ db_path = os.path.join(output_dir, "index.db")
1519
+ os.makedirs(os.path.dirname(os.path.abspath(db_path)), exist_ok=True)
1520
+
1521
+ # 图片落盘存储(与 index.db 同目录的 images/ 子目录)
1522
+ from .parsers.image_store import ImageStore
1523
+ images_root = Path(db_path).parent / "images"
1524
+ image_store = ImageStore(images_root)
1525
+ # PST 附件落盘存储(与 index.db 同目录的 pst_attachments/ 子目录,ADR-0005)
1526
+ from .parsers.pst_attachment_store import PstAttachmentStore
1527
+ pst_att_store = PstAttachmentStore(Path(db_path).parent / "pst_attachments")
1528
+ if force:
1529
+ image_store.purge_all()
1530
+ pst_att_store.purge_all()
1531
+
1532
+ # 计算 rel_path 用的 base 目录(paths 里第一个目录,即 search_path)
1533
+ base_dir = ""
1534
+ for p in paths:
1535
+ if os.path.isdir(p):
1536
+ base_dir = os.path.abspath(p)
1537
+ break
1538
+
1539
+ # Expand globs, files, and directories via resolve_paths
1540
+ expanded = resolve_paths(
1541
+ paths,
1542
+ ignore_dirs=ignore_dirs,
1543
+ respect_gitignore=respect_gitignore,
1544
+ max_files=max_files,
1545
+ )
1546
+ total_files = len(expanded)
1547
+ processed_counter = [0]
1548
+ _progress_lock = asyncio.Lock()
1549
+ if not expanded:
1550
+ raise FileNotFoundError(f"No files found for patterns: {paths}")
1551
+
1552
+ # Pre-compute deterministic doc_ids: basename + path hash (always, for determinism)
1553
+ _base_dir = os.path.dirname(os.path.abspath(db_path)) if db_path else os.getcwd()
1554
+
1555
+ def _doc_id_for(fp):
1556
+ base = os.path.splitext(os.path.basename(fp))[0]
1557
+ rel = os.path.relpath(os.path.abspath(fp), _base_dir)
1558
+ h = hashlib.md5(rel.encode()).hexdigest()[:8]
1559
+ return f"{base}_{h}"
1560
+
1561
+ _fp_to_doc_id = {fp: _doc_id_for(fp) for fp in expanded}
1562
+
1563
+ # Resolve prune policy: full-scope walks default to pruning orphans.
1564
+ explicit_prune = prune is not None
1565
+ if prune is None:
1566
+ full_scope = any(
1567
+ os.path.isdir(p) or "**" in p
1568
+ for p in paths
1569
+ )
1570
+ prune = full_scope and cfg.prune_orphans_on_directory
1571
+
1572
+ # Open DB with advisory file lock so concurrent build_index() calls on the
1573
+ # same DB serialize cleanly instead of racing on writes.
1574
+ fts = FTS5Index(db_path=db_path, tokenize_log_path=os.path.join(os.path.dirname(db_path), "tokenize.log"))
1575
+ _lock_handle = _acquire_index_lock(db_path)
1576
+
1577
+ # 如果锁获取失败(_NullLock),说明有其他进程在索引,跳过本次
1578
+ if isinstance(_lock_handle, _NullLock):
1579
+ logger.warning("Index is locked by another process, skipping this indexing run")
1580
+ return []
1581
+
1582
+ # Incremental indexing: batch check file hashes via DB
1583
+ to_index = []
1584
+ skipped = []
1585
+ file_hashes = {}
1586
+ pruned_paths: list[str] = []
1587
+
1588
+ if force:
1589
+ # Full rebuild: clear all failed file records
1590
+ fts.clear_all_failed_files()
1591
+ # 视觉解析队列一并清空(重建过程会重新登记)
1592
+ fts.vision_clear()
1593
+
1594
+ if not force:
1595
+ # Batch fetch all stored hashes in one query (instead of N queries)
1596
+ all_meta = fts.get_all_index_meta()
1597
+ elif cfg.allowed_source_types:
1598
+ # force=True with source_type filter: still need all_meta for orphan cleanup
1599
+ all_meta = fts.get_all_index_meta()
1600
+ else:
1601
+ all_meta = {}
1602
+
1603
+ # Pre-pass: detect moves/renames and remap source_path BEFORE pruning,
1604
+ # so the orphan cleanup below doesn't drop a doc whose file just moved.
1605
+ if not force:
1606
+ _detect_and_apply_moves(
1607
+ expanded, all_meta, fts, _fp_to_doc_id, file_hash_with_salts
1608
+ )
1609
+
1610
+ # Orphan cleanup: docs in DB whose source_path is not in the new scope.
1611
+ # Implicit prune (defaulted from a directory walk): only drop docs whose
1612
+ # files are also gone from disk — preserves unrelated files for partial
1613
+ # reindexes.
1614
+ # Explicit `prune=True`: reduce the index to exactly `paths`.
1615
+ if prune and all_meta:
1616
+ expanded_abs = {os.path.abspath(p) for p in expanded}
1617
+ # Pre-compute allowed extensions from source_type filter (if any)
1618
+ source_type_exts = None
1619
+ if cfg.allowed_source_types:
1620
+ from .pathutil import get_allowed_extensions_for_source_types
1621
+ source_type_exts = get_allowed_extensions_for_source_types(cfg.allowed_source_types)
1622
+ prune_doc_ids: list[str] = []
1623
+ for stored_path in list(all_meta.keys()):
1624
+ if stored_path in expanded_abs:
1625
+ continue
1626
+ if not explicit_prune and os.path.isfile(stored_path):
1627
+ # File exists on disk but was filtered out — only keep it if
1628
+ # it would have been included WITHOUT the source_type filter.
1629
+ if source_type_exts is not None:
1630
+ ext = os.path.splitext(stored_path)[1].lower()
1631
+ if ext not in source_type_exts:
1632
+ pass # excluded by source_type → prune
1633
+ else:
1634
+ continue
1635
+ else:
1636
+ continue
1637
+ doc_id = fts.get_doc_id_by_source_path(stored_path)
1638
+ # 多文档来源(PST 派生 "<file>#<entry_id>"):级联删除派生文档
1639
+ derived_ids = fts.get_doc_ids_by_source_prefix(stored_path + "#")
1640
+ ids = ([doc_id] if doc_id else []) + derived_ids
1641
+ if ids:
1642
+ prune_doc_ids.extend(ids)
1643
+ pruned_paths.append(stored_path)
1644
+ all_meta.pop(stored_path, None)
1645
+ if prune_doc_ids:
1646
+ fts.delete_documents(prune_doc_ids)
1647
+ logger.info("Pruned %d orphan document(s) from index", len(prune_doc_ids))
1648
+ # 被 prune 的源文件同步移出视觉解析队列
1649
+ for stored_path in pruned_paths:
1650
+ fts.vision_remove(stored_path)
1651
+
1652
+ # Clean up shadow MD files for pruned binary sources
1653
+ from .parsers.registry import is_binary_extension
1654
+ for pruned_path in pruned_paths:
1655
+ ext = os.path.splitext(pruned_path)[1].lower()
1656
+ if is_binary_extension(ext):
1657
+ md_path = shadow_md_path(pruned_path)
1658
+ if os.path.exists(md_path):
1659
+ try:
1660
+ os.remove(md_path)
1661
+ logger.debug("Removed orphan shadow MD: %s", md_path)
1662
+ except OSError as e:
1663
+ logger.debug("Failed to remove shadow MD %s: %s", md_path, e)
1664
+ # 清理被删文档的图片
1665
+ if base_dir:
1666
+ pruned_rel = os.path.relpath(pruned_path, base_dir).replace(os.sep, "/")
1667
+ image_store.purge_doc(pruned_rel)
1668
+ # 清理被删 PST 的落盘附件(ADR-0005)
1669
+ if ext == ".pst" and base_dir:
1670
+ pruned_rel = os.path.relpath(pruned_path, base_dir).replace(os.sep, "/")
1671
+ pst_att_store.purge_doc(pruned_rel)
1672
+
1673
+ for fp in expanded:
1674
+ abs_fp = os.path.abspath(fp)
1675
+ # 指纹含解析器格式盐(PST:ADR-0005 输出格式变化时自动重建)
1676
+ fh = file_hash_with_salts(abs_fp)
1677
+ if not fh:
1678
+ # File disappeared after glob expansion (e.g. broken symlink)
1679
+ logger.debug("Skipping missing file: %s", abs_fp)
1680
+ continue
1681
+ file_hashes[abs_fp] = fh
1682
+ if not force:
1683
+ stored_hash = all_meta.get(abs_fp)
1684
+ if stored_hash == fh:
1685
+ # source_path lookup catches both same-name and moved files.
1686
+ # Multi-doc sources (e.g. PST: docs keyed "<file>#<entry_id>")
1687
+ # have no doc at the exact path — check the derived prefix.
1688
+ if fts.get_doc_id_by_source_path(abs_fp) is not None \
1689
+ or fts.has_docs_with_source_prefix(abs_fp + "#"):
1690
+ skipped.append(fp)
1691
+ processed_counter[0] += 1
1692
+ if progress_callback:
1693
+ progress_callback(fp, processed_counter[0], total_files)
1694
+ continue
1695
+ to_index.append(fp)
1696
+
1697
+ if skipped:
1698
+ logger.info("Skipped %d unchanged file(s)", len(skipped))
1699
+
1700
+ # Filter out files that have exceeded the consecutive failure threshold
1701
+ excluded_count = 0
1702
+ if not force and to_index:
1703
+ failed_records = fts.get_all_failed_files() # {path: (fail_count, file_hash, last_error)}
1704
+ max_fail_count = cfg.max_index_fail_count
1705
+ excluded = []
1706
+ remaining = []
1707
+ for fp in to_index:
1708
+ abs_fp = os.path.abspath(fp)
1709
+ record = failed_records.get(abs_fp)
1710
+ if record and record[0] >= max_fail_count:
1711
+ # Check if file content has changed since last failure
1712
+ current_hash = file_hashes.get(abs_fp, "")
1713
+ if record[1] and record[1] == current_hash:
1714
+ excluded.append(fp)
1715
+ continue
1716
+ remaining.append(fp)
1717
+ to_index = remaining
1718
+ if excluded:
1719
+ excluded_count = len(excluded)
1720
+ logger.warning("Skipped %d file(s) with %d+ consecutive failures", excluded_count, max_fail_count)
1721
+ for fp in excluded:
1722
+ abs_fp = os.path.abspath(fp)
1723
+ record = failed_records.get(abs_fp)
1724
+ last_error = record[2] if record else ""
1725
+ if last_error:
1726
+ logger.warning(" Skipped %s: %s", fp, last_error)
1727
+ else:
1728
+ logger.warning(" Skipped %s: unknown error", fp)
1729
+ processed_counter[0] += 1
1730
+ if progress_callback:
1731
+ progress_callback(fp, processed_counter[0], total_files)
1732
+
1733
+ logger.info("Building indexes for %d file(s) (concurrency=%d)...", len(to_index), max_concurrency)
1734
+
1735
+ build_start = time.monotonic()
1736
+ semaphore = asyncio.Semaphore(max_concurrency)
1737
+ # PST 专用并发上限:PST 解析启 sidecar 子进程 + 加载 GB 级文件 + 全量附件提取,
1738
+ # 多个并发会耗尽内存/CPU/IO 触发 sidecar 崩溃(exit 0xC000013A)。单独低并发限流,
1739
+ # 同时缓解"极度缓慢"和"崩溃"两个根因。
1740
+ pst_semaphore = asyncio.Semaphore(getattr(cfg, "max_pst_concurrency", 1))
1741
+ # Collect per-file timing and source_type for stats
1742
+ _file_timings: dict[str, tuple[str, float]] = {} # fp -> (source_type, elapsed_s)
1743
+ _failed_paths: list[str] = []
1744
+
1745
+ # Progress bar for parsing stage
1746
+ _parse_bar = tqdm(total=len(to_index), desc="Parsing", unit="file",
1747
+ dynamic_ncols=True, disable=not to_index)
1748
+
1749
+ async def _index_one(fp: str) -> dict | None:
1750
+ ext = os.path.splitext(fp)[1].lower()
1751
+ # PST 重量级(sidecar 子进程 + GB 级文件 + 全量附件提取):先过 pst_semaphore
1752
+ # 限流(等待时不占通用 semaphore 槽),再过通用并发信号量。
1753
+ pst_gate = pst_semaphore if ext == ".pst" else nullcontext()
1754
+ async with pst_gate, semaphore:
1755
+ fname = os.path.basename(fp)
1756
+ _parse_bar.set_postfix_str(fname, refresh=False)
1757
+ t0 = time.monotonic()
1758
+ try:
1759
+ rel_path = ""
1760
+ if base_dir:
1761
+ rel_path = os.path.relpath(fp, base_dir).replace(os.sep, "/")
1762
+ # 该文件重抽前先清掉它的旧图(幂等;force 模式已 purge_all,此处为空操作)
1763
+ if rel_path:
1764
+ image_store.purge_doc(rel_path)
1765
+ # PST 重索引:清掉旧落盘附件(幂等,重新提取)
1766
+ if ext == ".pst":
1767
+ pst_att_store.purge_doc(rel_path)
1768
+ common = dict(
1769
+ if_add_node_summary=if_add_node_summary,
1770
+ if_add_doc_description=if_add_doc_description,
1771
+ if_add_node_text=if_add_node_text,
1772
+ if_add_node_id=if_add_node_id,
1773
+ image_store=image_store,
1774
+ rel_path=rel_path,
1775
+ pst_attachment_store=pst_att_store,
1776
+ sub_progress_callback=(
1777
+ (lambda n: sub_progress_callback(fp, n))
1778
+ if sub_progress_callback else None
1779
+ ),
1780
+ **kwargs,
1781
+ )
1782
+
1783
+ # Use ParserRegistry for dispatch (built-in parsers auto-registered)
1784
+ from .parsers import get_parser, SOURCE_TYPE_MAP
1785
+ parser_fn = get_parser(ext)
1786
+ if parser_fn is not None:
1787
+ result = await parser_fn(fp, **common)
1788
+ else:
1789
+ # Unknown extension: fall back to text_to_tree
1790
+ result = await text_to_tree(text_path=fp, **common)
1791
+
1792
+ # Tag source_type for search routing
1793
+ source_type = SOURCE_TYPE_MAP.get(ext, "text")
1794
+ result["source_type"] = source_type
1795
+
1796
+ # 图像文件:占位节点已进索引,登记到视觉解析队列(后台 worker 消费)
1797
+ if source_type == "image" and result.get("vision_pending"):
1798
+ fts.vision_enqueue(os.path.abspath(fp), rel_path)
1799
+
1800
+ # Generate shadow MD for binary files (concurrent with parsing)
1801
+ if cfg.enable_shadow_md:
1802
+ from .parsers.registry import is_binary_extension
1803
+ if is_binary_extension(ext):
1804
+ try:
1805
+ _generate_shadow_md(os.path.abspath(fp))
1806
+ except Exception as e:
1807
+ logger.debug("Shadow MD generation failed for %s: %s", fp, e)
1808
+
1809
+ _file_timings[fp] = (source_type, time.monotonic() - t0)
1810
+ # Call progress callback if provided
1811
+ async with _progress_lock:
1812
+ processed_counter[0] += 1
1813
+ if progress_callback:
1814
+ progress_callback(fp, processed_counter[0], total_files)
1815
+ return result
1816
+ except Exception as e:
1817
+ logger.warning("Failed to index %s: %s", fp, e)
1818
+ _failed_paths.append(fp)
1819
+ abs_fp = os.path.abspath(fp)
1820
+ fts.upsert_failed_file(abs_fp, str(e), file_hashes.get(abs_fp, ""))
1821
+ _file_timings[fp] = ("(failed)", time.monotonic() - t0)
1822
+ async with _progress_lock:
1823
+ processed_counter[0] += 1
1824
+ if progress_callback:
1825
+ progress_callback(fp, processed_counter[0], total_files)
1826
+ return None
1827
+ finally:
1828
+ _parse_bar.update()
1829
+
1830
+ raw_results = await asyncio.gather(*(_index_one(fp) for fp in to_index))
1831
+ _parse_bar.close()
1832
+
1833
+ # Save results to DB and collect Document objects
1834
+ result_map = {fp: r for fp, r in zip(to_index, raw_results) if r is not None}
1835
+ documents = []
1836
+
1837
+ # Batch load all skipped documents in one query (instead of N individual loads)
1838
+ # Key by source_path (unique and stable) instead of doc_id (may change)
1839
+ if skipped:
1840
+ all_docs_from_db = {
1841
+ d.metadata.get("source_path", ""): d
1842
+ for d in fts.load_all_documents()
1843
+ if d.metadata.get("source_path")
1844
+ }
1845
+ else:
1846
+ all_docs_from_db = {}
1847
+
1848
+ # Progress bar for Indexing stage (batch commit every N files to reduce fsync)
1849
+ _COMMIT_BATCH = 500
1850
+ _has_work = bool(result_map)
1851
+ _save_bar = tqdm(total=len(expanded), desc="Indexing", unit="file",
1852
+ dynamic_ncols=True, disable=not _has_work)
1853
+ _pending_commits = 0
1854
+ # Aggregate node-level diff stats across all reindexed docs.
1855
+ diff_totals = {"added": 0, "changed": 0, "removed": 0, "kept": 0}
1856
+ optimize_threshold = cfg.auto_optimize_threshold
1857
+ docs_since_optimize = 0
1858
+ for fp in expanded:
1859
+ name = _fp_to_doc_id[fp]
1860
+ if fp in result_map:
1861
+ _save_bar.set_postfix_str(os.path.basename(fp), refresh=False)
1862
+ result = result_map[fp]
1863
+ abs_fp = os.path.abspath(fp)
1864
+ file_h = file_hashes.get(abs_fp, "")
1865
+
1866
+ if result.get("multi_docs") is not None:
1867
+ # 多文档来源(PST 等):一个文件 → N 个派生文档。
1868
+ # 派生 doc_id = <file_doc_id>__<entry>;source_path = <file>#<entry>。
1869
+ # 已被移除的派生文档(邮件删除)按差集清除;其余走节点级增量。
1870
+ trees = result["multi_docs"]
1871
+ new_docs = []
1872
+ for t in trees:
1873
+ sp = t.get("source_path", "")
1874
+ entry = sp.rsplit("#", 1)[-1] if "#" in sp else str(len(new_docs))
1875
+ new_docs.append(Document(
1876
+ doc_id=f"{name}__{entry}",
1877
+ doc_name=t.get("doc_name", name),
1878
+ structure=t.get("structure", []),
1879
+ doc_description=t.get("doc_description", ""),
1880
+ metadata={"source_path": sp},
1881
+ source_type=result.get("source_type", ""),
1882
+ ))
1883
+ new_ids = {d.doc_id for d in new_docs}
1884
+ old_ids = set(fts.get_doc_ids_by_source_prefix(abs_fp + "#"))
1885
+ removed_ids = sorted(old_ids - new_ids)
1886
+ if removed_ids:
1887
+ fts.delete_documents(removed_ids)
1888
+ # 被移除派生文档(邮件删除)的落盘附件级联清理(ADR-0005)
1889
+ if base_dir:
1890
+ pst_rel = os.path.relpath(abs_fp, base_dir).replace(os.sep, "/")
1891
+ for rid in removed_ids:
1892
+ entry = rid.rsplit("__", 1)[-1]
1893
+ pst_att_store.purge_email(pst_rel, entry)
1894
+ logger.info("Removed %d stale derived doc(s) for %s",
1895
+ len(removed_ids), fp)
1896
+ for doc in new_docs:
1897
+ fts.index_document(doc, auto_commit=False)
1898
+ d = fts.last_node_diff
1899
+ for k in diff_totals:
1900
+ diff_totals[k] += d[k]
1901
+ _pending_commits += 1
1902
+ docs_since_optimize += 1
1903
+ if _pending_commits >= _COMMIT_BATCH:
1904
+ fts.commit()
1905
+ _pending_commits = 0
1906
+ if optimize_threshold and docs_since_optimize >= optimize_threshold:
1907
+ fts.optimize()
1908
+ docs_since_optimize = 0
1909
+ # 邮件元数据入库(pst_email_meta 表,供列表分页查询,ADR-0005)
1910
+ for t, doc in zip(trees, new_docs):
1911
+ meta = t.get("email_meta")
1912
+ if meta:
1913
+ fts.upsert_email_meta(doc.doc_id, abs_fp, meta)
1914
+ # 文件指纹记在物理文件路径上(派生文档不写 index_meta)
1915
+ fts.set_index_meta(abs_fp, file_h)
1916
+ fts.clear_failed_file(abs_fp)
1917
+ logger.debug("Indexed %d derived docs: %s -> %s",
1918
+ len(new_docs), fp, db_path)
1919
+ documents.extend(new_docs)
1920
+ _save_bar.update()
1921
+ continue
1922
+
1923
+ doc = Document(
1924
+ doc_id=name,
1925
+ doc_name=result.get("doc_name", name),
1926
+ structure=result.get("structure", []),
1927
+ doc_description=result.get("doc_description", ""),
1928
+ metadata={"source_path": result.get("source_path", "")},
1929
+ source_type=result.get("source_type", ""),
1930
+ )
1931
+ # index_document writes nodes, fts_nodes, documents AND index_meta
1932
+ # in a single atomic transaction (auto_commit handles batching).
1933
+ fts.index_document(doc, auto_commit=False, file_hash=file_h)
1934
+ # Clear any prior failure record for this file
1935
+ fts.clear_failed_file(abs_fp)
1936
+ d = fts.last_node_diff
1937
+ for k in diff_totals:
1938
+ diff_totals[k] += d[k]
1939
+ _pending_commits += 1
1940
+ docs_since_optimize += 1
1941
+ if _pending_commits >= _COMMIT_BATCH:
1942
+ fts.commit()
1943
+ _pending_commits = 0
1944
+ if optimize_threshold and docs_since_optimize >= optimize_threshold:
1945
+ fts.optimize()
1946
+ docs_since_optimize = 0
1947
+ logger.debug("Indexed: %s -> %s (doc_id=%s)", fp, db_path, name)
1948
+ else:
1949
+ # Skipped file: use batch-loaded docs (lookup by source_path, not doc_id)
1950
+ abs_fp = os.path.abspath(fp)
1951
+ doc = all_docs_from_db.get(abs_fp)
1952
+ if doc is None:
1953
+ # 多文档来源(PST 等):物理路径无精确匹配,按派生前缀收集
1954
+ derived = [d for sp, d in all_docs_from_db.items()
1955
+ if sp.startswith(abs_fp + "#")]
1956
+ if derived:
1957
+ documents.extend(derived)
1958
+ _save_bar.update()
1959
+ continue
1960
+ logger.debug("Skipped file %s has no document in DB (excluded by failure threshold)", fp)
1961
+ _save_bar.update()
1962
+ continue
1963
+ documents.append(doc)
1964
+ _save_bar.update()
1965
+ # Final commit for remaining pending writes
1966
+ if _pending_commits > 0:
1967
+ logger.info("Committing remaining %d documents to database...", _pending_commits)
1968
+ fts.commit()
1969
+ _save_bar.close()
1970
+
1971
+ # Fold WAL sidecar back into the main DB so long-running daemons don't
1972
+ # accumulate a multi-GB -wal file across many incremental builds.
1973
+ fts.wal_checkpoint("TRUNCATE")
1974
+
1975
+ # ---------------------------------------------------------------
1976
+ # Build IndexStats
1977
+ # ---------------------------------------------------------------
1978
+ build_elapsed = time.monotonic() - build_start
1979
+
1980
+ # Count total nodes in newly indexed documents
1981
+ total_nodes = 0
1982
+ for doc in documents:
1983
+ total_nodes += len(flatten_tree(doc.structure))
1984
+
1985
+ # Per source_type aggregation
1986
+ per_type: dict[str, dict] = {}
1987
+ for fp, (stype, elapsed) in _file_timings.items():
1988
+ if stype == "(failed)":
1989
+ continue
1990
+ entry = per_type.setdefault(stype, {"count": 0, "nodes": 0, "time_s": 0.0})
1991
+ entry["count"] += 1
1992
+ entry["time_s"] += elapsed
1993
+ # Count nodes for this file
1994
+ result = result_map.get(fp)
1995
+ if result:
1996
+ if result.get("multi_docs") is not None:
1997
+ entry["nodes"] += sum(
1998
+ len(flatten_tree(t.get("structure", [])))
1999
+ for t in result["multi_docs"]
2000
+ )
2001
+ else:
2002
+ entry["nodes"] += len(flatten_tree(result.get("structure", [])))
2003
+
2004
+ # Database size
2005
+ db_size = 0
2006
+ if db_path and os.path.isfile(db_path):
2007
+ try:
2008
+ db_size = os.path.getsize(db_path)
2009
+ except OSError:
2010
+ pass
2011
+
2012
+ stats = IndexStats(
2013
+ total_files=len(expanded),
2014
+ indexed_files=len(to_index) - len(_failed_paths),
2015
+ skipped_files=len(skipped),
2016
+ failed_files=len(_failed_paths),
2017
+ excluded_files=excluded_count,
2018
+ total_nodes=total_nodes,
2019
+ total_time_s=build_elapsed,
2020
+ per_type=per_type,
2021
+ db_path=db_path,
2022
+ db_size_bytes=db_size,
2023
+ failed_paths=_failed_paths,
2024
+ node_diff=diff_totals,
2025
+ pruned_paths=pruned_paths,
2026
+ )
2027
+
2028
+ # Attach stats to the returned list for easy access
2029
+ class _DocumentList(list):
2030
+ """List subclass that carries IndexStats."""
2031
+ stats: IndexStats = None # type: ignore[assignment]
2032
+
2033
+ doc_list = _DocumentList(documents)
2034
+ doc_list.stats = stats
2035
+
2036
+ fts.close()
2037
+ _lock_handle.release()
2038
+ return doc_list