treesearchlib 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- treesearch/__init__.py +53 -0
- treesearch/__main__.py +6 -0
- treesearch/_bin/pst-extract.exe +0 -0
- treesearch/cli.py +554 -0
- treesearch/config.py +206 -0
- treesearch/fts.py +2293 -0
- treesearch/heuristics.py +425 -0
- treesearch/indexer.py +2038 -0
- treesearch/parsers/__init__.py +62 -0
- treesearch/parsers/anydoc_parser.py +193 -0
- treesearch/parsers/ast_parser.py +136 -0
- treesearch/parsers/docx_parser.py +304 -0
- treesearch/parsers/email_html_md.py +60 -0
- treesearch/parsers/excel_parser.py +218 -0
- treesearch/parsers/html_parser.py +172 -0
- treesearch/parsers/image_metadata.py +345 -0
- treesearch/parsers/image_parser.py +59 -0
- treesearch/parsers/image_store.py +182 -0
- treesearch/parsers/markitdown_parser.py +258 -0
- treesearch/parsers/mhtml_parser.py +108 -0
- treesearch/parsers/pdf_parser.py +409 -0
- treesearch/parsers/pst_attachment_store.py +156 -0
- treesearch/parsers/pst_parser.py +733 -0
- treesearch/parsers/registry.py +405 -0
- treesearch/parsers/treesitter_parser.py +433 -0
- treesearch/pathutil.py +227 -0
- treesearch/py.typed +0 -0
- treesearch/ripgrep.py +159 -0
- treesearch/search.py +935 -0
- treesearch/tokenizer.py +176 -0
- treesearch/tree.py +393 -0
- treesearch/tree_searcher.py +1006 -0
- treesearch/treesearch.py +574 -0
- treesearch/watch.py +305 -0
- treesearchlib-1.1.0.dist-info/METADATA +124 -0
- treesearchlib-1.1.0.dist-info/RECORD +39 -0
- treesearchlib-1.1.0.dist-info/WHEEL +5 -0
- treesearchlib-1.1.0.dist-info/entry_points.txt +2 -0
- treesearchlib-1.1.0.dist-info/top_level.txt +1 -0
treesearch/indexer.py
ADDED
|
@@ -0,0 +1,2038 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""
|
|
3
|
+
@author:XuMing(xuming624@qq.com)
|
|
4
|
+
@description: Async-first document indexer. Builds tree structure from Markdown or plain text.
|
|
5
|
+
|
|
6
|
+
Supports batch indexing via ``build_index()`` which accepts glob patterns and
|
|
7
|
+
processes multiple files concurrently.
|
|
8
|
+
"""
|
|
9
|
+
import asyncio
|
|
10
|
+
import hashlib
|
|
11
|
+
import json
|
|
12
|
+
import logging
|
|
13
|
+
import os
|
|
14
|
+
import re
|
|
15
|
+
import time
|
|
16
|
+
from contextlib import nullcontext
|
|
17
|
+
from dataclasses import dataclass, field
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from typing import Optional
|
|
20
|
+
|
|
21
|
+
from tqdm import tqdm
|
|
22
|
+
|
|
23
|
+
from .tree import (
|
|
24
|
+
Document, assign_node_ids, flatten_tree, format_structure, remove_fields,
|
|
25
|
+
)
|
|
26
|
+
from .pathutil import resolve_paths, DEFAULT_IGNORE_DIRS, MAX_DIR_FILES, shadow_md_path
|
|
27
|
+
|
|
28
|
+
logger = logging.getLogger(__name__)
|
|
29
|
+
|
|
30
|
+
# PST 解析器输出格式版本盐(折进 PST 文件的指纹):pst_parser 的输出格式
|
|
31
|
+
# 变化(正文转写/元数据/附件落盘,ADR-0005)时 bump,让旧 PST 索引在下次
|
|
32
|
+
# 增量索引自动重建——只影响 PST,不像 INDEX_SCHEMA_VERSION 那样连累全库。
|
|
33
|
+
# 历史:
|
|
34
|
+
# ":pst2" — 2026-07-29 ADR-0005:HTML 正文转写 + 附件全量落盘 + 邮件元数据。
|
|
35
|
+
# ":pst3" — 2026-07-30:邮件头转储正文(Outlook 系统报告类)可读化重排
|
|
36
|
+
# (_reformat_header_dump:解码 encoded-word、折叠超长地址列表)。
|
|
37
|
+
# ":pst4" — 2026-07-30:头转储检测前移到 HTML 转写之前 + 逻辑块解析
|
|
38
|
+
# (body_html 装天书的路径不再漏检)。
|
|
39
|
+
# ":pst5" — 2026-07-30:头转储带完整 RFC822 源码时按 MIME 解析提取内嵌
|
|
40
|
+
# 正文(base64 附件只列名);头行正则兼容空值头(Subject:)。
|
|
41
|
+
PST_PARSER_FINGERPRINT_SALT = ":pst5"
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
# ============================================================================
|
|
45
|
+
# Shared helpers
|
|
46
|
+
# ============================================================================
|
|
47
|
+
|
|
48
|
+
def _generate_shadow_md(binary_path: str) -> None:
|
|
49
|
+
"""Convert a binary file to a hidden Markdown text copy using markitdown.
|
|
50
|
+
|
|
51
|
+
The shadow file (``._<name>.<ext>.md``) lives alongside the source and is
|
|
52
|
+
used by ripgrep fallback search when FTS5 has no results.
|
|
53
|
+
|
|
54
|
+
Skips generation if the shadow file is newer than the source (incremental).
|
|
55
|
+
"""
|
|
56
|
+
md_path = shadow_md_path(binary_path)
|
|
57
|
+
# Incremental: skip if shadow is up-to-date
|
|
58
|
+
try:
|
|
59
|
+
if os.path.exists(md_path) and os.path.getmtime(md_path) >= os.path.getmtime(binary_path):
|
|
60
|
+
return
|
|
61
|
+
except OSError:
|
|
62
|
+
pass
|
|
63
|
+
|
|
64
|
+
try:
|
|
65
|
+
from markitdown import MarkItDown
|
|
66
|
+
except ImportError:
|
|
67
|
+
logger.debug("markitdown not installed, skipping shadow MD for %s", binary_path)
|
|
68
|
+
return
|
|
69
|
+
|
|
70
|
+
md = MarkItDown()
|
|
71
|
+
result = md.convert(binary_path)
|
|
72
|
+
text = result.text_content or ""
|
|
73
|
+
if not text.strip():
|
|
74
|
+
logger.debug("markitdown returned empty content for %s", binary_path)
|
|
75
|
+
return
|
|
76
|
+
|
|
77
|
+
try:
|
|
78
|
+
with open(md_path, "w", encoding="utf-8") as f:
|
|
79
|
+
f.write(text)
|
|
80
|
+
logger.debug("Generated shadow MD: %s", md_path)
|
|
81
|
+
except OSError as e:
|
|
82
|
+
logger.warning("Failed to write shadow MD %s: %s", md_path, e)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _children_indices(node_list: list[dict], parent_idx: int, parent_level: int) -> list[int]:
|
|
86
|
+
"""Return indices of all descendants of node_list[parent_idx]."""
|
|
87
|
+
indices = []
|
|
88
|
+
for j in range(parent_idx + 1, len(node_list)):
|
|
89
|
+
if node_list[j]["level"] <= parent_level:
|
|
90
|
+
break
|
|
91
|
+
indices.append(j)
|
|
92
|
+
return indices
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
# ============================================================================
|
|
96
|
+
# Summary generation (shared by MD and Text)
|
|
97
|
+
# ============================================================================
|
|
98
|
+
|
|
99
|
+
def _summarize_node(node: dict, threshold: int = 600) -> str:
|
|
100
|
+
"""Generate a summary for a single node. Short nodes use their own text.
|
|
101
|
+
|
|
102
|
+
Args:
|
|
103
|
+
threshold: character count threshold. Nodes shorter than this use full text as summary.
|
|
104
|
+
For long nodes: head 250 chars + tail 100 chars (captures intro and conclusion).
|
|
105
|
+
"""
|
|
106
|
+
text = node.get("text", "")
|
|
107
|
+
if len(text) < threshold:
|
|
108
|
+
return text
|
|
109
|
+
head = text[:250].replace("\n", " ").strip()
|
|
110
|
+
tail = text[-100:].replace("\n", " ").strip()
|
|
111
|
+
return f"{head} ... {tail}"
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def generate_summaries(structure, threshold: int = 600):
|
|
115
|
+
"""Generate summaries for all nodes in a tree."""
|
|
116
|
+
nodes = flatten_tree(structure)
|
|
117
|
+
summaries = [_summarize_node(n, threshold=threshold) for n in nodes]
|
|
118
|
+
|
|
119
|
+
for node, summary in zip(nodes, summaries):
|
|
120
|
+
if node.get("nodes"):
|
|
121
|
+
node["prefix_summary"] = summary
|
|
122
|
+
else:
|
|
123
|
+
node["summary"] = summary
|
|
124
|
+
return structure
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def generate_doc_description(structure) -> str:
|
|
128
|
+
"""Generate a document description from its tree structure (no LLM).
|
|
129
|
+
|
|
130
|
+
Extracts top-level titles and first substantial text paragraph.
|
|
131
|
+
"""
|
|
132
|
+
nodes = flatten_tree(structure)
|
|
133
|
+
titles = [n.get("title", "") for n in nodes if n.get("title")][:5]
|
|
134
|
+
title_str = " > ".join(titles)
|
|
135
|
+
text = ""
|
|
136
|
+
for n in nodes:
|
|
137
|
+
t = n.get("text", "")
|
|
138
|
+
if t and len(t) > 20:
|
|
139
|
+
text = t[:200]
|
|
140
|
+
break
|
|
141
|
+
return f"{title_str}. {text}" if text else title_str
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def _finalize_tree(
|
|
145
|
+
tree,
|
|
146
|
+
doc_name: str,
|
|
147
|
+
source_path: str = "",
|
|
148
|
+
source_type: str = "",
|
|
149
|
+
*,
|
|
150
|
+
if_add_node_id: bool = True,
|
|
151
|
+
if_add_node_summary: bool = True,
|
|
152
|
+
summary_chars_threshold: int = 600,
|
|
153
|
+
if_add_node_text: bool = False,
|
|
154
|
+
if_add_doc_description: bool = False,
|
|
155
|
+
) -> dict:
|
|
156
|
+
"""Common post-processing for all *_to_tree functions.
|
|
157
|
+
|
|
158
|
+
Steps: split_oversized_nodes -> assign_node_ids -> format_structure -> generate_summaries -> doc_description.
|
|
159
|
+
"""
|
|
160
|
+
# Split oversized nodes before assigning IDs (so sub-nodes get proper IDs)
|
|
161
|
+
from .config import get_config
|
|
162
|
+
max_node_chars = get_config().max_node_chars
|
|
163
|
+
if max_node_chars:
|
|
164
|
+
tree = _split_oversized_nodes(tree, max_node_chars)
|
|
165
|
+
|
|
166
|
+
if if_add_node_id:
|
|
167
|
+
assign_node_ids(tree)
|
|
168
|
+
|
|
169
|
+
base_order = ["title", "node_id", "summary", "prefix_summary"]
|
|
170
|
+
text_fields = ["text"] if if_add_node_text or if_add_node_summary else []
|
|
171
|
+
tail_fields = ["line_start", "line_end", "nodes"]
|
|
172
|
+
order = base_order + text_fields + tail_fields
|
|
173
|
+
|
|
174
|
+
tree = format_structure(tree, order=order)
|
|
175
|
+
|
|
176
|
+
if if_add_node_summary:
|
|
177
|
+
logger.debug("Generating summaries...")
|
|
178
|
+
tree = generate_summaries(tree, threshold=summary_chars_threshold)
|
|
179
|
+
if not if_add_node_text:
|
|
180
|
+
order_no_text = [f for f in order if f != "text"]
|
|
181
|
+
tree = format_structure(tree, order=order_no_text)
|
|
182
|
+
|
|
183
|
+
result = {"doc_name": doc_name, "structure": tree}
|
|
184
|
+
if source_path:
|
|
185
|
+
result["source_path"] = source_path
|
|
186
|
+
if source_type:
|
|
187
|
+
result["source_type"] = source_type
|
|
188
|
+
|
|
189
|
+
if if_add_doc_description:
|
|
190
|
+
logger.debug("Generating document description...")
|
|
191
|
+
result["doc_description"] = generate_doc_description(tree)
|
|
192
|
+
|
|
193
|
+
return result
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
# ============================================================================
|
|
197
|
+
# Markdown indexer
|
|
198
|
+
# ============================================================================
|
|
199
|
+
|
|
200
|
+
def _extract_md_headings(content: str) -> tuple[list[dict], list[str]]:
|
|
201
|
+
"""Extract heading markers from Markdown content."""
|
|
202
|
+
header_re = re.compile(r"^(#{1,6})\s+(.+)$")
|
|
203
|
+
code_fence = re.compile(r"^```")
|
|
204
|
+
markers = []
|
|
205
|
+
lines = content.split("\n")
|
|
206
|
+
in_code = False
|
|
207
|
+
|
|
208
|
+
for num, line in enumerate(lines, 1):
|
|
209
|
+
stripped = line.strip()
|
|
210
|
+
if code_fence.match(stripped):
|
|
211
|
+
in_code = not in_code
|
|
212
|
+
continue
|
|
213
|
+
if in_code or not stripped:
|
|
214
|
+
continue
|
|
215
|
+
m = header_re.match(stripped)
|
|
216
|
+
if m:
|
|
217
|
+
markers.append({
|
|
218
|
+
"title": m.group(2).strip(),
|
|
219
|
+
"line_num": num,
|
|
220
|
+
"level": len(m.group(1)),
|
|
221
|
+
})
|
|
222
|
+
return markers, lines
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _cut_md_text(markers: list[dict], lines: list[str]) -> list[dict]:
|
|
226
|
+
"""Cut text content between headings."""
|
|
227
|
+
nodes = []
|
|
228
|
+
for i, mk in enumerate(markers):
|
|
229
|
+
start = mk["line_num"] - 1
|
|
230
|
+
end = markers[i + 1]["line_num"] - 1 if i + 1 < len(markers) else len(lines)
|
|
231
|
+
nodes.append({
|
|
232
|
+
"title": mk["title"],
|
|
233
|
+
"line_num": mk["line_num"],
|
|
234
|
+
"line_start": mk["line_num"],
|
|
235
|
+
"line_end": end,
|
|
236
|
+
"level": mk["level"],
|
|
237
|
+
"text": "\n".join(lines[start:end]).strip(),
|
|
238
|
+
})
|
|
239
|
+
return nodes
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def _update_char_counts(node_list: list[dict]) -> list[dict]:
|
|
243
|
+
"""Compute cumulative character counts (self + descendants) for thinning."""
|
|
244
|
+
for i in range(len(node_list) - 1, -1, -1):
|
|
245
|
+
text = node_list[i].get("text", "")
|
|
246
|
+
for ci in _children_indices(node_list, i, node_list[i]["level"]):
|
|
247
|
+
ct = node_list[ci].get("text", "")
|
|
248
|
+
if ct:
|
|
249
|
+
text += "\n" + ct
|
|
250
|
+
node_list[i]["text_char_count"] = len(text)
|
|
251
|
+
return node_list
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def _thin_tree(node_list: list[dict], min_chars: int) -> list[dict]:
|
|
255
|
+
"""Merge small sub-trees into their parent nodes."""
|
|
256
|
+
to_remove = set()
|
|
257
|
+
for i in range(len(node_list) - 1, -1, -1):
|
|
258
|
+
if i in to_remove:
|
|
259
|
+
continue
|
|
260
|
+
if node_list[i].get("text_char_count", 0) < min_chars:
|
|
261
|
+
children = _children_indices(node_list, i, node_list[i]["level"])
|
|
262
|
+
merged_parts = []
|
|
263
|
+
for ci in sorted(children):
|
|
264
|
+
if ci not in to_remove:
|
|
265
|
+
ct = node_list[ci].get("text", "")
|
|
266
|
+
if ct.strip():
|
|
267
|
+
merged_parts.append(ct)
|
|
268
|
+
to_remove.add(ci)
|
|
269
|
+
if merged_parts:
|
|
270
|
+
base = node_list[i].get("text", "")
|
|
271
|
+
node_list[i]["text"] = base + "\n\n" + "\n\n".join(merged_parts) if base else "\n\n".join(merged_parts)
|
|
272
|
+
node_list[i]["text_char_count"] = len(node_list[i]["text"])
|
|
273
|
+
|
|
274
|
+
for idx in sorted(to_remove, reverse=True):
|
|
275
|
+
node_list.pop(idx)
|
|
276
|
+
return node_list
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
def _build_tree(node_list: list[dict]) -> list[dict]:
|
|
280
|
+
"""Build hierarchical tree from flat node list using a stack algorithm."""
|
|
281
|
+
if not node_list:
|
|
282
|
+
return []
|
|
283
|
+
stack = []
|
|
284
|
+
roots = []
|
|
285
|
+
counter = 1
|
|
286
|
+
|
|
287
|
+
for node in node_list:
|
|
288
|
+
level = node["level"]
|
|
289
|
+
tree_node = {
|
|
290
|
+
"title": node["title"],
|
|
291
|
+
"node_id": str(counter),
|
|
292
|
+
"text": node.get("text", ""),
|
|
293
|
+
"line_start": node.get("line_start", node.get("line_num")),
|
|
294
|
+
"line_end": node.get("line_end"),
|
|
295
|
+
"nodes": [],
|
|
296
|
+
}
|
|
297
|
+
counter += 1
|
|
298
|
+
|
|
299
|
+
while stack and stack[-1][1] >= level:
|
|
300
|
+
stack.pop()
|
|
301
|
+
|
|
302
|
+
if not stack:
|
|
303
|
+
roots.append(tree_node)
|
|
304
|
+
else:
|
|
305
|
+
stack[-1][0]["nodes"].append(tree_node)
|
|
306
|
+
|
|
307
|
+
stack.append((tree_node, level))
|
|
308
|
+
return roots
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
# ============================================================================
|
|
312
|
+
# Structure-aware node splitting for oversized nodes
|
|
313
|
+
# ============================================================================
|
|
314
|
+
|
|
315
|
+
def _split_text_by_paragraphs(text: str, max_chars: int) -> list[str]:
|
|
316
|
+
"""Split text into chunks at paragraph boundaries (double newline).
|
|
317
|
+
|
|
318
|
+
Each chunk stays under max_chars. Falls back to single newline
|
|
319
|
+
boundaries, then hard character cut if paragraphs are still too large.
|
|
320
|
+
"""
|
|
321
|
+
if len(text) <= max_chars:
|
|
322
|
+
return [text]
|
|
323
|
+
|
|
324
|
+
# Try splitting by double newline (paragraph boundary)
|
|
325
|
+
chunks = _split_at_boundary(text, max_chars, "\n\n")
|
|
326
|
+
if chunks:
|
|
327
|
+
return chunks
|
|
328
|
+
|
|
329
|
+
# Fallback: split by single newline (line boundary)
|
|
330
|
+
chunks = _split_at_boundary(text, max_chars, "\n")
|
|
331
|
+
if chunks:
|
|
332
|
+
return chunks
|
|
333
|
+
|
|
334
|
+
# Last resort: hard character cut (should rarely happen)
|
|
335
|
+
return [text[i:i + max_chars] for i in range(0, len(text), max_chars)]
|
|
336
|
+
|
|
337
|
+
|
|
338
|
+
def _split_at_boundary(text: str, max_chars: int, separator: str) -> list[str]:
|
|
339
|
+
"""Split text into chunks at the given separator, each under max_chars.
|
|
340
|
+
|
|
341
|
+
Returns None if any segment between separators exceeds max_chars
|
|
342
|
+
(caller should try a finer-grained separator).
|
|
343
|
+
"""
|
|
344
|
+
segments = text.split(separator)
|
|
345
|
+
chunks = []
|
|
346
|
+
current = []
|
|
347
|
+
current_len = 0
|
|
348
|
+
|
|
349
|
+
for seg in segments:
|
|
350
|
+
seg_with_sep = (separator + seg) if current else seg
|
|
351
|
+
new_len = current_len + len(seg_with_sep)
|
|
352
|
+
|
|
353
|
+
if new_len > max_chars and current:
|
|
354
|
+
# Flush current chunk
|
|
355
|
+
chunks.append(separator.join(current))
|
|
356
|
+
current = [seg]
|
|
357
|
+
current_len = len(seg)
|
|
358
|
+
# If a single segment exceeds max_chars, this separator is too coarse
|
|
359
|
+
if current_len > max_chars:
|
|
360
|
+
return None
|
|
361
|
+
else:
|
|
362
|
+
current.append(seg)
|
|
363
|
+
current_len = new_len if current_len > 0 else len(seg)
|
|
364
|
+
|
|
365
|
+
if current:
|
|
366
|
+
chunks.append(separator.join(current))
|
|
367
|
+
|
|
368
|
+
return chunks if chunks else None
|
|
369
|
+
|
|
370
|
+
|
|
371
|
+
def _split_oversized_nodes(tree: list[dict], max_chars: int) -> list[dict]:
|
|
372
|
+
"""Recursively split oversized tree nodes into smaller sub-nodes.
|
|
373
|
+
|
|
374
|
+
For leaf nodes with text exceeding max_chars:
|
|
375
|
+
- Split text at paragraph boundaries (structure-aware)
|
|
376
|
+
- Create child nodes with sequential titles: "Part 1", "Part 2", etc.
|
|
377
|
+
- Parent node retains the original title with empty text
|
|
378
|
+
|
|
379
|
+
For non-leaf nodes: recurse into children first, then check the parent's
|
|
380
|
+
own text (the text before the first child heading).
|
|
381
|
+
"""
|
|
382
|
+
if not max_chars:
|
|
383
|
+
return tree
|
|
384
|
+
|
|
385
|
+
result = []
|
|
386
|
+
for node in tree:
|
|
387
|
+
# Recurse into children first
|
|
388
|
+
children = node.get("nodes", [])
|
|
389
|
+
if children:
|
|
390
|
+
node["nodes"] = _split_oversized_nodes(children, max_chars)
|
|
391
|
+
|
|
392
|
+
text = node.get("text", "")
|
|
393
|
+
if len(text) <= max_chars:
|
|
394
|
+
result.append(node)
|
|
395
|
+
continue
|
|
396
|
+
|
|
397
|
+
# Node text exceeds max_chars: split into sub-nodes
|
|
398
|
+
chunks = _split_text_by_paragraphs(text, max_chars)
|
|
399
|
+
if len(chunks) <= 1:
|
|
400
|
+
result.append(node)
|
|
401
|
+
continue
|
|
402
|
+
|
|
403
|
+
title = node.get("title", "")
|
|
404
|
+
line_start = node.get("line_start")
|
|
405
|
+
line_end = node.get("line_end")
|
|
406
|
+
existing_children = node.get("nodes", [])
|
|
407
|
+
|
|
408
|
+
# Create sub-nodes from text chunks
|
|
409
|
+
sub_nodes = []
|
|
410
|
+
for i, chunk in enumerate(chunks):
|
|
411
|
+
sub_node = {
|
|
412
|
+
"title": f"{title} (part {i + 1})",
|
|
413
|
+
"text": chunk,
|
|
414
|
+
"line_start": line_start,
|
|
415
|
+
"line_end": line_end,
|
|
416
|
+
"nodes": [],
|
|
417
|
+
}
|
|
418
|
+
sub_nodes.append(sub_node)
|
|
419
|
+
|
|
420
|
+
# Attach existing children to the last sub-node (they belong to the tail of the text)
|
|
421
|
+
if existing_children:
|
|
422
|
+
sub_nodes[-1]["nodes"] = existing_children
|
|
423
|
+
|
|
424
|
+
# Parent node becomes a container with truncated text
|
|
425
|
+
node["text"] = ""
|
|
426
|
+
node["nodes"] = sub_nodes
|
|
427
|
+
result.append(node)
|
|
428
|
+
|
|
429
|
+
return result
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
async def md_to_tree(
|
|
433
|
+
md_path: Optional[str] = None,
|
|
434
|
+
md_content: Optional[str] = None,
|
|
435
|
+
*,
|
|
436
|
+
if_thinning: bool = False,
|
|
437
|
+
min_thinning_chars: int = 15000,
|
|
438
|
+
if_add_node_summary: bool = True,
|
|
439
|
+
summary_chars_threshold: int = 600,
|
|
440
|
+
if_add_doc_description: bool = False,
|
|
441
|
+
if_add_node_text: bool = False,
|
|
442
|
+
if_add_node_id: bool = True,
|
|
443
|
+
**kwargs,
|
|
444
|
+
) -> dict:
|
|
445
|
+
"""
|
|
446
|
+
Build a tree index from a Markdown file or string.
|
|
447
|
+
|
|
448
|
+
Returns: {'doc_name': str, 'structure': list, 'doc_description'?: str}
|
|
449
|
+
"""
|
|
450
|
+
if md_path and md_content:
|
|
451
|
+
raise ValueError("Specify only one of md_path or md_content")
|
|
452
|
+
if not md_path and not md_content:
|
|
453
|
+
raise ValueError("Must specify md_path or md_content")
|
|
454
|
+
|
|
455
|
+
if md_path:
|
|
456
|
+
with open(md_path, "r", encoding="utf-8", errors="replace") as f:
|
|
457
|
+
md_content = f.read()
|
|
458
|
+
doc_name = os.path.splitext(os.path.basename(md_path))[0]
|
|
459
|
+
else:
|
|
460
|
+
doc_name = "untitled"
|
|
461
|
+
|
|
462
|
+
logger.debug("Extracting headings from markdown...")
|
|
463
|
+
markers, lines = _extract_md_headings(md_content)
|
|
464
|
+
nodes = _cut_md_text(markers, lines)
|
|
465
|
+
|
|
466
|
+
if if_thinning and min_thinning_chars:
|
|
467
|
+
nodes = _update_char_counts(nodes)
|
|
468
|
+
logger.debug("Thinning tree (threshold=%d chars)...", min_thinning_chars)
|
|
469
|
+
nodes = _thin_tree(nodes, min_thinning_chars)
|
|
470
|
+
|
|
471
|
+
logger.debug("Building tree from %d nodes...", len(nodes))
|
|
472
|
+
tree = _build_tree(nodes)
|
|
473
|
+
|
|
474
|
+
# 无标题回退:为纯文本 Markdown 创建一个根节点
|
|
475
|
+
if not tree and md_content.strip():
|
|
476
|
+
total_lines = len(md_content.split("\n"))
|
|
477
|
+
tree = [{
|
|
478
|
+
"title": doc_name,
|
|
479
|
+
"node_id": "0",
|
|
480
|
+
"text": md_content.strip(),
|
|
481
|
+
"line_start": 1,
|
|
482
|
+
"line_end": total_lines,
|
|
483
|
+
"nodes": [],
|
|
484
|
+
}]
|
|
485
|
+
|
|
486
|
+
return _finalize_tree(
|
|
487
|
+
tree, doc_name,
|
|
488
|
+
source_path=os.path.abspath(md_path) if md_path else "",
|
|
489
|
+
if_add_node_id=if_add_node_id,
|
|
490
|
+
if_add_node_summary=if_add_node_summary,
|
|
491
|
+
summary_chars_threshold=summary_chars_threshold,
|
|
492
|
+
if_add_node_text=if_add_node_text,
|
|
493
|
+
if_add_doc_description=if_add_doc_description,
|
|
494
|
+
)
|
|
495
|
+
|
|
496
|
+
|
|
497
|
+
# ============================================================================
|
|
498
|
+
# Plain text indexer
|
|
499
|
+
# ============================================================================
|
|
500
|
+
|
|
501
|
+
# --- Heading detection patterns ---
|
|
502
|
+
|
|
503
|
+
_RE_NUMERIC = re.compile(r"^(?P<prefix>(?:\d+\.)+\d*)\s*(?P<title>.+)$")
|
|
504
|
+
_RE_PAREN_NUM = re.compile(r"^(?:\(?\d+\))\s+(?P<title>.+)$")
|
|
505
|
+
_RE_ROMAN = re.compile(r"^(?P<prefix>[IVXLCDM]+)\.\s+(?P<title>.+)$")
|
|
506
|
+
_RE_LETTER = re.compile(r"^(?P<prefix>[A-Z])[.)]\s+(?P<title>.+)$")
|
|
507
|
+
_RE_CN_SECTION = re.compile(r"^(?:第[一二三四五六七八九十百千万零\d]+[章节篇部])\s*(?P<title>.*)$")
|
|
508
|
+
_RE_CN_NUM = re.compile(r"^(?P<prefix>[一二三四五六七八九十百千万零]+)[、..]\s*(?P<title>.+)$")
|
|
509
|
+
_RE_CN_PAREN = re.compile(r"^[((](?P<prefix>[一二三四五六七八九十百千万零\d]+)[))]\s*(?P<title>.+)$")
|
|
510
|
+
_RE_RST_UNDERLINE = re.compile(r"^[=\-~^+#]{3,}$")
|
|
511
|
+
_RE_ALL_CAPS = re.compile(r"^[A-Z][A-Z\s\-:,&/]{2,}$")
|
|
512
|
+
|
|
513
|
+
_ROMAN_VALID = {
|
|
514
|
+
"I", "II", "III", "IV", "V", "VI", "VII", "VIII", "IX", "X",
|
|
515
|
+
"XI", "XII", "XIII", "XIV", "XV", "XVI", "XVII", "XVIII", "XIX", "XX",
|
|
516
|
+
}
|
|
517
|
+
|
|
518
|
+
|
|
519
|
+
def _is_short(line: str, limit: int = 80) -> bool:
|
|
520
|
+
return 0 < len(line.strip()) < limit
|
|
521
|
+
|
|
522
|
+
|
|
523
|
+
def _has_blank_neighbor(lines: list, idx: int) -> bool:
|
|
524
|
+
prev_blank = (idx == 0) or (not lines[idx - 1].strip())
|
|
525
|
+
next_blank = (idx >= len(lines) - 1) or (not lines[idx + 1].strip())
|
|
526
|
+
return prev_blank or next_blank
|
|
527
|
+
|
|
528
|
+
|
|
529
|
+
def _detect_headings(lines: list[str]) -> list[dict]:
|
|
530
|
+
"""Detect headings from raw text lines using pattern matching."""
|
|
531
|
+
headings = []
|
|
532
|
+
in_code = False
|
|
533
|
+
|
|
534
|
+
for idx, raw in enumerate(lines):
|
|
535
|
+
line = raw.strip()
|
|
536
|
+
if line.startswith("```"):
|
|
537
|
+
in_code = not in_code
|
|
538
|
+
continue
|
|
539
|
+
if in_code or not line:
|
|
540
|
+
continue
|
|
541
|
+
num = idx + 1
|
|
542
|
+
|
|
543
|
+
# Chinese chapter/section
|
|
544
|
+
m = _RE_CN_SECTION.match(line)
|
|
545
|
+
if m:
|
|
546
|
+
level = 1 if any(c in line for c in "章篇部") else 2
|
|
547
|
+
headings.append({"title": line, "line_num": num, "level": level})
|
|
548
|
+
continue
|
|
549
|
+
|
|
550
|
+
m = _RE_CN_NUM.match(line)
|
|
551
|
+
if m:
|
|
552
|
+
headings.append({"title": line, "line_num": num, "level": 1})
|
|
553
|
+
continue
|
|
554
|
+
|
|
555
|
+
m = _RE_CN_PAREN.match(line)
|
|
556
|
+
if m:
|
|
557
|
+
headings.append({"title": line, "line_num": num, "level": 2})
|
|
558
|
+
continue
|
|
559
|
+
|
|
560
|
+
# Numeric hierarchical (e.g. "1.2 Introduction", "3.1.1 Methods")
|
|
561
|
+
# Require short line and title starts with a letter to avoid matching
|
|
562
|
+
# math expressions like "0.1 + (-0.6) + 0.9 = 0.4..."
|
|
563
|
+
m = _RE_NUMERIC.match(line)
|
|
564
|
+
if m and _is_short(line) and re.match(r"[A-Za-z\u4e00-\u9fff]", m.group("title")):
|
|
565
|
+
level = len(m.group("prefix").rstrip(".").split("."))
|
|
566
|
+
headings.append({"title": line, "line_num": num, "level": level})
|
|
567
|
+
continue
|
|
568
|
+
|
|
569
|
+
# Parenthesized number
|
|
570
|
+
m = _RE_PAREN_NUM.match(line)
|
|
571
|
+
if m:
|
|
572
|
+
headings.append({"title": line, "line_num": num, "level": 2})
|
|
573
|
+
continue
|
|
574
|
+
|
|
575
|
+
# Roman numeral
|
|
576
|
+
m = _RE_ROMAN.match(line)
|
|
577
|
+
if m and m.group("prefix") in _ROMAN_VALID:
|
|
578
|
+
headings.append({"title": line, "line_num": num, "level": 1})
|
|
579
|
+
continue
|
|
580
|
+
|
|
581
|
+
# Letter heading (e.g. "A. Introduction", "B) Methods")
|
|
582
|
+
# Only match short lines to avoid false positives like "D. All these works..."
|
|
583
|
+
m = _RE_LETTER.match(line)
|
|
584
|
+
if m and _is_short(line, 60):
|
|
585
|
+
headings.append({"title": line, "line_num": num, "level": 2})
|
|
586
|
+
continue
|
|
587
|
+
|
|
588
|
+
# RST underline style
|
|
589
|
+
if idx > 0 and _RE_RST_UNDERLINE.match(line):
|
|
590
|
+
prev = lines[idx - 1].strip()
|
|
591
|
+
if prev and _is_short(prev):
|
|
592
|
+
if not headings or headings[-1]["line_num"] != idx:
|
|
593
|
+
level = {"=": 1, "-": 2, "~": 3, "^": 4}.get(line[0], 2)
|
|
594
|
+
headings.append({"title": prev, "line_num": idx, "level": level})
|
|
595
|
+
continue
|
|
596
|
+
|
|
597
|
+
# ALL CAPS
|
|
598
|
+
if _RE_ALL_CAPS.match(line) and _is_short(line) and _has_blank_neighbor(lines, idx):
|
|
599
|
+
headings.append({"title": line, "line_num": num, "level": 1})
|
|
600
|
+
|
|
601
|
+
return headings
|
|
602
|
+
|
|
603
|
+
|
|
604
|
+
def _preprocess_text(text: str) -> str:
|
|
605
|
+
"""Normalize line endings and collapse excessive blank lines."""
|
|
606
|
+
text = text.replace("\r\n", "\n").replace("\r", "\n").replace("\f", "\n")
|
|
607
|
+
return re.sub(r"\n{3,}", "\n\n", text)
|
|
608
|
+
|
|
609
|
+
|
|
610
|
+
|
|
611
|
+
async def text_to_tree(
|
|
612
|
+
text_path: Optional[str] = None,
|
|
613
|
+
text_content: Optional[str] = None,
|
|
614
|
+
*,
|
|
615
|
+
if_thinning: bool = False,
|
|
616
|
+
min_thinning_chars: int = 15000,
|
|
617
|
+
if_add_node_summary: bool = True,
|
|
618
|
+
summary_chars_threshold: int = 600,
|
|
619
|
+
if_add_doc_description: bool = False,
|
|
620
|
+
if_add_node_text: bool = False,
|
|
621
|
+
if_add_node_id: bool = True,
|
|
622
|
+
**kwargs,
|
|
623
|
+
) -> dict:
|
|
624
|
+
"""
|
|
625
|
+
Build a tree index from plain text (pure rule-based, no LLM).
|
|
626
|
+
|
|
627
|
+
Args:
|
|
628
|
+
text_path: path to a .txt file
|
|
629
|
+
text_content: raw text string (alternative to text_path)
|
|
630
|
+
Returns:
|
|
631
|
+
{'doc_name': str, 'structure': list, 'doc_description'?: str}
|
|
632
|
+
"""
|
|
633
|
+
if text_path and text_content:
|
|
634
|
+
raise ValueError("Specify only one of text_path or text_content")
|
|
635
|
+
if not text_path and not text_content:
|
|
636
|
+
raise ValueError("Must specify text_path or text_content")
|
|
637
|
+
|
|
638
|
+
if text_path:
|
|
639
|
+
with open(text_path, "r", encoding="utf-8", errors="replace") as f:
|
|
640
|
+
raw = f.read()
|
|
641
|
+
doc_name = os.path.splitext(os.path.basename(text_path))[0]
|
|
642
|
+
else:
|
|
643
|
+
raw = text_content
|
|
644
|
+
doc_name = "untitled"
|
|
645
|
+
|
|
646
|
+
text = _preprocess_text(raw)
|
|
647
|
+
lines = text.split("\n")
|
|
648
|
+
logger.debug("Text loaded: %d lines", len(lines))
|
|
649
|
+
|
|
650
|
+
# Step 1: heading detection (pure rule-based)
|
|
651
|
+
headings = _detect_headings(lines)
|
|
652
|
+
markers = [{"title": h["title"], "line_num": h["line_num"], "level": h["level"]} for h in headings]
|
|
653
|
+
logger.debug("Rule-based detection: %d headings", len(markers))
|
|
654
|
+
|
|
655
|
+
# Fallback: single root node if no headings detected
|
|
656
|
+
if not markers:
|
|
657
|
+
markers = [{"title": doc_name, "line_num": 1, "level": 1}]
|
|
658
|
+
|
|
659
|
+
# Step 2: extract text
|
|
660
|
+
nodes = _cut_md_text(markers, lines)
|
|
661
|
+
|
|
662
|
+
# Step 3: thinning
|
|
663
|
+
if if_thinning and min_thinning_chars:
|
|
664
|
+
nodes = _update_char_counts(nodes)
|
|
665
|
+
logger.debug("Thinning tree (threshold=%d chars)...", min_thinning_chars)
|
|
666
|
+
nodes = _thin_tree(nodes, min_thinning_chars)
|
|
667
|
+
|
|
668
|
+
# Step 4: build tree
|
|
669
|
+
logger.debug("Building tree from %d nodes...", len(nodes))
|
|
670
|
+
tree = _build_tree(nodes)
|
|
671
|
+
|
|
672
|
+
return _finalize_tree(
|
|
673
|
+
tree, doc_name,
|
|
674
|
+
source_path=os.path.abspath(text_path) if text_path else "",
|
|
675
|
+
if_add_node_id=if_add_node_id,
|
|
676
|
+
if_add_node_summary=if_add_node_summary,
|
|
677
|
+
summary_chars_threshold=summary_chars_threshold,
|
|
678
|
+
if_add_node_text=if_add_node_text,
|
|
679
|
+
if_add_doc_description=if_add_doc_description,
|
|
680
|
+
)
|
|
681
|
+
|
|
682
|
+
|
|
683
|
+
# ============================================================================
|
|
684
|
+
# Code file indexer
|
|
685
|
+
# ============================================================================
|
|
686
|
+
|
|
687
|
+
def _detect_code_headings(lines: list[str], ext: str, source: str = "") -> list[dict]:
|
|
688
|
+
"""Detect classes and methods from code lines.
|
|
689
|
+
|
|
690
|
+
For ``.py`` files, tries AST-based parsing first (richer signatures),
|
|
691
|
+
falling back to regex if AST fails (e.g. syntax errors).
|
|
692
|
+
"""
|
|
693
|
+
# Python: use AST parser for accurate structure extraction
|
|
694
|
+
if ext == ".py" and source:
|
|
695
|
+
from .parsers.ast_parser import parse_python_structure
|
|
696
|
+
headings = parse_python_structure(source)
|
|
697
|
+
if headings:
|
|
698
|
+
return headings
|
|
699
|
+
# AST failed, fall through to regex
|
|
700
|
+
|
|
701
|
+
headings = []
|
|
702
|
+
patterns = []
|
|
703
|
+
if ext == ".py":
|
|
704
|
+
patterns = [
|
|
705
|
+
(re.compile(r"^(class\s+\w+.*)"), 1),
|
|
706
|
+
(re.compile(r"^(\s*def\s+\w+.*)"), 2)
|
|
707
|
+
]
|
|
708
|
+
elif ext in (".java", ".ts", ".js", ".cpp", ".cc", ".cs", ".php"):
|
|
709
|
+
patterns = [
|
|
710
|
+
(re.compile(r"^(\s*(?:public|private|protected|static|abstract|final\s+)*class\s+\w+.*)"), 1),
|
|
711
|
+
(re.compile(r"^(\s*(?:public|private|protected|static|abstract|final\s+)*interface\s+\w+.*)"), 1),
|
|
712
|
+
(re.compile(r"^(\s*(?:public|private|protected|static|abstract|final\s+)*(?:[\w<>\[\]]+\s+)+\w+\s*\(.*)"), 2),
|
|
713
|
+
(re.compile(r"^(\s*function\s+\w+.*)"), 2)
|
|
714
|
+
]
|
|
715
|
+
elif ext == ".go":
|
|
716
|
+
patterns = [
|
|
717
|
+
(re.compile(r"^(\s*type\s+\w+\s+struct.*)"), 1),
|
|
718
|
+
(re.compile(r"^(\s*type\s+\w+\s+interface.*)"), 1),
|
|
719
|
+
(re.compile(r"^(\s*func\s+(?:\([^)]+\)\s+)?\w+.*)"), 2)
|
|
720
|
+
]
|
|
721
|
+
elif ext == ".html":
|
|
722
|
+
patterns = [
|
|
723
|
+
(re.compile(r"^\s*<h1.*>(.*)</h1>"), 1),
|
|
724
|
+
(re.compile(r"^\s*<h2.*>(.*)</h2>"), 2),
|
|
725
|
+
(re.compile(r"^\s*<h3.*>(.*)</h3>"), 3),
|
|
726
|
+
(re.compile(r"^\s*<div.*id=\"(.*)\".*>"), 2),
|
|
727
|
+
(re.compile(r"^\s*<section.*id=\"(.*)\".*>"), 2)
|
|
728
|
+
]
|
|
729
|
+
elif ext == ".xml":
|
|
730
|
+
patterns = [
|
|
731
|
+
(re.compile(r"^\s*<(\w+).*>\s*$"), 1),
|
|
732
|
+
]
|
|
733
|
+
|
|
734
|
+
if not patterns:
|
|
735
|
+
return []
|
|
736
|
+
|
|
737
|
+
for idx, raw in enumerate(lines):
|
|
738
|
+
line = raw.rstrip()
|
|
739
|
+
if not line:
|
|
740
|
+
continue
|
|
741
|
+
num = idx + 1
|
|
742
|
+
|
|
743
|
+
for pat, level in patterns:
|
|
744
|
+
m = pat.match(line)
|
|
745
|
+
if m:
|
|
746
|
+
title = m.group(1).strip().rstrip(":{").strip()[:100]
|
|
747
|
+
headings.append({"title": title, "line_num": num, "level": level})
|
|
748
|
+
break
|
|
749
|
+
|
|
750
|
+
return headings
|
|
751
|
+
|
|
752
|
+
|
|
753
|
+
async def code_to_tree(
|
|
754
|
+
code_path: str,
|
|
755
|
+
*,
|
|
756
|
+
if_thinning: bool = False,
|
|
757
|
+
min_thinning_chars: int = 15000,
|
|
758
|
+
if_add_node_summary: bool = True,
|
|
759
|
+
summary_chars_threshold: int = 600,
|
|
760
|
+
if_add_doc_description: bool = False,
|
|
761
|
+
if_add_node_text: bool = False,
|
|
762
|
+
if_add_node_id: bool = True,
|
|
763
|
+
**kwargs,
|
|
764
|
+
) -> dict:
|
|
765
|
+
"""
|
|
766
|
+
Build a tree index from a code file.
|
|
767
|
+
|
|
768
|
+
Returns:
|
|
769
|
+
{'doc_name': str, 'structure': list, 'doc_description'?: str}
|
|
770
|
+
"""
|
|
771
|
+
with open(code_path, "r", encoding="utf-8", errors="replace") as f:
|
|
772
|
+
raw = f.read()
|
|
773
|
+
doc_name = os.path.splitext(os.path.basename(code_path))[0]
|
|
774
|
+
ext = os.path.splitext(code_path)[1].lower()
|
|
775
|
+
|
|
776
|
+
text = raw.replace("\r\n", "\n").replace("\r", "\n")
|
|
777
|
+
lines = text.split("\n")
|
|
778
|
+
logger.debug("Code loaded: %d lines", len(lines))
|
|
779
|
+
|
|
780
|
+
headings = _detect_code_headings(lines, ext, source=text)
|
|
781
|
+
markers = [{"title": h["title"], "line_num": h["line_num"], "level": h["level"]} for h in headings]
|
|
782
|
+
logger.debug("Code structure detection: %d methods/classes", len(markers))
|
|
783
|
+
|
|
784
|
+
if not markers:
|
|
785
|
+
markers = [{"title": doc_name, "line_num": 1, "level": 1}]
|
|
786
|
+
|
|
787
|
+
nodes = _cut_md_text(markers, lines)
|
|
788
|
+
|
|
789
|
+
if if_thinning and min_thinning_chars:
|
|
790
|
+
nodes = _update_char_counts(nodes)
|
|
791
|
+
logger.debug("Thinning tree (threshold=%d chars)...", min_thinning_chars)
|
|
792
|
+
nodes = _thin_tree(nodes, min_thinning_chars)
|
|
793
|
+
|
|
794
|
+
logger.debug("Building tree from %d nodes...", len(nodes))
|
|
795
|
+
tree = _build_tree(nodes)
|
|
796
|
+
|
|
797
|
+
return _finalize_tree(
|
|
798
|
+
tree, doc_name,
|
|
799
|
+
source_path=os.path.abspath(code_path),
|
|
800
|
+
if_add_node_id=if_add_node_id,
|
|
801
|
+
if_add_node_summary=if_add_node_summary,
|
|
802
|
+
summary_chars_threshold=summary_chars_threshold,
|
|
803
|
+
if_add_node_text=if_add_node_text,
|
|
804
|
+
if_add_doc_description=if_add_doc_description,
|
|
805
|
+
)
|
|
806
|
+
|
|
807
|
+
|
|
808
|
+
# ============================================================================
|
|
809
|
+
# JSON file indexer
|
|
810
|
+
# ============================================================================
|
|
811
|
+
|
|
812
|
+
def _json_to_nodes(data, prefix: str = "", level: int = 1) -> list[dict]:
|
|
813
|
+
"""Recursively convert JSON data into flat node list."""
|
|
814
|
+
nodes = []
|
|
815
|
+
if isinstance(data, dict):
|
|
816
|
+
for key, value in data.items():
|
|
817
|
+
path = f"{prefix}.{key}" if prefix else key
|
|
818
|
+
if isinstance(value, (dict, list)):
|
|
819
|
+
nodes.append({"title": path, "level": level, "text": ""})
|
|
820
|
+
nodes.extend(_json_to_nodes(value, prefix=path, level=level + 1))
|
|
821
|
+
else:
|
|
822
|
+
nodes.append({"title": path, "level": level, "text": f"{key}: {value}"})
|
|
823
|
+
elif isinstance(data, list):
|
|
824
|
+
for i, item in enumerate(data):
|
|
825
|
+
path = f"{prefix}[{i}]"
|
|
826
|
+
if isinstance(item, (dict, list)):
|
|
827
|
+
nodes.append({"title": path, "level": level, "text": ""})
|
|
828
|
+
nodes.extend(_json_to_nodes(item, prefix=path, level=level + 1))
|
|
829
|
+
else:
|
|
830
|
+
nodes.append({"title": path, "level": level, "text": str(item)})
|
|
831
|
+
return nodes
|
|
832
|
+
|
|
833
|
+
|
|
834
|
+
def _load_json_lenient(raw: str):
|
|
835
|
+
"""Load JSON tolerantly: handle control characters, comments, and trailing commas."""
|
|
836
|
+
# First try with strict=False to tolerate control characters
|
|
837
|
+
try:
|
|
838
|
+
return json.loads(raw, strict=False)
|
|
839
|
+
except json.JSONDecodeError:
|
|
840
|
+
pass
|
|
841
|
+
|
|
842
|
+
# Strip trailing commas before } or ]
|
|
843
|
+
stripped = re.sub(r',\s*([}\]])', r'\1', raw)
|
|
844
|
+
# Strip // comments, but preserve // inside quoted strings (e.g. URLs)
|
|
845
|
+
stripped = re.sub(r'("(?:[^"\\]|\\.)*")|//[^\n]*', r'\1', stripped)
|
|
846
|
+
return json.loads(stripped, strict=False)
|
|
847
|
+
|
|
848
|
+
|
|
849
|
+
async def json_to_tree(
|
|
850
|
+
json_path: str,
|
|
851
|
+
*,
|
|
852
|
+
if_add_node_summary: bool = True,
|
|
853
|
+
summary_chars_threshold: int = 600,
|
|
854
|
+
if_add_doc_description: bool = False,
|
|
855
|
+
if_add_node_text: bool = False,
|
|
856
|
+
if_add_node_id: bool = True,
|
|
857
|
+
**kwargs,
|
|
858
|
+
) -> dict:
|
|
859
|
+
"""Build a tree index from a JSON file."""
|
|
860
|
+
with open(json_path, "r", encoding="utf-8", errors="replace") as f:
|
|
861
|
+
raw = f.read()
|
|
862
|
+
doc_name = os.path.splitext(os.path.basename(json_path))[0]
|
|
863
|
+
|
|
864
|
+
try:
|
|
865
|
+
data = _load_json_lenient(raw)
|
|
866
|
+
except (json.JSONDecodeError, ValueError):
|
|
867
|
+
# JSON 解析失败,降级为纯文本
|
|
868
|
+
logger.debug("JSON parse failed for %s, falling back to text", json_path)
|
|
869
|
+
result = await text_to_tree(
|
|
870
|
+
text_content=raw,
|
|
871
|
+
if_add_node_summary=if_add_node_summary,
|
|
872
|
+
summary_chars_threshold=summary_chars_threshold,
|
|
873
|
+
if_add_doc_description=if_add_doc_description,
|
|
874
|
+
if_add_node_text=if_add_node_text,
|
|
875
|
+
if_add_node_id=if_add_node_id,
|
|
876
|
+
**kwargs,
|
|
877
|
+
)
|
|
878
|
+
result["doc_name"] = doc_name
|
|
879
|
+
result["source_path"] = os.path.abspath(json_path)
|
|
880
|
+
return result
|
|
881
|
+
|
|
882
|
+
flat_nodes = _json_to_nodes(data)
|
|
883
|
+
if not flat_nodes:
|
|
884
|
+
flat_nodes = [{"title": doc_name, "level": 1, "text": json.dumps(data, ensure_ascii=False)[:500]}]
|
|
885
|
+
|
|
886
|
+
# Assign line_num for _build_tree compatibility
|
|
887
|
+
for i, node in enumerate(flat_nodes):
|
|
888
|
+
node["line_num"] = i + 1
|
|
889
|
+
node["line_start"] = i + 1
|
|
890
|
+
node["line_end"] = i + 1
|
|
891
|
+
|
|
892
|
+
tree = _build_tree(flat_nodes)
|
|
893
|
+
|
|
894
|
+
return _finalize_tree(
|
|
895
|
+
tree, doc_name,
|
|
896
|
+
source_path=os.path.abspath(json_path),
|
|
897
|
+
if_add_node_id=if_add_node_id,
|
|
898
|
+
if_add_node_summary=if_add_node_summary,
|
|
899
|
+
summary_chars_threshold=summary_chars_threshold,
|
|
900
|
+
if_add_node_text=if_add_node_text,
|
|
901
|
+
if_add_doc_description=if_add_doc_description,
|
|
902
|
+
)
|
|
903
|
+
|
|
904
|
+
|
|
905
|
+
# ============================================================================
|
|
906
|
+
# JSONL file indexer
|
|
907
|
+
# ============================================================================
|
|
908
|
+
|
|
909
|
+
def _jsonl_to_nodes(records: list[dict], key_field: str = None) -> list[dict]:
|
|
910
|
+
"""Convert a list of JSONL records into flat node list.
|
|
911
|
+
|
|
912
|
+
Each record becomes a level-1 node. If key_field is specified and exists
|
|
913
|
+
in the record, it is used as the title; otherwise uses the record index.
|
|
914
|
+
Nested structures within each record are expanded as child nodes.
|
|
915
|
+
"""
|
|
916
|
+
nodes = []
|
|
917
|
+
for i, record in enumerate(records):
|
|
918
|
+
# Determine title for this record
|
|
919
|
+
if key_field and isinstance(record, dict) and key_field in record:
|
|
920
|
+
title = str(record[key_field])
|
|
921
|
+
elif isinstance(record, dict):
|
|
922
|
+
# Auto-detect: use first string-valued field as title
|
|
923
|
+
title = None
|
|
924
|
+
for k, v in record.items():
|
|
925
|
+
if isinstance(v, str) and len(v) < 200:
|
|
926
|
+
title = f"{k}: {v}"
|
|
927
|
+
break
|
|
928
|
+
if title is None:
|
|
929
|
+
title = f"record[{i}]"
|
|
930
|
+
else:
|
|
931
|
+
title = f"record[{i}]"
|
|
932
|
+
|
|
933
|
+
# Build text from record content
|
|
934
|
+
if isinstance(record, dict):
|
|
935
|
+
text_parts = []
|
|
936
|
+
child_nodes = []
|
|
937
|
+
for k, v in record.items():
|
|
938
|
+
if isinstance(v, (dict, list)):
|
|
939
|
+
child_nodes.extend(_json_to_nodes(v, prefix=f"record[{i}].{k}", level=2))
|
|
940
|
+
else:
|
|
941
|
+
text_parts.append(f"{k}: {v}")
|
|
942
|
+
text = "\n".join(text_parts)
|
|
943
|
+
nodes.append({"title": title, "level": 1, "text": text})
|
|
944
|
+
nodes.extend(child_nodes)
|
|
945
|
+
else:
|
|
946
|
+
nodes.append({"title": title, "level": 1, "text": str(record)})
|
|
947
|
+
|
|
948
|
+
return nodes
|
|
949
|
+
|
|
950
|
+
|
|
951
|
+
async def jsonl_to_tree(
|
|
952
|
+
jsonl_path: str,
|
|
953
|
+
*,
|
|
954
|
+
key_field: str = None,
|
|
955
|
+
if_add_node_summary: bool = True,
|
|
956
|
+
summary_chars_threshold: int = 600,
|
|
957
|
+
if_add_doc_description: bool = False,
|
|
958
|
+
if_add_node_text: bool = False,
|
|
959
|
+
if_add_node_id: bool = True,
|
|
960
|
+
**kwargs,
|
|
961
|
+
) -> dict:
|
|
962
|
+
"""Build a tree index from a JSONL file (one JSON object per line).
|
|
963
|
+
|
|
964
|
+
Args:
|
|
965
|
+
jsonl_path: path to the .jsonl file
|
|
966
|
+
key_field: optional field name to use as record title
|
|
967
|
+
"""
|
|
968
|
+
records = []
|
|
969
|
+
with open(jsonl_path, "r", encoding="utf-8", errors="replace") as f:
|
|
970
|
+
for line_num, line in enumerate(f, 1):
|
|
971
|
+
line = line.strip()
|
|
972
|
+
if not line:
|
|
973
|
+
continue
|
|
974
|
+
try:
|
|
975
|
+
records.append(json.loads(line))
|
|
976
|
+
except json.JSONDecodeError as e:
|
|
977
|
+
logger.warning("Skipping invalid JSON at line %d in %s: %s", line_num, jsonl_path, e)
|
|
978
|
+
|
|
979
|
+
doc_name = os.path.splitext(os.path.basename(jsonl_path))[0]
|
|
980
|
+
logger.debug("JSONL loaded: %d records from %s", len(records), jsonl_path)
|
|
981
|
+
|
|
982
|
+
flat_nodes = _jsonl_to_nodes(records, key_field=key_field)
|
|
983
|
+
if not flat_nodes:
|
|
984
|
+
flat_nodes = [{"title": doc_name, "level": 1, "text": ""}]
|
|
985
|
+
|
|
986
|
+
for i, node in enumerate(flat_nodes):
|
|
987
|
+
node["line_num"] = i + 1
|
|
988
|
+
node["line_start"] = i + 1
|
|
989
|
+
node["line_end"] = i + 1
|
|
990
|
+
|
|
991
|
+
tree = _build_tree(flat_nodes)
|
|
992
|
+
|
|
993
|
+
return _finalize_tree(
|
|
994
|
+
tree, doc_name,
|
|
995
|
+
source_path=os.path.abspath(jsonl_path),
|
|
996
|
+
if_add_node_id=if_add_node_id,
|
|
997
|
+
if_add_node_summary=if_add_node_summary,
|
|
998
|
+
summary_chars_threshold=summary_chars_threshold,
|
|
999
|
+
if_add_node_text=if_add_node_text,
|
|
1000
|
+
if_add_doc_description=if_add_doc_description,
|
|
1001
|
+
)
|
|
1002
|
+
|
|
1003
|
+
|
|
1004
|
+
# ============================================================================
|
|
1005
|
+
# CSV file indexer
|
|
1006
|
+
# ============================================================================
|
|
1007
|
+
|
|
1008
|
+
async def csv_to_tree(
|
|
1009
|
+
csv_path: str,
|
|
1010
|
+
*,
|
|
1011
|
+
if_add_node_summary: bool = True,
|
|
1012
|
+
summary_chars_threshold: int = 600,
|
|
1013
|
+
if_add_doc_description: bool = False,
|
|
1014
|
+
if_add_node_text: bool = False,
|
|
1015
|
+
if_add_node_id: bool = True,
|
|
1016
|
+
**kwargs,
|
|
1017
|
+
) -> dict:
|
|
1018
|
+
"""Build a tree index from a CSV file. Each row becomes a leaf node under a header node."""
|
|
1019
|
+
import csv as csvmod
|
|
1020
|
+
|
|
1021
|
+
with open(csv_path, "r", encoding="utf-8", errors="replace") as f:
|
|
1022
|
+
reader = csvmod.reader(f)
|
|
1023
|
+
rows = list(reader)
|
|
1024
|
+
|
|
1025
|
+
doc_name = os.path.splitext(os.path.basename(csv_path))[0]
|
|
1026
|
+
if not rows:
|
|
1027
|
+
return {"doc_name": doc_name, "structure": [{"title": doc_name, "node_id": "0001", "nodes": []}]}
|
|
1028
|
+
|
|
1029
|
+
headers = rows[0]
|
|
1030
|
+
flat_nodes = [{"title": doc_name, "level": 1, "text": f"Columns: {', '.join(headers)}", "line_num": 1, "line_start": 1, "line_end": 1}]
|
|
1031
|
+
|
|
1032
|
+
for i, row in enumerate(rows[1:], start=2):
|
|
1033
|
+
row_text = "; ".join(f"{h}: {v}" for h, v in zip(headers, row) if v.strip())
|
|
1034
|
+
title = row_text[:80] if row_text else f"Row {i}"
|
|
1035
|
+
flat_nodes.append({"title": title, "level": 2, "text": row_text, "line_num": i, "line_start": i, "line_end": i})
|
|
1036
|
+
|
|
1037
|
+
tree = _build_tree(flat_nodes)
|
|
1038
|
+
|
|
1039
|
+
return _finalize_tree(
|
|
1040
|
+
tree, doc_name,
|
|
1041
|
+
source_path=os.path.abspath(csv_path),
|
|
1042
|
+
if_add_node_id=if_add_node_id,
|
|
1043
|
+
if_add_node_summary=if_add_node_summary,
|
|
1044
|
+
summary_chars_threshold=summary_chars_threshold,
|
|
1045
|
+
if_add_node_text=if_add_node_text,
|
|
1046
|
+
if_add_doc_description=if_add_doc_description,
|
|
1047
|
+
)
|
|
1048
|
+
|
|
1049
|
+
|
|
1050
|
+
# ============================================================================
|
|
1051
|
+
# Index statistics
|
|
1052
|
+
# ============================================================================
|
|
1053
|
+
|
|
1054
|
+
@dataclass
|
|
1055
|
+
class IndexStats:
|
|
1056
|
+
"""Statistics collected during an indexing run.
|
|
1057
|
+
|
|
1058
|
+
Attributes:
|
|
1059
|
+
total_files: Total files discovered (including skipped).
|
|
1060
|
+
indexed_files: Files actually (re-)indexed in this run.
|
|
1061
|
+
skipped_files: Files skipped because they were unchanged.
|
|
1062
|
+
failed_files: Files that failed to parse.
|
|
1063
|
+
total_nodes: Total tree nodes generated across all indexed files.
|
|
1064
|
+
total_time_s: Total wall-clock time for the indexing run.
|
|
1065
|
+
per_type: Breakdown by source_type with counts, node totals, and timings.
|
|
1066
|
+
db_path: Path to the SQLite database file.
|
|
1067
|
+
db_size_bytes: Size of the database file on disk (0 for in-memory).
|
|
1068
|
+
failed_paths: List of file paths that failed to index.
|
|
1069
|
+
node_diff: Aggregate node-level diff across reindexed docs:
|
|
1070
|
+
``{"added", "changed", "removed", "kept"}``.
|
|
1071
|
+
Useful for verifying incremental behaviour.
|
|
1072
|
+
pruned_paths: Source paths whose documents were removed from the
|
|
1073
|
+
index because the file no longer exists in the indexed scope.
|
|
1074
|
+
"""
|
|
1075
|
+
total_files: int = 0
|
|
1076
|
+
indexed_files: int = 0
|
|
1077
|
+
skipped_files: int = 0
|
|
1078
|
+
failed_files: int = 0
|
|
1079
|
+
excluded_files: int = 0
|
|
1080
|
+
total_nodes: int = 0
|
|
1081
|
+
total_time_s: float = 0.0
|
|
1082
|
+
per_type: dict = field(default_factory=dict)
|
|
1083
|
+
db_path: str = ""
|
|
1084
|
+
db_size_bytes: int = 0
|
|
1085
|
+
failed_paths: list = field(default_factory=list)
|
|
1086
|
+
node_diff: dict = field(
|
|
1087
|
+
default_factory=lambda: {"added": 0, "changed": 0, "removed": 0, "kept": 0}
|
|
1088
|
+
)
|
|
1089
|
+
pruned_paths: list = field(default_factory=list)
|
|
1090
|
+
|
|
1091
|
+
def summary(self) -> str:
|
|
1092
|
+
"""Return a human-readable summary string."""
|
|
1093
|
+
lines = []
|
|
1094
|
+
lines.append(f"Index Statistics")
|
|
1095
|
+
lines.append(f" Total files discovered: {self.total_files}")
|
|
1096
|
+
lines.append(f" Indexed (new/changed): {self.indexed_files}")
|
|
1097
|
+
lines.append(f" Skipped (unchanged): {self.skipped_files}")
|
|
1098
|
+
if self.failed_files:
|
|
1099
|
+
lines.append(f" Failed: {self.failed_files}")
|
|
1100
|
+
if self.excluded_files:
|
|
1101
|
+
lines.append(f" Exceeded failures: {self.excluded_files}")
|
|
1102
|
+
if self.pruned_paths:
|
|
1103
|
+
lines.append(f" Pruned (orphans): {len(self.pruned_paths)}")
|
|
1104
|
+
lines.append(f" Total nodes generated: {self.total_nodes}")
|
|
1105
|
+
nd = self.node_diff
|
|
1106
|
+
if any(nd.values()):
|
|
1107
|
+
lines.append(
|
|
1108
|
+
f" Node-level diff: +{nd['added']} ~{nd['changed']} "
|
|
1109
|
+
f"-{nd['removed']} (kept {nd['kept']})"
|
|
1110
|
+
)
|
|
1111
|
+
lines.append(f" Total time: {self.total_time_s:.3f}s")
|
|
1112
|
+
if self.db_path:
|
|
1113
|
+
size_str = _format_size(self.db_size_bytes)
|
|
1114
|
+
lines.append(f" Database: {self.db_path} ({size_str})")
|
|
1115
|
+
|
|
1116
|
+
if self.per_type:
|
|
1117
|
+
lines.append(f"")
|
|
1118
|
+
lines.append(f" Per file type:")
|
|
1119
|
+
# Sort by file count descending
|
|
1120
|
+
for stype, info in sorted(self.per_type.items(), key=lambda x: -x[1]["count"]):
|
|
1121
|
+
cnt = info["count"]
|
|
1122
|
+
nodes = info["nodes"]
|
|
1123
|
+
t = info["time_s"]
|
|
1124
|
+
lines.append(f" {stype:12s} {cnt:4d} file(s) {nodes:5d} nodes {t:.3f}s")
|
|
1125
|
+
|
|
1126
|
+
if self.failed_paths:
|
|
1127
|
+
lines.append(f"")
|
|
1128
|
+
lines.append(f" Failed files:")
|
|
1129
|
+
for fp in self.failed_paths[:10]:
|
|
1130
|
+
lines.append(f" - {fp}")
|
|
1131
|
+
if len(self.failed_paths) > 10:
|
|
1132
|
+
lines.append(f" ... and {len(self.failed_paths) - 10} more")
|
|
1133
|
+
|
|
1134
|
+
return "\n".join(lines)
|
|
1135
|
+
|
|
1136
|
+
|
|
1137
|
+
def _format_size(size_bytes: int) -> str:
|
|
1138
|
+
"""Format bytes into human-readable string."""
|
|
1139
|
+
if size_bytes < 1024:
|
|
1140
|
+
return f"{size_bytes} B"
|
|
1141
|
+
elif size_bytes < 1024 * 1024:
|
|
1142
|
+
return f"{size_bytes / 1024:.1f} KB"
|
|
1143
|
+
else:
|
|
1144
|
+
return f"{size_bytes / (1024 * 1024):.1f} MB"
|
|
1145
|
+
|
|
1146
|
+
|
|
1147
|
+
# ============================================================================
|
|
1148
|
+
# Batch indexing API
|
|
1149
|
+
# ============================================================================
|
|
1150
|
+
|
|
1151
|
+
class _NullLock:
|
|
1152
|
+
"""No-op lock context for in-memory or non-POSIX paths."""
|
|
1153
|
+
def __enter__(self):
|
|
1154
|
+
return self
|
|
1155
|
+
def __exit__(self, *a):
|
|
1156
|
+
return False
|
|
1157
|
+
def release(self) -> None:
|
|
1158
|
+
return None
|
|
1159
|
+
|
|
1160
|
+
|
|
1161
|
+
def _acquire_index_lock(db_path: str):
|
|
1162
|
+
"""Acquire an exclusive advisory lock on ``{db_path}.lock``.
|
|
1163
|
+
|
|
1164
|
+
Returns a handle whose ``release()`` method closes the file (which
|
|
1165
|
+
releases the lock). Returns a no-op handle for in-memory or non-existent
|
|
1166
|
+
paths, and on platforms without ``fcntl`` (e.g. Windows).
|
|
1167
|
+
"""
|
|
1168
|
+
if not db_path or db_path == ":memory:":
|
|
1169
|
+
return _NullLock()
|
|
1170
|
+
|
|
1171
|
+
# Unix: 使用 fcntl.flock
|
|
1172
|
+
try:
|
|
1173
|
+
import fcntl
|
|
1174
|
+
except ImportError:
|
|
1175
|
+
pass
|
|
1176
|
+
else:
|
|
1177
|
+
lock_path = db_path + ".lock"
|
|
1178
|
+
os.makedirs(os.path.dirname(os.path.abspath(lock_path)), exist_ok=True)
|
|
1179
|
+
f = open(lock_path, "w")
|
|
1180
|
+
try:
|
|
1181
|
+
fcntl.flock(f.fileno(), fcntl.LOCK_EX)
|
|
1182
|
+
except OSError as e:
|
|
1183
|
+
f.close()
|
|
1184
|
+
logger.warning("Failed to acquire index lock %s: %s", lock_path, e)
|
|
1185
|
+
return _NullLock()
|
|
1186
|
+
|
|
1187
|
+
class _Handle:
|
|
1188
|
+
def __init__(self, fh):
|
|
1189
|
+
self._fh = fh
|
|
1190
|
+
def __enter__(self):
|
|
1191
|
+
return self
|
|
1192
|
+
def __exit__(self, *a):
|
|
1193
|
+
self.release()
|
|
1194
|
+
return False
|
|
1195
|
+
def release(self) -> None:
|
|
1196
|
+
if self._fh is not None:
|
|
1197
|
+
try:
|
|
1198
|
+
fcntl.flock(self._fh.fileno(), fcntl.LOCK_UN)
|
|
1199
|
+
finally:
|
|
1200
|
+
self._fh.close()
|
|
1201
|
+
self._fh = None
|
|
1202
|
+
|
|
1203
|
+
return _Handle(f)
|
|
1204
|
+
|
|
1205
|
+
# Windows: 使用 msvcrt.locking
|
|
1206
|
+
try:
|
|
1207
|
+
import msvcrt
|
|
1208
|
+
except ImportError:
|
|
1209
|
+
logger.warning("No file locking available on this platform (no fcntl, no msvcrt)")
|
|
1210
|
+
return _NullLock()
|
|
1211
|
+
|
|
1212
|
+
lock_path = db_path + ".lock"
|
|
1213
|
+
os.makedirs(os.path.dirname(os.path.abspath(lock_path)), exist_ok=True)
|
|
1214
|
+
f = open(lock_path, "w+")
|
|
1215
|
+
|
|
1216
|
+
class _WindowsHandle:
|
|
1217
|
+
def __init__(self, fh):
|
|
1218
|
+
self._fh = fh
|
|
1219
|
+
def __enter__(self):
|
|
1220
|
+
return self
|
|
1221
|
+
def __exit__(self, *a):
|
|
1222
|
+
self.release()
|
|
1223
|
+
return False
|
|
1224
|
+
def release(self) -> None:
|
|
1225
|
+
if self._fh is not None:
|
|
1226
|
+
try:
|
|
1227
|
+
msvcrt.locking(self._fh.fileno(), msvcrt.LK_UNLCK, 0)
|
|
1228
|
+
except (OSError, IOError):
|
|
1229
|
+
pass # Lock auto-released on close on Windows
|
|
1230
|
+
finally:
|
|
1231
|
+
self._fh.close()
|
|
1232
|
+
self._fh = None
|
|
1233
|
+
|
|
1234
|
+
import time as _time
|
|
1235
|
+
# 重试:瞬态锁冲突(同进程内并发的 build_index)不该让本次索引直接放弃并返回 []
|
|
1236
|
+
# (会把 documents 清空,导致 files API 的 indexed 标志全部消失)。
|
|
1237
|
+
last_err = None
|
|
1238
|
+
for _attempt in range(10):
|
|
1239
|
+
try:
|
|
1240
|
+
msvcrt.locking(f.fileno(), msvcrt.LK_NBLCK, 1)
|
|
1241
|
+
return _WindowsHandle(f)
|
|
1242
|
+
except (OSError, IOError) as e:
|
|
1243
|
+
last_err = e
|
|
1244
|
+
_time.sleep(0.2)
|
|
1245
|
+
f.close()
|
|
1246
|
+
logger.warning("Failed to acquire Windows index lock %s after retries: %s", lock_path, last_err)
|
|
1247
|
+
return _NullLock()
|
|
1248
|
+
|
|
1249
|
+
|
|
1250
|
+
def file_hash(fp: str, mode: Optional[str] = None) -> str:
|
|
1251
|
+
"""Compute a fingerprint for incremental indexing.
|
|
1252
|
+
|
|
1253
|
+
Format: ``"v{INDEX_SCHEMA_VERSION}:{mode}:{payload}"`` so that:
|
|
1254
|
+
- bumping ``INDEX_SCHEMA_VERSION`` invalidates every stored hash, forcing a
|
|
1255
|
+
clean rebuild after a parser/tokenizer/schema change.
|
|
1256
|
+
- switching ``fingerprint_mode`` between ``stat`` and ``content`` likewise
|
|
1257
|
+
invalidates so users can opt-in safely.
|
|
1258
|
+
|
|
1259
|
+
Modes:
|
|
1260
|
+
- ``stat`` (default): ``(mtime_ns:size)`` — fast, catches normal edits.
|
|
1261
|
+
- ``content``: full md5 for files <``content_fingerprint_size_threshold``;
|
|
1262
|
+
large files are sampled at head/middle/tail with
|
|
1263
|
+
``content_fingerprint_sample_bytes`` per region. Robust against
|
|
1264
|
+
``touch``/CI-replay scenarios at the cost of one read per file.
|
|
1265
|
+
|
|
1266
|
+
Returns empty string if the file does not exist.
|
|
1267
|
+
"""
|
|
1268
|
+
from .config import INDEX_SCHEMA_VERSION, get_config
|
|
1269
|
+
cfg = get_config()
|
|
1270
|
+
if mode is None:
|
|
1271
|
+
mode = cfg.fingerprint_mode
|
|
1272
|
+
|
|
1273
|
+
try:
|
|
1274
|
+
st = os.stat(fp)
|
|
1275
|
+
except (FileNotFoundError, OSError):
|
|
1276
|
+
return ""
|
|
1277
|
+
|
|
1278
|
+
if mode == "content":
|
|
1279
|
+
threshold = cfg.content_fingerprint_size_threshold
|
|
1280
|
+
sample = cfg.content_fingerprint_sample_bytes
|
|
1281
|
+
h = hashlib.md5()
|
|
1282
|
+
try:
|
|
1283
|
+
with open(fp, "rb") as f:
|
|
1284
|
+
if st.st_size <= threshold:
|
|
1285
|
+
h.update(f.read())
|
|
1286
|
+
else:
|
|
1287
|
+
h.update(f.read(sample))
|
|
1288
|
+
if st.st_size > 2 * sample:
|
|
1289
|
+
f.seek(st.st_size // 2)
|
|
1290
|
+
h.update(f.read(sample))
|
|
1291
|
+
f.seek(max(0, st.st_size - sample))
|
|
1292
|
+
h.update(f.read(sample))
|
|
1293
|
+
except OSError:
|
|
1294
|
+
return ""
|
|
1295
|
+
payload = f"{st.st_size}:{h.hexdigest()}"
|
|
1296
|
+
else:
|
|
1297
|
+
payload = f"{st.st_mtime_ns}:{st.st_size}"
|
|
1298
|
+
|
|
1299
|
+
return f"v{INDEX_SCHEMA_VERSION}:{mode}:{payload}"
|
|
1300
|
+
|
|
1301
|
+
|
|
1302
|
+
def file_hash_with_salts(fp: str, mode: Optional[str] = None) -> str:
|
|
1303
|
+
"""``file_hash`` + 解析器格式盐(插在版本号后,保持末尾 size 可解析)。
|
|
1304
|
+
|
|
1305
|
+
PST 解析器输出格式变化(ADR-0005)时 bump PST_PARSER_FINGERPRINT_SALT,
|
|
1306
|
+
旧 PST 索引自动重建,不像 INDEX_SCHEMA_VERSION 那样连累全库。
|
|
1307
|
+
增量比较与移动检测都必须用本函数(口径一致才能匹配)。
|
|
1308
|
+
|
|
1309
|
+
图像文件(jpg/jpeg/png/webp)改用解码像素指纹:写回 EXIF/XMP 元数据不改像素,
|
|
1310
|
+
故不触发「写回 → hash 变 → 增量重解析」死循环(ADR-0009 / 工单 03)。解码失败
|
|
1311
|
+
回退到 stat/content。注:口径切换使旧图像 hash 失效 → 首迁移重索引(已知代价)。
|
|
1312
|
+
"""
|
|
1313
|
+
ext = os.path.splitext(fp)[1].lower()
|
|
1314
|
+
if ext in (".jpg", ".jpeg", ".png", ".webp"): # 与 image_metadata.INTERPRETED_IMAGE_EXTS 一致
|
|
1315
|
+
try:
|
|
1316
|
+
from .config import INDEX_SCHEMA_VERSION
|
|
1317
|
+
from .parsers.image_metadata import content_fingerprint
|
|
1318
|
+
return f"v{INDEX_SCHEMA_VERSION}:image:{content_fingerprint(fp)}"
|
|
1319
|
+
except Exception:
|
|
1320
|
+
pass # 解码失败回退到 stat/content
|
|
1321
|
+
h = file_hash(fp, mode)
|
|
1322
|
+
if h and fp.lower().endswith(".pst"):
|
|
1323
|
+
ver, sep, rest = h.partition(":")
|
|
1324
|
+
return f"{ver}{sep}{PST_PARSER_FINGERPRINT_SALT.strip(':')}:{rest}"
|
|
1325
|
+
return h
|
|
1326
|
+
|
|
1327
|
+
|
|
1328
|
+
def _extract_size_from_fingerprint(fp_hash: str) -> Optional[str]:
|
|
1329
|
+
"""Extract the file size portion from a fingerprint string.
|
|
1330
|
+
|
|
1331
|
+
Supports both stat (``v{VER}:stat:{mtime}:{size}``) and content
|
|
1332
|
+
(``v{VER}:content:{size}:{md5}``) formats.
|
|
1333
|
+
"""
|
|
1334
|
+
parts = fp_hash.split(":")
|
|
1335
|
+
if len(parts) < 3:
|
|
1336
|
+
return None
|
|
1337
|
+
candidate = parts[-1]
|
|
1338
|
+
return candidate if candidate.isdigit() else None
|
|
1339
|
+
|
|
1340
|
+
|
|
1341
|
+
def _detect_and_apply_moves(
|
|
1342
|
+
expanded: list[str],
|
|
1343
|
+
all_meta: dict[str, str],
|
|
1344
|
+
fts,
|
|
1345
|
+
fp_to_doc_id: dict[str, str],
|
|
1346
|
+
file_hash_fn,
|
|
1347
|
+
) -> list[tuple[str, str]]:
|
|
1348
|
+
"""Detect file moves/renames and apply them in-place via ``fts.rename_document``.
|
|
1349
|
+
|
|
1350
|
+
Two-pass strategy:
|
|
1351
|
+
|
|
1352
|
+
1. **Fingerprint match** (exact): compute the file fingerprint and look it up
|
|
1353
|
+
in a reverse index built from stored metadata. This works perfectly when
|
|
1354
|
+
the fingerprint is content-based (MD5), and also catches the rare case
|
|
1355
|
+
where ``stat`` mtime is preserved after a move.
|
|
1356
|
+
|
|
1357
|
+
2. **Size heuristic** (fallback): for stat-mode fingerprints where mtime
|
|
1358
|
+
changed on move, match by ``(basename, file_size)``. This is reliable
|
|
1359
|
+
because: same basename + same byte size + old path gone = move.
|
|
1360
|
+
|
|
1361
|
+
Mutates ``all_meta`` in-place by popping old entries and inserting new ones.
|
|
1362
|
+
Returns a list of ``(old_path, new_path)`` tuples for logging.
|
|
1363
|
+
"""
|
|
1364
|
+
from .config import INDEX_SCHEMA_VERSION
|
|
1365
|
+
|
|
1366
|
+
_current_prefix = f"v{INDEX_SCHEMA_VERSION}:"
|
|
1367
|
+
|
|
1368
|
+
# --- Build reverse index: fingerprint → stored_path ---
|
|
1369
|
+
hash_to_old_path: dict[str, str] = {}
|
|
1370
|
+
for sp, fh in all_meta.items():
|
|
1371
|
+
if fh.startswith(_current_prefix):
|
|
1372
|
+
hash_to_old_path.setdefault(fh, sp)
|
|
1373
|
+
|
|
1374
|
+
moved: list[tuple[str, str]] = []
|
|
1375
|
+
expanded_abs = {os.path.abspath(p) for p in expanded}
|
|
1376
|
+
matched_new = set() # abs_fp of new files already claimed
|
|
1377
|
+
|
|
1378
|
+
# --- Pass 1: fingerprint-based matching ---
|
|
1379
|
+
if hash_to_old_path:
|
|
1380
|
+
for fp in expanded:
|
|
1381
|
+
abs_fp = os.path.abspath(fp)
|
|
1382
|
+
if abs_fp in all_meta:
|
|
1383
|
+
continue
|
|
1384
|
+
fh = file_hash_fn(abs_fp)
|
|
1385
|
+
if not fh:
|
|
1386
|
+
continue
|
|
1387
|
+
old_path = hash_to_old_path.get(fh)
|
|
1388
|
+
if old_path and old_path != abs_fp and not os.path.isfile(old_path):
|
|
1389
|
+
moved_doc_id = fts.get_doc_id_by_source_path(old_path)
|
|
1390
|
+
if moved_doc_id:
|
|
1391
|
+
new_doc_id = fp_to_doc_id.get(fp, os.path.splitext(os.path.basename(abs_fp))[0])
|
|
1392
|
+
if fts.rename_document(moved_doc_id, new_doc_id, new_doc_id, abs_fp):
|
|
1393
|
+
fts.set_index_meta(abs_fp, fh)
|
|
1394
|
+
all_meta.pop(old_path, None)
|
|
1395
|
+
all_meta[abs_fp] = fh
|
|
1396
|
+
hash_to_old_path[fh] = abs_fp
|
|
1397
|
+
matched_new.add(abs_fp)
|
|
1398
|
+
moved.append((old_path, abs_fp))
|
|
1399
|
+
logger.info(
|
|
1400
|
+
"Detected moved file: %s -> %s (doc_id %s -> %s)",
|
|
1401
|
+
old_path, abs_fp, moved_doc_id, new_doc_id,
|
|
1402
|
+
)
|
|
1403
|
+
else:
|
|
1404
|
+
logger.debug(
|
|
1405
|
+
"Move-detection skipped for %s -> %s "
|
|
1406
|
+
"(doc_id collision); will full re-index",
|
|
1407
|
+
old_path, abs_fp,
|
|
1408
|
+
)
|
|
1409
|
+
|
|
1410
|
+
# --- Pass 2: size-based heuristic for stat-mode moves ---
|
|
1411
|
+
# Collect orphan candidates: indexed paths whose files disappeared from disk
|
|
1412
|
+
# and were not already remapped by Pass 1.
|
|
1413
|
+
orphan_by_key: dict[str, list[tuple[str, str]]] = {} # "basename:size" → [(path, hash)]
|
|
1414
|
+
for sp, sh in list(all_meta.items()):
|
|
1415
|
+
if sp in expanded_abs or os.path.isfile(sp):
|
|
1416
|
+
continue
|
|
1417
|
+
size_str = _extract_size_from_fingerprint(sh)
|
|
1418
|
+
if not size_str:
|
|
1419
|
+
continue
|
|
1420
|
+
bn = os.path.basename(sp)
|
|
1421
|
+
orphan_by_key.setdefault(f"{bn}:{size_str}", []).append((sp, sh))
|
|
1422
|
+
|
|
1423
|
+
if orphan_by_key:
|
|
1424
|
+
for fp in expanded:
|
|
1425
|
+
abs_fp = os.path.abspath(fp)
|
|
1426
|
+
if abs_fp in all_meta or abs_fp in matched_new:
|
|
1427
|
+
continue
|
|
1428
|
+
try:
|
|
1429
|
+
st = os.stat(abs_fp)
|
|
1430
|
+
except OSError:
|
|
1431
|
+
continue
|
|
1432
|
+
bn = os.path.basename(abs_fp)
|
|
1433
|
+
key = f"{bn}:{st.st_size}"
|
|
1434
|
+
candidates = orphan_by_key.get(key, [])
|
|
1435
|
+
for old_path, stored_hash in candidates:
|
|
1436
|
+
if old_path not in all_meta:
|
|
1437
|
+
continue # already claimed by another new file
|
|
1438
|
+
moved_doc_id = fts.get_doc_id_by_source_path(old_path)
|
|
1439
|
+
if not moved_doc_id:
|
|
1440
|
+
continue
|
|
1441
|
+
new_doc_id = fp_to_doc_id.get(fp, os.path.splitext(bn)[0])
|
|
1442
|
+
if fts.rename_document(moved_doc_id, new_doc_id, new_doc_id, abs_fp):
|
|
1443
|
+
new_fh = file_hash_fn(abs_fp)
|
|
1444
|
+
fts.set_index_meta(abs_fp, new_fh)
|
|
1445
|
+
all_meta.pop(old_path, None)
|
|
1446
|
+
all_meta[abs_fp] = new_fh
|
|
1447
|
+
moved.append((old_path, abs_fp))
|
|
1448
|
+
logger.info(
|
|
1449
|
+
"Detected moved/renamed file (size heuristic): %s -> %s",
|
|
1450
|
+
old_path, abs_fp,
|
|
1451
|
+
)
|
|
1452
|
+
break
|
|
1453
|
+
|
|
1454
|
+
return moved
|
|
1455
|
+
|
|
1456
|
+
|
|
1457
|
+
async def build_index(
|
|
1458
|
+
paths: list[str],
|
|
1459
|
+
output_dir: str = "./indexes",
|
|
1460
|
+
*,
|
|
1461
|
+
db_path: str = "",
|
|
1462
|
+
if_add_node_summary: Optional[bool] = None,
|
|
1463
|
+
if_add_doc_description: Optional[bool] = None,
|
|
1464
|
+
if_add_node_text: Optional[bool] = None,
|
|
1465
|
+
if_add_node_id: Optional[bool] = None,
|
|
1466
|
+
max_concurrency: Optional[int] = None,
|
|
1467
|
+
force: bool = False,
|
|
1468
|
+
ignore_dirs: frozenset[str] = DEFAULT_IGNORE_DIRS,
|
|
1469
|
+
respect_gitignore: bool = True,
|
|
1470
|
+
max_files: int = MAX_DIR_FILES,
|
|
1471
|
+
prune: Optional[bool] = None,
|
|
1472
|
+
progress_callback: Optional[callable] = None,
|
|
1473
|
+
sub_progress_callback: Optional[callable] = None,
|
|
1474
|
+
**kwargs,
|
|
1475
|
+
) -> list[Document]:
|
|
1476
|
+
"""
|
|
1477
|
+
Build tree indexes for multiple files. Returns list of Document objects ready for search.
|
|
1478
|
+
|
|
1479
|
+
All parameters default to ``get_config()`` values when not explicitly set.
|
|
1480
|
+
|
|
1481
|
+
Args:
|
|
1482
|
+
paths: list of file paths, glob patterns, or directories
|
|
1483
|
+
(e.g. ``["docs/*.md", "paper.txt", "src/"]``)
|
|
1484
|
+
output_dir: directory for the database file (used to derive db_path if db_path is empty)
|
|
1485
|
+
db_path: path to the SQLite database file. If empty, defaults to ``{output_dir}/index.db``.
|
|
1486
|
+
max_concurrency: max concurrent indexing tasks
|
|
1487
|
+
force: force re-index even if file unchanged (default: False)
|
|
1488
|
+
ignore_dirs: directory names to skip during recursive walk
|
|
1489
|
+
respect_gitignore: honour ``.gitignore`` files when walking directories
|
|
1490
|
+
max_files: safety cap on files discovered per directory walk
|
|
1491
|
+
prune: if True, delete documents whose source files are no longer
|
|
1492
|
+
reachable through ``paths`` (orphan cleanup). When ``None``,
|
|
1493
|
+
defaults to True iff at least one entry of ``paths`` is a directory
|
|
1494
|
+
or recursive glob (full-scope reindex), False otherwise.
|
|
1495
|
+
**kwargs: passed through to individual parsers
|
|
1496
|
+
|
|
1497
|
+
Returns:
|
|
1498
|
+
list of Document objects (directly usable with search())
|
|
1499
|
+
"""
|
|
1500
|
+
from .config import get_config
|
|
1501
|
+
from .fts import FTS5Index
|
|
1502
|
+
cfg = get_config()
|
|
1503
|
+
|
|
1504
|
+
# Resolve defaults from config
|
|
1505
|
+
if if_add_node_summary is None:
|
|
1506
|
+
if_add_node_summary = cfg.if_add_node_summary
|
|
1507
|
+
if if_add_doc_description is None:
|
|
1508
|
+
if_add_doc_description = cfg.if_add_doc_description
|
|
1509
|
+
if if_add_node_text is None:
|
|
1510
|
+
if_add_node_text = cfg.if_add_node_text
|
|
1511
|
+
if if_add_node_id is None:
|
|
1512
|
+
if_add_node_id = cfg.if_add_node_id
|
|
1513
|
+
if max_concurrency is None:
|
|
1514
|
+
max_concurrency = cfg.max_concurrency
|
|
1515
|
+
|
|
1516
|
+
# Resolve db_path
|
|
1517
|
+
if not db_path:
|
|
1518
|
+
db_path = os.path.join(output_dir, "index.db")
|
|
1519
|
+
os.makedirs(os.path.dirname(os.path.abspath(db_path)), exist_ok=True)
|
|
1520
|
+
|
|
1521
|
+
# 图片落盘存储(与 index.db 同目录的 images/ 子目录)
|
|
1522
|
+
from .parsers.image_store import ImageStore
|
|
1523
|
+
images_root = Path(db_path).parent / "images"
|
|
1524
|
+
image_store = ImageStore(images_root)
|
|
1525
|
+
# PST 附件落盘存储(与 index.db 同目录的 pst_attachments/ 子目录,ADR-0005)
|
|
1526
|
+
from .parsers.pst_attachment_store import PstAttachmentStore
|
|
1527
|
+
pst_att_store = PstAttachmentStore(Path(db_path).parent / "pst_attachments")
|
|
1528
|
+
if force:
|
|
1529
|
+
image_store.purge_all()
|
|
1530
|
+
pst_att_store.purge_all()
|
|
1531
|
+
|
|
1532
|
+
# 计算 rel_path 用的 base 目录(paths 里第一个目录,即 search_path)
|
|
1533
|
+
base_dir = ""
|
|
1534
|
+
for p in paths:
|
|
1535
|
+
if os.path.isdir(p):
|
|
1536
|
+
base_dir = os.path.abspath(p)
|
|
1537
|
+
break
|
|
1538
|
+
|
|
1539
|
+
# Expand globs, files, and directories via resolve_paths
|
|
1540
|
+
expanded = resolve_paths(
|
|
1541
|
+
paths,
|
|
1542
|
+
ignore_dirs=ignore_dirs,
|
|
1543
|
+
respect_gitignore=respect_gitignore,
|
|
1544
|
+
max_files=max_files,
|
|
1545
|
+
)
|
|
1546
|
+
total_files = len(expanded)
|
|
1547
|
+
processed_counter = [0]
|
|
1548
|
+
_progress_lock = asyncio.Lock()
|
|
1549
|
+
if not expanded:
|
|
1550
|
+
raise FileNotFoundError(f"No files found for patterns: {paths}")
|
|
1551
|
+
|
|
1552
|
+
# Pre-compute deterministic doc_ids: basename + path hash (always, for determinism)
|
|
1553
|
+
_base_dir = os.path.dirname(os.path.abspath(db_path)) if db_path else os.getcwd()
|
|
1554
|
+
|
|
1555
|
+
def _doc_id_for(fp):
|
|
1556
|
+
base = os.path.splitext(os.path.basename(fp))[0]
|
|
1557
|
+
rel = os.path.relpath(os.path.abspath(fp), _base_dir)
|
|
1558
|
+
h = hashlib.md5(rel.encode()).hexdigest()[:8]
|
|
1559
|
+
return f"{base}_{h}"
|
|
1560
|
+
|
|
1561
|
+
_fp_to_doc_id = {fp: _doc_id_for(fp) for fp in expanded}
|
|
1562
|
+
|
|
1563
|
+
# Resolve prune policy: full-scope walks default to pruning orphans.
|
|
1564
|
+
explicit_prune = prune is not None
|
|
1565
|
+
if prune is None:
|
|
1566
|
+
full_scope = any(
|
|
1567
|
+
os.path.isdir(p) or "**" in p
|
|
1568
|
+
for p in paths
|
|
1569
|
+
)
|
|
1570
|
+
prune = full_scope and cfg.prune_orphans_on_directory
|
|
1571
|
+
|
|
1572
|
+
# Open DB with advisory file lock so concurrent build_index() calls on the
|
|
1573
|
+
# same DB serialize cleanly instead of racing on writes.
|
|
1574
|
+
fts = FTS5Index(db_path=db_path, tokenize_log_path=os.path.join(os.path.dirname(db_path), "tokenize.log"))
|
|
1575
|
+
_lock_handle = _acquire_index_lock(db_path)
|
|
1576
|
+
|
|
1577
|
+
# 如果锁获取失败(_NullLock),说明有其他进程在索引,跳过本次
|
|
1578
|
+
if isinstance(_lock_handle, _NullLock):
|
|
1579
|
+
logger.warning("Index is locked by another process, skipping this indexing run")
|
|
1580
|
+
return []
|
|
1581
|
+
|
|
1582
|
+
# Incremental indexing: batch check file hashes via DB
|
|
1583
|
+
to_index = []
|
|
1584
|
+
skipped = []
|
|
1585
|
+
file_hashes = {}
|
|
1586
|
+
pruned_paths: list[str] = []
|
|
1587
|
+
|
|
1588
|
+
if force:
|
|
1589
|
+
# Full rebuild: clear all failed file records
|
|
1590
|
+
fts.clear_all_failed_files()
|
|
1591
|
+
# 视觉解析队列一并清空(重建过程会重新登记)
|
|
1592
|
+
fts.vision_clear()
|
|
1593
|
+
|
|
1594
|
+
if not force:
|
|
1595
|
+
# Batch fetch all stored hashes in one query (instead of N queries)
|
|
1596
|
+
all_meta = fts.get_all_index_meta()
|
|
1597
|
+
elif cfg.allowed_source_types:
|
|
1598
|
+
# force=True with source_type filter: still need all_meta for orphan cleanup
|
|
1599
|
+
all_meta = fts.get_all_index_meta()
|
|
1600
|
+
else:
|
|
1601
|
+
all_meta = {}
|
|
1602
|
+
|
|
1603
|
+
# Pre-pass: detect moves/renames and remap source_path BEFORE pruning,
|
|
1604
|
+
# so the orphan cleanup below doesn't drop a doc whose file just moved.
|
|
1605
|
+
if not force:
|
|
1606
|
+
_detect_and_apply_moves(
|
|
1607
|
+
expanded, all_meta, fts, _fp_to_doc_id, file_hash_with_salts
|
|
1608
|
+
)
|
|
1609
|
+
|
|
1610
|
+
# Orphan cleanup: docs in DB whose source_path is not in the new scope.
|
|
1611
|
+
# Implicit prune (defaulted from a directory walk): only drop docs whose
|
|
1612
|
+
# files are also gone from disk — preserves unrelated files for partial
|
|
1613
|
+
# reindexes.
|
|
1614
|
+
# Explicit `prune=True`: reduce the index to exactly `paths`.
|
|
1615
|
+
if prune and all_meta:
|
|
1616
|
+
expanded_abs = {os.path.abspath(p) for p in expanded}
|
|
1617
|
+
# Pre-compute allowed extensions from source_type filter (if any)
|
|
1618
|
+
source_type_exts = None
|
|
1619
|
+
if cfg.allowed_source_types:
|
|
1620
|
+
from .pathutil import get_allowed_extensions_for_source_types
|
|
1621
|
+
source_type_exts = get_allowed_extensions_for_source_types(cfg.allowed_source_types)
|
|
1622
|
+
prune_doc_ids: list[str] = []
|
|
1623
|
+
for stored_path in list(all_meta.keys()):
|
|
1624
|
+
if stored_path in expanded_abs:
|
|
1625
|
+
continue
|
|
1626
|
+
if not explicit_prune and os.path.isfile(stored_path):
|
|
1627
|
+
# File exists on disk but was filtered out — only keep it if
|
|
1628
|
+
# it would have been included WITHOUT the source_type filter.
|
|
1629
|
+
if source_type_exts is not None:
|
|
1630
|
+
ext = os.path.splitext(stored_path)[1].lower()
|
|
1631
|
+
if ext not in source_type_exts:
|
|
1632
|
+
pass # excluded by source_type → prune
|
|
1633
|
+
else:
|
|
1634
|
+
continue
|
|
1635
|
+
else:
|
|
1636
|
+
continue
|
|
1637
|
+
doc_id = fts.get_doc_id_by_source_path(stored_path)
|
|
1638
|
+
# 多文档来源(PST 派生 "<file>#<entry_id>"):级联删除派生文档
|
|
1639
|
+
derived_ids = fts.get_doc_ids_by_source_prefix(stored_path + "#")
|
|
1640
|
+
ids = ([doc_id] if doc_id else []) + derived_ids
|
|
1641
|
+
if ids:
|
|
1642
|
+
prune_doc_ids.extend(ids)
|
|
1643
|
+
pruned_paths.append(stored_path)
|
|
1644
|
+
all_meta.pop(stored_path, None)
|
|
1645
|
+
if prune_doc_ids:
|
|
1646
|
+
fts.delete_documents(prune_doc_ids)
|
|
1647
|
+
logger.info("Pruned %d orphan document(s) from index", len(prune_doc_ids))
|
|
1648
|
+
# 被 prune 的源文件同步移出视觉解析队列
|
|
1649
|
+
for stored_path in pruned_paths:
|
|
1650
|
+
fts.vision_remove(stored_path)
|
|
1651
|
+
|
|
1652
|
+
# Clean up shadow MD files for pruned binary sources
|
|
1653
|
+
from .parsers.registry import is_binary_extension
|
|
1654
|
+
for pruned_path in pruned_paths:
|
|
1655
|
+
ext = os.path.splitext(pruned_path)[1].lower()
|
|
1656
|
+
if is_binary_extension(ext):
|
|
1657
|
+
md_path = shadow_md_path(pruned_path)
|
|
1658
|
+
if os.path.exists(md_path):
|
|
1659
|
+
try:
|
|
1660
|
+
os.remove(md_path)
|
|
1661
|
+
logger.debug("Removed orphan shadow MD: %s", md_path)
|
|
1662
|
+
except OSError as e:
|
|
1663
|
+
logger.debug("Failed to remove shadow MD %s: %s", md_path, e)
|
|
1664
|
+
# 清理被删文档的图片
|
|
1665
|
+
if base_dir:
|
|
1666
|
+
pruned_rel = os.path.relpath(pruned_path, base_dir).replace(os.sep, "/")
|
|
1667
|
+
image_store.purge_doc(pruned_rel)
|
|
1668
|
+
# 清理被删 PST 的落盘附件(ADR-0005)
|
|
1669
|
+
if ext == ".pst" and base_dir:
|
|
1670
|
+
pruned_rel = os.path.relpath(pruned_path, base_dir).replace(os.sep, "/")
|
|
1671
|
+
pst_att_store.purge_doc(pruned_rel)
|
|
1672
|
+
|
|
1673
|
+
for fp in expanded:
|
|
1674
|
+
abs_fp = os.path.abspath(fp)
|
|
1675
|
+
# 指纹含解析器格式盐(PST:ADR-0005 输出格式变化时自动重建)
|
|
1676
|
+
fh = file_hash_with_salts(abs_fp)
|
|
1677
|
+
if not fh:
|
|
1678
|
+
# File disappeared after glob expansion (e.g. broken symlink)
|
|
1679
|
+
logger.debug("Skipping missing file: %s", abs_fp)
|
|
1680
|
+
continue
|
|
1681
|
+
file_hashes[abs_fp] = fh
|
|
1682
|
+
if not force:
|
|
1683
|
+
stored_hash = all_meta.get(abs_fp)
|
|
1684
|
+
if stored_hash == fh:
|
|
1685
|
+
# source_path lookup catches both same-name and moved files.
|
|
1686
|
+
# Multi-doc sources (e.g. PST: docs keyed "<file>#<entry_id>")
|
|
1687
|
+
# have no doc at the exact path — check the derived prefix.
|
|
1688
|
+
if fts.get_doc_id_by_source_path(abs_fp) is not None \
|
|
1689
|
+
or fts.has_docs_with_source_prefix(abs_fp + "#"):
|
|
1690
|
+
skipped.append(fp)
|
|
1691
|
+
processed_counter[0] += 1
|
|
1692
|
+
if progress_callback:
|
|
1693
|
+
progress_callback(fp, processed_counter[0], total_files)
|
|
1694
|
+
continue
|
|
1695
|
+
to_index.append(fp)
|
|
1696
|
+
|
|
1697
|
+
if skipped:
|
|
1698
|
+
logger.info("Skipped %d unchanged file(s)", len(skipped))
|
|
1699
|
+
|
|
1700
|
+
# Filter out files that have exceeded the consecutive failure threshold
|
|
1701
|
+
excluded_count = 0
|
|
1702
|
+
if not force and to_index:
|
|
1703
|
+
failed_records = fts.get_all_failed_files() # {path: (fail_count, file_hash, last_error)}
|
|
1704
|
+
max_fail_count = cfg.max_index_fail_count
|
|
1705
|
+
excluded = []
|
|
1706
|
+
remaining = []
|
|
1707
|
+
for fp in to_index:
|
|
1708
|
+
abs_fp = os.path.abspath(fp)
|
|
1709
|
+
record = failed_records.get(abs_fp)
|
|
1710
|
+
if record and record[0] >= max_fail_count:
|
|
1711
|
+
# Check if file content has changed since last failure
|
|
1712
|
+
current_hash = file_hashes.get(abs_fp, "")
|
|
1713
|
+
if record[1] and record[1] == current_hash:
|
|
1714
|
+
excluded.append(fp)
|
|
1715
|
+
continue
|
|
1716
|
+
remaining.append(fp)
|
|
1717
|
+
to_index = remaining
|
|
1718
|
+
if excluded:
|
|
1719
|
+
excluded_count = len(excluded)
|
|
1720
|
+
logger.warning("Skipped %d file(s) with %d+ consecutive failures", excluded_count, max_fail_count)
|
|
1721
|
+
for fp in excluded:
|
|
1722
|
+
abs_fp = os.path.abspath(fp)
|
|
1723
|
+
record = failed_records.get(abs_fp)
|
|
1724
|
+
last_error = record[2] if record else ""
|
|
1725
|
+
if last_error:
|
|
1726
|
+
logger.warning(" Skipped %s: %s", fp, last_error)
|
|
1727
|
+
else:
|
|
1728
|
+
logger.warning(" Skipped %s: unknown error", fp)
|
|
1729
|
+
processed_counter[0] += 1
|
|
1730
|
+
if progress_callback:
|
|
1731
|
+
progress_callback(fp, processed_counter[0], total_files)
|
|
1732
|
+
|
|
1733
|
+
logger.info("Building indexes for %d file(s) (concurrency=%d)...", len(to_index), max_concurrency)
|
|
1734
|
+
|
|
1735
|
+
build_start = time.monotonic()
|
|
1736
|
+
semaphore = asyncio.Semaphore(max_concurrency)
|
|
1737
|
+
# PST 专用并发上限:PST 解析启 sidecar 子进程 + 加载 GB 级文件 + 全量附件提取,
|
|
1738
|
+
# 多个并发会耗尽内存/CPU/IO 触发 sidecar 崩溃(exit 0xC000013A)。单独低并发限流,
|
|
1739
|
+
# 同时缓解"极度缓慢"和"崩溃"两个根因。
|
|
1740
|
+
pst_semaphore = asyncio.Semaphore(getattr(cfg, "max_pst_concurrency", 1))
|
|
1741
|
+
# Collect per-file timing and source_type for stats
|
|
1742
|
+
_file_timings: dict[str, tuple[str, float]] = {} # fp -> (source_type, elapsed_s)
|
|
1743
|
+
_failed_paths: list[str] = []
|
|
1744
|
+
|
|
1745
|
+
# Progress bar for parsing stage
|
|
1746
|
+
_parse_bar = tqdm(total=len(to_index), desc="Parsing", unit="file",
|
|
1747
|
+
dynamic_ncols=True, disable=not to_index)
|
|
1748
|
+
|
|
1749
|
+
async def _index_one(fp: str) -> dict | None:
|
|
1750
|
+
ext = os.path.splitext(fp)[1].lower()
|
|
1751
|
+
# PST 重量级(sidecar 子进程 + GB 级文件 + 全量附件提取):先过 pst_semaphore
|
|
1752
|
+
# 限流(等待时不占通用 semaphore 槽),再过通用并发信号量。
|
|
1753
|
+
pst_gate = pst_semaphore if ext == ".pst" else nullcontext()
|
|
1754
|
+
async with pst_gate, semaphore:
|
|
1755
|
+
fname = os.path.basename(fp)
|
|
1756
|
+
_parse_bar.set_postfix_str(fname, refresh=False)
|
|
1757
|
+
t0 = time.monotonic()
|
|
1758
|
+
try:
|
|
1759
|
+
rel_path = ""
|
|
1760
|
+
if base_dir:
|
|
1761
|
+
rel_path = os.path.relpath(fp, base_dir).replace(os.sep, "/")
|
|
1762
|
+
# 该文件重抽前先清掉它的旧图(幂等;force 模式已 purge_all,此处为空操作)
|
|
1763
|
+
if rel_path:
|
|
1764
|
+
image_store.purge_doc(rel_path)
|
|
1765
|
+
# PST 重索引:清掉旧落盘附件(幂等,重新提取)
|
|
1766
|
+
if ext == ".pst":
|
|
1767
|
+
pst_att_store.purge_doc(rel_path)
|
|
1768
|
+
common = dict(
|
|
1769
|
+
if_add_node_summary=if_add_node_summary,
|
|
1770
|
+
if_add_doc_description=if_add_doc_description,
|
|
1771
|
+
if_add_node_text=if_add_node_text,
|
|
1772
|
+
if_add_node_id=if_add_node_id,
|
|
1773
|
+
image_store=image_store,
|
|
1774
|
+
rel_path=rel_path,
|
|
1775
|
+
pst_attachment_store=pst_att_store,
|
|
1776
|
+
sub_progress_callback=(
|
|
1777
|
+
(lambda n: sub_progress_callback(fp, n))
|
|
1778
|
+
if sub_progress_callback else None
|
|
1779
|
+
),
|
|
1780
|
+
**kwargs,
|
|
1781
|
+
)
|
|
1782
|
+
|
|
1783
|
+
# Use ParserRegistry for dispatch (built-in parsers auto-registered)
|
|
1784
|
+
from .parsers import get_parser, SOURCE_TYPE_MAP
|
|
1785
|
+
parser_fn = get_parser(ext)
|
|
1786
|
+
if parser_fn is not None:
|
|
1787
|
+
result = await parser_fn(fp, **common)
|
|
1788
|
+
else:
|
|
1789
|
+
# Unknown extension: fall back to text_to_tree
|
|
1790
|
+
result = await text_to_tree(text_path=fp, **common)
|
|
1791
|
+
|
|
1792
|
+
# Tag source_type for search routing
|
|
1793
|
+
source_type = SOURCE_TYPE_MAP.get(ext, "text")
|
|
1794
|
+
result["source_type"] = source_type
|
|
1795
|
+
|
|
1796
|
+
# 图像文件:占位节点已进索引,登记到视觉解析队列(后台 worker 消费)
|
|
1797
|
+
if source_type == "image" and result.get("vision_pending"):
|
|
1798
|
+
fts.vision_enqueue(os.path.abspath(fp), rel_path)
|
|
1799
|
+
|
|
1800
|
+
# Generate shadow MD for binary files (concurrent with parsing)
|
|
1801
|
+
if cfg.enable_shadow_md:
|
|
1802
|
+
from .parsers.registry import is_binary_extension
|
|
1803
|
+
if is_binary_extension(ext):
|
|
1804
|
+
try:
|
|
1805
|
+
_generate_shadow_md(os.path.abspath(fp))
|
|
1806
|
+
except Exception as e:
|
|
1807
|
+
logger.debug("Shadow MD generation failed for %s: %s", fp, e)
|
|
1808
|
+
|
|
1809
|
+
_file_timings[fp] = (source_type, time.monotonic() - t0)
|
|
1810
|
+
# Call progress callback if provided
|
|
1811
|
+
async with _progress_lock:
|
|
1812
|
+
processed_counter[0] += 1
|
|
1813
|
+
if progress_callback:
|
|
1814
|
+
progress_callback(fp, processed_counter[0], total_files)
|
|
1815
|
+
return result
|
|
1816
|
+
except Exception as e:
|
|
1817
|
+
logger.warning("Failed to index %s: %s", fp, e)
|
|
1818
|
+
_failed_paths.append(fp)
|
|
1819
|
+
abs_fp = os.path.abspath(fp)
|
|
1820
|
+
fts.upsert_failed_file(abs_fp, str(e), file_hashes.get(abs_fp, ""))
|
|
1821
|
+
_file_timings[fp] = ("(failed)", time.monotonic() - t0)
|
|
1822
|
+
async with _progress_lock:
|
|
1823
|
+
processed_counter[0] += 1
|
|
1824
|
+
if progress_callback:
|
|
1825
|
+
progress_callback(fp, processed_counter[0], total_files)
|
|
1826
|
+
return None
|
|
1827
|
+
finally:
|
|
1828
|
+
_parse_bar.update()
|
|
1829
|
+
|
|
1830
|
+
raw_results = await asyncio.gather(*(_index_one(fp) for fp in to_index))
|
|
1831
|
+
_parse_bar.close()
|
|
1832
|
+
|
|
1833
|
+
# Save results to DB and collect Document objects
|
|
1834
|
+
result_map = {fp: r for fp, r in zip(to_index, raw_results) if r is not None}
|
|
1835
|
+
documents = []
|
|
1836
|
+
|
|
1837
|
+
# Batch load all skipped documents in one query (instead of N individual loads)
|
|
1838
|
+
# Key by source_path (unique and stable) instead of doc_id (may change)
|
|
1839
|
+
if skipped:
|
|
1840
|
+
all_docs_from_db = {
|
|
1841
|
+
d.metadata.get("source_path", ""): d
|
|
1842
|
+
for d in fts.load_all_documents()
|
|
1843
|
+
if d.metadata.get("source_path")
|
|
1844
|
+
}
|
|
1845
|
+
else:
|
|
1846
|
+
all_docs_from_db = {}
|
|
1847
|
+
|
|
1848
|
+
# Progress bar for Indexing stage (batch commit every N files to reduce fsync)
|
|
1849
|
+
_COMMIT_BATCH = 500
|
|
1850
|
+
_has_work = bool(result_map)
|
|
1851
|
+
_save_bar = tqdm(total=len(expanded), desc="Indexing", unit="file",
|
|
1852
|
+
dynamic_ncols=True, disable=not _has_work)
|
|
1853
|
+
_pending_commits = 0
|
|
1854
|
+
# Aggregate node-level diff stats across all reindexed docs.
|
|
1855
|
+
diff_totals = {"added": 0, "changed": 0, "removed": 0, "kept": 0}
|
|
1856
|
+
optimize_threshold = cfg.auto_optimize_threshold
|
|
1857
|
+
docs_since_optimize = 0
|
|
1858
|
+
for fp in expanded:
|
|
1859
|
+
name = _fp_to_doc_id[fp]
|
|
1860
|
+
if fp in result_map:
|
|
1861
|
+
_save_bar.set_postfix_str(os.path.basename(fp), refresh=False)
|
|
1862
|
+
result = result_map[fp]
|
|
1863
|
+
abs_fp = os.path.abspath(fp)
|
|
1864
|
+
file_h = file_hashes.get(abs_fp, "")
|
|
1865
|
+
|
|
1866
|
+
if result.get("multi_docs") is not None:
|
|
1867
|
+
# 多文档来源(PST 等):一个文件 → N 个派生文档。
|
|
1868
|
+
# 派生 doc_id = <file_doc_id>__<entry>;source_path = <file>#<entry>。
|
|
1869
|
+
# 已被移除的派生文档(邮件删除)按差集清除;其余走节点级增量。
|
|
1870
|
+
trees = result["multi_docs"]
|
|
1871
|
+
new_docs = []
|
|
1872
|
+
for t in trees:
|
|
1873
|
+
sp = t.get("source_path", "")
|
|
1874
|
+
entry = sp.rsplit("#", 1)[-1] if "#" in sp else str(len(new_docs))
|
|
1875
|
+
new_docs.append(Document(
|
|
1876
|
+
doc_id=f"{name}__{entry}",
|
|
1877
|
+
doc_name=t.get("doc_name", name),
|
|
1878
|
+
structure=t.get("structure", []),
|
|
1879
|
+
doc_description=t.get("doc_description", ""),
|
|
1880
|
+
metadata={"source_path": sp},
|
|
1881
|
+
source_type=result.get("source_type", ""),
|
|
1882
|
+
))
|
|
1883
|
+
new_ids = {d.doc_id for d in new_docs}
|
|
1884
|
+
old_ids = set(fts.get_doc_ids_by_source_prefix(abs_fp + "#"))
|
|
1885
|
+
removed_ids = sorted(old_ids - new_ids)
|
|
1886
|
+
if removed_ids:
|
|
1887
|
+
fts.delete_documents(removed_ids)
|
|
1888
|
+
# 被移除派生文档(邮件删除)的落盘附件级联清理(ADR-0005)
|
|
1889
|
+
if base_dir:
|
|
1890
|
+
pst_rel = os.path.relpath(abs_fp, base_dir).replace(os.sep, "/")
|
|
1891
|
+
for rid in removed_ids:
|
|
1892
|
+
entry = rid.rsplit("__", 1)[-1]
|
|
1893
|
+
pst_att_store.purge_email(pst_rel, entry)
|
|
1894
|
+
logger.info("Removed %d stale derived doc(s) for %s",
|
|
1895
|
+
len(removed_ids), fp)
|
|
1896
|
+
for doc in new_docs:
|
|
1897
|
+
fts.index_document(doc, auto_commit=False)
|
|
1898
|
+
d = fts.last_node_diff
|
|
1899
|
+
for k in diff_totals:
|
|
1900
|
+
diff_totals[k] += d[k]
|
|
1901
|
+
_pending_commits += 1
|
|
1902
|
+
docs_since_optimize += 1
|
|
1903
|
+
if _pending_commits >= _COMMIT_BATCH:
|
|
1904
|
+
fts.commit()
|
|
1905
|
+
_pending_commits = 0
|
|
1906
|
+
if optimize_threshold and docs_since_optimize >= optimize_threshold:
|
|
1907
|
+
fts.optimize()
|
|
1908
|
+
docs_since_optimize = 0
|
|
1909
|
+
# 邮件元数据入库(pst_email_meta 表,供列表分页查询,ADR-0005)
|
|
1910
|
+
for t, doc in zip(trees, new_docs):
|
|
1911
|
+
meta = t.get("email_meta")
|
|
1912
|
+
if meta:
|
|
1913
|
+
fts.upsert_email_meta(doc.doc_id, abs_fp, meta)
|
|
1914
|
+
# 文件指纹记在物理文件路径上(派生文档不写 index_meta)
|
|
1915
|
+
fts.set_index_meta(abs_fp, file_h)
|
|
1916
|
+
fts.clear_failed_file(abs_fp)
|
|
1917
|
+
logger.debug("Indexed %d derived docs: %s -> %s",
|
|
1918
|
+
len(new_docs), fp, db_path)
|
|
1919
|
+
documents.extend(new_docs)
|
|
1920
|
+
_save_bar.update()
|
|
1921
|
+
continue
|
|
1922
|
+
|
|
1923
|
+
doc = Document(
|
|
1924
|
+
doc_id=name,
|
|
1925
|
+
doc_name=result.get("doc_name", name),
|
|
1926
|
+
structure=result.get("structure", []),
|
|
1927
|
+
doc_description=result.get("doc_description", ""),
|
|
1928
|
+
metadata={"source_path": result.get("source_path", "")},
|
|
1929
|
+
source_type=result.get("source_type", ""),
|
|
1930
|
+
)
|
|
1931
|
+
# index_document writes nodes, fts_nodes, documents AND index_meta
|
|
1932
|
+
# in a single atomic transaction (auto_commit handles batching).
|
|
1933
|
+
fts.index_document(doc, auto_commit=False, file_hash=file_h)
|
|
1934
|
+
# Clear any prior failure record for this file
|
|
1935
|
+
fts.clear_failed_file(abs_fp)
|
|
1936
|
+
d = fts.last_node_diff
|
|
1937
|
+
for k in diff_totals:
|
|
1938
|
+
diff_totals[k] += d[k]
|
|
1939
|
+
_pending_commits += 1
|
|
1940
|
+
docs_since_optimize += 1
|
|
1941
|
+
if _pending_commits >= _COMMIT_BATCH:
|
|
1942
|
+
fts.commit()
|
|
1943
|
+
_pending_commits = 0
|
|
1944
|
+
if optimize_threshold and docs_since_optimize >= optimize_threshold:
|
|
1945
|
+
fts.optimize()
|
|
1946
|
+
docs_since_optimize = 0
|
|
1947
|
+
logger.debug("Indexed: %s -> %s (doc_id=%s)", fp, db_path, name)
|
|
1948
|
+
else:
|
|
1949
|
+
# Skipped file: use batch-loaded docs (lookup by source_path, not doc_id)
|
|
1950
|
+
abs_fp = os.path.abspath(fp)
|
|
1951
|
+
doc = all_docs_from_db.get(abs_fp)
|
|
1952
|
+
if doc is None:
|
|
1953
|
+
# 多文档来源(PST 等):物理路径无精确匹配,按派生前缀收集
|
|
1954
|
+
derived = [d for sp, d in all_docs_from_db.items()
|
|
1955
|
+
if sp.startswith(abs_fp + "#")]
|
|
1956
|
+
if derived:
|
|
1957
|
+
documents.extend(derived)
|
|
1958
|
+
_save_bar.update()
|
|
1959
|
+
continue
|
|
1960
|
+
logger.debug("Skipped file %s has no document in DB (excluded by failure threshold)", fp)
|
|
1961
|
+
_save_bar.update()
|
|
1962
|
+
continue
|
|
1963
|
+
documents.append(doc)
|
|
1964
|
+
_save_bar.update()
|
|
1965
|
+
# Final commit for remaining pending writes
|
|
1966
|
+
if _pending_commits > 0:
|
|
1967
|
+
logger.info("Committing remaining %d documents to database...", _pending_commits)
|
|
1968
|
+
fts.commit()
|
|
1969
|
+
_save_bar.close()
|
|
1970
|
+
|
|
1971
|
+
# Fold WAL sidecar back into the main DB so long-running daemons don't
|
|
1972
|
+
# accumulate a multi-GB -wal file across many incremental builds.
|
|
1973
|
+
fts.wal_checkpoint("TRUNCATE")
|
|
1974
|
+
|
|
1975
|
+
# ---------------------------------------------------------------
|
|
1976
|
+
# Build IndexStats
|
|
1977
|
+
# ---------------------------------------------------------------
|
|
1978
|
+
build_elapsed = time.monotonic() - build_start
|
|
1979
|
+
|
|
1980
|
+
# Count total nodes in newly indexed documents
|
|
1981
|
+
total_nodes = 0
|
|
1982
|
+
for doc in documents:
|
|
1983
|
+
total_nodes += len(flatten_tree(doc.structure))
|
|
1984
|
+
|
|
1985
|
+
# Per source_type aggregation
|
|
1986
|
+
per_type: dict[str, dict] = {}
|
|
1987
|
+
for fp, (stype, elapsed) in _file_timings.items():
|
|
1988
|
+
if stype == "(failed)":
|
|
1989
|
+
continue
|
|
1990
|
+
entry = per_type.setdefault(stype, {"count": 0, "nodes": 0, "time_s": 0.0})
|
|
1991
|
+
entry["count"] += 1
|
|
1992
|
+
entry["time_s"] += elapsed
|
|
1993
|
+
# Count nodes for this file
|
|
1994
|
+
result = result_map.get(fp)
|
|
1995
|
+
if result:
|
|
1996
|
+
if result.get("multi_docs") is not None:
|
|
1997
|
+
entry["nodes"] += sum(
|
|
1998
|
+
len(flatten_tree(t.get("structure", [])))
|
|
1999
|
+
for t in result["multi_docs"]
|
|
2000
|
+
)
|
|
2001
|
+
else:
|
|
2002
|
+
entry["nodes"] += len(flatten_tree(result.get("structure", [])))
|
|
2003
|
+
|
|
2004
|
+
# Database size
|
|
2005
|
+
db_size = 0
|
|
2006
|
+
if db_path and os.path.isfile(db_path):
|
|
2007
|
+
try:
|
|
2008
|
+
db_size = os.path.getsize(db_path)
|
|
2009
|
+
except OSError:
|
|
2010
|
+
pass
|
|
2011
|
+
|
|
2012
|
+
stats = IndexStats(
|
|
2013
|
+
total_files=len(expanded),
|
|
2014
|
+
indexed_files=len(to_index) - len(_failed_paths),
|
|
2015
|
+
skipped_files=len(skipped),
|
|
2016
|
+
failed_files=len(_failed_paths),
|
|
2017
|
+
excluded_files=excluded_count,
|
|
2018
|
+
total_nodes=total_nodes,
|
|
2019
|
+
total_time_s=build_elapsed,
|
|
2020
|
+
per_type=per_type,
|
|
2021
|
+
db_path=db_path,
|
|
2022
|
+
db_size_bytes=db_size,
|
|
2023
|
+
failed_paths=_failed_paths,
|
|
2024
|
+
node_diff=diff_totals,
|
|
2025
|
+
pruned_paths=pruned_paths,
|
|
2026
|
+
)
|
|
2027
|
+
|
|
2028
|
+
# Attach stats to the returned list for easy access
|
|
2029
|
+
class _DocumentList(list):
|
|
2030
|
+
"""List subclass that carries IndexStats."""
|
|
2031
|
+
stats: IndexStats = None # type: ignore[assignment]
|
|
2032
|
+
|
|
2033
|
+
doc_list = _DocumentList(documents)
|
|
2034
|
+
doc_list.stats = stats
|
|
2035
|
+
|
|
2036
|
+
fts.close()
|
|
2037
|
+
_lock_handle.release()
|
|
2038
|
+
return doc_list
|