treesearchlib 1.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. treesearch/__init__.py +53 -0
  2. treesearch/__main__.py +6 -0
  3. treesearch/_bin/pst-extract.exe +0 -0
  4. treesearch/cli.py +554 -0
  5. treesearch/config.py +206 -0
  6. treesearch/fts.py +2293 -0
  7. treesearch/heuristics.py +425 -0
  8. treesearch/indexer.py +2038 -0
  9. treesearch/parsers/__init__.py +62 -0
  10. treesearch/parsers/anydoc_parser.py +193 -0
  11. treesearch/parsers/ast_parser.py +136 -0
  12. treesearch/parsers/docx_parser.py +304 -0
  13. treesearch/parsers/email_html_md.py +60 -0
  14. treesearch/parsers/excel_parser.py +218 -0
  15. treesearch/parsers/html_parser.py +172 -0
  16. treesearch/parsers/image_metadata.py +345 -0
  17. treesearch/parsers/image_parser.py +59 -0
  18. treesearch/parsers/image_store.py +182 -0
  19. treesearch/parsers/markitdown_parser.py +258 -0
  20. treesearch/parsers/mhtml_parser.py +108 -0
  21. treesearch/parsers/pdf_parser.py +409 -0
  22. treesearch/parsers/pst_attachment_store.py +156 -0
  23. treesearch/parsers/pst_parser.py +733 -0
  24. treesearch/parsers/registry.py +405 -0
  25. treesearch/parsers/treesitter_parser.py +433 -0
  26. treesearch/pathutil.py +227 -0
  27. treesearch/py.typed +0 -0
  28. treesearch/ripgrep.py +159 -0
  29. treesearch/search.py +935 -0
  30. treesearch/tokenizer.py +176 -0
  31. treesearch/tree.py +393 -0
  32. treesearch/tree_searcher.py +1006 -0
  33. treesearch/treesearch.py +574 -0
  34. treesearch/watch.py +305 -0
  35. treesearchlib-1.1.0.dist-info/METADATA +124 -0
  36. treesearchlib-1.1.0.dist-info/RECORD +39 -0
  37. treesearchlib-1.1.0.dist-info/WHEEL +5 -0
  38. treesearchlib-1.1.0.dist-info/entry_points.txt +2 -0
  39. treesearchlib-1.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,425 @@
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ @author:XuMing(xuming624@qq.com)
4
+ @description: Heuristic scoring for tree search.
5
+
6
+ All scoring logic is centralized here so it can be tested, tuned, and
7
+ extended independently of the search pipeline.
8
+
9
+ No LLM or embedding dependencies. Pure rule-based scoring.
10
+
11
+ Scoring philosophy:
12
+ - FTS5 lexical score (body text BM25) is the dominant signal
13
+ - Title/phrase matches are bonuses, not requirements
14
+ - Body text term overlap provides content-aware scoring even when titles are generic
15
+ - Ancestor support propagates path-level relevance
16
+ """
17
+ import logging
18
+ import math
19
+ import re
20
+ from dataclasses import dataclass, field
21
+
22
+ logger = logging.getLogger(__name__)
23
+
24
+
25
+ # ---------------------------------------------------------------------------
26
+ # Query Plan
27
+ # ---------------------------------------------------------------------------
28
+
29
+ @dataclass
30
+ class QueryPlan:
31
+ """Structured representation of a parsed query.
32
+
33
+ Attributes:
34
+ raw: original query string
35
+ terms: cleaned individual terms
36
+ phrases: exact phrase fragments (from quoted substrings or consecutive terms)
37
+ is_code_query: whether the query likely targets code (function/class/import)
38
+ is_structural_query: whether the query targets structural location (chapter/section)
39
+ """
40
+ raw: str = ""
41
+ terms: list[str] = field(default_factory=list)
42
+ phrases: list[str] = field(default_factory=list)
43
+ is_code_query: bool = False
44
+ is_structural_query: bool = False
45
+
46
+
47
+ # Patterns for query intent detection
48
+ _CODE_SIGNALS = re.compile(
49
+ r'\b(function|func|def|class|import|module|method|param|return|error|exception|api|endpoint)\b',
50
+ re.IGNORECASE,
51
+ )
52
+ _STRUCT_SIGNALS = re.compile(
53
+ r'\b(chapter|section|appendix|part|table of contents|toc)\b|'
54
+ r'第[一二三四五六七八九十\d]+[章节篇部]|'
55
+ r'\b[Qq]\d+\b|\bv\d+\.\d+',
56
+ re.IGNORECASE,
57
+ )
58
+ _QUOTED_PHRASE = re.compile(r'"([^"]+)"')
59
+
60
+ # Stop words to ignore during term overlap computation
61
+ _STOP_WORDS = frozenset({
62
+ # English
63
+ "a", "an", "the", "is", "are", "was", "were", "be", "been", "being",
64
+ "have", "has", "had", "do", "does", "did", "will", "would", "could",
65
+ "should", "may", "might", "shall", "can", "of", "in", "to", "for",
66
+ "with", "on", "at", "by", "from", "as", "into", "through", "during",
67
+ "before", "after", "above", "below", "between", "and", "but", "or",
68
+ "not", "no", "so", "if", "than", "too", "very", "just", "about",
69
+ "also", "then", "this", "that", "these", "those", "it", "its",
70
+ "what", "which", "who", "whom", "how", "when", "where", "why",
71
+ "all", "each", "every", "both", "few", "more", "most", "other",
72
+ "some", "such", "only", "own", "same", "we", "they", "he", "she",
73
+ "us", "our", "their", "your", "my", "i", "me", "you",
74
+ # Chinese
75
+ "的", "了", "在", "是", "我", "有", "和", "就", "不", "人", "都",
76
+ "一", "一个", "上", "也", "很", "到", "说", "要", "去", "你",
77
+ "会", "着", "没有", "看", "好", "自己", "这", "他", "她", "它",
78
+ "吗", "吧", "呢", "啊", "呀", "哦", "嗯", "嘛", "哈",
79
+ "怎样", "怎么", "什么", "哪", "哪个", "哪些", "为什么", "如何",
80
+ "可以", "能", "把", "被", "让", "给", "对", "从", "向", "跟",
81
+ "还", "又", "再", "已", "已经", "正在", "将", "将要",
82
+ })
83
+
84
+
85
+ def build_query_plan(query: str) -> QueryPlan:
86
+ """Parse a raw query string into a structured QueryPlan.
87
+
88
+ Steps:
89
+ 1. Extract quoted phrases
90
+ 2. Tokenize remaining text (CJK-aware: uses jieba for Chinese)
91
+ 3. Filter stop words
92
+ 4. Detect code / structural intent via regex
93
+ """
94
+ from .tokenizer import tokenize, _RE_HAS_CJK
95
+
96
+ plan = QueryPlan(raw=query)
97
+
98
+ # Extract quoted phrases
99
+ for m in _QUOTED_PHRASE.finditer(query):
100
+ plan.phrases.append(m.group(1).strip())
101
+ remaining = _QUOTED_PHRASE.sub("", query)
102
+
103
+ # Tokenize with CJK support (jieba for Chinese, whitespace for English)
104
+ remaining = remaining.strip()
105
+ tokens = tokenize(remaining, use_stemmer=False, remove_stopwords=False)
106
+ # Further filter: remove stop words and single-char English tokens
107
+ raw_terms = [t.lower() for t in tokens if t.strip()]
108
+ plan.terms = [t for t in raw_terms if t not in _STOP_WORDS and (
109
+ len(t) > 1 or _RE_HAS_CJK.match(t)
110
+ )]
111
+ # Deduplicate while preserving order
112
+ seen = set()
113
+ deduped = []
114
+ for t in plan.terms:
115
+ if t not in seen:
116
+ seen.add(t)
117
+ deduped.append(t)
118
+ plan.terms = deduped
119
+ # Fallback: if all terms were stop words, keep original
120
+ if not plan.terms and raw_terms:
121
+ plan.terms = raw_terms
122
+
123
+ # Build implicit phrase from consecutive terms (2-gram)
124
+ if len(raw_terms) >= 2 and not plan.phrases:
125
+ plan.phrases.append(" ".join(raw_terms))
126
+
127
+ # Intent detection
128
+ plan.is_code_query = bool(_CODE_SIGNALS.search(query))
129
+ plan.is_structural_query = bool(_STRUCT_SIGNALS.search(query))
130
+
131
+ return plan
132
+
133
+
134
+ # ---------------------------------------------------------------------------
135
+ # Term overlap ratio (content-aware scoring)
136
+ # ---------------------------------------------------------------------------
137
+
138
+ def compute_term_overlap(text: str, terms: list[str], idf: dict[str, float] | None = None) -> float:
139
+ """Compute IDF-weighted fraction of query terms that appear in the text.
140
+
141
+ When ``idf`` is provided, rare terms contribute more than common terms.
142
+ Falls back to uniform weighting when ``idf`` is None.
143
+
144
+ Returns a value in [0.0, 1.0].
145
+ """
146
+ if not text or not terms:
147
+ return 0.0
148
+ text_lower = text.lower()
149
+ if idf:
150
+ total_w = sum(idf.get(t, 1.0) for t in terms)
151
+ if total_w <= 0:
152
+ return 0.0
153
+ hit_w = sum(idf.get(t, 1.0) for t in terms if t in text_lower)
154
+ return hit_w / total_w
155
+ # Uniform fallback
156
+ matched = sum(1 for t in terms if t in text_lower)
157
+ return matched / len(terms)
158
+
159
+
160
+ def estimate_idf(terms: list[str], corpus_texts: list[str]) -> dict[str, float]:
161
+ """Estimate IDF weights for query terms from a corpus of node texts.
162
+
163
+ Uses smooth IDF: log((N + 1) / (df + 1)) + 1 to avoid zero weights.
164
+ Corpus is typically all node texts from a single document.
165
+
166
+ Optimization: pre-lowercase corpus texts once, then check all terms.
167
+ Previous version did `text.lower()` inside the inner loop (N * T calls).
168
+
169
+ Args:
170
+ terms: query terms (lowercased)
171
+ corpus_texts: list of node text strings
172
+
173
+ Returns:
174
+ {term: idf_weight} for each term
175
+ """
176
+ n = len(corpus_texts)
177
+ if n == 0:
178
+ return {t: 1.0 for t in terms}
179
+ df: dict[str, int] = {t: 0 for t in terms}
180
+ for text in corpus_texts:
181
+ text_lower = text.lower()
182
+ for t in terms:
183
+ if t in text_lower:
184
+ df[t] += 1
185
+ idf = {}
186
+ for t in terms:
187
+ idf[t] = math.log((n + 1) / (df[t] + 1)) + 1.0
188
+ return idf
189
+
190
+
191
+ # ---------------------------------------------------------------------------
192
+ # Anchor Scorer
193
+ # ---------------------------------------------------------------------------
194
+
195
+ def score_anchor(
196
+ fts_score: float,
197
+ depth: int,
198
+ has_title_match: bool = False,
199
+ has_phrase_match: bool = False,
200
+ body_term_overlap: float = 0.0,
201
+ max_depth: int = 6,
202
+ ) -> float:
203
+ """Score a candidate anchor node.
204
+
205
+ Anchors should be high-level entry points, so deeper nodes get penalized.
206
+ FTS5 score (which already incorporates body text BM25) is the primary signal.
207
+
208
+ Args:
209
+ fts_score: normalized FTS5 relevance score [0, 1]
210
+ depth: node depth in tree (0 = root)
211
+ has_title_match: whether query terms appear in node title
212
+ has_phrase_match: whether exact phrase appears in node
213
+ body_term_overlap: fraction of query terms found in node body [0, 1]
214
+ max_depth: max depth for normalization
215
+
216
+ Returns:
217
+ anchor score in [0, 1] range
218
+ """
219
+ # Depth penalty: deeper nodes are less ideal as anchors
220
+ depth_penalty = min(depth / max(max_depth, 1), 1.0) * 0.10
221
+
222
+ # Title match bonus: query terms in node title strongly indicate relevance.
223
+ # Academic papers: section titles precisely describe content.
224
+ # Financial docs: titles like "Net Sales", "Operating Income" are highly discriminative.
225
+ title_bonus = 0.15 if has_title_match else 0.0
226
+
227
+ # Phrase match bonus
228
+ phrase_bonus = 0.07 if has_phrase_match else 0.0
229
+
230
+ # Body content overlap bonus -- reward nodes whose text contains query terms
231
+ body_bonus = 0.10 * body_term_overlap
232
+
233
+ score = fts_score + title_bonus + phrase_bonus + body_bonus - depth_penalty
234
+ return max(0.0, min(1.0, score))
235
+
236
+
237
+ # ---------------------------------------------------------------------------
238
+ # Walk Scorer
239
+ # ---------------------------------------------------------------------------
240
+
241
+ def score_walk_node(
242
+ lexical_score: float,
243
+ *,
244
+ has_title_match: bool = False,
245
+ has_phrase_match: bool = False,
246
+ body_term_overlap: float = 0.0,
247
+ ancestor_support: float = 0.0,
248
+ hop: int = 0,
249
+ is_redundant: bool = False,
250
+ max_hops: int = 3,
251
+ ) -> float:
252
+ """Score a node during tree walk expansion.
253
+
254
+ Design: FTS5 lexical_score already captures BM25 relevance over body text.
255
+ We weight it heavily so that content-rich nodes with query term hits rank
256
+ high even when their title is generic (e.g. "Results", "Discussion").
257
+
258
+ Args:
259
+ lexical_score: FTS5 normalized score for this node [0, 1]
260
+ has_title_match: query terms appear in title
261
+ has_phrase_match: exact phrase appears in node
262
+ body_term_overlap: fraction of query terms in node body [0, 1]
263
+ ancestor_support: max score among ancestors on current path [0, 1]
264
+ hop: number of hops from anchor
265
+ is_redundant: whether this node overlaps with already-visited paths
266
+ max_hops: max hops for normalization
267
+
268
+ Returns:
269
+ walk score (higher = more worth expanding)
270
+ """
271
+ # Base: FTS5 lexical relevance is the primary signal (heavy weight)
272
+ score = 0.45 * lexical_score
273
+
274
+ # Body text content overlap -- works even when FTS5 score is 0 (unexpanded nodes)
275
+ score += 0.15 * body_term_overlap
276
+
277
+ # Title match bonus (reduced from old 0.20 -> 0.08)
278
+ if has_title_match:
279
+ score += 0.08
280
+
281
+ # Phrase match bonus
282
+ if has_phrase_match:
283
+ score += 0.07
284
+
285
+ # Ancestor support: path consistency
286
+ score += 0.12 * ancestor_support
287
+
288
+ # Hop penalty: further from anchor = less relevant
289
+ hop_ratio = min(hop / max(max_hops, 1), 1.0)
290
+ score -= 0.08 * hop_ratio
291
+
292
+ # Redundancy penalty
293
+ if is_redundant:
294
+ score -= 0.08
295
+
296
+ return max(0.0, score)
297
+
298
+
299
+ # ---------------------------------------------------------------------------
300
+ # Path Scorer
301
+ # ---------------------------------------------------------------------------
302
+
303
+ def score_path(
304
+ leaf_score: float,
305
+ path_titles: list[str],
306
+ path_texts: list[str],
307
+ query_terms: list[str],
308
+ path_length: int,
309
+ leaf_fts_score: float = 0.0,
310
+ max_path_length: int = 6,
311
+ ) -> float:
312
+ """Score a complete root-to-leaf path.
313
+
314
+ Combines leaf node quality with path-level content coverage.
315
+ Uses both title and body text for path scoring to avoid title-dependency.
316
+
317
+ Args:
318
+ leaf_score: walk score of the terminal node
319
+ path_titles: titles along the path (root to leaf)
320
+ path_texts: body texts along the path (root to leaf)
321
+ query_terms: cleaned query terms from QueryPlan
322
+ path_length: number of nodes in path
323
+ leaf_fts_score: FTS5 score of the leaf node [0, 1]
324
+ max_path_length: max path length for normalization
325
+
326
+ Returns:
327
+ path score (higher = better answer path)
328
+ """
329
+ # Leaf score dominates (walk-level quality)
330
+ score = 0.30 * leaf_score
331
+
332
+ # Leaf FTS5 score direct contribution (content relevance of the answer node)
333
+ score += 0.30 * leaf_fts_score
334
+
335
+ # Path content coverage: how many query terms appear in ANY node's text along the path
336
+ if query_terms and path_texts:
337
+ all_text = " ".join(path_texts).lower()
338
+ covered = sum(1 for t in query_terms if t in all_text)
339
+ text_coverage = covered / len(query_terms)
340
+ score += 0.20 * text_coverage
341
+
342
+ # Path title consistency: how many path titles contain query terms
343
+ if path_titles and query_terms:
344
+ match_count = 0
345
+ for title in path_titles:
346
+ title_lower = title.lower()
347
+ if any(t in title_lower for t in query_terms):
348
+ match_count += 1
349
+ consistency = match_count / len(path_titles)
350
+ score += 0.10 * consistency
351
+
352
+ # Context coverage: how many query terms appear somewhere in path titles
353
+ if query_terms and path_titles:
354
+ all_titles_text = " ".join(path_titles).lower()
355
+ covered = sum(1 for t in query_terms if t in all_titles_text)
356
+ coverage = covered / len(query_terms)
357
+ score += 0.08 * coverage
358
+
359
+ # Readability bonus: shorter paths are easier to present
360
+ length_ratio = min(path_length / max(max_path_length, 1), 1.0)
361
+ readability = 1.0 - length_ratio * 0.5
362
+ score += 0.07 * readability
363
+
364
+ return max(0.0, min(1.0, score))
365
+
366
+
367
+ # ---------------------------------------------------------------------------
368
+ # Utility: term matching helpers
369
+ # ---------------------------------------------------------------------------
370
+
371
+ def check_title_match(title: str, terms: list[str]) -> bool:
372
+ """Check if any query term appears in the node title."""
373
+ if not title or not terms:
374
+ return False
375
+ title_lower = title.lower()
376
+ return any(t in title_lower for t in terms)
377
+
378
+
379
+ def check_phrase_match(text: str, phrases: list[str]) -> bool:
380
+ """Check if any exact phrase appears in the text."""
381
+ if not text or not phrases:
382
+ return False
383
+ text_lower = text.lower()
384
+ return any(p.lower() in text_lower for p in phrases)
385
+
386
+
387
+ # ---------------------------------------------------------------------------
388
+ # Generic Section Detection
389
+ # ---------------------------------------------------------------------------
390
+
391
+ # Sections that typically contain broad overview text rather than specific answers.
392
+ # These get high BM25 scores because they mention many terms, but rarely contain
393
+ # the precise information a question targets.
394
+ _GENERIC_SECTIONS = frozenset({
395
+ "abstract", "introduction", "conclusion", "conclusions",
396
+ "related work", "acknowledgments", "acknowledgements",
397
+ "conclusion and outlook", "conclusions and outlook",
398
+ "conclusion and future work", "conclusions and future work",
399
+ "background", "overview",
400
+ })
401
+
402
+
403
+ def is_generic_section(title: str, depth: int) -> bool:
404
+ """Check if a node is a generic overview section.
405
+
406
+ Only applies to top-level sections (depth 0-1) whose base title
407
+ (before ::: delimiter) is in the generic set.
408
+
409
+ Args:
410
+ title: node title string
411
+ depth: node depth in tree (0 = root)
412
+
413
+ Returns:
414
+ True if the node is a generic section that should be demoted
415
+ """
416
+ if depth > 1:
417
+ return False
418
+ if not title:
419
+ return False
420
+ # For ::: delimited titles, only check the base (leftmost) part
421
+ base_title = title.split(" ::: ")[0].strip().lower() if " ::: " in title else title.strip().lower()
422
+ # Root node (depth=0, paper title) is almost never relevant
423
+ if depth == 0:
424
+ return True
425
+ return base_title in _GENERIC_SECTIONS