treesearchlib 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- treesearch/__init__.py +53 -0
- treesearch/__main__.py +6 -0
- treesearch/_bin/pst-extract.exe +0 -0
- treesearch/cli.py +554 -0
- treesearch/config.py +206 -0
- treesearch/fts.py +2293 -0
- treesearch/heuristics.py +425 -0
- treesearch/indexer.py +2038 -0
- treesearch/parsers/__init__.py +62 -0
- treesearch/parsers/anydoc_parser.py +193 -0
- treesearch/parsers/ast_parser.py +136 -0
- treesearch/parsers/docx_parser.py +304 -0
- treesearch/parsers/email_html_md.py +60 -0
- treesearch/parsers/excel_parser.py +218 -0
- treesearch/parsers/html_parser.py +172 -0
- treesearch/parsers/image_metadata.py +345 -0
- treesearch/parsers/image_parser.py +59 -0
- treesearch/parsers/image_store.py +182 -0
- treesearch/parsers/markitdown_parser.py +258 -0
- treesearch/parsers/mhtml_parser.py +108 -0
- treesearch/parsers/pdf_parser.py +409 -0
- treesearch/parsers/pst_attachment_store.py +156 -0
- treesearch/parsers/pst_parser.py +733 -0
- treesearch/parsers/registry.py +405 -0
- treesearch/parsers/treesitter_parser.py +433 -0
- treesearch/pathutil.py +227 -0
- treesearch/py.typed +0 -0
- treesearch/ripgrep.py +159 -0
- treesearch/search.py +935 -0
- treesearch/tokenizer.py +176 -0
- treesearch/tree.py +393 -0
- treesearch/tree_searcher.py +1006 -0
- treesearch/treesearch.py +574 -0
- treesearch/watch.py +305 -0
- treesearchlib-1.1.0.dist-info/METADATA +124 -0
- treesearchlib-1.1.0.dist-info/RECORD +39 -0
- treesearchlib-1.1.0.dist-info/WHEEL +5 -0
- treesearchlib-1.1.0.dist-info/entry_points.txt +2 -0
- treesearchlib-1.1.0.dist-info/top_level.txt +1 -0
treesearch/heuristics.py
ADDED
|
@@ -0,0 +1,425 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""
|
|
3
|
+
@author:XuMing(xuming624@qq.com)
|
|
4
|
+
@description: Heuristic scoring for tree search.
|
|
5
|
+
|
|
6
|
+
All scoring logic is centralized here so it can be tested, tuned, and
|
|
7
|
+
extended independently of the search pipeline.
|
|
8
|
+
|
|
9
|
+
No LLM or embedding dependencies. Pure rule-based scoring.
|
|
10
|
+
|
|
11
|
+
Scoring philosophy:
|
|
12
|
+
- FTS5 lexical score (body text BM25) is the dominant signal
|
|
13
|
+
- Title/phrase matches are bonuses, not requirements
|
|
14
|
+
- Body text term overlap provides content-aware scoring even when titles are generic
|
|
15
|
+
- Ancestor support propagates path-level relevance
|
|
16
|
+
"""
|
|
17
|
+
import logging
|
|
18
|
+
import math
|
|
19
|
+
import re
|
|
20
|
+
from dataclasses import dataclass, field
|
|
21
|
+
|
|
22
|
+
logger = logging.getLogger(__name__)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
# ---------------------------------------------------------------------------
|
|
26
|
+
# Query Plan
|
|
27
|
+
# ---------------------------------------------------------------------------
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class QueryPlan:
|
|
31
|
+
"""Structured representation of a parsed query.
|
|
32
|
+
|
|
33
|
+
Attributes:
|
|
34
|
+
raw: original query string
|
|
35
|
+
terms: cleaned individual terms
|
|
36
|
+
phrases: exact phrase fragments (from quoted substrings or consecutive terms)
|
|
37
|
+
is_code_query: whether the query likely targets code (function/class/import)
|
|
38
|
+
is_structural_query: whether the query targets structural location (chapter/section)
|
|
39
|
+
"""
|
|
40
|
+
raw: str = ""
|
|
41
|
+
terms: list[str] = field(default_factory=list)
|
|
42
|
+
phrases: list[str] = field(default_factory=list)
|
|
43
|
+
is_code_query: bool = False
|
|
44
|
+
is_structural_query: bool = False
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
# Patterns for query intent detection
|
|
48
|
+
_CODE_SIGNALS = re.compile(
|
|
49
|
+
r'\b(function|func|def|class|import|module|method|param|return|error|exception|api|endpoint)\b',
|
|
50
|
+
re.IGNORECASE,
|
|
51
|
+
)
|
|
52
|
+
_STRUCT_SIGNALS = re.compile(
|
|
53
|
+
r'\b(chapter|section|appendix|part|table of contents|toc)\b|'
|
|
54
|
+
r'第[一二三四五六七八九十\d]+[章节篇部]|'
|
|
55
|
+
r'\b[Qq]\d+\b|\bv\d+\.\d+',
|
|
56
|
+
re.IGNORECASE,
|
|
57
|
+
)
|
|
58
|
+
_QUOTED_PHRASE = re.compile(r'"([^"]+)"')
|
|
59
|
+
|
|
60
|
+
# Stop words to ignore during term overlap computation
|
|
61
|
+
_STOP_WORDS = frozenset({
|
|
62
|
+
# English
|
|
63
|
+
"a", "an", "the", "is", "are", "was", "were", "be", "been", "being",
|
|
64
|
+
"have", "has", "had", "do", "does", "did", "will", "would", "could",
|
|
65
|
+
"should", "may", "might", "shall", "can", "of", "in", "to", "for",
|
|
66
|
+
"with", "on", "at", "by", "from", "as", "into", "through", "during",
|
|
67
|
+
"before", "after", "above", "below", "between", "and", "but", "or",
|
|
68
|
+
"not", "no", "so", "if", "than", "too", "very", "just", "about",
|
|
69
|
+
"also", "then", "this", "that", "these", "those", "it", "its",
|
|
70
|
+
"what", "which", "who", "whom", "how", "when", "where", "why",
|
|
71
|
+
"all", "each", "every", "both", "few", "more", "most", "other",
|
|
72
|
+
"some", "such", "only", "own", "same", "we", "they", "he", "she",
|
|
73
|
+
"us", "our", "their", "your", "my", "i", "me", "you",
|
|
74
|
+
# Chinese
|
|
75
|
+
"的", "了", "在", "是", "我", "有", "和", "就", "不", "人", "都",
|
|
76
|
+
"一", "一个", "上", "也", "很", "到", "说", "要", "去", "你",
|
|
77
|
+
"会", "着", "没有", "看", "好", "自己", "这", "他", "她", "它",
|
|
78
|
+
"吗", "吧", "呢", "啊", "呀", "哦", "嗯", "嘛", "哈",
|
|
79
|
+
"怎样", "怎么", "什么", "哪", "哪个", "哪些", "为什么", "如何",
|
|
80
|
+
"可以", "能", "把", "被", "让", "给", "对", "从", "向", "跟",
|
|
81
|
+
"还", "又", "再", "已", "已经", "正在", "将", "将要",
|
|
82
|
+
})
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def build_query_plan(query: str) -> QueryPlan:
|
|
86
|
+
"""Parse a raw query string into a structured QueryPlan.
|
|
87
|
+
|
|
88
|
+
Steps:
|
|
89
|
+
1. Extract quoted phrases
|
|
90
|
+
2. Tokenize remaining text (CJK-aware: uses jieba for Chinese)
|
|
91
|
+
3. Filter stop words
|
|
92
|
+
4. Detect code / structural intent via regex
|
|
93
|
+
"""
|
|
94
|
+
from .tokenizer import tokenize, _RE_HAS_CJK
|
|
95
|
+
|
|
96
|
+
plan = QueryPlan(raw=query)
|
|
97
|
+
|
|
98
|
+
# Extract quoted phrases
|
|
99
|
+
for m in _QUOTED_PHRASE.finditer(query):
|
|
100
|
+
plan.phrases.append(m.group(1).strip())
|
|
101
|
+
remaining = _QUOTED_PHRASE.sub("", query)
|
|
102
|
+
|
|
103
|
+
# Tokenize with CJK support (jieba for Chinese, whitespace for English)
|
|
104
|
+
remaining = remaining.strip()
|
|
105
|
+
tokens = tokenize(remaining, use_stemmer=False, remove_stopwords=False)
|
|
106
|
+
# Further filter: remove stop words and single-char English tokens
|
|
107
|
+
raw_terms = [t.lower() for t in tokens if t.strip()]
|
|
108
|
+
plan.terms = [t for t in raw_terms if t not in _STOP_WORDS and (
|
|
109
|
+
len(t) > 1 or _RE_HAS_CJK.match(t)
|
|
110
|
+
)]
|
|
111
|
+
# Deduplicate while preserving order
|
|
112
|
+
seen = set()
|
|
113
|
+
deduped = []
|
|
114
|
+
for t in plan.terms:
|
|
115
|
+
if t not in seen:
|
|
116
|
+
seen.add(t)
|
|
117
|
+
deduped.append(t)
|
|
118
|
+
plan.terms = deduped
|
|
119
|
+
# Fallback: if all terms were stop words, keep original
|
|
120
|
+
if not plan.terms and raw_terms:
|
|
121
|
+
plan.terms = raw_terms
|
|
122
|
+
|
|
123
|
+
# Build implicit phrase from consecutive terms (2-gram)
|
|
124
|
+
if len(raw_terms) >= 2 and not plan.phrases:
|
|
125
|
+
plan.phrases.append(" ".join(raw_terms))
|
|
126
|
+
|
|
127
|
+
# Intent detection
|
|
128
|
+
plan.is_code_query = bool(_CODE_SIGNALS.search(query))
|
|
129
|
+
plan.is_structural_query = bool(_STRUCT_SIGNALS.search(query))
|
|
130
|
+
|
|
131
|
+
return plan
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
# ---------------------------------------------------------------------------
|
|
135
|
+
# Term overlap ratio (content-aware scoring)
|
|
136
|
+
# ---------------------------------------------------------------------------
|
|
137
|
+
|
|
138
|
+
def compute_term_overlap(text: str, terms: list[str], idf: dict[str, float] | None = None) -> float:
|
|
139
|
+
"""Compute IDF-weighted fraction of query terms that appear in the text.
|
|
140
|
+
|
|
141
|
+
When ``idf`` is provided, rare terms contribute more than common terms.
|
|
142
|
+
Falls back to uniform weighting when ``idf`` is None.
|
|
143
|
+
|
|
144
|
+
Returns a value in [0.0, 1.0].
|
|
145
|
+
"""
|
|
146
|
+
if not text or not terms:
|
|
147
|
+
return 0.0
|
|
148
|
+
text_lower = text.lower()
|
|
149
|
+
if idf:
|
|
150
|
+
total_w = sum(idf.get(t, 1.0) for t in terms)
|
|
151
|
+
if total_w <= 0:
|
|
152
|
+
return 0.0
|
|
153
|
+
hit_w = sum(idf.get(t, 1.0) for t in terms if t in text_lower)
|
|
154
|
+
return hit_w / total_w
|
|
155
|
+
# Uniform fallback
|
|
156
|
+
matched = sum(1 for t in terms if t in text_lower)
|
|
157
|
+
return matched / len(terms)
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def estimate_idf(terms: list[str], corpus_texts: list[str]) -> dict[str, float]:
|
|
161
|
+
"""Estimate IDF weights for query terms from a corpus of node texts.
|
|
162
|
+
|
|
163
|
+
Uses smooth IDF: log((N + 1) / (df + 1)) + 1 to avoid zero weights.
|
|
164
|
+
Corpus is typically all node texts from a single document.
|
|
165
|
+
|
|
166
|
+
Optimization: pre-lowercase corpus texts once, then check all terms.
|
|
167
|
+
Previous version did `text.lower()` inside the inner loop (N * T calls).
|
|
168
|
+
|
|
169
|
+
Args:
|
|
170
|
+
terms: query terms (lowercased)
|
|
171
|
+
corpus_texts: list of node text strings
|
|
172
|
+
|
|
173
|
+
Returns:
|
|
174
|
+
{term: idf_weight} for each term
|
|
175
|
+
"""
|
|
176
|
+
n = len(corpus_texts)
|
|
177
|
+
if n == 0:
|
|
178
|
+
return {t: 1.0 for t in terms}
|
|
179
|
+
df: dict[str, int] = {t: 0 for t in terms}
|
|
180
|
+
for text in corpus_texts:
|
|
181
|
+
text_lower = text.lower()
|
|
182
|
+
for t in terms:
|
|
183
|
+
if t in text_lower:
|
|
184
|
+
df[t] += 1
|
|
185
|
+
idf = {}
|
|
186
|
+
for t in terms:
|
|
187
|
+
idf[t] = math.log((n + 1) / (df[t] + 1)) + 1.0
|
|
188
|
+
return idf
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
# ---------------------------------------------------------------------------
|
|
192
|
+
# Anchor Scorer
|
|
193
|
+
# ---------------------------------------------------------------------------
|
|
194
|
+
|
|
195
|
+
def score_anchor(
|
|
196
|
+
fts_score: float,
|
|
197
|
+
depth: int,
|
|
198
|
+
has_title_match: bool = False,
|
|
199
|
+
has_phrase_match: bool = False,
|
|
200
|
+
body_term_overlap: float = 0.0,
|
|
201
|
+
max_depth: int = 6,
|
|
202
|
+
) -> float:
|
|
203
|
+
"""Score a candidate anchor node.
|
|
204
|
+
|
|
205
|
+
Anchors should be high-level entry points, so deeper nodes get penalized.
|
|
206
|
+
FTS5 score (which already incorporates body text BM25) is the primary signal.
|
|
207
|
+
|
|
208
|
+
Args:
|
|
209
|
+
fts_score: normalized FTS5 relevance score [0, 1]
|
|
210
|
+
depth: node depth in tree (0 = root)
|
|
211
|
+
has_title_match: whether query terms appear in node title
|
|
212
|
+
has_phrase_match: whether exact phrase appears in node
|
|
213
|
+
body_term_overlap: fraction of query terms found in node body [0, 1]
|
|
214
|
+
max_depth: max depth for normalization
|
|
215
|
+
|
|
216
|
+
Returns:
|
|
217
|
+
anchor score in [0, 1] range
|
|
218
|
+
"""
|
|
219
|
+
# Depth penalty: deeper nodes are less ideal as anchors
|
|
220
|
+
depth_penalty = min(depth / max(max_depth, 1), 1.0) * 0.10
|
|
221
|
+
|
|
222
|
+
# Title match bonus: query terms in node title strongly indicate relevance.
|
|
223
|
+
# Academic papers: section titles precisely describe content.
|
|
224
|
+
# Financial docs: titles like "Net Sales", "Operating Income" are highly discriminative.
|
|
225
|
+
title_bonus = 0.15 if has_title_match else 0.0
|
|
226
|
+
|
|
227
|
+
# Phrase match bonus
|
|
228
|
+
phrase_bonus = 0.07 if has_phrase_match else 0.0
|
|
229
|
+
|
|
230
|
+
# Body content overlap bonus -- reward nodes whose text contains query terms
|
|
231
|
+
body_bonus = 0.10 * body_term_overlap
|
|
232
|
+
|
|
233
|
+
score = fts_score + title_bonus + phrase_bonus + body_bonus - depth_penalty
|
|
234
|
+
return max(0.0, min(1.0, score))
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
# ---------------------------------------------------------------------------
|
|
238
|
+
# Walk Scorer
|
|
239
|
+
# ---------------------------------------------------------------------------
|
|
240
|
+
|
|
241
|
+
def score_walk_node(
|
|
242
|
+
lexical_score: float,
|
|
243
|
+
*,
|
|
244
|
+
has_title_match: bool = False,
|
|
245
|
+
has_phrase_match: bool = False,
|
|
246
|
+
body_term_overlap: float = 0.0,
|
|
247
|
+
ancestor_support: float = 0.0,
|
|
248
|
+
hop: int = 0,
|
|
249
|
+
is_redundant: bool = False,
|
|
250
|
+
max_hops: int = 3,
|
|
251
|
+
) -> float:
|
|
252
|
+
"""Score a node during tree walk expansion.
|
|
253
|
+
|
|
254
|
+
Design: FTS5 lexical_score already captures BM25 relevance over body text.
|
|
255
|
+
We weight it heavily so that content-rich nodes with query term hits rank
|
|
256
|
+
high even when their title is generic (e.g. "Results", "Discussion").
|
|
257
|
+
|
|
258
|
+
Args:
|
|
259
|
+
lexical_score: FTS5 normalized score for this node [0, 1]
|
|
260
|
+
has_title_match: query terms appear in title
|
|
261
|
+
has_phrase_match: exact phrase appears in node
|
|
262
|
+
body_term_overlap: fraction of query terms in node body [0, 1]
|
|
263
|
+
ancestor_support: max score among ancestors on current path [0, 1]
|
|
264
|
+
hop: number of hops from anchor
|
|
265
|
+
is_redundant: whether this node overlaps with already-visited paths
|
|
266
|
+
max_hops: max hops for normalization
|
|
267
|
+
|
|
268
|
+
Returns:
|
|
269
|
+
walk score (higher = more worth expanding)
|
|
270
|
+
"""
|
|
271
|
+
# Base: FTS5 lexical relevance is the primary signal (heavy weight)
|
|
272
|
+
score = 0.45 * lexical_score
|
|
273
|
+
|
|
274
|
+
# Body text content overlap -- works even when FTS5 score is 0 (unexpanded nodes)
|
|
275
|
+
score += 0.15 * body_term_overlap
|
|
276
|
+
|
|
277
|
+
# Title match bonus (reduced from old 0.20 -> 0.08)
|
|
278
|
+
if has_title_match:
|
|
279
|
+
score += 0.08
|
|
280
|
+
|
|
281
|
+
# Phrase match bonus
|
|
282
|
+
if has_phrase_match:
|
|
283
|
+
score += 0.07
|
|
284
|
+
|
|
285
|
+
# Ancestor support: path consistency
|
|
286
|
+
score += 0.12 * ancestor_support
|
|
287
|
+
|
|
288
|
+
# Hop penalty: further from anchor = less relevant
|
|
289
|
+
hop_ratio = min(hop / max(max_hops, 1), 1.0)
|
|
290
|
+
score -= 0.08 * hop_ratio
|
|
291
|
+
|
|
292
|
+
# Redundancy penalty
|
|
293
|
+
if is_redundant:
|
|
294
|
+
score -= 0.08
|
|
295
|
+
|
|
296
|
+
return max(0.0, score)
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
# ---------------------------------------------------------------------------
|
|
300
|
+
# Path Scorer
|
|
301
|
+
# ---------------------------------------------------------------------------
|
|
302
|
+
|
|
303
|
+
def score_path(
|
|
304
|
+
leaf_score: float,
|
|
305
|
+
path_titles: list[str],
|
|
306
|
+
path_texts: list[str],
|
|
307
|
+
query_terms: list[str],
|
|
308
|
+
path_length: int,
|
|
309
|
+
leaf_fts_score: float = 0.0,
|
|
310
|
+
max_path_length: int = 6,
|
|
311
|
+
) -> float:
|
|
312
|
+
"""Score a complete root-to-leaf path.
|
|
313
|
+
|
|
314
|
+
Combines leaf node quality with path-level content coverage.
|
|
315
|
+
Uses both title and body text for path scoring to avoid title-dependency.
|
|
316
|
+
|
|
317
|
+
Args:
|
|
318
|
+
leaf_score: walk score of the terminal node
|
|
319
|
+
path_titles: titles along the path (root to leaf)
|
|
320
|
+
path_texts: body texts along the path (root to leaf)
|
|
321
|
+
query_terms: cleaned query terms from QueryPlan
|
|
322
|
+
path_length: number of nodes in path
|
|
323
|
+
leaf_fts_score: FTS5 score of the leaf node [0, 1]
|
|
324
|
+
max_path_length: max path length for normalization
|
|
325
|
+
|
|
326
|
+
Returns:
|
|
327
|
+
path score (higher = better answer path)
|
|
328
|
+
"""
|
|
329
|
+
# Leaf score dominates (walk-level quality)
|
|
330
|
+
score = 0.30 * leaf_score
|
|
331
|
+
|
|
332
|
+
# Leaf FTS5 score direct contribution (content relevance of the answer node)
|
|
333
|
+
score += 0.30 * leaf_fts_score
|
|
334
|
+
|
|
335
|
+
# Path content coverage: how many query terms appear in ANY node's text along the path
|
|
336
|
+
if query_terms and path_texts:
|
|
337
|
+
all_text = " ".join(path_texts).lower()
|
|
338
|
+
covered = sum(1 for t in query_terms if t in all_text)
|
|
339
|
+
text_coverage = covered / len(query_terms)
|
|
340
|
+
score += 0.20 * text_coverage
|
|
341
|
+
|
|
342
|
+
# Path title consistency: how many path titles contain query terms
|
|
343
|
+
if path_titles and query_terms:
|
|
344
|
+
match_count = 0
|
|
345
|
+
for title in path_titles:
|
|
346
|
+
title_lower = title.lower()
|
|
347
|
+
if any(t in title_lower for t in query_terms):
|
|
348
|
+
match_count += 1
|
|
349
|
+
consistency = match_count / len(path_titles)
|
|
350
|
+
score += 0.10 * consistency
|
|
351
|
+
|
|
352
|
+
# Context coverage: how many query terms appear somewhere in path titles
|
|
353
|
+
if query_terms and path_titles:
|
|
354
|
+
all_titles_text = " ".join(path_titles).lower()
|
|
355
|
+
covered = sum(1 for t in query_terms if t in all_titles_text)
|
|
356
|
+
coverage = covered / len(query_terms)
|
|
357
|
+
score += 0.08 * coverage
|
|
358
|
+
|
|
359
|
+
# Readability bonus: shorter paths are easier to present
|
|
360
|
+
length_ratio = min(path_length / max(max_path_length, 1), 1.0)
|
|
361
|
+
readability = 1.0 - length_ratio * 0.5
|
|
362
|
+
score += 0.07 * readability
|
|
363
|
+
|
|
364
|
+
return max(0.0, min(1.0, score))
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
# ---------------------------------------------------------------------------
|
|
368
|
+
# Utility: term matching helpers
|
|
369
|
+
# ---------------------------------------------------------------------------
|
|
370
|
+
|
|
371
|
+
def check_title_match(title: str, terms: list[str]) -> bool:
|
|
372
|
+
"""Check if any query term appears in the node title."""
|
|
373
|
+
if not title or not terms:
|
|
374
|
+
return False
|
|
375
|
+
title_lower = title.lower()
|
|
376
|
+
return any(t in title_lower for t in terms)
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
def check_phrase_match(text: str, phrases: list[str]) -> bool:
|
|
380
|
+
"""Check if any exact phrase appears in the text."""
|
|
381
|
+
if not text or not phrases:
|
|
382
|
+
return False
|
|
383
|
+
text_lower = text.lower()
|
|
384
|
+
return any(p.lower() in text_lower for p in phrases)
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
# ---------------------------------------------------------------------------
|
|
388
|
+
# Generic Section Detection
|
|
389
|
+
# ---------------------------------------------------------------------------
|
|
390
|
+
|
|
391
|
+
# Sections that typically contain broad overview text rather than specific answers.
|
|
392
|
+
# These get high BM25 scores because they mention many terms, but rarely contain
|
|
393
|
+
# the precise information a question targets.
|
|
394
|
+
_GENERIC_SECTIONS = frozenset({
|
|
395
|
+
"abstract", "introduction", "conclusion", "conclusions",
|
|
396
|
+
"related work", "acknowledgments", "acknowledgements",
|
|
397
|
+
"conclusion and outlook", "conclusions and outlook",
|
|
398
|
+
"conclusion and future work", "conclusions and future work",
|
|
399
|
+
"background", "overview",
|
|
400
|
+
})
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
def is_generic_section(title: str, depth: int) -> bool:
|
|
404
|
+
"""Check if a node is a generic overview section.
|
|
405
|
+
|
|
406
|
+
Only applies to top-level sections (depth 0-1) whose base title
|
|
407
|
+
(before ::: delimiter) is in the generic set.
|
|
408
|
+
|
|
409
|
+
Args:
|
|
410
|
+
title: node title string
|
|
411
|
+
depth: node depth in tree (0 = root)
|
|
412
|
+
|
|
413
|
+
Returns:
|
|
414
|
+
True if the node is a generic section that should be demoted
|
|
415
|
+
"""
|
|
416
|
+
if depth > 1:
|
|
417
|
+
return False
|
|
418
|
+
if not title:
|
|
419
|
+
return False
|
|
420
|
+
# For ::: delimited titles, only check the base (leftmost) part
|
|
421
|
+
base_title = title.split(" ::: ")[0].strip().lower() if " ::: " in title else title.strip().lower()
|
|
422
|
+
# Root node (depth=0, paper title) is almost never relevant
|
|
423
|
+
if depth == 0:
|
|
424
|
+
return True
|
|
425
|
+
return base_title in _GENERIC_SECTIONS
|