treesearchlib 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- treesearch/__init__.py +53 -0
- treesearch/__main__.py +6 -0
- treesearch/_bin/pst-extract.exe +0 -0
- treesearch/cli.py +554 -0
- treesearch/config.py +206 -0
- treesearch/fts.py +2293 -0
- treesearch/heuristics.py +425 -0
- treesearch/indexer.py +2038 -0
- treesearch/parsers/__init__.py +62 -0
- treesearch/parsers/anydoc_parser.py +193 -0
- treesearch/parsers/ast_parser.py +136 -0
- treesearch/parsers/docx_parser.py +304 -0
- treesearch/parsers/email_html_md.py +60 -0
- treesearch/parsers/excel_parser.py +218 -0
- treesearch/parsers/html_parser.py +172 -0
- treesearch/parsers/image_metadata.py +345 -0
- treesearch/parsers/image_parser.py +59 -0
- treesearch/parsers/image_store.py +182 -0
- treesearch/parsers/markitdown_parser.py +258 -0
- treesearch/parsers/mhtml_parser.py +108 -0
- treesearch/parsers/pdf_parser.py +409 -0
- treesearch/parsers/pst_attachment_store.py +156 -0
- treesearch/parsers/pst_parser.py +733 -0
- treesearch/parsers/registry.py +405 -0
- treesearch/parsers/treesitter_parser.py +433 -0
- treesearch/pathutil.py +227 -0
- treesearch/py.typed +0 -0
- treesearch/ripgrep.py +159 -0
- treesearch/search.py +935 -0
- treesearch/tokenizer.py +176 -0
- treesearch/tree.py +393 -0
- treesearch/tree_searcher.py +1006 -0
- treesearch/treesearch.py +574 -0
- treesearch/watch.py +305 -0
- treesearchlib-1.1.0.dist-info/METADATA +124 -0
- treesearchlib-1.1.0.dist-info/RECORD +39 -0
- treesearchlib-1.1.0.dist-info/WHEEL +5 -0
- treesearchlib-1.1.0.dist-info/entry_points.txt +2 -0
- treesearchlib-1.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""邮件 HTML 正文 → Markdown 转写(ADR-0005 决策 3)。
|
|
3
|
+
|
|
4
|
+
body_html 存在时优先转写为 Markdown 作为邮件文档正文(保持标题/列表/表格
|
|
5
|
+
结构),搜索语料与预览同源。转换复用项目已有依赖 markitdown。
|
|
6
|
+
|
|
7
|
+
图片一律剥除(留 alt 占位文字):
|
|
8
|
+
- ``cid:`` 内嵌图——本期不做 cid→附件映射重写(ADR-0005 决策 4)
|
|
9
|
+
- ``data:`` base64 图——避免 MB 级 base64 混进索引文本
|
|
10
|
+
- 远程图——预览时加载外联图会向发件方泄漏已读回执(tracking pixel)
|
|
11
|
+
"""
|
|
12
|
+
import io
|
|
13
|
+
import logging
|
|
14
|
+
import re
|
|
15
|
+
|
|
16
|
+
logger = logging.getLogger(__name__)
|
|
17
|
+
|
|
18
|
+
_CONVERTER = None # MarkItDown 懒加载单例(初始化有开销,逐邮件新建太慢)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _get_converter():
|
|
22
|
+
global _CONVERTER
|
|
23
|
+
if _CONVERTER is None:
|
|
24
|
+
from markitdown import MarkItDown
|
|
25
|
+
|
|
26
|
+
_CONVERTER = MarkItDown()
|
|
27
|
+
return _CONVERTER
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
# markdown 图片语法:
|
|
31
|
+
_RE_MD_IMAGE = re.compile(r"!\[([^\]]*)\]\([^)]*\)")
|
|
32
|
+
|
|
33
|
+
# 转换后残留的 HTML img 标签(markitdown 对复杂嵌套偶有泄漏)
|
|
34
|
+
_RE_HTML_IMAGE = re.compile(r"<img\b[^>]*>", re.IGNORECASE)
|
|
35
|
+
|
|
36
|
+
_RE_BLANK_LINES = re.compile(r"\n{3,}")
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def email_html_to_md(html: str) -> str:
|
|
40
|
+
"""把邮件 body_html 转写为 Markdown 文本。
|
|
41
|
+
|
|
42
|
+
Returns:
|
|
43
|
+
转写后的 Markdown;输入为空或转换失败时返回 ""(调用方退回纯文本正文)。
|
|
44
|
+
"""
|
|
45
|
+
if not html or not html.strip():
|
|
46
|
+
return ""
|
|
47
|
+
try:
|
|
48
|
+
result = _get_converter().convert_stream(
|
|
49
|
+
io.BytesIO(html.encode("utf-8", errors="replace")),
|
|
50
|
+
file_extension=".html",
|
|
51
|
+
)
|
|
52
|
+
md = result.text_content or ""
|
|
53
|
+
except Exception as e:
|
|
54
|
+
logger.debug("email html->md conversion failed: %s", e)
|
|
55
|
+
return ""
|
|
56
|
+
|
|
57
|
+
# 剥图片:md 语法留 alt 占位文字;残留 HTML img 标签整段移除
|
|
58
|
+
md = _RE_MD_IMAGE.sub(lambda m: m.group(1) or "", md)
|
|
59
|
+
md = _RE_HTML_IMAGE.sub("", md)
|
|
60
|
+
return _RE_BLANK_LINES.sub("\n\n", md).strip()
|
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""
|
|
3
|
+
@author:XuMing(xuming624@qq.com)
|
|
4
|
+
@description: Excel/Spreadsheet parser for TreeSearch.
|
|
5
|
+
|
|
6
|
+
Requires optional dependency: ``pip install openpyxl``
|
|
7
|
+
Extracts sheets, headers, and row data from Excel files and builds tree structure.
|
|
8
|
+
"""
|
|
9
|
+
import logging
|
|
10
|
+
import os
|
|
11
|
+
from itertools import chain
|
|
12
|
+
from typing import Optional
|
|
13
|
+
|
|
14
|
+
logger = logging.getLogger(__name__)
|
|
15
|
+
|
|
16
|
+
# Extensions supported by openpyxl
|
|
17
|
+
EXCEL_EXTENSIONS = frozenset({".xlsx", ".xlsm", ".xltx", ".xltm"})
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _non_empty_count(row) -> int:
|
|
21
|
+
"""Count non-empty cells in a row tuple (None / blank 字符串视为空)."""
|
|
22
|
+
if not row:
|
|
23
|
+
return 0
|
|
24
|
+
return sum(1 for c in row if c is not None and str(c).strip())
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _extract_excel_data(
|
|
28
|
+
excel_path: str,
|
|
29
|
+
*,
|
|
30
|
+
max_rows_per_sheet: int = 10000,
|
|
31
|
+
max_consecutive_empty_rows: int = 100,
|
|
32
|
+
) -> list[dict]:
|
|
33
|
+
"""Extract sheet data from an Excel file.
|
|
34
|
+
|
|
35
|
+
Returns a flat node list with:
|
|
36
|
+
- Level 1: Sheet name
|
|
37
|
+
- Level 2: Header row (columns)
|
|
38
|
+
- Level 3: Data rows
|
|
39
|
+
|
|
40
|
+
Args:
|
|
41
|
+
excel_path: Path to the Excel file.
|
|
42
|
+
max_rows_per_sheet: Maximum number of rows to process per sheet.
|
|
43
|
+
max_consecutive_empty_rows: Stop parsing a sheet after this many
|
|
44
|
+
consecutive empty rows.
|
|
45
|
+
"""
|
|
46
|
+
try:
|
|
47
|
+
from openpyxl import load_workbook
|
|
48
|
+
except ImportError:
|
|
49
|
+
raise ImportError(
|
|
50
|
+
"Excel support requires 'openpyxl'. Install with: pip install openpyxl"
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
wb = load_workbook(excel_path, read_only=True, data_only=True)
|
|
54
|
+
nodes = []
|
|
55
|
+
row_counter = 1
|
|
56
|
+
|
|
57
|
+
for sheet_name in wb.sheetnames:
|
|
58
|
+
ws = wb[sheet_name]
|
|
59
|
+
rows_iter = ws.iter_rows(values_only=True)
|
|
60
|
+
|
|
61
|
+
# Read header row first
|
|
62
|
+
try:
|
|
63
|
+
header_row = next(rows_iter)
|
|
64
|
+
except StopIteration:
|
|
65
|
+
nodes.append({
|
|
66
|
+
"title": sheet_name,
|
|
67
|
+
"level": 1,
|
|
68
|
+
"text": "(empty sheet)",
|
|
69
|
+
"line_num": row_counter,
|
|
70
|
+
"line_start": row_counter,
|
|
71
|
+
"line_end": row_counter,
|
|
72
|
+
})
|
|
73
|
+
row_counter += 1
|
|
74
|
+
continue
|
|
75
|
+
|
|
76
|
+
# Title-row detection: many sheets start with a merged title cell
|
|
77
|
+
# (e.g., "2025年通讯录" spanning all columns) above the real headers.
|
|
78
|
+
# If row 1 has only 1 non-empty cell while row 2 has ≥2, treat row 2
|
|
79
|
+
# as the real header. Strict threshold avoids false positives on
|
|
80
|
+
# regular tables where row 1 happens to have a blank trailing cell.
|
|
81
|
+
try:
|
|
82
|
+
second_row = next(rows_iter)
|
|
83
|
+
except StopIteration:
|
|
84
|
+
second_row = None
|
|
85
|
+
first_data_row = second_row
|
|
86
|
+
if (second_row is not None
|
|
87
|
+
and _non_empty_count(header_row) <= 1
|
|
88
|
+
and _non_empty_count(second_row) >= 2):
|
|
89
|
+
# row 1 was title; row 2 becomes header; first data row unknown yet
|
|
90
|
+
header_row = second_row
|
|
91
|
+
first_data_row = None
|
|
92
|
+
|
|
93
|
+
headers = [str(cell) if cell is not None else "" for cell in header_row]
|
|
94
|
+
header_text = f"Columns: {', '.join(h for h in headers if h)}"
|
|
95
|
+
|
|
96
|
+
# Sheet node (level 1)
|
|
97
|
+
sheet_start = row_counter
|
|
98
|
+
sheet_text_parts = [header_text]
|
|
99
|
+
|
|
100
|
+
# Collect data rows with early termination.
|
|
101
|
+
# If we peeked row 2 for title detection but kept row 1 as header,
|
|
102
|
+
# row 2 is the first data row and must be prepended back to the stream.
|
|
103
|
+
data_rows_iter = (
|
|
104
|
+
chain([first_data_row], rows_iter)
|
|
105
|
+
if first_data_row is not None
|
|
106
|
+
else rows_iter
|
|
107
|
+
)
|
|
108
|
+
data_rows_text = []
|
|
109
|
+
consecutive_empty = 0
|
|
110
|
+
rows_processed = 0
|
|
111
|
+
truncated = False
|
|
112
|
+
|
|
113
|
+
for row in data_rows_iter:
|
|
114
|
+
rows_processed += 1
|
|
115
|
+
if rows_processed > max_rows_per_sheet:
|
|
116
|
+
truncated = True
|
|
117
|
+
break
|
|
118
|
+
|
|
119
|
+
cells = [str(cell) if cell is not None else "" for cell in row]
|
|
120
|
+
is_empty = not any(c.strip() for c in cells)
|
|
121
|
+
|
|
122
|
+
if is_empty:
|
|
123
|
+
consecutive_empty += 1
|
|
124
|
+
if consecutive_empty >= max_consecutive_empty_rows:
|
|
125
|
+
truncated = True
|
|
126
|
+
break
|
|
127
|
+
continue
|
|
128
|
+
|
|
129
|
+
# Reset counter on non-empty row
|
|
130
|
+
consecutive_empty = 0
|
|
131
|
+
|
|
132
|
+
row_text = "; ".join(
|
|
133
|
+
f"{h}: {v}" for h, v in zip(headers, cells) if v.strip()
|
|
134
|
+
)
|
|
135
|
+
if row_text:
|
|
136
|
+
data_rows_text.append(row_text)
|
|
137
|
+
|
|
138
|
+
# Combine sheet content
|
|
139
|
+
if data_rows_text:
|
|
140
|
+
# Show up to 200 rows in text to avoid excessive node size
|
|
141
|
+
displayed_rows = data_rows_text[:200]
|
|
142
|
+
sheet_text_parts.extend(displayed_rows)
|
|
143
|
+
if len(data_rows_text) > 200:
|
|
144
|
+
sheet_text_parts.append(f"... ({len(data_rows_text) - 200} more rows)")
|
|
145
|
+
|
|
146
|
+
if truncated:
|
|
147
|
+
sheet_text_parts.append(
|
|
148
|
+
f"(parsing stopped: reached limit of {max_rows_per_sheet} rows "
|
|
149
|
+
f"or {max_consecutive_empty_rows} consecutive empty rows)"
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
sheet_text = "\n".join(sheet_text_parts)
|
|
153
|
+
row_count = len(data_rows_text)
|
|
154
|
+
|
|
155
|
+
nodes.append({
|
|
156
|
+
"title": f"{sheet_name} ({row_count} rows)",
|
|
157
|
+
"level": 1,
|
|
158
|
+
"text": sheet_text,
|
|
159
|
+
"line_num": sheet_start,
|
|
160
|
+
"line_start": sheet_start,
|
|
161
|
+
"line_end": sheet_start + row_count,
|
|
162
|
+
})
|
|
163
|
+
row_counter += row_count + 1
|
|
164
|
+
|
|
165
|
+
wb.close()
|
|
166
|
+
return nodes
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
async def excel_to_tree(
|
|
170
|
+
excel_path: str,
|
|
171
|
+
*,
|
|
172
|
+
model: Optional[str] = None,
|
|
173
|
+
if_add_node_summary: bool = True,
|
|
174
|
+
summary_chars_threshold: int = 600,
|
|
175
|
+
if_add_doc_description: bool = False,
|
|
176
|
+
if_add_node_text: bool = False,
|
|
177
|
+
if_add_node_id: bool = True,
|
|
178
|
+
**kwargs,
|
|
179
|
+
) -> dict:
|
|
180
|
+
"""Build a tree index from an Excel file.
|
|
181
|
+
|
|
182
|
+
Each worksheet becomes a level-1 node. Headers and row data
|
|
183
|
+
are extracted as structured text content.
|
|
184
|
+
|
|
185
|
+
Returns:
|
|
186
|
+
{'doc_name': str, 'structure': list, 'source_path': str}
|
|
187
|
+
"""
|
|
188
|
+
from ..config import get_config
|
|
189
|
+
|
|
190
|
+
doc_name = os.path.splitext(os.path.basename(excel_path))[0]
|
|
191
|
+
logger.debug("Parsing Excel: %s", excel_path)
|
|
192
|
+
|
|
193
|
+
cfg = get_config()
|
|
194
|
+
nodes = _extract_excel_data(
|
|
195
|
+
excel_path,
|
|
196
|
+
max_rows_per_sheet=cfg.xlsx_max_rows_per_sheet,
|
|
197
|
+
max_consecutive_empty_rows=cfg.xlsx_max_consecutive_empty_rows,
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
if not nodes:
|
|
201
|
+
# Empty workbook, create a single root node
|
|
202
|
+
nodes = [{"title": doc_name, "level": 1, "text": "(empty workbook)",
|
|
203
|
+
"line_num": 1, "line_start": 1, "line_end": 1}]
|
|
204
|
+
|
|
205
|
+
from ..indexer import _build_tree, _finalize_tree
|
|
206
|
+
|
|
207
|
+
tree = _build_tree(nodes)
|
|
208
|
+
|
|
209
|
+
return _finalize_tree(
|
|
210
|
+
tree, doc_name,
|
|
211
|
+
source_path=os.path.abspath(excel_path),
|
|
212
|
+
source_type="excel",
|
|
213
|
+
if_add_node_id=if_add_node_id,
|
|
214
|
+
if_add_node_summary=if_add_node_summary,
|
|
215
|
+
summary_chars_threshold=summary_chars_threshold,
|
|
216
|
+
if_add_node_text=if_add_node_text,
|
|
217
|
+
if_add_doc_description=if_add_doc_description,
|
|
218
|
+
)
|
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""
|
|
3
|
+
@author:XuMing(xuming624@qq.com)
|
|
4
|
+
@description: HTML parser for TreeSearch.
|
|
5
|
+
|
|
6
|
+
Requires optional dependency: ``pip install beautifulsoup4``
|
|
7
|
+
Extracts heading structure and text from HTML using BeautifulSoup.
|
|
8
|
+
"""
|
|
9
|
+
import logging
|
|
10
|
+
import os
|
|
11
|
+
import re
|
|
12
|
+
from typing import Optional
|
|
13
|
+
|
|
14
|
+
logger = logging.getLogger(__name__)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _extract_html_structure(html_content: str) -> tuple[list[dict], str]:
|
|
18
|
+
"""Extract headings and body text from HTML content.
|
|
19
|
+
|
|
20
|
+
Returns:
|
|
21
|
+
(headings, plain_text) where headings is a list of
|
|
22
|
+
{'title': str, 'line_num': int, 'level': int}
|
|
23
|
+
"""
|
|
24
|
+
try:
|
|
25
|
+
from bs4 import BeautifulSoup
|
|
26
|
+
except ImportError:
|
|
27
|
+
raise ImportError(
|
|
28
|
+
"HTML parser requires 'beautifulsoup4'. Install with: pip install beautifulsoup4"
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
soup = BeautifulSoup(html_content, "html.parser")
|
|
32
|
+
|
|
33
|
+
# Remove script and style elements
|
|
34
|
+
for tag in soup(["script", "style"]):
|
|
35
|
+
tag.decompose()
|
|
36
|
+
|
|
37
|
+
# Extract plain text
|
|
38
|
+
plain_text = soup.get_text(separator="\n")
|
|
39
|
+
# Collapse blank lines
|
|
40
|
+
plain_text = re.sub(r"\n{3,}", "\n\n", plain_text).strip()
|
|
41
|
+
lines = plain_text.split("\n")
|
|
42
|
+
|
|
43
|
+
# Extract headings (h1-h6)
|
|
44
|
+
headings = []
|
|
45
|
+
heading_tags = soup.find_all(re.compile(r"^h[1-6]$"))
|
|
46
|
+
for tag in heading_tags:
|
|
47
|
+
level = int(tag.name[1])
|
|
48
|
+
title = tag.get_text(strip=True)
|
|
49
|
+
if not title:
|
|
50
|
+
continue
|
|
51
|
+
# Find approximate line number by searching in plain text
|
|
52
|
+
line_num = 1
|
|
53
|
+
title_lower = title.lower().strip()
|
|
54
|
+
for i, line in enumerate(lines):
|
|
55
|
+
if title_lower in line.lower().strip():
|
|
56
|
+
line_num = i + 1
|
|
57
|
+
break
|
|
58
|
+
headings.append({
|
|
59
|
+
"title": title,
|
|
60
|
+
"line_num": line_num,
|
|
61
|
+
"level": level,
|
|
62
|
+
})
|
|
63
|
+
|
|
64
|
+
return headings, plain_text
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
async def html_to_tree(
|
|
68
|
+
html_path: str,
|
|
69
|
+
*,
|
|
70
|
+
model: Optional[str] = None,
|
|
71
|
+
if_add_node_summary: bool = True,
|
|
72
|
+
summary_chars_threshold: int = 600,
|
|
73
|
+
if_add_doc_description: bool = False,
|
|
74
|
+
if_add_node_text: bool = False,
|
|
75
|
+
if_add_node_id: bool = True,
|
|
76
|
+
**kwargs,
|
|
77
|
+
) -> dict:
|
|
78
|
+
"""Build a tree index from an HTML file.
|
|
79
|
+
|
|
80
|
+
Uses BeautifulSoup to extract headings (h1-h6) and text content.
|
|
81
|
+
Falls back to ``text_to_tree`` if no headings are found.
|
|
82
|
+
|
|
83
|
+
Returns:
|
|
84
|
+
{'doc_name': str, 'structure': list, 'source_path': str}
|
|
85
|
+
"""
|
|
86
|
+
doc_name = os.path.splitext(os.path.basename(html_path))[0]
|
|
87
|
+
logger.debug("Parsing HTML: %s", html_path)
|
|
88
|
+
|
|
89
|
+
with open(html_path, "r", encoding="utf-8", errors="replace") as f:
|
|
90
|
+
html_content = f.read()
|
|
91
|
+
|
|
92
|
+
return await html_content_to_tree(
|
|
93
|
+
html_content,
|
|
94
|
+
doc_name=doc_name,
|
|
95
|
+
source_path=os.path.abspath(html_path),
|
|
96
|
+
model=model,
|
|
97
|
+
if_add_node_summary=if_add_node_summary,
|
|
98
|
+
summary_chars_threshold=summary_chars_threshold,
|
|
99
|
+
if_add_doc_description=if_add_doc_description,
|
|
100
|
+
if_add_node_text=if_add_node_text,
|
|
101
|
+
if_add_node_id=if_add_node_id,
|
|
102
|
+
**kwargs,
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
async def html_content_to_tree(
|
|
107
|
+
html_content: str,
|
|
108
|
+
*,
|
|
109
|
+
doc_name: str,
|
|
110
|
+
source_path: str,
|
|
111
|
+
model: Optional[str] = None,
|
|
112
|
+
if_add_node_summary: bool = True,
|
|
113
|
+
summary_chars_threshold: int = 600,
|
|
114
|
+
if_add_doc_description: bool = False,
|
|
115
|
+
if_add_node_text: bool = False,
|
|
116
|
+
if_add_node_id: bool = True,
|
|
117
|
+
source_type: str = "html",
|
|
118
|
+
**kwargs,
|
|
119
|
+
) -> dict:
|
|
120
|
+
"""Build a tree index from HTML content in memory.
|
|
121
|
+
|
|
122
|
+
Shared core of ``html_to_tree``; also used by the MHTML parser
|
|
123
|
+
(which unpacks the HTML part from the MIME archive first).
|
|
124
|
+
"""
|
|
125
|
+
headings, plain_text = _extract_html_structure(html_content)
|
|
126
|
+
|
|
127
|
+
if not headings:
|
|
128
|
+
from ..indexer import text_to_tree
|
|
129
|
+
result = await text_to_tree(
|
|
130
|
+
text_content=plain_text,
|
|
131
|
+
model=model,
|
|
132
|
+
if_add_node_summary=if_add_node_summary,
|
|
133
|
+
summary_chars_threshold=summary_chars_threshold,
|
|
134
|
+
if_add_doc_description=if_add_doc_description,
|
|
135
|
+
if_add_node_text=if_add_node_text,
|
|
136
|
+
if_add_node_id=if_add_node_id,
|
|
137
|
+
**kwargs,
|
|
138
|
+
)
|
|
139
|
+
result["doc_name"] = doc_name
|
|
140
|
+
result["source_path"] = source_path
|
|
141
|
+
return result
|
|
142
|
+
|
|
143
|
+
# Build nodes from headings
|
|
144
|
+
from ..indexer import _build_tree, _finalize_tree
|
|
145
|
+
|
|
146
|
+
lines = plain_text.split("\n")
|
|
147
|
+
nodes = []
|
|
148
|
+
for i, hd in enumerate(headings):
|
|
149
|
+
start = hd["line_num"] - 1
|
|
150
|
+
end = headings[i + 1]["line_num"] - 1 if i + 1 < len(headings) else len(lines)
|
|
151
|
+
text = "\n".join(lines[start:end]).strip()
|
|
152
|
+
nodes.append({
|
|
153
|
+
"title": hd["title"],
|
|
154
|
+
"line_num": hd["line_num"],
|
|
155
|
+
"line_start": hd["line_num"],
|
|
156
|
+
"line_end": end,
|
|
157
|
+
"level": hd["level"],
|
|
158
|
+
"text": text,
|
|
159
|
+
})
|
|
160
|
+
|
|
161
|
+
tree = _build_tree(nodes)
|
|
162
|
+
|
|
163
|
+
return _finalize_tree(
|
|
164
|
+
tree, doc_name,
|
|
165
|
+
source_path=source_path,
|
|
166
|
+
source_type=source_type,
|
|
167
|
+
if_add_node_id=if_add_node_id,
|
|
168
|
+
if_add_node_summary=if_add_node_summary,
|
|
169
|
+
summary_chars_threshold=summary_chars_threshold,
|
|
170
|
+
if_add_node_text=if_add_node_text,
|
|
171
|
+
if_add_doc_description=if_add_doc_description,
|
|
172
|
+
)
|