treesearchlib 1.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. treesearch/__init__.py +53 -0
  2. treesearch/__main__.py +6 -0
  3. treesearch/_bin/pst-extract.exe +0 -0
  4. treesearch/cli.py +554 -0
  5. treesearch/config.py +206 -0
  6. treesearch/fts.py +2293 -0
  7. treesearch/heuristics.py +425 -0
  8. treesearch/indexer.py +2038 -0
  9. treesearch/parsers/__init__.py +62 -0
  10. treesearch/parsers/anydoc_parser.py +193 -0
  11. treesearch/parsers/ast_parser.py +136 -0
  12. treesearch/parsers/docx_parser.py +304 -0
  13. treesearch/parsers/email_html_md.py +60 -0
  14. treesearch/parsers/excel_parser.py +218 -0
  15. treesearch/parsers/html_parser.py +172 -0
  16. treesearch/parsers/image_metadata.py +345 -0
  17. treesearch/parsers/image_parser.py +59 -0
  18. treesearch/parsers/image_store.py +182 -0
  19. treesearch/parsers/markitdown_parser.py +258 -0
  20. treesearch/parsers/mhtml_parser.py +108 -0
  21. treesearch/parsers/pdf_parser.py +409 -0
  22. treesearch/parsers/pst_attachment_store.py +156 -0
  23. treesearch/parsers/pst_parser.py +733 -0
  24. treesearch/parsers/registry.py +405 -0
  25. treesearch/parsers/treesitter_parser.py +433 -0
  26. treesearch/pathutil.py +227 -0
  27. treesearch/py.typed +0 -0
  28. treesearch/ripgrep.py +159 -0
  29. treesearch/search.py +935 -0
  30. treesearch/tokenizer.py +176 -0
  31. treesearch/tree.py +393 -0
  32. treesearch/tree_searcher.py +1006 -0
  33. treesearch/treesearch.py +574 -0
  34. treesearch/watch.py +305 -0
  35. treesearchlib-1.1.0.dist-info/METADATA +124 -0
  36. treesearchlib-1.1.0.dist-info/RECORD +39 -0
  37. treesearchlib-1.1.0.dist-info/WHEEL +5 -0
  38. treesearchlib-1.1.0.dist-info/entry_points.txt +2 -0
  39. treesearchlib-1.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,60 @@
1
+ # -*- coding: utf-8 -*-
2
+ """邮件 HTML 正文 → Markdown 转写(ADR-0005 决策 3)。
3
+
4
+ body_html 存在时优先转写为 Markdown 作为邮件文档正文(保持标题/列表/表格
5
+ 结构),搜索语料与预览同源。转换复用项目已有依赖 markitdown。
6
+
7
+ 图片一律剥除(留 alt 占位文字):
8
+ - ``cid:`` 内嵌图——本期不做 cid→附件映射重写(ADR-0005 决策 4)
9
+ - ``data:`` base64 图——避免 MB 级 base64 混进索引文本
10
+ - 远程图——预览时加载外联图会向发件方泄漏已读回执(tracking pixel)
11
+ """
12
+ import io
13
+ import logging
14
+ import re
15
+
16
+ logger = logging.getLogger(__name__)
17
+
18
+ _CONVERTER = None # MarkItDown 懒加载单例(初始化有开销,逐邮件新建太慢)
19
+
20
+
21
+ def _get_converter():
22
+ global _CONVERTER
23
+ if _CONVERTER is None:
24
+ from markitdown import MarkItDown
25
+
26
+ _CONVERTER = MarkItDown()
27
+ return _CONVERTER
28
+
29
+
30
+ # markdown 图片语法:![alt](url)
31
+ _RE_MD_IMAGE = re.compile(r"!\[([^\]]*)\]\([^)]*\)")
32
+
33
+ # 转换后残留的 HTML img 标签(markitdown 对复杂嵌套偶有泄漏)
34
+ _RE_HTML_IMAGE = re.compile(r"<img\b[^>]*>", re.IGNORECASE)
35
+
36
+ _RE_BLANK_LINES = re.compile(r"\n{3,}")
37
+
38
+
39
+ def email_html_to_md(html: str) -> str:
40
+ """把邮件 body_html 转写为 Markdown 文本。
41
+
42
+ Returns:
43
+ 转写后的 Markdown;输入为空或转换失败时返回 ""(调用方退回纯文本正文)。
44
+ """
45
+ if not html or not html.strip():
46
+ return ""
47
+ try:
48
+ result = _get_converter().convert_stream(
49
+ io.BytesIO(html.encode("utf-8", errors="replace")),
50
+ file_extension=".html",
51
+ )
52
+ md = result.text_content or ""
53
+ except Exception as e:
54
+ logger.debug("email html->md conversion failed: %s", e)
55
+ return ""
56
+
57
+ # 剥图片:md 语法留 alt 占位文字;残留 HTML img 标签整段移除
58
+ md = _RE_MD_IMAGE.sub(lambda m: m.group(1) or "", md)
59
+ md = _RE_HTML_IMAGE.sub("", md)
60
+ return _RE_BLANK_LINES.sub("\n\n", md).strip()
@@ -0,0 +1,218 @@
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ @author:XuMing(xuming624@qq.com)
4
+ @description: Excel/Spreadsheet parser for TreeSearch.
5
+
6
+ Requires optional dependency: ``pip install openpyxl``
7
+ Extracts sheets, headers, and row data from Excel files and builds tree structure.
8
+ """
9
+ import logging
10
+ import os
11
+ from itertools import chain
12
+ from typing import Optional
13
+
14
+ logger = logging.getLogger(__name__)
15
+
16
+ # Extensions supported by openpyxl
17
+ EXCEL_EXTENSIONS = frozenset({".xlsx", ".xlsm", ".xltx", ".xltm"})
18
+
19
+
20
+ def _non_empty_count(row) -> int:
21
+ """Count non-empty cells in a row tuple (None / blank 字符串视为空)."""
22
+ if not row:
23
+ return 0
24
+ return sum(1 for c in row if c is not None and str(c).strip())
25
+
26
+
27
+ def _extract_excel_data(
28
+ excel_path: str,
29
+ *,
30
+ max_rows_per_sheet: int = 10000,
31
+ max_consecutive_empty_rows: int = 100,
32
+ ) -> list[dict]:
33
+ """Extract sheet data from an Excel file.
34
+
35
+ Returns a flat node list with:
36
+ - Level 1: Sheet name
37
+ - Level 2: Header row (columns)
38
+ - Level 3: Data rows
39
+
40
+ Args:
41
+ excel_path: Path to the Excel file.
42
+ max_rows_per_sheet: Maximum number of rows to process per sheet.
43
+ max_consecutive_empty_rows: Stop parsing a sheet after this many
44
+ consecutive empty rows.
45
+ """
46
+ try:
47
+ from openpyxl import load_workbook
48
+ except ImportError:
49
+ raise ImportError(
50
+ "Excel support requires 'openpyxl'. Install with: pip install openpyxl"
51
+ )
52
+
53
+ wb = load_workbook(excel_path, read_only=True, data_only=True)
54
+ nodes = []
55
+ row_counter = 1
56
+
57
+ for sheet_name in wb.sheetnames:
58
+ ws = wb[sheet_name]
59
+ rows_iter = ws.iter_rows(values_only=True)
60
+
61
+ # Read header row first
62
+ try:
63
+ header_row = next(rows_iter)
64
+ except StopIteration:
65
+ nodes.append({
66
+ "title": sheet_name,
67
+ "level": 1,
68
+ "text": "(empty sheet)",
69
+ "line_num": row_counter,
70
+ "line_start": row_counter,
71
+ "line_end": row_counter,
72
+ })
73
+ row_counter += 1
74
+ continue
75
+
76
+ # Title-row detection: many sheets start with a merged title cell
77
+ # (e.g., "2025年通讯录" spanning all columns) above the real headers.
78
+ # If row 1 has only 1 non-empty cell while row 2 has ≥2, treat row 2
79
+ # as the real header. Strict threshold avoids false positives on
80
+ # regular tables where row 1 happens to have a blank trailing cell.
81
+ try:
82
+ second_row = next(rows_iter)
83
+ except StopIteration:
84
+ second_row = None
85
+ first_data_row = second_row
86
+ if (second_row is not None
87
+ and _non_empty_count(header_row) <= 1
88
+ and _non_empty_count(second_row) >= 2):
89
+ # row 1 was title; row 2 becomes header; first data row unknown yet
90
+ header_row = second_row
91
+ first_data_row = None
92
+
93
+ headers = [str(cell) if cell is not None else "" for cell in header_row]
94
+ header_text = f"Columns: {', '.join(h for h in headers if h)}"
95
+
96
+ # Sheet node (level 1)
97
+ sheet_start = row_counter
98
+ sheet_text_parts = [header_text]
99
+
100
+ # Collect data rows with early termination.
101
+ # If we peeked row 2 for title detection but kept row 1 as header,
102
+ # row 2 is the first data row and must be prepended back to the stream.
103
+ data_rows_iter = (
104
+ chain([first_data_row], rows_iter)
105
+ if first_data_row is not None
106
+ else rows_iter
107
+ )
108
+ data_rows_text = []
109
+ consecutive_empty = 0
110
+ rows_processed = 0
111
+ truncated = False
112
+
113
+ for row in data_rows_iter:
114
+ rows_processed += 1
115
+ if rows_processed > max_rows_per_sheet:
116
+ truncated = True
117
+ break
118
+
119
+ cells = [str(cell) if cell is not None else "" for cell in row]
120
+ is_empty = not any(c.strip() for c in cells)
121
+
122
+ if is_empty:
123
+ consecutive_empty += 1
124
+ if consecutive_empty >= max_consecutive_empty_rows:
125
+ truncated = True
126
+ break
127
+ continue
128
+
129
+ # Reset counter on non-empty row
130
+ consecutive_empty = 0
131
+
132
+ row_text = "; ".join(
133
+ f"{h}: {v}" for h, v in zip(headers, cells) if v.strip()
134
+ )
135
+ if row_text:
136
+ data_rows_text.append(row_text)
137
+
138
+ # Combine sheet content
139
+ if data_rows_text:
140
+ # Show up to 200 rows in text to avoid excessive node size
141
+ displayed_rows = data_rows_text[:200]
142
+ sheet_text_parts.extend(displayed_rows)
143
+ if len(data_rows_text) > 200:
144
+ sheet_text_parts.append(f"... ({len(data_rows_text) - 200} more rows)")
145
+
146
+ if truncated:
147
+ sheet_text_parts.append(
148
+ f"(parsing stopped: reached limit of {max_rows_per_sheet} rows "
149
+ f"or {max_consecutive_empty_rows} consecutive empty rows)"
150
+ )
151
+
152
+ sheet_text = "\n".join(sheet_text_parts)
153
+ row_count = len(data_rows_text)
154
+
155
+ nodes.append({
156
+ "title": f"{sheet_name} ({row_count} rows)",
157
+ "level": 1,
158
+ "text": sheet_text,
159
+ "line_num": sheet_start,
160
+ "line_start": sheet_start,
161
+ "line_end": sheet_start + row_count,
162
+ })
163
+ row_counter += row_count + 1
164
+
165
+ wb.close()
166
+ return nodes
167
+
168
+
169
+ async def excel_to_tree(
170
+ excel_path: str,
171
+ *,
172
+ model: Optional[str] = None,
173
+ if_add_node_summary: bool = True,
174
+ summary_chars_threshold: int = 600,
175
+ if_add_doc_description: bool = False,
176
+ if_add_node_text: bool = False,
177
+ if_add_node_id: bool = True,
178
+ **kwargs,
179
+ ) -> dict:
180
+ """Build a tree index from an Excel file.
181
+
182
+ Each worksheet becomes a level-1 node. Headers and row data
183
+ are extracted as structured text content.
184
+
185
+ Returns:
186
+ {'doc_name': str, 'structure': list, 'source_path': str}
187
+ """
188
+ from ..config import get_config
189
+
190
+ doc_name = os.path.splitext(os.path.basename(excel_path))[0]
191
+ logger.debug("Parsing Excel: %s", excel_path)
192
+
193
+ cfg = get_config()
194
+ nodes = _extract_excel_data(
195
+ excel_path,
196
+ max_rows_per_sheet=cfg.xlsx_max_rows_per_sheet,
197
+ max_consecutive_empty_rows=cfg.xlsx_max_consecutive_empty_rows,
198
+ )
199
+
200
+ if not nodes:
201
+ # Empty workbook, create a single root node
202
+ nodes = [{"title": doc_name, "level": 1, "text": "(empty workbook)",
203
+ "line_num": 1, "line_start": 1, "line_end": 1}]
204
+
205
+ from ..indexer import _build_tree, _finalize_tree
206
+
207
+ tree = _build_tree(nodes)
208
+
209
+ return _finalize_tree(
210
+ tree, doc_name,
211
+ source_path=os.path.abspath(excel_path),
212
+ source_type="excel",
213
+ if_add_node_id=if_add_node_id,
214
+ if_add_node_summary=if_add_node_summary,
215
+ summary_chars_threshold=summary_chars_threshold,
216
+ if_add_node_text=if_add_node_text,
217
+ if_add_doc_description=if_add_doc_description,
218
+ )
@@ -0,0 +1,172 @@
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ @author:XuMing(xuming624@qq.com)
4
+ @description: HTML parser for TreeSearch.
5
+
6
+ Requires optional dependency: ``pip install beautifulsoup4``
7
+ Extracts heading structure and text from HTML using BeautifulSoup.
8
+ """
9
+ import logging
10
+ import os
11
+ import re
12
+ from typing import Optional
13
+
14
+ logger = logging.getLogger(__name__)
15
+
16
+
17
+ def _extract_html_structure(html_content: str) -> tuple[list[dict], str]:
18
+ """Extract headings and body text from HTML content.
19
+
20
+ Returns:
21
+ (headings, plain_text) where headings is a list of
22
+ {'title': str, 'line_num': int, 'level': int}
23
+ """
24
+ try:
25
+ from bs4 import BeautifulSoup
26
+ except ImportError:
27
+ raise ImportError(
28
+ "HTML parser requires 'beautifulsoup4'. Install with: pip install beautifulsoup4"
29
+ )
30
+
31
+ soup = BeautifulSoup(html_content, "html.parser")
32
+
33
+ # Remove script and style elements
34
+ for tag in soup(["script", "style"]):
35
+ tag.decompose()
36
+
37
+ # Extract plain text
38
+ plain_text = soup.get_text(separator="\n")
39
+ # Collapse blank lines
40
+ plain_text = re.sub(r"\n{3,}", "\n\n", plain_text).strip()
41
+ lines = plain_text.split("\n")
42
+
43
+ # Extract headings (h1-h6)
44
+ headings = []
45
+ heading_tags = soup.find_all(re.compile(r"^h[1-6]$"))
46
+ for tag in heading_tags:
47
+ level = int(tag.name[1])
48
+ title = tag.get_text(strip=True)
49
+ if not title:
50
+ continue
51
+ # Find approximate line number by searching in plain text
52
+ line_num = 1
53
+ title_lower = title.lower().strip()
54
+ for i, line in enumerate(lines):
55
+ if title_lower in line.lower().strip():
56
+ line_num = i + 1
57
+ break
58
+ headings.append({
59
+ "title": title,
60
+ "line_num": line_num,
61
+ "level": level,
62
+ })
63
+
64
+ return headings, plain_text
65
+
66
+
67
+ async def html_to_tree(
68
+ html_path: str,
69
+ *,
70
+ model: Optional[str] = None,
71
+ if_add_node_summary: bool = True,
72
+ summary_chars_threshold: int = 600,
73
+ if_add_doc_description: bool = False,
74
+ if_add_node_text: bool = False,
75
+ if_add_node_id: bool = True,
76
+ **kwargs,
77
+ ) -> dict:
78
+ """Build a tree index from an HTML file.
79
+
80
+ Uses BeautifulSoup to extract headings (h1-h6) and text content.
81
+ Falls back to ``text_to_tree`` if no headings are found.
82
+
83
+ Returns:
84
+ {'doc_name': str, 'structure': list, 'source_path': str}
85
+ """
86
+ doc_name = os.path.splitext(os.path.basename(html_path))[0]
87
+ logger.debug("Parsing HTML: %s", html_path)
88
+
89
+ with open(html_path, "r", encoding="utf-8", errors="replace") as f:
90
+ html_content = f.read()
91
+
92
+ return await html_content_to_tree(
93
+ html_content,
94
+ doc_name=doc_name,
95
+ source_path=os.path.abspath(html_path),
96
+ model=model,
97
+ if_add_node_summary=if_add_node_summary,
98
+ summary_chars_threshold=summary_chars_threshold,
99
+ if_add_doc_description=if_add_doc_description,
100
+ if_add_node_text=if_add_node_text,
101
+ if_add_node_id=if_add_node_id,
102
+ **kwargs,
103
+ )
104
+
105
+
106
+ async def html_content_to_tree(
107
+ html_content: str,
108
+ *,
109
+ doc_name: str,
110
+ source_path: str,
111
+ model: Optional[str] = None,
112
+ if_add_node_summary: bool = True,
113
+ summary_chars_threshold: int = 600,
114
+ if_add_doc_description: bool = False,
115
+ if_add_node_text: bool = False,
116
+ if_add_node_id: bool = True,
117
+ source_type: str = "html",
118
+ **kwargs,
119
+ ) -> dict:
120
+ """Build a tree index from HTML content in memory.
121
+
122
+ Shared core of ``html_to_tree``; also used by the MHTML parser
123
+ (which unpacks the HTML part from the MIME archive first).
124
+ """
125
+ headings, plain_text = _extract_html_structure(html_content)
126
+
127
+ if not headings:
128
+ from ..indexer import text_to_tree
129
+ result = await text_to_tree(
130
+ text_content=plain_text,
131
+ model=model,
132
+ if_add_node_summary=if_add_node_summary,
133
+ summary_chars_threshold=summary_chars_threshold,
134
+ if_add_doc_description=if_add_doc_description,
135
+ if_add_node_text=if_add_node_text,
136
+ if_add_node_id=if_add_node_id,
137
+ **kwargs,
138
+ )
139
+ result["doc_name"] = doc_name
140
+ result["source_path"] = source_path
141
+ return result
142
+
143
+ # Build nodes from headings
144
+ from ..indexer import _build_tree, _finalize_tree
145
+
146
+ lines = plain_text.split("\n")
147
+ nodes = []
148
+ for i, hd in enumerate(headings):
149
+ start = hd["line_num"] - 1
150
+ end = headings[i + 1]["line_num"] - 1 if i + 1 < len(headings) else len(lines)
151
+ text = "\n".join(lines[start:end]).strip()
152
+ nodes.append({
153
+ "title": hd["title"],
154
+ "line_num": hd["line_num"],
155
+ "line_start": hd["line_num"],
156
+ "line_end": end,
157
+ "level": hd["level"],
158
+ "text": text,
159
+ })
160
+
161
+ tree = _build_tree(nodes)
162
+
163
+ return _finalize_tree(
164
+ tree, doc_name,
165
+ source_path=source_path,
166
+ source_type=source_type,
167
+ if_add_node_id=if_add_node_id,
168
+ if_add_node_summary=if_add_node_summary,
169
+ summary_chars_threshold=summary_chars_threshold,
170
+ if_add_node_text=if_add_node_text,
171
+ if_add_doc_description=if_add_doc_description,
172
+ )