treesearchlib 1.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. treesearch/__init__.py +53 -0
  2. treesearch/__main__.py +6 -0
  3. treesearch/_bin/pst-extract.exe +0 -0
  4. treesearch/cli.py +554 -0
  5. treesearch/config.py +206 -0
  6. treesearch/fts.py +2293 -0
  7. treesearch/heuristics.py +425 -0
  8. treesearch/indexer.py +2038 -0
  9. treesearch/parsers/__init__.py +62 -0
  10. treesearch/parsers/anydoc_parser.py +193 -0
  11. treesearch/parsers/ast_parser.py +136 -0
  12. treesearch/parsers/docx_parser.py +304 -0
  13. treesearch/parsers/email_html_md.py +60 -0
  14. treesearch/parsers/excel_parser.py +218 -0
  15. treesearch/parsers/html_parser.py +172 -0
  16. treesearch/parsers/image_metadata.py +345 -0
  17. treesearch/parsers/image_parser.py +59 -0
  18. treesearch/parsers/image_store.py +182 -0
  19. treesearch/parsers/markitdown_parser.py +258 -0
  20. treesearch/parsers/mhtml_parser.py +108 -0
  21. treesearch/parsers/pdf_parser.py +409 -0
  22. treesearch/parsers/pst_attachment_store.py +156 -0
  23. treesearch/parsers/pst_parser.py +733 -0
  24. treesearch/parsers/registry.py +405 -0
  25. treesearch/parsers/treesitter_parser.py +433 -0
  26. treesearch/pathutil.py +227 -0
  27. treesearch/py.typed +0 -0
  28. treesearch/ripgrep.py +159 -0
  29. treesearch/search.py +935 -0
  30. treesearch/tokenizer.py +176 -0
  31. treesearch/tree.py +393 -0
  32. treesearch/tree_searcher.py +1006 -0
  33. treesearch/treesearch.py +574 -0
  34. treesearch/watch.py +305 -0
  35. treesearchlib-1.1.0.dist-info/METADATA +124 -0
  36. treesearchlib-1.1.0.dist-info/RECORD +39 -0
  37. treesearchlib-1.1.0.dist-info/WHEEL +5 -0
  38. treesearchlib-1.1.0.dist-info/entry_points.txt +2 -0
  39. treesearchlib-1.1.0.dist-info/top_level.txt +1 -0
treesearch/config.py ADDED
@@ -0,0 +1,206 @@
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ @author:XuMing(xuming624@qq.com)
4
+ @description: Unified configuration for TreeSearch.
5
+
6
+ Priority (high -> low):
7
+ 1. set_config(TreeSearchConfig(...))
8
+ 2. Environment variables
9
+ 3. Built-in defaults
10
+
11
+ Environment variables:
12
+ Tokenizer: TREESEARCH_CJK_TOKENIZER
13
+ """
14
+ import logging
15
+ import os
16
+ from dataclasses import dataclass, field
17
+ from typing import Literal, Optional
18
+
19
+ logger = logging.getLogger(__name__)
20
+
21
+ # ---------------------------------------------------------------------------
22
+ # Index schema version
23
+ # ---------------------------------------------------------------------------
24
+ # Bump whenever a change in tree builder, tokenizer, FTS schema, or node_id
25
+ # algorithm would invalidate previously-built indexes. The version is folded
26
+ # into every file's fingerprint so old index_meta entries automatically miss
27
+ # and the file is re-indexed on next run.
28
+ #
29
+ # History:
30
+ # "1" — original (mtime_ns:size) fingerprint with sequential int node_ids.
31
+ # "2" — stable hash node_ids + node-level diff + atomic per-doc transaction.
32
+ # "3" — docx parser emits \n\n between paragraphs for correct preview rendering,
33
+ # and renders tables as GFM markdown tables (header + separator + rows).
34
+ INDEX_SCHEMA_VERSION = "3"
35
+
36
+ # ---------------------------------------------------------------------------
37
+ # Environment variable names
38
+ # ---------------------------------------------------------------------------
39
+ _ENV_CJK_TOKENIZER = "TREESEARCH_CJK_TOKENIZER"
40
+ _ENV_FINGERPRINT_MODE = "TREESEARCH_FINGERPRINT_MODE"
41
+ _ENV_PRUNE = "TREESEARCH_PRUNE"
42
+ _ENV_SHADOW_MD = "TREESEARCH_ENABLE_SHADOW_MD"
43
+ _ENV_ALLOWED_SOURCE_TYPES = "TREESEARCH_ALLOWED_SOURCE_TYPES"
44
+
45
+
46
+ def _env_int(cfg: "TreeSearchConfig", attr: str, min_val: int, env_name: str) -> None:
47
+ """Read an integer env var and set it on *cfg* if present and >= *min_val*."""
48
+ raw = os.getenv(env_name)
49
+ if raw is not None:
50
+ try:
51
+ v = int(raw)
52
+ if v >= min_val:
53
+ setattr(cfg, attr, v)
54
+ except ValueError:
55
+ pass
56
+
57
+
58
+ # ---------------------------------------------------------------------------
59
+ # Configuration dataclass
60
+ # ---------------------------------------------------------------------------
61
+
62
+ @dataclass
63
+ class TreeSearchConfig:
64
+ """Single configuration class for TreeSearch.
65
+
66
+ Priority: set_config() > env vars > defaults.
67
+ """
68
+ # Search
69
+ max_nodes_per_doc: int = 5
70
+ top_k_docs: int = 3
71
+
72
+ # Index
73
+ if_add_node_summary: bool = True
74
+ if_add_doc_description: bool = False
75
+ if_add_node_text: bool = True
76
+ if_add_node_id: bool = True
77
+ if_thinning: bool = False
78
+ min_thinning_chars: int = 15000 # min chars to keep a sub-tree during thinning
79
+ summary_chars_threshold: int = 600 # nodes shorter than this use full text as summary
80
+ max_concurrency: int = field(default_factory=lambda: min(os.cpu_count() or 4, 256))
81
+ # PST 解析的并发上限。PST 重量级:每文件启 sidecar 子进程 + 加载 GB 级文件 +
82
+ # 全量附件提取,多个并发易耗尽内存/CPU/IO 触发 sidecar 崩溃(exit 0xC000013A)。
83
+ # 1 = PST 串行索引(最稳);调大可加速但会增加资源压力。
84
+ max_pst_concurrency: int = 1
85
+ max_dir_files: int = 10_000 # safety cap for directory walk
86
+
87
+ # Text length limits
88
+ max_node_chars: int = 8000 # max characters per node text when indexing into FTS5
89
+ max_result_chars: int = 32000 # max total characters of returned search result texts
90
+
91
+ # FTS
92
+ fts_db_path: str = "" # empty = same DB as tree storage (default: index.db)
93
+ fts_title_weight: float = 5.0
94
+ fts_summary_weight: float = 2.0
95
+ fts_body_weight: float = 10.0
96
+ fts_code_weight: float = 1.0
97
+ fts_front_matter_weight: float = 2.0
98
+
99
+ # Tree Search
100
+ search_mode: str = "auto" # "auto" | "flat" | "tree" | "auto" degrades to flat for code-only docs
101
+ anchor_top_k: int = 5 # max anchor nodes per document
102
+ max_anchor_per_doc: int = 3 # anchors to expand per document
103
+ max_expansions: int = 40 # max total node expansions in tree walk
104
+ max_hops: int = 3 # max depth offset from anchor
105
+ max_siblings: int = 2 # max sibling nodes to expand per step
106
+ min_frontier_score: float = 0.1 # stop if best frontier score below this
107
+ early_stop_score: float = 0.95 # stop early if a path reaches this score
108
+ path_top_k: int = 3 # top paths to return
109
+
110
+ # Tokenizer
111
+ cjk_tokenizer: str = "auto" # "auto" | "jieba" | "bigram" | "char"
112
+
113
+ # Incremental indexing
114
+ fingerprint_mode: Literal["stat", "content"] = "stat"
115
+ # "stat": fast `(mtime_ns:size)` fingerprint. Re-indexes after a `touch`.
116
+ # "content": samples first/middle/last 64KB of large files (full md5 for
117
+ # files <1MB) — robust against `touch`/CI replay; ~1ms cost on small
118
+ # files, dozens of ms on multi-GB files. Opt-in.
119
+ content_fingerprint_size_threshold: int = 1_000_000 # bytes; full md5 below this
120
+ content_fingerprint_sample_bytes: int = 64 * 1024 # head/mid/tail sample size
121
+
122
+ # Auto FTS5 maintenance: run `optimize` every N reindexed documents
123
+ # within a single build_index call. 0 disables.
124
+ auto_optimize_threshold: int = 1000
125
+
126
+ # Default prune policy when build_index sees a directory in `paths`.
127
+ # Set to False to never auto-delete orphans even on directory walks.
128
+ prune_orphans_on_directory: bool = True
129
+
130
+ # Failed files auto-skip: after N consecutive parse failures, skip the file
131
+ max_index_fail_count: int = 3
132
+
133
+ # Excel parser limits
134
+ xlsx_max_rows_per_sheet: int = 10000 # max rows to index per sheet
135
+ xlsx_max_consecutive_empty_rows: int = 100 # stop parsing after this many consecutive empty rows
136
+
137
+ # Shadow Markdown: generate a hidden .md copy for binary files (PDF, DOCX, etc.)
138
+ # so ripgrep fallback can search them. Disable to speed up indexing.
139
+ enable_shadow_md: bool = True
140
+
141
+ # Allowed source types: if non-empty, only files matching these source types
142
+ # will be indexed. Empty list means all types are allowed (backward compatible).
143
+ # Valid values: markdown, code, text, json, jsonl, csv, html, xml, pdf, doc,
144
+ # docx, pptx, excel
145
+ allowed_source_types: list[str] = field(default_factory=list)
146
+
147
+ @classmethod
148
+ def from_env(cls) -> "TreeSearchConfig":
149
+ """Create config from environment variables, falling back to defaults."""
150
+ config = cls()
151
+
152
+ env_cjk = os.getenv(_ENV_CJK_TOKENIZER)
153
+ if env_cjk:
154
+ config.cjk_tokenizer = env_cjk
155
+
156
+ env_fp = os.getenv(_ENV_FINGERPRINT_MODE)
157
+ if env_fp in ("stat", "content"):
158
+ config.fingerprint_mode = env_fp
159
+
160
+ env_prune = os.getenv(_ENV_PRUNE)
161
+ if env_prune is not None:
162
+ config.prune_orphans_on_directory = env_prune.lower() in ("1", "true", "yes")
163
+
164
+ env_shadow = os.getenv(_ENV_SHADOW_MD)
165
+ if env_shadow is not None:
166
+ config.enable_shadow_md = env_shadow.lower() in ("1", "true", "yes")
167
+
168
+ _env_int(config, "xlsx_max_rows_per_sheet", 1, "TREESEARCH_XLSX_MAX_ROWS_PER_SHEET")
169
+ _env_int(config, "xlsx_max_consecutive_empty_rows", 1, "TREESEARCH_XLSX_MAX_CONSECUTIVE_EMPTY_ROWS")
170
+
171
+ env_source_types = os.getenv(_ENV_ALLOWED_SOURCE_TYPES)
172
+ if env_source_types:
173
+ config.allowed_source_types = [
174
+ t.strip() for t in env_source_types.split(",") if t.strip()
175
+ ]
176
+
177
+ return config
178
+
179
+
180
+ # ---------------------------------------------------------------------------
181
+ # Global singleton
182
+ # ---------------------------------------------------------------------------
183
+ _default_config: Optional[TreeSearchConfig] = None
184
+
185
+
186
+ def get_config(reload: bool = False) -> TreeSearchConfig:
187
+ """Get global configuration (lazy singleton).
188
+
189
+ First call reads env vars + defaults. Subsequent calls return cached instance.
190
+ """
191
+ global _default_config
192
+ if reload or _default_config is None:
193
+ _default_config = TreeSearchConfig.from_env()
194
+ return _default_config
195
+
196
+
197
+ def set_config(config: TreeSearchConfig) -> None:
198
+ """Set global configuration (highest priority)."""
199
+ global _default_config
200
+ _default_config = config
201
+
202
+
203
+ def reset_config() -> None:
204
+ """Reset global config. Next get_config() re-initializes from env."""
205
+ global _default_config
206
+ _default_config = None