treesearchlib 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- treesearch/__init__.py +53 -0
- treesearch/__main__.py +6 -0
- treesearch/_bin/pst-extract.exe +0 -0
- treesearch/cli.py +554 -0
- treesearch/config.py +206 -0
- treesearch/fts.py +2293 -0
- treesearch/heuristics.py +425 -0
- treesearch/indexer.py +2038 -0
- treesearch/parsers/__init__.py +62 -0
- treesearch/parsers/anydoc_parser.py +193 -0
- treesearch/parsers/ast_parser.py +136 -0
- treesearch/parsers/docx_parser.py +304 -0
- treesearch/parsers/email_html_md.py +60 -0
- treesearch/parsers/excel_parser.py +218 -0
- treesearch/parsers/html_parser.py +172 -0
- treesearch/parsers/image_metadata.py +345 -0
- treesearch/parsers/image_parser.py +59 -0
- treesearch/parsers/image_store.py +182 -0
- treesearch/parsers/markitdown_parser.py +258 -0
- treesearch/parsers/mhtml_parser.py +108 -0
- treesearch/parsers/pdf_parser.py +409 -0
- treesearch/parsers/pst_attachment_store.py +156 -0
- treesearch/parsers/pst_parser.py +733 -0
- treesearch/parsers/registry.py +405 -0
- treesearch/parsers/treesitter_parser.py +433 -0
- treesearch/pathutil.py +227 -0
- treesearch/py.typed +0 -0
- treesearch/ripgrep.py +159 -0
- treesearch/search.py +935 -0
- treesearch/tokenizer.py +176 -0
- treesearch/tree.py +393 -0
- treesearch/tree_searcher.py +1006 -0
- treesearch/treesearch.py +574 -0
- treesearch/watch.py +305 -0
- treesearchlib-1.1.0.dist-info/METADATA +124 -0
- treesearchlib-1.1.0.dist-info/RECORD +39 -0
- treesearchlib-1.1.0.dist-info/WHEEL +5 -0
- treesearchlib-1.1.0.dist-info/entry_points.txt +2 -0
- treesearchlib-1.1.0.dist-info/top_level.txt +1 -0
treesearch/config.py
ADDED
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""
|
|
3
|
+
@author:XuMing(xuming624@qq.com)
|
|
4
|
+
@description: Unified configuration for TreeSearch.
|
|
5
|
+
|
|
6
|
+
Priority (high -> low):
|
|
7
|
+
1. set_config(TreeSearchConfig(...))
|
|
8
|
+
2. Environment variables
|
|
9
|
+
3. Built-in defaults
|
|
10
|
+
|
|
11
|
+
Environment variables:
|
|
12
|
+
Tokenizer: TREESEARCH_CJK_TOKENIZER
|
|
13
|
+
"""
|
|
14
|
+
import logging
|
|
15
|
+
import os
|
|
16
|
+
from dataclasses import dataclass, field
|
|
17
|
+
from typing import Literal, Optional
|
|
18
|
+
|
|
19
|
+
logger = logging.getLogger(__name__)
|
|
20
|
+
|
|
21
|
+
# ---------------------------------------------------------------------------
|
|
22
|
+
# Index schema version
|
|
23
|
+
# ---------------------------------------------------------------------------
|
|
24
|
+
# Bump whenever a change in tree builder, tokenizer, FTS schema, or node_id
|
|
25
|
+
# algorithm would invalidate previously-built indexes. The version is folded
|
|
26
|
+
# into every file's fingerprint so old index_meta entries automatically miss
|
|
27
|
+
# and the file is re-indexed on next run.
|
|
28
|
+
#
|
|
29
|
+
# History:
|
|
30
|
+
# "1" — original (mtime_ns:size) fingerprint with sequential int node_ids.
|
|
31
|
+
# "2" — stable hash node_ids + node-level diff + atomic per-doc transaction.
|
|
32
|
+
# "3" — docx parser emits \n\n between paragraphs for correct preview rendering,
|
|
33
|
+
# and renders tables as GFM markdown tables (header + separator + rows).
|
|
34
|
+
INDEX_SCHEMA_VERSION = "3"
|
|
35
|
+
|
|
36
|
+
# ---------------------------------------------------------------------------
|
|
37
|
+
# Environment variable names
|
|
38
|
+
# ---------------------------------------------------------------------------
|
|
39
|
+
_ENV_CJK_TOKENIZER = "TREESEARCH_CJK_TOKENIZER"
|
|
40
|
+
_ENV_FINGERPRINT_MODE = "TREESEARCH_FINGERPRINT_MODE"
|
|
41
|
+
_ENV_PRUNE = "TREESEARCH_PRUNE"
|
|
42
|
+
_ENV_SHADOW_MD = "TREESEARCH_ENABLE_SHADOW_MD"
|
|
43
|
+
_ENV_ALLOWED_SOURCE_TYPES = "TREESEARCH_ALLOWED_SOURCE_TYPES"
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _env_int(cfg: "TreeSearchConfig", attr: str, min_val: int, env_name: str) -> None:
|
|
47
|
+
"""Read an integer env var and set it on *cfg* if present and >= *min_val*."""
|
|
48
|
+
raw = os.getenv(env_name)
|
|
49
|
+
if raw is not None:
|
|
50
|
+
try:
|
|
51
|
+
v = int(raw)
|
|
52
|
+
if v >= min_val:
|
|
53
|
+
setattr(cfg, attr, v)
|
|
54
|
+
except ValueError:
|
|
55
|
+
pass
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
# ---------------------------------------------------------------------------
|
|
59
|
+
# Configuration dataclass
|
|
60
|
+
# ---------------------------------------------------------------------------
|
|
61
|
+
|
|
62
|
+
@dataclass
|
|
63
|
+
class TreeSearchConfig:
|
|
64
|
+
"""Single configuration class for TreeSearch.
|
|
65
|
+
|
|
66
|
+
Priority: set_config() > env vars > defaults.
|
|
67
|
+
"""
|
|
68
|
+
# Search
|
|
69
|
+
max_nodes_per_doc: int = 5
|
|
70
|
+
top_k_docs: int = 3
|
|
71
|
+
|
|
72
|
+
# Index
|
|
73
|
+
if_add_node_summary: bool = True
|
|
74
|
+
if_add_doc_description: bool = False
|
|
75
|
+
if_add_node_text: bool = True
|
|
76
|
+
if_add_node_id: bool = True
|
|
77
|
+
if_thinning: bool = False
|
|
78
|
+
min_thinning_chars: int = 15000 # min chars to keep a sub-tree during thinning
|
|
79
|
+
summary_chars_threshold: int = 600 # nodes shorter than this use full text as summary
|
|
80
|
+
max_concurrency: int = field(default_factory=lambda: min(os.cpu_count() or 4, 256))
|
|
81
|
+
# PST 解析的并发上限。PST 重量级:每文件启 sidecar 子进程 + 加载 GB 级文件 +
|
|
82
|
+
# 全量附件提取,多个并发易耗尽内存/CPU/IO 触发 sidecar 崩溃(exit 0xC000013A)。
|
|
83
|
+
# 1 = PST 串行索引(最稳);调大可加速但会增加资源压力。
|
|
84
|
+
max_pst_concurrency: int = 1
|
|
85
|
+
max_dir_files: int = 10_000 # safety cap for directory walk
|
|
86
|
+
|
|
87
|
+
# Text length limits
|
|
88
|
+
max_node_chars: int = 8000 # max characters per node text when indexing into FTS5
|
|
89
|
+
max_result_chars: int = 32000 # max total characters of returned search result texts
|
|
90
|
+
|
|
91
|
+
# FTS
|
|
92
|
+
fts_db_path: str = "" # empty = same DB as tree storage (default: index.db)
|
|
93
|
+
fts_title_weight: float = 5.0
|
|
94
|
+
fts_summary_weight: float = 2.0
|
|
95
|
+
fts_body_weight: float = 10.0
|
|
96
|
+
fts_code_weight: float = 1.0
|
|
97
|
+
fts_front_matter_weight: float = 2.0
|
|
98
|
+
|
|
99
|
+
# Tree Search
|
|
100
|
+
search_mode: str = "auto" # "auto" | "flat" | "tree" | "auto" degrades to flat for code-only docs
|
|
101
|
+
anchor_top_k: int = 5 # max anchor nodes per document
|
|
102
|
+
max_anchor_per_doc: int = 3 # anchors to expand per document
|
|
103
|
+
max_expansions: int = 40 # max total node expansions in tree walk
|
|
104
|
+
max_hops: int = 3 # max depth offset from anchor
|
|
105
|
+
max_siblings: int = 2 # max sibling nodes to expand per step
|
|
106
|
+
min_frontier_score: float = 0.1 # stop if best frontier score below this
|
|
107
|
+
early_stop_score: float = 0.95 # stop early if a path reaches this score
|
|
108
|
+
path_top_k: int = 3 # top paths to return
|
|
109
|
+
|
|
110
|
+
# Tokenizer
|
|
111
|
+
cjk_tokenizer: str = "auto" # "auto" | "jieba" | "bigram" | "char"
|
|
112
|
+
|
|
113
|
+
# Incremental indexing
|
|
114
|
+
fingerprint_mode: Literal["stat", "content"] = "stat"
|
|
115
|
+
# "stat": fast `(mtime_ns:size)` fingerprint. Re-indexes after a `touch`.
|
|
116
|
+
# "content": samples first/middle/last 64KB of large files (full md5 for
|
|
117
|
+
# files <1MB) — robust against `touch`/CI replay; ~1ms cost on small
|
|
118
|
+
# files, dozens of ms on multi-GB files. Opt-in.
|
|
119
|
+
content_fingerprint_size_threshold: int = 1_000_000 # bytes; full md5 below this
|
|
120
|
+
content_fingerprint_sample_bytes: int = 64 * 1024 # head/mid/tail sample size
|
|
121
|
+
|
|
122
|
+
# Auto FTS5 maintenance: run `optimize` every N reindexed documents
|
|
123
|
+
# within a single build_index call. 0 disables.
|
|
124
|
+
auto_optimize_threshold: int = 1000
|
|
125
|
+
|
|
126
|
+
# Default prune policy when build_index sees a directory in `paths`.
|
|
127
|
+
# Set to False to never auto-delete orphans even on directory walks.
|
|
128
|
+
prune_orphans_on_directory: bool = True
|
|
129
|
+
|
|
130
|
+
# Failed files auto-skip: after N consecutive parse failures, skip the file
|
|
131
|
+
max_index_fail_count: int = 3
|
|
132
|
+
|
|
133
|
+
# Excel parser limits
|
|
134
|
+
xlsx_max_rows_per_sheet: int = 10000 # max rows to index per sheet
|
|
135
|
+
xlsx_max_consecutive_empty_rows: int = 100 # stop parsing after this many consecutive empty rows
|
|
136
|
+
|
|
137
|
+
# Shadow Markdown: generate a hidden .md copy for binary files (PDF, DOCX, etc.)
|
|
138
|
+
# so ripgrep fallback can search them. Disable to speed up indexing.
|
|
139
|
+
enable_shadow_md: bool = True
|
|
140
|
+
|
|
141
|
+
# Allowed source types: if non-empty, only files matching these source types
|
|
142
|
+
# will be indexed. Empty list means all types are allowed (backward compatible).
|
|
143
|
+
# Valid values: markdown, code, text, json, jsonl, csv, html, xml, pdf, doc,
|
|
144
|
+
# docx, pptx, excel
|
|
145
|
+
allowed_source_types: list[str] = field(default_factory=list)
|
|
146
|
+
|
|
147
|
+
@classmethod
|
|
148
|
+
def from_env(cls) -> "TreeSearchConfig":
|
|
149
|
+
"""Create config from environment variables, falling back to defaults."""
|
|
150
|
+
config = cls()
|
|
151
|
+
|
|
152
|
+
env_cjk = os.getenv(_ENV_CJK_TOKENIZER)
|
|
153
|
+
if env_cjk:
|
|
154
|
+
config.cjk_tokenizer = env_cjk
|
|
155
|
+
|
|
156
|
+
env_fp = os.getenv(_ENV_FINGERPRINT_MODE)
|
|
157
|
+
if env_fp in ("stat", "content"):
|
|
158
|
+
config.fingerprint_mode = env_fp
|
|
159
|
+
|
|
160
|
+
env_prune = os.getenv(_ENV_PRUNE)
|
|
161
|
+
if env_prune is not None:
|
|
162
|
+
config.prune_orphans_on_directory = env_prune.lower() in ("1", "true", "yes")
|
|
163
|
+
|
|
164
|
+
env_shadow = os.getenv(_ENV_SHADOW_MD)
|
|
165
|
+
if env_shadow is not None:
|
|
166
|
+
config.enable_shadow_md = env_shadow.lower() in ("1", "true", "yes")
|
|
167
|
+
|
|
168
|
+
_env_int(config, "xlsx_max_rows_per_sheet", 1, "TREESEARCH_XLSX_MAX_ROWS_PER_SHEET")
|
|
169
|
+
_env_int(config, "xlsx_max_consecutive_empty_rows", 1, "TREESEARCH_XLSX_MAX_CONSECUTIVE_EMPTY_ROWS")
|
|
170
|
+
|
|
171
|
+
env_source_types = os.getenv(_ENV_ALLOWED_SOURCE_TYPES)
|
|
172
|
+
if env_source_types:
|
|
173
|
+
config.allowed_source_types = [
|
|
174
|
+
t.strip() for t in env_source_types.split(",") if t.strip()
|
|
175
|
+
]
|
|
176
|
+
|
|
177
|
+
return config
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
# ---------------------------------------------------------------------------
|
|
181
|
+
# Global singleton
|
|
182
|
+
# ---------------------------------------------------------------------------
|
|
183
|
+
_default_config: Optional[TreeSearchConfig] = None
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def get_config(reload: bool = False) -> TreeSearchConfig:
|
|
187
|
+
"""Get global configuration (lazy singleton).
|
|
188
|
+
|
|
189
|
+
First call reads env vars + defaults. Subsequent calls return cached instance.
|
|
190
|
+
"""
|
|
191
|
+
global _default_config
|
|
192
|
+
if reload or _default_config is None:
|
|
193
|
+
_default_config = TreeSearchConfig.from_env()
|
|
194
|
+
return _default_config
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def set_config(config: TreeSearchConfig) -> None:
|
|
198
|
+
"""Set global configuration (highest priority)."""
|
|
199
|
+
global _default_config
|
|
200
|
+
_default_config = config
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def reset_config() -> None:
|
|
204
|
+
"""Reset global config. Next get_config() re-initializes from env."""
|
|
205
|
+
global _default_config
|
|
206
|
+
_default_config = None
|