cli-tools-kit 0.6.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,268 @@
1
+ """Build one text document per installable tool.
2
+
3
+ A tool's document is its CLAUDE.md plus its README.md. When it has neither,
4
+ the tool is asked for its --advertise metadata and the name/desc from that is
5
+ used instead.
6
+
7
+ :func:`corpus_fingerprint` hashes those same inputs without running anything,
8
+ so a caller can tell whether a stored grouping is still current.
9
+ """
10
+
11
+ import hashlib
12
+ import json
13
+ import os
14
+ import subprocess
15
+ import sys
16
+ from typing import Dict, List, Tuple
17
+
18
+ MAX_CHARS = 6000
19
+ BLURB_MAX_CHARS = 700
20
+ BLURB_PARAGRAPHS = 2
21
+ BLURB_HEADINGS = 8
22
+ DOC_FILES = ("CLAUDE.md", "README.md")
23
+ ADVERTISE_TIMEOUT = 5
24
+ SKIP_PREFIXES = ("_", ".")
25
+ SKIP_NAMES = {"dev"}
26
+
27
+
28
+ def tool_dirs(root: str) -> List[str]:
29
+ """Names of the top-level directories that are installable tools."""
30
+ found = []
31
+ for name in sorted(os.listdir(root)):
32
+ if name.startswith(SKIP_PREFIXES) or name in SKIP_NAMES:
33
+ continue
34
+ path = os.path.join(root, name)
35
+ if not os.path.isdir(path):
36
+ continue
37
+ if os.path.isfile(os.path.join(path, "main.py")) and os.path.isfile(
38
+ os.path.join(path, "requirements.txt")
39
+ ):
40
+ found.append(name)
41
+ return found
42
+
43
+
44
+ def _advertise_entries(tool_path: str) -> List[dict]:
45
+ """Run the tool's --advertise and return its metadata entries."""
46
+ main_py = os.path.join(tool_path, "main.py")
47
+ try:
48
+ proc = subprocess.run(
49
+ [sys.executable, main_py, "--advertise"],
50
+ capture_output=True,
51
+ text=True,
52
+ timeout=ADVERTISE_TIMEOUT,
53
+ cwd=tool_path,
54
+ )
55
+ except (subprocess.SubprocessError, OSError):
56
+ return []
57
+
58
+ for line in reversed((proc.stdout or "").strip().splitlines()):
59
+ line = line.strip()
60
+ if not line.startswith(("[", "{")):
61
+ continue
62
+ try:
63
+ data = json.loads(line)
64
+ except json.JSONDecodeError:
65
+ continue
66
+ entries = data if isinstance(data, list) else [data]
67
+ return [e for e in entries if isinstance(e, dict)]
68
+ return []
69
+
70
+
71
+ def _advertise_text(tool_path: str) -> str:
72
+ """The tool's advertised names, descriptions and taxonomy words."""
73
+ parts = []
74
+ for entry in _advertise_entries(tool_path):
75
+ for key in ("name", "desc", "capability", "domain"):
76
+ value = entry.get(key)
77
+ if value:
78
+ parts.append(str(value))
79
+ return "\n".join(parts)
80
+
81
+
82
+ def tool_capabilities(root: str) -> Dict[str, str]:
83
+ """Map each tool to its advertised capability word, "" when it has none.
84
+
85
+ The capability is hand-written, validated against CAPABILITY_VOCAB, and
86
+ needs no network to read — which is what makes it usable as the last-resort
87
+ grouping on a host with no API key.
88
+ """
89
+ caps = {}
90
+ for name in tool_dirs(root):
91
+ capability = ""
92
+ for entry in _advertise_entries(os.path.join(root, name)):
93
+ value = str(entry.get("capability") or "").strip()
94
+ if value:
95
+ capability = value
96
+ break
97
+ caps[name] = capability
98
+ return caps
99
+
100
+
101
+ def tool_documents(root: str) -> Dict[str, str]:
102
+ """Map each tool directory name to its document text."""
103
+ docs = {}
104
+ for name in tool_dirs(root):
105
+ path = os.path.join(root, name)
106
+ chunks = [name.replace("_", " ")]
107
+ for doc_name in DOC_FILES:
108
+ doc_path = os.path.join(path, doc_name)
109
+ if os.path.isfile(doc_path):
110
+ try:
111
+ with open(doc_path, encoding="utf-8", errors="replace") as fh:
112
+ chunks.append(fh.read())
113
+ except OSError:
114
+ pass
115
+ if len(chunks) == 1:
116
+ advertised = _advertise_text(path)
117
+ if advertised:
118
+ chunks.append(advertised)
119
+ docs[name] = "\n\n".join(chunks)[:MAX_CHARS]
120
+ return docs
121
+
122
+
123
+ def _file_digest(path: str) -> str:
124
+ """sha256 of a file's bytes, "" when it cannot be read."""
125
+ try:
126
+ with open(path, "rb") as fh:
127
+ return hashlib.sha256(fh.read()).hexdigest()
128
+ except OSError:
129
+ return ""
130
+
131
+
132
+ def corpus_fingerprint(root: str) -> str:
133
+ """Hash the inputs :func:`tool_documents` reads, without running them.
134
+
135
+ Covers the set of tool directories and the contents of every CLAUDE.md and
136
+ README.md. A tool with neither is represented by its main.py instead, since
137
+ that is what the --advertise fallback reads. Content hashes rather than
138
+ mtimes: a Syncthing checkout restamps mtimes fleet-wide without changing a
139
+ byte.
140
+ """
141
+ digest = hashlib.sha256()
142
+ for name in tool_dirs(root):
143
+ path = os.path.join(root, name)
144
+ digest.update(name.encode())
145
+ documented = False
146
+ for doc_name in DOC_FILES:
147
+ doc_path = os.path.join(path, doc_name)
148
+ if os.path.isfile(doc_path):
149
+ documented = True
150
+ digest.update(doc_name.encode())
151
+ digest.update(_file_digest(doc_path).encode())
152
+ if not documented:
153
+ digest.update(b"main.py")
154
+ digest.update(_file_digest(os.path.join(path, "main.py")).encode())
155
+ return digest.hexdigest()
156
+
157
+
158
+ # --- condensed blurbs -----------------------------------------------------
159
+ #
160
+ # tool_documents() feeds an embedder, which tolerates bulk. The LLM grouping
161
+ # wants the opposite: a short, comparable description of what each tool is for.
162
+ # A tool's CLAUDE.md is mostly operational prose — commit policy, install
163
+ # mechanics, fleet paths — that says nothing about its subject, and the six
164
+ # tools with the longest docs would otherwise drown out the fourteen with none.
165
+
166
+ _SKIP_LINE_PREFIXES = ("|", ">", "<!--", "<", "!", "```", "---", "===", "*Note")
167
+ _DOC_STOP_HEADINGS = {
168
+ "installation", "install", "usage", "requirements", "dependencies",
169
+ "license", "licence", "testing", "tests", "development", "changelog",
170
+ "troubleshooting", "configuration", "config", "files", "layout",
171
+ "architecture overview", "environment variables", "cli protocol",
172
+ }
173
+
174
+
175
+ def _clean_markdown(text: str) -> Tuple[List[str], List[str]]:
176
+ """Split a markdown doc into (prose paragraphs, heading names).
177
+
178
+ Code fences, tables, block quotes, badges and raw HTML are dropped: they
179
+ carry syntax, not subject.
180
+ """
181
+ paragraphs: List[str] = []
182
+ headings: List[str] = []
183
+ buffer: List[str] = []
184
+ in_fence = False
185
+
186
+ def flush():
187
+ if buffer:
188
+ para = " ".join(buffer).strip()
189
+ # A paragraph of mostly punctuation or paths is not a description.
190
+ if len(para) >= 40 and sum(c.isalpha() for c in para) > len(para) / 2:
191
+ paragraphs.append(para)
192
+ buffer.clear()
193
+
194
+ for raw in text.splitlines():
195
+ line = raw.strip()
196
+ if line.startswith("```"):
197
+ in_fence = not in_fence
198
+ flush()
199
+ continue
200
+ if in_fence:
201
+ continue
202
+ if not line:
203
+ flush()
204
+ continue
205
+ if line.startswith("#"):
206
+ flush()
207
+ name = line.lstrip("#").strip().strip("*_`")
208
+ if name and name.lower() not in _DOC_STOP_HEADINGS:
209
+ headings.append(name)
210
+ continue
211
+ if line.startswith(_SKIP_LINE_PREFIXES):
212
+ flush()
213
+ continue
214
+ buffer.append(line.lstrip("-*+ ").strip())
215
+ flush()
216
+ return paragraphs, headings
217
+
218
+
219
+ def _advertise_fields(tool_path: str) -> List[str]:
220
+ """Advertised name/desc/capability/domain lines, deduplicated in order."""
221
+ text = _advertise_text(tool_path)
222
+ seen = set()
223
+ fields = []
224
+ for line in text.splitlines():
225
+ line = line.strip()
226
+ if line and line.lower() not in seen:
227
+ seen.add(line.lower())
228
+ fields.append(line)
229
+ return fields
230
+
231
+
232
+ def tool_blurb(root: str, name: str) -> str:
233
+ """A short, comparable description of one tool, for the LLM grouping.
234
+
235
+ Built from, in order: the advertised name/desc/capability/domain, the first
236
+ couple of real prose paragraphs of CLAUDE.md or README.md, and that doc's
237
+ section headings (which name features in very few tokens). Capped at
238
+ BLURB_MAX_CHARS so a heavily documented tool cannot outweigh a bare one.
239
+ """
240
+ path = os.path.join(root, name)
241
+ parts = [f"directory: {name}"]
242
+
243
+ fields = _advertise_fields(path)
244
+ if fields:
245
+ parts.append("advertises: " + " | ".join(fields))
246
+
247
+ for doc_name in DOC_FILES:
248
+ doc_path = os.path.join(path, doc_name)
249
+ if not os.path.isfile(doc_path):
250
+ continue
251
+ try:
252
+ with open(doc_path, encoding="utf-8", errors="replace") as fh:
253
+ text = fh.read(MAX_CHARS * 2)
254
+ except OSError:
255
+ continue
256
+ paragraphs, headings = _clean_markdown(text)
257
+ if paragraphs:
258
+ parts.append(" ".join(paragraphs[:BLURB_PARAGRAPHS]))
259
+ if headings:
260
+ parts.append("sections: " + "; ".join(headings[:BLURB_HEADINGS]))
261
+ break
262
+
263
+ return "\n".join(parts)[:BLURB_MAX_CHARS]
264
+
265
+
266
+ def tool_blurbs(root: str) -> Dict[str, str]:
267
+ """One condensed blurb per tool."""
268
+ return {name: tool_blurb(root, name) for name in tool_dirs(root)}
@@ -0,0 +1,170 @@
1
+ """Text embeddings with a local backend first and Gemini as fallback.
2
+
3
+ Two backends behind one function. The local OpenAI-compatible
4
+ ``/v1/embeddings`` endpoint (LM Studio on this fleet) is tried first because it
5
+ is free and offline; Gemini is used when it is not reachable. If neither works
6
+ the caller gets :class:`EmbeddingUnavailable` rather than a silent empty list.
7
+ """
8
+
9
+ import json
10
+ import os
11
+ import urllib.error
12
+ import urllib.request
13
+ from typing import List
14
+
15
+ _DEFAULT_HOST = "http://localhost:11434"
16
+ _DEFAULT_MODEL = "text-embedding-nomic-embed-text-v1.5"
17
+ _TIMEOUT = 60.0
18
+
19
+ _HERE = os.path.dirname(os.path.abspath(__file__))
20
+ ROOT_DIR = os.path.dirname(os.path.dirname(_HERE))
21
+
22
+
23
+ class EmbeddingUnavailable(RuntimeError):
24
+ """No embedding backend could produce vectors."""
25
+
26
+
27
+ def _load_env() -> None:
28
+ """Load the repo-root .env so GEMINI_API_KEY is present."""
29
+ try:
30
+ from dotenv import load_dotenv
31
+ except ImportError:
32
+ return
33
+ load_dotenv(os.path.join(ROOT_DIR, ".env"))
34
+
35
+
36
+ def local_host() -> str:
37
+ """Base URL of the local OpenAI-compatible server."""
38
+ host = (
39
+ os.getenv("TOOLS_EMBED_HOST", "")
40
+ or os.getenv("LMSTUDIO_HOST", "")
41
+ or os.getenv("OLLAMA_HOST", "")
42
+ or _DEFAULT_HOST
43
+ )
44
+ if not host.startswith("http"):
45
+ host = f"http://{host}"
46
+ return host.rstrip("/")
47
+
48
+
49
+ def local_model() -> str:
50
+ """Name of the local embedding model."""
51
+ return os.getenv("TOOLS_EMBED_MODEL", "") or _DEFAULT_MODEL
52
+
53
+
54
+ def _embed_local(texts: List[str]) -> List[List[float]]:
55
+ """Call the local /v1/embeddings endpoint. Returns [] on any failure."""
56
+ payload = json.dumps({"model": local_model(), "input": texts}).encode()
57
+ req = urllib.request.Request(
58
+ f"{local_host()}/v1/embeddings",
59
+ data=payload,
60
+ headers={"Content-Type": "application/json"},
61
+ )
62
+ try:
63
+ with urllib.request.urlopen(req, timeout=_TIMEOUT) as resp:
64
+ body = json.loads(resp.read().decode())
65
+ except (urllib.error.URLError, OSError, json.JSONDecodeError, ValueError):
66
+ return []
67
+
68
+ rows = body.get("data")
69
+ if not isinstance(rows, list) or len(rows) != len(texts):
70
+ return []
71
+ rows = sorted(rows, key=lambda r: r.get("index", 0))
72
+ vectors = []
73
+ for row in rows:
74
+ vec = row.get("embedding")
75
+ if not isinstance(vec, list) or not vec:
76
+ return []
77
+ vectors.append([float(x) for x in vec])
78
+ return vectors
79
+
80
+
81
+ def _embed_gemini_client(texts: List[str]) -> List[List[float]]:
82
+ """Try the shared GeminiClient's own embed(). Returns [] if unusable."""
83
+ try:
84
+ from _shared.gemini import GeminiClient
85
+ except ImportError:
86
+ return []
87
+ try:
88
+ client = GeminiClient()
89
+ result = client.embed(texts)
90
+ except Exception:
91
+ return []
92
+ if not isinstance(result, list) or len(result) != len(texts):
93
+ return []
94
+ if not all(isinstance(v, list) and v for v in result):
95
+ return []
96
+ return [[float(x) for x in v] for v in result]
97
+
98
+
99
+ def _embed_gemini_rest(texts: List[str]) -> List[List[float]]:
100
+ """Direct REST call to the Gemini embedding endpoint. [] on failure."""
101
+ api_key = os.getenv("GEMINI_API_KEY", "")
102
+ if not api_key:
103
+ return []
104
+ model = "models/gemini-embedding-001"
105
+ url = (
106
+ f"https://generativelanguage.googleapis.com/v1beta/{model}"
107
+ f":batchEmbedContents?key={api_key}"
108
+ )
109
+ payload = json.dumps(
110
+ {
111
+ "requests": [
112
+ {"model": model, "content": {"parts": [{"text": t}]}}
113
+ for t in texts
114
+ ]
115
+ }
116
+ ).encode()
117
+ req = urllib.request.Request(
118
+ url, data=payload, headers={"Content-Type": "application/json"}
119
+ )
120
+ try:
121
+ with urllib.request.urlopen(req, timeout=_TIMEOUT) as resp:
122
+ body = json.loads(resp.read().decode())
123
+ except (urllib.error.URLError, OSError, json.JSONDecodeError, ValueError):
124
+ return []
125
+
126
+ rows = body.get("embeddings")
127
+ if not isinstance(rows, list) or len(rows) != len(texts):
128
+ return []
129
+ vectors = []
130
+ for row in rows:
131
+ vec = (row or {}).get("values")
132
+ if not isinstance(vec, list) or not vec:
133
+ return []
134
+ vectors.append([float(x) for x in vec])
135
+ return vectors
136
+
137
+
138
+ def embed_texts(texts: List[str]) -> List[List[float]]:
139
+ """Embed a list of texts, local backend first, Gemini second.
140
+
141
+ Raises EmbeddingUnavailable if no backend answers.
142
+ """
143
+ texts = [t if t else " " for t in texts]
144
+ if not texts:
145
+ return []
146
+
147
+ vectors = _embed_local(texts)
148
+ if vectors:
149
+ return vectors
150
+
151
+ _load_env()
152
+ for backend in (_embed_gemini_client, _embed_gemini_rest):
153
+ vectors = backend(texts)
154
+ if vectors:
155
+ return vectors
156
+
157
+ raise EmbeddingUnavailable(
158
+ "No embedding backend answered: neither the local "
159
+ f"/v1/embeddings server at {local_host()} nor Gemini."
160
+ )
161
+
162
+
163
+ def backend_name() -> str:
164
+ """Name of the backend a fresh embed_texts call would reach first."""
165
+ try:
166
+ if _embed_local([" "]):
167
+ return f"local:{local_model()}"
168
+ except Exception:
169
+ pass
170
+ return "gemini"
@@ -0,0 +1,108 @@
1
+ """Read the stored tool grouping back.
2
+
3
+ ``data/tool_groups.json`` is written by regroup.py, or by
4
+ :func:`ensure_groups` when the installer starts and the file no longer matches
5
+ the tools on disk. It is a build artifact: it is regenerated on demand and
6
+ tracked in git, so every host reads the same groups. Entries in "overrides"
7
+ beat the computed assignment.
8
+ """
9
+
10
+ import json
11
+ import os
12
+ from typing import Dict
13
+
14
+ GROUPS_FILE = os.path.join("data", "tool_groups.json")
15
+ UNGROUPED = "Ungrouped"
16
+
17
+
18
+ def groups_path(root: str) -> str:
19
+ """Absolute path of the groups file for a repo root."""
20
+ return os.path.join(root, GROUPS_FILE)
21
+
22
+
23
+ def read_groups_file(root: str) -> dict:
24
+ """Return the raw JSON, or {} when it is missing or unreadable."""
25
+ path = groups_path(root)
26
+ try:
27
+ with open(path, encoding="utf-8") as fh:
28
+ data = json.load(fh)
29
+ except (OSError, json.JSONDecodeError):
30
+ return {}
31
+ return data if isinstance(data, dict) else {}
32
+
33
+
34
+ def load_groups(root: str) -> Dict[str, str]:
35
+ """Map each tool directory name to its group label.
36
+
37
+ Tools absent from the file are not in the result; callers treat a missing
38
+ tool as UNGROUPED.
39
+ """
40
+ data = read_groups_file(root)
41
+ mapping: Dict[str, str] = {}
42
+
43
+ labels = data.get("labels")
44
+ if isinstance(labels, dict):
45
+ for label, members in labels.items():
46
+ if not isinstance(members, list):
47
+ continue
48
+ for tool in members:
49
+ if isinstance(tool, str):
50
+ mapping[tool] = str(label)
51
+
52
+ overrides = data.get("overrides")
53
+ if isinstance(overrides, dict):
54
+ for tool, label in overrides.items():
55
+ if isinstance(tool, str) and isinstance(label, str):
56
+ mapping[tool] = label
57
+
58
+ return mapping
59
+
60
+
61
+ def group_of(root: str, tool: str) -> str:
62
+ """Group label for one tool, UNGROUPED when it has none."""
63
+ return load_groups(root).get(tool, UNGROUPED)
64
+
65
+
66
+ def is_stale(root: str, k: int) -> bool:
67
+ """True when the stored grouping no longer matches the corpus on disk.
68
+
69
+ Stale means: no labels yet, a different k, or a corpus fingerprint that has
70
+ moved — a tool added or removed, or a CLAUDE.md/README.md edited.
71
+ """
72
+ from .corpus import corpus_fingerprint
73
+
74
+ data = read_groups_file(root)
75
+ if not data.get("labels"):
76
+ return True
77
+ if data.get("k") != k:
78
+ return True
79
+ return data.get("fingerprint") != corpus_fingerprint(root)
80
+
81
+
82
+ # The installer calls ensure_groups on its way to opening a window, so the LLM
83
+ # tier gets a wall-clock cap. Gemini answers well inside it; a local model on a
84
+ # busy host may not, and then the capability tier takes over — instantly, and
85
+ # without disturbing bands that are already there.
86
+ STARTUP_BUDGET_SECONDS = 60.0
87
+
88
+
89
+ def ensure_groups(
90
+ root: str, k: int = None, budget: float = STARTUP_BUDGET_SECONDS
91
+ ) -> Dict[str, str]:
92
+ """Groups for the installer, recomputed first if the corpus has changed.
93
+
94
+ Falls back to whatever is stored when the rebuild cannot run at all (read-
95
+ only tree, no tools): a stale grouping beats no grouping, and the GUI must
96
+ still open.
97
+ """
98
+ from .build import DEFAULT_K
99
+
100
+ k = DEFAULT_K if k is None else k
101
+ try:
102
+ if is_stale(root, k):
103
+ from .build import build_groups
104
+
105
+ build_groups(root, k=k, budget=budget)
106
+ except Exception:
107
+ pass
108
+ return load_groups(root)