cli-tools-kit 0.6.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cli_tools_kit/__init__.py +59 -0
- cli_tools_kit/__main__.py +43 -0
- cli_tools_kit/advertise.py +77 -0
- cli_tools_kit/cron_installer.py +159 -0
- cli_tools_kit/gui_installer.py +5928 -0
- cli_tools_kit/host.py +239 -0
- cli_tools_kit/identity.py +229 -0
- cli_tools_kit/onboarding.py +261 -0
- cli_tools_kit/skills.py +97 -0
- cli_tools_kit/sources.py +335 -0
- cli_tools_kit/taxonomy/__init__.py +40 -0
- cli_tools_kit/taxonomy/build.py +171 -0
- cli_tools_kit/taxonomy/capability.py +116 -0
- cli_tools_kit/taxonomy/cluster.py +261 -0
- cli_tools_kit/taxonomy/corpus.py +268 -0
- cli_tools_kit/taxonomy/embedder.py +170 -0
- cli_tools_kit/taxonomy/groups.py +108 -0
- cli_tools_kit/taxonomy/llm_groups.py +748 -0
- cli_tools_kit/tool_installer.py +572 -0
- cli_tools_kit/tui_installer.py +462 -0
- cli_tools_kit-0.6.0.dist-info/METADATA +505 -0
- cli_tools_kit-0.6.0.dist-info/RECORD +26 -0
- cli_tools_kit-0.6.0.dist-info/WHEEL +5 -0
- cli_tools_kit-0.6.0.dist-info/entry_points.txt +3 -0
- cli_tools_kit-0.6.0.dist-info/licenses/LICENSE +21 -0
- cli_tools_kit-0.6.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,748 @@
|
|
|
1
|
+
"""Ask Gemini to sort the tools into named categories.
|
|
2
|
+
|
|
3
|
+
k-means over document embeddings produced defensible clusters with
|
|
4
|
+
indefensible names: the label was the most distinctive *token* in a cluster,
|
|
5
|
+
so a band holding arxiv_summary, scrape and extract_pdf_annotations came out
|
|
6
|
+
called "youtube". Naming a category is a language task, so a language model
|
|
7
|
+
does it — and it assigns the members at the same time, which also fixes the
|
|
8
|
+
clusters that were only held together by shared boilerplate.
|
|
9
|
+
|
|
10
|
+
Three steps, not one. Asked to name and fill six categories in a single call,
|
|
11
|
+
flash-lite reliably produced five plausible ones and swept the remaining third
|
|
12
|
+
of the collection into a "System & Utility" bin. The pipeline instead:
|
|
13
|
+
|
|
14
|
+
1. name — propose the k categories from one-line advertised summaries;
|
|
15
|
+
2. assign — file every tool into those fixed names, using the full blurbs;
|
|
16
|
+
3. review — show the model what each category ended up holding, and let it
|
|
17
|
+
move the misfits and rename a category to match its actual contents.
|
|
18
|
+
|
|
19
|
+
Step 3 is a patch, not a rewrite: only valid moves and renames are applied, so
|
|
20
|
+
a confused reply degrades to the step-2 result rather than replacing it.
|
|
21
|
+
|
|
22
|
+
A rebuild only happens when the corpus fingerprint moves (see
|
|
23
|
+
groups.ensure_groups), so this is about three cheap calls per documentation
|
|
24
|
+
edit. When it cannot run, build.py falls back to the offline embedding path.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
import json
|
|
28
|
+
import math
|
|
29
|
+
import os
|
|
30
|
+
import re
|
|
31
|
+
import time
|
|
32
|
+
import urllib.error
|
|
33
|
+
import urllib.request
|
|
34
|
+
from typing import Dict, List, Optional, Sequence, Tuple
|
|
35
|
+
|
|
36
|
+
MODEL = "gemini-3.5-flash-lite"
|
|
37
|
+
# Tried in order on the local OpenAI-compatible server (LM Studio). The small
|
|
38
|
+
# one first: this is a classification job, not a writing job.
|
|
39
|
+
LOCAL_MODELS = ("google/gemma-4-e2b", "qwen/qwen3.6-35b-a3b")
|
|
40
|
+
GEMINI_TIMEOUT = 60.0
|
|
41
|
+
LOCAL_TIMEOUT = 300.0
|
|
42
|
+
# A local model is small and its context is short (the default gemma-4-e2b load
|
|
43
|
+
# offers 4096 tokens, and the assignment prompt for 36 tools is 4145). On that
|
|
44
|
+
# backend the tools are described by their one-line summary and filed in
|
|
45
|
+
# batches that fit this budget, rather than all at once with full blurbs.
|
|
46
|
+
LOCAL_TOKEN_BUDGET = 2400
|
|
47
|
+
|
|
48
|
+
# LM Studio rejects response_format "json_object" ("must be 'json_schema' or
|
|
49
|
+
# 'text'"), and a small model holds its shape far better with a schema than
|
|
50
|
+
# with an instruction, so each step declares one.
|
|
51
|
+
_STRING_MAP = {"type": "object", "additionalProperties": {"type": "string"}}
|
|
52
|
+
CATEGORIES_SCHEMA = {
|
|
53
|
+
"type": "object",
|
|
54
|
+
"properties": {
|
|
55
|
+
"categories": {
|
|
56
|
+
"type": "array",
|
|
57
|
+
"items": {
|
|
58
|
+
"type": "object",
|
|
59
|
+
"properties": {
|
|
60
|
+
"name": {"type": "string"},
|
|
61
|
+
"scope": {"type": "string"},
|
|
62
|
+
},
|
|
63
|
+
"required": ["name"],
|
|
64
|
+
},
|
|
65
|
+
}
|
|
66
|
+
},
|
|
67
|
+
"required": ["categories"],
|
|
68
|
+
}
|
|
69
|
+
ASSIGNMENT_SCHEMA = {
|
|
70
|
+
"type": "object",
|
|
71
|
+
"properties": {"assignment": _STRING_MAP},
|
|
72
|
+
"required": ["assignment"],
|
|
73
|
+
}
|
|
74
|
+
REVIEW_SCHEMA = {
|
|
75
|
+
"type": "object",
|
|
76
|
+
"properties": {"moves": _STRING_MAP, "renames": _STRING_MAP},
|
|
77
|
+
}
|
|
78
|
+
MAX_NAME_CHARS = 24
|
|
79
|
+
_RETRIES = 2
|
|
80
|
+
_BANNED_NAMES = {
|
|
81
|
+
"misc", "miscellaneous", "other", "others", "utilities", "utility",
|
|
82
|
+
"tools", "general", "various", "assorted", "extras",
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
class LLMGroupingUnavailable(RuntimeError):
|
|
87
|
+
"""The model could not be reached, or would not answer usably."""
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def model_name() -> str:
|
|
91
|
+
"""Gemini model used for the grouping; TOOLS_GROUPS_MODEL overrides."""
|
|
92
|
+
return os.getenv("TOOLS_GROUPS_MODEL", "") or MODEL
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def local_models() -> List[str]:
|
|
96
|
+
"""Local models to try, in order; TOOLS_GROUPS_LOCAL_MODEL overrides."""
|
|
97
|
+
pinned = os.getenv("TOOLS_GROUPS_LOCAL_MODEL", "").strip()
|
|
98
|
+
return [pinned] if pinned else list(LOCAL_MODELS)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _have_api_key() -> bool:
|
|
102
|
+
"""True when a Gemini key is reachable (the repo .env counts)."""
|
|
103
|
+
try:
|
|
104
|
+
from .embedder import _load_env
|
|
105
|
+
_load_env()
|
|
106
|
+
except ImportError: # pragma: no cover - depends on the venv
|
|
107
|
+
pass
|
|
108
|
+
return bool(os.getenv("GEMINI_API_KEY", "").strip())
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def backend_order() -> List[str]:
|
|
112
|
+
"""Which backends to try, in order.
|
|
113
|
+
|
|
114
|
+
TOOLS_GROUPS_BACKEND pins one ("local" or "gemini"). The default is to use
|
|
115
|
+
Gemini when a key is present and the local LM Studio server otherwise —
|
|
116
|
+
so this fleet keeps the better model, and a checkout without a key still
|
|
117
|
+
gets a real grouping instead of the capability fallback.
|
|
118
|
+
"""
|
|
119
|
+
pinned = os.getenv("TOOLS_GROUPS_BACKEND", "").strip().lower()
|
|
120
|
+
if pinned in ("local", "gemini"):
|
|
121
|
+
return [pinned]
|
|
122
|
+
return ["gemini", "local"] if _have_api_key() else ["local", "gemini"]
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def size_band(n: int, k: int) -> Tuple[int, int]:
|
|
126
|
+
"""Smallest and largest sensible category size for n tools in k groups."""
|
|
127
|
+
return max(2, (n // k) // 2), math.ceil(n / k) + 3
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
# --- prompts --------------------------------------------------------------
|
|
131
|
+
|
|
132
|
+
def _one_liner(name: str, blurb: str) -> str:
|
|
133
|
+
"""The advertised summary line of a blurb, for the naming call."""
|
|
134
|
+
for line in blurb.splitlines():
|
|
135
|
+
if line.startswith("advertises:"):
|
|
136
|
+
return f"{name}: {line[len('advertises:'):].strip()[:120]}"
|
|
137
|
+
return f"{name}: {blurb.splitlines()[-1][:120] if blurb else ''}"
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def naming_prompt(blurbs: Dict[str, str], k: int) -> str:
|
|
141
|
+
"""Stage one: propose the k category names from one-line summaries."""
|
|
142
|
+
n = len(blurbs)
|
|
143
|
+
low, high = size_band(n, k)
|
|
144
|
+
listing = "\n".join(_one_liner(name, blurbs[name]) for name in sorted(blurbs))
|
|
145
|
+
return (
|
|
146
|
+
f"Here are {n} personal command-line and GUI tools, one per line as "
|
|
147
|
+
"'id: advertised name | description | capability word'.\n\n"
|
|
148
|
+
f"{listing}\n\n"
|
|
149
|
+
f"Propose exactly {k} category names to file them under in a graphical "
|
|
150
|
+
"installer's sidebar. Each name is one or two words, title case, at "
|
|
151
|
+
f"most {MAX_NAME_CHARS} characters, and names a subject a user would "
|
|
152
|
+
"look under. No catch-all ('Misc', 'Other', 'Utilities', 'Tools', "
|
|
153
|
+
f"'General'). Together the {k} must cover all {n} tools with roughly "
|
|
154
|
+
f"{low}-{high} tools each, so choose boundaries that split the "
|
|
155
|
+
"collection evenly rather than one broad name that would absorb "
|
|
156
|
+
"everything left over.\n"
|
|
157
|
+
'Answer as JSON: {"categories": [{"name": "...", "scope": "one line '
|
|
158
|
+
'on what belongs here"}]}'
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def assignment_prompt(
|
|
163
|
+
blurbs: Dict[str, str],
|
|
164
|
+
categories: Sequence[dict],
|
|
165
|
+
k: int,
|
|
166
|
+
total: Optional[int] = None,
|
|
167
|
+
) -> str:
|
|
168
|
+
"""Step two: file every tool under one of the fixed category names.
|
|
169
|
+
|
|
170
|
+
``total`` is the size of the whole collection when this prompt covers only
|
|
171
|
+
a batch of it, which changes the size guidance: a batch must not be
|
|
172
|
+
balanced against itself.
|
|
173
|
+
"""
|
|
174
|
+
n = len(blurbs)
|
|
175
|
+
block = "\n".join(
|
|
176
|
+
f"- {c['name']}: {c.get('scope', '')}" for c in categories
|
|
177
|
+
)
|
|
178
|
+
bodies = "\n\n".join(f"### {name}\n{blurbs[name]}" for name in sorted(blurbs))
|
|
179
|
+
if total and total != n:
|
|
180
|
+
sizing = (
|
|
181
|
+
f"These {n} are one batch of {total} tools being filed into the "
|
|
182
|
+
"same categories, so do not try to balance the batch: file each "
|
|
183
|
+
"tool where it belongs, even if that fills one category."
|
|
184
|
+
)
|
|
185
|
+
else:
|
|
186
|
+
low, high = size_band(n, k)
|
|
187
|
+
sizing = f"Each category ends up with roughly {low}-{high} tools."
|
|
188
|
+
return (
|
|
189
|
+
f"File each of these {n} tools under exactly one of the {k} "
|
|
190
|
+
f"categories.\n\nCategories:\n{block}\n\n"
|
|
191
|
+
f"{sizing}\n"
|
|
192
|
+
"Rules: every tool id appears exactly once; use the ids verbatim, "
|
|
193
|
+
"with their original capitalisation; file by subject and ignore the "
|
|
194
|
+
"install, git and sync prose a description may carry.\n"
|
|
195
|
+
'Answer as JSON: {"assignment": {"tool_id": "Category Name", ...}}\n\n'
|
|
196
|
+
f"{bodies}"
|
|
197
|
+
)
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def _estimate_tokens(text: str) -> int:
|
|
201
|
+
"""Rough token count — four characters per token is close enough here."""
|
|
202
|
+
return len(text) // 4 + 1
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def batch_for_budget(
|
|
206
|
+
blurbs: Dict[str, str], overhead: int, budget: int = LOCAL_TOKEN_BUDGET
|
|
207
|
+
) -> List[List[str]]:
|
|
208
|
+
"""Split tool names into batches whose prompts fit the token budget."""
|
|
209
|
+
batches: List[List[str]] = [[]]
|
|
210
|
+
used = overhead
|
|
211
|
+
for name in sorted(blurbs):
|
|
212
|
+
cost = _estimate_tokens(blurbs[name]) + 8
|
|
213
|
+
if batches[-1] and used + cost > budget:
|
|
214
|
+
batches.append([])
|
|
215
|
+
used = overhead
|
|
216
|
+
batches[-1].append(name)
|
|
217
|
+
used += cost
|
|
218
|
+
return [batch for batch in batches if batch]
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
# --- reply handling -------------------------------------------------------
|
|
222
|
+
|
|
223
|
+
def _clean_name(raw) -> str:
|
|
224
|
+
"""Tidy a category name and keep it inside MAX_NAME_CHARS.
|
|
225
|
+
|
|
226
|
+
An over-long name is cut back to a whole word: "System & Development Tools"
|
|
227
|
+
truncated blind reads "System & Development Too".
|
|
228
|
+
"""
|
|
229
|
+
name = str(raw or "").strip().strip('"').strip()
|
|
230
|
+
if not name or name.lower() in _BANNED_NAMES:
|
|
231
|
+
return ""
|
|
232
|
+
if len(name) > MAX_NAME_CHARS:
|
|
233
|
+
cut = name[:MAX_NAME_CHARS + 1]
|
|
234
|
+
if " " in cut.strip():
|
|
235
|
+
cut = cut[:cut.rstrip().rfind(" ")]
|
|
236
|
+
name = cut.strip()
|
|
237
|
+
return name.strip(" &-,/·").strip()
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def _extract_json(text: str) -> dict:
|
|
241
|
+
"""Parse the reply, tolerating a code fence or prose around the JSON."""
|
|
242
|
+
text = (text or "").strip()
|
|
243
|
+
if not text:
|
|
244
|
+
raise ValueError("empty reply")
|
|
245
|
+
fenced = re.search(r"```(?:json)?\s*(.*?)```", text, re.S)
|
|
246
|
+
if fenced:
|
|
247
|
+
text = fenced.group(1).strip()
|
|
248
|
+
try:
|
|
249
|
+
return json.loads(text)
|
|
250
|
+
except json.JSONDecodeError:
|
|
251
|
+
pass
|
|
252
|
+
start, end = text.find("{"), text.rfind("}")
|
|
253
|
+
if start == -1 or end <= start:
|
|
254
|
+
raise ValueError("no JSON object in the reply")
|
|
255
|
+
return json.loads(text[start:end + 1])
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def parse_categories(reply: str, k: int) -> List[dict]:
|
|
259
|
+
"""Stage-one reply -> [{"name", "scope"}], names cleaned and deduplicated."""
|
|
260
|
+
data = _extract_json(reply)
|
|
261
|
+
raw = data.get("categories") if isinstance(data, dict) else None
|
|
262
|
+
if not isinstance(raw, list):
|
|
263
|
+
raise ValueError("no 'categories' list in the reply")
|
|
264
|
+
|
|
265
|
+
out: List[dict] = []
|
|
266
|
+
seen = set()
|
|
267
|
+
for entry in raw:
|
|
268
|
+
if isinstance(entry, str):
|
|
269
|
+
entry = {"name": entry}
|
|
270
|
+
if not isinstance(entry, dict):
|
|
271
|
+
continue
|
|
272
|
+
name = _clean_name(entry.get("name"))
|
|
273
|
+
if not name or name.casefold() in seen:
|
|
274
|
+
continue
|
|
275
|
+
seen.add(name.casefold())
|
|
276
|
+
out.append({"name": name, "scope": str(entry.get("scope") or "").strip()})
|
|
277
|
+
if len(out) != k:
|
|
278
|
+
raise ValueError(f"got {len(out)} category names, need exactly {k}")
|
|
279
|
+
return out
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def parse_assignment(
|
|
283
|
+
reply: str, names: Sequence[str], categories: Sequence[dict]
|
|
284
|
+
) -> Tuple[Dict[str, List[str]], List[str]]:
|
|
285
|
+
"""Stage-two reply -> {category: [tool, ...]} plus a list of complaints.
|
|
286
|
+
|
|
287
|
+
Never raises on a merely sloppy answer. Tool ids and category names are
|
|
288
|
+
matched case-insensitively (flash-lite writes "llmchat" for "LMChat"),
|
|
289
|
+
unknown ids are dropped, and a tool the model forgot lands in the smallest
|
|
290
|
+
category. The complaints are what a retry is told about.
|
|
291
|
+
"""
|
|
292
|
+
data = _extract_json(reply)
|
|
293
|
+
raw = data.get("assignment") if isinstance(data, dict) else None
|
|
294
|
+
if not isinstance(raw, dict) or not raw:
|
|
295
|
+
raise ValueError("no 'assignment' object in the reply")
|
|
296
|
+
|
|
297
|
+
by_tool = {n.casefold(): n for n in names}
|
|
298
|
+
by_category = {c["name"].casefold(): c["name"] for c in categories}
|
|
299
|
+
labels: Dict[str, List[str]] = {c["name"]: [] for c in categories}
|
|
300
|
+
complaints: List[str] = []
|
|
301
|
+
assigned = set()
|
|
302
|
+
|
|
303
|
+
for tool, category in raw.items():
|
|
304
|
+
real = by_tool.get(str(tool).strip().casefold())
|
|
305
|
+
if real is None:
|
|
306
|
+
complaints.append(f"unknown tool {tool!r}")
|
|
307
|
+
continue
|
|
308
|
+
if real in assigned:
|
|
309
|
+
complaints.append(f"{real} assigned twice")
|
|
310
|
+
continue
|
|
311
|
+
label = by_category.get(str(category).strip().casefold())
|
|
312
|
+
if label is None:
|
|
313
|
+
complaints.append(f"{real} put in unknown category {category!r}")
|
|
314
|
+
continue
|
|
315
|
+
labels[label].append(real)
|
|
316
|
+
assigned.add(real)
|
|
317
|
+
|
|
318
|
+
missing = [n for n in names if n not in assigned]
|
|
319
|
+
if missing:
|
|
320
|
+
complaints.append("unassigned: " + ", ".join(missing))
|
|
321
|
+
smallest = min(labels, key=lambda label: (len(labels[label]), label))
|
|
322
|
+
labels[smallest].extend(missing)
|
|
323
|
+
|
|
324
|
+
_, high = size_band(len(names), len(categories))
|
|
325
|
+
for label, members in labels.items():
|
|
326
|
+
if len(members) > high:
|
|
327
|
+
complaints.append(
|
|
328
|
+
f"category {label!r} holds {len(members)} tools, more than {high}"
|
|
329
|
+
)
|
|
330
|
+
|
|
331
|
+
labels = {label: sorted(members) for label, members in labels.items() if members}
|
|
332
|
+
if not labels:
|
|
333
|
+
raise ValueError("no tool was assigned")
|
|
334
|
+
return labels, complaints
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def review_prompt(
|
|
338
|
+
labels: Dict[str, List[str]],
|
|
339
|
+
blurbs: Dict[str, str],
|
|
340
|
+
categories: Sequence[dict] = (),
|
|
341
|
+
) -> str:
|
|
342
|
+
"""Step three: show the finished bands and invite a patch.
|
|
343
|
+
|
|
344
|
+
Each band is shown with the scope it was created for, so a move is judged
|
|
345
|
+
against what the category was meant to hold and not only its name.
|
|
346
|
+
"""
|
|
347
|
+
scopes = {c["name"]: c.get("scope", "") for c in categories}
|
|
348
|
+
blocks = []
|
|
349
|
+
for label in sorted(labels):
|
|
350
|
+
members = "\n".join(
|
|
351
|
+
" " + _one_liner(name, blurbs.get(name, "")) for name in labels[label]
|
|
352
|
+
)
|
|
353
|
+
scope = scopes.get(label)
|
|
354
|
+
header = f"{label} ({len(labels[label])})"
|
|
355
|
+
if scope:
|
|
356
|
+
header += f" — meant for: {scope}"
|
|
357
|
+
blocks.append(f"{header}:\n{members}")
|
|
358
|
+
return (
|
|
359
|
+
"These are the finished categories of a tool installer's sidebar, each "
|
|
360
|
+
"with the tools filed under it.\n\n" + "\n\n".join(blocks) + "\n\n"
|
|
361
|
+
"Review the result and correct it. Look for a tool whose subject does "
|
|
362
|
+
"not match the category it sits in, and for a category whose name no "
|
|
363
|
+
"longer describes what it actually holds. Judge a tool by what it does "
|
|
364
|
+
"for its user: reading a chat service is not the same subject as "
|
|
365
|
+
"running an AI agent, even though both involve conversations.\n"
|
|
366
|
+
"Rules: move a tool only when another existing category is clearly a "
|
|
367
|
+
"better home — do not shuffle borderline cases; keep every category "
|
|
368
|
+
"non-empty; a new name is one or two words, title case, at most "
|
|
369
|
+
f"{MAX_NAME_CHARS} characters, and must not be a catch-all ('Misc', "
|
|
370
|
+
"'Other', 'Utilities', 'Tools', 'General').\n"
|
|
371
|
+
"Change nothing you are not confident about. An empty patch is a valid "
|
|
372
|
+
"answer.\n"
|
|
373
|
+
'Answer as JSON: {"moves": {"tool_id": "Destination Category"}, '
|
|
374
|
+
'"renames": {"Old Name": "New Name"}}'
|
|
375
|
+
)
|
|
376
|
+
|
|
377
|
+
|
|
378
|
+
def parse_review(
|
|
379
|
+
reply: str, labels: Dict[str, List[str]]
|
|
380
|
+
) -> Tuple[Dict[str, List[str]], List[str]]:
|
|
381
|
+
"""Apply a review patch to the labels, returning (labels, applied notes).
|
|
382
|
+
|
|
383
|
+
Everything is checked before it is applied: an unknown tool or destination,
|
|
384
|
+
a move that would empty a category, a rename onto an existing name or to a
|
|
385
|
+
banned one, are all ignored. The step-2 result is the floor.
|
|
386
|
+
"""
|
|
387
|
+
data = _extract_json(reply)
|
|
388
|
+
if not isinstance(data, dict):
|
|
389
|
+
raise ValueError("review reply is not an object")
|
|
390
|
+
|
|
391
|
+
result = {label: list(members) for label, members in labels.items()}
|
|
392
|
+
home = {tool: label for label, members in result.items() for tool in members}
|
|
393
|
+
notes: List[str] = []
|
|
394
|
+
|
|
395
|
+
moves = data.get("moves")
|
|
396
|
+
if isinstance(moves, dict):
|
|
397
|
+
by_tool = {t.casefold(): t for t in home}
|
|
398
|
+
by_label = {label.casefold(): label for label in result}
|
|
399
|
+
for tool, destination in moves.items():
|
|
400
|
+
real = by_tool.get(str(tool).strip().casefold())
|
|
401
|
+
target = by_label.get(str(destination).strip().casefold())
|
|
402
|
+
if real is None or target is None or target == home[real]:
|
|
403
|
+
continue
|
|
404
|
+
source = home[real]
|
|
405
|
+
if len(result[source]) <= 1:
|
|
406
|
+
continue # never empty a category
|
|
407
|
+
result[source].remove(real)
|
|
408
|
+
result[target].append(real)
|
|
409
|
+
home[real] = target
|
|
410
|
+
notes.append(f"moved {real}: {source} -> {target}")
|
|
411
|
+
|
|
412
|
+
renames = data.get("renames")
|
|
413
|
+
if isinstance(renames, dict):
|
|
414
|
+
by_label = {label.casefold(): label for label in result}
|
|
415
|
+
for old, new in renames.items():
|
|
416
|
+
source = by_label.get(str(old).strip().casefold())
|
|
417
|
+
name = _clean_name(new)
|
|
418
|
+
if source is None or not name or name == source:
|
|
419
|
+
continue
|
|
420
|
+
if name.casefold() in {label.casefold() for label in result}:
|
|
421
|
+
continue
|
|
422
|
+
result[name] = result.pop(source)
|
|
423
|
+
by_label.pop(source.casefold(), None)
|
|
424
|
+
by_label[name.casefold()] = name
|
|
425
|
+
notes.append(f"renamed {source} -> {name}")
|
|
426
|
+
|
|
427
|
+
return {label: sorted(members) for label, members in result.items()}, notes
|
|
428
|
+
|
|
429
|
+
|
|
430
|
+
# --- the call ------------------------------------------------------------
|
|
431
|
+
|
|
432
|
+
def _gemini_rest(prompt: str) -> str:
|
|
433
|
+
"""One generateContent call over plain HTTP. "" when it cannot be made.
|
|
434
|
+
|
|
435
|
+
The fallback for anyone without the first-party GeminiClient: it needs
|
|
436
|
+
nothing but GEMINI_API_KEY and the standard library, which is what makes
|
|
437
|
+
the LLM tier available to a third-party checkout at all.
|
|
438
|
+
"""
|
|
439
|
+
api_key = os.getenv("GEMINI_API_KEY", "")
|
|
440
|
+
if not api_key:
|
|
441
|
+
return ""
|
|
442
|
+
url = (f"https://generativelanguage.googleapis.com/v1beta/models/"
|
|
443
|
+
f"{model_name()}:generateContent?key={api_key}")
|
|
444
|
+
payload = json.dumps({
|
|
445
|
+
"contents": [{"parts": [{"text": prompt}]}],
|
|
446
|
+
"generationConfig": {"temperature": 0, "responseMimeType": "application/json"},
|
|
447
|
+
}).encode()
|
|
448
|
+
req = urllib.request.Request(
|
|
449
|
+
url, data=payload, headers={"Content-Type": "application/json"})
|
|
450
|
+
try:
|
|
451
|
+
with urllib.request.urlopen(req, timeout=GEMINI_TIMEOUT) as resp:
|
|
452
|
+
data = json.loads(resp.read().decode())
|
|
453
|
+
except Exception:
|
|
454
|
+
return ""
|
|
455
|
+
try:
|
|
456
|
+
parts = data["candidates"][0]["content"]["parts"]
|
|
457
|
+
except (KeyError, IndexError, TypeError):
|
|
458
|
+
return ""
|
|
459
|
+
return "".join(p.get("text", "") for p in parts if isinstance(p, dict))
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
def _ask_gemini(prompt: str, fresh: bool = False) -> str:
|
|
463
|
+
"""One Gemini call. Raises LLMGroupingUnavailable if it cannot be made.
|
|
464
|
+
|
|
465
|
+
Prefers the first-party GeminiClient when it is importable — it brings the
|
|
466
|
+
response cache and the model rotation — and falls back to a direct REST
|
|
467
|
+
call, so a checkout without it still gets the LLM grouping.
|
|
468
|
+
"""
|
|
469
|
+
try:
|
|
470
|
+
from .embedder import _load_env
|
|
471
|
+
_load_env() # GEMINI_API_KEY lives in the tree's .env
|
|
472
|
+
except ImportError: # pragma: no cover - defensive
|
|
473
|
+
pass
|
|
474
|
+
|
|
475
|
+
try:
|
|
476
|
+
from _shared.gemini import GeminiClient
|
|
477
|
+
except ImportError:
|
|
478
|
+
GeminiClient = None
|
|
479
|
+
|
|
480
|
+
if GeminiClient is not None:
|
|
481
|
+
try:
|
|
482
|
+
client = GeminiClient(models=[model_name()])
|
|
483
|
+
reply = client.generate(prompt, json_mode=True, label="", skip_cache=fresh)
|
|
484
|
+
if reply:
|
|
485
|
+
return reply
|
|
486
|
+
except Exception:
|
|
487
|
+
pass # fall through to REST rather than losing the tier
|
|
488
|
+
|
|
489
|
+
reply = _gemini_rest(prompt)
|
|
490
|
+
if not reply:
|
|
491
|
+
raise LLMGroupingUnavailable("Gemini returned nothing")
|
|
492
|
+
return reply
|
|
493
|
+
|
|
494
|
+
|
|
495
|
+
def _local_host() -> str:
|
|
496
|
+
"""Base URL of the local OpenAI-compatible server."""
|
|
497
|
+
try:
|
|
498
|
+
from .embedder import local_host
|
|
499
|
+
return local_host()
|
|
500
|
+
except ImportError: # pragma: no cover - depends on the venv
|
|
501
|
+
host = os.getenv("TOOLS_EMBED_HOST", "") or "http://localhost:11434"
|
|
502
|
+
return (host if host.startswith("http") else f"http://{host}").rstrip("/")
|
|
503
|
+
|
|
504
|
+
|
|
505
|
+
def _chat_local(prompt: str, model: str, schema: Optional[dict] = None) -> str: # noqa: C901
|
|
506
|
+
"""One /v1/chat/completions call. Returns "" on any failure.
|
|
507
|
+
|
|
508
|
+
Retries once without the schema: an older server, or one whose model has no
|
|
509
|
+
grammar support, rejects response_format rather than ignoring it.
|
|
510
|
+
"""
|
|
511
|
+
for use_schema in (bool(schema), False):
|
|
512
|
+
body = {
|
|
513
|
+
"model": model,
|
|
514
|
+
"messages": [{"role": "user", "content": prompt}],
|
|
515
|
+
"temperature": 0,
|
|
516
|
+
}
|
|
517
|
+
if use_schema:
|
|
518
|
+
body["response_format"] = {
|
|
519
|
+
"type": "json_schema",
|
|
520
|
+
"json_schema": {
|
|
521
|
+
"name": "tool_grouping", "strict": True, "schema": schema,
|
|
522
|
+
},
|
|
523
|
+
}
|
|
524
|
+
request = urllib.request.Request(
|
|
525
|
+
f"{_local_host()}/v1/chat/completions",
|
|
526
|
+
data=json.dumps(body).encode(),
|
|
527
|
+
headers={"Content-Type": "application/json"},
|
|
528
|
+
)
|
|
529
|
+
left = _remaining()
|
|
530
|
+
timeout = LOCAL_TIMEOUT if left is None else min(LOCAL_TIMEOUT, max(1.0, left))
|
|
531
|
+
try:
|
|
532
|
+
with urllib.request.urlopen(request, timeout=timeout) as response:
|
|
533
|
+
data = json.loads(response.read().decode())
|
|
534
|
+
except (urllib.error.URLError, OSError, json.JSONDecodeError, ValueError):
|
|
535
|
+
continue
|
|
536
|
+
try:
|
|
537
|
+
content = data["choices"][0]["message"]["content"] or ""
|
|
538
|
+
except (KeyError, IndexError, TypeError):
|
|
539
|
+
continue
|
|
540
|
+
if content:
|
|
541
|
+
return content
|
|
542
|
+
return ""
|
|
543
|
+
|
|
544
|
+
|
|
545
|
+
# The backend and model that answered first; the rest of the pipeline stays on
|
|
546
|
+
# it rather than re-probing before every step.
|
|
547
|
+
_ACTIVE: Optional[Tuple[str, str]] = None
|
|
548
|
+
# Wall-clock deadline for the whole pipeline, or None for no limit. The
|
|
549
|
+
# installer sets one because it calls this on the way to opening a window; a
|
|
550
|
+
# local model on a busy host can take minutes per batch, and a GUI that waits
|
|
551
|
+
# for it is worse than a grouping filed by capability words.
|
|
552
|
+
_DEADLINE: Optional[float] = None
|
|
553
|
+
|
|
554
|
+
|
|
555
|
+
def _remaining() -> Optional[float]:
|
|
556
|
+
"""Seconds left of the deadline, or None when there is none."""
|
|
557
|
+
return None if _DEADLINE is None else _DEADLINE - time.monotonic()
|
|
558
|
+
|
|
559
|
+
|
|
560
|
+
def active_backend() -> str:
|
|
561
|
+
"""Which backend answered ("local", "gemini"), or "" before the first call."""
|
|
562
|
+
return _ACTIVE[0] if _ACTIVE else ""
|
|
563
|
+
|
|
564
|
+
|
|
565
|
+
def active_label() -> str:
|
|
566
|
+
"""What produced the current grouping, e.g. "local:google/gemma-4-e2b"."""
|
|
567
|
+
if _ACTIVE is None:
|
|
568
|
+
return ""
|
|
569
|
+
backend, model = _ACTIVE
|
|
570
|
+
return f"{backend}:{model}"
|
|
571
|
+
|
|
572
|
+
|
|
573
|
+
def _ask(prompt: str, schema: Optional[dict] = None, fresh: bool = False) -> str:
|
|
574
|
+
"""Ask whichever backend is available, and stay on the one that answers.
|
|
575
|
+
|
|
576
|
+
``fresh`` bypasses the response cache. The naming step uses it: an
|
|
577
|
+
identical prompt otherwise replays one answer forever, so a mediocre set of
|
|
578
|
+
names could never be shaken off by asking again.
|
|
579
|
+
"""
|
|
580
|
+
global _ACTIVE
|
|
581
|
+
|
|
582
|
+
left = _remaining()
|
|
583
|
+
if left is not None and left <= 0:
|
|
584
|
+
raise LLMGroupingUnavailable("out of time")
|
|
585
|
+
|
|
586
|
+
if _ACTIVE is not None:
|
|
587
|
+
backend, model = _ACTIVE
|
|
588
|
+
if backend == "gemini":
|
|
589
|
+
return _ask_gemini(prompt, fresh)
|
|
590
|
+
reply = _chat_local(prompt, model, schema)
|
|
591
|
+
if reply:
|
|
592
|
+
return reply
|
|
593
|
+
raise LLMGroupingUnavailable(f"local model {model} stopped answering")
|
|
594
|
+
|
|
595
|
+
problems = []
|
|
596
|
+
for backend in backend_order():
|
|
597
|
+
if backend == "gemini":
|
|
598
|
+
try:
|
|
599
|
+
reply = _ask_gemini(prompt, fresh)
|
|
600
|
+
except LLMGroupingUnavailable as exc:
|
|
601
|
+
problems.append(str(exc))
|
|
602
|
+
continue
|
|
603
|
+
_ACTIVE = ("gemini", model_name())
|
|
604
|
+
return reply
|
|
605
|
+
for model in local_models():
|
|
606
|
+
reply = _chat_local(prompt, model, schema)
|
|
607
|
+
if reply:
|
|
608
|
+
_ACTIVE = ("local", model)
|
|
609
|
+
return reply
|
|
610
|
+
problems.append(f"local model {model} did not answer")
|
|
611
|
+
raise LLMGroupingUnavailable("; ".join(problems) or "no backend configured")
|
|
612
|
+
|
|
613
|
+
|
|
614
|
+
def llm_groups(
|
|
615
|
+
blurbs: Dict[str, str],
|
|
616
|
+
k: int,
|
|
617
|
+
budget: Optional[float] = None,
|
|
618
|
+
categories: Optional[Sequence[dict]] = None,
|
|
619
|
+
) -> Dict[str, List[str]]:
|
|
620
|
+
"""Category name -> members, decided by the model.
|
|
621
|
+
|
|
622
|
+
``categories`` skips the naming step and files into names that already
|
|
623
|
+
exist. The installer passes the stored ones: a rebuild after an edited
|
|
624
|
+
README should not rename every band, and asking for six fresh names each
|
|
625
|
+
time produced overlapping ones ("Artificial Intelligence" beside "AI
|
|
626
|
+
Agents"). The review step can still rename one whose contents drifted.
|
|
627
|
+
|
|
628
|
+
``budget`` caps the whole pipeline in seconds; past it the next call raises
|
|
629
|
+
rather than starting. Raises LLMGroupingUnavailable when the model cannot
|
|
630
|
+
be reached, its answers stay unusable, or the budget runs out.
|
|
631
|
+
"""
|
|
632
|
+
global _ACTIVE, _DEADLINE
|
|
633
|
+
_ACTIVE = None
|
|
634
|
+
_DEADLINE = None if budget is None else time.monotonic() + budget
|
|
635
|
+
|
|
636
|
+
names = sorted(blurbs)
|
|
637
|
+
if not names:
|
|
638
|
+
raise ValueError("no tools to group")
|
|
639
|
+
|
|
640
|
+
if categories and len(categories) == k:
|
|
641
|
+
return _fill(blurbs, list(categories), k, names)
|
|
642
|
+
return _fill(blurbs, None, k, names)
|
|
643
|
+
|
|
644
|
+
|
|
645
|
+
def _fill(
|
|
646
|
+
blurbs: Dict[str, str],
|
|
647
|
+
categories: Optional[List[dict]],
|
|
648
|
+
k: int,
|
|
649
|
+
names: Sequence[str],
|
|
650
|
+
) -> Dict[str, List[str]]:
|
|
651
|
+
"""Name the categories if needed, then assign and review."""
|
|
652
|
+
problem = ""
|
|
653
|
+
for _ in range(_RETRIES):
|
|
654
|
+
if categories is not None:
|
|
655
|
+
break
|
|
656
|
+
prompt = naming_prompt(blurbs, k)
|
|
657
|
+
if problem:
|
|
658
|
+
prompt += f"\n\nYour previous answer was rejected: {problem}."
|
|
659
|
+
try:
|
|
660
|
+
reply = _ask(prompt, CATEGORIES_SCHEMA, fresh=True)
|
|
661
|
+
categories = parse_categories(reply, k)
|
|
662
|
+
break
|
|
663
|
+
except (ValueError, json.JSONDecodeError) as exc:
|
|
664
|
+
problem = str(exc)
|
|
665
|
+
if categories is None:
|
|
666
|
+
raise LLMGroupingUnavailable(f"could not name {k} categories: {problem}")
|
|
667
|
+
|
|
668
|
+
if active_backend() == "local":
|
|
669
|
+
labels = _assign_in_batches(blurbs, categories, k, names)
|
|
670
|
+
return _review(labels, _compact(blurbs), categories)
|
|
671
|
+
|
|
672
|
+
problem = ""
|
|
673
|
+
best: Dict[str, List[str]] = {}
|
|
674
|
+
for attempt in range(_RETRIES):
|
|
675
|
+
prompt = assignment_prompt(blurbs, categories, k)
|
|
676
|
+
if problem:
|
|
677
|
+
prompt += (
|
|
678
|
+
f"\n\nYour previous answer was rejected: {problem}. Answer "
|
|
679
|
+
"again, covering every tool exactly once and keeping the "
|
|
680
|
+
"categories to a sensible size."
|
|
681
|
+
)
|
|
682
|
+
try:
|
|
683
|
+
reply = _ask(prompt, ASSIGNMENT_SCHEMA)
|
|
684
|
+
labels, complaints = parse_assignment(reply, names, categories)
|
|
685
|
+
except (ValueError, json.JSONDecodeError) as exc:
|
|
686
|
+
problem = str(exc)
|
|
687
|
+
continue
|
|
688
|
+
if not complaints:
|
|
689
|
+
return _review(labels, blurbs, categories)
|
|
690
|
+
best = best or labels
|
|
691
|
+
if attempt == _RETRIES - 1:
|
|
692
|
+
# repaired: usable, if not pristine
|
|
693
|
+
return _review(best, blurbs, categories)
|
|
694
|
+
problem = "; ".join(complaints[:5])
|
|
695
|
+
raise LLMGroupingUnavailable(f"unusable assignment: {problem}")
|
|
696
|
+
|
|
697
|
+
|
|
698
|
+
def _compact(blurbs: Dict[str, str]) -> Dict[str, str]:
|
|
699
|
+
"""One line per tool — what a short-context model gets instead of blurbs."""
|
|
700
|
+
return {name: _one_liner(name, blurb) for name, blurb in blurbs.items()}
|
|
701
|
+
|
|
702
|
+
|
|
703
|
+
def _assign_in_batches(
|
|
704
|
+
blurbs: Dict[str, str],
|
|
705
|
+
categories: Sequence[dict],
|
|
706
|
+
k: int,
|
|
707
|
+
names: Sequence[str],
|
|
708
|
+
) -> Dict[str, List[str]]:
|
|
709
|
+
"""Step two for a short-context backend: compact input, several batches.
|
|
710
|
+
|
|
711
|
+
A batch that fails is not fatal — its tools come back unassigned and the
|
|
712
|
+
usual repair files them, so one bad reply costs a few placements rather
|
|
713
|
+
than the whole grouping.
|
|
714
|
+
"""
|
|
715
|
+
compact = _compact(blurbs)
|
|
716
|
+
overhead = _estimate_tokens(assignment_prompt({}, categories, k, total=len(names)))
|
|
717
|
+
merged: Dict[str, str] = {}
|
|
718
|
+
for batch in batch_for_budget(compact, overhead):
|
|
719
|
+
subset = {name: compact[name] for name in batch}
|
|
720
|
+
prompt = assignment_prompt(subset, categories, k, total=len(names))
|
|
721
|
+
try:
|
|
722
|
+
data = _extract_json(_ask(prompt, ASSIGNMENT_SCHEMA))
|
|
723
|
+
except (LLMGroupingUnavailable, ValueError, json.JSONDecodeError):
|
|
724
|
+
continue
|
|
725
|
+
assignment = data.get("assignment") if isinstance(data, dict) else None
|
|
726
|
+
if isinstance(assignment, dict):
|
|
727
|
+
merged.update({str(t): str(c) for t, c in assignment.items()})
|
|
728
|
+
|
|
729
|
+
if not merged:
|
|
730
|
+
raise LLMGroupingUnavailable("no batch of the assignment step answered")
|
|
731
|
+
labels, _ = parse_assignment(json.dumps({"assignment": merged}), names, categories)
|
|
732
|
+
return labels
|
|
733
|
+
|
|
734
|
+
|
|
735
|
+
def _review(
|
|
736
|
+
labels: Dict[str, List[str]],
|
|
737
|
+
blurbs: Dict[str, str],
|
|
738
|
+
categories: Sequence[dict] = (),
|
|
739
|
+
) -> Dict[str, List[str]]:
|
|
740
|
+
"""Step three, best effort: a failed review keeps the step-2 result."""
|
|
741
|
+
try:
|
|
742
|
+
reply = _ask(review_prompt(labels, blurbs, categories), REVIEW_SCHEMA)
|
|
743
|
+
reviewed, notes = parse_review(reply, labels)
|
|
744
|
+
except (LLMGroupingUnavailable, ValueError, json.JSONDecodeError):
|
|
745
|
+
return labels
|
|
746
|
+
for note in notes:
|
|
747
|
+
print(f" review: {note}")
|
|
748
|
+
return reviewed
|