cli-tools-kit 0.6.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,748 @@
1
+ """Ask Gemini to sort the tools into named categories.
2
+
3
+ k-means over document embeddings produced defensible clusters with
4
+ indefensible names: the label was the most distinctive *token* in a cluster,
5
+ so a band holding arxiv_summary, scrape and extract_pdf_annotations came out
6
+ called "youtube". Naming a category is a language task, so a language model
7
+ does it — and it assigns the members at the same time, which also fixes the
8
+ clusters that were only held together by shared boilerplate.
9
+
10
+ Three steps, not one. Asked to name and fill six categories in a single call,
11
+ flash-lite reliably produced five plausible ones and swept the remaining third
12
+ of the collection into a "System & Utility" bin. The pipeline instead:
13
+
14
+ 1. name — propose the k categories from one-line advertised summaries;
15
+ 2. assign — file every tool into those fixed names, using the full blurbs;
16
+ 3. review — show the model what each category ended up holding, and let it
17
+ move the misfits and rename a category to match its actual contents.
18
+
19
+ Step 3 is a patch, not a rewrite: only valid moves and renames are applied, so
20
+ a confused reply degrades to the step-2 result rather than replacing it.
21
+
22
+ A rebuild only happens when the corpus fingerprint moves (see
23
+ groups.ensure_groups), so this is about three cheap calls per documentation
24
+ edit. When it cannot run, build.py falls back to the offline embedding path.
25
+ """
26
+
27
+ import json
28
+ import math
29
+ import os
30
+ import re
31
+ import time
32
+ import urllib.error
33
+ import urllib.request
34
+ from typing import Dict, List, Optional, Sequence, Tuple
35
+
36
+ MODEL = "gemini-3.5-flash-lite"
37
+ # Tried in order on the local OpenAI-compatible server (LM Studio). The small
38
+ # one first: this is a classification job, not a writing job.
39
+ LOCAL_MODELS = ("google/gemma-4-e2b", "qwen/qwen3.6-35b-a3b")
40
+ GEMINI_TIMEOUT = 60.0
41
+ LOCAL_TIMEOUT = 300.0
42
+ # A local model is small and its context is short (the default gemma-4-e2b load
43
+ # offers 4096 tokens, and the assignment prompt for 36 tools is 4145). On that
44
+ # backend the tools are described by their one-line summary and filed in
45
+ # batches that fit this budget, rather than all at once with full blurbs.
46
+ LOCAL_TOKEN_BUDGET = 2400
47
+
48
+ # LM Studio rejects response_format "json_object" ("must be 'json_schema' or
49
+ # 'text'"), and a small model holds its shape far better with a schema than
50
+ # with an instruction, so each step declares one.
51
+ _STRING_MAP = {"type": "object", "additionalProperties": {"type": "string"}}
52
+ CATEGORIES_SCHEMA = {
53
+ "type": "object",
54
+ "properties": {
55
+ "categories": {
56
+ "type": "array",
57
+ "items": {
58
+ "type": "object",
59
+ "properties": {
60
+ "name": {"type": "string"},
61
+ "scope": {"type": "string"},
62
+ },
63
+ "required": ["name"],
64
+ },
65
+ }
66
+ },
67
+ "required": ["categories"],
68
+ }
69
+ ASSIGNMENT_SCHEMA = {
70
+ "type": "object",
71
+ "properties": {"assignment": _STRING_MAP},
72
+ "required": ["assignment"],
73
+ }
74
+ REVIEW_SCHEMA = {
75
+ "type": "object",
76
+ "properties": {"moves": _STRING_MAP, "renames": _STRING_MAP},
77
+ }
78
+ MAX_NAME_CHARS = 24
79
+ _RETRIES = 2
80
+ _BANNED_NAMES = {
81
+ "misc", "miscellaneous", "other", "others", "utilities", "utility",
82
+ "tools", "general", "various", "assorted", "extras",
83
+ }
84
+
85
+
86
+ class LLMGroupingUnavailable(RuntimeError):
87
+ """The model could not be reached, or would not answer usably."""
88
+
89
+
90
+ def model_name() -> str:
91
+ """Gemini model used for the grouping; TOOLS_GROUPS_MODEL overrides."""
92
+ return os.getenv("TOOLS_GROUPS_MODEL", "") or MODEL
93
+
94
+
95
+ def local_models() -> List[str]:
96
+ """Local models to try, in order; TOOLS_GROUPS_LOCAL_MODEL overrides."""
97
+ pinned = os.getenv("TOOLS_GROUPS_LOCAL_MODEL", "").strip()
98
+ return [pinned] if pinned else list(LOCAL_MODELS)
99
+
100
+
101
+ def _have_api_key() -> bool:
102
+ """True when a Gemini key is reachable (the repo .env counts)."""
103
+ try:
104
+ from .embedder import _load_env
105
+ _load_env()
106
+ except ImportError: # pragma: no cover - depends on the venv
107
+ pass
108
+ return bool(os.getenv("GEMINI_API_KEY", "").strip())
109
+
110
+
111
+ def backend_order() -> List[str]:
112
+ """Which backends to try, in order.
113
+
114
+ TOOLS_GROUPS_BACKEND pins one ("local" or "gemini"). The default is to use
115
+ Gemini when a key is present and the local LM Studio server otherwise —
116
+ so this fleet keeps the better model, and a checkout without a key still
117
+ gets a real grouping instead of the capability fallback.
118
+ """
119
+ pinned = os.getenv("TOOLS_GROUPS_BACKEND", "").strip().lower()
120
+ if pinned in ("local", "gemini"):
121
+ return [pinned]
122
+ return ["gemini", "local"] if _have_api_key() else ["local", "gemini"]
123
+
124
+
125
+ def size_band(n: int, k: int) -> Tuple[int, int]:
126
+ """Smallest and largest sensible category size for n tools in k groups."""
127
+ return max(2, (n // k) // 2), math.ceil(n / k) + 3
128
+
129
+
130
+ # --- prompts --------------------------------------------------------------
131
+
132
+ def _one_liner(name: str, blurb: str) -> str:
133
+ """The advertised summary line of a blurb, for the naming call."""
134
+ for line in blurb.splitlines():
135
+ if line.startswith("advertises:"):
136
+ return f"{name}: {line[len('advertises:'):].strip()[:120]}"
137
+ return f"{name}: {blurb.splitlines()[-1][:120] if blurb else ''}"
138
+
139
+
140
+ def naming_prompt(blurbs: Dict[str, str], k: int) -> str:
141
+ """Stage one: propose the k category names from one-line summaries."""
142
+ n = len(blurbs)
143
+ low, high = size_band(n, k)
144
+ listing = "\n".join(_one_liner(name, blurbs[name]) for name in sorted(blurbs))
145
+ return (
146
+ f"Here are {n} personal command-line and GUI tools, one per line as "
147
+ "'id: advertised name | description | capability word'.\n\n"
148
+ f"{listing}\n\n"
149
+ f"Propose exactly {k} category names to file them under in a graphical "
150
+ "installer's sidebar. Each name is one or two words, title case, at "
151
+ f"most {MAX_NAME_CHARS} characters, and names a subject a user would "
152
+ "look under. No catch-all ('Misc', 'Other', 'Utilities', 'Tools', "
153
+ f"'General'). Together the {k} must cover all {n} tools with roughly "
154
+ f"{low}-{high} tools each, so choose boundaries that split the "
155
+ "collection evenly rather than one broad name that would absorb "
156
+ "everything left over.\n"
157
+ 'Answer as JSON: {"categories": [{"name": "...", "scope": "one line '
158
+ 'on what belongs here"}]}'
159
+ )
160
+
161
+
162
+ def assignment_prompt(
163
+ blurbs: Dict[str, str],
164
+ categories: Sequence[dict],
165
+ k: int,
166
+ total: Optional[int] = None,
167
+ ) -> str:
168
+ """Step two: file every tool under one of the fixed category names.
169
+
170
+ ``total`` is the size of the whole collection when this prompt covers only
171
+ a batch of it, which changes the size guidance: a batch must not be
172
+ balanced against itself.
173
+ """
174
+ n = len(blurbs)
175
+ block = "\n".join(
176
+ f"- {c['name']}: {c.get('scope', '')}" for c in categories
177
+ )
178
+ bodies = "\n\n".join(f"### {name}\n{blurbs[name]}" for name in sorted(blurbs))
179
+ if total and total != n:
180
+ sizing = (
181
+ f"These {n} are one batch of {total} tools being filed into the "
182
+ "same categories, so do not try to balance the batch: file each "
183
+ "tool where it belongs, even if that fills one category."
184
+ )
185
+ else:
186
+ low, high = size_band(n, k)
187
+ sizing = f"Each category ends up with roughly {low}-{high} tools."
188
+ return (
189
+ f"File each of these {n} tools under exactly one of the {k} "
190
+ f"categories.\n\nCategories:\n{block}\n\n"
191
+ f"{sizing}\n"
192
+ "Rules: every tool id appears exactly once; use the ids verbatim, "
193
+ "with their original capitalisation; file by subject and ignore the "
194
+ "install, git and sync prose a description may carry.\n"
195
+ 'Answer as JSON: {"assignment": {"tool_id": "Category Name", ...}}\n\n'
196
+ f"{bodies}"
197
+ )
198
+
199
+
200
+ def _estimate_tokens(text: str) -> int:
201
+ """Rough token count — four characters per token is close enough here."""
202
+ return len(text) // 4 + 1
203
+
204
+
205
+ def batch_for_budget(
206
+ blurbs: Dict[str, str], overhead: int, budget: int = LOCAL_TOKEN_BUDGET
207
+ ) -> List[List[str]]:
208
+ """Split tool names into batches whose prompts fit the token budget."""
209
+ batches: List[List[str]] = [[]]
210
+ used = overhead
211
+ for name in sorted(blurbs):
212
+ cost = _estimate_tokens(blurbs[name]) + 8
213
+ if batches[-1] and used + cost > budget:
214
+ batches.append([])
215
+ used = overhead
216
+ batches[-1].append(name)
217
+ used += cost
218
+ return [batch for batch in batches if batch]
219
+
220
+
221
+ # --- reply handling -------------------------------------------------------
222
+
223
+ def _clean_name(raw) -> str:
224
+ """Tidy a category name and keep it inside MAX_NAME_CHARS.
225
+
226
+ An over-long name is cut back to a whole word: "System & Development Tools"
227
+ truncated blind reads "System & Development Too".
228
+ """
229
+ name = str(raw or "").strip().strip('"').strip()
230
+ if not name or name.lower() in _BANNED_NAMES:
231
+ return ""
232
+ if len(name) > MAX_NAME_CHARS:
233
+ cut = name[:MAX_NAME_CHARS + 1]
234
+ if " " in cut.strip():
235
+ cut = cut[:cut.rstrip().rfind(" ")]
236
+ name = cut.strip()
237
+ return name.strip(" &-,/·").strip()
238
+
239
+
240
+ def _extract_json(text: str) -> dict:
241
+ """Parse the reply, tolerating a code fence or prose around the JSON."""
242
+ text = (text or "").strip()
243
+ if not text:
244
+ raise ValueError("empty reply")
245
+ fenced = re.search(r"```(?:json)?\s*(.*?)```", text, re.S)
246
+ if fenced:
247
+ text = fenced.group(1).strip()
248
+ try:
249
+ return json.loads(text)
250
+ except json.JSONDecodeError:
251
+ pass
252
+ start, end = text.find("{"), text.rfind("}")
253
+ if start == -1 or end <= start:
254
+ raise ValueError("no JSON object in the reply")
255
+ return json.loads(text[start:end + 1])
256
+
257
+
258
+ def parse_categories(reply: str, k: int) -> List[dict]:
259
+ """Stage-one reply -> [{"name", "scope"}], names cleaned and deduplicated."""
260
+ data = _extract_json(reply)
261
+ raw = data.get("categories") if isinstance(data, dict) else None
262
+ if not isinstance(raw, list):
263
+ raise ValueError("no 'categories' list in the reply")
264
+
265
+ out: List[dict] = []
266
+ seen = set()
267
+ for entry in raw:
268
+ if isinstance(entry, str):
269
+ entry = {"name": entry}
270
+ if not isinstance(entry, dict):
271
+ continue
272
+ name = _clean_name(entry.get("name"))
273
+ if not name or name.casefold() in seen:
274
+ continue
275
+ seen.add(name.casefold())
276
+ out.append({"name": name, "scope": str(entry.get("scope") or "").strip()})
277
+ if len(out) != k:
278
+ raise ValueError(f"got {len(out)} category names, need exactly {k}")
279
+ return out
280
+
281
+
282
+ def parse_assignment(
283
+ reply: str, names: Sequence[str], categories: Sequence[dict]
284
+ ) -> Tuple[Dict[str, List[str]], List[str]]:
285
+ """Stage-two reply -> {category: [tool, ...]} plus a list of complaints.
286
+
287
+ Never raises on a merely sloppy answer. Tool ids and category names are
288
+ matched case-insensitively (flash-lite writes "llmchat" for "LMChat"),
289
+ unknown ids are dropped, and a tool the model forgot lands in the smallest
290
+ category. The complaints are what a retry is told about.
291
+ """
292
+ data = _extract_json(reply)
293
+ raw = data.get("assignment") if isinstance(data, dict) else None
294
+ if not isinstance(raw, dict) or not raw:
295
+ raise ValueError("no 'assignment' object in the reply")
296
+
297
+ by_tool = {n.casefold(): n for n in names}
298
+ by_category = {c["name"].casefold(): c["name"] for c in categories}
299
+ labels: Dict[str, List[str]] = {c["name"]: [] for c in categories}
300
+ complaints: List[str] = []
301
+ assigned = set()
302
+
303
+ for tool, category in raw.items():
304
+ real = by_tool.get(str(tool).strip().casefold())
305
+ if real is None:
306
+ complaints.append(f"unknown tool {tool!r}")
307
+ continue
308
+ if real in assigned:
309
+ complaints.append(f"{real} assigned twice")
310
+ continue
311
+ label = by_category.get(str(category).strip().casefold())
312
+ if label is None:
313
+ complaints.append(f"{real} put in unknown category {category!r}")
314
+ continue
315
+ labels[label].append(real)
316
+ assigned.add(real)
317
+
318
+ missing = [n for n in names if n not in assigned]
319
+ if missing:
320
+ complaints.append("unassigned: " + ", ".join(missing))
321
+ smallest = min(labels, key=lambda label: (len(labels[label]), label))
322
+ labels[smallest].extend(missing)
323
+
324
+ _, high = size_band(len(names), len(categories))
325
+ for label, members in labels.items():
326
+ if len(members) > high:
327
+ complaints.append(
328
+ f"category {label!r} holds {len(members)} tools, more than {high}"
329
+ )
330
+
331
+ labels = {label: sorted(members) for label, members in labels.items() if members}
332
+ if not labels:
333
+ raise ValueError("no tool was assigned")
334
+ return labels, complaints
335
+
336
+
337
+ def review_prompt(
338
+ labels: Dict[str, List[str]],
339
+ blurbs: Dict[str, str],
340
+ categories: Sequence[dict] = (),
341
+ ) -> str:
342
+ """Step three: show the finished bands and invite a patch.
343
+
344
+ Each band is shown with the scope it was created for, so a move is judged
345
+ against what the category was meant to hold and not only its name.
346
+ """
347
+ scopes = {c["name"]: c.get("scope", "") for c in categories}
348
+ blocks = []
349
+ for label in sorted(labels):
350
+ members = "\n".join(
351
+ " " + _one_liner(name, blurbs.get(name, "")) for name in labels[label]
352
+ )
353
+ scope = scopes.get(label)
354
+ header = f"{label} ({len(labels[label])})"
355
+ if scope:
356
+ header += f" — meant for: {scope}"
357
+ blocks.append(f"{header}:\n{members}")
358
+ return (
359
+ "These are the finished categories of a tool installer's sidebar, each "
360
+ "with the tools filed under it.\n\n" + "\n\n".join(blocks) + "\n\n"
361
+ "Review the result and correct it. Look for a tool whose subject does "
362
+ "not match the category it sits in, and for a category whose name no "
363
+ "longer describes what it actually holds. Judge a tool by what it does "
364
+ "for its user: reading a chat service is not the same subject as "
365
+ "running an AI agent, even though both involve conversations.\n"
366
+ "Rules: move a tool only when another existing category is clearly a "
367
+ "better home — do not shuffle borderline cases; keep every category "
368
+ "non-empty; a new name is one or two words, title case, at most "
369
+ f"{MAX_NAME_CHARS} characters, and must not be a catch-all ('Misc', "
370
+ "'Other', 'Utilities', 'Tools', 'General').\n"
371
+ "Change nothing you are not confident about. An empty patch is a valid "
372
+ "answer.\n"
373
+ 'Answer as JSON: {"moves": {"tool_id": "Destination Category"}, '
374
+ '"renames": {"Old Name": "New Name"}}'
375
+ )
376
+
377
+
378
+ def parse_review(
379
+ reply: str, labels: Dict[str, List[str]]
380
+ ) -> Tuple[Dict[str, List[str]], List[str]]:
381
+ """Apply a review patch to the labels, returning (labels, applied notes).
382
+
383
+ Everything is checked before it is applied: an unknown tool or destination,
384
+ a move that would empty a category, a rename onto an existing name or to a
385
+ banned one, are all ignored. The step-2 result is the floor.
386
+ """
387
+ data = _extract_json(reply)
388
+ if not isinstance(data, dict):
389
+ raise ValueError("review reply is not an object")
390
+
391
+ result = {label: list(members) for label, members in labels.items()}
392
+ home = {tool: label for label, members in result.items() for tool in members}
393
+ notes: List[str] = []
394
+
395
+ moves = data.get("moves")
396
+ if isinstance(moves, dict):
397
+ by_tool = {t.casefold(): t for t in home}
398
+ by_label = {label.casefold(): label for label in result}
399
+ for tool, destination in moves.items():
400
+ real = by_tool.get(str(tool).strip().casefold())
401
+ target = by_label.get(str(destination).strip().casefold())
402
+ if real is None or target is None or target == home[real]:
403
+ continue
404
+ source = home[real]
405
+ if len(result[source]) <= 1:
406
+ continue # never empty a category
407
+ result[source].remove(real)
408
+ result[target].append(real)
409
+ home[real] = target
410
+ notes.append(f"moved {real}: {source} -> {target}")
411
+
412
+ renames = data.get("renames")
413
+ if isinstance(renames, dict):
414
+ by_label = {label.casefold(): label for label in result}
415
+ for old, new in renames.items():
416
+ source = by_label.get(str(old).strip().casefold())
417
+ name = _clean_name(new)
418
+ if source is None or not name or name == source:
419
+ continue
420
+ if name.casefold() in {label.casefold() for label in result}:
421
+ continue
422
+ result[name] = result.pop(source)
423
+ by_label.pop(source.casefold(), None)
424
+ by_label[name.casefold()] = name
425
+ notes.append(f"renamed {source} -> {name}")
426
+
427
+ return {label: sorted(members) for label, members in result.items()}, notes
428
+
429
+
430
+ # --- the call ------------------------------------------------------------
431
+
432
+ def _gemini_rest(prompt: str) -> str:
433
+ """One generateContent call over plain HTTP. "" when it cannot be made.
434
+
435
+ The fallback for anyone without the first-party GeminiClient: it needs
436
+ nothing but GEMINI_API_KEY and the standard library, which is what makes
437
+ the LLM tier available to a third-party checkout at all.
438
+ """
439
+ api_key = os.getenv("GEMINI_API_KEY", "")
440
+ if not api_key:
441
+ return ""
442
+ url = (f"https://generativelanguage.googleapis.com/v1beta/models/"
443
+ f"{model_name()}:generateContent?key={api_key}")
444
+ payload = json.dumps({
445
+ "contents": [{"parts": [{"text": prompt}]}],
446
+ "generationConfig": {"temperature": 0, "responseMimeType": "application/json"},
447
+ }).encode()
448
+ req = urllib.request.Request(
449
+ url, data=payload, headers={"Content-Type": "application/json"})
450
+ try:
451
+ with urllib.request.urlopen(req, timeout=GEMINI_TIMEOUT) as resp:
452
+ data = json.loads(resp.read().decode())
453
+ except Exception:
454
+ return ""
455
+ try:
456
+ parts = data["candidates"][0]["content"]["parts"]
457
+ except (KeyError, IndexError, TypeError):
458
+ return ""
459
+ return "".join(p.get("text", "") for p in parts if isinstance(p, dict))
460
+
461
+
462
+ def _ask_gemini(prompt: str, fresh: bool = False) -> str:
463
+ """One Gemini call. Raises LLMGroupingUnavailable if it cannot be made.
464
+
465
+ Prefers the first-party GeminiClient when it is importable — it brings the
466
+ response cache and the model rotation — and falls back to a direct REST
467
+ call, so a checkout without it still gets the LLM grouping.
468
+ """
469
+ try:
470
+ from .embedder import _load_env
471
+ _load_env() # GEMINI_API_KEY lives in the tree's .env
472
+ except ImportError: # pragma: no cover - defensive
473
+ pass
474
+
475
+ try:
476
+ from _shared.gemini import GeminiClient
477
+ except ImportError:
478
+ GeminiClient = None
479
+
480
+ if GeminiClient is not None:
481
+ try:
482
+ client = GeminiClient(models=[model_name()])
483
+ reply = client.generate(prompt, json_mode=True, label="", skip_cache=fresh)
484
+ if reply:
485
+ return reply
486
+ except Exception:
487
+ pass # fall through to REST rather than losing the tier
488
+
489
+ reply = _gemini_rest(prompt)
490
+ if not reply:
491
+ raise LLMGroupingUnavailable("Gemini returned nothing")
492
+ return reply
493
+
494
+
495
+ def _local_host() -> str:
496
+ """Base URL of the local OpenAI-compatible server."""
497
+ try:
498
+ from .embedder import local_host
499
+ return local_host()
500
+ except ImportError: # pragma: no cover - depends on the venv
501
+ host = os.getenv("TOOLS_EMBED_HOST", "") or "http://localhost:11434"
502
+ return (host if host.startswith("http") else f"http://{host}").rstrip("/")
503
+
504
+
505
+ def _chat_local(prompt: str, model: str, schema: Optional[dict] = None) -> str: # noqa: C901
506
+ """One /v1/chat/completions call. Returns "" on any failure.
507
+
508
+ Retries once without the schema: an older server, or one whose model has no
509
+ grammar support, rejects response_format rather than ignoring it.
510
+ """
511
+ for use_schema in (bool(schema), False):
512
+ body = {
513
+ "model": model,
514
+ "messages": [{"role": "user", "content": prompt}],
515
+ "temperature": 0,
516
+ }
517
+ if use_schema:
518
+ body["response_format"] = {
519
+ "type": "json_schema",
520
+ "json_schema": {
521
+ "name": "tool_grouping", "strict": True, "schema": schema,
522
+ },
523
+ }
524
+ request = urllib.request.Request(
525
+ f"{_local_host()}/v1/chat/completions",
526
+ data=json.dumps(body).encode(),
527
+ headers={"Content-Type": "application/json"},
528
+ )
529
+ left = _remaining()
530
+ timeout = LOCAL_TIMEOUT if left is None else min(LOCAL_TIMEOUT, max(1.0, left))
531
+ try:
532
+ with urllib.request.urlopen(request, timeout=timeout) as response:
533
+ data = json.loads(response.read().decode())
534
+ except (urllib.error.URLError, OSError, json.JSONDecodeError, ValueError):
535
+ continue
536
+ try:
537
+ content = data["choices"][0]["message"]["content"] or ""
538
+ except (KeyError, IndexError, TypeError):
539
+ continue
540
+ if content:
541
+ return content
542
+ return ""
543
+
544
+
545
+ # The backend and model that answered first; the rest of the pipeline stays on
546
+ # it rather than re-probing before every step.
547
+ _ACTIVE: Optional[Tuple[str, str]] = None
548
+ # Wall-clock deadline for the whole pipeline, or None for no limit. The
549
+ # installer sets one because it calls this on the way to opening a window; a
550
+ # local model on a busy host can take minutes per batch, and a GUI that waits
551
+ # for it is worse than a grouping filed by capability words.
552
+ _DEADLINE: Optional[float] = None
553
+
554
+
555
+ def _remaining() -> Optional[float]:
556
+ """Seconds left of the deadline, or None when there is none."""
557
+ return None if _DEADLINE is None else _DEADLINE - time.monotonic()
558
+
559
+
560
+ def active_backend() -> str:
561
+ """Which backend answered ("local", "gemini"), or "" before the first call."""
562
+ return _ACTIVE[0] if _ACTIVE else ""
563
+
564
+
565
+ def active_label() -> str:
566
+ """What produced the current grouping, e.g. "local:google/gemma-4-e2b"."""
567
+ if _ACTIVE is None:
568
+ return ""
569
+ backend, model = _ACTIVE
570
+ return f"{backend}:{model}"
571
+
572
+
573
+ def _ask(prompt: str, schema: Optional[dict] = None, fresh: bool = False) -> str:
574
+ """Ask whichever backend is available, and stay on the one that answers.
575
+
576
+ ``fresh`` bypasses the response cache. The naming step uses it: an
577
+ identical prompt otherwise replays one answer forever, so a mediocre set of
578
+ names could never be shaken off by asking again.
579
+ """
580
+ global _ACTIVE
581
+
582
+ left = _remaining()
583
+ if left is not None and left <= 0:
584
+ raise LLMGroupingUnavailable("out of time")
585
+
586
+ if _ACTIVE is not None:
587
+ backend, model = _ACTIVE
588
+ if backend == "gemini":
589
+ return _ask_gemini(prompt, fresh)
590
+ reply = _chat_local(prompt, model, schema)
591
+ if reply:
592
+ return reply
593
+ raise LLMGroupingUnavailable(f"local model {model} stopped answering")
594
+
595
+ problems = []
596
+ for backend in backend_order():
597
+ if backend == "gemini":
598
+ try:
599
+ reply = _ask_gemini(prompt, fresh)
600
+ except LLMGroupingUnavailable as exc:
601
+ problems.append(str(exc))
602
+ continue
603
+ _ACTIVE = ("gemini", model_name())
604
+ return reply
605
+ for model in local_models():
606
+ reply = _chat_local(prompt, model, schema)
607
+ if reply:
608
+ _ACTIVE = ("local", model)
609
+ return reply
610
+ problems.append(f"local model {model} did not answer")
611
+ raise LLMGroupingUnavailable("; ".join(problems) or "no backend configured")
612
+
613
+
614
+ def llm_groups(
615
+ blurbs: Dict[str, str],
616
+ k: int,
617
+ budget: Optional[float] = None,
618
+ categories: Optional[Sequence[dict]] = None,
619
+ ) -> Dict[str, List[str]]:
620
+ """Category name -> members, decided by the model.
621
+
622
+ ``categories`` skips the naming step and files into names that already
623
+ exist. The installer passes the stored ones: a rebuild after an edited
624
+ README should not rename every band, and asking for six fresh names each
625
+ time produced overlapping ones ("Artificial Intelligence" beside "AI
626
+ Agents"). The review step can still rename one whose contents drifted.
627
+
628
+ ``budget`` caps the whole pipeline in seconds; past it the next call raises
629
+ rather than starting. Raises LLMGroupingUnavailable when the model cannot
630
+ be reached, its answers stay unusable, or the budget runs out.
631
+ """
632
+ global _ACTIVE, _DEADLINE
633
+ _ACTIVE = None
634
+ _DEADLINE = None if budget is None else time.monotonic() + budget
635
+
636
+ names = sorted(blurbs)
637
+ if not names:
638
+ raise ValueError("no tools to group")
639
+
640
+ if categories and len(categories) == k:
641
+ return _fill(blurbs, list(categories), k, names)
642
+ return _fill(blurbs, None, k, names)
643
+
644
+
645
+ def _fill(
646
+ blurbs: Dict[str, str],
647
+ categories: Optional[List[dict]],
648
+ k: int,
649
+ names: Sequence[str],
650
+ ) -> Dict[str, List[str]]:
651
+ """Name the categories if needed, then assign and review."""
652
+ problem = ""
653
+ for _ in range(_RETRIES):
654
+ if categories is not None:
655
+ break
656
+ prompt = naming_prompt(blurbs, k)
657
+ if problem:
658
+ prompt += f"\n\nYour previous answer was rejected: {problem}."
659
+ try:
660
+ reply = _ask(prompt, CATEGORIES_SCHEMA, fresh=True)
661
+ categories = parse_categories(reply, k)
662
+ break
663
+ except (ValueError, json.JSONDecodeError) as exc:
664
+ problem = str(exc)
665
+ if categories is None:
666
+ raise LLMGroupingUnavailable(f"could not name {k} categories: {problem}")
667
+
668
+ if active_backend() == "local":
669
+ labels = _assign_in_batches(blurbs, categories, k, names)
670
+ return _review(labels, _compact(blurbs), categories)
671
+
672
+ problem = ""
673
+ best: Dict[str, List[str]] = {}
674
+ for attempt in range(_RETRIES):
675
+ prompt = assignment_prompt(blurbs, categories, k)
676
+ if problem:
677
+ prompt += (
678
+ f"\n\nYour previous answer was rejected: {problem}. Answer "
679
+ "again, covering every tool exactly once and keeping the "
680
+ "categories to a sensible size."
681
+ )
682
+ try:
683
+ reply = _ask(prompt, ASSIGNMENT_SCHEMA)
684
+ labels, complaints = parse_assignment(reply, names, categories)
685
+ except (ValueError, json.JSONDecodeError) as exc:
686
+ problem = str(exc)
687
+ continue
688
+ if not complaints:
689
+ return _review(labels, blurbs, categories)
690
+ best = best or labels
691
+ if attempt == _RETRIES - 1:
692
+ # repaired: usable, if not pristine
693
+ return _review(best, blurbs, categories)
694
+ problem = "; ".join(complaints[:5])
695
+ raise LLMGroupingUnavailable(f"unusable assignment: {problem}")
696
+
697
+
698
+ def _compact(blurbs: Dict[str, str]) -> Dict[str, str]:
699
+ """One line per tool — what a short-context model gets instead of blurbs."""
700
+ return {name: _one_liner(name, blurb) for name, blurb in blurbs.items()}
701
+
702
+
703
+ def _assign_in_batches(
704
+ blurbs: Dict[str, str],
705
+ categories: Sequence[dict],
706
+ k: int,
707
+ names: Sequence[str],
708
+ ) -> Dict[str, List[str]]:
709
+ """Step two for a short-context backend: compact input, several batches.
710
+
711
+ A batch that fails is not fatal — its tools come back unassigned and the
712
+ usual repair files them, so one bad reply costs a few placements rather
713
+ than the whole grouping.
714
+ """
715
+ compact = _compact(blurbs)
716
+ overhead = _estimate_tokens(assignment_prompt({}, categories, k, total=len(names)))
717
+ merged: Dict[str, str] = {}
718
+ for batch in batch_for_budget(compact, overhead):
719
+ subset = {name: compact[name] for name in batch}
720
+ prompt = assignment_prompt(subset, categories, k, total=len(names))
721
+ try:
722
+ data = _extract_json(_ask(prompt, ASSIGNMENT_SCHEMA))
723
+ except (LLMGroupingUnavailable, ValueError, json.JSONDecodeError):
724
+ continue
725
+ assignment = data.get("assignment") if isinstance(data, dict) else None
726
+ if isinstance(assignment, dict):
727
+ merged.update({str(t): str(c) for t, c in assignment.items()})
728
+
729
+ if not merged:
730
+ raise LLMGroupingUnavailable("no batch of the assignment step answered")
731
+ labels, _ = parse_assignment(json.dumps({"assignment": merged}), names, categories)
732
+ return labels
733
+
734
+
735
+ def _review(
736
+ labels: Dict[str, List[str]],
737
+ blurbs: Dict[str, str],
738
+ categories: Sequence[dict] = (),
739
+ ) -> Dict[str, List[str]]:
740
+ """Step three, best effort: a failed review keeps the step-2 result."""
741
+ try:
742
+ reply = _ask(review_prompt(labels, blurbs, categories), REVIEW_SCHEMA)
743
+ reviewed, notes = parse_review(reply, labels)
744
+ except (LLMGroupingUnavailable, ValueError, json.JSONDecodeError):
745
+ return labels
746
+ for note in notes:
747
+ print(f" review: {note}")
748
+ return reviewed