blun-king-cli 9.1.527 → 9.1.550

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (108) hide show
  1. package/LIESMICH.txt +13 -834
  2. package/README.md +41 -799
  3. package/bin/agent-resume-snapshot.cjs +31 -0
  4. package/bin/assistant-message-offload-policy.cjs +23 -2
  5. package/bin/codebase-search-runtime.cjs +23 -0
  6. package/bin/context-performance-policy.cjs +2 -5
  7. package/bin/context-pressure-policy.cjs +20 -0
  8. package/bin/cron-run-output.cjs +45 -0
  9. package/bin/cron-run-store.cjs +145 -0
  10. package/bin/durable-task-resume-policy.cjs +130 -0
  11. package/bin/durable-task-resume-runtime.cjs +117 -0
  12. package/bin/durable-task-resume-store.cjs +88 -0
  13. package/bin/editable-tool-approval-policy.cjs +540 -0
  14. package/bin/editable-tool-approval-runtime.cjs +99 -0
  15. package/bin/empty-response-retry-policy.cjs +29 -0
  16. package/bin/history-offload-pressure-policy.cjs +33 -0
  17. package/bin/html-to-research-markdown.cjs +146 -0
  18. package/bin/programmatic-context-isolation.cjs +25 -0
  19. package/bin/programmatic-tool-runtime.mjs +627 -0
  20. package/bin/read-continuation-policy.cjs +36 -5
  21. package/bin/scoped-cron-run-policy.cjs +358 -0
  22. package/bin/skill-activation-performance-policy.cjs +9 -0
  23. package/bin/startup-preferences.cjs +3 -3
  24. package/bin/structured-agent-swarm-output.cjs +325 -0
  25. package/bin/structured-subagent-output.cjs +252 -0
  26. package/bin/subagent-context-fork-policy.cjs +155 -0
  27. package/bin/subagent-skill-policy.cjs +204 -0
  28. package/bin/telegram-approval-relay.cjs +2 -1
  29. package/bin/telegram-direct-focus-policy.cjs +25 -1
  30. package/bin/todo-list-turn-policy.cjs +111 -1
  31. package/bin/tool-result-offload-policy.cjs +43 -1
  32. package/bin/turn-thinking-policy.cjs +6 -15
  33. package/bin/turn-tool-performance-policy.cjs +5 -4
  34. package/bin/update-notice.js +14 -18
  35. package/bin/user-message-offload-policy.cjs +12 -1
  36. package/blun.mjs +1406 -265
  37. package/codebase-index/README.md +82 -0
  38. package/codebase-index/codebase_index.py +469 -0
  39. package/package.json +24 -38
  40. package/telegram-plugin/bin/telegram-typing-keepalive.cjs +89 -0
  41. package/telegram-plugin/dist/bridge.mjs +8 -1
  42. package/CHANGELOG.md +0 -260
  43. package/agent-spine-plugin/CHANGELOG.md +0 -406
  44. package/agent-spine-plugin/CONTRIBUTING.md +0 -52
  45. package/agent-spine-plugin/README.md +0 -344
  46. package/agent-spine-plugin/SECURITY.md +0 -47
  47. package/agent-spine-plugin/docs/acceptance.md +0 -61
  48. package/agent-spine-plugin/docs/architecture.md +0 -183
  49. package/agent-spine-plugin/docs/attention.md +0 -121
  50. package/agent-spine-plugin/docs/automatic-continuity.md +0 -79
  51. package/agent-spine-plugin/docs/channel-runtime.md +0 -92
  52. package/agent-spine-plugin/docs/coordination.md +0 -138
  53. package/agent-spine-plugin/docs/feed-transport.md +0 -99
  54. package/agent-spine-plugin/docs/gateway-runtime.md +0 -116
  55. package/agent-spine-plugin/docs/harness-reference.md +0 -45
  56. package/agent-spine-plugin/docs/host-integration.md +0 -129
  57. package/agent-spine-plugin/docs/https-transport.md +0 -116
  58. package/agent-spine-plugin/docs/learning.md +0 -133
  59. package/agent-spine-plugin/docs/object-transport.md +0 -93
  60. package/agent-spine-plugin/docs/peer-transport.md +0 -88
  61. package/agent-spine-plugin/docs/preflight-recall.md +0 -69
  62. package/agent-spine-plugin/docs/preservation-contract.md +0 -53
  63. package/agent-spine-plugin/docs/quality-gates.md +0 -50
  64. package/agent-spine-plugin/docs/relationships.md +0 -73
  65. package/agent-spine-plugin/docs/releasing.md +0 -83
  66. package/agent-spine-plugin/docs/roadmap.md +0 -307
  67. package/agent-spine-plugin/docs/selfstarter.md +0 -88
  68. package/agent-spine-plugin/docs/session-briefing.md +0 -74
  69. package/agent-spine-plugin/docs/shared-memory.md +0 -259
  70. package/agent-spine-plugin/docs/source-roots.md +0 -86
  71. package/agent-spine-plugin/docs/sqlite-transport.md +0 -76
  72. package/agent-spine-plugin/scripts/check-hosts.js +0 -195
  73. package/agent-spine-plugin/scripts/check-install.js +0 -569
  74. package/agent-spine-plugin/scripts/check-syntax.js +0 -29
  75. package/agent-spine-plugin/scripts/github-actions.js +0 -11
  76. package/agent-spine-plugin/scripts/release-check.js +0 -128
  77. package/agent-spine-plugin/scripts/run-acceptance.js +0 -19
  78. package/agent-spine-plugin/scripts/run-checks.js +0 -46
  79. package/agent-spine-plugin/scripts/run-tests-hermetic.js +0 -73
  80. package/agent-spine-plugin/spine-example/1-identity.md +0 -12
  81. package/agent-spine-plugin/spine-example/2-voice.md +0 -6
  82. package/agent-spine-plugin/spine-example/3-conduct.md +0 -8
  83. package/agent-spine-plugin/spine-example/4-history.md +0 -4
  84. package/bin/package-regression-policy.cjs +0 -77
  85. package/release-planned-removals.json +0 -15
  86. package/scripts/check-active-profile-plugin-startup.js +0 -36
  87. package/scripts/check-approval-observability-regression.js +0 -111
  88. package/scripts/check-approval-queue-shortcuts-regression.js +0 -65
  89. package/scripts/check-bundled-agent-spine-regression.js +0 -48
  90. package/scripts/check-copy-command-regression.js +0 -74
  91. package/scripts/check-historical-tool-result-preview-regression.js +0 -77
  92. package/scripts/check-mcp-startup-wait-budget.js +0 -48
  93. package/scripts/check-package-regression.js +0 -38
  94. package/scripts/check-plugin-startup-regression.js +0 -53
  95. package/scripts/check-queue-controls-regression.js +0 -189
  96. package/scripts/check-release-metadata.js +0 -103
  97. package/scripts/check-reload-agent-spine-regression.js +0 -76
  98. package/scripts/check-resume-replay-regression.js +0 -100
  99. package/scripts/check-session-cancel-regression.js +0 -43
  100. package/scripts/check-session-picker-resume-metrics-regression.js +0 -97
  101. package/scripts/check-session-start-hook-context-regression.js +0 -228
  102. package/scripts/check-shell-terminal-isolation-regression.js +0 -81
  103. package/scripts/check-slash-escape-regression.js +0 -89
  104. package/scripts/check-telegram-bridge-watchdog.js +0 -60
  105. package/scripts/check-telegram-loop-exactly-once-regression.js +0 -71
  106. package/scripts/check-todo-loop-regression.js +0 -78
  107. package/scripts/check-todo-recovery-catalog-regression.js +0 -50
  108. /package/{scripts → bin}/fix-node-pty-perms.js +0 -0
@@ -0,0 +1,82 @@
1
+ # codebase-index
2
+
3
+ Lokaler semantischer Index über einen Git-Code-Baum — F1 aus Papas
4
+ Feature-Liste. Konzept + Messprotokoll:
5
+ `handoffs/codebase-verstaendnis-lokal-konzept.md`.
6
+
7
+ Der Index ist ein **Lese-Einstieg, keine Wahrheit**: exakte Namen bleiben
8
+ `grep`/`glob`; Begriffe ohne bekannten Namen gehen an den Index. Jede
9
+ Antwort trägt die Selbstauskunft ("semantisch, kann falsch liegen") und
10
+ den Index-Stand.
11
+
12
+ ## Voraussetzungen
13
+
14
+ Auf dieser Maschine bereits vorhanden (kein Installations-Kapitel):
15
+ Python 3.11, `fastembed` 0.8.0 (ONNX, CPU), `numpy`, `psutil`.
16
+ Modell: `BAAI/bge-small-en-v1.5` (33 MB, einmaliger HF-Cache-Download).
17
+
18
+ ## Benutzung
19
+
20
+ ```bash
21
+ # Erstaufbau (Hintergrund-Task! ~40 min auf dem App-Baum, 38k Chunks)
22
+ python codebase_index.py build <repo>
23
+
24
+ # Delta nach Änderungen (Content-Hash je Datei; Ziel < 30 s)
25
+ python codebase_index.py update <repo>
26
+
27
+ # Frage stellen (Top-5 mit Score, Frische, Selbstauskunft)
28
+ python codebase_index.py query <repo> "wo wird die retry-Kappe gesetzt?"
29
+
30
+ # Qualitäts-Gate: Stichprobe aus dem Subjekt-Baum, >= 2/3 in Top-5
31
+ python codebase_index.py sample <repo>
32
+ ```
33
+
34
+ `query` und `sample` waehlen standardmaessig `--model auto`: Ein vorhandener
35
+ Jina-Code-Index wird wegen des gemessenen 3/3-Qualitaetsgates bevorzugt, ein
36
+ vorhandener BGE-Index bleibt Fallback. `--model jina-code` und `--model bge`
37
+ waehlen explizit. Build und Update bleiben modellgebunden und akzeptieren kein
38
+ `auto`.
39
+
40
+ Vorhandene Indizes werden ueber die kanonische Repository-Identitaet im
41
+ Manifest gefunden, auch wenn ein aelterer BLUN-Stand einen anderen
42
+ Workspace-Hash als Ordnernamen erzeugt hat. Fremde Repositories, unbekannte
43
+ Modelle, falsche Dimensionen, fehlende Vektordateien und Links aus dem
44
+ Index-State heraus werden ignoriert.
45
+
46
+ Index-Ort: `~/.blun/codebase-index/<workspace-hash>/` mit
47
+ `vectors.npy` (float32-Matrix), `manifest.json` (Hashes, Spans, Meta,
48
+ Stand) und `query-log.jsonl` (jede Query — der Moat: welche Begriffe
49
+ nichts trafen, wächst mit jeder Session). Löschen des Ordners = sauberer
50
+ Rückweg, Neuaufbau jederzeit reproduzierbar.
51
+
52
+ ## Design-Entscheidungen (alle an Messungen festgemacht)
53
+
54
+ - **Streaming in vorallokierte Matrix** (`np.lib.format.open_memmap`),
55
+ niemals Vektoren-Liste: deterministische Schreibweise, kein
56
+ Doppelbestand. **Korrektur zur Erst-Einordnung:** die ~9-GB-RAM-Spitze
57
+ kommt NICHT von der Liste (~60-100 MB), sondern von der ONNX-Runtime
58
+ (Thread-Pool + Arena) — Run 4 mit memmap läuft ebenfalls bei ~8,9 GB.
59
+ Mitigation (Thread-Begrenzung/Batch) wird separat gemessen.
60
+ - **Inkrementell statt Vollindex:** Vollindex CPU = 2389 s (~40 min,
61
+ Run 3, unbelastet). Erstaufbau als Hintergrund-Task, danach nur Delta:
62
+ Content-Hash je Datei im Manifest, geänderte neu einbetten, gelöschte
63
+ entfernen. Gate: Delta < 30 s (gemessen, nicht geschätzt).
64
+ - **Frische-Auskunft in jeder Query-Antwort:** Index-HEAD vs. Repo-HEAD,
65
+ bei Abweichung "update fällig". Ein Werkzeug ohne Frische-Angabe
66
+ erzeugt Verdikte auf Altstand.
67
+ - **Stichprobe aus dem SUBJEKT-Baum:** Run 3 scheiterte mit 0/3 an einem
68
+ Repo-Mix (Erwartungsdateien aus dem SDK, Subjekt war der App-Baum).
69
+ Die Gate-Queries zeigen auf Dateien, die via `git ls-files` im Subjekt
70
+ verifiziert sind (`api-chat.js`, `terminal-manager.js`,
71
+ `renderer-view-shell.js`).
72
+
73
+ ## Gemessene Werte (App-Baum, 1191 Dateien / 37901 Chunks)
74
+
75
+ | Messung | Wert | Quelle |
76
+ | --- | --- | --- |
77
+ | Vollindex CPU | Run 3: 2389 s (Liste) · Run 4: 2431 s (memmap) | handoffs/f1-benchmark-run3.log, f1-build-run4.log |
78
+ | Query-Latenz | Ø 10,9 ms | Run 3 |
79
+ | Index-Größe | 58,2 MB float32 | Run 3/4 |
80
+ | RAM-Spitze Bau | Run 3: 10,5 GB · Run 4 (memmap): ~8,9-10,2 GB — Treiber ONNX-Arena (Probe: 4,2 GB schon bei 2048 Chunks; OMP_NUM_THREADS 4/2 gemessen: KEIN Effekt auf RSS) | Run 3 rss_peak, Run 4 psutil, f1-onnx-rss-probe |
81
+ | Delta-Update | **3,0 s / 3,1 s am App-Baum (Gate <30 s PASS)** · Mechanik 0,8 s (Mini-Repo) | update-Läufe 03.08. |
82
+ | Stichprobe | V2 (namensbasiert): 0/3 FAIL, Ground-Truth-schwach · **V3 (paraphrasiert): jina-code 3/3 PASS (Top-1 überall), bge 2/3** | sample-Läufe 03.08., Commit fe0b69d |
@@ -0,0 +1,469 @@
1
+ #!/usr/bin/env python3
2
+ """codebase-index — lokaler semantischer Index ueber einen Git-Code-Baum.
3
+
4
+ Lokaler Index nach dem geprueften Streaming-Konzept:
5
+ - Streaming in vorallokierte float32-Matrix (memmap), KEINE Vektoren-Liste
6
+ (Run-3-Befund: Liste trieb RAM-Spitze auf 10,5 GB).
7
+ - Inkrementell: Content-Hash je Datei im Manifest, Delta statt Vollindex.
8
+ - Query mit Frische-Auskunft + Selbstauskunft in JEDER Antwort.
9
+ - Query-Fehlschlag-Log lokal (query-log.jsonl) — der Moat.
10
+
11
+ Aufruf:
12
+ python codebase_index.py build <repo>
13
+ python codebase_index.py update <repo>
14
+ python codebase_index.py query <repo> "<frage>" [--top 5]
15
+ python codebase_index.py sample <repo>
16
+
17
+ Index-Ort: ~/.blun/codebase-index/<workspace-hash>/ (sichtbar, mit Rueckweg).
18
+ """
19
+ import hashlib
20
+ import json
21
+ import os
22
+ import subprocess
23
+ import sys
24
+ import time
25
+ from datetime import datetime, timezone
26
+
27
+ import numpy as np
28
+
29
+ CODE_EXT = (
30
+ ".ts", ".tsx", ".js", ".mjs", ".cjs", ".py", ".html", ".css",
31
+ ".json", ".md", ".toml", ".yml", ".yaml", ".sh", ".ps1", ".sql",
32
+ )
33
+ CHUNK_SIZE = 1000
34
+ CHUNK_OVERLAP = 100
35
+ EMBED_DIM = 384
36
+ BATCH = 256
37
+ MODEL_NAME = "BAAI/bge-small-en-v1.5"
38
+ JINA_CODE_MODEL_NAME = "jinaai/jina-embeddings-v2-base-code"
39
+ KNOWN_MODELS = {
40
+ MODEL_NAME: 384,
41
+ JINA_CODE_MODEL_NAME: 768,
42
+ }
43
+ MODEL_ALIASES = {
44
+ "bge": MODEL_NAME,
45
+ "jina-code": JINA_CODE_MODEL_NAME,
46
+ }
47
+ AUTO_MODEL_ORDER = (JINA_CODE_MODEL_NAME, MODEL_NAME)
48
+
49
+ # Rausch-Filter: Backup-Kopien und Vendor-
50
+ # Buelle machten 66% aller Index-Rows aus und dominieren Top-5.
51
+ EXCLUDE_PREFIXES = ("backup-", "vendor/")
52
+
53
+ # Aktives Modell + Dim zur Laufzeit (wird per --model ueberschrieben).
54
+ _active_model_name = MODEL_NAME
55
+ _active_dim = EMBED_DIM
56
+
57
+ # Stichprobe: Fragen PARAPHRASIERT aus gelesenem
58
+ # Dateiinhalt — keine wörtlichen Bezeichner/Funktionsnamen/Kommentare aus
59
+ # den Dateien (misst semantische Suche, nicht Wortgleichheit).
60
+ # api-chat.js: routet Chat-Nachrichten ans Modell, Fallback-Kette, Stream.
61
+ # terminal-manager.js: startet Shell-Prozesse mit Sandbox-/Approval-Modi.
62
+ # renderer-view-shell.js: wechselt Login-/Willkommens-/Chat-Ansicht.
63
+ SAMPLE_QUERIES = [
64
+ ("how are conversation messages routed to the right model with a fallback chain", "api-chat"),
65
+ ("starting sandboxed shell processes with approval modes and output history", "terminal-manager"),
66
+ ("switching between sign-in screen, welcome page and main conversation view", "renderer-view-shell"),
67
+ ]
68
+
69
+ SELSTAUSKUNFT = (
70
+ "HINWEIS: semantischer Index, kann falsch liegen — Fundstelle vor "
71
+ "Verwendung in der Datei verifizieren."
72
+ )
73
+
74
+
75
+ def canonical_repo_path(repo: str) -> str:
76
+ return os.path.normcase(os.path.realpath(os.path.abspath(repo)))
77
+
78
+
79
+ def index_root() -> str:
80
+ return os.path.join(os.path.expanduser("~"), ".blun", "codebase-index")
81
+
82
+
83
+ def index_dir(repo: str) -> str:
84
+ raw = canonical_repo_path(repo) + "|" + _active_model_name
85
+ key = hashlib.sha256(raw.encode()).hexdigest()[:12]
86
+ return os.path.join(index_root(), key)
87
+
88
+
89
+ def normalize_model_selector(selector: str) -> str:
90
+ if selector == "auto":
91
+ return selector
92
+ model = MODEL_ALIASES.get(selector, selector)
93
+ if model not in KNOWN_MODELS:
94
+ allowed = ", ".join(("auto", *MODEL_ALIASES, *KNOWN_MODELS))
95
+ raise ValueError(f"Unsupported index model {selector!r}; choose one of: {allowed}")
96
+ return model
97
+
98
+
99
+ def find_existing_indexes(repo: str) -> list[tuple[str, dict]]:
100
+ root = index_root()
101
+ if not os.path.isdir(root):
102
+ return []
103
+ canonical_repo = canonical_repo_path(repo)
104
+ canonical_root = os.path.realpath(root)
105
+ candidates = []
106
+ for entry in os.scandir(root):
107
+ if not entry.is_dir(follow_symlinks=False):
108
+ continue
109
+ directory = os.path.realpath(entry.path)
110
+ try:
111
+ if os.path.commonpath((canonical_root, directory)) != canonical_root:
112
+ continue
113
+ except ValueError:
114
+ continue
115
+ manifest_path = os.path.join(directory, "manifest.json")
116
+ vectors_path = os.path.join(directory, "vectors.npy")
117
+ if not os.path.isfile(manifest_path) or not os.path.isfile(vectors_path):
118
+ continue
119
+ try:
120
+ manifest = load_manifest(directory)
121
+ except (OSError, ValueError, TypeError):
122
+ continue
123
+ model = manifest.get("model")
124
+ if model not in KNOWN_MODELS or manifest.get("dim") != KNOWN_MODELS[model]:
125
+ continue
126
+ manifest_repo = manifest.get("repo")
127
+ if not isinstance(manifest_repo, str):
128
+ continue
129
+ if canonical_repo_path(manifest_repo) != canonical_repo:
130
+ continue
131
+ candidates.append((directory, manifest))
132
+ return sorted(candidates, key=lambda item: item[0])
133
+
134
+
135
+ def resolve_existing_index(repo: str, selector: str = "auto") -> tuple[str, dict]:
136
+ requested = normalize_model_selector(selector)
137
+ candidates = find_existing_indexes(repo)
138
+ model_order = AUTO_MODEL_ORDER if requested == "auto" else (requested,)
139
+ for model in model_order:
140
+ matches = [item for item in candidates if item[1]["model"] == model]
141
+ if matches:
142
+ return max(
143
+ matches,
144
+ key=lambda item: (str(item[1].get("built_at", "")), item[0]),
145
+ )
146
+ available = sorted({item[1]["model"] for item in candidates})
147
+ suffix = f"; available for this repository: {', '.join(available)}" if available else ""
148
+ raise FileNotFoundError(
149
+ f"No compatible codebase index for {os.path.abspath(repo)!r} and selector {selector!r}{suffix}"
150
+ )
151
+
152
+
153
+ def activate_model(model: str, dim: int | None = None) -> None:
154
+ global _active_model_name, _active_dim
155
+ normalized = normalize_model_selector(model)
156
+ if normalized == "auto":
157
+ raise ValueError("auto can select an existing index only; it cannot build a new one")
158
+ expected_dim = KNOWN_MODELS[normalized]
159
+ if dim is not None and dim != expected_dim:
160
+ raise ValueError(
161
+ f"Index dimension {dim} does not match {normalized} ({expected_dim})"
162
+ )
163
+ _active_model_name = normalized
164
+ _active_dim = expected_dim
165
+
166
+
167
+ def git_files(repo: str) -> list[str]:
168
+ out = subprocess.run(
169
+ ["git", "ls-files"], cwd=repo, capture_output=True, text=True, check=True
170
+ ).stdout.splitlines()
171
+ return [
172
+ f for f in out
173
+ if f.lower().endswith(CODE_EXT) and not f.startswith(EXCLUDE_PREFIXES)
174
+ ]
175
+
176
+
177
+ def git_head(repo: str) -> str:
178
+ return subprocess.run(
179
+ ["git", "rev-parse", "--short", "HEAD"],
180
+ cwd=repo, capture_output=True, text=True, check=True,
181
+ ).stdout.strip()
182
+
183
+
184
+ def sha_file(path: str) -> str:
185
+ h = hashlib.sha256()
186
+ with open(path, "rb") as fh:
187
+ for block in iter(lambda: fh.read(1 << 20), b""):
188
+ h.update(block)
189
+ return h.hexdigest()
190
+
191
+
192
+ def chunk_text(text: str) -> list[str]:
193
+ step = CHUNK_SIZE - CHUNK_OVERLAP
194
+ return [
195
+ text[i : i + CHUNK_SIZE]
196
+ for i in range(0, len(text), step)
197
+ if text[i : i + CHUNK_SIZE].strip()
198
+ ]
199
+
200
+
201
+ def load_model():
202
+ from fastembed import TextEmbedding
203
+
204
+ return TextEmbedding(_active_model_name)
205
+
206
+
207
+ def collect_chunks(repo: str, files: list[str]):
208
+ """Liest Dateien, liefert (texts, meta, file_spans, read_errors)."""
209
+ texts, meta, spans, read_errors = [], [], {}, 0
210
+ for rel in files:
211
+ try:
212
+ with open(os.path.join(repo, rel), encoding="utf-8", errors="replace") as fh:
213
+ chunks = chunk_text(fh.read())
214
+ except OSError:
215
+ read_errors += 1
216
+ continue
217
+ start = len(texts)
218
+ for idx, chunk in enumerate(chunks):
219
+ texts.append(chunk)
220
+ meta.append(f"{rel}#{idx}")
221
+ spans[rel] = [start, len(texts)]
222
+ return texts, meta, spans, read_errors
223
+
224
+
225
+ def embed_into(model, texts: list[str], matrix, offset: int) -> None:
226
+ """Streamt Embeddings batchweise DIREKT in die (vorallokierte) Matrix."""
227
+ for i in range(0, len(texts), BATCH):
228
+ vecs = list(model.embed(texts[i : i + BATCH]))
229
+ matrix[offset + i : offset + i + len(vecs)] = np.asarray(
230
+ vecs, dtype=np.float32
231
+ )
232
+ if i % (BATCH * 8) == 0:
233
+ print(f"PROGRESS embedded={offset + i}", flush=True)
234
+
235
+
236
+ def write_matrix(dirpath: str, total: int):
237
+ return np.lib.format.open_memmap(
238
+ os.path.join(dirpath, "vectors.npy"),
239
+ mode="w+", dtype=np.float32, shape=(total, _active_dim),
240
+ )
241
+
242
+
243
+ def save_manifest(dirpath: str, repo: str, files_hashes: dict, spans: dict,
244
+ meta: list[str], build_s: float) -> None:
245
+ manifest = {
246
+ "repo": os.path.abspath(repo),
247
+ "head": git_head(repo),
248
+ "built_at": datetime.now(timezone.utc).isoformat(),
249
+ "model": _active_model_name,
250
+ "dim": _active_dim,
251
+ "chunk_size": CHUNK_SIZE,
252
+ "chunk_overlap": CHUNK_OVERLAP,
253
+ "chunks": len(meta),
254
+ "build_s": round(build_s, 2),
255
+ "files": files_hashes,
256
+ "spans": spans,
257
+ "meta": meta,
258
+ }
259
+ with open(os.path.join(dirpath, "manifest.json"), "w", encoding="utf-8") as fh:
260
+ json.dump(manifest, fh)
261
+
262
+
263
+ def load_manifest(dirpath: str) -> dict:
264
+ with open(os.path.join(dirpath, "manifest.json"), encoding="utf-8") as fh:
265
+ return json.load(fh)
266
+
267
+
268
+ def cmd_build(repo: str) -> None:
269
+ t0 = time.perf_counter()
270
+ files = git_files(repo)
271
+ print(f"BUILD files={len(files)} repo={repo}")
272
+ texts, meta, spans, read_errors = collect_chunks(repo, files)
273
+ print(f"CHUNKS total={len(texts)} read_errors={read_errors}")
274
+
275
+ dirpath = index_dir(repo)
276
+ os.makedirs(dirpath, exist_ok=True)
277
+ matrix = write_matrix(dirpath, len(texts))
278
+ model = load_model()
279
+ t_embed = time.perf_counter()
280
+ embed_into(model, texts, matrix, 0)
281
+ matrix.flush()
282
+ build_s = time.perf_counter() - t0
283
+ embed_s = time.perf_counter() - t_embed
284
+
285
+ hashes = {rel: sha_file(os.path.join(repo, rel)) for rel in spans}
286
+ save_manifest(dirpath, repo, hashes, spans, meta, build_s)
287
+ mb = len(texts) * _active_dim * 4 / 1e6
288
+ print(
289
+ f"DONE build_s={build_s:.1f} embed_s={embed_s:.1f} chunks={len(texts)} "
290
+ f"index_mb={mb:.1f} dir={dirpath}",
291
+ flush=True,
292
+ )
293
+
294
+
295
+ def cmd_update(repo: str) -> None:
296
+ """Delta: geaenderte/neue Dateien neu einbetten, geloeschte entfernen."""
297
+ t0 = time.perf_counter()
298
+ dirpath = index_dir(repo)
299
+ man = load_manifest(dirpath)
300
+ old_hashes: dict = man["files"]
301
+ old_spans: dict = man["spans"]
302
+ old_meta: list[str] = man["meta"]
303
+ old_mat = np.load(os.path.join(dirpath, "vectors.npy"))
304
+
305
+ files = git_files(repo)
306
+ texts_new, meta_new, spans_new, read_errors = collect_chunks(repo, files)
307
+ new_hashes = {rel: sha_file(os.path.join(repo, rel)) for rel in spans_new}
308
+
309
+ changed = {r for r, h in new_hashes.items() if old_hashes.get(r) != h}
310
+ deleted = set(old_hashes) - set(new_hashes)
311
+ if not changed and not deleted:
312
+ print(f"UPDATE noop delta_s={time.perf_counter() - t0:.1f} head={git_head(repo)}")
313
+ return
314
+
315
+ # Behaltene Dateien: weder geaendert noch geloescht.
316
+ kept_rels = [r for r in old_spans if r not in changed and r not in deleted]
317
+
318
+ changed_chunks: dict[str, list[str]] = {}
319
+ for rel in sorted(changed):
320
+ with open(os.path.join(repo, rel), encoding="utf-8", errors="replace") as fh:
321
+ changed_chunks[rel] = chunk_text(fh.read())
322
+
323
+ total = sum(old_spans[r][1] - old_spans[r][0] for r in kept_rels) + sum(
324
+ len(c) for c in changed_chunks.values()
325
+ )
326
+ matrix = write_matrix(dirpath, total)
327
+ model = load_model()
328
+
329
+ spans_out, meta_out, cursor = {}, [], 0
330
+ for rel in kept_rels:
331
+ a, b = old_spans[rel]
332
+ matrix[cursor : cursor + (b - a)] = old_mat[a:b]
333
+ spans_out[rel] = [cursor, cursor + (b - a)]
334
+ meta_out.extend(old_meta[a:b])
335
+ cursor += b - a
336
+ for rel in sorted(changed_chunks):
337
+ chunks = changed_chunks[rel]
338
+ embed_into(model, chunks, matrix, cursor)
339
+ spans_out[rel] = [cursor, cursor + len(chunks)]
340
+ meta_out.extend(f"{rel}#{i}" for i in range(len(chunks)))
341
+ cursor += len(chunks)
342
+ matrix.flush()
343
+
344
+ save_manifest(dirpath, repo, new_hashes, spans_out, meta_out,
345
+ time.perf_counter() - t0)
346
+ print(
347
+ f"UPDATE changed={len(changed)} deleted={len(deleted)} "
348
+ f"delta_s={time.perf_counter() - t0:.1f} chunks={total}",
349
+ flush=True,
350
+ )
351
+
352
+
353
+ def freshness(repo: str, man: dict) -> str:
354
+ current = git_head(repo)
355
+ same = "== HEAD" if current == man["head"] else f"Index {man['head']} != HEAD {current} — 'update' faellig"
356
+ return f"Index-Stand: {man['head']} ({man['built_at']}), {man['chunks']} Chunks | {same}"
357
+
358
+
359
+ def cmd_query(repo: str, question: str, top: int, selector: str = "auto") -> int:
360
+ dirpath, man = resolve_existing_index(repo, selector)
361
+ activate_model(man["model"], man["dim"])
362
+ mat = np.load(os.path.join(dirpath, "vectors.npy"))
363
+ norms = np.linalg.norm(mat, axis=1, keepdims=True)
364
+ mat_n = mat / np.maximum(norms, 1e-12)
365
+
366
+ model = load_model()
367
+ t0 = time.perf_counter()
368
+ qv = np.asarray(list(model.embed([question])), dtype=np.float32)[0]
369
+ qv = qv / max(np.linalg.norm(qv), 1e-12)
370
+ scores = mat_n @ qv
371
+ idx = np.argsort(scores)[::-1][:top]
372
+ latency_ms = (time.perf_counter() - t0) * 1000
373
+
374
+ hits = [(man["meta"][i], round(float(scores[i]), 4)) for i in idx]
375
+ print(SELSTAUSKUNFT)
376
+ print(f"Index-Modell: {man['model']}")
377
+ print(freshness(repo, man))
378
+ for m, s in hits:
379
+ print(f" {s:.4f} {m}")
380
+ print(f"query_ms={latency_ms:.1f}")
381
+
382
+ log_path = os.path.join(dirpath, "query-log.jsonl")
383
+ with open(log_path, "a", encoding="utf-8") as fh:
384
+ fh.write(json.dumps({
385
+ "ts": datetime.now(timezone.utc).isoformat(),
386
+ "query": question,
387
+ "top": hits,
388
+ "query_ms": round(latency_ms, 1),
389
+ "index_head": man["head"],
390
+ }) + "\n")
391
+ return 0
392
+
393
+
394
+ def cmd_sample(repo: str, selector: str = "auto") -> int:
395
+ """Gate: Stichprobe aus dem Subjekt-Baum, Ziel >= 2/3 in Top-5."""
396
+ dirpath, man = resolve_existing_index(repo, selector)
397
+ activate_model(man["model"], man["dim"])
398
+ mat = np.load(os.path.join(dirpath, "vectors.npy"))
399
+ norms = np.linalg.norm(mat, axis=1, keepdims=True)
400
+ mat_n = mat / np.maximum(norms, 1e-12)
401
+ model = load_model()
402
+
403
+ hits_count = 0
404
+ for question, expected in SAMPLE_QUERIES:
405
+ qv = np.asarray(list(model.embed([question])), dtype=np.float32)[0]
406
+ qv = qv / max(np.linalg.norm(qv), 1e-12)
407
+ scores = mat_n @ qv
408
+ idx = np.argsort(scores)[::-1][:5]
409
+ hits = [(man["meta"][i], round(float(scores[i]), 4)) for i in idx]
410
+ ok = any(expected in m for m, _ in hits)
411
+ hits_count += ok
412
+ print(f"SAMPLE {'HIT ' if ok else 'MISS'} {question!r} expected={expected}")
413
+ for m, s in hits:
414
+ print(f" {s:.4f} {m}")
415
+ verdict = "PASS" if hits_count >= 2 else "FAIL"
416
+ print(f"SAMPLE_RESULT {hits_count}/3 gate=2/3 -> {verdict}")
417
+ return 0 if hits_count >= 2 else 1
418
+
419
+
420
+ def main() -> int:
421
+ global _active_model_name, _active_dim
422
+ if len(sys.argv) < 3:
423
+ print(__doc__)
424
+ return 2
425
+ cmd, repo = sys.argv[1], sys.argv[2]
426
+ selector = sys.argv[sys.argv.index("--model") + 1] if "--model" in sys.argv else (
427
+ "auto" if cmd in ("query", "sample") else MODEL_NAME
428
+ )
429
+ if cmd in ("build", "update"):
430
+ try:
431
+ activate_model(selector)
432
+ except ValueError as error:
433
+ print(f"MODEL_ERROR {error}")
434
+ return 2
435
+ from fastembed import TextEmbedding
436
+
437
+ dims = {m["model"]: m.get("dim") for m in TextEmbedding.list_supported_models()}
438
+ if _active_model_name not in dims:
439
+ print(f"MODEL_UNKNOWN {_active_model_name} — nicht im fastembed-Angebot")
440
+ return 2
441
+ _active_dim = int(dims[_active_model_name])
442
+ print(f"MODEL {_active_model_name} dim={_active_dim}")
443
+ if cmd == "build":
444
+ cmd_build(repo)
445
+ return 0
446
+ if cmd == "update":
447
+ cmd_update(repo)
448
+ return 0
449
+ if cmd == "query":
450
+ top = 5
451
+ if "--top" in sys.argv:
452
+ top = int(sys.argv[sys.argv.index("--top") + 1])
453
+ try:
454
+ return cmd_query(repo, sys.argv[3], top, selector)
455
+ except (FileNotFoundError, ValueError) as error:
456
+ print(f"INDEX_ERROR {error}")
457
+ return 2
458
+ if cmd == "sample":
459
+ try:
460
+ return cmd_sample(repo, selector)
461
+ except (FileNotFoundError, ValueError) as error:
462
+ print(f"INDEX_ERROR {error}")
463
+ return 2
464
+ print(f"unknown command: {cmd}")
465
+ return 2
466
+
467
+
468
+ if __name__ == "__main__":
469
+ sys.exit(main())
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "blun-king-cli",
3
- "version": "9.1.527",
3
+ "version": "9.1.550",
4
4
  "description": "BLUN CLI - your own AI agent with a Telegram channel. Get it done. With BLUN.",
5
5
  "license": "MIT",
6
6
  "bin": {
@@ -8,10 +8,7 @@
8
8
  "king": "bin/king.js"
9
9
  },
10
10
  "scripts": {
11
- "test": "node --test test/*.test.js",
12
- "prepack": "node scripts/check-release-metadata.js && node scripts/check-copy-command-regression.js && node scripts/check-shell-terminal-isolation-regression.js && node scripts/check-todo-loop-regression.js && node scripts/check-telegram-loop-exactly-once-regression.js && node scripts/check-session-picker-resume-metrics-regression.js && node scripts/check-bundled-agent-spine-regression.js && node scripts/check-reload-agent-spine-regression.js && node scripts/check-todo-recovery-catalog-regression.js && node scripts/check-historical-tool-result-preview-regression.js && node scripts/check-session-start-hook-context-regression.js && node scripts/check-queue-controls-regression.js && node scripts/check-approval-queue-shortcuts-regression.js && node scripts/check-approval-observability-regression.js && node scripts/check-slash-escape-regression.js && node scripts/check-telegram-bridge-watchdog.js && node scripts/check-resume-replay-regression.js && node scripts/check-session-cancel-regression.js && node scripts/check-plugin-startup-regression.js && node scripts/check-active-profile-plugin-startup.js && node scripts/check-mcp-startup-wait-budget.js",
13
- "release:verify": "node scripts/check-release-metadata.js --external",
14
- "postinstall": "node scripts/fix-node-pty-perms.js"
11
+ "postinstall": "node bin/fix-node-pty-perms.js"
15
12
  },
16
13
  "engines": {
17
14
  "node": ">=24.15.0"
@@ -31,38 +28,27 @@
31
28
  },
32
29
  "files": [
33
30
  "bin/",
31
+ "!bin/package-regression-policy.cjs",
34
32
  "blun.mjs",
33
+ "codebase-index/codebase_index.py",
34
+ "codebase-index/README.md",
35
35
  "dist-web/",
36
36
  "native/",
37
- "release-planned-removals.json",
38
- "scripts/check-package-regression.js",
39
- "scripts/check-copy-command-regression.js",
40
- "scripts/check-active-profile-plugin-startup.js",
41
- "scripts/check-approval-queue-shortcuts-regression.js",
42
- "scripts/check-bundled-agent-spine-regression.js",
43
- "scripts/check-reload-agent-spine-regression.js",
44
- "scripts/check-approval-observability-regression.js",
45
- "scripts/check-slash-escape-regression.js",
46
- "scripts/check-mcp-startup-wait-budget.js",
47
- "scripts/check-plugin-startup-regression.js",
48
- "scripts/check-queue-controls-regression.js",
49
- "scripts/check-telegram-bridge-watchdog.js",
50
- "scripts/check-resume-replay-regression.js",
51
- "scripts/check-session-cancel-regression.js",
52
- "scripts/check-shell-terminal-isolation-regression.js",
53
- "scripts/check-session-picker-resume-metrics-regression.js",
54
- "scripts/check-release-metadata.js",
55
- "scripts/check-historical-tool-result-preview-regression.js",
56
- "scripts/check-session-start-hook-context-regression.js",
57
- "scripts/check-telegram-loop-exactly-once-regression.js",
58
- "scripts/check-todo-recovery-catalog-regression.js",
59
- "scripts/check-todo-loop-regression.js",
60
- "scripts/fix-node-pty-perms.js",
61
37
  "standard-skills/",
62
38
  "standard-tools/",
63
- "agent-spine-plugin/",
39
+ "agent-spine-plugin/.claude-plugin/",
40
+ "agent-spine-plugin/.codex-plugin/",
41
+ "agent-spine-plugin/.mcp.json",
42
+ "agent-spine-plugin/assets/",
43
+ "agent-spine-plugin/bin/",
44
+ "agent-spine-plugin/blun.plugin.json",
45
+ "agent-spine-plugin/hooks/",
46
+ "agent-spine-plugin/LICENSE",
47
+ "agent-spine-plugin/package.json",
48
+ "agent-spine-plugin/skill/",
49
+ "agent-spine-plugin/skills/",
50
+ "agent-spine-plugin/src/",
64
51
  "telegram-plugin/",
65
- "CHANGELOG.md",
66
52
  "LIESMICH.txt"
67
53
  ],
68
54
  "keywords": [
@@ -74,13 +60,13 @@
74
60
  ],
75
61
  "author": "BLUN",
76
62
  "main": "index.js",
77
- "directories": {
78
- "test": "test"
79
- },
80
63
  "type": "commonjs",
81
64
  "dependencies": {
82
- "blun-king-cli": "^9.1.62",
83
- "node-addon-api": "^7.1.1"
84
- },
85
- "devDependencies": {}
65
+ "@mozilla/readability": "0.6.0",
66
+ "linkedom": "0.18.12",
67
+ "node-addon-api": "^7.1.1",
68
+ "quickjs-emscripten": "0.32.0",
69
+ "turndown": "7.2.4",
70
+ "turndown-plugin-gfm": "1.0.2"
71
+ }
86
72
  }