esbi-cli 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- esbi_cli/__init__.py +8 -0
- esbi_cli/ask/__init__.py +0 -0
- esbi_cli/ask/answer.py +256 -0
- esbi_cli/bench/__init__.py +0 -0
- esbi_cli/bench/cases.py +57 -0
- esbi_cli/bench/metrics.py +23 -0
- esbi_cli/bench/report.py +117 -0
- esbi_cli/bench/runner.py +114 -0
- esbi_cli/capture/__init__.py +0 -0
- esbi_cli/capture/inbox.py +63 -0
- esbi_cli/capture/legacy.py +49 -0
- esbi_cli/cli.py +1387 -0
- esbi_cli/config.py +344 -0
- esbi_cli/doctor.py +391 -0
- esbi_cli/evaluate.py +91 -0
- esbi_cli/export.py +137 -0
- esbi_cli/extract/__init__.py +107 -0
- esbi_cli/extract/clip.py +30 -0
- esbi_cli/extract/html.py +60 -0
- esbi_cli/extract/image.py +58 -0
- esbi_cli/extract/pdf.py +109 -0
- esbi_cli/gitops.py +101 -0
- esbi_cli/index.py +303 -0
- esbi_cli/ingest/__init__.py +0 -0
- esbi_cli/ingest/apply.py +480 -0
- esbi_cli/ingest/chunks.py +49 -0
- esbi_cli/ingest/connect.py +87 -0
- esbi_cli/ingest/digest.py +91 -0
- esbi_cli/ingest/pipeline.py +176 -0
- esbi_cli/ingest/plan.py +231 -0
- esbi_cli/ingest/read.py +105 -0
- esbi_cli/ingest/retrieve.py +59 -0
- esbi_cli/init.py +176 -0
- esbi_cli/interrupts.py +90 -0
- esbi_cli/lang.py +341 -0
- esbi_cli/links.py +10 -0
- esbi_cli/lint/__init__.py +0 -0
- esbi_cli/lint/checks.py +178 -0
- esbi_cli/lint/report.py +60 -0
- esbi_cli/llm/__init__.py +0 -0
- esbi_cli/llm/adapter.py +393 -0
- esbi_cli/llm/schemas.py +146 -0
- esbi_cli/mail/__init__.py +0 -0
- esbi_cli/mail/convert.py +194 -0
- esbi_cli/mail/credentials.py +65 -0
- esbi_cli/mail/fetch.py +154 -0
- esbi_cli/mail/imap.py +92 -0
- esbi_cli/netguard.py +127 -0
- esbi_cli/privacy.py +81 -0
- esbi_cli/queue.py +179 -0
- esbi_cli/reingest.py +165 -0
- esbi_cli/report/__init__.py +0 -0
- esbi_cli/report/daily_index.py +235 -0
- esbi_cli/report/index_md.py +21 -0
- esbi_cli/report/readstate.py +26 -0
- esbi_cli/run.py +100 -0
- esbi_cli/runlock.py +31 -0
- esbi_cli/runlog.py +80 -0
- esbi_cli/schedule.py +106 -0
- esbi_cli/templates/SCHEMA.md +52 -0
- esbi_cli/templates/clipper-template.json +17 -0
- esbi_cli/templates/clipper-youtube-template.json +18 -0
- esbi_cli/templates/config.example.toml +108 -0
- esbi_cli/update.py +247 -0
- esbi_cli/vault.py +188 -0
- esbi_cli/wizards/clipper.sh +271 -0
- esbi_cli/wizards/email.sh +265 -0
- esbi_cli-0.2.1.dist-info/METADATA +167 -0
- esbi_cli-0.2.1.dist-info/RECORD +72 -0
- esbi_cli-0.2.1.dist-info/WHEEL +4 -0
- esbi_cli-0.2.1.dist-info/entry_points.txt +3 -0
- esbi_cli-0.2.1.dist-info/licenses/LICENSE +21 -0
esbi_cli/doctor.py
ADDED
|
@@ -0,0 +1,391 @@
|
|
|
1
|
+
"""`sb doctor`: check the whole setup and say what to fix. Warnings are inconvenient, FAIL is broken."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
import plistlib
|
|
6
|
+
import re
|
|
7
|
+
import shutil
|
|
8
|
+
import subprocess
|
|
9
|
+
import sys
|
|
10
|
+
from dataclasses import dataclass, replace
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
import httpx
|
|
14
|
+
|
|
15
|
+
from esbi_cli import __version__, lang, update
|
|
16
|
+
from esbi_cli import schedule as launchd
|
|
17
|
+
from esbi_cli.config import Config, find_config, load_config
|
|
18
|
+
from esbi_cli.gitops import has_git
|
|
19
|
+
from esbi_cli.llm.adapter import make_llm, make_ocr
|
|
20
|
+
from esbi_cli.mail.credentials import CredentialError, get_password
|
|
21
|
+
from esbi_cli.privacy import remote_host, remote_warning
|
|
22
|
+
from esbi_cli.queue import Queue
|
|
23
|
+
from esbi_cli.runlog import RunLog
|
|
24
|
+
|
|
25
|
+
KEY_VARS = {"openai": "OPENAI_API_KEY", "anthropic": "ANTHROPIC_API_KEY"}
|
|
26
|
+
SERVER_PROBE_TIMEOUT_SECONDS = 3 # a model server that does not answer this fast is down
|
|
27
|
+
CLAUDE_STATUS_TIMEOUT_SECONDS = 20
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class Check:
|
|
32
|
+
level: str # "ok" | "WARN" | "FAIL"
|
|
33
|
+
name: str
|
|
34
|
+
text: str
|
|
35
|
+
fix: str = ""
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _model(task: str, cfg: Config, fallback: bool = False) -> Check:
|
|
39
|
+
llm = cfg.llm[task]
|
|
40
|
+
if fallback: # the fallback inherits the section's settings, with its own model
|
|
41
|
+
llm = replace(llm, model=llm.fallback)
|
|
42
|
+
task = f"{task} fallback"
|
|
43
|
+
provider, _, name = llm.model.partition("/")
|
|
44
|
+
label = f"model {task}"
|
|
45
|
+
if provider == "ollama":
|
|
46
|
+
base = (llm.base_url or "http://localhost:11434").rstrip("/")
|
|
47
|
+
try:
|
|
48
|
+
tags = httpx.get(f"{base}/api/tags", timeout=SERVER_PROBE_TIMEOUT_SECONDS).json()[
|
|
49
|
+
"models"
|
|
50
|
+
]
|
|
51
|
+
except (httpx.HTTPError, KeyError, ValueError):
|
|
52
|
+
return Check(
|
|
53
|
+
"FAIL",
|
|
54
|
+
label,
|
|
55
|
+
f"Ollama is not reachable at {base}",
|
|
56
|
+
"brew services start ollama (or open the Ollama app)",
|
|
57
|
+
)
|
|
58
|
+
have = {m["name"] for m in tags}
|
|
59
|
+
if name in have or f"{name}:latest" in have:
|
|
60
|
+
return Check("ok", label, f"{llm.model} is installed")
|
|
61
|
+
return Check("FAIL", label, f"{llm.model} is not installed", f"ollama pull {name}")
|
|
62
|
+
if provider == "lmstudio":
|
|
63
|
+
base = (llm.base_url or "http://localhost:1234/v1").rstrip("/")
|
|
64
|
+
try:
|
|
65
|
+
served = {
|
|
66
|
+
m["id"]
|
|
67
|
+
for m in httpx.get(f"{base}/models", timeout=SERVER_PROBE_TIMEOUT_SECONDS).json()[
|
|
68
|
+
"data"
|
|
69
|
+
]
|
|
70
|
+
}
|
|
71
|
+
except (httpx.HTTPError, KeyError, ValueError):
|
|
72
|
+
return Check(
|
|
73
|
+
"FAIL",
|
|
74
|
+
label,
|
|
75
|
+
f"LM Studio is not reachable at {base}",
|
|
76
|
+
"open LM Studio, load a model, and start the local server (Developer tab)",
|
|
77
|
+
)
|
|
78
|
+
if name in served:
|
|
79
|
+
return Check("ok", label, f"{llm.model} is served by LM Studio")
|
|
80
|
+
return Check(
|
|
81
|
+
"FAIL",
|
|
82
|
+
label,
|
|
83
|
+
f"LM Studio does not serve {name!r}",
|
|
84
|
+
f"use one of: {', '.join(sorted(served)) or '(none loaded)'}",
|
|
85
|
+
)
|
|
86
|
+
if provider == "claude-cli":
|
|
87
|
+
if not shutil.which("claude"):
|
|
88
|
+
return Check(
|
|
89
|
+
"FAIL", label, "the `claude` command is not installed", "https://claude.com/code"
|
|
90
|
+
)
|
|
91
|
+
try:
|
|
92
|
+
who = json.loads(
|
|
93
|
+
subprocess.run(
|
|
94
|
+
["claude", "auth", "status"],
|
|
95
|
+
capture_output=True,
|
|
96
|
+
text=True,
|
|
97
|
+
timeout=CLAUDE_STATUS_TIMEOUT_SECONDS,
|
|
98
|
+
).stdout
|
|
99
|
+
)
|
|
100
|
+
except (OSError, subprocess.TimeoutExpired, ValueError):
|
|
101
|
+
who = {}
|
|
102
|
+
if who.get("loggedIn"):
|
|
103
|
+
return Check(
|
|
104
|
+
"ok", label, f"{llm.model} (subscription of {who.get('email', 'your account')})"
|
|
105
|
+
)
|
|
106
|
+
return Check("FAIL", label, "`claude` is not logged in", "claude auth login")
|
|
107
|
+
if provider in KEY_VARS:
|
|
108
|
+
env = llm.api_key_env or KEY_VARS[provider]
|
|
109
|
+
if os.environ.get(env):
|
|
110
|
+
return Check("ok", label, f"{llm.model} (key in ${env})")
|
|
111
|
+
return Check("FAIL", label, f"{llm.model}: ${env} is not set", f"export {env}=...")
|
|
112
|
+
return Check(
|
|
113
|
+
"FAIL",
|
|
114
|
+
label,
|
|
115
|
+
f"unknown provider in {llm.model!r}",
|
|
116
|
+
"use ollama/, lmstudio/, openai/, anthropic/ or claude-cli/",
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _ocr(cfg: Config) -> list[Check]:
|
|
121
|
+
"""Reading images and scanned PDFs is optional; when set up it must be a model on this machine."""
|
|
122
|
+
if "ocr" not in cfg.llm:
|
|
123
|
+
return [
|
|
124
|
+
Check(
|
|
125
|
+
"ok",
|
|
126
|
+
"ocr",
|
|
127
|
+
"off (optional): add [llm.ocr] to read images and scanned PDFs "
|
|
128
|
+
"(`sb init --ocr`, https://rubenamaury.github.io/esbi-cli/docs/reference/configuration/)",
|
|
129
|
+
)
|
|
130
|
+
]
|
|
131
|
+
try:
|
|
132
|
+
make_ocr(cfg.llm["ocr"])
|
|
133
|
+
except ValueError as exc:
|
|
134
|
+
return [Check("FAIL", "ocr", str(exc), 'model = "ollama/qwen3-vl:2b-instruct"')]
|
|
135
|
+
return [_model("ocr", cfg)]
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _sends_text_out(llm_cfg) -> bool:
|
|
139
|
+
try:
|
|
140
|
+
return bool(make_llm(llm_cfg).sends_text_out) # building a model makes no request
|
|
141
|
+
except ValueError:
|
|
142
|
+
return False # an unknown provider is reported by the model check
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _privacy(cfg: Config) -> Check:
|
|
146
|
+
"""Email must be read only by a model on this machine, and cloud models must never get it."""
|
|
147
|
+
private = cfg.llm.get("private")
|
|
148
|
+
if private and _sends_text_out(private):
|
|
149
|
+
return Check(
|
|
150
|
+
"FAIL",
|
|
151
|
+
"email privacy",
|
|
152
|
+
"[llm.private] (or its fallback) sends text away, so email would leave this machine",
|
|
153
|
+
'use a local model: [llm.private] model = "ollama/llama3.2:latest", with no cloud fallback',
|
|
154
|
+
)
|
|
155
|
+
cloud = [t for t, m in cfg.llm.items() if t not in ("private", "embed") and _sends_text_out(m)]
|
|
156
|
+
if not cloud or private:
|
|
157
|
+
return Check(
|
|
158
|
+
"ok",
|
|
159
|
+
"email privacy",
|
|
160
|
+
"email is read only by a model on this machine" if private else "all models are local",
|
|
161
|
+
)
|
|
162
|
+
return Check(
|
|
163
|
+
"WARN",
|
|
164
|
+
"email privacy",
|
|
165
|
+
f"{', '.join(cloud)} send text away and no [llm.private] is set: emails will be refused",
|
|
166
|
+
'add [llm.private] model = "ollama/llama3.2:latest" to config.toml',
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _remote_servers(cfg: Config) -> list[Check]:
|
|
171
|
+
"""A "local" runtime pointed at another machine still sends the notes' text out. (A remote
|
|
172
|
+
[llm.private] is already a FAIL in _privacy: there email would go to that host.)"""
|
|
173
|
+
checks = []
|
|
174
|
+
for task, llm in cfg.llm.items():
|
|
175
|
+
for label, model in ((task, llm.model), (f"{task} fallback", llm.fallback)):
|
|
176
|
+
host = model and task != "private" and remote_host(model, llm.base_url)
|
|
177
|
+
if host:
|
|
178
|
+
checks.append(
|
|
179
|
+
Check(
|
|
180
|
+
"WARN",
|
|
181
|
+
f"server {label}",
|
|
182
|
+
remote_warning(model, host),
|
|
183
|
+
"serve the model from this machine, or use it only for text you would share",
|
|
184
|
+
)
|
|
185
|
+
)
|
|
186
|
+
return checks
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _network(cfg: Config) -> list[Check]:
|
|
190
|
+
if not cfg.network.use_environment_proxy:
|
|
191
|
+
return []
|
|
192
|
+
return [
|
|
193
|
+
Check(
|
|
194
|
+
"WARN",
|
|
195
|
+
"network",
|
|
196
|
+
"[network].use_environment_proxy is on: address checks are done by the proxy, not by "
|
|
197
|
+
"esbi-cli, so a link in an email could make the proxy reach an internal host",
|
|
198
|
+
"set it to false unless this network only has a proxy",
|
|
199
|
+
)
|
|
200
|
+
]
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def _email(cfg: Config) -> Check:
|
|
204
|
+
if not cfg.email.enabled:
|
|
205
|
+
return Check("ok", "email", "off (optional)")
|
|
206
|
+
try:
|
|
207
|
+
get_password(cfg.email.user or "")
|
|
208
|
+
return Check("ok", "email", f"password for {cfg.email.user} is in the Keychain")
|
|
209
|
+
except CredentialError as exc:
|
|
210
|
+
return Check("FAIL", "email", str(exc), "sb email set-password")
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _version(cfg: Config) -> Check:
|
|
214
|
+
"""The installed version, and whether a newer one is known (from the daily cache)."""
|
|
215
|
+
if not update.checks_enabled(cfg.update.check):
|
|
216
|
+
return Check("ok", "version", f"{__version__} (update check is off)")
|
|
217
|
+
release = update.cached_latest()
|
|
218
|
+
if release is None: # offline, or GitHub did not answer: not a problem with this setup
|
|
219
|
+
return Check("ok", "version", __version__)
|
|
220
|
+
if update.is_newer(release.version, __version__):
|
|
221
|
+
return Check("WARN", "version", f"{release.version} is available, run `sb update`")
|
|
222
|
+
return Check("ok", "version", f"{__version__} (latest)")
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
AGENTS_DIR = Path("~/Library/LaunchAgents").expanduser()
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def _installed_plist() -> dict:
|
|
229
|
+
"""The installed LaunchAgent, or {} when there is none (or it cannot be read)."""
|
|
230
|
+
try:
|
|
231
|
+
return plistlib.loads((AGENTS_DIR / f"{launchd.LABEL}.plist").read_bytes())
|
|
232
|
+
except (OSError, ValueError):
|
|
233
|
+
return {}
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def _installed_time() -> tuple[int, int] | None:
|
|
237
|
+
"""The (hour, minute) in the installed LaunchAgent, if there is one."""
|
|
238
|
+
try:
|
|
239
|
+
at = _installed_plist()["StartCalendarInterval"]
|
|
240
|
+
return at["Hour"], at["Minute"]
|
|
241
|
+
except KeyError:
|
|
242
|
+
return None
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _job(cfg: Config) -> Check:
|
|
246
|
+
try:
|
|
247
|
+
loaded = launchd.is_loaded(os.getuid(), launchctl=launchd.run_launchctl)
|
|
248
|
+
except OSError:
|
|
249
|
+
return Check("WARN", "nightly job", "launchd is not available here")
|
|
250
|
+
if not loaded:
|
|
251
|
+
return Check("WARN", "nightly job", "not installed", "sb schedule install")
|
|
252
|
+
installed = _installed_time()
|
|
253
|
+
if installed is not None and installed != cfg.nightly_at:
|
|
254
|
+
return Check(
|
|
255
|
+
"WARN",
|
|
256
|
+
"nightly job",
|
|
257
|
+
f"installed for {installed[0]:02d}:{installed[1]:02d} but config.toml says {cfg.nightly_time}",
|
|
258
|
+
"sb schedule install",
|
|
259
|
+
)
|
|
260
|
+
command = " ".join(map(str, _installed_plist().get("ProgramArguments", [])))
|
|
261
|
+
if "/Cellar/esbi-cli/" in command: # `brew upgrade` deletes that folder: the job would stop
|
|
262
|
+
return Check(
|
|
263
|
+
"WARN",
|
|
264
|
+
"nightly job",
|
|
265
|
+
"points into a versioned Homebrew folder (Cellar) that the next `brew upgrade` removes",
|
|
266
|
+
"sb schedule install",
|
|
267
|
+
)
|
|
268
|
+
prefix = Path(
|
|
269
|
+
sys.prefix
|
|
270
|
+
) # a uv tool or pipx environment can be rebuilt elsewhere by an upgrade
|
|
271
|
+
if update.install_method(prefix, Path(__file__).resolve().parent) in ("uv-tool", "pipx"):
|
|
272
|
+
current = launchd.stable_prefix(prefix) / "bin" / "sb"
|
|
273
|
+
if str(current) not in command:
|
|
274
|
+
old = re.search(r"\S+/bin/sb\b", command)
|
|
275
|
+
return Check(
|
|
276
|
+
"WARN",
|
|
277
|
+
"nightly job",
|
|
278
|
+
f"runs {old[0] if old else 'another environment'}, not the {current} that is "
|
|
279
|
+
"running now",
|
|
280
|
+
"sb schedule install",
|
|
281
|
+
)
|
|
282
|
+
return Check("ok", "nightly job", f"installed and loaded, runs at {cfg.nightly_time}")
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def _last_run(vault: Path) -> Check:
|
|
286
|
+
runs = [r for r in RunLog(vault / ".esbi" / "runs.jsonl").runs() if r.trigger == "scheduled"]
|
|
287
|
+
if not runs:
|
|
288
|
+
return Check(
|
|
289
|
+
"WARN",
|
|
290
|
+
"last run",
|
|
291
|
+
"no scheduled run yet",
|
|
292
|
+
"the job does it after the nightly time; or try `sb run --if-due`",
|
|
293
|
+
)
|
|
294
|
+
last = runs[-1]
|
|
295
|
+
text = f"{last.started:%Y-%m-%d %H:%M}: {last.ingested} ingested, {last.failed} failed"
|
|
296
|
+
if last.stopped_by == "llm_unavailable":
|
|
297
|
+
return Check(
|
|
298
|
+
"WARN", "last run", f"{text}; the model was unreachable", "check the model line above"
|
|
299
|
+
)
|
|
300
|
+
return Check("ok", "last run", text)
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def _vault(root: Path, viewer: str = "obsidian") -> list[Check]:
|
|
304
|
+
if not (root / "SCHEMA.md").is_file() or not (root / "wiki").is_dir():
|
|
305
|
+
return [
|
|
306
|
+
Check(
|
|
307
|
+
"FAIL",
|
|
308
|
+
"vault",
|
|
309
|
+
f"{root} is not a esbi-cli vault (SCHEMA.md or wiki/ missing)",
|
|
310
|
+
"fix [paths].vault in config.toml",
|
|
311
|
+
)
|
|
312
|
+
]
|
|
313
|
+
checks = [Check("ok", "vault", str(root))]
|
|
314
|
+
if not (root / ".git").exists():
|
|
315
|
+
checks.append(
|
|
316
|
+
Check(
|
|
317
|
+
"ok", "vault history", "off: git is not installed (optional, `sb init` turns it on)"
|
|
318
|
+
)
|
|
319
|
+
if not has_git()
|
|
320
|
+
else Check(
|
|
321
|
+
"WARN",
|
|
322
|
+
"vault history",
|
|
323
|
+
"no version history (the vault is not a git repository)",
|
|
324
|
+
f"git -C {root} init",
|
|
325
|
+
)
|
|
326
|
+
)
|
|
327
|
+
if viewer == "none":
|
|
328
|
+
checks.append(Check("ok", "obsidian", 'not used ([notes].viewer = "none")'))
|
|
329
|
+
elif not (root / ".obsidian").is_dir():
|
|
330
|
+
checks.append(
|
|
331
|
+
Check(
|
|
332
|
+
"WARN",
|
|
333
|
+
"obsidian",
|
|
334
|
+
"this folder was never opened as a vault in Obsidian",
|
|
335
|
+
f"Obsidian > Open folder as vault > {root}",
|
|
336
|
+
)
|
|
337
|
+
)
|
|
338
|
+
else:
|
|
339
|
+
checks.append(Check("ok", "obsidian", "opened as a vault"))
|
|
340
|
+
counts = Queue(root / ".esbi" / "queue.sqlite3").counts()
|
|
341
|
+
text = ", ".join(f"{n} {state}" for state, n in sorted(counts.items())) or "empty"
|
|
342
|
+
if counts.get("failed"):
|
|
343
|
+
checks.append(Check("WARN", "queue", text, "sb status, then `sb retry` or `sb drop`"))
|
|
344
|
+
else:
|
|
345
|
+
checks.append(Check("ok", "queue", text))
|
|
346
|
+
return checks + [_last_run(root)]
|
|
347
|
+
|
|
348
|
+
|
|
349
|
+
def run_checks(config_arg: Path | None) -> list[Check]:
|
|
350
|
+
try:
|
|
351
|
+
path = find_config(config_arg)
|
|
352
|
+
cfg = load_config(path)
|
|
353
|
+
except (FileNotFoundError, ValueError) as exc:
|
|
354
|
+
return [
|
|
355
|
+
Check(
|
|
356
|
+
"FAIL",
|
|
357
|
+
"config",
|
|
358
|
+
str(exc),
|
|
359
|
+
"correct the setting named in the message"
|
|
360
|
+
if isinstance(exc, ValueError)
|
|
361
|
+
else "run `sb init`, or point at your file with --config PATH",
|
|
362
|
+
)
|
|
363
|
+
]
|
|
364
|
+
checks = [
|
|
365
|
+
Check("ok", "config", str(path)),
|
|
366
|
+
_version(cfg),
|
|
367
|
+
Check("ok", "notes language", f"{cfg.language} ({lang.name(cfg.language)})"),
|
|
368
|
+
*_vault(cfg.vault, cfg.viewer),
|
|
369
|
+
]
|
|
370
|
+
for task in ("summarize", "synthesize", "ask", "private", "embed"):
|
|
371
|
+
if task in cfg.llm:
|
|
372
|
+
checks.append(_model(task, cfg))
|
|
373
|
+
if cfg.llm[task].fallback:
|
|
374
|
+
checks.append(_model(task, cfg, fallback=True))
|
|
375
|
+
checks += _ocr(cfg)
|
|
376
|
+
checks.append(_privacy(cfg))
|
|
377
|
+
checks += _remote_servers(cfg)
|
|
378
|
+
checks += [*_network(cfg), _email(cfg), _job(cfg)]
|
|
379
|
+
sb = shutil.which("sb")
|
|
380
|
+
project = Path(__file__).resolve().parents[2]
|
|
381
|
+
checks.append(
|
|
382
|
+
Check("ok", "global install", sb)
|
|
383
|
+
if sb
|
|
384
|
+
else Check(
|
|
385
|
+
"WARN",
|
|
386
|
+
"global install",
|
|
387
|
+
"`sb` is not on your PATH",
|
|
388
|
+
f"uv tool install --editable {project}",
|
|
389
|
+
)
|
|
390
|
+
)
|
|
391
|
+
return checks
|
esbi_cli/evaluate.py
ADDED
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
"""Score retrieval (and answers) on questions whose source is known: the number to beat before
|
|
2
|
+
changing how `sb ask` finds pages."""
|
|
3
|
+
|
|
4
|
+
import json
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from esbi_cli.ask.answer import answer_question, rewrite_question
|
|
9
|
+
from esbi_cli.ingest.retrieve import find_candidates
|
|
10
|
+
from esbi_cli.llm.adapter import LLM
|
|
11
|
+
from esbi_cli.vault import Vault, fold
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass
|
|
15
|
+
class Golden:
|
|
16
|
+
question: str
|
|
17
|
+
expect: list[str] # titles, or the start of titles: long source titles are cut
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass
|
|
21
|
+
class Case:
|
|
22
|
+
question: str
|
|
23
|
+
expect: list[str]
|
|
24
|
+
retrieved: list[str]
|
|
25
|
+
rank: int | None # 1 = first page returned; None = not among them
|
|
26
|
+
grounded: bool | None = None # only when answers were asked for
|
|
27
|
+
cited: bool | None = None
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class EvalReport:
|
|
32
|
+
k: int
|
|
33
|
+
cases: list[Case] = field(default_factory=list)
|
|
34
|
+
|
|
35
|
+
@property
|
|
36
|
+
def recall(self) -> float:
|
|
37
|
+
return sum(c.rank is not None for c in self.cases) / len(self.cases) if self.cases else 0.0
|
|
38
|
+
|
|
39
|
+
@property
|
|
40
|
+
def mrr(self) -> float:
|
|
41
|
+
return (
|
|
42
|
+
sum(1 / c.rank for c in self.cases if c.rank) / len(self.cases) if self.cases else 0.0
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
def rate(self, attribute: str) -> float | None:
|
|
46
|
+
values = [getattr(c, attribute) for c in self.cases if getattr(c, attribute) is not None]
|
|
47
|
+
return sum(values) / len(values) if values else None
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def load_golden(path: Path) -> list[Golden]:
|
|
51
|
+
golden = []
|
|
52
|
+
for number, line in enumerate(path.read_text(encoding="utf-8").splitlines(), 1):
|
|
53
|
+
if not line.strip():
|
|
54
|
+
continue
|
|
55
|
+
try:
|
|
56
|
+
row = json.loads(line)
|
|
57
|
+
golden.append(Golden(str(row["question"]), [str(e) for e in row["expect"]]))
|
|
58
|
+
except (ValueError, KeyError, TypeError) as exc:
|
|
59
|
+
raise ValueError(
|
|
60
|
+
f'{path.name} line {number}: need {{"question": ..., "expect": [...]}}'
|
|
61
|
+
) from exc
|
|
62
|
+
return golden
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _matches(title: str, expect: list[str]) -> bool:
|
|
66
|
+
return any(fold(title).startswith(fold(e)) for e in expect)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def evaluate(
|
|
70
|
+
vault: Vault,
|
|
71
|
+
golden: list[Golden],
|
|
72
|
+
k: int = 6,
|
|
73
|
+
rewrite_llm: LLM | None = None,
|
|
74
|
+
answer_llm: LLM | None = None,
|
|
75
|
+
) -> EvalReport:
|
|
76
|
+
report = EvalReport(k)
|
|
77
|
+
for item in golden:
|
|
78
|
+
extra = rewrite_question(rewrite_llm, item.question, vault.language) if rewrite_llm else []
|
|
79
|
+
titles = [
|
|
80
|
+
c.title for c in find_candidates(vault, item.question, max_results=k, extra=extra)
|
|
81
|
+
]
|
|
82
|
+
rank = next((i + 1 for i, t in enumerate(titles) if _matches(t, item.expect)), None)
|
|
83
|
+
case = Case(item.question, item.expect, titles, rank)
|
|
84
|
+
if answer_llm:
|
|
85
|
+
answer = answer_question(
|
|
86
|
+
vault, answer_llm, item.question, rewrite=rewrite_llm is not None
|
|
87
|
+
)
|
|
88
|
+
case.grounded = answer.grounded
|
|
89
|
+
case.cited = any(_matches(c, item.expect) for c in answer.citations)
|
|
90
|
+
report.cases.append(case)
|
|
91
|
+
return report
|
esbi_cli/export.py
ADDED
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
"""A read-only static copy of the wiki (plain HTML files), for people who browse without Obsidian."""
|
|
2
|
+
|
|
3
|
+
import html
|
|
4
|
+
import re
|
|
5
|
+
import shutil
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from urllib.parse import quote
|
|
8
|
+
|
|
9
|
+
from markdown_it import MarkdownIt
|
|
10
|
+
|
|
11
|
+
from esbi_cli import lang
|
|
12
|
+
from esbi_cli.vault import Page, Vault, fold, slugify
|
|
13
|
+
|
|
14
|
+
MARKER = ".esbi-site"
|
|
15
|
+
_EMBED = re.compile(r"!\[\[([^\]|]+?)(?:\|(\d+))?\]\]")
|
|
16
|
+
_LINK = re.compile(r"\[\[([^\]|#]+?)(?:#[^\]|]*)?(?:\|([^\]]+?))?\]\]")
|
|
17
|
+
_STYLE = """
|
|
18
|
+
:root { color-scheme: light dark; --fg: #1f2328; --bg: #ffffff; --muted: #6a737d; --line: #d8dee4; --link: #0b5cad; }
|
|
19
|
+
@media (prefers-color-scheme: dark) { :root { --fg: #e6edf3; --bg: #0d1117; --muted: #8b949e; --line: #30363d; --link: #58a6ff; } }
|
|
20
|
+
body { font: 16px/1.6 system-ui, sans-serif; color: var(--fg); background: var(--bg); max-width: 46rem; margin: 0 auto; padding: 1rem; }
|
|
21
|
+
a { color: var(--link); } nav { border-bottom: 1px solid var(--line); margin-bottom: 1rem; padding-bottom: .5rem; }
|
|
22
|
+
img { max-width: 100%; height: auto; } pre { overflow-x: auto; } blockquote { border-left: 3px solid var(--line); margin-left: 0; padding-left: 1rem; color: var(--muted); }
|
|
23
|
+
li.src small { color: var(--muted); } table { border-collapse: collapse; } td, th { border: 1px solid var(--line); padding: .25rem .5rem; }
|
|
24
|
+
"""
|
|
25
|
+
_MERMAID = (
|
|
26
|
+
'<script type="module">import mermaid from "https://cdn.jsdelivr.net/npm/mermaid@10.9.3/dist/'
|
|
27
|
+
'mermaid.esm.min.mjs"; mermaid.initialize({startOnLoad: true});</script>'
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _page_html(title: str, body: str, root: str, language: str) -> str:
|
|
32
|
+
script = _MERMAID if 'class="mermaid"' in body else ""
|
|
33
|
+
return (
|
|
34
|
+
f'<!doctype html><html lang="{language}"><head><meta charset="utf-8">'
|
|
35
|
+
f'<meta name="viewport" content="width=device-width, initial-scale=1">'
|
|
36
|
+
f"<title>{html.escape(title)}</title><style>{_STYLE}</style></head><body>"
|
|
37
|
+
f'<nav><a href="{root}index.html">esbi-cli</a></nav>{body}{script}</body></html>'
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def export_site(vault: Vault, out: Path) -> int:
|
|
42
|
+
"""Write the wiki as HTML into `out` (replacing an earlier export). Returns the page count."""
|
|
43
|
+
out = out.expanduser().resolve()
|
|
44
|
+
if out.exists() and any(out.iterdir()):
|
|
45
|
+
old_marker = out / ".second-brain-site" # legacy: the name before esbi-cli
|
|
46
|
+
if not (out / MARKER).exists() and not old_marker.exists():
|
|
47
|
+
raise ValueError(
|
|
48
|
+
f"{out} is not an export of this wiki and has other files: not touching it"
|
|
49
|
+
)
|
|
50
|
+
for child in out.iterdir():
|
|
51
|
+
shutil.rmtree(child) if child.is_dir() else child.unlink()
|
|
52
|
+
out.mkdir(parents=True, exist_ok=True)
|
|
53
|
+
(out / MARKER).write_text("written by `sb export`; safe to replace\n", encoding="utf-8")
|
|
54
|
+
|
|
55
|
+
pages = list(vault.iter_pages())
|
|
56
|
+
taken: dict[str, set[str]] = {}
|
|
57
|
+
where: dict[Page, str] = {} # page -> "kind/slug.html"
|
|
58
|
+
names: dict[str, str] = {}
|
|
59
|
+
for page in pages:
|
|
60
|
+
slug, used = slugify(page.title), taken.setdefault(page.kind, set())
|
|
61
|
+
n = 2
|
|
62
|
+
while slug in used:
|
|
63
|
+
slug, n = f"{slugify(page.title)}-{n}", n + 1
|
|
64
|
+
used.add(slug)
|
|
65
|
+
where[id(page)] = f"{page.kind}/{slug}.html"
|
|
66
|
+
for name in (page.path.stem, page.title, *page.aliases):
|
|
67
|
+
names.setdefault(fold(str(name)), where[id(page)])
|
|
68
|
+
|
|
69
|
+
# html False: a note is written from untrusted pages and mail, and the export is a website, so
|
|
70
|
+
# raw HTML in a note is shown as text. Links and images are written as Markdown below.
|
|
71
|
+
md = MarkdownIt("commonmark", {"html": False}).enable("table")
|
|
72
|
+
for page in pages:
|
|
73
|
+
|
|
74
|
+
def embed(m: re.Match) -> str:
|
|
75
|
+
return f", safe='/')})"
|
|
76
|
+
|
|
77
|
+
def link(m: re.Match) -> str:
|
|
78
|
+
target = names.get(fold(m[1].strip()))
|
|
79
|
+
label = (m[2] or m[1].strip()).replace("[", "\\[").replace("]", "\\]")
|
|
80
|
+
return f"[{label}](../{quote(target, safe='/')})" if target else label
|
|
81
|
+
|
|
82
|
+
text = _LINK.sub(link, _EMBED.sub(embed, page.body))
|
|
83
|
+
body = md.render(text)
|
|
84
|
+
body = body.replace('<pre><code class="language-mermaid">', '<pre class="mermaid"><code>')
|
|
85
|
+
body = re.sub(
|
|
86
|
+
r'<pre class="mermaid"><code>(.*?)</code></pre>',
|
|
87
|
+
r'<pre class="mermaid">\1</pre>',
|
|
88
|
+
body,
|
|
89
|
+
flags=re.S,
|
|
90
|
+
)
|
|
91
|
+
body = body.replace("<li>[x] ", "<li>☑ ").replace("<li>[ ] ", "<li>☐ ")
|
|
92
|
+
target = out / where[id(page)]
|
|
93
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
94
|
+
target.write_text(_page_html(page.title, body, "../", vault.language), encoding="utf-8")
|
|
95
|
+
|
|
96
|
+
if (vault.root / "attachments").is_dir():
|
|
97
|
+
shutil.copytree(vault.root / "attachments", out / "attachments", dirs_exist_ok=True)
|
|
98
|
+
(out / "index.html").write_text(_index(pages, where, vault.language), encoding="utf-8")
|
|
99
|
+
return len(pages)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _index(pages: list[Page], where: dict, language: str) -> str:
|
|
103
|
+
def items(selected: list[Page]) -> str:
|
|
104
|
+
rows = []
|
|
105
|
+
for p in selected:
|
|
106
|
+
summary = html.escape(str(p.meta.get("summary") or ""))
|
|
107
|
+
rows.append(
|
|
108
|
+
f'<li class="src"><a href="{where[id(p)]}">{html.escape(p.title)}</a>'
|
|
109
|
+
+ (f"<br><small>{summary}</small>" if summary else "")
|
|
110
|
+
+ "</li>"
|
|
111
|
+
)
|
|
112
|
+
return f"<ul>{''.join(rows)}</ul>"
|
|
113
|
+
|
|
114
|
+
sections = []
|
|
115
|
+
sources = sorted(
|
|
116
|
+
(p for p in pages if p.kind == "sources"),
|
|
117
|
+
key=lambda p: (str(p.meta.get("processed", "")), p.title),
|
|
118
|
+
reverse=True,
|
|
119
|
+
)
|
|
120
|
+
for heading, selected in (
|
|
121
|
+
(lang.t(language, "sources"), sources),
|
|
122
|
+
(
|
|
123
|
+
lang.t(language, "concepts"),
|
|
124
|
+
sorted((p for p in pages if p.kind == "concepts"), key=lambda p: fold(p.title)),
|
|
125
|
+
),
|
|
126
|
+
(
|
|
127
|
+
lang.t(language, "entities"),
|
|
128
|
+
sorted((p for p in pages if p.kind == "entities"), key=lambda p: fold(p.title)),
|
|
129
|
+
),
|
|
130
|
+
(
|
|
131
|
+
lang.t(language, "syntheses"),
|
|
132
|
+
sorted((p for p in pages if p.kind == "syntheses"), key=lambda p: fold(p.title)),
|
|
133
|
+
),
|
|
134
|
+
):
|
|
135
|
+
if selected:
|
|
136
|
+
sections.append(f"<h2>{heading} ({len(selected)})</h2>{items(selected)}")
|
|
137
|
+
return _page_html("esbi-cli", "<h1>esbi-cli</h1>" + "".join(sections), "", language)
|