esbi-cli 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. esbi_cli/__init__.py +8 -0
  2. esbi_cli/ask/__init__.py +0 -0
  3. esbi_cli/ask/answer.py +256 -0
  4. esbi_cli/bench/__init__.py +0 -0
  5. esbi_cli/bench/cases.py +57 -0
  6. esbi_cli/bench/metrics.py +23 -0
  7. esbi_cli/bench/report.py +117 -0
  8. esbi_cli/bench/runner.py +114 -0
  9. esbi_cli/capture/__init__.py +0 -0
  10. esbi_cli/capture/inbox.py +63 -0
  11. esbi_cli/capture/legacy.py +49 -0
  12. esbi_cli/cli.py +1387 -0
  13. esbi_cli/config.py +344 -0
  14. esbi_cli/doctor.py +391 -0
  15. esbi_cli/evaluate.py +91 -0
  16. esbi_cli/export.py +137 -0
  17. esbi_cli/extract/__init__.py +107 -0
  18. esbi_cli/extract/clip.py +30 -0
  19. esbi_cli/extract/html.py +60 -0
  20. esbi_cli/extract/image.py +58 -0
  21. esbi_cli/extract/pdf.py +109 -0
  22. esbi_cli/gitops.py +101 -0
  23. esbi_cli/index.py +303 -0
  24. esbi_cli/ingest/__init__.py +0 -0
  25. esbi_cli/ingest/apply.py +480 -0
  26. esbi_cli/ingest/chunks.py +49 -0
  27. esbi_cli/ingest/connect.py +87 -0
  28. esbi_cli/ingest/digest.py +91 -0
  29. esbi_cli/ingest/pipeline.py +176 -0
  30. esbi_cli/ingest/plan.py +231 -0
  31. esbi_cli/ingest/read.py +105 -0
  32. esbi_cli/ingest/retrieve.py +59 -0
  33. esbi_cli/init.py +176 -0
  34. esbi_cli/interrupts.py +90 -0
  35. esbi_cli/lang.py +341 -0
  36. esbi_cli/links.py +10 -0
  37. esbi_cli/lint/__init__.py +0 -0
  38. esbi_cli/lint/checks.py +178 -0
  39. esbi_cli/lint/report.py +60 -0
  40. esbi_cli/llm/__init__.py +0 -0
  41. esbi_cli/llm/adapter.py +393 -0
  42. esbi_cli/llm/schemas.py +146 -0
  43. esbi_cli/mail/__init__.py +0 -0
  44. esbi_cli/mail/convert.py +194 -0
  45. esbi_cli/mail/credentials.py +65 -0
  46. esbi_cli/mail/fetch.py +154 -0
  47. esbi_cli/mail/imap.py +92 -0
  48. esbi_cli/netguard.py +127 -0
  49. esbi_cli/privacy.py +81 -0
  50. esbi_cli/queue.py +179 -0
  51. esbi_cli/reingest.py +165 -0
  52. esbi_cli/report/__init__.py +0 -0
  53. esbi_cli/report/daily_index.py +235 -0
  54. esbi_cli/report/index_md.py +21 -0
  55. esbi_cli/report/readstate.py +26 -0
  56. esbi_cli/run.py +100 -0
  57. esbi_cli/runlock.py +31 -0
  58. esbi_cli/runlog.py +80 -0
  59. esbi_cli/schedule.py +106 -0
  60. esbi_cli/templates/SCHEMA.md +52 -0
  61. esbi_cli/templates/clipper-template.json +17 -0
  62. esbi_cli/templates/clipper-youtube-template.json +18 -0
  63. esbi_cli/templates/config.example.toml +108 -0
  64. esbi_cli/update.py +247 -0
  65. esbi_cli/vault.py +188 -0
  66. esbi_cli/wizards/clipper.sh +271 -0
  67. esbi_cli/wizards/email.sh +265 -0
  68. esbi_cli-0.2.1.dist-info/METADATA +167 -0
  69. esbi_cli-0.2.1.dist-info/RECORD +72 -0
  70. esbi_cli-0.2.1.dist-info/WHEEL +4 -0
  71. esbi_cli-0.2.1.dist-info/entry_points.txt +3 -0
  72. esbi_cli-0.2.1.dist-info/licenses/LICENSE +21 -0
esbi_cli/config.py ADDED
@@ -0,0 +1,344 @@
1
+ import difflib
2
+ import os
3
+ import re
4
+ import sys
5
+ import tomllib
6
+ import types
7
+ from dataclasses import MISSING, dataclass, field, fields
8
+ from pathlib import Path
9
+ from typing import Union, get_args, get_origin, get_type_hints
10
+
11
+ from esbi_cli import lang, netguard
12
+ from esbi_cli.gitops import ignore_state
13
+
14
+ # Never a path relative to the current folder: a ./config.toml in a cloned repository could point the
15
+ # model, or the mailbox, at someone else's server.
16
+ DEFAULT_CONFIG_PATHS = (
17
+ Path("~/.config/esbi-cli/config.toml").expanduser(),
18
+ Path(__file__).resolve().parents[2] / "config.toml", # next to an editable install: any folder
19
+ Path("~/.config/secondbrain/config.toml").expanduser(), # legacy: the name before esbi-cli
20
+ )
21
+
22
+
23
+ @dataclass
24
+ class LLMConfig:
25
+ model: str # "<provider>/<name>", provider in {ollama, openai, anthropic}
26
+ base_url: str | None = None
27
+ api_key_env: str | None = None
28
+ num_ctx: int = 8192
29
+ temperature: float = 0.2
30
+ timeout_seconds: float = 300.0 # one answer; `timeout` is the deprecated name
31
+ fallback: str | None = None # "<provider>/<name>" used when this model is out of reach
32
+ max_tokens: int | None = None # caps the answer; stops a small model that loops on one input
33
+
34
+
35
+ @dataclass
36
+ class EmailConfig:
37
+ enabled: bool = False
38
+ imap_host: str = "imap.gmail.com"
39
+ mailbox: str = "esbi-cli"
40
+ user: str | None = None
41
+ follow_links: bool = False # queue the links found in a mail; its pages are read as email
42
+ follow_links_max: int = 3 # at most this many links per mail (1 to 10)
43
+
44
+
45
+ @dataclass
46
+ class BenchConfig:
47
+ models: list[str] = field(default_factory=list)
48
+ cases: int = 3
49
+ prices: dict[str, float] = field(default_factory=dict) # USD per million tokens
50
+
51
+
52
+ @dataclass
53
+ class UpdateConfig:
54
+ check: bool = (
55
+ True # look for a newer release once a day (one anonymous HTTPS GET); see update.py
56
+ )
57
+
58
+
59
+ @dataclass
60
+ class NetworkConfig:
61
+ # False: the fetch guard ignores HTTP(S)_PROXY and friends, so its address check is the truth.
62
+ # True: for a network that only has a proxy; the proxy then decides where requests go.
63
+ use_environment_proxy: bool = False
64
+
65
+
66
+ @dataclass
67
+ class Config:
68
+ vault: Path
69
+ language: str = lang.DEFAULT # what the notes are written in: see lang.py
70
+ viewer: str = (
71
+ "obsidian" # "obsidian" opens notes with obsidian:// links; "none" just prints paths
72
+ )
73
+ legacy_vault: Path | None = None
74
+ max_source_chars: int = 4000 # up to this length a source is read in one go
75
+ chunk_chars: int = 8000 # longer sources are read in chunks of about this size
76
+ max_chunks: int = 16 # ...at most this many (huge sources are sampled)
77
+ ocr_max_pages: int = 10 # a scanned PDF is read (OCR) up to this many pages
78
+ find_connections: bool = True # relate each new source to pages already in the wiki
79
+ rewrite_questions: bool = False # `sb ask` first rewrites the question into search terms
80
+ nightly_time: str = "03:00" # HH:MM, 24 hours: when the nightly job runs
81
+ max_sources_per_run: int = 20
82
+ max_tokens_per_run: int | None = 300_000
83
+ flag_contradictions: bool = (
84
+ False # small models flag tenuous ones; opt in with a stronger model
85
+ )
86
+ llm: dict[str, LLMConfig] = field(default_factory=dict)
87
+ email: EmailConfig = field(default_factory=EmailConfig)
88
+ bench: BenchConfig = field(default_factory=BenchConfig)
89
+ update: UpdateConfig = field(default_factory=UpdateConfig)
90
+ network: NetworkConfig = field(default_factory=NetworkConfig)
91
+
92
+ @property
93
+ def nightly_at(self) -> tuple[int, int]:
94
+ return parse_time(self.nightly_time)
95
+
96
+ def llm_for(self, task: str) -> LLMConfig:
97
+ try:
98
+ return self.llm[task]
99
+ except KeyError:
100
+ raise KeyError(f"No [llm.{task}] section in config") from None
101
+
102
+
103
+ def find_config(path: Path | None = None) -> Path:
104
+ """The config file that applies: --config, $ESBI_CONFIG, then the default places. A file named
105
+ on purpose that does not exist is an error; only the default places may fall through."""
106
+ env = os.environ.get("ESBI_CONFIG") or os.environ.get("SECONDBRAIN_CONFIG") # legacy name
107
+ if explicit := path or (Path(env) if env else None):
108
+ if not explicit.is_file():
109
+ raise FileNotFoundError(f"config file not found: {explicit}")
110
+ return explicit
111
+ for candidate in DEFAULT_CONFIG_PATHS:
112
+ if candidate.is_file():
113
+ return candidate
114
+ searched = ", ".join(str(c) for c in DEFAULT_CONFIG_PATHS)
115
+ raise FileNotFoundError(f"No config.toml found (looked in: {searched})")
116
+
117
+
118
+ _loaded_path: Path | None = None # the file load_config read for the command that is running
119
+
120
+
121
+ def wants_update_check() -> bool:
122
+ """`[update].check` of the config the running command used (else the one that applies by
123
+ default), read without validating anything else and without any side effect. On any trouble:
124
+ True (the setting's default); the command that is running reports a bad config itself."""
125
+ try:
126
+ raw = tomllib.loads((_loaded_path or find_config()).read_text(encoding="utf-8"))
127
+ return raw.get("update", {}).get("check", True) is not False
128
+ except (OSError, ValueError, AttributeError):
129
+ return True
130
+
131
+
132
+ def _adopt_old_state_folder(vault: Path) -> None:
133
+ """legacy: before esbi-cli the vault's state folder was `.secondbrain`. Rename it once and keep
134
+ git ignoring it."""
135
+ old, new = vault / ".secondbrain", vault / ".esbi"
136
+ if old.is_dir() and not new.exists():
137
+ old.rename(new)
138
+ if new.is_dir():
139
+ ignore_state(vault)
140
+
141
+
142
+ def reset_loaded() -> None:
143
+ """Forget what the previous command loaded: every invocation starts from the safe defaults."""
144
+ global _loaded_path
145
+ _loaded_path = None
146
+ netguard.use_environment_proxy = False
147
+
148
+
149
+ def load_config(path: Path | None = None) -> Config:
150
+ global _loaded_path
151
+ found = find_config(path)
152
+ cfg = _parse(tomllib.loads(found.read_text(encoding="utf-8")))
153
+ _loaded_path = found
154
+ _adopt_old_state_folder(cfg.vault)
155
+ netguard.use_environment_proxy = cfg.network.use_environment_proxy
156
+ return cfg
157
+
158
+
159
+ def parse_time(text: str) -> tuple[int, int]:
160
+ """`HH:MM` (24 hours) as (hour, minute)."""
161
+ match = re.fullmatch(r"(\d{1,2}):(\d{2})", str(text).strip())
162
+ if not match or int(match[1]) > 23 or int(match[2]) > 59:
163
+ raise ValueError(f"[run].nightly_time must look like HH:MM (24 hours), got {text!r}")
164
+ return int(match[1]), int(match[2])
165
+
166
+
167
+ # Which keys each section accepts. The [llm.*], [email], [bench], [update] and [network] ones come from their dataclass.
168
+ _TABLES = {
169
+ "paths": ("vault", "legacy_vault"),
170
+ "notes": ("language", "viewer"),
171
+ "run": (
172
+ "max_source_chars",
173
+ "chunk_chars",
174
+ "max_chunks",
175
+ "ocr_max_pages",
176
+ "find_connections",
177
+ "rewrite_questions",
178
+ "nightly_time",
179
+ "max_sources_per_run",
180
+ "max_tokens_per_run",
181
+ "flag_contradictions",
182
+ ),
183
+ }
184
+ _SECTIONS = ("llm", "email", "bench", "update", "network")
185
+ _LLM_TASKS = ("summarize", "synthesize", "private", "ocr", "ask", "embed")
186
+ _TYPE_NAMES = {
187
+ int: "a whole number",
188
+ float: "a number",
189
+ str: "text",
190
+ bool: "true or false",
191
+ list: "a list",
192
+ dict: "a table",
193
+ }
194
+
195
+
196
+ def _fits(value, tp) -> bool:
197
+ if get_origin(tp) in (Union, types.UnionType):
198
+ return any(_fits(value, arg) for arg in get_args(tp))
199
+ if tp is type(None):
200
+ return value is None
201
+ if get_origin(tp) is list:
202
+ return isinstance(value, list) and all(_fits(v, get_args(tp)[0]) for v in value)
203
+ if get_origin(tp) is dict:
204
+ return isinstance(value, dict) and all(_fits(v, get_args(tp)[1]) for v in value.values())
205
+ if tp is float:
206
+ return isinstance(value, int | float) and not isinstance(value, bool)
207
+ if tp is int:
208
+ return isinstance(value, int) and not isinstance(value, bool)
209
+ return isinstance(value, str if tp is Path else tp)
210
+
211
+
212
+ def _expects(tp) -> str:
213
+ tp = next((a for a in get_args(tp) if a is not type(None)), tp) # `X | None` expects an X
214
+ return _TYPE_NAMES.get(get_origin(tp) or tp, "a different kind of value")
215
+
216
+
217
+ def _check(name: str, table, hints: dict, required: tuple[str, ...] = ()) -> None:
218
+ """Refuse what would otherwise surface as a Python traceback: a section that is not a table, a
219
+ key we do not know (with the closest match), a missing key, or a value of the wrong type."""
220
+ if not isinstance(table, dict):
221
+ raise ValueError(f"[{name}] must be a table with keys, not a single value")
222
+ for key, value in table.items():
223
+ if key not in hints:
224
+ close = difflib.get_close_matches(key, hints, n=1)
225
+ hint = f" (did you mean {close[0]!r}?)" if close else ""
226
+ raise ValueError(
227
+ f"[{name}] has an unknown key {key!r}{hint}. Valid keys: {', '.join(hints)}"
228
+ )
229
+ if not _fits(value, hints[key]):
230
+ raise ValueError(f"[{name}].{key} must be {_expects(hints[key])}, got {value!r}")
231
+ if missing := [k for k in required if k not in table]:
232
+ raise ValueError(f"[{name}] needs {', '.join(missing)}")
233
+
234
+
235
+ def _dataclass_hints(cls) -> tuple[dict, tuple[str, ...]]:
236
+ hints = get_type_hints(cls)
237
+ required = tuple(
238
+ f.name for f in fields(cls) if f.default is MISSING and f.default_factory is MISSING
239
+ )
240
+ return {f.name: hints[f.name] for f in fields(cls)}, required
241
+
242
+
243
+ # keys that early versions wrote into config.toml and that nothing ever read: accepted, ignored
244
+ _RETIRED = {"run": ("daily_cost_cap_usd",)}
245
+ _RETIRED_TASKS = (
246
+ "link",
247
+ "lint",
248
+ ) # [llm.link] and [llm.lint] were written for tasks that never existed
249
+
250
+
251
+ # `[llm.*]` keys renamed to carry their unit: the old name keeps working, with a notice
252
+ _RENAMED_LLM_KEYS = {"timeout": "timeout_seconds"}
253
+ _noticed: set[tuple[str, str]] = set() # one notice per section and key in a process
254
+
255
+
256
+ def _accept_renamed_keys(raw: dict) -> None:
257
+ """Rewrite old key names to the new ones (the new one wins if both are set) before checking."""
258
+ for task, section in raw.get("llm", {}).items():
259
+ if not isinstance(section, dict) or task not in (*_LLM_TASKS, *_RETIRED_TASKS):
260
+ continue # _validate names the problem
261
+ for old, new in _RENAMED_LLM_KEYS.items():
262
+ if old in section:
263
+ value = section.pop(old)
264
+ section.setdefault(new, value)
265
+ if (task, old) not in _noticed:
266
+ _noticed.add((task, old))
267
+ print(f"notice: [llm.{task}] {old} is now {new}", file=sys.stderr)
268
+
269
+
270
+ def _validate(raw: dict) -> None:
271
+ for name, table in raw.items():
272
+ if name in _RETIRED and isinstance(table, dict):
273
+ table = {k: v for k, v in table.items() if k not in _RETIRED[name]}
274
+ if name not in (*_TABLES, *_SECTIONS):
275
+ close = difflib.get_close_matches(name, [*_TABLES, *_SECTIONS], n=1)
276
+ raise ValueError(
277
+ f"unknown section [{name}]" + (f" (did you mean [{close[0]}]?)" if close else "")
278
+ )
279
+ if name in _TABLES:
280
+ hints = get_type_hints(Config)
281
+ _check(name, table, {k: hints.get(k, str) for k in _TABLES[name]})
282
+ for task, section in raw.get("llm", {}).items():
283
+ if not isinstance(section, dict):
284
+ raise ValueError("[llm] holds one table per model task, like [llm.summarize]")
285
+ if task not in _LLM_TASKS and task not in _RETIRED_TASKS:
286
+ close = difflib.get_close_matches(task, _LLM_TASKS, n=1)
287
+ raise ValueError(
288
+ f"unknown section [llm.{task}]"
289
+ + (
290
+ f" (did you mean [llm.{close[0]}]?)"
291
+ if close
292
+ else f"; valid: {', '.join(_LLM_TASKS)}"
293
+ )
294
+ )
295
+ _check(f"llm.{task}", section, *_dataclass_hints(LLMConfig))
296
+ for name, cls in (
297
+ ("email", EmailConfig),
298
+ ("bench", BenchConfig),
299
+ ("update", UpdateConfig),
300
+ ("network", NetworkConfig),
301
+ ):
302
+ _check(name, raw.get(name, {}), *_dataclass_hints(cls))
303
+
304
+
305
+ def _parse(raw: dict) -> Config:
306
+ _accept_renamed_keys(raw)
307
+ _validate(raw)
308
+ paths = raw.get("paths", {})
309
+ if "vault" not in paths:
310
+ raise ValueError("config.toml needs [paths].vault")
311
+ run = raw.get("run", {})
312
+ llm = {name: LLMConfig(**section) for name, section in raw.get("llm", {}).items()}
313
+ cfg = Config(
314
+ vault=Path(paths["vault"]).expanduser(),
315
+ language=raw.get("notes", {}).get("language", lang.DEFAULT),
316
+ viewer=raw.get("notes", {}).get("viewer", "obsidian"),
317
+ legacy_vault=Path(paths["legacy_vault"]).expanduser() if "legacy_vault" in paths else None,
318
+ max_source_chars=run.get("max_source_chars", 4000),
319
+ chunk_chars=run.get("chunk_chars", 8000),
320
+ max_chunks=run.get("max_chunks", 16),
321
+ ocr_max_pages=run.get("ocr_max_pages", 10),
322
+ find_connections=run.get("find_connections", True),
323
+ rewrite_questions=run.get("rewrite_questions", False),
324
+ nightly_time=run.get("nightly_time", "03:00"),
325
+ max_sources_per_run=run.get("max_sources_per_run", 20),
326
+ max_tokens_per_run=run.get("max_tokens_per_run", 300_000),
327
+ flag_contradictions=run.get("flag_contradictions", False),
328
+ llm=llm,
329
+ email=EmailConfig(**raw.get("email", {})),
330
+ bench=BenchConfig(**raw.get("bench", {})),
331
+ update=UpdateConfig(**raw.get("update", {})),
332
+ network=NetworkConfig(**raw.get("network", {})),
333
+ )
334
+ if not 1 <= cfg.email.follow_links_max <= 10:
335
+ raise ValueError(
336
+ f"[email].follow_links_max must be between 1 and 10, got {cfg.email.follow_links_max}"
337
+ )
338
+ lang.get(cfg.language) # an unsupported language is an error line, not a wrong note
339
+ if cfg.viewer not in ("obsidian", "none"):
340
+ raise ValueError(f'[notes].viewer must be "obsidian" or "none", got {cfg.viewer!r}')
341
+ parse_time(
342
+ cfg.nightly_time
343
+ ) # refuse a bad time when the config is read, not when the job fires
344
+ return cfg