esbi-cli 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. esbi_cli/__init__.py +8 -0
  2. esbi_cli/ask/__init__.py +0 -0
  3. esbi_cli/ask/answer.py +256 -0
  4. esbi_cli/bench/__init__.py +0 -0
  5. esbi_cli/bench/cases.py +57 -0
  6. esbi_cli/bench/metrics.py +23 -0
  7. esbi_cli/bench/report.py +117 -0
  8. esbi_cli/bench/runner.py +114 -0
  9. esbi_cli/capture/__init__.py +0 -0
  10. esbi_cli/capture/inbox.py +63 -0
  11. esbi_cli/capture/legacy.py +49 -0
  12. esbi_cli/cli.py +1387 -0
  13. esbi_cli/config.py +344 -0
  14. esbi_cli/doctor.py +391 -0
  15. esbi_cli/evaluate.py +91 -0
  16. esbi_cli/export.py +137 -0
  17. esbi_cli/extract/__init__.py +107 -0
  18. esbi_cli/extract/clip.py +30 -0
  19. esbi_cli/extract/html.py +60 -0
  20. esbi_cli/extract/image.py +58 -0
  21. esbi_cli/extract/pdf.py +109 -0
  22. esbi_cli/gitops.py +101 -0
  23. esbi_cli/index.py +303 -0
  24. esbi_cli/ingest/__init__.py +0 -0
  25. esbi_cli/ingest/apply.py +480 -0
  26. esbi_cli/ingest/chunks.py +49 -0
  27. esbi_cli/ingest/connect.py +87 -0
  28. esbi_cli/ingest/digest.py +91 -0
  29. esbi_cli/ingest/pipeline.py +176 -0
  30. esbi_cli/ingest/plan.py +231 -0
  31. esbi_cli/ingest/read.py +105 -0
  32. esbi_cli/ingest/retrieve.py +59 -0
  33. esbi_cli/init.py +176 -0
  34. esbi_cli/interrupts.py +90 -0
  35. esbi_cli/lang.py +341 -0
  36. esbi_cli/links.py +10 -0
  37. esbi_cli/lint/__init__.py +0 -0
  38. esbi_cli/lint/checks.py +178 -0
  39. esbi_cli/lint/report.py +60 -0
  40. esbi_cli/llm/__init__.py +0 -0
  41. esbi_cli/llm/adapter.py +393 -0
  42. esbi_cli/llm/schemas.py +146 -0
  43. esbi_cli/mail/__init__.py +0 -0
  44. esbi_cli/mail/convert.py +194 -0
  45. esbi_cli/mail/credentials.py +65 -0
  46. esbi_cli/mail/fetch.py +154 -0
  47. esbi_cli/mail/imap.py +92 -0
  48. esbi_cli/netguard.py +127 -0
  49. esbi_cli/privacy.py +81 -0
  50. esbi_cli/queue.py +179 -0
  51. esbi_cli/reingest.py +165 -0
  52. esbi_cli/report/__init__.py +0 -0
  53. esbi_cli/report/daily_index.py +235 -0
  54. esbi_cli/report/index_md.py +21 -0
  55. esbi_cli/report/readstate.py +26 -0
  56. esbi_cli/run.py +100 -0
  57. esbi_cli/runlock.py +31 -0
  58. esbi_cli/runlog.py +80 -0
  59. esbi_cli/schedule.py +106 -0
  60. esbi_cli/templates/SCHEMA.md +52 -0
  61. esbi_cli/templates/clipper-template.json +17 -0
  62. esbi_cli/templates/clipper-youtube-template.json +18 -0
  63. esbi_cli/templates/config.example.toml +108 -0
  64. esbi_cli/update.py +247 -0
  65. esbi_cli/vault.py +188 -0
  66. esbi_cli/wizards/clipper.sh +271 -0
  67. esbi_cli/wizards/email.sh +265 -0
  68. esbi_cli-0.2.1.dist-info/METADATA +167 -0
  69. esbi_cli-0.2.1.dist-info/RECORD +72 -0
  70. esbi_cli-0.2.1.dist-info/WHEEL +4 -0
  71. esbi_cli-0.2.1.dist-info/entry_points.txt +3 -0
  72. esbi_cli-0.2.1.dist-info/licenses/LICENSE +21 -0
esbi_cli/index.py ADDED
@@ -0,0 +1,303 @@
1
+ """A persistent index of the wiki's pages, in one SQLite file inside the vault.
2
+
3
+ Finding a page by name, a source by URL, or the pages related to a text used to mean reading and
4
+ parsing every page of the vault each time (a lookup took 240 ms at 1,000 pages and 1.2 s at 5,000,
5
+ and one ingest does about 30). The index keeps names, URLs, hashes and the full text, and only
6
+ re-reads the files whose size or modification time changed. It is disposable: delete the file and
7
+ the next lookup rebuilds it from the vault."""
8
+
9
+ import re
10
+ import sqlite3
11
+ import sys
12
+ from array import array
13
+ from math import sqrt
14
+ from operator import mul
15
+ from pathlib import Path
16
+
17
+ from esbi_cli.llm.adapter import LLMError
18
+ from esbi_cli.privacy import public_body
19
+ from esbi_cli.vault import PAGE_KINDS, Page, fold, parse_page
20
+
21
+ VERSION = "2" # 2: the vectors table
22
+ CHUNK_CHARS = 3000 # a longer section is embedded by its first part; split it if that matters
23
+ QUERY_CHARS = 2000 # an ingest query is a whole source: the model reads its beginning
24
+ POOL = 50 # candidates taken from each side before fusing
25
+ RRF_K = 60 # reciprocal rank fusion constant (the usual one; 10 and 30 measured no differently)
26
+ _SCHEMA = """
27
+ CREATE TABLE IF NOT EXISTS meta (key TEXT PRIMARY KEY, value TEXT);
28
+ CREATE TABLE IF NOT EXISTS pages (
29
+ id INTEGER PRIMARY KEY, path TEXT UNIQUE, kind TEXT, ord INTEGER, title TEXT, summary TEXT,
30
+ url TEXT, content_hash TEXT, mtime_ns INTEGER, size INTEGER);
31
+ CREATE TABLE IF NOT EXISTS names (name TEXT, page_id INTEGER);
32
+ CREATE INDEX IF NOT EXISTS names_name ON names (name);
33
+ CREATE INDEX IF NOT EXISTS pages_url ON pages (url);
34
+ CREATE INDEX IF NOT EXISTS pages_hash ON pages (content_hash);
35
+ CREATE VIRTUAL TABLE IF NOT EXISTS fts USING fts5(
36
+ title, aliases, body, tokenize='unicode61 remove_diacritics 2');
37
+ CREATE TABLE IF NOT EXISTS vectors (page_id INTEGER, chunk INTEGER, vec BLOB);
38
+ CREATE INDEX IF NOT EXISTS vectors_page ON vectors (page_id);
39
+ """
40
+
41
+
42
+ def chunks(title: str, body: str) -> list[str]:
43
+ """The texts a page is embedded as: its introduction and each `## ` section, every one led by
44
+ the page title (a section alone often does not say what it is about)."""
45
+ head, *rest = re.split(r"^## ", body, flags=re.M)
46
+ parts = [head.strip(), *("## " + r.strip() for r in rest)]
47
+ return [f"{title}\n{p[:CHUNK_CHARS]}" for p in parts if p] or [title]
48
+
49
+
50
+ def _unit(vector: list[float]) -> array:
51
+ norm = sqrt(sum(x * x for x in vector)) or 1.0
52
+ return array("f", (x / norm for x in vector))
53
+
54
+
55
+ def rrf(*rankings: list[int]) -> list[int]:
56
+ """Reciprocal rank fusion: an item scores 1/(k+rank) in every list it appears in."""
57
+ score: dict[int, float] = {}
58
+ for ranking in rankings:
59
+ for rank, item in enumerate(ranking):
60
+ score[item] = score.get(item, 0.0) + 1 / (RRF_K + rank + 1)
61
+ return sorted(score, key=lambda item: -score[item])
62
+
63
+
64
+ class Index:
65
+ def __init__(self, root: Path, embedder=None, private=None):
66
+ """`embedder` adds the dense side to `search` (None: keywords only, as always). `private()`
67
+ returns (titles, source titles) of email-derived pages, asked only when the embedder
68
+ sends text out."""
69
+ self.embedder, self.private = embedder, private
70
+ self.embed_down = False # set once a call fails: the rest of this run is keywords only
71
+ self.root = root
72
+ self.db_path = root / ".esbi" / "index.sqlite3"
73
+ self.db_path.parent.mkdir(parents=True, exist_ok=True)
74
+ self.con = self._open()
75
+
76
+ def _open(self) -> sqlite3.Connection:
77
+ con = sqlite3.connect(self.db_path)
78
+ current = None
79
+ try:
80
+ current = con.execute("SELECT value FROM meta WHERE key = 'version'").fetchone()
81
+ except sqlite3.DatabaseError: # no tables yet, or not a database at all
82
+ pass
83
+ if current is None or current[0] != VERSION: # a different layout: rebuild from the vault
84
+ con.close()
85
+ self.db_path.unlink(missing_ok=True)
86
+ con = sqlite3.connect(self.db_path)
87
+ con.executescript(_SCHEMA)
88
+ con.execute("INSERT OR REPLACE INTO meta VALUES ('version', ?)", (VERSION,))
89
+ con.commit()
90
+ return con
91
+
92
+ def _on_disk(self) -> dict[str, tuple[int, int, str, int]]:
93
+ found = {}
94
+ for order, kind in enumerate(PAGE_KINDS):
95
+ folder = self.root / "wiki" / kind
96
+ for path in folder.glob("*.md") if folder.is_dir() else ():
97
+ stat = path.stat()
98
+ found[str(path)] = (stat.st_mtime_ns, stat.st_size, kind, order)
99
+ return found
100
+
101
+ def sync(self) -> None:
102
+ """Bring the index in line with the files: re-read only what is new or changed."""
103
+ disk = self._on_disk()
104
+ known = {
105
+ path: (mtime_ns, size_bytes)
106
+ for path, mtime_ns, size_bytes in self.con.execute(
107
+ "SELECT path, mtime_ns, size FROM pages" # the column `size` is in bytes
108
+ )
109
+ }
110
+ for path in known.keys() - disk.keys():
111
+ self._delete(path)
112
+ for path, (mtime_ns, size_bytes, kind, order) in disk.items():
113
+ if known.get(path) != (mtime_ns, size_bytes):
114
+ page = parse_page(Path(path), Path(path).read_text(encoding="utf-8"))
115
+ self._store(page, mtime_ns, size_bytes, kind, order)
116
+ self.con.commit()
117
+
118
+ def upsert(self, page: Page) -> None:
119
+ """Record a page the worker has just written, so the next lookup needs no re-read."""
120
+ path = page.path.resolve()
121
+ stat = path.stat()
122
+ kind = path.parent.name
123
+ if kind in PAGE_KINDS and path.parent.parent.name == "wiki":
124
+ self._store(page, stat.st_mtime_ns, stat.st_size, kind, PAGE_KINDS.index(kind))
125
+ self.con.commit()
126
+
127
+ def _delete(self, path: str) -> None:
128
+ row = self.con.execute("SELECT id FROM pages WHERE path = ?", (path,)).fetchone()
129
+ if row:
130
+ for table, column in (
131
+ ("names", "page_id"),
132
+ ("vectors", "page_id"),
133
+ ("fts", "rowid"),
134
+ ("pages", "id"),
135
+ ):
136
+ self.con.execute(f"DELETE FROM {table} WHERE {column} = ?", (row[0],))
137
+
138
+ def _store(self, page: Page, mtime_ns: int, size_bytes: int, kind: str, order: int) -> None:
139
+ path = str(page.path.resolve())
140
+ self._delete(path)
141
+ meta = page.meta
142
+ cur = self.con.execute(
143
+ "INSERT INTO pages (path, kind, ord, title, summary, url, content_hash, mtime_ns, size) "
144
+ "VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
145
+ (
146
+ path,
147
+ kind,
148
+ order,
149
+ page.title,
150
+ str(meta.get("summary", "")),
151
+ str(meta["url"]) if meta.get("url") else None,
152
+ str(meta["content_hash"]) if meta.get("content_hash") else None,
153
+ mtime_ns,
154
+ size_bytes,
155
+ ),
156
+ )
157
+ page_id = cur.lastrowid
158
+ names = {fold(str(n)) for n in (page.path.stem, page.title, *page.aliases)}
159
+ self.con.executemany("INSERT INTO names VALUES (?, ?)", [(n, page_id) for n in names])
160
+ self.con.execute(
161
+ "INSERT INTO fts (rowid, title, aliases, body) VALUES (?, ?, ?, ?)",
162
+ (page_id, page.title, " ".join(page.aliases), page.body),
163
+ )
164
+
165
+ def find_path(self, title: str, kinds: tuple[str, ...]) -> Path | None:
166
+ self.sync()
167
+ marks = ",".join("?" * len(kinds))
168
+ row = self.con.execute(
169
+ "SELECT p.path FROM pages p JOIN names n ON n.page_id = p.id "
170
+ f"WHERE n.name = ? AND p.kind IN ({marks}) ORDER BY p.ord, p.path LIMIT 1",
171
+ (fold(title), *kinds),
172
+ ).fetchone()
173
+ return Path(row[0]) if row else None
174
+
175
+ def find_source_path(self, key: str, value: str) -> Path | None:
176
+ self.sync()
177
+ column = {"url": "url", "content_hash": "content_hash"}.get(key)
178
+ if column is None: # any other frontmatter key: not indexed, the caller scans
179
+ raise KeyError(key)
180
+ row = self.con.execute(
181
+ f"SELECT path FROM pages WHERE kind = 'sources' AND {column} = ? ORDER BY path LIMIT 1",
182
+ (value,),
183
+ ).fetchone()
184
+ return Path(row[0]) if row else None
185
+
186
+ def search(
187
+ self,
188
+ words: list[str],
189
+ max_results: int,
190
+ exclude=frozenset(),
191
+ query: str | None = None,
192
+ private: bool = False,
193
+ ) -> list[tuple[str, str, str]]:
194
+ """(title, kind, summary) of the best matches, best first: keyword matches for any of
195
+ `words`, fused with the pages whose meaning is closest to `query` when there is an
196
+ embedder. `private`: the query is email text, which a remote embedder must not see."""
197
+ self.sync()
198
+ pool = max_results + len(exclude)
199
+ dense = self._dense(query, private)
200
+ if dense is not None:
201
+ pool = max(pool, POOL)
202
+ ids = self._keyword_ids(words, pool)
203
+ if dense is not None:
204
+ ids = rrf(ids, dense)
205
+ rows = []
206
+ for page_id in ids:
207
+ row = self.con.execute(
208
+ "SELECT title, kind, summary FROM pages WHERE id = ?", (page_id,)
209
+ ).fetchone()
210
+ if row[0] not in exclude:
211
+ rows.append(row)
212
+ return rows[:max_results]
213
+
214
+ def _keyword_ids(self, words: list[str], pool: int) -> list[int]:
215
+ if not words:
216
+ return []
217
+ query = " OR ".join(f'"{w}"' for w in words)
218
+ rows = self.con.execute(
219
+ "SELECT rowid FROM fts WHERE fts MATCH ? ORDER BY bm25(fts, 5.0, 3.0, 1.0) LIMIT ?",
220
+ (query, pool),
221
+ ).fetchall()
222
+ return [r[0] for r in rows]
223
+
224
+ def _dense(self, query: str | None, private: bool) -> list[int] | None:
225
+ """Page ids by meaning, best first; None when there is no dense side to use (no embedder,
226
+ nothing to embed, or it failed: then search is keywords only, and says so once)."""
227
+ emb = self.embedder
228
+ if emb is None or not query or self.embed_down or (private and emb.sends_text_out):
229
+ return None
230
+ try:
231
+ self._embed_pending(emb)
232
+ vector = _unit(emb.embed([query[:QUERY_CHARS]], query=True)[0])
233
+ except LLMError as exc:
234
+ self.embed_down = True
235
+ print(
236
+ f"warning: embeddings unavailable ({exc}); searching by keyword only",
237
+ file=sys.stderr,
238
+ )
239
+ return None
240
+ best: dict[int, float] = {}
241
+ # Known ceiling: plain Python over every chunk (about 30 microseconds each); numpy or
242
+ # sqlite-vec when a vault passes some tens of thousands of chunks
243
+ for page_id, blob in self.con.execute("SELECT page_id, vec FROM vectors WHERE chunk >= 0"):
244
+ other = array("f")
245
+ other.frombytes(blob)
246
+ score = sum(map(mul, vector, other))
247
+ if score > best.get(page_id, -2.0):
248
+ best[page_id] = score
249
+ return sorted(best, key=lambda i: -best[i])[:POOL]
250
+
251
+ def _embed_pending(self, emb) -> None:
252
+ """Embed the pages that have no vectors yet (new or changed: `_delete` drops theirs)."""
253
+ stored = self.con.execute("SELECT value FROM meta WHERE key = 'embedder'").fetchone()
254
+ if stored is None or stored[0] != emb.identity: # another model: its vectors are useless
255
+ self.con.execute("DELETE FROM vectors")
256
+ self.con.execute("INSERT OR REPLACE INTO meta VALUES ('embedder', ?)", (emb.identity,))
257
+ rows = self.con.execute(
258
+ "SELECT p.id, p.title, f.body FROM pages p JOIN fts f ON f.rowid = p.id "
259
+ "WHERE p.id NOT IN (SELECT page_id FROM vectors)"
260
+ ).fetchall()
261
+ if not rows:
262
+ self.con.commit()
263
+ return
264
+ hidden, shared = self.private() if emb.sends_text_out and self.private else ((), ())
265
+ todo = []
266
+ for page_id, title, body in rows:
267
+ if title in hidden: # email stays here: a marker says "left out on purpose"
268
+ self.con.execute("INSERT INTO vectors VALUES (?, -1, x'')", (page_id,))
269
+ else:
270
+ todo.append((page_id, chunks(title, public_body(body, shared) if shared else body)))
271
+ for i in range(0, len(todo), 8): # whole pages per call, so a stop never leaves half a page
272
+ group = todo[i : i + 8]
273
+ texts = [c for _, cs in group for c in cs]
274
+ vectors = iter(emb.embed(texts))
275
+ for page_id, cs in group:
276
+ for n in range(len(cs)):
277
+ self.con.execute(
278
+ "INSERT INTO vectors VALUES (?, ?, ?)",
279
+ (page_id, n, _unit(next(vectors)).tobytes()),
280
+ )
281
+ self.con.commit()
282
+ self.con.commit()
283
+
284
+ def body_matches(self, names: list[str], sync: bool = True) -> set[str] | None:
285
+ """Paths of the pages whose text contains any of `names` as whole words (case and accents
286
+ ignored): a superset of what lint then checks exactly. None when a name cannot be
287
+ searched (no letters or digits in it), so the caller falls back to looking at every page.
288
+ A caller asking many times in a row syncs once itself and passes `sync=False`."""
289
+ if sync:
290
+ self.sync()
291
+ phrases = []
292
+ for name in names:
293
+ if not re.search(r"\w", name):
294
+ return None
295
+ phrases.append('"' + name.replace('"', '""') + '"')
296
+ try:
297
+ rows = self.con.execute(
298
+ "SELECT p.path FROM fts JOIN pages p ON p.id = fts.rowid WHERE fts MATCH ?",
299
+ (f"body : ({' OR '.join(phrases)})",),
300
+ ).fetchall()
301
+ except sqlite3.OperationalError:
302
+ return None
303
+ return {r[0] for r in rows}
File without changes