esbi-cli 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- esbi_cli/__init__.py +8 -0
- esbi_cli/ask/__init__.py +0 -0
- esbi_cli/ask/answer.py +256 -0
- esbi_cli/bench/__init__.py +0 -0
- esbi_cli/bench/cases.py +57 -0
- esbi_cli/bench/metrics.py +23 -0
- esbi_cli/bench/report.py +117 -0
- esbi_cli/bench/runner.py +114 -0
- esbi_cli/capture/__init__.py +0 -0
- esbi_cli/capture/inbox.py +63 -0
- esbi_cli/capture/legacy.py +49 -0
- esbi_cli/cli.py +1387 -0
- esbi_cli/config.py +344 -0
- esbi_cli/doctor.py +391 -0
- esbi_cli/evaluate.py +91 -0
- esbi_cli/export.py +137 -0
- esbi_cli/extract/__init__.py +107 -0
- esbi_cli/extract/clip.py +30 -0
- esbi_cli/extract/html.py +60 -0
- esbi_cli/extract/image.py +58 -0
- esbi_cli/extract/pdf.py +109 -0
- esbi_cli/gitops.py +101 -0
- esbi_cli/index.py +303 -0
- esbi_cli/ingest/__init__.py +0 -0
- esbi_cli/ingest/apply.py +480 -0
- esbi_cli/ingest/chunks.py +49 -0
- esbi_cli/ingest/connect.py +87 -0
- esbi_cli/ingest/digest.py +91 -0
- esbi_cli/ingest/pipeline.py +176 -0
- esbi_cli/ingest/plan.py +231 -0
- esbi_cli/ingest/read.py +105 -0
- esbi_cli/ingest/retrieve.py +59 -0
- esbi_cli/init.py +176 -0
- esbi_cli/interrupts.py +90 -0
- esbi_cli/lang.py +341 -0
- esbi_cli/links.py +10 -0
- esbi_cli/lint/__init__.py +0 -0
- esbi_cli/lint/checks.py +178 -0
- esbi_cli/lint/report.py +60 -0
- esbi_cli/llm/__init__.py +0 -0
- esbi_cli/llm/adapter.py +393 -0
- esbi_cli/llm/schemas.py +146 -0
- esbi_cli/mail/__init__.py +0 -0
- esbi_cli/mail/convert.py +194 -0
- esbi_cli/mail/credentials.py +65 -0
- esbi_cli/mail/fetch.py +154 -0
- esbi_cli/mail/imap.py +92 -0
- esbi_cli/netguard.py +127 -0
- esbi_cli/privacy.py +81 -0
- esbi_cli/queue.py +179 -0
- esbi_cli/reingest.py +165 -0
- esbi_cli/report/__init__.py +0 -0
- esbi_cli/report/daily_index.py +235 -0
- esbi_cli/report/index_md.py +21 -0
- esbi_cli/report/readstate.py +26 -0
- esbi_cli/run.py +100 -0
- esbi_cli/runlock.py +31 -0
- esbi_cli/runlog.py +80 -0
- esbi_cli/schedule.py +106 -0
- esbi_cli/templates/SCHEMA.md +52 -0
- esbi_cli/templates/clipper-template.json +17 -0
- esbi_cli/templates/clipper-youtube-template.json +18 -0
- esbi_cli/templates/config.example.toml +108 -0
- esbi_cli/update.py +247 -0
- esbi_cli/vault.py +188 -0
- esbi_cli/wizards/clipper.sh +271 -0
- esbi_cli/wizards/email.sh +265 -0
- esbi_cli-0.2.1.dist-info/METADATA +167 -0
- esbi_cli-0.2.1.dist-info/RECORD +72 -0
- esbi_cli-0.2.1.dist-info/WHEEL +4 -0
- esbi_cli-0.2.1.dist-info/entry_points.txt +3 -0
- esbi_cli-0.2.1.dist-info/licenses/LICENSE +21 -0
esbi_cli/index.py
ADDED
|
@@ -0,0 +1,303 @@
|
|
|
1
|
+
"""A persistent index of the wiki's pages, in one SQLite file inside the vault.
|
|
2
|
+
|
|
3
|
+
Finding a page by name, a source by URL, or the pages related to a text used to mean reading and
|
|
4
|
+
parsing every page of the vault each time (a lookup took 240 ms at 1,000 pages and 1.2 s at 5,000,
|
|
5
|
+
and one ingest does about 30). The index keeps names, URLs, hashes and the full text, and only
|
|
6
|
+
re-reads the files whose size or modification time changed. It is disposable: delete the file and
|
|
7
|
+
the next lookup rebuilds it from the vault."""
|
|
8
|
+
|
|
9
|
+
import re
|
|
10
|
+
import sqlite3
|
|
11
|
+
import sys
|
|
12
|
+
from array import array
|
|
13
|
+
from math import sqrt
|
|
14
|
+
from operator import mul
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
from esbi_cli.llm.adapter import LLMError
|
|
18
|
+
from esbi_cli.privacy import public_body
|
|
19
|
+
from esbi_cli.vault import PAGE_KINDS, Page, fold, parse_page
|
|
20
|
+
|
|
21
|
+
VERSION = "2" # 2: the vectors table
|
|
22
|
+
CHUNK_CHARS = 3000 # a longer section is embedded by its first part; split it if that matters
|
|
23
|
+
QUERY_CHARS = 2000 # an ingest query is a whole source: the model reads its beginning
|
|
24
|
+
POOL = 50 # candidates taken from each side before fusing
|
|
25
|
+
RRF_K = 60 # reciprocal rank fusion constant (the usual one; 10 and 30 measured no differently)
|
|
26
|
+
_SCHEMA = """
|
|
27
|
+
CREATE TABLE IF NOT EXISTS meta (key TEXT PRIMARY KEY, value TEXT);
|
|
28
|
+
CREATE TABLE IF NOT EXISTS pages (
|
|
29
|
+
id INTEGER PRIMARY KEY, path TEXT UNIQUE, kind TEXT, ord INTEGER, title TEXT, summary TEXT,
|
|
30
|
+
url TEXT, content_hash TEXT, mtime_ns INTEGER, size INTEGER);
|
|
31
|
+
CREATE TABLE IF NOT EXISTS names (name TEXT, page_id INTEGER);
|
|
32
|
+
CREATE INDEX IF NOT EXISTS names_name ON names (name);
|
|
33
|
+
CREATE INDEX IF NOT EXISTS pages_url ON pages (url);
|
|
34
|
+
CREATE INDEX IF NOT EXISTS pages_hash ON pages (content_hash);
|
|
35
|
+
CREATE VIRTUAL TABLE IF NOT EXISTS fts USING fts5(
|
|
36
|
+
title, aliases, body, tokenize='unicode61 remove_diacritics 2');
|
|
37
|
+
CREATE TABLE IF NOT EXISTS vectors (page_id INTEGER, chunk INTEGER, vec BLOB);
|
|
38
|
+
CREATE INDEX IF NOT EXISTS vectors_page ON vectors (page_id);
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def chunks(title: str, body: str) -> list[str]:
|
|
43
|
+
"""The texts a page is embedded as: its introduction and each `## ` section, every one led by
|
|
44
|
+
the page title (a section alone often does not say what it is about)."""
|
|
45
|
+
head, *rest = re.split(r"^## ", body, flags=re.M)
|
|
46
|
+
parts = [head.strip(), *("## " + r.strip() for r in rest)]
|
|
47
|
+
return [f"{title}\n{p[:CHUNK_CHARS]}" for p in parts if p] or [title]
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _unit(vector: list[float]) -> array:
|
|
51
|
+
norm = sqrt(sum(x * x for x in vector)) or 1.0
|
|
52
|
+
return array("f", (x / norm for x in vector))
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def rrf(*rankings: list[int]) -> list[int]:
|
|
56
|
+
"""Reciprocal rank fusion: an item scores 1/(k+rank) in every list it appears in."""
|
|
57
|
+
score: dict[int, float] = {}
|
|
58
|
+
for ranking in rankings:
|
|
59
|
+
for rank, item in enumerate(ranking):
|
|
60
|
+
score[item] = score.get(item, 0.0) + 1 / (RRF_K + rank + 1)
|
|
61
|
+
return sorted(score, key=lambda item: -score[item])
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class Index:
|
|
65
|
+
def __init__(self, root: Path, embedder=None, private=None):
|
|
66
|
+
"""`embedder` adds the dense side to `search` (None: keywords only, as always). `private()`
|
|
67
|
+
returns (titles, source titles) of email-derived pages, asked only when the embedder
|
|
68
|
+
sends text out."""
|
|
69
|
+
self.embedder, self.private = embedder, private
|
|
70
|
+
self.embed_down = False # set once a call fails: the rest of this run is keywords only
|
|
71
|
+
self.root = root
|
|
72
|
+
self.db_path = root / ".esbi" / "index.sqlite3"
|
|
73
|
+
self.db_path.parent.mkdir(parents=True, exist_ok=True)
|
|
74
|
+
self.con = self._open()
|
|
75
|
+
|
|
76
|
+
def _open(self) -> sqlite3.Connection:
|
|
77
|
+
con = sqlite3.connect(self.db_path)
|
|
78
|
+
current = None
|
|
79
|
+
try:
|
|
80
|
+
current = con.execute("SELECT value FROM meta WHERE key = 'version'").fetchone()
|
|
81
|
+
except sqlite3.DatabaseError: # no tables yet, or not a database at all
|
|
82
|
+
pass
|
|
83
|
+
if current is None or current[0] != VERSION: # a different layout: rebuild from the vault
|
|
84
|
+
con.close()
|
|
85
|
+
self.db_path.unlink(missing_ok=True)
|
|
86
|
+
con = sqlite3.connect(self.db_path)
|
|
87
|
+
con.executescript(_SCHEMA)
|
|
88
|
+
con.execute("INSERT OR REPLACE INTO meta VALUES ('version', ?)", (VERSION,))
|
|
89
|
+
con.commit()
|
|
90
|
+
return con
|
|
91
|
+
|
|
92
|
+
def _on_disk(self) -> dict[str, tuple[int, int, str, int]]:
|
|
93
|
+
found = {}
|
|
94
|
+
for order, kind in enumerate(PAGE_KINDS):
|
|
95
|
+
folder = self.root / "wiki" / kind
|
|
96
|
+
for path in folder.glob("*.md") if folder.is_dir() else ():
|
|
97
|
+
stat = path.stat()
|
|
98
|
+
found[str(path)] = (stat.st_mtime_ns, stat.st_size, kind, order)
|
|
99
|
+
return found
|
|
100
|
+
|
|
101
|
+
def sync(self) -> None:
|
|
102
|
+
"""Bring the index in line with the files: re-read only what is new or changed."""
|
|
103
|
+
disk = self._on_disk()
|
|
104
|
+
known = {
|
|
105
|
+
path: (mtime_ns, size_bytes)
|
|
106
|
+
for path, mtime_ns, size_bytes in self.con.execute(
|
|
107
|
+
"SELECT path, mtime_ns, size FROM pages" # the column `size` is in bytes
|
|
108
|
+
)
|
|
109
|
+
}
|
|
110
|
+
for path in known.keys() - disk.keys():
|
|
111
|
+
self._delete(path)
|
|
112
|
+
for path, (mtime_ns, size_bytes, kind, order) in disk.items():
|
|
113
|
+
if known.get(path) != (mtime_ns, size_bytes):
|
|
114
|
+
page = parse_page(Path(path), Path(path).read_text(encoding="utf-8"))
|
|
115
|
+
self._store(page, mtime_ns, size_bytes, kind, order)
|
|
116
|
+
self.con.commit()
|
|
117
|
+
|
|
118
|
+
def upsert(self, page: Page) -> None:
|
|
119
|
+
"""Record a page the worker has just written, so the next lookup needs no re-read."""
|
|
120
|
+
path = page.path.resolve()
|
|
121
|
+
stat = path.stat()
|
|
122
|
+
kind = path.parent.name
|
|
123
|
+
if kind in PAGE_KINDS and path.parent.parent.name == "wiki":
|
|
124
|
+
self._store(page, stat.st_mtime_ns, stat.st_size, kind, PAGE_KINDS.index(kind))
|
|
125
|
+
self.con.commit()
|
|
126
|
+
|
|
127
|
+
def _delete(self, path: str) -> None:
|
|
128
|
+
row = self.con.execute("SELECT id FROM pages WHERE path = ?", (path,)).fetchone()
|
|
129
|
+
if row:
|
|
130
|
+
for table, column in (
|
|
131
|
+
("names", "page_id"),
|
|
132
|
+
("vectors", "page_id"),
|
|
133
|
+
("fts", "rowid"),
|
|
134
|
+
("pages", "id"),
|
|
135
|
+
):
|
|
136
|
+
self.con.execute(f"DELETE FROM {table} WHERE {column} = ?", (row[0],))
|
|
137
|
+
|
|
138
|
+
def _store(self, page: Page, mtime_ns: int, size_bytes: int, kind: str, order: int) -> None:
|
|
139
|
+
path = str(page.path.resolve())
|
|
140
|
+
self._delete(path)
|
|
141
|
+
meta = page.meta
|
|
142
|
+
cur = self.con.execute(
|
|
143
|
+
"INSERT INTO pages (path, kind, ord, title, summary, url, content_hash, mtime_ns, size) "
|
|
144
|
+
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
|
145
|
+
(
|
|
146
|
+
path,
|
|
147
|
+
kind,
|
|
148
|
+
order,
|
|
149
|
+
page.title,
|
|
150
|
+
str(meta.get("summary", "")),
|
|
151
|
+
str(meta["url"]) if meta.get("url") else None,
|
|
152
|
+
str(meta["content_hash"]) if meta.get("content_hash") else None,
|
|
153
|
+
mtime_ns,
|
|
154
|
+
size_bytes,
|
|
155
|
+
),
|
|
156
|
+
)
|
|
157
|
+
page_id = cur.lastrowid
|
|
158
|
+
names = {fold(str(n)) for n in (page.path.stem, page.title, *page.aliases)}
|
|
159
|
+
self.con.executemany("INSERT INTO names VALUES (?, ?)", [(n, page_id) for n in names])
|
|
160
|
+
self.con.execute(
|
|
161
|
+
"INSERT INTO fts (rowid, title, aliases, body) VALUES (?, ?, ?, ?)",
|
|
162
|
+
(page_id, page.title, " ".join(page.aliases), page.body),
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
def find_path(self, title: str, kinds: tuple[str, ...]) -> Path | None:
|
|
166
|
+
self.sync()
|
|
167
|
+
marks = ",".join("?" * len(kinds))
|
|
168
|
+
row = self.con.execute(
|
|
169
|
+
"SELECT p.path FROM pages p JOIN names n ON n.page_id = p.id "
|
|
170
|
+
f"WHERE n.name = ? AND p.kind IN ({marks}) ORDER BY p.ord, p.path LIMIT 1",
|
|
171
|
+
(fold(title), *kinds),
|
|
172
|
+
).fetchone()
|
|
173
|
+
return Path(row[0]) if row else None
|
|
174
|
+
|
|
175
|
+
def find_source_path(self, key: str, value: str) -> Path | None:
|
|
176
|
+
self.sync()
|
|
177
|
+
column = {"url": "url", "content_hash": "content_hash"}.get(key)
|
|
178
|
+
if column is None: # any other frontmatter key: not indexed, the caller scans
|
|
179
|
+
raise KeyError(key)
|
|
180
|
+
row = self.con.execute(
|
|
181
|
+
f"SELECT path FROM pages WHERE kind = 'sources' AND {column} = ? ORDER BY path LIMIT 1",
|
|
182
|
+
(value,),
|
|
183
|
+
).fetchone()
|
|
184
|
+
return Path(row[0]) if row else None
|
|
185
|
+
|
|
186
|
+
def search(
|
|
187
|
+
self,
|
|
188
|
+
words: list[str],
|
|
189
|
+
max_results: int,
|
|
190
|
+
exclude=frozenset(),
|
|
191
|
+
query: str | None = None,
|
|
192
|
+
private: bool = False,
|
|
193
|
+
) -> list[tuple[str, str, str]]:
|
|
194
|
+
"""(title, kind, summary) of the best matches, best first: keyword matches for any of
|
|
195
|
+
`words`, fused with the pages whose meaning is closest to `query` when there is an
|
|
196
|
+
embedder. `private`: the query is email text, which a remote embedder must not see."""
|
|
197
|
+
self.sync()
|
|
198
|
+
pool = max_results + len(exclude)
|
|
199
|
+
dense = self._dense(query, private)
|
|
200
|
+
if dense is not None:
|
|
201
|
+
pool = max(pool, POOL)
|
|
202
|
+
ids = self._keyword_ids(words, pool)
|
|
203
|
+
if dense is not None:
|
|
204
|
+
ids = rrf(ids, dense)
|
|
205
|
+
rows = []
|
|
206
|
+
for page_id in ids:
|
|
207
|
+
row = self.con.execute(
|
|
208
|
+
"SELECT title, kind, summary FROM pages WHERE id = ?", (page_id,)
|
|
209
|
+
).fetchone()
|
|
210
|
+
if row[0] not in exclude:
|
|
211
|
+
rows.append(row)
|
|
212
|
+
return rows[:max_results]
|
|
213
|
+
|
|
214
|
+
def _keyword_ids(self, words: list[str], pool: int) -> list[int]:
|
|
215
|
+
if not words:
|
|
216
|
+
return []
|
|
217
|
+
query = " OR ".join(f'"{w}"' for w in words)
|
|
218
|
+
rows = self.con.execute(
|
|
219
|
+
"SELECT rowid FROM fts WHERE fts MATCH ? ORDER BY bm25(fts, 5.0, 3.0, 1.0) LIMIT ?",
|
|
220
|
+
(query, pool),
|
|
221
|
+
).fetchall()
|
|
222
|
+
return [r[0] for r in rows]
|
|
223
|
+
|
|
224
|
+
def _dense(self, query: str | None, private: bool) -> list[int] | None:
|
|
225
|
+
"""Page ids by meaning, best first; None when there is no dense side to use (no embedder,
|
|
226
|
+
nothing to embed, or it failed: then search is keywords only, and says so once)."""
|
|
227
|
+
emb = self.embedder
|
|
228
|
+
if emb is None or not query or self.embed_down or (private and emb.sends_text_out):
|
|
229
|
+
return None
|
|
230
|
+
try:
|
|
231
|
+
self._embed_pending(emb)
|
|
232
|
+
vector = _unit(emb.embed([query[:QUERY_CHARS]], query=True)[0])
|
|
233
|
+
except LLMError as exc:
|
|
234
|
+
self.embed_down = True
|
|
235
|
+
print(
|
|
236
|
+
f"warning: embeddings unavailable ({exc}); searching by keyword only",
|
|
237
|
+
file=sys.stderr,
|
|
238
|
+
)
|
|
239
|
+
return None
|
|
240
|
+
best: dict[int, float] = {}
|
|
241
|
+
# Known ceiling: plain Python over every chunk (about 30 microseconds each); numpy or
|
|
242
|
+
# sqlite-vec when a vault passes some tens of thousands of chunks
|
|
243
|
+
for page_id, blob in self.con.execute("SELECT page_id, vec FROM vectors WHERE chunk >= 0"):
|
|
244
|
+
other = array("f")
|
|
245
|
+
other.frombytes(blob)
|
|
246
|
+
score = sum(map(mul, vector, other))
|
|
247
|
+
if score > best.get(page_id, -2.0):
|
|
248
|
+
best[page_id] = score
|
|
249
|
+
return sorted(best, key=lambda i: -best[i])[:POOL]
|
|
250
|
+
|
|
251
|
+
def _embed_pending(self, emb) -> None:
|
|
252
|
+
"""Embed the pages that have no vectors yet (new or changed: `_delete` drops theirs)."""
|
|
253
|
+
stored = self.con.execute("SELECT value FROM meta WHERE key = 'embedder'").fetchone()
|
|
254
|
+
if stored is None or stored[0] != emb.identity: # another model: its vectors are useless
|
|
255
|
+
self.con.execute("DELETE FROM vectors")
|
|
256
|
+
self.con.execute("INSERT OR REPLACE INTO meta VALUES ('embedder', ?)", (emb.identity,))
|
|
257
|
+
rows = self.con.execute(
|
|
258
|
+
"SELECT p.id, p.title, f.body FROM pages p JOIN fts f ON f.rowid = p.id "
|
|
259
|
+
"WHERE p.id NOT IN (SELECT page_id FROM vectors)"
|
|
260
|
+
).fetchall()
|
|
261
|
+
if not rows:
|
|
262
|
+
self.con.commit()
|
|
263
|
+
return
|
|
264
|
+
hidden, shared = self.private() if emb.sends_text_out and self.private else ((), ())
|
|
265
|
+
todo = []
|
|
266
|
+
for page_id, title, body in rows:
|
|
267
|
+
if title in hidden: # email stays here: a marker says "left out on purpose"
|
|
268
|
+
self.con.execute("INSERT INTO vectors VALUES (?, -1, x'')", (page_id,))
|
|
269
|
+
else:
|
|
270
|
+
todo.append((page_id, chunks(title, public_body(body, shared) if shared else body)))
|
|
271
|
+
for i in range(0, len(todo), 8): # whole pages per call, so a stop never leaves half a page
|
|
272
|
+
group = todo[i : i + 8]
|
|
273
|
+
texts = [c for _, cs in group for c in cs]
|
|
274
|
+
vectors = iter(emb.embed(texts))
|
|
275
|
+
for page_id, cs in group:
|
|
276
|
+
for n in range(len(cs)):
|
|
277
|
+
self.con.execute(
|
|
278
|
+
"INSERT INTO vectors VALUES (?, ?, ?)",
|
|
279
|
+
(page_id, n, _unit(next(vectors)).tobytes()),
|
|
280
|
+
)
|
|
281
|
+
self.con.commit()
|
|
282
|
+
self.con.commit()
|
|
283
|
+
|
|
284
|
+
def body_matches(self, names: list[str], sync: bool = True) -> set[str] | None:
|
|
285
|
+
"""Paths of the pages whose text contains any of `names` as whole words (case and accents
|
|
286
|
+
ignored): a superset of what lint then checks exactly. None when a name cannot be
|
|
287
|
+
searched (no letters or digits in it), so the caller falls back to looking at every page.
|
|
288
|
+
A caller asking many times in a row syncs once itself and passes `sync=False`."""
|
|
289
|
+
if sync:
|
|
290
|
+
self.sync()
|
|
291
|
+
phrases = []
|
|
292
|
+
for name in names:
|
|
293
|
+
if not re.search(r"\w", name):
|
|
294
|
+
return None
|
|
295
|
+
phrases.append('"' + name.replace('"', '""') + '"')
|
|
296
|
+
try:
|
|
297
|
+
rows = self.con.execute(
|
|
298
|
+
"SELECT p.path FROM fts JOIN pages p ON p.id = fts.rowid WHERE fts MATCH ?",
|
|
299
|
+
(f"body : ({' OR '.join(phrases)})",),
|
|
300
|
+
).fetchall()
|
|
301
|
+
except sqlite3.OperationalError:
|
|
302
|
+
return None
|
|
303
|
+
return {r[0] for r in rows}
|
|
File without changes
|