esbi-cli 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- esbi_cli/__init__.py +8 -0
- esbi_cli/ask/__init__.py +0 -0
- esbi_cli/ask/answer.py +256 -0
- esbi_cli/bench/__init__.py +0 -0
- esbi_cli/bench/cases.py +57 -0
- esbi_cli/bench/metrics.py +23 -0
- esbi_cli/bench/report.py +117 -0
- esbi_cli/bench/runner.py +114 -0
- esbi_cli/capture/__init__.py +0 -0
- esbi_cli/capture/inbox.py +63 -0
- esbi_cli/capture/legacy.py +49 -0
- esbi_cli/cli.py +1387 -0
- esbi_cli/config.py +344 -0
- esbi_cli/doctor.py +391 -0
- esbi_cli/evaluate.py +91 -0
- esbi_cli/export.py +137 -0
- esbi_cli/extract/__init__.py +107 -0
- esbi_cli/extract/clip.py +30 -0
- esbi_cli/extract/html.py +60 -0
- esbi_cli/extract/image.py +58 -0
- esbi_cli/extract/pdf.py +109 -0
- esbi_cli/gitops.py +101 -0
- esbi_cli/index.py +303 -0
- esbi_cli/ingest/__init__.py +0 -0
- esbi_cli/ingest/apply.py +480 -0
- esbi_cli/ingest/chunks.py +49 -0
- esbi_cli/ingest/connect.py +87 -0
- esbi_cli/ingest/digest.py +91 -0
- esbi_cli/ingest/pipeline.py +176 -0
- esbi_cli/ingest/plan.py +231 -0
- esbi_cli/ingest/read.py +105 -0
- esbi_cli/ingest/retrieve.py +59 -0
- esbi_cli/init.py +176 -0
- esbi_cli/interrupts.py +90 -0
- esbi_cli/lang.py +341 -0
- esbi_cli/links.py +10 -0
- esbi_cli/lint/__init__.py +0 -0
- esbi_cli/lint/checks.py +178 -0
- esbi_cli/lint/report.py +60 -0
- esbi_cli/llm/__init__.py +0 -0
- esbi_cli/llm/adapter.py +393 -0
- esbi_cli/llm/schemas.py +146 -0
- esbi_cli/mail/__init__.py +0 -0
- esbi_cli/mail/convert.py +194 -0
- esbi_cli/mail/credentials.py +65 -0
- esbi_cli/mail/fetch.py +154 -0
- esbi_cli/mail/imap.py +92 -0
- esbi_cli/netguard.py +127 -0
- esbi_cli/privacy.py +81 -0
- esbi_cli/queue.py +179 -0
- esbi_cli/reingest.py +165 -0
- esbi_cli/report/__init__.py +0 -0
- esbi_cli/report/daily_index.py +235 -0
- esbi_cli/report/index_md.py +21 -0
- esbi_cli/report/readstate.py +26 -0
- esbi_cli/run.py +100 -0
- esbi_cli/runlock.py +31 -0
- esbi_cli/runlog.py +80 -0
- esbi_cli/schedule.py +106 -0
- esbi_cli/templates/SCHEMA.md +52 -0
- esbi_cli/templates/clipper-template.json +17 -0
- esbi_cli/templates/clipper-youtube-template.json +18 -0
- esbi_cli/templates/config.example.toml +108 -0
- esbi_cli/update.py +247 -0
- esbi_cli/vault.py +188 -0
- esbi_cli/wizards/clipper.sh +271 -0
- esbi_cli/wizards/email.sh +265 -0
- esbi_cli-0.2.1.dist-info/METADATA +167 -0
- esbi_cli-0.2.1.dist-info/RECORD +72 -0
- esbi_cli-0.2.1.dist-info/WHEEL +4 -0
- esbi_cli-0.2.1.dist-info/entry_points.txt +3 -0
- esbi_cli-0.2.1.dist-info/licenses/LICENSE +21 -0
esbi_cli/llm/schemas.py
ADDED
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
"""The edit plan: the only thing the LLM produces. The worker validates and applies it."""
|
|
2
|
+
|
|
3
|
+
from typing import Annotated
|
|
4
|
+
|
|
5
|
+
from annotated_types import MaxLen
|
|
6
|
+
from pydantic import BaseModel, BeforeValidator, Field, field_validator
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def _clamp(max_items: int):
|
|
10
|
+
def clamp(value):
|
|
11
|
+
return value[:max_items] if isinstance(value, list) else value
|
|
12
|
+
|
|
13
|
+
return BeforeValidator(clamp)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class ConceptEdit(BaseModel):
|
|
17
|
+
"""A concept or entity this source touches."""
|
|
18
|
+
|
|
19
|
+
title: str = Field(
|
|
20
|
+
min_length=2, max_length=60, description="Canonical name, short (at most 6 words)"
|
|
21
|
+
)
|
|
22
|
+
aliases: list[str] = Field(default_factory=list, description="Other names or acronyms")
|
|
23
|
+
description: str = Field(
|
|
24
|
+
min_length=10,
|
|
25
|
+
description="1-3 sentences on what this source adds to this concept or entity",
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class Term(BaseModel):
|
|
30
|
+
"""A technical term used in the source, with what it means."""
|
|
31
|
+
|
|
32
|
+
term: str = Field(
|
|
33
|
+
min_length=2, max_length=60, description="Exactly as it appears in the source"
|
|
34
|
+
)
|
|
35
|
+
definition: str = Field(min_length=10, description="Definition in one sentence")
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class Relation(BaseModel):
|
|
39
|
+
"""A link between two ideas, drawn later as an edge of the concept map."""
|
|
40
|
+
|
|
41
|
+
a: str = Field(min_length=2, max_length=50, description="Name of a concept or term, 1-4 words")
|
|
42
|
+
relation: str = Field(
|
|
43
|
+
min_length=2, max_length=40, description="A short label for how the two relate"
|
|
44
|
+
)
|
|
45
|
+
b: str = Field(min_length=2, max_length=50, description="Name of a concept or term, 1-4 words")
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
class ChunkNotes(BaseModel):
|
|
49
|
+
"""What one reads out of one chunk of a long source."""
|
|
50
|
+
|
|
51
|
+
points: Annotated[list[str], _clamp(8), MaxLen(8)] = Field(
|
|
52
|
+
min_length=1,
|
|
53
|
+
description="3-8 concrete statements: facts, methods, results, arguments",
|
|
54
|
+
)
|
|
55
|
+
terms: Annotated[list[Term], _clamp(6), MaxLen(6)] = Field(default_factory=list)
|
|
56
|
+
quotes: Annotated[list[str], _clamp(3), MaxLen(3)] = Field(
|
|
57
|
+
default_factory=list, description="Sentences copied EXACTLY from the chunk"
|
|
58
|
+
)
|
|
59
|
+
relations: Annotated[list[Relation], _clamp(6), MaxLen(6)] = Field(default_factory=list)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class Insight(BaseModel):
|
|
63
|
+
idea: str = Field(min_length=10, description="The idea, in one or two sentences")
|
|
64
|
+
why: str = Field(min_length=10, description="What follows from the idea, in one sentence")
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class Digest(BaseModel):
|
|
68
|
+
"""The long-form part of a note, written from the chunk notes."""
|
|
69
|
+
|
|
70
|
+
# a list, not one string: a string has no bound a model's grammar can enforce, and a small model
|
|
71
|
+
# repeated paragraphs in it until the token cap in 5 of 20 calls (0 of 10 as a list of at most 5)
|
|
72
|
+
paragraphs: Annotated[list[str], _clamp(4), MaxLen(4)] = Field(
|
|
73
|
+
min_length=3,
|
|
74
|
+
description="Detailed summary: the problem, the approach, the findings, the implications",
|
|
75
|
+
)
|
|
76
|
+
insights: Annotated[list[Insight], _clamp(6), MaxLen(6)] = Field(min_length=1)
|
|
77
|
+
open_questions: Annotated[list[str], _clamp(5), MaxLen(5)] = Field(default_factory=list)
|
|
78
|
+
|
|
79
|
+
@field_validator("paragraphs")
|
|
80
|
+
@classmethod
|
|
81
|
+
def _enough_to_read(cls, value: list[str]) -> list[str]:
|
|
82
|
+
if sum(len(p.strip()) for p in value) < 200:
|
|
83
|
+
raise ValueError("the detailed summary is too short: write real paragraphs")
|
|
84
|
+
return value
|
|
85
|
+
|
|
86
|
+
@property
|
|
87
|
+
def abstract(self) -> str:
|
|
88
|
+
return "\n\n".join(p.strip() for p in self.paragraphs)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
class Connection(BaseModel):
|
|
92
|
+
"""How a new source relates to a page already in the wiki."""
|
|
93
|
+
|
|
94
|
+
page: str = Field(description="EXACT title of an existing page")
|
|
95
|
+
relation: str = Field(
|
|
96
|
+
min_length=2, max_length=40, description="A short label for how the two relate"
|
|
97
|
+
)
|
|
98
|
+
why: str = Field(min_length=10, description="One sentence: why they are related")
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
class ConnectionPlan(BaseModel):
|
|
102
|
+
connections: Annotated[list[Connection], _clamp(8), MaxLen(8)] = Field(default_factory=list)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
class Contradiction(BaseModel):
|
|
106
|
+
page: str = Field(description="EXACT title of the existing page that is contradicted")
|
|
107
|
+
note: str = Field(min_length=5, description="What is contradicted, in one sentence")
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
class EditPlan(BaseModel):
|
|
111
|
+
title: str = Field(min_length=3, description="Title of the source")
|
|
112
|
+
one_liner: str = Field(min_length=10, description="One-sentence summary for the index")
|
|
113
|
+
summary: str = Field(
|
|
114
|
+
min_length=30, description="Executive summary: 2-3 sentences, what it is and why it matters"
|
|
115
|
+
)
|
|
116
|
+
abstract: str = Field(
|
|
117
|
+
default="",
|
|
118
|
+
description="Detailed summary, 3-5 paragraphs: problem, approach, findings, implications",
|
|
119
|
+
)
|
|
120
|
+
insights: Annotated[list[Insight], _clamp(8), MaxLen(8)] = Field(default_factory=list)
|
|
121
|
+
terms: Annotated[list[Term], _clamp(10), MaxLen(10)] = Field(default_factory=list)
|
|
122
|
+
quotes: Annotated[list[str], _clamp(6), MaxLen(6)] = Field(
|
|
123
|
+
default_factory=list, description="Sentences copied EXACTLY from the source"
|
|
124
|
+
)
|
|
125
|
+
relations: Annotated[list[Relation], _clamp(10), MaxLen(10)] = Field(default_factory=list)
|
|
126
|
+
open_questions: Annotated[list[str], _clamp(5), MaxLen(5)] = Field(default_factory=list)
|
|
127
|
+
key_points: Annotated[list[str], _clamp(8), MaxLen(8)] = Field(description="3-8 key points")
|
|
128
|
+
tags: Annotated[list[str], _clamp(6), MaxLen(6)] = Field(default_factory=list)
|
|
129
|
+
concepts: Annotated[list[ConceptEdit], _clamp(6), MaxLen(6)] = Field(
|
|
130
|
+
min_length=1, description="2-6 central concepts or techniques (never empty)"
|
|
131
|
+
)
|
|
132
|
+
entities: Annotated[list[ConceptEdit], _clamp(5), MaxLen(5)] = Field(default_factory=list)
|
|
133
|
+
related_pages: Annotated[list[str], _clamp(6), MaxLen(6)] = Field(
|
|
134
|
+
default_factory=list,
|
|
135
|
+
description="EXACT titles of related existing pages (only from the given list)",
|
|
136
|
+
)
|
|
137
|
+
contradictions: Annotated[list[Contradiction], _clamp(4), MaxLen(4)] = Field(
|
|
138
|
+
default_factory=list
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
@field_validator("one_liner")
|
|
142
|
+
@classmethod
|
|
143
|
+
def _one_liner_is_a_sentence(cls, value: str) -> str:
|
|
144
|
+
if value.strip().lower().startswith(("http://", "https://")) or " " not in value.strip():
|
|
145
|
+
raise ValueError("must be a descriptive sentence, not a URL")
|
|
146
|
+
return value
|
|
File without changes
|
esbi_cli/mail/convert.py
ADDED
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
"""Turn a raw email (RFC 5322 bytes) into a Web Clipper-style note for the vault's inbox."""
|
|
2
|
+
|
|
3
|
+
import email
|
|
4
|
+
import hashlib
|
|
5
|
+
import io
|
|
6
|
+
import re
|
|
7
|
+
from dataclasses import dataclass, field
|
|
8
|
+
from datetime import date
|
|
9
|
+
from email import policy
|
|
10
|
+
from email.message import EmailMessage
|
|
11
|
+
from email.utils import parsedate_to_datetime
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from urllib.parse import urlparse
|
|
14
|
+
|
|
15
|
+
import lxml.html
|
|
16
|
+
import trafilatura
|
|
17
|
+
from PIL import Image
|
|
18
|
+
|
|
19
|
+
from esbi_cli import lang
|
|
20
|
+
from esbi_cli.vault import Page, safe_title
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass
|
|
24
|
+
class ClipNote:
|
|
25
|
+
filename: str
|
|
26
|
+
content: str # empty when the mail is only attachments
|
|
27
|
+
source: str # stable id of the mail (`mail:<message-id>`), used to spot duplicates
|
|
28
|
+
pdfs: list[tuple[str, bytes]] = field(default_factory=list) # (file name, bytes) attached
|
|
29
|
+
images: list[tuple[str, bytes]] = field(default_factory=list) # (safe file name, bytes) kept
|
|
30
|
+
links: list[str] = field(default_factory=list) # http(s) links worth following, in order
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
# Newsletters pad the preview text with these (zero-width and joiner characters, soft hyphens)
|
|
34
|
+
INVISIBLE = re.compile("[\u034f\u200b-\u200f\u2060\u00ad\ufeff]")
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _clean(text: str) -> str:
|
|
38
|
+
"""Drop the invisible padding, then the blank lines it leaves behind."""
|
|
39
|
+
text = INVISIBLE.sub("", text)
|
|
40
|
+
text = "\n".join(line.rstrip() for line in text.splitlines())
|
|
41
|
+
return re.sub(r"\n{3,}", "\n\n", text).strip()
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _html_to_text(html: str) -> str:
|
|
45
|
+
"""Article extraction when the email looks like one, else the page's visible text."""
|
|
46
|
+
extracted = trafilatura.extract(html, output_format="markdown", favor_recall=True)
|
|
47
|
+
if extracted and len(extracted.strip()) > 40:
|
|
48
|
+
return extracted.strip()
|
|
49
|
+
tree = lxml.html.fromstring(html)
|
|
50
|
+
for junk in tree.xpath("//script | //style | //head"):
|
|
51
|
+
junk.drop_tree()
|
|
52
|
+
return re.sub(r"\n{3,}", "\n\n", tree.text_content()).strip()
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
MIN_PLAIN_CHARS = 100 # shorter plain parts are usually a "view in browser" stub
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _body(msg: EmailMessage) -> str:
|
|
59
|
+
plain = msg.get_body(preferencelist=("plain",))
|
|
60
|
+
html = msg.get_body(preferencelist=("html",))
|
|
61
|
+
plain_text = plain.get_content().strip() if plain else ""
|
|
62
|
+
if plain_text and (len(plain_text) >= MIN_PLAIN_CHARS or html is None):
|
|
63
|
+
return plain_text
|
|
64
|
+
return _html_to_text(html.get_content()) if html else plain_text
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
# Image attachments: what the bytes are decides, never the declared type or the file name.
|
|
68
|
+
IMAGE_FORMATS = { # declared content type -> (format Pillow must find in the bytes, extension)
|
|
69
|
+
"image/png": ("PNG", ".png"),
|
|
70
|
+
"image/jpeg": ("JPEG", ".jpg"),
|
|
71
|
+
"image/webp": ("WEBP", ".webp"),
|
|
72
|
+
"image/tiff": ("TIFF", ".tiff"),
|
|
73
|
+
}
|
|
74
|
+
MIN_IMAGE_SIDE_PX = 200 # signature logos, icons and tracking pixels are smaller
|
|
75
|
+
MIN_IMAGE_BYTES = 5_000 # ...or flat: a screenshot or a photo is never this light
|
|
76
|
+
MAX_IMAGE_BYTES = 5_000_000
|
|
77
|
+
MAX_MAIL_IMAGE_BYTES = 15_000_000
|
|
78
|
+
MAX_IMAGE_PIXELS = 50_000_000 # the decoder allocates width x height: refuse absurd sizes
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _image_name(data: bytes, declared: str, filename: str | None) -> str | None:
|
|
82
|
+
"""`<safe stem><extension>` for an image worth keeping, else None."""
|
|
83
|
+
if declared not in IMAGE_FORMATS or not MIN_IMAGE_BYTES <= len(data) <= MAX_IMAGE_BYTES:
|
|
84
|
+
return None
|
|
85
|
+
try:
|
|
86
|
+
with Image.open(io.BytesIO(data)) as img:
|
|
87
|
+
fmt, (width, height) = img.format, img.size
|
|
88
|
+
except Exception: # Pillow raises several unrelated error types for bytes that are no image
|
|
89
|
+
return None
|
|
90
|
+
expected, suffix = IMAGE_FORMATS[declared]
|
|
91
|
+
if (
|
|
92
|
+
fmt != expected
|
|
93
|
+
or min(width, height) < MIN_IMAGE_SIDE_PX
|
|
94
|
+
or width * height > MAX_IMAGE_PIXELS
|
|
95
|
+
):
|
|
96
|
+
return None
|
|
97
|
+
stem = Path((filename or "").replace("\\", "/")).stem # never a path: only the last part
|
|
98
|
+
return f"{safe_title(stem, 40) or 'image'}{suffix}"
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _images(msg: EmailMessage) -> list[tuple[str, bytes]]:
|
|
102
|
+
kept, total = [], 0
|
|
103
|
+
for part in msg.walk():
|
|
104
|
+
if part.is_multipart() or not part.get_content_type().startswith("image/"):
|
|
105
|
+
continue
|
|
106
|
+
data = part.get_content()
|
|
107
|
+
name = _image_name(data, part.get_content_type(), part.get_filename())
|
|
108
|
+
if name and total + len(data) <= MAX_MAIL_IMAGE_BYTES:
|
|
109
|
+
kept.append((name, data))
|
|
110
|
+
total += len(data)
|
|
111
|
+
return kept
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
_URL = re.compile(r"https?://[^\s<>\"')\]]+")
|
|
115
|
+
_NOT_AN_ARTICLE = re.compile(
|
|
116
|
+
r"unsubscri|opt-?out|preferences|view-?in-?browser|view_in_browser|web-?version|"
|
|
117
|
+
r"manage[-_]?subscription|confirm|verif|reset|log-?in|sign-?in|magic|activat|password|token=|"
|
|
118
|
+
r"\.(?:gif|png|jpe?g|svg|webp|css|js|ico)(?:\?|$)",
|
|
119
|
+
re.IGNORECASE,
|
|
120
|
+
)
|
|
121
|
+
_REDIRECT_HOSTS = (
|
|
122
|
+
"click",
|
|
123
|
+
"clicks",
|
|
124
|
+
"track",
|
|
125
|
+
"tracking",
|
|
126
|
+
"links",
|
|
127
|
+
"trk",
|
|
128
|
+
"email",
|
|
129
|
+
"em",
|
|
130
|
+
) # click.news.test
|
|
131
|
+
MAX_LINK_CHARS = 500
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _links(text: str) -> list[str]:
|
|
135
|
+
"""http(s) links in the mail's text, in order and without repeats, minus tracking redirects,
|
|
136
|
+
unsubscribe/preferences/view-in-browser housekeeping and pictures. ponytail: only links that
|
|
137
|
+
show in the text; a link hidden behind anchor text in an HTML-only mail is not found."""
|
|
138
|
+
found: list[str] = []
|
|
139
|
+
for url in (u.rstrip(".,;:!?") for u in _URL.findall(text)):
|
|
140
|
+
host = urlparse(url).hostname or ""
|
|
141
|
+
if (
|
|
142
|
+
len(url) <= MAX_LINK_CHARS
|
|
143
|
+
and host
|
|
144
|
+
and not _NOT_AN_ARTICLE.search(url)
|
|
145
|
+
and not (host.count(".") >= 2 and host.partition(".")[0] in _REDIRECT_HOSTS)
|
|
146
|
+
and host != "list-manage.com"
|
|
147
|
+
and not host.endswith(".list-manage.com")
|
|
148
|
+
and url not in found
|
|
149
|
+
):
|
|
150
|
+
found.append(url)
|
|
151
|
+
return found
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _sent_date(msg: EmailMessage) -> date:
|
|
155
|
+
try:
|
|
156
|
+
return parsedate_to_datetime(str(msg["Date"])).date()
|
|
157
|
+
except (TypeError, ValueError):
|
|
158
|
+
return date.today()
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def email_to_clip(raw: bytes, language: str = lang.DEFAULT) -> ClipNote:
|
|
162
|
+
msg = email.message_from_bytes(raw, policy=policy.default)
|
|
163
|
+
subject = str(msg["Subject"] or lang.t(language, "no_subject")).strip()
|
|
164
|
+
body = _clean(_body(msg))
|
|
165
|
+
pdfs = [
|
|
166
|
+
(part.get_filename() or "attachment.pdf", part.get_content())
|
|
167
|
+
for part in msg.iter_attachments()
|
|
168
|
+
if part.get_content_type() == "application/pdf"
|
|
169
|
+
]
|
|
170
|
+
images = _images(msg)
|
|
171
|
+
if not body and not pdfs and not images:
|
|
172
|
+
raise ValueError("the email has no readable text")
|
|
173
|
+
sent = _sent_date(msg).isoformat()
|
|
174
|
+
# an id is used as a source name and written into notes: letters, digits and . @ + = _ - only
|
|
175
|
+
message_id = (
|
|
176
|
+
re.sub(r"[^\w.@+=-]", "", str(msg["Message-ID"] or ""))[:150]
|
|
177
|
+
or hashlib.sha256(raw).hexdigest()[:16]
|
|
178
|
+
)
|
|
179
|
+
meta = {
|
|
180
|
+
"title": subject,
|
|
181
|
+
"source": f"mail:{message_id}",
|
|
182
|
+
"kind": "email",
|
|
183
|
+
"from": str(msg["From"] or ""),
|
|
184
|
+
"date": sent,
|
|
185
|
+
}
|
|
186
|
+
content = Page(Path("clip.md"), meta, body).render() if body else ""
|
|
187
|
+
return ClipNote(
|
|
188
|
+
filename=f"{sent} {safe_title(subject, 60)}.md",
|
|
189
|
+
content=content,
|
|
190
|
+
source=meta["source"],
|
|
191
|
+
pdfs=pdfs,
|
|
192
|
+
images=images,
|
|
193
|
+
links=_links(body),
|
|
194
|
+
)
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""The IMAP app password lives in the macOS Keychain (via keyring), never in a file."""
|
|
2
|
+
|
|
3
|
+
import threading
|
|
4
|
+
|
|
5
|
+
import keyring
|
|
6
|
+
from keyring.errors import KeyringError
|
|
7
|
+
|
|
8
|
+
SERVICE = "esbi-cli-imap"
|
|
9
|
+
OLD_SERVICE = "secondbrain-imap" # legacy: the name before esbi-cli
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class CredentialError(RuntimeError):
|
|
13
|
+
pass
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def save_password(user: str, password: str, backend=None) -> None:
|
|
17
|
+
# Google shows app passwords in groups separated by spaces; the spaces are not part of it
|
|
18
|
+
try:
|
|
19
|
+
(backend or keyring).set_password(SERVICE, user, "".join(password.split()))
|
|
20
|
+
except KeyringError as exc:
|
|
21
|
+
hint = ""
|
|
22
|
+
if "-25244" in str(exc): # the existing item was made by another program (a reinstall)
|
|
23
|
+
hint = (
|
|
24
|
+
" The old item belongs to another program; delete it and try again: "
|
|
25
|
+
f"security delete-generic-password -s {SERVICE} -a {user}"
|
|
26
|
+
)
|
|
27
|
+
raise CredentialError(f"Could not write to the Keychain: {exc}.{hint}") from exc
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def get_password(user: str, backend=None, timeout_seconds: float = 20) -> str:
|
|
31
|
+
# macOS can show "allow this program to use the item?" and block until someone clicks; an
|
|
32
|
+
# unattended run must not hang there holding the run lock, so the read has a time limit.
|
|
33
|
+
# A daemon thread, so a read still blocked at exit cannot keep the process alive.
|
|
34
|
+
outcome: dict = {}
|
|
35
|
+
|
|
36
|
+
def read() -> None:
|
|
37
|
+
try:
|
|
38
|
+
store = backend or keyring
|
|
39
|
+
outcome["password"] = store.get_password(SERVICE, user)
|
|
40
|
+
if not outcome["password"] and (old := store.get_password(OLD_SERVICE, user)):
|
|
41
|
+
store.set_password(SERVICE, user, old) # move it to the new name
|
|
42
|
+
outcome["password"] = old
|
|
43
|
+
except BaseException as exc: # handed to the caller below
|
|
44
|
+
outcome["error"] = exc
|
|
45
|
+
|
|
46
|
+
thread = threading.Thread(target=read, daemon=True)
|
|
47
|
+
thread.start()
|
|
48
|
+
thread.join(timeout_seconds)
|
|
49
|
+
if thread.is_alive():
|
|
50
|
+
raise CredentialError(
|
|
51
|
+
"The Keychain is waiting for permission: a dialog on the Mac asks whether `sb` may use "
|
|
52
|
+
"the item. Click Always Allow, or store the password again with `sb email set-password`."
|
|
53
|
+
)
|
|
54
|
+
if isinstance(outcome.get("error"), KeyringError): # no backend (Linux/CI), locked or denied
|
|
55
|
+
raise CredentialError(f"Could not read the Keychain: {outcome['error']}") from outcome[
|
|
56
|
+
"error"
|
|
57
|
+
]
|
|
58
|
+
if "error" in outcome:
|
|
59
|
+
raise outcome["error"]
|
|
60
|
+
password = outcome["password"]
|
|
61
|
+
if not password:
|
|
62
|
+
raise CredentialError(
|
|
63
|
+
f"No IMAP password in the Keychain for {user}. Store it with `sb email set-password`."
|
|
64
|
+
)
|
|
65
|
+
return password
|
esbi_cli/mail/fetch.py
ADDED
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
"""Pull unseen mail from the dedicated mailbox into the vault's inbox/ as clip notes."""
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Protocol
|
|
7
|
+
|
|
8
|
+
from esbi_cli.mail.convert import email_to_clip
|
|
9
|
+
from esbi_cli.queue import Queue, normalize_target
|
|
10
|
+
from esbi_cli.vault import Vault, free_path, safe_title
|
|
11
|
+
|
|
12
|
+
MAIL_LINK_ORIGIN = "mail-link" # queue origin of a link found in a mail: its page is email
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class MailClient(Protocol):
|
|
16
|
+
def recent(self, days: int = 14) -> list[tuple[str, bytes]]:
|
|
17
|
+
"""(uid, raw RFC 5322 message) for every message of the last `days`, seen or not."""
|
|
18
|
+
...
|
|
19
|
+
|
|
20
|
+
def mark_seen(self, uid: str) -> None: ...
|
|
21
|
+
|
|
22
|
+
def close(self) -> None: ...
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass
|
|
26
|
+
class FetchResult:
|
|
27
|
+
saved: int = 0
|
|
28
|
+
duplicates: int = 0
|
|
29
|
+
failed: int = 0
|
|
30
|
+
images: int = 0 # image attachments kept
|
|
31
|
+
links: int = 0 # links found in mail and queued
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _known_sources(vault: Vault) -> set[str]:
|
|
35
|
+
"""`mail:<id>` sources already saved: waiting in inbox/, moved to raw/inbox/, or ingested."""
|
|
36
|
+
known = set()
|
|
37
|
+
for folder in (vault.root / "inbox", vault.root / "raw" / "inbox"):
|
|
38
|
+
for path in folder.glob("*.md") if folder.is_dir() else ():
|
|
39
|
+
known.add(str(vault.read_page(path).meta.get("source")))
|
|
40
|
+
known |= {str(p.meta.get("url")) for p in vault.iter_pages(("sources",))}
|
|
41
|
+
return known
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _kept_file(vault: Vault, data: bytes, suffix: str) -> bool:
|
|
45
|
+
"""Is this very file already waiting in inbox/ or kept in raw/inbox/?"""
|
|
46
|
+
return any(
|
|
47
|
+
f.stat().st_size == len(data) and f.read_bytes() == data
|
|
48
|
+
for folder in (vault.root / "inbox", vault.root / "raw" / "inbox")
|
|
49
|
+
if folder.is_dir()
|
|
50
|
+
for f in folder.iterdir()
|
|
51
|
+
if f.suffix.lower() == suffix
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _save(vault: Vault, inbox: Path, clip) -> int:
|
|
56
|
+
"""Write the note and its attachments; returns how many images were new."""
|
|
57
|
+
if clip.content:
|
|
58
|
+
free_path(inbox, clip.filename).write_text(clip.content, encoding="utf-8")
|
|
59
|
+
stem, images = clip.filename.removesuffix(".md"), 0
|
|
60
|
+
for name, data in clip.pdfs:
|
|
61
|
+
_remember_mail_file(vault, data)
|
|
62
|
+
if not _kept_file(vault, data, ".pdf"):
|
|
63
|
+
free_path(inbox, f"{stem} - {safe_title(Path(name).stem, 40)}.pdf").write_bytes(data)
|
|
64
|
+
for name, data in clip.images: # `name` is already safe: stem and extension from the bytes
|
|
65
|
+
_remember_mail_file(vault, data)
|
|
66
|
+
if not _kept_file(vault, data, Path(name).suffix):
|
|
67
|
+
free_path(inbox, f"{stem} - {name}").write_bytes(data)
|
|
68
|
+
images += 1
|
|
69
|
+
return images
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _queue_links(vault: Vault, queue: Queue, links: list[str], max_links: int) -> int:
|
|
73
|
+
"""Queue links not already known (queued, done or a source) as email-derived items. Never
|
|
74
|
+
fetches: the nightly run reads them, with its limits and the private model."""
|
|
75
|
+
queued = 0
|
|
76
|
+
for url in links:
|
|
77
|
+
if queued == max_links:
|
|
78
|
+
break
|
|
79
|
+
if not vault.find_source("url", normalize_target(url)) and queue.add(
|
|
80
|
+
url, origin=MAIL_LINK_ORIGIN
|
|
81
|
+
):
|
|
82
|
+
queued += 1
|
|
83
|
+
return queued
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def fetch_mail(
|
|
87
|
+
client: MailClient,
|
|
88
|
+
vault: Vault,
|
|
89
|
+
queue: Queue | None = None,
|
|
90
|
+
follow_links: bool = False,
|
|
91
|
+
max_links: int = 3,
|
|
92
|
+
) -> FetchResult:
|
|
93
|
+
"""Save each recent mail as inbox/<date> <subject>.md (and its PDF and image attachments), mark
|
|
94
|
+
it seen. With `follow_links` and a queue, the links in a newly saved mail are queued.
|
|
95
|
+
|
|
96
|
+
Never deletes mail. Mail already captured is skipped by Message-ID, so the window can overlap
|
|
97
|
+
runs. A mail that cannot be converted is reported once: its id is remembered in
|
|
98
|
+
`.esbi/mail-failed.txt` and skipped from then on.
|
|
99
|
+
"""
|
|
100
|
+
result = FetchResult()
|
|
101
|
+
inbox = vault.root / "inbox"
|
|
102
|
+
inbox.mkdir(exist_ok=True)
|
|
103
|
+
known = _known_sources(vault)
|
|
104
|
+
failed_file = vault.root / ".esbi" / "mail-failed.txt"
|
|
105
|
+
failed_before = set(failed_file.read_text().split()) if failed_file.exists() else set()
|
|
106
|
+
for uid, raw in client.recent():
|
|
107
|
+
key = hashlib.sha256(raw).hexdigest()[:16] # the uid is not stable across mailboxes
|
|
108
|
+
if key in failed_before:
|
|
109
|
+
continue
|
|
110
|
+
try:
|
|
111
|
+
clip = email_to_clip(raw, vault.language)
|
|
112
|
+
except Exception: # one unreadable mail must not block the rest
|
|
113
|
+
result.failed += 1
|
|
114
|
+
failed_file.parent.mkdir(exist_ok=True)
|
|
115
|
+
with failed_file.open("a") as fh:
|
|
116
|
+
fh.write(key + "\n")
|
|
117
|
+
continue
|
|
118
|
+
if clip.source in known:
|
|
119
|
+
result.duplicates += 1
|
|
120
|
+
else:
|
|
121
|
+
try:
|
|
122
|
+
result.images += _save(vault, inbox, clip)
|
|
123
|
+
if follow_links and queue is not None:
|
|
124
|
+
result.links += _queue_links(vault, queue, clip.links, max_links)
|
|
125
|
+
except OSError: # a name the disk refuses: report it once, not on every run
|
|
126
|
+
result.failed += 1
|
|
127
|
+
failed_file.parent.mkdir(exist_ok=True)
|
|
128
|
+
with failed_file.open("a") as fh:
|
|
129
|
+
fh.write(key + "\n")
|
|
130
|
+
continue
|
|
131
|
+
known.add(clip.source)
|
|
132
|
+
result.saved += 1
|
|
133
|
+
client.mark_seen(uid)
|
|
134
|
+
return result
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def _hashes_path(vault: Vault) -> Path:
|
|
138
|
+
return vault.root / ".esbi" / "mail-pdfs.txt"
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _remember_mail_file(vault: Vault, data: bytes) -> None:
|
|
142
|
+
"""A PDF or image saved from a mail is email: its hash is kept so the privacy rules still know
|
|
143
|
+
that once it sits in inbox/ as a plain file."""
|
|
144
|
+
path, digest = _hashes_path(vault), hashlib.sha256(data).hexdigest()
|
|
145
|
+
path.parent.mkdir(exist_ok=True)
|
|
146
|
+
known = set(path.read_text().split()) if path.exists() else set()
|
|
147
|
+
if digest not in known:
|
|
148
|
+
with path.open("a") as fh:
|
|
149
|
+
fh.write(digest + "\n")
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def is_mail_file(vault: Vault, data: bytes) -> bool:
|
|
153
|
+
path = _hashes_path(vault)
|
|
154
|
+
return path.exists() and hashlib.sha256(data).hexdigest() in path.read_text().split()
|
esbi_cli/mail/imap.py
ADDED
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""IMAP access to the dedicated mailbox (a Gmail label works as a mailbox name)."""
|
|
2
|
+
|
|
3
|
+
import imaplib
|
|
4
|
+
import ssl
|
|
5
|
+
from collections.abc import Callable
|
|
6
|
+
from datetime import date, timedelta
|
|
7
|
+
|
|
8
|
+
MONTHS = (
|
|
9
|
+
"Jan",
|
|
10
|
+
"Feb",
|
|
11
|
+
"Mar",
|
|
12
|
+
"Apr",
|
|
13
|
+
"May",
|
|
14
|
+
"Jun",
|
|
15
|
+
"Jul",
|
|
16
|
+
"Aug",
|
|
17
|
+
"Sep",
|
|
18
|
+
"Oct",
|
|
19
|
+
"Nov",
|
|
20
|
+
"Dec",
|
|
21
|
+
) # IMAP wants English
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class MailError(RuntimeError):
|
|
25
|
+
pass
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _verified_ssl(host: str) -> imaplib.IMAP4:
|
|
29
|
+
"""imaplib's own default context does not check the certificate: the app password would go to
|
|
30
|
+
whoever answers. This one verifies the chain and the host name."""
|
|
31
|
+
return imaplib.IMAP4_SSL(host, ssl_context=ssl.create_default_context())
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class ImapMailClient:
|
|
35
|
+
def __init__(
|
|
36
|
+
self,
|
|
37
|
+
host: str,
|
|
38
|
+
user: str,
|
|
39
|
+
password: str,
|
|
40
|
+
mailbox: str,
|
|
41
|
+
factory: Callable[[str], imaplib.IMAP4] = _verified_ssl,
|
|
42
|
+
):
|
|
43
|
+
self.host, self.user, self.mailbox = host, user, mailbox
|
|
44
|
+
self._password = password
|
|
45
|
+
self._factory = factory
|
|
46
|
+
self._imap: imaplib.IMAP4 | None = None
|
|
47
|
+
|
|
48
|
+
def _connect(self) -> imaplib.IMAP4:
|
|
49
|
+
if self._imap is None:
|
|
50
|
+
try:
|
|
51
|
+
imap = self._factory(self.host)
|
|
52
|
+
imap.login(self.user, self._password)
|
|
53
|
+
quoted = '"' + self.mailbox.replace("\\", "\\\\").replace('"', '\\"') + '"'
|
|
54
|
+
status, _ = imap.select(quoted, readonly=False)
|
|
55
|
+
except (imaplib.IMAP4.error, OSError) as exc:
|
|
56
|
+
raise MailError(
|
|
57
|
+
f"IMAP login/select failed for {self.user}@{self.host}: {exc}"
|
|
58
|
+
) from None
|
|
59
|
+
if status != "OK":
|
|
60
|
+
raise MailError(f"Mailbox {self.mailbox!r} not found on {self.host}")
|
|
61
|
+
self._imap = imap
|
|
62
|
+
return self._imap
|
|
63
|
+
|
|
64
|
+
def recent(self, days: int = 14, today: date | None = None) -> list[tuple[str, bytes]]:
|
|
65
|
+
"""Every message of the last `days`, opened in Gmail or not: the worker dedupes by
|
|
66
|
+
Message-ID, so the window can overlap runs without saving a mail twice."""
|
|
67
|
+
imap = self._connect()
|
|
68
|
+
since = (today or date.today()) - timedelta(days=days)
|
|
69
|
+
try:
|
|
70
|
+
_, data = imap.uid(
|
|
71
|
+
"SEARCH", None, "SINCE", f"{since.day:02d}-{MONTHS[since.month - 1]}-{since.year}"
|
|
72
|
+
)
|
|
73
|
+
mails = []
|
|
74
|
+
for uid in data[0].decode().split():
|
|
75
|
+
# BODY.PEEK[] reads the message without setting \Seen: only the worker does that,
|
|
76
|
+
# after the note is safely written
|
|
77
|
+
_, parts = imap.uid("FETCH", uid, "(BODY.PEEK[])")
|
|
78
|
+
mails.append((uid, parts[0][1]))
|
|
79
|
+
return mails
|
|
80
|
+
except (imaplib.IMAP4.error, OSError) as exc:
|
|
81
|
+
raise MailError(f"Reading mail failed: {exc}") from None
|
|
82
|
+
|
|
83
|
+
def mark_seen(self, uid: str) -> None:
|
|
84
|
+
self._connect().uid("STORE", uid, "+FLAGS", "(\\Seen)")
|
|
85
|
+
|
|
86
|
+
def close(self) -> None:
|
|
87
|
+
if self._imap is not None:
|
|
88
|
+
try:
|
|
89
|
+
self._imap.logout()
|
|
90
|
+
except (imaplib.IMAP4.error, OSError):
|
|
91
|
+
pass
|
|
92
|
+
self._imap = None
|