esbi-cli 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. esbi_cli/__init__.py +8 -0
  2. esbi_cli/ask/__init__.py +0 -0
  3. esbi_cli/ask/answer.py +256 -0
  4. esbi_cli/bench/__init__.py +0 -0
  5. esbi_cli/bench/cases.py +57 -0
  6. esbi_cli/bench/metrics.py +23 -0
  7. esbi_cli/bench/report.py +117 -0
  8. esbi_cli/bench/runner.py +114 -0
  9. esbi_cli/capture/__init__.py +0 -0
  10. esbi_cli/capture/inbox.py +63 -0
  11. esbi_cli/capture/legacy.py +49 -0
  12. esbi_cli/cli.py +1387 -0
  13. esbi_cli/config.py +344 -0
  14. esbi_cli/doctor.py +391 -0
  15. esbi_cli/evaluate.py +91 -0
  16. esbi_cli/export.py +137 -0
  17. esbi_cli/extract/__init__.py +107 -0
  18. esbi_cli/extract/clip.py +30 -0
  19. esbi_cli/extract/html.py +60 -0
  20. esbi_cli/extract/image.py +58 -0
  21. esbi_cli/extract/pdf.py +109 -0
  22. esbi_cli/gitops.py +101 -0
  23. esbi_cli/index.py +303 -0
  24. esbi_cli/ingest/__init__.py +0 -0
  25. esbi_cli/ingest/apply.py +480 -0
  26. esbi_cli/ingest/chunks.py +49 -0
  27. esbi_cli/ingest/connect.py +87 -0
  28. esbi_cli/ingest/digest.py +91 -0
  29. esbi_cli/ingest/pipeline.py +176 -0
  30. esbi_cli/ingest/plan.py +231 -0
  31. esbi_cli/ingest/read.py +105 -0
  32. esbi_cli/ingest/retrieve.py +59 -0
  33. esbi_cli/init.py +176 -0
  34. esbi_cli/interrupts.py +90 -0
  35. esbi_cli/lang.py +341 -0
  36. esbi_cli/links.py +10 -0
  37. esbi_cli/lint/__init__.py +0 -0
  38. esbi_cli/lint/checks.py +178 -0
  39. esbi_cli/lint/report.py +60 -0
  40. esbi_cli/llm/__init__.py +0 -0
  41. esbi_cli/llm/adapter.py +393 -0
  42. esbi_cli/llm/schemas.py +146 -0
  43. esbi_cli/mail/__init__.py +0 -0
  44. esbi_cli/mail/convert.py +194 -0
  45. esbi_cli/mail/credentials.py +65 -0
  46. esbi_cli/mail/fetch.py +154 -0
  47. esbi_cli/mail/imap.py +92 -0
  48. esbi_cli/netguard.py +127 -0
  49. esbi_cli/privacy.py +81 -0
  50. esbi_cli/queue.py +179 -0
  51. esbi_cli/reingest.py +165 -0
  52. esbi_cli/report/__init__.py +0 -0
  53. esbi_cli/report/daily_index.py +235 -0
  54. esbi_cli/report/index_md.py +21 -0
  55. esbi_cli/report/readstate.py +26 -0
  56. esbi_cli/run.py +100 -0
  57. esbi_cli/runlock.py +31 -0
  58. esbi_cli/runlog.py +80 -0
  59. esbi_cli/schedule.py +106 -0
  60. esbi_cli/templates/SCHEMA.md +52 -0
  61. esbi_cli/templates/clipper-template.json +17 -0
  62. esbi_cli/templates/clipper-youtube-template.json +18 -0
  63. esbi_cli/templates/config.example.toml +108 -0
  64. esbi_cli/update.py +247 -0
  65. esbi_cli/vault.py +188 -0
  66. esbi_cli/wizards/clipper.sh +271 -0
  67. esbi_cli/wizards/email.sh +265 -0
  68. esbi_cli-0.2.1.dist-info/METADATA +167 -0
  69. esbi_cli-0.2.1.dist-info/RECORD +72 -0
  70. esbi_cli-0.2.1.dist-info/WHEEL +4 -0
  71. esbi_cli-0.2.1.dist-info/entry_points.txt +3 -0
  72. esbi_cli-0.2.1.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,146 @@
1
+ """The edit plan: the only thing the LLM produces. The worker validates and applies it."""
2
+
3
+ from typing import Annotated
4
+
5
+ from annotated_types import MaxLen
6
+ from pydantic import BaseModel, BeforeValidator, Field, field_validator
7
+
8
+
9
+ def _clamp(max_items: int):
10
+ def clamp(value):
11
+ return value[:max_items] if isinstance(value, list) else value
12
+
13
+ return BeforeValidator(clamp)
14
+
15
+
16
+ class ConceptEdit(BaseModel):
17
+ """A concept or entity this source touches."""
18
+
19
+ title: str = Field(
20
+ min_length=2, max_length=60, description="Canonical name, short (at most 6 words)"
21
+ )
22
+ aliases: list[str] = Field(default_factory=list, description="Other names or acronyms")
23
+ description: str = Field(
24
+ min_length=10,
25
+ description="1-3 sentences on what this source adds to this concept or entity",
26
+ )
27
+
28
+
29
+ class Term(BaseModel):
30
+ """A technical term used in the source, with what it means."""
31
+
32
+ term: str = Field(
33
+ min_length=2, max_length=60, description="Exactly as it appears in the source"
34
+ )
35
+ definition: str = Field(min_length=10, description="Definition in one sentence")
36
+
37
+
38
+ class Relation(BaseModel):
39
+ """A link between two ideas, drawn later as an edge of the concept map."""
40
+
41
+ a: str = Field(min_length=2, max_length=50, description="Name of a concept or term, 1-4 words")
42
+ relation: str = Field(
43
+ min_length=2, max_length=40, description="A short label for how the two relate"
44
+ )
45
+ b: str = Field(min_length=2, max_length=50, description="Name of a concept or term, 1-4 words")
46
+
47
+
48
+ class ChunkNotes(BaseModel):
49
+ """What one reads out of one chunk of a long source."""
50
+
51
+ points: Annotated[list[str], _clamp(8), MaxLen(8)] = Field(
52
+ min_length=1,
53
+ description="3-8 concrete statements: facts, methods, results, arguments",
54
+ )
55
+ terms: Annotated[list[Term], _clamp(6), MaxLen(6)] = Field(default_factory=list)
56
+ quotes: Annotated[list[str], _clamp(3), MaxLen(3)] = Field(
57
+ default_factory=list, description="Sentences copied EXACTLY from the chunk"
58
+ )
59
+ relations: Annotated[list[Relation], _clamp(6), MaxLen(6)] = Field(default_factory=list)
60
+
61
+
62
+ class Insight(BaseModel):
63
+ idea: str = Field(min_length=10, description="The idea, in one or two sentences")
64
+ why: str = Field(min_length=10, description="What follows from the idea, in one sentence")
65
+
66
+
67
+ class Digest(BaseModel):
68
+ """The long-form part of a note, written from the chunk notes."""
69
+
70
+ # a list, not one string: a string has no bound a model's grammar can enforce, and a small model
71
+ # repeated paragraphs in it until the token cap in 5 of 20 calls (0 of 10 as a list of at most 5)
72
+ paragraphs: Annotated[list[str], _clamp(4), MaxLen(4)] = Field(
73
+ min_length=3,
74
+ description="Detailed summary: the problem, the approach, the findings, the implications",
75
+ )
76
+ insights: Annotated[list[Insight], _clamp(6), MaxLen(6)] = Field(min_length=1)
77
+ open_questions: Annotated[list[str], _clamp(5), MaxLen(5)] = Field(default_factory=list)
78
+
79
+ @field_validator("paragraphs")
80
+ @classmethod
81
+ def _enough_to_read(cls, value: list[str]) -> list[str]:
82
+ if sum(len(p.strip()) for p in value) < 200:
83
+ raise ValueError("the detailed summary is too short: write real paragraphs")
84
+ return value
85
+
86
+ @property
87
+ def abstract(self) -> str:
88
+ return "\n\n".join(p.strip() for p in self.paragraphs)
89
+
90
+
91
+ class Connection(BaseModel):
92
+ """How a new source relates to a page already in the wiki."""
93
+
94
+ page: str = Field(description="EXACT title of an existing page")
95
+ relation: str = Field(
96
+ min_length=2, max_length=40, description="A short label for how the two relate"
97
+ )
98
+ why: str = Field(min_length=10, description="One sentence: why they are related")
99
+
100
+
101
+ class ConnectionPlan(BaseModel):
102
+ connections: Annotated[list[Connection], _clamp(8), MaxLen(8)] = Field(default_factory=list)
103
+
104
+
105
+ class Contradiction(BaseModel):
106
+ page: str = Field(description="EXACT title of the existing page that is contradicted")
107
+ note: str = Field(min_length=5, description="What is contradicted, in one sentence")
108
+
109
+
110
+ class EditPlan(BaseModel):
111
+ title: str = Field(min_length=3, description="Title of the source")
112
+ one_liner: str = Field(min_length=10, description="One-sentence summary for the index")
113
+ summary: str = Field(
114
+ min_length=30, description="Executive summary: 2-3 sentences, what it is and why it matters"
115
+ )
116
+ abstract: str = Field(
117
+ default="",
118
+ description="Detailed summary, 3-5 paragraphs: problem, approach, findings, implications",
119
+ )
120
+ insights: Annotated[list[Insight], _clamp(8), MaxLen(8)] = Field(default_factory=list)
121
+ terms: Annotated[list[Term], _clamp(10), MaxLen(10)] = Field(default_factory=list)
122
+ quotes: Annotated[list[str], _clamp(6), MaxLen(6)] = Field(
123
+ default_factory=list, description="Sentences copied EXACTLY from the source"
124
+ )
125
+ relations: Annotated[list[Relation], _clamp(10), MaxLen(10)] = Field(default_factory=list)
126
+ open_questions: Annotated[list[str], _clamp(5), MaxLen(5)] = Field(default_factory=list)
127
+ key_points: Annotated[list[str], _clamp(8), MaxLen(8)] = Field(description="3-8 key points")
128
+ tags: Annotated[list[str], _clamp(6), MaxLen(6)] = Field(default_factory=list)
129
+ concepts: Annotated[list[ConceptEdit], _clamp(6), MaxLen(6)] = Field(
130
+ min_length=1, description="2-6 central concepts or techniques (never empty)"
131
+ )
132
+ entities: Annotated[list[ConceptEdit], _clamp(5), MaxLen(5)] = Field(default_factory=list)
133
+ related_pages: Annotated[list[str], _clamp(6), MaxLen(6)] = Field(
134
+ default_factory=list,
135
+ description="EXACT titles of related existing pages (only from the given list)",
136
+ )
137
+ contradictions: Annotated[list[Contradiction], _clamp(4), MaxLen(4)] = Field(
138
+ default_factory=list
139
+ )
140
+
141
+ @field_validator("one_liner")
142
+ @classmethod
143
+ def _one_liner_is_a_sentence(cls, value: str) -> str:
144
+ if value.strip().lower().startswith(("http://", "https://")) or " " not in value.strip():
145
+ raise ValueError("must be a descriptive sentence, not a URL")
146
+ return value
File without changes
@@ -0,0 +1,194 @@
1
+ """Turn a raw email (RFC 5322 bytes) into a Web Clipper-style note for the vault's inbox."""
2
+
3
+ import email
4
+ import hashlib
5
+ import io
6
+ import re
7
+ from dataclasses import dataclass, field
8
+ from datetime import date
9
+ from email import policy
10
+ from email.message import EmailMessage
11
+ from email.utils import parsedate_to_datetime
12
+ from pathlib import Path
13
+ from urllib.parse import urlparse
14
+
15
+ import lxml.html
16
+ import trafilatura
17
+ from PIL import Image
18
+
19
+ from esbi_cli import lang
20
+ from esbi_cli.vault import Page, safe_title
21
+
22
+
23
+ @dataclass
24
+ class ClipNote:
25
+ filename: str
26
+ content: str # empty when the mail is only attachments
27
+ source: str # stable id of the mail (`mail:<message-id>`), used to spot duplicates
28
+ pdfs: list[tuple[str, bytes]] = field(default_factory=list) # (file name, bytes) attached
29
+ images: list[tuple[str, bytes]] = field(default_factory=list) # (safe file name, bytes) kept
30
+ links: list[str] = field(default_factory=list) # http(s) links worth following, in order
31
+
32
+
33
+ # Newsletters pad the preview text with these (zero-width and joiner characters, soft hyphens)
34
+ INVISIBLE = re.compile("[\u034f\u200b-\u200f\u2060\u00ad\ufeff]")
35
+
36
+
37
+ def _clean(text: str) -> str:
38
+ """Drop the invisible padding, then the blank lines it leaves behind."""
39
+ text = INVISIBLE.sub("", text)
40
+ text = "\n".join(line.rstrip() for line in text.splitlines())
41
+ return re.sub(r"\n{3,}", "\n\n", text).strip()
42
+
43
+
44
+ def _html_to_text(html: str) -> str:
45
+ """Article extraction when the email looks like one, else the page's visible text."""
46
+ extracted = trafilatura.extract(html, output_format="markdown", favor_recall=True)
47
+ if extracted and len(extracted.strip()) > 40:
48
+ return extracted.strip()
49
+ tree = lxml.html.fromstring(html)
50
+ for junk in tree.xpath("//script | //style | //head"):
51
+ junk.drop_tree()
52
+ return re.sub(r"\n{3,}", "\n\n", tree.text_content()).strip()
53
+
54
+
55
+ MIN_PLAIN_CHARS = 100 # shorter plain parts are usually a "view in browser" stub
56
+
57
+
58
+ def _body(msg: EmailMessage) -> str:
59
+ plain = msg.get_body(preferencelist=("plain",))
60
+ html = msg.get_body(preferencelist=("html",))
61
+ plain_text = plain.get_content().strip() if plain else ""
62
+ if plain_text and (len(plain_text) >= MIN_PLAIN_CHARS or html is None):
63
+ return plain_text
64
+ return _html_to_text(html.get_content()) if html else plain_text
65
+
66
+
67
+ # Image attachments: what the bytes are decides, never the declared type or the file name.
68
+ IMAGE_FORMATS = { # declared content type -> (format Pillow must find in the bytes, extension)
69
+ "image/png": ("PNG", ".png"),
70
+ "image/jpeg": ("JPEG", ".jpg"),
71
+ "image/webp": ("WEBP", ".webp"),
72
+ "image/tiff": ("TIFF", ".tiff"),
73
+ }
74
+ MIN_IMAGE_SIDE_PX = 200 # signature logos, icons and tracking pixels are smaller
75
+ MIN_IMAGE_BYTES = 5_000 # ...or flat: a screenshot or a photo is never this light
76
+ MAX_IMAGE_BYTES = 5_000_000
77
+ MAX_MAIL_IMAGE_BYTES = 15_000_000
78
+ MAX_IMAGE_PIXELS = 50_000_000 # the decoder allocates width x height: refuse absurd sizes
79
+
80
+
81
+ def _image_name(data: bytes, declared: str, filename: str | None) -> str | None:
82
+ """`<safe stem><extension>` for an image worth keeping, else None."""
83
+ if declared not in IMAGE_FORMATS or not MIN_IMAGE_BYTES <= len(data) <= MAX_IMAGE_BYTES:
84
+ return None
85
+ try:
86
+ with Image.open(io.BytesIO(data)) as img:
87
+ fmt, (width, height) = img.format, img.size
88
+ except Exception: # Pillow raises several unrelated error types for bytes that are no image
89
+ return None
90
+ expected, suffix = IMAGE_FORMATS[declared]
91
+ if (
92
+ fmt != expected
93
+ or min(width, height) < MIN_IMAGE_SIDE_PX
94
+ or width * height > MAX_IMAGE_PIXELS
95
+ ):
96
+ return None
97
+ stem = Path((filename or "").replace("\\", "/")).stem # never a path: only the last part
98
+ return f"{safe_title(stem, 40) or 'image'}{suffix}"
99
+
100
+
101
+ def _images(msg: EmailMessage) -> list[tuple[str, bytes]]:
102
+ kept, total = [], 0
103
+ for part in msg.walk():
104
+ if part.is_multipart() or not part.get_content_type().startswith("image/"):
105
+ continue
106
+ data = part.get_content()
107
+ name = _image_name(data, part.get_content_type(), part.get_filename())
108
+ if name and total + len(data) <= MAX_MAIL_IMAGE_BYTES:
109
+ kept.append((name, data))
110
+ total += len(data)
111
+ return kept
112
+
113
+
114
+ _URL = re.compile(r"https?://[^\s<>\"')\]]+")
115
+ _NOT_AN_ARTICLE = re.compile(
116
+ r"unsubscri|opt-?out|preferences|view-?in-?browser|view_in_browser|web-?version|"
117
+ r"manage[-_]?subscription|confirm|verif|reset|log-?in|sign-?in|magic|activat|password|token=|"
118
+ r"\.(?:gif|png|jpe?g|svg|webp|css|js|ico)(?:\?|$)",
119
+ re.IGNORECASE,
120
+ )
121
+ _REDIRECT_HOSTS = (
122
+ "click",
123
+ "clicks",
124
+ "track",
125
+ "tracking",
126
+ "links",
127
+ "trk",
128
+ "email",
129
+ "em",
130
+ ) # click.news.test
131
+ MAX_LINK_CHARS = 500
132
+
133
+
134
+ def _links(text: str) -> list[str]:
135
+ """http(s) links in the mail's text, in order and without repeats, minus tracking redirects,
136
+ unsubscribe/preferences/view-in-browser housekeeping and pictures. ponytail: only links that
137
+ show in the text; a link hidden behind anchor text in an HTML-only mail is not found."""
138
+ found: list[str] = []
139
+ for url in (u.rstrip(".,;:!?") for u in _URL.findall(text)):
140
+ host = urlparse(url).hostname or ""
141
+ if (
142
+ len(url) <= MAX_LINK_CHARS
143
+ and host
144
+ and not _NOT_AN_ARTICLE.search(url)
145
+ and not (host.count(".") >= 2 and host.partition(".")[0] in _REDIRECT_HOSTS)
146
+ and host != "list-manage.com"
147
+ and not host.endswith(".list-manage.com")
148
+ and url not in found
149
+ ):
150
+ found.append(url)
151
+ return found
152
+
153
+
154
+ def _sent_date(msg: EmailMessage) -> date:
155
+ try:
156
+ return parsedate_to_datetime(str(msg["Date"])).date()
157
+ except (TypeError, ValueError):
158
+ return date.today()
159
+
160
+
161
+ def email_to_clip(raw: bytes, language: str = lang.DEFAULT) -> ClipNote:
162
+ msg = email.message_from_bytes(raw, policy=policy.default)
163
+ subject = str(msg["Subject"] or lang.t(language, "no_subject")).strip()
164
+ body = _clean(_body(msg))
165
+ pdfs = [
166
+ (part.get_filename() or "attachment.pdf", part.get_content())
167
+ for part in msg.iter_attachments()
168
+ if part.get_content_type() == "application/pdf"
169
+ ]
170
+ images = _images(msg)
171
+ if not body and not pdfs and not images:
172
+ raise ValueError("the email has no readable text")
173
+ sent = _sent_date(msg).isoformat()
174
+ # an id is used as a source name and written into notes: letters, digits and . @ + = _ - only
175
+ message_id = (
176
+ re.sub(r"[^\w.@+=-]", "", str(msg["Message-ID"] or ""))[:150]
177
+ or hashlib.sha256(raw).hexdigest()[:16]
178
+ )
179
+ meta = {
180
+ "title": subject,
181
+ "source": f"mail:{message_id}",
182
+ "kind": "email",
183
+ "from": str(msg["From"] or ""),
184
+ "date": sent,
185
+ }
186
+ content = Page(Path("clip.md"), meta, body).render() if body else ""
187
+ return ClipNote(
188
+ filename=f"{sent} {safe_title(subject, 60)}.md",
189
+ content=content,
190
+ source=meta["source"],
191
+ pdfs=pdfs,
192
+ images=images,
193
+ links=_links(body),
194
+ )
@@ -0,0 +1,65 @@
1
+ """The IMAP app password lives in the macOS Keychain (via keyring), never in a file."""
2
+
3
+ import threading
4
+
5
+ import keyring
6
+ from keyring.errors import KeyringError
7
+
8
+ SERVICE = "esbi-cli-imap"
9
+ OLD_SERVICE = "secondbrain-imap" # legacy: the name before esbi-cli
10
+
11
+
12
+ class CredentialError(RuntimeError):
13
+ pass
14
+
15
+
16
+ def save_password(user: str, password: str, backend=None) -> None:
17
+ # Google shows app passwords in groups separated by spaces; the spaces are not part of it
18
+ try:
19
+ (backend or keyring).set_password(SERVICE, user, "".join(password.split()))
20
+ except KeyringError as exc:
21
+ hint = ""
22
+ if "-25244" in str(exc): # the existing item was made by another program (a reinstall)
23
+ hint = (
24
+ " The old item belongs to another program; delete it and try again: "
25
+ f"security delete-generic-password -s {SERVICE} -a {user}"
26
+ )
27
+ raise CredentialError(f"Could not write to the Keychain: {exc}.{hint}") from exc
28
+
29
+
30
+ def get_password(user: str, backend=None, timeout_seconds: float = 20) -> str:
31
+ # macOS can show "allow this program to use the item?" and block until someone clicks; an
32
+ # unattended run must not hang there holding the run lock, so the read has a time limit.
33
+ # A daemon thread, so a read still blocked at exit cannot keep the process alive.
34
+ outcome: dict = {}
35
+
36
+ def read() -> None:
37
+ try:
38
+ store = backend or keyring
39
+ outcome["password"] = store.get_password(SERVICE, user)
40
+ if not outcome["password"] and (old := store.get_password(OLD_SERVICE, user)):
41
+ store.set_password(SERVICE, user, old) # move it to the new name
42
+ outcome["password"] = old
43
+ except BaseException as exc: # handed to the caller below
44
+ outcome["error"] = exc
45
+
46
+ thread = threading.Thread(target=read, daemon=True)
47
+ thread.start()
48
+ thread.join(timeout_seconds)
49
+ if thread.is_alive():
50
+ raise CredentialError(
51
+ "The Keychain is waiting for permission: a dialog on the Mac asks whether `sb` may use "
52
+ "the item. Click Always Allow, or store the password again with `sb email set-password`."
53
+ )
54
+ if isinstance(outcome.get("error"), KeyringError): # no backend (Linux/CI), locked or denied
55
+ raise CredentialError(f"Could not read the Keychain: {outcome['error']}") from outcome[
56
+ "error"
57
+ ]
58
+ if "error" in outcome:
59
+ raise outcome["error"]
60
+ password = outcome["password"]
61
+ if not password:
62
+ raise CredentialError(
63
+ f"No IMAP password in the Keychain for {user}. Store it with `sb email set-password`."
64
+ )
65
+ return password
esbi_cli/mail/fetch.py ADDED
@@ -0,0 +1,154 @@
1
+ """Pull unseen mail from the dedicated mailbox into the vault's inbox/ as clip notes."""
2
+
3
+ import hashlib
4
+ from dataclasses import dataclass
5
+ from pathlib import Path
6
+ from typing import Protocol
7
+
8
+ from esbi_cli.mail.convert import email_to_clip
9
+ from esbi_cli.queue import Queue, normalize_target
10
+ from esbi_cli.vault import Vault, free_path, safe_title
11
+
12
+ MAIL_LINK_ORIGIN = "mail-link" # queue origin of a link found in a mail: its page is email
13
+
14
+
15
+ class MailClient(Protocol):
16
+ def recent(self, days: int = 14) -> list[tuple[str, bytes]]:
17
+ """(uid, raw RFC 5322 message) for every message of the last `days`, seen or not."""
18
+ ...
19
+
20
+ def mark_seen(self, uid: str) -> None: ...
21
+
22
+ def close(self) -> None: ...
23
+
24
+
25
+ @dataclass
26
+ class FetchResult:
27
+ saved: int = 0
28
+ duplicates: int = 0
29
+ failed: int = 0
30
+ images: int = 0 # image attachments kept
31
+ links: int = 0 # links found in mail and queued
32
+
33
+
34
+ def _known_sources(vault: Vault) -> set[str]:
35
+ """`mail:<id>` sources already saved: waiting in inbox/, moved to raw/inbox/, or ingested."""
36
+ known = set()
37
+ for folder in (vault.root / "inbox", vault.root / "raw" / "inbox"):
38
+ for path in folder.glob("*.md") if folder.is_dir() else ():
39
+ known.add(str(vault.read_page(path).meta.get("source")))
40
+ known |= {str(p.meta.get("url")) for p in vault.iter_pages(("sources",))}
41
+ return known
42
+
43
+
44
+ def _kept_file(vault: Vault, data: bytes, suffix: str) -> bool:
45
+ """Is this very file already waiting in inbox/ or kept in raw/inbox/?"""
46
+ return any(
47
+ f.stat().st_size == len(data) and f.read_bytes() == data
48
+ for folder in (vault.root / "inbox", vault.root / "raw" / "inbox")
49
+ if folder.is_dir()
50
+ for f in folder.iterdir()
51
+ if f.suffix.lower() == suffix
52
+ )
53
+
54
+
55
+ def _save(vault: Vault, inbox: Path, clip) -> int:
56
+ """Write the note and its attachments; returns how many images were new."""
57
+ if clip.content:
58
+ free_path(inbox, clip.filename).write_text(clip.content, encoding="utf-8")
59
+ stem, images = clip.filename.removesuffix(".md"), 0
60
+ for name, data in clip.pdfs:
61
+ _remember_mail_file(vault, data)
62
+ if not _kept_file(vault, data, ".pdf"):
63
+ free_path(inbox, f"{stem} - {safe_title(Path(name).stem, 40)}.pdf").write_bytes(data)
64
+ for name, data in clip.images: # `name` is already safe: stem and extension from the bytes
65
+ _remember_mail_file(vault, data)
66
+ if not _kept_file(vault, data, Path(name).suffix):
67
+ free_path(inbox, f"{stem} - {name}").write_bytes(data)
68
+ images += 1
69
+ return images
70
+
71
+
72
+ def _queue_links(vault: Vault, queue: Queue, links: list[str], max_links: int) -> int:
73
+ """Queue links not already known (queued, done or a source) as email-derived items. Never
74
+ fetches: the nightly run reads them, with its limits and the private model."""
75
+ queued = 0
76
+ for url in links:
77
+ if queued == max_links:
78
+ break
79
+ if not vault.find_source("url", normalize_target(url)) and queue.add(
80
+ url, origin=MAIL_LINK_ORIGIN
81
+ ):
82
+ queued += 1
83
+ return queued
84
+
85
+
86
+ def fetch_mail(
87
+ client: MailClient,
88
+ vault: Vault,
89
+ queue: Queue | None = None,
90
+ follow_links: bool = False,
91
+ max_links: int = 3,
92
+ ) -> FetchResult:
93
+ """Save each recent mail as inbox/<date> <subject>.md (and its PDF and image attachments), mark
94
+ it seen. With `follow_links` and a queue, the links in a newly saved mail are queued.
95
+
96
+ Never deletes mail. Mail already captured is skipped by Message-ID, so the window can overlap
97
+ runs. A mail that cannot be converted is reported once: its id is remembered in
98
+ `.esbi/mail-failed.txt` and skipped from then on.
99
+ """
100
+ result = FetchResult()
101
+ inbox = vault.root / "inbox"
102
+ inbox.mkdir(exist_ok=True)
103
+ known = _known_sources(vault)
104
+ failed_file = vault.root / ".esbi" / "mail-failed.txt"
105
+ failed_before = set(failed_file.read_text().split()) if failed_file.exists() else set()
106
+ for uid, raw in client.recent():
107
+ key = hashlib.sha256(raw).hexdigest()[:16] # the uid is not stable across mailboxes
108
+ if key in failed_before:
109
+ continue
110
+ try:
111
+ clip = email_to_clip(raw, vault.language)
112
+ except Exception: # one unreadable mail must not block the rest
113
+ result.failed += 1
114
+ failed_file.parent.mkdir(exist_ok=True)
115
+ with failed_file.open("a") as fh:
116
+ fh.write(key + "\n")
117
+ continue
118
+ if clip.source in known:
119
+ result.duplicates += 1
120
+ else:
121
+ try:
122
+ result.images += _save(vault, inbox, clip)
123
+ if follow_links and queue is not None:
124
+ result.links += _queue_links(vault, queue, clip.links, max_links)
125
+ except OSError: # a name the disk refuses: report it once, not on every run
126
+ result.failed += 1
127
+ failed_file.parent.mkdir(exist_ok=True)
128
+ with failed_file.open("a") as fh:
129
+ fh.write(key + "\n")
130
+ continue
131
+ known.add(clip.source)
132
+ result.saved += 1
133
+ client.mark_seen(uid)
134
+ return result
135
+
136
+
137
+ def _hashes_path(vault: Vault) -> Path:
138
+ return vault.root / ".esbi" / "mail-pdfs.txt"
139
+
140
+
141
+ def _remember_mail_file(vault: Vault, data: bytes) -> None:
142
+ """A PDF or image saved from a mail is email: its hash is kept so the privacy rules still know
143
+ that once it sits in inbox/ as a plain file."""
144
+ path, digest = _hashes_path(vault), hashlib.sha256(data).hexdigest()
145
+ path.parent.mkdir(exist_ok=True)
146
+ known = set(path.read_text().split()) if path.exists() else set()
147
+ if digest not in known:
148
+ with path.open("a") as fh:
149
+ fh.write(digest + "\n")
150
+
151
+
152
+ def is_mail_file(vault: Vault, data: bytes) -> bool:
153
+ path = _hashes_path(vault)
154
+ return path.exists() and hashlib.sha256(data).hexdigest() in path.read_text().split()
esbi_cli/mail/imap.py ADDED
@@ -0,0 +1,92 @@
1
+ """IMAP access to the dedicated mailbox (a Gmail label works as a mailbox name)."""
2
+
3
+ import imaplib
4
+ import ssl
5
+ from collections.abc import Callable
6
+ from datetime import date, timedelta
7
+
8
+ MONTHS = (
9
+ "Jan",
10
+ "Feb",
11
+ "Mar",
12
+ "Apr",
13
+ "May",
14
+ "Jun",
15
+ "Jul",
16
+ "Aug",
17
+ "Sep",
18
+ "Oct",
19
+ "Nov",
20
+ "Dec",
21
+ ) # IMAP wants English
22
+
23
+
24
+ class MailError(RuntimeError):
25
+ pass
26
+
27
+
28
+ def _verified_ssl(host: str) -> imaplib.IMAP4:
29
+ """imaplib's own default context does not check the certificate: the app password would go to
30
+ whoever answers. This one verifies the chain and the host name."""
31
+ return imaplib.IMAP4_SSL(host, ssl_context=ssl.create_default_context())
32
+
33
+
34
+ class ImapMailClient:
35
+ def __init__(
36
+ self,
37
+ host: str,
38
+ user: str,
39
+ password: str,
40
+ mailbox: str,
41
+ factory: Callable[[str], imaplib.IMAP4] = _verified_ssl,
42
+ ):
43
+ self.host, self.user, self.mailbox = host, user, mailbox
44
+ self._password = password
45
+ self._factory = factory
46
+ self._imap: imaplib.IMAP4 | None = None
47
+
48
+ def _connect(self) -> imaplib.IMAP4:
49
+ if self._imap is None:
50
+ try:
51
+ imap = self._factory(self.host)
52
+ imap.login(self.user, self._password)
53
+ quoted = '"' + self.mailbox.replace("\\", "\\\\").replace('"', '\\"') + '"'
54
+ status, _ = imap.select(quoted, readonly=False)
55
+ except (imaplib.IMAP4.error, OSError) as exc:
56
+ raise MailError(
57
+ f"IMAP login/select failed for {self.user}@{self.host}: {exc}"
58
+ ) from None
59
+ if status != "OK":
60
+ raise MailError(f"Mailbox {self.mailbox!r} not found on {self.host}")
61
+ self._imap = imap
62
+ return self._imap
63
+
64
+ def recent(self, days: int = 14, today: date | None = None) -> list[tuple[str, bytes]]:
65
+ """Every message of the last `days`, opened in Gmail or not: the worker dedupes by
66
+ Message-ID, so the window can overlap runs without saving a mail twice."""
67
+ imap = self._connect()
68
+ since = (today or date.today()) - timedelta(days=days)
69
+ try:
70
+ _, data = imap.uid(
71
+ "SEARCH", None, "SINCE", f"{since.day:02d}-{MONTHS[since.month - 1]}-{since.year}"
72
+ )
73
+ mails = []
74
+ for uid in data[0].decode().split():
75
+ # BODY.PEEK[] reads the message without setting \Seen: only the worker does that,
76
+ # after the note is safely written
77
+ _, parts = imap.uid("FETCH", uid, "(BODY.PEEK[])")
78
+ mails.append((uid, parts[0][1]))
79
+ return mails
80
+ except (imaplib.IMAP4.error, OSError) as exc:
81
+ raise MailError(f"Reading mail failed: {exc}") from None
82
+
83
+ def mark_seen(self, uid: str) -> None:
84
+ self._connect().uid("STORE", uid, "+FLAGS", "(\\Seen)")
85
+
86
+ def close(self) -> None:
87
+ if self._imap is not None:
88
+ try:
89
+ self._imap.logout()
90
+ except (imaplib.IMAP4.error, OSError):
91
+ pass
92
+ self._imap = None