esbi-cli 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. esbi_cli/__init__.py +8 -0
  2. esbi_cli/ask/__init__.py +0 -0
  3. esbi_cli/ask/answer.py +256 -0
  4. esbi_cli/bench/__init__.py +0 -0
  5. esbi_cli/bench/cases.py +57 -0
  6. esbi_cli/bench/metrics.py +23 -0
  7. esbi_cli/bench/report.py +117 -0
  8. esbi_cli/bench/runner.py +114 -0
  9. esbi_cli/capture/__init__.py +0 -0
  10. esbi_cli/capture/inbox.py +63 -0
  11. esbi_cli/capture/legacy.py +49 -0
  12. esbi_cli/cli.py +1387 -0
  13. esbi_cli/config.py +344 -0
  14. esbi_cli/doctor.py +391 -0
  15. esbi_cli/evaluate.py +91 -0
  16. esbi_cli/export.py +137 -0
  17. esbi_cli/extract/__init__.py +107 -0
  18. esbi_cli/extract/clip.py +30 -0
  19. esbi_cli/extract/html.py +60 -0
  20. esbi_cli/extract/image.py +58 -0
  21. esbi_cli/extract/pdf.py +109 -0
  22. esbi_cli/gitops.py +101 -0
  23. esbi_cli/index.py +303 -0
  24. esbi_cli/ingest/__init__.py +0 -0
  25. esbi_cli/ingest/apply.py +480 -0
  26. esbi_cli/ingest/chunks.py +49 -0
  27. esbi_cli/ingest/connect.py +87 -0
  28. esbi_cli/ingest/digest.py +91 -0
  29. esbi_cli/ingest/pipeline.py +176 -0
  30. esbi_cli/ingest/plan.py +231 -0
  31. esbi_cli/ingest/read.py +105 -0
  32. esbi_cli/ingest/retrieve.py +59 -0
  33. esbi_cli/init.py +176 -0
  34. esbi_cli/interrupts.py +90 -0
  35. esbi_cli/lang.py +341 -0
  36. esbi_cli/links.py +10 -0
  37. esbi_cli/lint/__init__.py +0 -0
  38. esbi_cli/lint/checks.py +178 -0
  39. esbi_cli/lint/report.py +60 -0
  40. esbi_cli/llm/__init__.py +0 -0
  41. esbi_cli/llm/adapter.py +393 -0
  42. esbi_cli/llm/schemas.py +146 -0
  43. esbi_cli/mail/__init__.py +0 -0
  44. esbi_cli/mail/convert.py +194 -0
  45. esbi_cli/mail/credentials.py +65 -0
  46. esbi_cli/mail/fetch.py +154 -0
  47. esbi_cli/mail/imap.py +92 -0
  48. esbi_cli/netguard.py +127 -0
  49. esbi_cli/privacy.py +81 -0
  50. esbi_cli/queue.py +179 -0
  51. esbi_cli/reingest.py +165 -0
  52. esbi_cli/report/__init__.py +0 -0
  53. esbi_cli/report/daily_index.py +235 -0
  54. esbi_cli/report/index_md.py +21 -0
  55. esbi_cli/report/readstate.py +26 -0
  56. esbi_cli/run.py +100 -0
  57. esbi_cli/runlock.py +31 -0
  58. esbi_cli/runlog.py +80 -0
  59. esbi_cli/schedule.py +106 -0
  60. esbi_cli/templates/SCHEMA.md +52 -0
  61. esbi_cli/templates/clipper-template.json +17 -0
  62. esbi_cli/templates/clipper-youtube-template.json +18 -0
  63. esbi_cli/templates/config.example.toml +108 -0
  64. esbi_cli/update.py +247 -0
  65. esbi_cli/vault.py +188 -0
  66. esbi_cli/wizards/clipper.sh +271 -0
  67. esbi_cli/wizards/email.sh +265 -0
  68. esbi_cli-0.2.1.dist-info/METADATA +167 -0
  69. esbi_cli-0.2.1.dist-info/RECORD +72 -0
  70. esbi_cli-0.2.1.dist-info/WHEEL +4 -0
  71. esbi_cli-0.2.1.dist-info/entry_points.txt +3 -0
  72. esbi_cli-0.2.1.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,107 @@
1
+ """Turn a URL or a local file into clean markdown text."""
2
+
3
+ import re
4
+ from dataclasses import dataclass, field
5
+ from pathlib import Path
6
+ from urllib.parse import urlparse
7
+
8
+ import httpx
9
+
10
+ from esbi_cli.netguard import CannotResolve, UnsafeURL, safe_get
11
+
12
+ # Identify honestly with a contact URL: Wikipedia (and others) answer 403 to generic browser-like agents
13
+ USER_AGENT = "esbi-cli/0.1 (personal knowledge tool; +https://github.com/RubenAmaury/esbi-cli)"
14
+
15
+
16
+ class ExtractError(RuntimeError):
17
+ pass
18
+
19
+
20
+ @dataclass
21
+ class Figure:
22
+ """An image taken from a PDF (as PNG), with the caption printed next to it if there is one."""
23
+
24
+ data: bytes
25
+ page: int
26
+ caption: str | None = None
27
+
28
+
29
+ @dataclass
30
+ class ExtractedDoc:
31
+ title: str
32
+ text: str
33
+ kind: str # article | paper
34
+ url: str | None = None # canonical URL, or None for local files
35
+ pdf_bytes: bytes | None = None # original PDF, saved into raw/
36
+ figures: list[Figure] = field(default_factory=list) # PDF images: copied into the vault
37
+ image_links: list[tuple[str, str]] = field(
38
+ default_factory=list
39
+ ) # web images: (alt, url), linked
40
+ image_bytes: bytes | None = None # the original image file, saved into raw/
41
+ image_suffix: str = "" # its extension, e.g. ".jpg"
42
+ warnings: list[str] = field(default_factory=list) # said to the user (e.g. OCR page cap)
43
+
44
+
45
+ def is_url(target: str) -> bool:
46
+ return urlparse(target).scheme in ("http", "https")
47
+
48
+
49
+ _GITHUB_REPO = re.compile(r"^https?://(?:www\.)?github\.com/([\w.-]+)/([\w.-]+?)(?:\.git)?/?$")
50
+
51
+
52
+ def _get(url: str) -> httpx.Response:
53
+ try:
54
+ resp = safe_get(url, headers={"User-Agent": USER_AGENT}, timeout_seconds=30)
55
+ resp.raise_for_status()
56
+ except CannotResolve as exc:
57
+ raise ExtractError(f"Could not fetch {url}: {exc}") from exc
58
+ except UnsafeURL as exc:
59
+ raise ExtractError(f"Refused to fetch {url}: {exc}") from exc
60
+ except httpx.HTTPError as exc:
61
+ raise ExtractError(f"Could not fetch {url}: {exc}") from exc
62
+ return resp
63
+
64
+
65
+ def extract_source(target: str, ocr=None, max_ocr_pages: int = 10) -> ExtractedDoc:
66
+ """Dispatch on the target: http(s) URL, local .pdf, .md or image path. `ocr` is the model that
67
+ reads images and scanned PDFs; without one, those are rejected."""
68
+ from esbi_cli.extract import clip, html, image, pdf
69
+
70
+ if is_url(target):
71
+ if repo := _GITHUB_REPO.match(target):
72
+ # the repo page is mostly navigation; its README is the content
73
+ owner, name = repo.groups()
74
+ try:
75
+ readme = _get(f"https://raw.githubusercontent.com/{owner}/{name}/HEAD/README.md")
76
+ except ExtractError:
77
+ readme = None # no README: read the page like any other
78
+ if readme is not None and len(readme.text.strip()) >= 200:
79
+ return ExtractedDoc(f"{owner}/{name}", readme.text.strip(), "article", target)
80
+ resp = _get(target)
81
+ final_url = str(resp.url)
82
+ content_type = resp.headers.get("content-type", "")
83
+ if "application/pdf" in content_type or final_url.lower().endswith(".pdf"):
84
+ doc = pdf.extract_pdf_bytes(
85
+ resp.content,
86
+ fallback_title=Path(urlparse(final_url).path).stem,
87
+ ocr=ocr,
88
+ max_ocr_pages=max_ocr_pages,
89
+ )
90
+ doc.url = target
91
+ return doc
92
+ doc = html.extract_html(resp.text, url=final_url)
93
+ doc.url = target
94
+ return doc
95
+
96
+ path = Path(target).expanduser()
97
+ if not path.is_file():
98
+ raise ExtractError(f"{target} is neither an http(s) URL nor an existing file")
99
+ if path.suffix.lower() == ".pdf":
100
+ return pdf.extract_pdf_bytes(
101
+ path.read_bytes(), fallback_title=path.stem, ocr=ocr, max_ocr_pages=max_ocr_pages
102
+ )
103
+ if path.suffix.lower() == ".md":
104
+ return clip.extract_clip(path)
105
+ if path.suffix.lower() in image.IMAGE_SUFFIXES:
106
+ return image.extract_image(path, ocr)
107
+ raise ExtractError(f"Unsupported file type: {path.suffix or path.name}")
@@ -0,0 +1,30 @@
1
+ """Markdown notes, typically Obsidian Web Clipper output: the clipped body is the content."""
2
+
3
+ import re
4
+ from pathlib import Path
5
+
6
+ from esbi_cli.extract import ExtractedDoc, ExtractError
7
+ from esbi_cli.vault import parse_page
8
+
9
+ KINDS = ("article", "paper", "email", "video")
10
+ MIN_CHARS = 40 # social posts are short; anything below this is an empty clip
11
+
12
+
13
+ def extract_clip(path: Path) -> ExtractedDoc:
14
+ page = parse_page(path, path.read_text(encoding="utf-8"))
15
+ text = page.body.strip()
16
+ if len(text) < MIN_CHARS:
17
+ raise ExtractError(f"{path.name} has almost no text ({len(text)} chars)")
18
+ url = page.meta.get("source") or page.meta.get("url")
19
+ title = str(page.meta.get("title") or path.stem).strip()
20
+ kind = page.meta.get("kind")
21
+ if (
22
+ kind == "video"
23
+ ): # the Clipper writes one transcript line per row: make each its own paragraph
24
+ text = re.sub(r"\n(?=\*\*\d+(?::\d{2}){1,2}\*\* · )", "\n\n", text)
25
+ return ExtractedDoc(
26
+ title=title,
27
+ text=text,
28
+ kind=kind if kind in KINDS else "article",
29
+ url=str(url) if url else None,
30
+ )
@@ -0,0 +1,60 @@
1
+ import re
2
+ from urllib.parse import urljoin, urlparse
3
+
4
+ import lxml.html
5
+ import trafilatura
6
+
7
+ from esbi_cli.extract import ExtractedDoc, ExtractError
8
+
9
+ MAX_IMAGES = 5
10
+ NOT_CONTENT = re.compile(r"logo|icon|avatar|sprite|pixel|tracking|badge|button", re.I)
11
+
12
+
13
+ def _image_links(html: str, base_url: str) -> list[tuple[str, str]]:
14
+ """(alt, url) of the article's content images. Linked, never downloaded."""
15
+ try:
16
+ tree = lxml.html.fromstring(html)
17
+ except (lxml.etree.ParserError, ValueError):
18
+ return []
19
+ region = next(iter(tree.xpath("//article") or tree.xpath("//main") or [tree]))
20
+ links, seen = [], set()
21
+ for img in region.iter("img"):
22
+ src = (img.get("src") or "").strip()
23
+ if (
24
+ not src
25
+ or src.startswith("data:")
26
+ or urlparse(src).path.lower().endswith((".svg", ".gif"))
27
+ ):
28
+ continue
29
+ try:
30
+ width_px, height_px = int(img.get("width") or 0), int(img.get("height") or 0)
31
+ except ValueError:
32
+ width_px = height_px = 0
33
+ if (width_px and width_px < 100) or (height_px and height_px < 100):
34
+ continue # a pixel, an icon
35
+ if NOT_CONTENT.search(src) or NOT_CONTENT.search(img.get("class") or ""):
36
+ continue
37
+ url = urljoin(base_url, src)
38
+ if urlparse(url).scheme in ("http", "https") and url not in seen:
39
+ seen.add(url)
40
+ links.append((" ".join((img.get("alt") or "").split()), url))
41
+ if len(links) == MAX_IMAGES:
42
+ break
43
+ return links
44
+
45
+
46
+ def extract_html(html: str, url: str) -> ExtractedDoc:
47
+ text = trafilatura.extract(
48
+ html, url=url, output_format="markdown", include_links=False, include_tables=True
49
+ )
50
+ if not text or len(text.strip()) < 200:
51
+ raise ExtractError(f"No readable article content found at {url}")
52
+ meta = trafilatura.extract_metadata(html, default_url=url)
53
+ title = (meta.title if meta and meta.title else None) or url
54
+ return ExtractedDoc(
55
+ title=title.strip(),
56
+ text=text.strip(),
57
+ kind="article",
58
+ url=url,
59
+ image_links=_image_links(html, url),
60
+ )
@@ -0,0 +1,58 @@
1
+ """Images (screenshots, photos of slides, scanned pages): an OCR model reads the text."""
2
+
3
+ import io
4
+ from pathlib import Path
5
+
6
+ from PIL import Image, ImageOps
7
+
8
+ from esbi_cli.extract import ExtractedDoc, ExtractError, Figure
9
+ from esbi_cli.llm.adapter import LLMError
10
+
11
+ IMAGE_SUFFIXES = (".png", ".jpg", ".jpeg", ".webp", ".tif", ".tiff")
12
+ MAX_SIDE_PX = 1600 # px: measured, a dense page at this size reads as well as at twice the size
13
+ MIN_CHARS = 40 # less than this is a photo or a diagram, not text worth a note
14
+ NO_OCR = (
15
+ "add an [llm.ocr] model to config.toml to read images and scanned PDFs "
16
+ "(`sb init --ocr`, https://rubenamaury.github.io/esbi-cli/docs/reference/configuration/)"
17
+ )
18
+
19
+
20
+ def read_text(ocr, png: bytes, name: str) -> str:
21
+ try:
22
+ return ocr.read_image(png)
23
+ except LLMError as exc: # this source fails; the text model and the other sources are fine
24
+ raise ExtractError(f"The OCR model could not read {name}: {exc}") from exc
25
+
26
+
27
+ def to_png(data: bytes, name: str) -> bytes:
28
+ """Upright (EXIF), RGB, at most MAX_SIDE_PX on the long side: what the model sees and the note shows."""
29
+ try:
30
+ img = ImageOps.exif_transpose(Image.open(io.BytesIO(data))).convert("RGB")
31
+ except Exception as exc: # Pillow raises several unrelated error types
32
+ raise ExtractError(f"Could not open {name} as an image: {exc}") from exc
33
+ img.thumbnail((MAX_SIDE_PX, MAX_SIDE_PX))
34
+ out = io.BytesIO()
35
+ img.save(out, "PNG")
36
+ return out.getvalue()
37
+
38
+
39
+ def extract_image(path: Path, ocr) -> ExtractedDoc:
40
+ if ocr is None:
41
+ raise ExtractError(f"{path.name}: {NO_OCR}")
42
+ data = path.read_bytes()
43
+ png = to_png(data, path.name)
44
+ text = read_text(ocr, png, path.name).strip()
45
+ if len(text) < MIN_CHARS:
46
+ raise ExtractError(f"{path.name} has no readable text (a photo or a diagram?)")
47
+ return ExtractedDoc(
48
+ title=path.stem.replace("-", " ").replace("_", " ").strip(),
49
+ text=text,
50
+ kind="article",
51
+ image_bytes=data,
52
+ image_suffix=path.suffix.lower(),
53
+ figures=[image_figure(png)],
54
+ )
55
+
56
+
57
+ def image_figure(png: bytes) -> Figure:
58
+ return Figure(data=png, page=0) # page 0: it is not a page of anything
@@ -0,0 +1,109 @@
1
+ import hashlib
2
+ import re
3
+
4
+ import pymupdf
5
+ import pymupdf4llm
6
+
7
+ from esbi_cli.extract import ExtractedDoc, ExtractError, Figure
8
+ from esbi_cli.extract.image import MAX_SIDE_PX, MIN_CHARS, NO_OCR, read_text
9
+
10
+ CAPTION = re.compile(r"\s*(figure|fig\.?|figura)\s*\d+", re.I)
11
+ MAX_FIGURES = 8
12
+ MIN_WIDTH_POINTS, MIN_HEIGHT_POINTS = (
13
+ 100,
14
+ 80,
15
+ ) # PDF points: smaller images are icons, logos, bullets
16
+ MAX_ASPECT = 6 # thinner strips are rules and banners, not figures
17
+
18
+
19
+ def _caption_for(captions: list, rect: pymupdf.Rect) -> str | None:
20
+ """The figure caption printed just below the image (or just above it)."""
21
+ below = [b for b in captions if rect.y1 - 5 <= b[1] < rect.y1 + 90]
22
+ above = [b for b in captions if 0 <= rect.y0 - b[3] < 60]
23
+ nearest = sorted(below, key=lambda b: b[1] - rect.y1) or above
24
+ return " ".join(nearest[0][4].split())[:200] if nearest else None
25
+
26
+
27
+ MAX_FIGURE_PX = 3000
28
+
29
+
30
+ def _dpi_for(width_points: float, height_points: float) -> int:
31
+ """150 dpi, lowered so that no side passes MAX_FIGURE_PX: a hostile PDF can claim a huge image."""
32
+ return max(
33
+ 1, int(min(150, MAX_FIGURE_PX * 72 / max(width_points, height_points, 1)))
34
+ ) # pymupdf wants an int
35
+
36
+
37
+ def _figures(doc: pymupdf.Document) -> list[Figure]:
38
+ """Up to MAX_FIGURES real figures: the captioned ones first, then the biggest."""
39
+ candidates = [] # (page number, rect, caption)
40
+ for number, page in enumerate(doc, 1):
41
+ captions = [b for b in page.get_text("blocks") if len(b) > 4 and CAPTION.match(b[4])]
42
+ for image in page.get_images(full=True):
43
+ for rect in page.get_image_rects(image[0]):
44
+ w, h = rect.width, rect.height
45
+ if w < MIN_WIDTH_POINTS or h < MIN_HEIGHT_POINTS or max(w / h, h / w) > MAX_ASPECT:
46
+ continue
47
+ candidates.append((number, rect, _caption_for(captions, rect)))
48
+ candidates.sort(key=lambda c: (c[2] is None, -c[1].get_area()))
49
+ figures, seen = [], set()
50
+ for number, rect, caption in candidates:
51
+ if len(figures) == MAX_FIGURES:
52
+ break
53
+ # rendering the image's area (not extracting the raw stream) keeps masks and overlays right
54
+ data = (
55
+ doc[number - 1]
56
+ .get_pixmap(clip=rect, dpi=_dpi_for(rect.width, rect.height))
57
+ .tobytes("png")
58
+ )
59
+ digest = hashlib.sha1(data).hexdigest()
60
+ if digest not in seen:
61
+ seen.add(digest)
62
+ figures.append(Figure(data=data, page=number, caption=caption))
63
+ return sorted(figures, key=lambda f: f.page) # reading order
64
+
65
+
66
+ def _ocr_pages(doc: pymupdf.Document, ocr, max_pages: int) -> tuple[str, list[str]]:
67
+ """A scanned PDF has no text layer: render its first `max_pages` pages and read them."""
68
+ pages = []
69
+ for number, page in enumerate(doc, 1):
70
+ if number > max_pages:
71
+ break
72
+ dpi = max(1, int(min(130, MAX_SIDE_PX * 72 / max(page.rect.width, page.rect.height, 1))))
73
+ png = page.get_pixmap(dpi=dpi).tobytes("png")
74
+ pages.append(read_text(ocr, png, f"page {number}").strip())
75
+ text = "\n\n".join(p for p in pages if p)
76
+ if len(text) < MIN_CHARS:
77
+ raise ExtractError("PDF has no readable text, even with OCR")
78
+ if doc.page_count <= max_pages:
79
+ return text, []
80
+ return text, [f"OCR read the first {max_pages} of {doc.page_count} pages ([run].ocr_max_pages)"]
81
+
82
+
83
+ def extract_pdf_bytes(
84
+ data: bytes, fallback_title: str, ocr=None, max_ocr_pages: int = 10
85
+ ) -> ExtractedDoc:
86
+ try:
87
+ doc = pymupdf.open(stream=data, filetype="pdf")
88
+ except Exception as exc: # pymupdf raises several unrelated error types
89
+ raise ExtractError(f"Could not open PDF: {exc}") from exc
90
+ with doc:
91
+ text = pymupdf4llm.to_markdown(doc)
92
+ meta_title = (doc.metadata or {}).get("title", "") or ""
93
+ warnings: list[str] = []
94
+ if len(text.strip()) < 200:
95
+ if ocr is None:
96
+ raise ExtractError(f"PDF has almost no extractable text (scanned?): {NO_OCR}")
97
+ text, warnings = _ocr_pages(doc, ocr, max_ocr_pages)
98
+ figures = [] # the pages are images: showing them all as figures would only add noise
99
+ else:
100
+ figures = _figures(doc)
101
+ title = meta_title.strip() or fallback_title.replace("-", " ").replace("_", " ").strip()
102
+ return ExtractedDoc(
103
+ title=title,
104
+ text=text.strip(),
105
+ kind="paper",
106
+ pdf_bytes=data,
107
+ figures=figures,
108
+ warnings=warnings,
109
+ )
esbi_cli/gitops.py ADDED
@@ -0,0 +1,101 @@
1
+ import subprocess
2
+ from pathlib import Path
3
+
4
+ # What the worker commits. SCHEMA.md and the golden questions are the user's own, but they are
5
+ # versioned so that a restore from the remote does not lose them.
6
+ VAULT_MANAGED = (
7
+ "wiki",
8
+ "raw",
9
+ "attachments",
10
+ "index.md",
11
+ "log.md",
12
+ "Home.md",
13
+ "SCHEMA.md",
14
+ ".esbi/golden.jsonl",
15
+ )
16
+ # The state folder is ignored, but for the golden questions.
17
+ STATE_IGNORE = (".esbi/*", "!.esbi/golden.jsonl")
18
+ GIT_PUSH_TIMEOUT_SECONDS = 120 # an unreachable remote must not hold a run
19
+
20
+
21
+ class GitError(RuntimeError):
22
+ pass
23
+
24
+
25
+ def has_git() -> bool:
26
+ """Is a working `git` installed? (On a Mac without the developer tools, /usr/bin/git exists
27
+ but fails, so asking it is the only honest test.)"""
28
+ try:
29
+ return subprocess.run(["git", "--version"], capture_output=True).returncode == 0
30
+ except OSError:
31
+ return False
32
+
33
+
34
+ def _git(root: Path, *args: str) -> subprocess.CompletedProcess:
35
+ try:
36
+ return subprocess.run(["git", *args], cwd=root, capture_output=True, text=True)
37
+ except FileNotFoundError:
38
+ raise GitError("git is not installed") from None
39
+
40
+
41
+ def _has_files(path: Path) -> bool:
42
+ """A managed path is committable only if it holds a file: git cannot track an empty folder,
43
+ and naming one in `git commit -- <path>` fails ("pathspec did not match")."""
44
+ return path.is_file() or (path.is_dir() and any(f.is_file() for f in path.rglob("*")))
45
+
46
+
47
+ def ignore_state(root: Path) -> None:
48
+ """Make `.gitignore` ignore the state folder except the golden questions. A vault from before
49
+ this (a plain `.esbi/` line, or none after the `.secondbrain` rename) is upgraded in place."""
50
+ ignore = root / ".gitignore"
51
+ if not ignore.exists():
52
+ return
53
+ lines = ignore.read_text().splitlines()
54
+ if STATE_IGNORE[0] in lines:
55
+ return
56
+ at = lines.index(".esbi/") if ".esbi/" in lines else len(lines)
57
+ lines[at : at + 1] = STATE_IGNORE
58
+ ignore.write_text("\n".join(lines) + "\n")
59
+
60
+
61
+ def commit_vault(root: Path, message: str) -> bool:
62
+ """Commit only worker-managed paths, leaving the user's own edits (.obsidian etc.) alone.
63
+
64
+ Returns True if a commit was made, False if there was nothing to commit.
65
+ """
66
+ if not (root / ".git").exists(): # history is optional: no repository, nothing to commit
67
+ return False
68
+ ignore_state(root)
69
+ paths = [p for p in VAULT_MANAGED if _has_files(root / p)]
70
+ # a path the user's own rules ignore would make `git add` fail and stop every commit
71
+ hidden = _git(root, "check-ignore", "--", *paths).stdout.split() if paths else []
72
+ paths = [p for p in paths if p not in hidden]
73
+ if not paths:
74
+ return False
75
+ added = _git(root, "add", "--", *paths)
76
+ if added.returncode != 0:
77
+ raise GitError(added.stderr.strip())
78
+ if _git(root, "diff", "--cached", "--quiet", "--", *paths).returncode == 0:
79
+ return False
80
+ done = _git(root, "commit", "-m", message, "--", *paths)
81
+ if done.returncode != 0:
82
+ raise GitError(done.stderr.strip() or done.stdout.strip())
83
+ return True
84
+
85
+
86
+ def push_vault(root: Path) -> str | None:
87
+ """Back the vault up to its `origin`, if it has one. Returns None when done or when there is
88
+ no remote, else the reason: a backup that is unreachable must never stop a run."""
89
+ if not (root / ".git").exists() or _git(root, "remote", "get-url", "origin").returncode != 0:
90
+ return None
91
+ try:
92
+ done = subprocess.run(
93
+ ["git", "push", "origin", "HEAD", "--tags"],
94
+ cwd=root,
95
+ capture_output=True,
96
+ text=True,
97
+ timeout=GIT_PUSH_TIMEOUT_SECONDS,
98
+ )
99
+ except subprocess.TimeoutExpired:
100
+ return "git push timed out"
101
+ return None if done.returncode == 0 else (done.stderr.strip() or done.stdout.strip())