git-ew 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- git_ew/__init__.py +166 -0
- git_ew/__main__.py +32 -0
- git_ew/_internal/__init__.py +17 -0
- git_ew/_internal/app.py +418 -0
- git_ew/_internal/cli.py +246 -0
- git_ew/_internal/config.py +148 -0
- git_ew/_internal/database.py +297 -0
- git_ew/_internal/debug.py +130 -0
- git_ew/_internal/email_fetcher.py +301 -0
- git_ew/_internal/email_parser.py +274 -0
- git_ew/_internal/email_sender.py +220 -0
- git_ew/_internal/mailing_lists/__init__.py +19 -0
- git_ew/_internal/mailing_lists/zsh_workers/__init__.py +17 -0
- git_ew/_internal/mailing_lists/zsh_workers/ingest.py +427 -0
- git_ew/_internal/mailing_lists/zsh_workers/sync_archives.py +300 -0
- git_ew/_internal/models.py +149 -0
- git_ew/_internal/secrets.py +59 -0
- git_ew/_internal/sync.py +156 -0
- git_ew/_internal/thread_utils.py +185 -0
- git_ew/py.typed +0 -0
- git_ew/static/css/style.css +608 -0
- git_ew/static/js/main.js +47 -0
- git_ew/templates/base.html +38 -0
- git_ew/templates/index.html +44 -0
- git_ew/templates/thread.html +153 -0
- git_ew-0.1.0.dist-info/METADATA +291 -0
- git_ew-0.1.0.dist-info/RECORD +30 -0
- git_ew-0.1.0.dist-info/WHEEL +4 -0
- git_ew-0.1.0.dist-info/entry_points.txt +5 -0
- git_ew-0.1.0.dist-info/licenses/LICENSE +15 -0
|
@@ -0,0 +1,300 @@
|
|
|
1
|
+
# SPDX-License-Identifier: ISC
|
|
2
|
+
|
|
3
|
+
# Copyright (c) 2021, Timothée Mazzucotelli and contributors
|
|
4
|
+
|
|
5
|
+
# Permission to use, copy, modify, and/or distribute this software for any
|
|
6
|
+
# purpose with or without fee is hereby granted, provided that the above
|
|
7
|
+
# copyright notice and this permission notice appear in all copies.
|
|
8
|
+
|
|
9
|
+
# THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES
|
|
10
|
+
# WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF
|
|
11
|
+
# MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR
|
|
12
|
+
# ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES
|
|
13
|
+
# WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN
|
|
14
|
+
# ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF
|
|
15
|
+
# OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE.
|
|
16
|
+
|
|
17
|
+
# Zsh-workers mailing list utilities.
|
|
18
|
+
|
|
19
|
+
import argparse
|
|
20
|
+
import logging
|
|
21
|
+
import sys
|
|
22
|
+
from datetime import date
|
|
23
|
+
from html.parser import HTMLParser
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
from time import strptime
|
|
26
|
+
from urllib.error import URLError
|
|
27
|
+
from urllib.request import urlopen, urlretrieve
|
|
28
|
+
|
|
29
|
+
BASE_URL = "https://www.zsh.org/mla/zsh-workers/"
|
|
30
|
+
"""Base URL for zsh-workers mailing list archives."""
|
|
31
|
+
_logger = logging.getLogger(__name__)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class LinkExtractor(HTMLParser):
|
|
35
|
+
"""Extract .tgz archive links and dates from HTML."""
|
|
36
|
+
|
|
37
|
+
def __init__(self):
|
|
38
|
+
super().__init__()
|
|
39
|
+
self.archives: dict[str, date | None] = {}
|
|
40
|
+
"""Archive filenames and their dates found in the page."""
|
|
41
|
+
self._in_pre = False
|
|
42
|
+
self._current_line = ""
|
|
43
|
+
|
|
44
|
+
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
|
45
|
+
"""Track pre tags and extract href attributes."""
|
|
46
|
+
if tag == "pre":
|
|
47
|
+
self._in_pre = True
|
|
48
|
+
elif tag == "a" and self._in_pre:
|
|
49
|
+
for attr, value in attrs:
|
|
50
|
+
if attr == "href" and value and value.endswith(".tgz"):
|
|
51
|
+
filename = value.split("/")[-1]
|
|
52
|
+
# Date will be extracted from text content after the link
|
|
53
|
+
self.archives[filename] = None
|
|
54
|
+
|
|
55
|
+
def handle_endtag(self, tag: str) -> None:
|
|
56
|
+
"""Track end of pre tag."""
|
|
57
|
+
if tag == "pre":
|
|
58
|
+
self._in_pre = False
|
|
59
|
+
|
|
60
|
+
def handle_data(self, data: str) -> None:
|
|
61
|
+
"""Extract date information from pre-formatted text."""
|
|
62
|
+
if not self._in_pre:
|
|
63
|
+
return
|
|
64
|
+
|
|
65
|
+
self._current_line += data
|
|
66
|
+
|
|
67
|
+
# Look for date patterns in the line (DD-MMM-YYYY format)
|
|
68
|
+
# Example: "12-Jun-1995"
|
|
69
|
+
if len(self._current_line) > 50: # Approximate line length # noqa: PLR2004
|
|
70
|
+
parts = self._current_line.split()
|
|
71
|
+
for i, part in enumerate(parts):
|
|
72
|
+
# Try to parse date from parts
|
|
73
|
+
if "-" in part and len(parts) > i + 1:
|
|
74
|
+
try:
|
|
75
|
+
# Try common date formats
|
|
76
|
+
date_str = part
|
|
77
|
+
# Convert French month names to English
|
|
78
|
+
date_str = date_str.replace("janv.", "Jan")
|
|
79
|
+
date_str = date_str.replace("févr.", "Feb")
|
|
80
|
+
date_str = date_str.replace("mars", "Mar")
|
|
81
|
+
date_str = date_str.replace("avril", "Apr")
|
|
82
|
+
date_str = date_str.replace("mai", "May")
|
|
83
|
+
date_str = date_str.replace("juin", "Jun")
|
|
84
|
+
date_str = date_str.replace("juil.", "Jul")
|
|
85
|
+
date_str = date_str.replace("août", "Aug")
|
|
86
|
+
date_str = date_str.replace("sept.", "Sep")
|
|
87
|
+
date_str = date_str.replace("oct.", "Oct")
|
|
88
|
+
date_str = date_str.replace("nov.", "Nov")
|
|
89
|
+
date_str = date_str.replace("déc.", "Dec")
|
|
90
|
+
|
|
91
|
+
parsed_time = strptime(date_str, "%d-%b-%Y")
|
|
92
|
+
parsed = date(parsed_time.tm_year, parsed_time.tm_mon, parsed_time.tm_mday)
|
|
93
|
+
# Associate this date with the last archived filename found
|
|
94
|
+
if self.archives:
|
|
95
|
+
last_filename = list(self.archives.keys())[-1]
|
|
96
|
+
if self.archives[last_filename] is None:
|
|
97
|
+
self.archives[last_filename] = parsed
|
|
98
|
+
except (ValueError, IndexError):
|
|
99
|
+
_logger.debug("Could not parse archive date from %r", part, exc_info=True)
|
|
100
|
+
|
|
101
|
+
self._current_line = ""
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def fetch_archive_list() -> dict[str, date | None]:
|
|
105
|
+
"""Fetch the list of available archives from zsh.org with dates.
|
|
106
|
+
|
|
107
|
+
Returns:
|
|
108
|
+
Dict mapping archive filenames to their dates (or None if date couldn't be parsed)
|
|
109
|
+
|
|
110
|
+
Raises:
|
|
111
|
+
URLError: If the page cannot be fetched.
|
|
112
|
+
"""
|
|
113
|
+
try:
|
|
114
|
+
with urlopen(BASE_URL) as response:
|
|
115
|
+
html = response.read().decode("utf-8")
|
|
116
|
+
except URLError as e:
|
|
117
|
+
raise URLError(f"Failed to fetch {BASE_URL}: {e}") from e
|
|
118
|
+
|
|
119
|
+
parser = LinkExtractor()
|
|
120
|
+
parser.feed(html)
|
|
121
|
+
# Return sorted by filename
|
|
122
|
+
return dict(sorted(parser.archives.items()))
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _get_matching_archives(
|
|
126
|
+
available_archives: dict[str, date | None],
|
|
127
|
+
since: date | None = None,
|
|
128
|
+
until: date | None = None,
|
|
129
|
+
) -> list[str]:
|
|
130
|
+
"""Return archive filenames within the requested date range.
|
|
131
|
+
|
|
132
|
+
Args:
|
|
133
|
+
available_archives: Dict of available archive filenames to dates.
|
|
134
|
+
since: Only include archives from this date onwards.
|
|
135
|
+
until: Only include archives up to this date.
|
|
136
|
+
|
|
137
|
+
Returns:
|
|
138
|
+
Archive filenames within the requested date range.
|
|
139
|
+
"""
|
|
140
|
+
matching = []
|
|
141
|
+
|
|
142
|
+
for filename, file_date in available_archives.items():
|
|
143
|
+
if since is not None and file_date is not None and file_date < since:
|
|
144
|
+
continue
|
|
145
|
+
if until is not None and file_date is not None and file_date > until:
|
|
146
|
+
continue
|
|
147
|
+
|
|
148
|
+
matching.append(filename)
|
|
149
|
+
|
|
150
|
+
return matching
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def get_missing_archives(
|
|
154
|
+
archive_dir: Path,
|
|
155
|
+
available_archives: dict[str, date | None],
|
|
156
|
+
since: date | None = None,
|
|
157
|
+
until: date | None = None,
|
|
158
|
+
) -> list[str]:
|
|
159
|
+
"""Determine which matching archives need to be downloaded.
|
|
160
|
+
|
|
161
|
+
Args:
|
|
162
|
+
archive_dir: Directory where archives are stored.
|
|
163
|
+
available_archives: Dict of available archive filenames to dates.
|
|
164
|
+
since: Only include archives from this date onwards.
|
|
165
|
+
until: Only include archives up to this date.
|
|
166
|
+
|
|
167
|
+
Returns:
|
|
168
|
+
Matching archive filenames that do not exist locally.
|
|
169
|
+
"""
|
|
170
|
+
existing = {file.name for file in archive_dir.glob("*.tgz")}
|
|
171
|
+
|
|
172
|
+
return [
|
|
173
|
+
filename for filename in _get_matching_archives(available_archives, since, until) if filename not in existing
|
|
174
|
+
]
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def download_archive(filename: str, archive_dir: Path) -> bool:
|
|
178
|
+
"""Download a single archive.
|
|
179
|
+
|
|
180
|
+
Args:
|
|
181
|
+
filename: Archive filename to download.
|
|
182
|
+
archive_dir: Directory to save the archive to.
|
|
183
|
+
|
|
184
|
+
Returns:
|
|
185
|
+
True if download succeeded, False otherwise.
|
|
186
|
+
"""
|
|
187
|
+
url = BASE_URL + filename
|
|
188
|
+
output_path = archive_dir / filename
|
|
189
|
+
|
|
190
|
+
try:
|
|
191
|
+
_logger.info("Downloading %s", filename)
|
|
192
|
+
urlretrieve(url, output_path) # noqa: S310
|
|
193
|
+
_logger.info("Downloaded %s", filename)
|
|
194
|
+
except URLError as error:
|
|
195
|
+
_logger.error("Failed to download %s: %s", filename, error) # noqa: TRY400
|
|
196
|
+
# Clean up partially downloaded file
|
|
197
|
+
if output_path.exists():
|
|
198
|
+
output_path.unlink()
|
|
199
|
+
return False
|
|
200
|
+
else:
|
|
201
|
+
return True
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _main() -> int:
|
|
205
|
+
"""Main entry point."""
|
|
206
|
+
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
|
207
|
+
parser = argparse.ArgumentParser(
|
|
208
|
+
description="Sync zsh-workers mailing list archives from zsh.org",
|
|
209
|
+
)
|
|
210
|
+
parser.add_argument(
|
|
211
|
+
"-d",
|
|
212
|
+
"--directory",
|
|
213
|
+
type=Path,
|
|
214
|
+
default=Path(".archives"),
|
|
215
|
+
help="Directory to store archives (default: .archives)",
|
|
216
|
+
)
|
|
217
|
+
parser.add_argument(
|
|
218
|
+
"-s",
|
|
219
|
+
"--since",
|
|
220
|
+
type=str,
|
|
221
|
+
help="Only download archives from this date onwards. "
|
|
222
|
+
"Format: YYYY (year) or YYYY-MM-DD (specific date). "
|
|
223
|
+
"Example: --since 2020 or --since 2020-01-15",
|
|
224
|
+
)
|
|
225
|
+
parser.add_argument(
|
|
226
|
+
"-n",
|
|
227
|
+
"--dry-run",
|
|
228
|
+
action="store_true",
|
|
229
|
+
help="Show what would be downloaded without actually downloading",
|
|
230
|
+
)
|
|
231
|
+
parser.add_argument(
|
|
232
|
+
"-v",
|
|
233
|
+
"--verbose",
|
|
234
|
+
action="store_true",
|
|
235
|
+
help="Show more detailed output",
|
|
236
|
+
)
|
|
237
|
+
|
|
238
|
+
args = parser.parse_args()
|
|
239
|
+
archive_dir = args.directory
|
|
240
|
+
|
|
241
|
+
# Parse --since argument
|
|
242
|
+
since_date: date | None = None
|
|
243
|
+
if args.since:
|
|
244
|
+
try:
|
|
245
|
+
since_date = (
|
|
246
|
+
date(int(args.since), 1, 1)
|
|
247
|
+
if len(args.since) == 4 # noqa: PLR2004
|
|
248
|
+
else date.fromisoformat(args.since)
|
|
249
|
+
)
|
|
250
|
+
except ValueError:
|
|
251
|
+
_logger.error("Invalid date format %r. Use YYYY or YYYY-MM-DD", args.since) # noqa: TRY400
|
|
252
|
+
return 1
|
|
253
|
+
|
|
254
|
+
# Create directory if it doesn't exist
|
|
255
|
+
archive_dir.mkdir(parents=True, exist_ok=True)
|
|
256
|
+
|
|
257
|
+
# Fetch available archives
|
|
258
|
+
_logger.info("Fetching archive list from zsh.org")
|
|
259
|
+
try:
|
|
260
|
+
available = fetch_archive_list()
|
|
261
|
+
except URLError as error:
|
|
262
|
+
_logger.error("Failed to fetch archive list: %s", error) # noqa: TRY400
|
|
263
|
+
return 1
|
|
264
|
+
|
|
265
|
+
_logger.info("Found %d archives available", len(available))
|
|
266
|
+
|
|
267
|
+
# Determine missing archives
|
|
268
|
+
missing = get_missing_archives(archive_dir, available, since_date)
|
|
269
|
+
|
|
270
|
+
if not missing:
|
|
271
|
+
_logger.info("All requested archives are already downloaded")
|
|
272
|
+
return 0
|
|
273
|
+
|
|
274
|
+
_logger.info("Found %d new archives to download", len(missing))
|
|
275
|
+
|
|
276
|
+
if args.verbose:
|
|
277
|
+
_logger.info("Missing archives:")
|
|
278
|
+
for name in missing:
|
|
279
|
+
_logger.info(" - %s", name)
|
|
280
|
+
|
|
281
|
+
if args.dry_run:
|
|
282
|
+
_logger.info("Dry-run: no archives downloaded")
|
|
283
|
+
return 0
|
|
284
|
+
|
|
285
|
+
success_count = 0
|
|
286
|
+
for filename in missing:
|
|
287
|
+
if download_archive(filename, archive_dir):
|
|
288
|
+
success_count += 1
|
|
289
|
+
|
|
290
|
+
_logger.info("Downloaded %d/%d archives", success_count, len(missing))
|
|
291
|
+
|
|
292
|
+
if success_count < len(missing):
|
|
293
|
+
_logger.warning("%d archive downloads failed", len(missing) - success_count)
|
|
294
|
+
return 1
|
|
295
|
+
|
|
296
|
+
return 0
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
if __name__ == "__main__":
|
|
300
|
+
sys.exit(_main())
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
# SPDX-License-Identifier: ISC
|
|
2
|
+
|
|
3
|
+
# Copyright (c) 2021, Timothée Mazzucotelli and contributors
|
|
4
|
+
|
|
5
|
+
# Permission to use, copy, modify, and/or distribute this software for any
|
|
6
|
+
# purpose with or without fee is hereby granted, provided that the above
|
|
7
|
+
# copyright notice and this permission notice appear in all copies.
|
|
8
|
+
|
|
9
|
+
# THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES
|
|
10
|
+
# WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF
|
|
11
|
+
# MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR
|
|
12
|
+
# ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES
|
|
13
|
+
# WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN
|
|
14
|
+
# ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF
|
|
15
|
+
# OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE.
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from datetime import datetime # noqa: TC003
|
|
20
|
+
from typing import TYPE_CHECKING
|
|
21
|
+
|
|
22
|
+
from sqlalchemy import Boolean, DateTime, ForeignKey, Integer, String, Text, create_engine
|
|
23
|
+
from sqlalchemy.orm import DeclarativeBase, Mapped, mapped_column, relationship, sessionmaker
|
|
24
|
+
|
|
25
|
+
if TYPE_CHECKING:
|
|
26
|
+
from sqlalchemy.engine import Engine
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class Base(DeclarativeBase):
|
|
30
|
+
"""Base class for all models."""
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class Thread(Base):
|
|
34
|
+
"""Represents an email thread (like a GitHub issue or PR)."""
|
|
35
|
+
|
|
36
|
+
__tablename__ = "threads"
|
|
37
|
+
"""Database table name."""
|
|
38
|
+
|
|
39
|
+
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
|
40
|
+
"""Thread unique identifier."""
|
|
41
|
+
subject: Mapped[str] = mapped_column(String(500), nullable=False)
|
|
42
|
+
"""Thread subject line."""
|
|
43
|
+
first_message_id: Mapped[str] = mapped_column(String(500), unique=True, nullable=False)
|
|
44
|
+
"""ID of the first message in the thread."""
|
|
45
|
+
created_at: Mapped[datetime] = mapped_column(DateTime, nullable=False)
|
|
46
|
+
"""When the thread was created."""
|
|
47
|
+
updated_at: Mapped[datetime] = mapped_column(DateTime, nullable=False)
|
|
48
|
+
"""When the thread was last updated."""
|
|
49
|
+
is_patch: Mapped[bool] = mapped_column(Boolean, default=False, nullable=False)
|
|
50
|
+
"""Whether this thread contains patches."""
|
|
51
|
+
status: Mapped[str] = mapped_column(String(50), default="open", nullable=False) # open, closed
|
|
52
|
+
"""Thread status (open or closed)."""
|
|
53
|
+
|
|
54
|
+
# Relationships
|
|
55
|
+
messages: Mapped[list[Message]] = relationship("Message", back_populates="thread", cascade="all, delete-orphan")
|
|
56
|
+
"""Messages in this thread."""
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class Message(Base):
|
|
60
|
+
"""Represents an individual email message in a thread."""
|
|
61
|
+
|
|
62
|
+
__tablename__ = "messages"
|
|
63
|
+
"""Database table name."""
|
|
64
|
+
|
|
65
|
+
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
|
66
|
+
"""Message unique identifier."""
|
|
67
|
+
message_id: Mapped[str] = mapped_column(String(500), unique=True, nullable=False, index=True)
|
|
68
|
+
"""Email message ID header."""
|
|
69
|
+
in_reply_to: Mapped[str | None] = mapped_column(String(500), nullable=True, index=True)
|
|
70
|
+
"""Message ID this reply refers to."""
|
|
71
|
+
thread_id: Mapped[int] = mapped_column(Integer, ForeignKey("threads.id"), nullable=False, index=True)
|
|
72
|
+
"""Thread this message belongs to."""
|
|
73
|
+
|
|
74
|
+
# Email metadata
|
|
75
|
+
from_email: Mapped[str] = mapped_column(String(255), nullable=False)
|
|
76
|
+
"""Sender email address."""
|
|
77
|
+
from_name: Mapped[str] = mapped_column(String(255), nullable=False)
|
|
78
|
+
"""Sender name."""
|
|
79
|
+
subject: Mapped[str] = mapped_column(String(500), nullable=False)
|
|
80
|
+
"""Email subject."""
|
|
81
|
+
date: Mapped[datetime] = mapped_column(DateTime, nullable=False, index=True)
|
|
82
|
+
"""Email date."""
|
|
83
|
+
|
|
84
|
+
# Content
|
|
85
|
+
body: Mapped[str] = mapped_column(Text, nullable=False)
|
|
86
|
+
"""Message body text."""
|
|
87
|
+
raw_email: Mapped[str] = mapped_column(Text, nullable=True)
|
|
88
|
+
"""Raw email content."""
|
|
89
|
+
|
|
90
|
+
# Patch information
|
|
91
|
+
is_patch: Mapped[bool] = mapped_column(Boolean, default=False, nullable=False)
|
|
92
|
+
"""Whether this message contains a patch."""
|
|
93
|
+
patch_content: Mapped[str | None] = mapped_column(Text, nullable=True)
|
|
94
|
+
"""Extracted patch content."""
|
|
95
|
+
|
|
96
|
+
# Relationships
|
|
97
|
+
thread: Mapped[Thread] = relationship("Thread", back_populates="messages")
|
|
98
|
+
"""Thread this message belongs to."""
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
class EmailSource(Base):
|
|
102
|
+
"""Represents an email archive source to fetch from."""
|
|
103
|
+
|
|
104
|
+
__tablename__ = "email_sources"
|
|
105
|
+
"""Database table name."""
|
|
106
|
+
|
|
107
|
+
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
|
108
|
+
"""Source unique identifier."""
|
|
109
|
+
name: Mapped[str] = mapped_column(String(255), nullable=False, unique=True)
|
|
110
|
+
"""Source name."""
|
|
111
|
+
source_type: Mapped[str] = mapped_column(String(50), nullable=False) # maildir, mbox, imap, archive_url
|
|
112
|
+
"""Type of email source (maildir, mbox, imap, etc)."""
|
|
113
|
+
config: Mapped[str] = mapped_column(Text, nullable=False) # JSON config for the source
|
|
114
|
+
"""JSON configuration for the source."""
|
|
115
|
+
last_synced: Mapped[datetime | None] = mapped_column(DateTime, nullable=True)
|
|
116
|
+
"""When this source was last synced."""
|
|
117
|
+
enabled: Mapped[bool] = mapped_column(Boolean, default=True, nullable=False)
|
|
118
|
+
"""Whether this source is enabled."""
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
class Configuration(Base):
|
|
122
|
+
"""Stores application configuration."""
|
|
123
|
+
|
|
124
|
+
__tablename__ = "configuration"
|
|
125
|
+
"""Database table name."""
|
|
126
|
+
|
|
127
|
+
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
|
128
|
+
"""Configuration unique identifier."""
|
|
129
|
+
key: Mapped[str] = mapped_column(String(255), nullable=False, unique=True)
|
|
130
|
+
"""Configuration key."""
|
|
131
|
+
value: Mapped[str] = mapped_column(Text, nullable=False)
|
|
132
|
+
"""Configuration value."""
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def get_engine(database_url: str = "sqlite+aiosqlite:///./git_ew.db") -> Engine:
|
|
136
|
+
"""Create database engine."""
|
|
137
|
+
return create_engine(database_url, echo=False)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def init_db(database_url: str = "sqlite:///./git_ew.db") -> Engine:
|
|
141
|
+
"""Initialize the database."""
|
|
142
|
+
engine = create_engine(database_url, echo=False)
|
|
143
|
+
Base.metadata.create_all(engine)
|
|
144
|
+
return engine
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def get_session_maker(engine: Engine) -> sessionmaker:
|
|
148
|
+
"""Get a session maker."""
|
|
149
|
+
return sessionmaker(bind=engine)
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
# SPDX-License-Identifier: ISC
|
|
2
|
+
|
|
3
|
+
# Copyright (c) 2021, Timothée Mazzucotelli and contributors
|
|
4
|
+
|
|
5
|
+
# Permission to use, copy, modify, and/or distribute this software for any
|
|
6
|
+
# purpose with or without fee is hereby granted, provided that the above
|
|
7
|
+
# copyright notice and this permission notice appear in all copies.
|
|
8
|
+
|
|
9
|
+
# THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES
|
|
10
|
+
# WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF
|
|
11
|
+
# MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR
|
|
12
|
+
# ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES
|
|
13
|
+
# WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN
|
|
14
|
+
# ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF
|
|
15
|
+
# OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE.
|
|
16
|
+
|
|
17
|
+
# Runtime secret resolution helpers.
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import shlex
|
|
22
|
+
import subprocess
|
|
23
|
+
from typing import Any, Literal, overload
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@overload
|
|
27
|
+
def resolve_password(config: dict[str, Any], *, required: Literal[True] = True) -> str: ...
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@overload
|
|
31
|
+
def resolve_password(config: dict[str, Any], *, required: Literal[False]) -> str | None: ...
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def resolve_password(config: dict[str, Any], *, required: bool = True) -> str | None:
|
|
35
|
+
"""Resolve a configured password.
|
|
36
|
+
|
|
37
|
+
Return `None` when no password is configured and `required` is false.
|
|
38
|
+
"""
|
|
39
|
+
command = config.get("password_command")
|
|
40
|
+
if command:
|
|
41
|
+
result = subprocess.run( # noqa: S603
|
|
42
|
+
shlex.split(command),
|
|
43
|
+
check=True,
|
|
44
|
+
capture_output=True,
|
|
45
|
+
text=True,
|
|
46
|
+
encoding="utf-8",
|
|
47
|
+
timeout=10,
|
|
48
|
+
)
|
|
49
|
+
password = result.stdout.strip()
|
|
50
|
+
if not password:
|
|
51
|
+
raise RuntimeError("password command returned no password")
|
|
52
|
+
return password
|
|
53
|
+
|
|
54
|
+
password = config.get("password")
|
|
55
|
+
if password:
|
|
56
|
+
return password
|
|
57
|
+
if not required:
|
|
58
|
+
return None
|
|
59
|
+
raise RuntimeError("neither password nor password_command is configured")
|
git_ew/_internal/sync.py
ADDED
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
# SPDX-License-Identifier: ISC
|
|
2
|
+
|
|
3
|
+
# Copyright (c) 2021, Timothée Mazzucotelli and contributors
|
|
4
|
+
|
|
5
|
+
# Permission to use, copy, modify, and/or distribute this software for any
|
|
6
|
+
# purpose with or without fee is hereby granted, provided that the above
|
|
7
|
+
# copyright notice and this permission notice appear in all copies.
|
|
8
|
+
|
|
9
|
+
# THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES
|
|
10
|
+
# WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF
|
|
11
|
+
# MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR
|
|
12
|
+
# ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES
|
|
13
|
+
# WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN
|
|
14
|
+
# ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF
|
|
15
|
+
# OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE.
|
|
16
|
+
|
|
17
|
+
# Email synchronization utilities for git-ew.
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import asyncio
|
|
22
|
+
import json
|
|
23
|
+
import logging
|
|
24
|
+
from typing import TypedDict
|
|
25
|
+
|
|
26
|
+
from git_ew._internal.database import Database
|
|
27
|
+
from git_ew._internal.email_fetcher import get_fetcher
|
|
28
|
+
|
|
29
|
+
_logger = logging.getLogger(__name__)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class _SyncStats(TypedDict):
|
|
33
|
+
"""Store email synchronization counters and errors."""
|
|
34
|
+
|
|
35
|
+
total_sources: int
|
|
36
|
+
processed_sources: int
|
|
37
|
+
total_messages: int
|
|
38
|
+
new_messages: int
|
|
39
|
+
new_threads: int
|
|
40
|
+
errors: list[str]
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
async def sync_all_sources(db: Database | None = None) -> _SyncStats:
|
|
44
|
+
"""Sync emails from all configured sources.
|
|
45
|
+
|
|
46
|
+
Args:
|
|
47
|
+
db: Database instance. If None, creates a new one.
|
|
48
|
+
|
|
49
|
+
Returns:
|
|
50
|
+
Dictionary with sync statistics.
|
|
51
|
+
"""
|
|
52
|
+
if db is None:
|
|
53
|
+
db = Database()
|
|
54
|
+
await db.init_db()
|
|
55
|
+
|
|
56
|
+
sources = await db.get_email_sources()
|
|
57
|
+
stats: _SyncStats = {
|
|
58
|
+
"total_sources": len(sources),
|
|
59
|
+
"processed_sources": 0,
|
|
60
|
+
"total_messages": 0,
|
|
61
|
+
"new_messages": 0,
|
|
62
|
+
"new_threads": 0,
|
|
63
|
+
"errors": [],
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
for source in sources:
|
|
67
|
+
if not source.enabled:
|
|
68
|
+
continue
|
|
69
|
+
|
|
70
|
+
try:
|
|
71
|
+
config = json.loads(source.config)
|
|
72
|
+
fetcher = get_fetcher(source.source_type, config)
|
|
73
|
+
|
|
74
|
+
source_messages = 0
|
|
75
|
+
async for parsed_email in fetcher.fetch_emails():
|
|
76
|
+
# Check if message already exists
|
|
77
|
+
existing = await db.get_message_by_id(parsed_email.message_id)
|
|
78
|
+
if existing:
|
|
79
|
+
if parsed_email.patch_content and not existing.patch_content:
|
|
80
|
+
await db.update_message_patch(parsed_email.message_id, parsed_email.patch_content)
|
|
81
|
+
continue
|
|
82
|
+
|
|
83
|
+
# Find or create thread
|
|
84
|
+
thread_id_str = parsed_email.get_thread_id()
|
|
85
|
+
thread = await db.get_thread_by_message_id(thread_id_str)
|
|
86
|
+
|
|
87
|
+
if not thread:
|
|
88
|
+
# Create new thread
|
|
89
|
+
thread = await db.create_thread(
|
|
90
|
+
subject=parsed_email.clean_subject,
|
|
91
|
+
first_message_id=thread_id_str,
|
|
92
|
+
is_patch=parsed_email.is_patch,
|
|
93
|
+
)
|
|
94
|
+
stats["new_threads"] += 1
|
|
95
|
+
|
|
96
|
+
# Create message
|
|
97
|
+
await db.create_message(
|
|
98
|
+
message_id=parsed_email.message_id,
|
|
99
|
+
thread_id=thread.id,
|
|
100
|
+
from_email=parsed_email.from_email,
|
|
101
|
+
from_name=parsed_email.from_name,
|
|
102
|
+
subject=parsed_email.subject,
|
|
103
|
+
date=parsed_email.date,
|
|
104
|
+
body=parsed_email.body,
|
|
105
|
+
in_reply_to=parsed_email.in_reply_to,
|
|
106
|
+
is_patch=parsed_email.is_patch,
|
|
107
|
+
patch_content=parsed_email.patch_content,
|
|
108
|
+
raw_email=parsed_email.raw,
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
source_messages += 1
|
|
112
|
+
stats["new_messages"] += 1
|
|
113
|
+
|
|
114
|
+
stats["processed_sources"] += 1
|
|
115
|
+
stats["total_messages"] += source_messages
|
|
116
|
+
|
|
117
|
+
_logger.info(f"✓ Synced {source_messages} messages from '{source.name}'")
|
|
118
|
+
|
|
119
|
+
except Exception as e:
|
|
120
|
+
error_msg = f"Error syncing source '{source.name}': {e}"
|
|
121
|
+
stats["errors"].append(error_msg)
|
|
122
|
+
_logger.exception("✗ %s", error_msg)
|
|
123
|
+
|
|
124
|
+
return stats
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
async def sync_command() -> int:
|
|
128
|
+
"""Run email sync as a standalone command."""
|
|
129
|
+
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
|
130
|
+
|
|
131
|
+
_logger.info("Starting email sync...")
|
|
132
|
+
_logger.info("-" * 50)
|
|
133
|
+
|
|
134
|
+
db = Database()
|
|
135
|
+
await db.init_db()
|
|
136
|
+
|
|
137
|
+
stats = await sync_all_sources(db)
|
|
138
|
+
|
|
139
|
+
_logger.info("-" * 50)
|
|
140
|
+
_logger.info("\nSync Summary:")
|
|
141
|
+
_logger.info(f" Sources processed: {stats['processed_sources']}/{stats['total_sources']}")
|
|
142
|
+
_logger.info(f" New threads: {stats['new_threads']}")
|
|
143
|
+
_logger.info(f" New messages: {stats['new_messages']}")
|
|
144
|
+
|
|
145
|
+
if stats["errors"]:
|
|
146
|
+
_logger.info(f"\n Errors: {len(stats['errors'])}")
|
|
147
|
+
for error in stats["errors"]:
|
|
148
|
+
_logger.info(f" - {error}")
|
|
149
|
+
|
|
150
|
+
return 0 if not stats["errors"] else 1
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
if __name__ == "__main__":
|
|
154
|
+
import sys
|
|
155
|
+
|
|
156
|
+
sys.exit(asyncio.run(sync_command()))
|