majordomo-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- majordomo/__init__.py +30 -0
- majordomo/activity.py +669 -0
- majordomo/agent.py +217 -0
- majordomo/asr.py +396 -0
- majordomo/brief.py +83 -0
- majordomo/chat.py +768 -0
- majordomo/cli.py +1446 -0
- majordomo/config.example.yml +229 -0
- majordomo/config.py +561 -0
- majordomo/context.py +139 -0
- majordomo/coordinator.py +255 -0
- majordomo/documents.py +165 -0
- majordomo/dotenv.py +93 -0
- majordomo/firstrun.py +416 -0
- majordomo/hook.py +186 -0
- majordomo/install.py +229 -0
- majordomo/jsonlog.py +50 -0
- majordomo/keys.py +321 -0
- majordomo/llm.py +680 -0
- majordomo/memory.py +530 -0
- majordomo/models.py +186 -0
- majordomo/panel.py +282 -0
- majordomo/paths.py +79 -0
- majordomo/prompts.py +672 -0
- majordomo/render.py +182 -0
- majordomo/resume.py +102 -0
- majordomo/router.py +82 -0
- majordomo/scaffold.py +233 -0
- majordomo/session.py +471 -0
- majordomo/speechgate.py +100 -0
- majordomo/state.py +223 -0
- majordomo/tools.py +668 -0
- majordomo/tray.py +84 -0
- majordomo/trigger.py +235 -0
- majordomo/tts.py +353 -0
- majordomo/workers/__init__.py +41 -0
- majordomo/workers/github.py +232 -0
- majordomo/workers/gmail.py +328 -0
- majordomo/workers/sessions.py +197 -0
- majordomo_cli-0.1.0.dist-info/METADATA +328 -0
- majordomo_cli-0.1.0.dist-info/RECORD +44 -0
- majordomo_cli-0.1.0.dist-info/WHEEL +5 -0
- majordomo_cli-0.1.0.dist-info/entry_points.txt +2 -0
- majordomo_cli-0.1.0.dist-info/top_level.txt +1 -0
majordomo/__init__.py
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""Majordomo — the head of household staff who runs the workers and briefs you."""
|
|
2
|
+
|
|
3
|
+
#: Single source of truth for the version. ``pyproject.toml`` reads this
|
|
4
|
+
#: attribute rather than carrying its own literal, so the two cannot disagree.
|
|
5
|
+
__version__ = "0.1.0"
|
|
6
|
+
|
|
7
|
+
#: The distribution name, which is **not** the name of this package and **not**
|
|
8
|
+
#: the name of the command either. `majordomo` is taken on PyPI; `mj` is untaken
|
|
9
|
+
#: but rejected by it ("The name 'mj' isn't allowed" — too short or too close to
|
|
10
|
+
#: something else, a rule the JSON API cannot be asked about). The command stays
|
|
11
|
+
#: `mj`, because `[project.scripts]` is independent of all this.
|
|
12
|
+
DISTRIBUTION = "majordomo-cli"
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def install_hint(extra: str) -> str:
|
|
16
|
+
"""How to add an optional extra, for whichever way this was installed.
|
|
17
|
+
|
|
18
|
+
Lives here, once, because four modules need to say it — ``asr``,
|
|
19
|
+
``tts``, ``documents``, ``tray`` — and each had its own copy naming
|
|
20
|
+
``pip install majordomo[...]``. Both halves of that are now wrong: the
|
|
21
|
+
distribution is ``mj``, and the recommended install is ``uv tool``, which
|
|
22
|
+
puts the package in an isolated environment where a plain ``pip install``
|
|
23
|
+
lands somewhere else entirely and appears to do nothing.
|
|
24
|
+
|
|
25
|
+
Both forms are named because we cannot tell from in here which one applies,
|
|
26
|
+
and guessing wrong wastes the reader's time on the one line that was supposed
|
|
27
|
+
to save it.
|
|
28
|
+
"""
|
|
29
|
+
spec = f'"{DISTRIBUTION}[{extra}]"'
|
|
30
|
+
return f"uv tool install --force {spec} (or, in a venv: pip install {spec})"
|
majordomo/activity.py
ADDED
|
@@ -0,0 +1,669 @@
|
|
|
1
|
+
"""Your own GitHub activity, cached locally.
|
|
2
|
+
|
|
3
|
+
The GitHub *worker* answers "what is waiting on me right now?" — a live question
|
|
4
|
+
whose answer is worthless if it's five minutes old. This answers a different
|
|
5
|
+
one: "what have I been doing?" That has no reason to touch the network, so it
|
|
6
|
+
doesn't. Fetch occasionally, store, and read from disk.
|
|
7
|
+
|
|
8
|
+
Two consequences worth stating, because they are the point:
|
|
9
|
+
|
|
10
|
+
- ``mj ask "what did I work on in June?"`` is instant and works on a plane.
|
|
11
|
+
- A GitHub outage costs you nothing. The cache answers, and says how old it is.
|
|
12
|
+
|
|
13
|
+
── STORAGE ──────────────────────────────────────────────────────────────────
|
|
14
|
+
``~/.majordomo/activity.jsonl`` — append-only, one event per line, exactly the
|
|
15
|
+
discipline ``state.py`` uses and for the same reason: a torn final line from a
|
|
16
|
+
process killed mid-write must cost one event, not the whole history.
|
|
17
|
+
|
|
18
|
+
Two different prunings, and the distinction matters:
|
|
19
|
+
|
|
20
|
+
- **The read window** (``recent``) filters in memory and never touches the file.
|
|
21
|
+
This is what ``activity_days`` controls, and it is cheap.
|
|
22
|
+
- **Compaction** (``compact``) does rewrite, but only past a size threshold and
|
|
23
|
+
only from ``refresh`` — never from a read path. It keeps a window twice as
|
|
24
|
+
wide as the read window, because dropping an event the moment it leaves view
|
|
25
|
+
means a later widening of ``activity_days`` finds nothing behind it.
|
|
26
|
+
|
|
27
|
+
Rewriting an append-only log is where such logs get corrupted, so the rewrite
|
|
28
|
+
goes through ``jsonlog.rewrite`` — write beside the target, then replace, which
|
|
29
|
+
is atomic. Shared with ``state.py`` so a fix to it reaches both.
|
|
30
|
+
─────────────────────────────────────────────────────────────────────────────
|
|
31
|
+
"""
|
|
32
|
+
from __future__ import annotations
|
|
33
|
+
|
|
34
|
+
import json
|
|
35
|
+
import re
|
|
36
|
+
from dataclasses import dataclass, field
|
|
37
|
+
from datetime import datetime, timedelta, timezone
|
|
38
|
+
from pathlib import Path
|
|
39
|
+
|
|
40
|
+
import httpx
|
|
41
|
+
|
|
42
|
+
from majordomo.config import Config, GitHubConfig
|
|
43
|
+
from majordomo import jsonlog
|
|
44
|
+
from majordomo.paths import activity_fetched_path, activity_path
|
|
45
|
+
from majordomo.workers.github import GitHubError, get_json, resolve_token
|
|
46
|
+
|
|
47
|
+
#: How the three searches map onto something readable. Ordered: the first
|
|
48
|
+
#: search to claim a URL wins, and "you opened this" beats "you commented on
|
|
49
|
+
#: this" as a description of the same pull request.
|
|
50
|
+
KIND_LABELS = {
|
|
51
|
+
"pr": "opened a PR",
|
|
52
|
+
"commit": "committed",
|
|
53
|
+
"comment": "commented",
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
#: A YYYY-MM heading. Anything else groups under "undated".
|
|
57
|
+
_MONTH = re.compile(r"^\d{4}-\d{2}$")
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@dataclass(frozen=True)
|
|
61
|
+
class ActivityEvent:
|
|
62
|
+
"""One thing you did, flattened out of three different API shapes."""
|
|
63
|
+
|
|
64
|
+
#: Dedup key. The html_url — unique, stable, and meaningful to a human.
|
|
65
|
+
id: str
|
|
66
|
+
kind: str # "pr" | "commit" | "comment"
|
|
67
|
+
at: str # ISO-8601
|
|
68
|
+
repo: str # owner/name
|
|
69
|
+
title: str
|
|
70
|
+
url: str | None = None
|
|
71
|
+
|
|
72
|
+
def to_json(self) -> dict:
|
|
73
|
+
return {
|
|
74
|
+
"id": self.id,
|
|
75
|
+
"kind": self.kind,
|
|
76
|
+
"at": self.at,
|
|
77
|
+
"repo": self.repo,
|
|
78
|
+
"title": self.title,
|
|
79
|
+
"url": self.url,
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@dataclass
|
|
84
|
+
class ReadResult:
|
|
85
|
+
events: list[ActivityEvent] = field(default_factory=list)
|
|
86
|
+
#: Lines that were not parseable as an event. Surfaced by
|
|
87
|
+
#: ``mj activity --debug`` rather than silently dropped.
|
|
88
|
+
skipped: int = 0
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
@dataclass
|
|
92
|
+
class RefreshResult:
|
|
93
|
+
fetched: int = 0
|
|
94
|
+
added: int = 0
|
|
95
|
+
error: str | None = None
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
# ---------------------------------------------------------------------------
|
|
99
|
+
# Reading — never raises
|
|
100
|
+
# ---------------------------------------------------------------------------
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _parse_event(raw: object) -> ActivityEvent | None:
|
|
104
|
+
if not isinstance(raw, dict):
|
|
105
|
+
return None
|
|
106
|
+
ident = raw.get("id")
|
|
107
|
+
kind = raw.get("kind")
|
|
108
|
+
at = raw.get("at")
|
|
109
|
+
if not isinstance(ident, str) or not ident:
|
|
110
|
+
return None
|
|
111
|
+
if not isinstance(kind, str) or not isinstance(at, str):
|
|
112
|
+
return None
|
|
113
|
+
url = raw.get("url")
|
|
114
|
+
return ActivityEvent(
|
|
115
|
+
id=ident,
|
|
116
|
+
kind=kind,
|
|
117
|
+
at=at,
|
|
118
|
+
repo=raw.get("repo") if isinstance(raw.get("repo"), str) else "",
|
|
119
|
+
title=raw.get("title") if isinstance(raw.get("title"), str) else "",
|
|
120
|
+
url=url if isinstance(url, str) else None,
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def read_events_detailed(path: Path | str | None = None) -> ReadResult:
|
|
125
|
+
"""Read the whole log. Never raises."""
|
|
126
|
+
target = Path(path) if path is not None else activity_path()
|
|
127
|
+
if not target.is_file():
|
|
128
|
+
return ReadResult()
|
|
129
|
+
|
|
130
|
+
try:
|
|
131
|
+
raw = target.read_text(encoding="utf-8")
|
|
132
|
+
except OSError:
|
|
133
|
+
return ReadResult()
|
|
134
|
+
|
|
135
|
+
if raw.startswith(""):
|
|
136
|
+
raw = raw[1:]
|
|
137
|
+
|
|
138
|
+
result = ReadResult()
|
|
139
|
+
for line in raw.split("\n"):
|
|
140
|
+
trimmed = line.strip()
|
|
141
|
+
if not trimmed:
|
|
142
|
+
continue
|
|
143
|
+
try:
|
|
144
|
+
decoded = json.loads(trimmed)
|
|
145
|
+
except ValueError:
|
|
146
|
+
result.skipped += 1
|
|
147
|
+
continue
|
|
148
|
+
event = _parse_event(decoded)
|
|
149
|
+
if event is None:
|
|
150
|
+
result.skipped += 1
|
|
151
|
+
else:
|
|
152
|
+
result.events.append(event)
|
|
153
|
+
return result
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def read_events(path: Path | str | None = None) -> list[ActivityEvent]:
|
|
157
|
+
return read_events_detailed(path).events
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _parse_at(value: str) -> datetime | None:
|
|
161
|
+
try:
|
|
162
|
+
parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
|
|
163
|
+
except ValueError:
|
|
164
|
+
return None
|
|
165
|
+
return parsed if parsed.tzinfo else parsed.replace(tzinfo=timezone.utc)
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def recent(
|
|
169
|
+
days: int = 90,
|
|
170
|
+
now: datetime | None = None,
|
|
171
|
+
path: Path | str | None = None,
|
|
172
|
+
) -> list[ActivityEvent]:
|
|
173
|
+
"""Events inside the window, newest first. Never raises.
|
|
174
|
+
|
|
175
|
+
The window is **calendar days**, counted back from midnight today. A rolling
|
|
176
|
+
``now - days`` made ``--days 0`` a cutoff of *this instant*, so it could only
|
|
177
|
+
ever match events in the future and always returned nothing — while the CLI
|
|
178
|
+
described it as "just today". Counting from midnight makes 0 mean today, 1
|
|
179
|
+
mean since yesterday morning, and 7 mean the last week, which is what the
|
|
180
|
+
flag reads as.
|
|
181
|
+
|
|
182
|
+
``now`` is injectable for the same reason it is in the sessions worker:
|
|
183
|
+
reading the wall clock directly makes every fixture-dated test pass today
|
|
184
|
+
and fail next quarter.
|
|
185
|
+
"""
|
|
186
|
+
moment = now or datetime.now(timezone.utc)
|
|
187
|
+
# Midnight *where you are*, not midnight UTC. "Today" is a local idea: at
|
|
188
|
+
# UTC-5, snapping to UTC midnight puts the cutoff five hours into your
|
|
189
|
+
# morning and `--days 0` silently drops everything you did before lunch.
|
|
190
|
+
local = moment.astimezone()
|
|
191
|
+
midnight = local.replace(hour=0, minute=0, second=0, microsecond=0)
|
|
192
|
+
cutoff = midnight - timedelta(days=max(0, days))
|
|
193
|
+
|
|
194
|
+
kept = []
|
|
195
|
+
for event in read_events(path):
|
|
196
|
+
stamp = _parse_at(event.at)
|
|
197
|
+
if stamp is None or stamp >= cutoff:
|
|
198
|
+
kept.append(event)
|
|
199
|
+
|
|
200
|
+
kept.sort(key=lambda e: e.at, reverse=True)
|
|
201
|
+
return kept
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def newest_at(path: Path | str | None = None) -> datetime | None:
|
|
205
|
+
"""When the most recent cached event happened, or None if empty."""
|
|
206
|
+
stamps = [s for s in (_parse_at(e.at) for e in read_events(path)) if s]
|
|
207
|
+
return max(stamps) if stamps else None
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def _marker_for(path: Path | str | None) -> Path:
|
|
211
|
+
"""The fetch marker beside a given log, so tests can use a temp path."""
|
|
212
|
+
if path is None:
|
|
213
|
+
return activity_fetched_path()
|
|
214
|
+
return Path(str(path) + ".fetched")
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def record_fetch(
|
|
218
|
+
now: datetime | None = None, path: Path | str | None = None
|
|
219
|
+
) -> None:
|
|
220
|
+
"""Note that we asked GitHub just now. Never raises."""
|
|
221
|
+
marker = _marker_for(path)
|
|
222
|
+
try:
|
|
223
|
+
marker.parent.mkdir(parents=True, exist_ok=True)
|
|
224
|
+
marker.write_text(
|
|
225
|
+
(now or datetime.now(timezone.utc)).isoformat(), encoding="utf-8"
|
|
226
|
+
)
|
|
227
|
+
except OSError:
|
|
228
|
+
# A missing marker only means we refresh more often than needed.
|
|
229
|
+
pass
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def last_fetch(path: Path | str | None = None) -> datetime | None:
|
|
233
|
+
"""When we last asked GitHub, or None if never. Never raises."""
|
|
234
|
+
marker = _marker_for(path)
|
|
235
|
+
if not marker.is_file():
|
|
236
|
+
return None
|
|
237
|
+
try:
|
|
238
|
+
return _parse_at(marker.read_text(encoding="utf-8").strip())
|
|
239
|
+
except OSError:
|
|
240
|
+
return None
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def is_stale(
|
|
244
|
+
max_age_hours: float = 6.0,
|
|
245
|
+
now: datetime | None = None,
|
|
246
|
+
path: Path | str | None = None,
|
|
247
|
+
) -> bool:
|
|
248
|
+
"""Should we refresh before answering? Never fetched counts as stale.
|
|
249
|
+
|
|
250
|
+
Measured against the last **fetch**, not the newest event. Using the newest
|
|
251
|
+
event means a week without pushing makes the cache permanently stale, so
|
|
252
|
+
every single question fires three search calls before answering — the cache
|
|
253
|
+
failing hardest for exactly the quiet weeks it exists to cover.
|
|
254
|
+
"""
|
|
255
|
+
fetched = last_fetch(path)
|
|
256
|
+
if fetched is None:
|
|
257
|
+
return True
|
|
258
|
+
moment = now or datetime.now(timezone.utc)
|
|
259
|
+
return (moment - fetched) > timedelta(hours=max_age_hours)
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
# ---------------------------------------------------------------------------
|
|
263
|
+
# Writing
|
|
264
|
+
# ---------------------------------------------------------------------------
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def append_events(
|
|
268
|
+
events: list[ActivityEvent],
|
|
269
|
+
path: Path | str | None = None,
|
|
270
|
+
) -> int:
|
|
271
|
+
"""Append events not already present. Returns how many were new.
|
|
272
|
+
|
|
273
|
+
Dedup is against what is on disk, by ``id``. Refreshing twice in a row must
|
|
274
|
+
be a no-op, because the search windows overlap by design.
|
|
275
|
+
"""
|
|
276
|
+
target = Path(path) if path is not None else activity_path()
|
|
277
|
+
known = {e.id for e in read_events(target)}
|
|
278
|
+
|
|
279
|
+
fresh = []
|
|
280
|
+
for event in events:
|
|
281
|
+
if event.id in known:
|
|
282
|
+
continue
|
|
283
|
+
known.add(event.id) # also dedups within this batch
|
|
284
|
+
fresh.append(event)
|
|
285
|
+
|
|
286
|
+
if not fresh:
|
|
287
|
+
return 0
|
|
288
|
+
|
|
289
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
290
|
+
with open(target, "a", encoding="utf-8") as fh:
|
|
291
|
+
for event in fresh:
|
|
292
|
+
fh.write(json.dumps(event.to_json(), ensure_ascii=False) + "\n")
|
|
293
|
+
return len(fresh)
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
# ---------------------------------------------------------------------------
|
|
297
|
+
# Compaction — the same shape as state.compact, deliberately
|
|
298
|
+
# ---------------------------------------------------------------------------
|
|
299
|
+
|
|
300
|
+
#: Compact once the log passes this. Roughly 10k events, or years of activity.
|
|
301
|
+
#: Re-exported from ``jsonlog`` so callers of this module keep working.
|
|
302
|
+
COMPACT_OVER_BYTES = jsonlog.COMPACT_OVER_BYTES
|
|
303
|
+
|
|
304
|
+
#: Keep this multiple of ``activity_days`` when compacting. Wider than the read
|
|
305
|
+
#: window on purpose: dropping an event the moment it leaves the window means a
|
|
306
|
+
#: later widening of ``activity_days`` finds nothing behind it, and re-fetching
|
|
307
|
+
#: only reaches as far back as GitHub's search will go.
|
|
308
|
+
KEEP_WINDOW_MULTIPLE = 2
|
|
309
|
+
|
|
310
|
+
#: Never compact to a window narrower than this, whatever ``activity_days`` says.
|
|
311
|
+
#: The read window is a display preference and can legitimately be 0 ("today");
|
|
312
|
+
#: the *cache* is history, and history you throw away does not come back — the
|
|
313
|
+
#: search API stops at 1000 results however far you ask it to reach.
|
|
314
|
+
MIN_KEEP_DAYS = 30
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def compact(
|
|
318
|
+
path: Path | str | None = None,
|
|
319
|
+
keep_days: int = 180,
|
|
320
|
+
now: datetime | None = None,
|
|
321
|
+
) -> int:
|
|
322
|
+
"""Rewrite the log keeping only what is still in range. Returns lines dropped.
|
|
323
|
+
|
|
324
|
+
Shares the rewrite with :func:`majordomo.state.compact` via ``jsonlog`` —
|
|
325
|
+
the atomic part, which is what matters. What each keeps differs and stays
|
|
326
|
+
here: sessions carry an old event forward to preserve a topic, and activity
|
|
327
|
+
simply drops anything outside the window.
|
|
328
|
+
|
|
329
|
+
Simpler than the state version in one respect: there is no topic to carry
|
|
330
|
+
forward, so an event outside the window is simply gone.
|
|
331
|
+
"""
|
|
332
|
+
target = Path(path) if path is not None else activity_path()
|
|
333
|
+
if not target.is_file():
|
|
334
|
+
return 0
|
|
335
|
+
|
|
336
|
+
moment = now or datetime.now(timezone.utc)
|
|
337
|
+
cutoff = moment - timedelta(days=keep_days)
|
|
338
|
+
|
|
339
|
+
events = read_events(target)
|
|
340
|
+
if not events:
|
|
341
|
+
return 0
|
|
342
|
+
|
|
343
|
+
# An unparseable timestamp is kept, matching `recent`: losing an event to a
|
|
344
|
+
# malformed date is worse than carrying it.
|
|
345
|
+
kept = [
|
|
346
|
+
e for e in events
|
|
347
|
+
if (stamp := _parse_at(e.at)) is None or stamp >= cutoff
|
|
348
|
+
]
|
|
349
|
+
|
|
350
|
+
dropped = len(events) - len(kept)
|
|
351
|
+
if dropped <= 0:
|
|
352
|
+
return 0
|
|
353
|
+
|
|
354
|
+
jsonlog.rewrite(target, kept)
|
|
355
|
+
|
|
356
|
+
return dropped
|
|
357
|
+
|
|
358
|
+
|
|
359
|
+
def maybe_compact(
|
|
360
|
+
path: Path | str | None = None,
|
|
361
|
+
keep_days: int = 180,
|
|
362
|
+
max_bytes: int = COMPACT_OVER_BYTES,
|
|
363
|
+
) -> int:
|
|
364
|
+
"""Compact only if the log has actually got big. Never raises."""
|
|
365
|
+
target = Path(path) if path is not None else activity_path()
|
|
366
|
+
try:
|
|
367
|
+
if not jsonlog.is_large(target, max_bytes):
|
|
368
|
+
return 0
|
|
369
|
+
return compact(target, keep_days=keep_days)
|
|
370
|
+
except OSError:
|
|
371
|
+
# Housekeeping must never be the reason a question goes unanswered.
|
|
372
|
+
return 0
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
# ---------------------------------------------------------------------------
|
|
376
|
+
# Fetching
|
|
377
|
+
# ---------------------------------------------------------------------------
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
def _repo_from_url(url: object) -> str:
|
|
381
|
+
"""'https://api.github.com/repos/nav/majordomo' -> 'nav/majordomo'."""
|
|
382
|
+
if not isinstance(url, str):
|
|
383
|
+
return ""
|
|
384
|
+
parts = [p for p in url.rstrip("/").split("/") if p]
|
|
385
|
+
return "/".join(parts[-2:]) if len(parts) >= 2 else ""
|
|
386
|
+
|
|
387
|
+
|
|
388
|
+
def _items(payload: object) -> list:
|
|
389
|
+
if isinstance(payload, dict):
|
|
390
|
+
found = payload.get("items")
|
|
391
|
+
return found if isinstance(found, list) else []
|
|
392
|
+
return payload if isinstance(payload, list) else []
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
def _total_count(payload: object) -> int | None:
|
|
396
|
+
"""How many results GitHub says exist, or None if it didn't say.
|
|
397
|
+
|
|
398
|
+
This is what makes truncation visible. Search responses carry it and we
|
|
399
|
+
used to read only ``items`` — so a search returning exactly ``per_page``
|
|
400
|
+
results was indistinguishable from one that happened to have that many.
|
|
401
|
+
"""
|
|
402
|
+
if isinstance(payload, dict):
|
|
403
|
+
total = payload.get("total_count")
|
|
404
|
+
if isinstance(total, int):
|
|
405
|
+
return total
|
|
406
|
+
return None
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
def _as_pr(item: dict) -> ActivityEvent | None:
|
|
410
|
+
url = item.get("html_url")
|
|
411
|
+
if not isinstance(url, str):
|
|
412
|
+
return None
|
|
413
|
+
return ActivityEvent(
|
|
414
|
+
id=url,
|
|
415
|
+
kind="pr",
|
|
416
|
+
at=str(item.get("updated_at") or item.get("created_at") or ""),
|
|
417
|
+
repo=_repo_from_url(item.get("repository_url")),
|
|
418
|
+
title=str(item.get("title") or "(untitled)"),
|
|
419
|
+
url=url,
|
|
420
|
+
)
|
|
421
|
+
|
|
422
|
+
|
|
423
|
+
def _as_commit(item: dict) -> ActivityEvent | None:
|
|
424
|
+
url = item.get("html_url")
|
|
425
|
+
if not isinstance(url, str):
|
|
426
|
+
return None
|
|
427
|
+
commit = item.get("commit") or {}
|
|
428
|
+
author = commit.get("author") or {}
|
|
429
|
+
repository = item.get("repository") or {}
|
|
430
|
+
message = str(commit.get("message") or "")
|
|
431
|
+
return ActivityEvent(
|
|
432
|
+
id=url,
|
|
433
|
+
kind="commit",
|
|
434
|
+
at=str(author.get("date") or ""),
|
|
435
|
+
repo=str(repository.get("full_name") or ""),
|
|
436
|
+
# A commit message is a paragraph; the subject is the fact.
|
|
437
|
+
title=message.split("\n", 1)[0][:200] or "(no message)",
|
|
438
|
+
url=url,
|
|
439
|
+
)
|
|
440
|
+
|
|
441
|
+
|
|
442
|
+
def _as_comment(item: dict) -> ActivityEvent | None:
|
|
443
|
+
url = item.get("html_url")
|
|
444
|
+
if not isinstance(url, str):
|
|
445
|
+
return None
|
|
446
|
+
return ActivityEvent(
|
|
447
|
+
id=url,
|
|
448
|
+
kind="comment",
|
|
449
|
+
at=str(item.get("updated_at") or ""),
|
|
450
|
+
repo=_repo_from_url(item.get("repository_url")),
|
|
451
|
+
title=str(item.get("title") or "(untitled)"),
|
|
452
|
+
url=url,
|
|
453
|
+
)
|
|
454
|
+
|
|
455
|
+
|
|
456
|
+
#: GitHub's search API will not return results beyond this offset. Ask for page
|
|
457
|
+
#: 11 at 100 per page and you get a 422, not an empty page — so this is a wall,
|
|
458
|
+
#: not a preference, and no configuration can move it.
|
|
459
|
+
SEARCH_RESULT_CEILING = 1_000
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
#: GitHub rejects a per_page above this and substitutes its own default.
|
|
463
|
+
MAX_PER_PAGE = 100
|
|
464
|
+
|
|
465
|
+
|
|
466
|
+
def _per_page(cfg: GitHubConfig) -> int:
|
|
467
|
+
"""Results per request, clamped to what the API will actually honour.
|
|
468
|
+
|
|
469
|
+
Clamping only inside ``_last_page`` was not enough: the unclamped value also
|
|
470
|
+
went into the request *and* into the ``len(items) < per_page`` stop
|
|
471
|
+
condition, where a 0 can never be reached — so every page was fetched
|
|
472
|
+
whether or not there was anything left. Same shape as the ``max_messages: 0``
|
|
473
|
+
bug in the Gmail worker: a limit that is only enforced in one of the places
|
|
474
|
+
it is read.
|
|
475
|
+
"""
|
|
476
|
+
return max(1, min(cfg.activity_per_page, MAX_PER_PAGE))
|
|
477
|
+
|
|
478
|
+
|
|
479
|
+
def _last_page(cfg: GitHubConfig) -> int:
|
|
480
|
+
"""The highest page worth requesting: your setting, or the API's wall."""
|
|
481
|
+
return max(1, min(cfg.activity_max_pages, SEARCH_RESULT_CEILING // _per_page(cfg)))
|
|
482
|
+
|
|
483
|
+
|
|
484
|
+
def _how_to_get_more(cfg: GitHubConfig) -> str:
|
|
485
|
+
"""What to actually do about a truncated search.
|
|
486
|
+
|
|
487
|
+
This used to say "raise sources.github.activity_max_pages" unconditionally.
|
|
488
|
+
Past the API's ceiling that advice is worse than none: following it turns a
|
|
489
|
+
silent truncation into a 422 on every refresh. Once you are at the wall the
|
|
490
|
+
only thing that works is asking for a narrower window.
|
|
491
|
+
"""
|
|
492
|
+
if cfg.activity_max_pages < SEARCH_RESULT_CEILING // _per_page(cfg):
|
|
493
|
+
return "raise sources.github.activity_max_pages to reach the rest"
|
|
494
|
+
return (
|
|
495
|
+
f"that is GitHub's {SEARCH_RESULT_CEILING}-result search limit, not a "
|
|
496
|
+
f"setting — narrow the window with --days to see further back"
|
|
497
|
+
)
|
|
498
|
+
|
|
499
|
+
|
|
500
|
+
def fetch(
|
|
501
|
+
cfg: GitHubConfig, token: str, since: str
|
|
502
|
+
) -> tuple[list[ActivityEvent], list[str]]:
|
|
503
|
+
"""Pull your activity since ``since`` (a YYYY-MM-DD date).
|
|
504
|
+
|
|
505
|
+
Returns ``(events, errors)``. Each search is independent and a failure in
|
|
506
|
+
one does **not** discard what the others returned: GitHub's search endpoints
|
|
507
|
+
rate-limit separately and commit search is the flakiest of the three, so
|
|
508
|
+
all-or-nothing meant one 403 on the last call threw away a complete set of
|
|
509
|
+
pull requests already in hand.
|
|
510
|
+
"""
|
|
511
|
+
headers = {
|
|
512
|
+
"Authorization": f"Bearer {token}",
|
|
513
|
+
"Accept": "application/vnd.github+json",
|
|
514
|
+
"X-GitHub-Api-Version": "2022-11-28",
|
|
515
|
+
}
|
|
516
|
+
per_page = _per_page(cfg)
|
|
517
|
+
|
|
518
|
+
# Ordered: the first search to claim a URL wins in append_events, and
|
|
519
|
+
# "you opened this PR" beats "you commented" as a description of the same PR.
|
|
520
|
+
# Each carries its own label. Deriving one from the query gave `author:@me`
|
|
521
|
+
# for both the pull-request and the commit search, so a truncation or error
|
|
522
|
+
# warning could not say which endpoint it came from — and those two fail for
|
|
523
|
+
# different reasons and at different rates.
|
|
524
|
+
searches = (
|
|
525
|
+
("pull requests", "/search/issues", f"author:@me type:pr updated:>={since}", _as_pr),
|
|
526
|
+
("commits", "/search/commits", f"author:@me author-date:>={since}", _as_commit),
|
|
527
|
+
("comments", "/search/issues", f"commenter:@me updated:>={since}", _as_comment),
|
|
528
|
+
)
|
|
529
|
+
|
|
530
|
+
events: list[ActivityEvent] = []
|
|
531
|
+
errors: list[str] = []
|
|
532
|
+
notices: list[str] = []
|
|
533
|
+
|
|
534
|
+
with httpx.Client(base_url=cfg.api_base, headers=headers, timeout=cfg.timeout) as client:
|
|
535
|
+
for label, url, query, convert in searches:
|
|
536
|
+
collected = 0
|
|
537
|
+
total = None
|
|
538
|
+
|
|
539
|
+
for page in range(1, _last_page(cfg) + 1):
|
|
540
|
+
try:
|
|
541
|
+
payload = get_json(
|
|
542
|
+
client,
|
|
543
|
+
url,
|
|
544
|
+
{"q": query, "per_page": per_page, "page": page},
|
|
545
|
+
)
|
|
546
|
+
except GitHubError as exc:
|
|
547
|
+
# Page 1 failing means this search returned nothing; a later
|
|
548
|
+
# page failing still leaves the earlier ones in `events`.
|
|
549
|
+
errors.append(f"{label}: {exc}")
|
|
550
|
+
break
|
|
551
|
+
|
|
552
|
+
if total is None:
|
|
553
|
+
total = _total_count(payload)
|
|
554
|
+
|
|
555
|
+
items = _items(payload)
|
|
556
|
+
for item in items:
|
|
557
|
+
event = convert(item) if isinstance(item, dict) else None
|
|
558
|
+
if event is not None:
|
|
559
|
+
events.append(event)
|
|
560
|
+
collected += len(items)
|
|
561
|
+
|
|
562
|
+
# A short page is the last page — asking for another wastes a
|
|
563
|
+
# request against a rate-limited endpoint.
|
|
564
|
+
if len(items) < per_page:
|
|
565
|
+
break
|
|
566
|
+
|
|
567
|
+
# Say so when GitHub had more than we took. Before this, a search
|
|
568
|
+
# returning exactly `per_page` results looked identical to one that
|
|
569
|
+
# had exactly that many — 100 commits in a 90-day window was a
|
|
570
|
+
# ceiling being reported as a count.
|
|
571
|
+
if total is not None and total > collected:
|
|
572
|
+
# A *notice*, not an error. Sharing the errors list made a
|
|
573
|
+
# completely successful refresh report "could not refresh",
|
|
574
|
+
# which is the opposite of what happened.
|
|
575
|
+
notices.append(
|
|
576
|
+
f"{label}: took {collected} of {total} — {_how_to_get_more(cfg)}"
|
|
577
|
+
)
|
|
578
|
+
|
|
579
|
+
return events, errors + notices
|
|
580
|
+
|
|
581
|
+
|
|
582
|
+
def refresh(
|
|
583
|
+
config: Config,
|
|
584
|
+
token: str | None = None,
|
|
585
|
+
now: datetime | None = None,
|
|
586
|
+
path: Path | str | None = None,
|
|
587
|
+
) -> RefreshResult:
|
|
588
|
+
"""Fetch and store. Never raises — a failure is reported, not thrown.
|
|
589
|
+
|
|
590
|
+
Same contract as a worker: this runs on the way to answering a question, and
|
|
591
|
+
a GitHub outage must cost you the *freshness* of the answer, not the answer.
|
|
592
|
+
"""
|
|
593
|
+
cfg = config.sources.github
|
|
594
|
+
moment = now or datetime.now(timezone.utc)
|
|
595
|
+
since = (moment - timedelta(days=cfg.activity_days)).date().isoformat()
|
|
596
|
+
|
|
597
|
+
try:
|
|
598
|
+
resolved = token or resolve_token(cfg)
|
|
599
|
+
events, errors = fetch(cfg, resolved, since)
|
|
600
|
+
except GitHubError as exc:
|
|
601
|
+
return RefreshResult(error=str(exc))
|
|
602
|
+
except Exception as exc: # pragma: no cover - defensive
|
|
603
|
+
return RefreshResult(error=f"unexpected: {exc}")
|
|
604
|
+
|
|
605
|
+
added = append_events(events, path)
|
|
606
|
+
|
|
607
|
+
# Housekeeping at the natural mutation point. Not on read: reads happen on
|
|
608
|
+
# the way to answering a question and must stay cheap, and this is already
|
|
609
|
+
# the slow path.
|
|
610
|
+
# Floored, because `activity_days: 0` is a legitimate setting — it means
|
|
611
|
+
# "show me today" on the read path — and multiplying it gives a keep-window
|
|
612
|
+
# of zero, which compacts the entire history away. The cache cannot be
|
|
613
|
+
# rebuilt past GitHub's 1000-result search ceiling, so that is permanent
|
|
614
|
+
# data loss triggered by a config value we deliberately made valid.
|
|
615
|
+
keep_days = max(MIN_KEEP_DAYS, cfg.activity_days * KEEP_WINDOW_MULTIPLE)
|
|
616
|
+
maybe_compact(path, keep_days=keep_days)
|
|
617
|
+
|
|
618
|
+
# Recorded even on a partial failure: we *did* ask, and the point of the
|
|
619
|
+
# marker is to stop every question re-firing the same searches.
|
|
620
|
+
record_fetch(moment, path)
|
|
621
|
+
|
|
622
|
+
return RefreshResult(
|
|
623
|
+
fetched=len(events),
|
|
624
|
+
added=added,
|
|
625
|
+
error="; ".join(errors) if errors else None,
|
|
626
|
+
)
|
|
627
|
+
|
|
628
|
+
|
|
629
|
+
# ---------------------------------------------------------------------------
|
|
630
|
+
# Rendering
|
|
631
|
+
# ---------------------------------------------------------------------------
|
|
632
|
+
|
|
633
|
+
|
|
634
|
+
def describe(event: ActivityEvent) -> str:
|
|
635
|
+
label = KIND_LABELS.get(event.kind, event.kind)
|
|
636
|
+
where = f" in {event.repo}" if event.repo else ""
|
|
637
|
+
day = event.at[:10]
|
|
638
|
+
return f"{day} {label}{where}: {event.title}"
|
|
639
|
+
|
|
640
|
+
|
|
641
|
+
def digest(events: list[ActivityEvent], limit: int = 60) -> str:
|
|
642
|
+
"""The activity block that goes into a prompt.
|
|
643
|
+
|
|
644
|
+
Grouped by month rather than listed flat: "what was I doing in June" is the
|
|
645
|
+
question this exists to answer, and a model reads a dated heading far more
|
|
646
|
+
reliably than it infers month boundaries from sixty ISO timestamps.
|
|
647
|
+
"""
|
|
648
|
+
if not events:
|
|
649
|
+
return "No recorded GitHub activity."
|
|
650
|
+
|
|
651
|
+
shown = events[:limit]
|
|
652
|
+
lines: list[str] = []
|
|
653
|
+
current = ""
|
|
654
|
+
for event in shown:
|
|
655
|
+
month = event.at[:7]
|
|
656
|
+
# An unparseable timestamp is kept (losing an event to a bad date is
|
|
657
|
+
# worse than showing it) but must not become a heading — "not-a-d:" is
|
|
658
|
+
# how a rendering bug looks in production.
|
|
659
|
+
if not _MONTH.match(month):
|
|
660
|
+
month = "undated"
|
|
661
|
+
if month != current:
|
|
662
|
+
current = month
|
|
663
|
+
lines.append(f"\n{month}:")
|
|
664
|
+
lines.append(f" {describe(event)}")
|
|
665
|
+
|
|
666
|
+
if len(events) > limit:
|
|
667
|
+
lines.append(f"\n({len(events) - limit} older entries not listed.)")
|
|
668
|
+
|
|
669
|
+
return "\n".join(lines).strip()
|