gmail-llm-cleanup 1.1.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gmail_llm_cleanup-1.1.2/LICENSE +21 -0
- gmail_llm_cleanup-1.1.2/PKG-INFO +606 -0
- gmail_llm_cleanup-1.1.2/README.md +565 -0
- gmail_llm_cleanup-1.1.2/pyproject.toml +67 -0
- gmail_llm_cleanup-1.1.2/setup.cfg +4 -0
- gmail_llm_cleanup-1.1.2/src/gmail_cleanup/__init__.py +2 -0
- gmail_llm_cleanup-1.1.2/src/gmail_cleanup/__main__.py +72 -0
- gmail_llm_cleanup-1.1.2/src/gmail_cleanup/applylog.py +274 -0
- gmail_llm_cleanup-1.1.2/src/gmail_cleanup/backends/__init__.py +18 -0
- gmail_llm_cleanup-1.1.2/src/gmail_cleanup/backends/claude.py +39 -0
- gmail_llm_cleanup-1.1.2/src/gmail_cleanup/backends/ollama.py +114 -0
- gmail_llm_cleanup-1.1.2/src/gmail_cleanup/backends/openai.py +129 -0
- gmail_llm_cleanup-1.1.2/src/gmail_cleanup/cli.py +948 -0
- gmail_llm_cleanup-1.1.2/src/gmail_cleanup/gmail_client.py +457 -0
- gmail_llm_cleanup-1.1.2/src/gmail_cleanup/portforward.py +159 -0
- gmail_llm_cleanup-1.1.2/src/gmail_cleanup/prompt.py +356 -0
- gmail_llm_cleanup-1.1.2/src/gmail_llm_cleanup.egg-info/PKG-INFO +606 -0
- gmail_llm_cleanup-1.1.2/src/gmail_llm_cleanup.egg-info/SOURCES.txt +35 -0
- gmail_llm_cleanup-1.1.2/src/gmail_llm_cleanup.egg-info/dependency_links.txt +1 -0
- gmail_llm_cleanup-1.1.2/src/gmail_llm_cleanup.egg-info/entry_points.txt +2 -0
- gmail_llm_cleanup-1.1.2/src/gmail_llm_cleanup.egg-info/requires.txt +18 -0
- gmail_llm_cleanup-1.1.2/src/gmail_llm_cleanup.egg-info/top_level.txt +1 -0
- gmail_llm_cleanup-1.1.2/tests/test_applylog.py +197 -0
- gmail_llm_cleanup-1.1.2/tests/test_backend_claude.py +92 -0
- gmail_llm_cleanup-1.1.2/tests/test_backend_ollama.py +123 -0
- gmail_llm_cleanup-1.1.2/tests/test_backend_openai.py +153 -0
- gmail_llm_cleanup-1.1.2/tests/test_backends_factory.py +51 -0
- gmail_llm_cleanup-1.1.2/tests/test_cli_apply_log.py +224 -0
- gmail_llm_cleanup-1.1.2/tests/test_cli_classify.py +276 -0
- gmail_llm_cleanup-1.1.2/tests/test_cli_misc.py +204 -0
- gmail_llm_cleanup-1.1.2/tests/test_cli_relabel.py +215 -0
- gmail_llm_cleanup-1.1.2/tests/test_dotenv_loading.py +106 -0
- gmail_llm_cleanup-1.1.2/tests/test_gmail_helpers.py +150 -0
- gmail_llm_cleanup-1.1.2/tests/test_gmail_retry.py +121 -0
- gmail_llm_cleanup-1.1.2/tests/test_gmail_search_skip.py +169 -0
- gmail_llm_cleanup-1.1.2/tests/test_portforward.py +129 -0
- gmail_llm_cleanup-1.1.2/tests/test_prompt.py +242 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Brett Rosequist
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,606 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: gmail-llm-cleanup
|
|
3
|
+
Version: 1.1.2
|
|
4
|
+
Summary: Triage a Gmail inbox with a local or cloud LLM
|
|
5
|
+
Author: Brett Rosequist
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/brosequist/gmail-cleanup-agent
|
|
8
|
+
Project-URL: Repository, https://github.com/brosequist/gmail-cleanup-agent
|
|
9
|
+
Project-URL: Issues, https://github.com/brosequist/gmail-cleanup-agent/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/brosequist/gmail-cleanup-agent/releases
|
|
11
|
+
Keywords: gmail,email,llm,ollama,claude,openai,classification,inbox,triage,cli
|
|
12
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
13
|
+
Classifier: Environment :: Console
|
|
14
|
+
Classifier: Intended Audience :: End Users/Desktop
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Topic :: Communications :: Email
|
|
21
|
+
Classifier: Topic :: Utilities
|
|
22
|
+
Requires-Python: >=3.11
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
License-File: LICENSE
|
|
25
|
+
Requires-Dist: google-api-python-client>=2.140.0
|
|
26
|
+
Requires-Dist: google-auth>=2.34.0
|
|
27
|
+
Requires-Dist: google-auth-oauthlib>=1.2.0
|
|
28
|
+
Requires-Dist: google-auth-httplib2>=0.2.0
|
|
29
|
+
Requires-Dist: httpx>=0.27.0
|
|
30
|
+
Requires-Dist: pyyaml>=6.0.2
|
|
31
|
+
Requires-Dist: click>=8.1.7
|
|
32
|
+
Requires-Dist: python-dotenv>=1.0.0
|
|
33
|
+
Provides-Extra: claude
|
|
34
|
+
Requires-Dist: anthropic>=0.40.0; extra == "claude"
|
|
35
|
+
Provides-Extra: openai
|
|
36
|
+
Requires-Dist: openai>=1.55.0; extra == "openai"
|
|
37
|
+
Provides-Extra: dev
|
|
38
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
39
|
+
Requires-Dist: pytest-cov>=5.0; extra == "dev"
|
|
40
|
+
Dynamic: license-file
|
|
41
|
+
|
|
42
|
+
# gmail-cleanup-agent
|
|
43
|
+
|
|
44
|
+
Triage a Gmail inbox you've given up on. Uses a local (or cloud) LLM to
|
|
45
|
+
classify every old email as **important** (keep + label it) or
|
|
46
|
+
**non-important** (move to trash), with a full audit log and a
|
|
47
|
+
sender whitelist that always wins.
|
|
48
|
+
|
|
49
|
+
Built for the inbox-bankruptcy use case: you have 50k–500k unlabeled
|
|
50
|
+
emails, most of them marketing / political / automated, but real
|
|
51
|
+
things are mixed in (family, receipts, school, medical, government).
|
|
52
|
+
Manual triage is intractable; a hard-coded rules engine misses too
|
|
53
|
+
much; a generic AI tool either touches things it shouldn't or sends
|
|
54
|
+
your mail to a vendor. This script is a middle path.
|
|
55
|
+
|
|
56
|
+
## Why this exists
|
|
57
|
+
|
|
58
|
+
I had ~313k old emails I'd been ignoring for a decade
|
|
59
|
+
(see [docs/results.md](docs/results.md) for the full run breakdown).
|
|
60
|
+
The math is brutal:
|
|
61
|
+
|
|
62
|
+
| Path | Time | Money | Privacy |
|
|
63
|
+
|---|---|---|---|
|
|
64
|
+
| **Manual** (3 sec / email, average) | ~260 hrs ≈ 6.5 weeks full-time | $0 | full |
|
|
65
|
+
| **This script, local LLM** on a consumer GPU | ~65 hrs of active LLM time (4 worker threads serialize on a single GPU; idle / downtime / failed-call retries excluded) | ~$6 of electricity | full — nothing leaves the machine |
|
|
66
|
+
| **This script, Claude Haiku 4.5** | ~60–100 min | ~$69 | sender/subject/snippet leave the machine |
|
|
67
|
+
| **This script, GPT-4o-mini** | ~60–100 min | ~$10 | same |
|
|
68
|
+
| **This script, Gemini 2.0 Flash** | ~60–100 min | ~$7 | same |
|
|
69
|
+
|
|
70
|
+
(Token + cost math for the API rows is in [docs/cost-math.md](docs/cost-math.md).
|
|
71
|
+
Local runtime is from my actual run on an RX 9070 XT + RX 9060 XT running
|
|
72
|
+
`qwen3.6:35b-a3b` at IQ3, concurrency 4. Your mileage will vary.)
|
|
73
|
+
|
|
74
|
+
The script is conservative by default: **dry-run is the default**,
|
|
75
|
+
**trash is reversible for 30 days**, and there's a **sender whitelist**
|
|
76
|
+
that the LLM cannot override.
|
|
77
|
+
|
|
78
|
+
**Expect an iterative workflow.** A first pass typically clears the
|
|
79
|
+
bulk of unwanted mail (~83 % of threads in the example run), but a
|
|
80
|
+
post-hoc audit of the *kept* side usually surfaces a few-percent
|
|
81
|
+
tail of false-keeps — recurring digests, old sign-in alerts, social-
|
|
82
|
+
network reply notifications — that warrant a much shorter follow-on
|
|
83
|
+
run with tightened rules. See [docs/results.md](docs/results.md) for
|
|
84
|
+
the procedure and concrete numbers from one real run.
|
|
85
|
+
|
|
86
|
+
## Features
|
|
87
|
+
|
|
88
|
+
- **Local-LLM-by-default** via Ollama or LM Studio. No email content
|
|
89
|
+
leaves your network unless you pick a cloud backend.
|
|
90
|
+
- **Pluggable backends** — Ollama, LM Studio (OpenAI-compatible),
|
|
91
|
+
Anthropic Claude, or real OpenAI. Same script, four wire formats.
|
|
92
|
+
- **Batched classification** — ~20 emails per LLM call.
|
|
93
|
+
- **Resumable** — checkpoints after every batch to `state.json`.
|
|
94
|
+
Ctrl+C and re-run with the same args; already-classified threads
|
|
95
|
+
are skipped at Gmail-list speed.
|
|
96
|
+
- **Dry-run by default** — every run produces a decision log first;
|
|
97
|
+
nothing touches Gmail until you pass `--apply`.
|
|
98
|
+
- **Trash, not permanent delete** — Gmail's 30-day recovery window
|
|
99
|
+
protects against false positives.
|
|
100
|
+
- **Sender whitelist** — addresses or domains that are *never*
|
|
101
|
+
classified as trash, regardless of LLM judgment.
|
|
102
|
+
- **Auto-create labels** — for important categories that don't exist
|
|
103
|
+
yet, the agent creates them based on your `labels.yaml` catalog.
|
|
104
|
+
- **Audit trail** — every decision logged with sender, subject,
|
|
105
|
+
action, label, and reason.
|
|
106
|
+
|
|
107
|
+
## What gets sent to the LLM
|
|
108
|
+
|
|
109
|
+
For each email:
|
|
110
|
+
|
|
111
|
+
- Sender (`From:` header)
|
|
112
|
+
- Subject
|
|
113
|
+
- Snippet (the ~200-char preview Gmail provides; capped at 300 chars
|
|
114
|
+
defensively by the prompt builder)
|
|
115
|
+
- **Age** in days (derived from Gmail's `internalDate`). Lets the
|
|
116
|
+
classifier apply different heuristics for "received last week" vs
|
|
117
|
+
"received 5 years ago" — especially useful for inbox-bankruptcy
|
|
118
|
+
on years-old mail where time-sensitive notices (sign-in alerts,
|
|
119
|
+
verification codes, expired sales) are no longer actionable.
|
|
120
|
+
- **`List-Unsubscribe: yes`** flag (RFC 2369). Personal mail almost
|
|
121
|
+
never has this; bulk / automated / marketing mail almost always
|
|
122
|
+
does. Very strong "this is automated" signal.
|
|
123
|
+
- **Never the full body** unless you explicitly pass `--include-body`
|
|
124
|
+
(off by default — and even then, snippet is preferred when
|
|
125
|
+
available).
|
|
126
|
+
- **Never the exact Date.** Only the derived age-in-days makes it
|
|
127
|
+
into the prompt, not the raw timestamp.
|
|
128
|
+
|
|
129
|
+
See [docs/privacy.md](docs/privacy.md) for the full breakdown.
|
|
130
|
+
|
|
131
|
+
## Labels: how kept mail gets organized
|
|
132
|
+
|
|
133
|
+
Every "keep" decision is also a **label-classification decision**. The
|
|
134
|
+
LLM doesn't just decide *whether* an email survives — it picks the
|
|
135
|
+
single best-matching label from a catalog you provide, and the script
|
|
136
|
+
applies that label in Gmail. After a run, every kept email is filed
|
|
137
|
+
under a specific category instead of sitting unsorted in your inbox.
|
|
138
|
+
|
|
139
|
+
The catalog lives in
|
|
140
|
+
[`config/labels.yaml`](#configlabelsyaml) and has two parts:
|
|
141
|
+
|
|
142
|
+
```yaml
|
|
143
|
+
existing: # labels that already exist in your Gmail
|
|
144
|
+
- Taxes
|
|
145
|
+
- Filed
|
|
146
|
+
auto_create: # categories the agent will create the first time
|
|
147
|
+
Family: "personal correspondence from friends or family members"
|
|
148
|
+
Receipts: "order confirmations, purchase receipts, ..."
|
|
149
|
+
Statements: "monthly/quarterly account statements ..."
|
|
150
|
+
Medical: "doctors, hospitals, pharmacies, lab results, EOBs"
|
|
151
|
+
Government: "DMV, IRS, immigration, voter, courts, social security"
|
|
152
|
+
Registrations: "event tickets, account creations, program enrollments"
|
|
153
|
+
# ...etc
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
The one-line descriptions go into the prompt as part of the catalog
|
|
157
|
+
section, so the model knows what each label means without you having
|
|
158
|
+
to write explicit rules for every category in
|
|
159
|
+
[`config/rules.md`](#configrulesmd).
|
|
160
|
+
|
|
161
|
+
A real run's outcome (per [docs/results.md](docs/results.md)):
|
|
162
|
+
|
|
163
|
+
| Label | Threads kept |
|
|
164
|
+
|---|---:|
|
|
165
|
+
| Receipts | 21,793 |
|
|
166
|
+
| Registrations | 11,640 |
|
|
167
|
+
| Organizations | 3,523 |
|
|
168
|
+
| School | 3,348 |
|
|
169
|
+
| Family | 2,721 |
|
|
170
|
+
| Medical | 1,844 |
|
|
171
|
+
| Government | 1,440 |
|
|
172
|
+
| Statements | 1,116 |
|
|
173
|
+
| Veterans | 488 |
|
|
174
|
+
| … | … |
|
|
175
|
+
| **Total kept (across 18 labels)** | **53,099** |
|
|
176
|
+
|
|
177
|
+
A few label-system properties worth knowing up front:
|
|
178
|
+
|
|
179
|
+
- **The model is forced to pick exactly one label per kept email** —
|
|
180
|
+
no comma-separated multi-labels, no "Misc/Other" fallback.
|
|
181
|
+
[`config/rules.md`](config/rules.example.md) includes a "Don't
|
|
182
|
+
reach for a catch-all label" principle: if no specific label
|
|
183
|
+
clearly fits, prefer trash. This keeps the label set tight.
|
|
184
|
+
- **Labels you don't include in the catalog won't be assigned.** The
|
|
185
|
+
model can only choose from the catalog you gave it.
|
|
186
|
+
- **You can reorganize later without re-classifying.** If after a
|
|
187
|
+
run you want to add a new category — splitting `Receipts` →
|
|
188
|
+
`Receipts` + `Travel` for booked-trip records, say — the
|
|
189
|
+
[`relabel`](#reorganizing-later-the-relabel-pass) subcommand
|
|
190
|
+
re-asks the model only for the label decision, never the
|
|
191
|
+
keep-vs-trash decision. Already-kept mail stays kept.
|
|
192
|
+
- **Tighten the catalog before the big run.** Each label adds tokens
|
|
193
|
+
to every batch's prompt; bloated catalogs cost real money on cloud
|
|
194
|
+
backends and slow down local runs. See the design tips in
|
|
195
|
+
[`config/labels.example.yaml`](config/labels.example.yaml).
|
|
196
|
+
|
|
197
|
+
## Install
|
|
198
|
+
|
|
199
|
+
Three options, in increasing order of isolation:
|
|
200
|
+
|
|
201
|
+
```bash
|
|
202
|
+
# 1) PyPI (any working directory; config lives in ./config/)
|
|
203
|
+
pipx install gmail-llm-cleanup # base
|
|
204
|
+
pipx install 'gmail-llm-cleanup[claude]' # + Anthropic
|
|
205
|
+
pipx install 'gmail-llm-cleanup[openai]' # + OpenAI / LM Studio / llama.cpp
|
|
206
|
+
|
|
207
|
+
# 2) Docker (no Python on the host)
|
|
208
|
+
docker run --rm -it \
|
|
209
|
+
-v "$PWD/config:/config" -v "$PWD:/work" -w /work \
|
|
210
|
+
ghcr.io/brosequist/gmail-llm-cleanup:latest classify --dry-run
|
|
211
|
+
|
|
212
|
+
# 3) Editable checkout (for hacking on the tool itself)
|
|
213
|
+
git clone https://github.com/brosequist/gmail-cleanup-agent
|
|
214
|
+
cd gmail-cleanup-agent
|
|
215
|
+
python -m venv .venv && source .venv/bin/activate
|
|
216
|
+
pip install -e '.[openai]' # or .[claude], or just .
|
|
217
|
+
```
|
|
218
|
+
|
|
219
|
+
The PyPI distribution name is `gmail-llm-cleanup` (the obvious
|
|
220
|
+
`gmail-cleanup-agent` name is already taken on PyPI by an unrelated
|
|
221
|
+
project). The Python import name (`gmail_cleanup`) and console script
|
|
222
|
+
(`gmail-cleanup`) are unaffected:
|
|
223
|
+
|
|
224
|
+
```bash
|
|
225
|
+
pipx install gmail-llm-cleanup
|
|
226
|
+
gmail-cleanup --help # binary on PATH
|
|
227
|
+
python -c "import gmail_cleanup" # import as a library
|
|
228
|
+
```
|
|
229
|
+
|
|
230
|
+
For the PyPI and Docker paths, the CLI looks for `config/` in your
|
|
231
|
+
current working directory by default. Override with
|
|
232
|
+
`GMAIL_CLEANUP_CONFIG_DIR=/path/to/config`.
|
|
233
|
+
|
|
234
|
+
## Quick start — local LLM
|
|
235
|
+
|
|
236
|
+
You need Python 3.11+ and a local LLM server. Pick one:
|
|
237
|
+
|
|
238
|
+
### Option A: Ollama
|
|
239
|
+
|
|
240
|
+
```bash
|
|
241
|
+
# 1. Install Ollama from https://ollama.ai, then pull a model
|
|
242
|
+
ollama pull qwen3:8b # 5 GB, runs on most modern GPUs
|
|
243
|
+
# or, larger / smarter:
|
|
244
|
+
ollama pull qwen3.6:35b-a3b # 15 GB, needs 16+ GB VRAM, much better quality
|
|
245
|
+
|
|
246
|
+
# 2. Make sure Ollama is serving (default: localhost:11434)
|
|
247
|
+
ollama serve & # or run `ollama` as a system service
|
|
248
|
+
|
|
249
|
+
# 3. Install this tool
|
|
250
|
+
git clone https://github.com/brosequist/gmail-cleanup-agent
|
|
251
|
+
cd gmail-cleanup-agent
|
|
252
|
+
python -m venv .venv && source .venv/bin/activate
|
|
253
|
+
pip install -e .
|
|
254
|
+
|
|
255
|
+
# 4. Configure — copy the template and uncomment the Ollama section
|
|
256
|
+
cp config/backend.env.example config/backend.env
|
|
257
|
+
# edit config/backend.env: uncomment GCA_BACKEND / OLLAMA_HOST / OLLAMA_MODEL
|
|
258
|
+
# (you can leave the other backend sections commented out)
|
|
259
|
+
```
|
|
260
|
+
|
|
261
|
+
`pip install -e .` registers a `gmail-cleanup` console script and
|
|
262
|
+
makes `python -m gmail_cleanup ...` work from any directory. The two
|
|
263
|
+
are interchangeable in the rest of this README.
|
|
264
|
+
|
|
265
|
+
`config/backend.env` is loaded automatically every time you run
|
|
266
|
+
`python -m gmail_cleanup ...`. Shell-exported env vars still override
|
|
267
|
+
what's in the file, so one-off overrides (e.g.
|
|
268
|
+
`OLLAMA_MODEL=qwen3.6:35b-a3b python -m gmail_cleanup classify ...`)
|
|
269
|
+
work without editing the file.
|
|
270
|
+
|
|
271
|
+
### Option B: LM Studio
|
|
272
|
+
|
|
273
|
+
```bash
|
|
274
|
+
# 1. Install LM Studio from https://lmstudio.ai, download a model
|
|
275
|
+
# inside it (Qwen 2.5 7B Instruct or Qwen 3 8B both work great),
|
|
276
|
+
# then start its local server (Developer tab → Start Server).
|
|
277
|
+
# Default URL: http://localhost:1234
|
|
278
|
+
|
|
279
|
+
# 2. Install this tool (same as above), with the openai extra
|
|
280
|
+
git clone https://github.com/brosequist/gmail-cleanup-agent
|
|
281
|
+
cd gmail-cleanup-agent
|
|
282
|
+
python -m venv .venv && source .venv/bin/activate
|
|
283
|
+
pip install -e '.[openai]' # base deps + openai SDK for the
|
|
284
|
+
# OpenAI-compatible client
|
|
285
|
+
|
|
286
|
+
# 3. Configure — LM Studio speaks the OpenAI wire format
|
|
287
|
+
cp config/backend.env.example config/backend.env
|
|
288
|
+
# edit config/backend.env: uncomment the "OpenAI-compatible" block and set
|
|
289
|
+
# OPENAI_BASE_URL=http://localhost:1234/v1
|
|
290
|
+
# OPENAI_API_KEY=not-needed # placeholder; LM Studio doesn't check
|
|
291
|
+
# OPENAI_MODEL=qwen2.5-7b-instruct # exact id from LM Studio's Server tab
|
|
292
|
+
```
|
|
293
|
+
|
|
294
|
+
The same `GCA_BACKEND=openai` setup also works with llama.cpp's
|
|
295
|
+
`server` binary, vLLM, and Ollama's `/v1` OpenAI shim — point
|
|
296
|
+
`OPENAI_BASE_URL` accordingly.
|
|
297
|
+
|
|
298
|
+
If you're using **llama.cpp's `llama-server`**, there are a few
|
|
299
|
+
sharp edges worth knowing about — sampling configs that work great
|
|
300
|
+
for chat actively break verbatim-reproduction tasks like this one,
|
|
301
|
+
reasoning models (Qwen3, DeepSeek-R1, ...) need
|
|
302
|
+
`OPENAI_DISABLE_THINKING=1`, and `--parallel N` on a single GPU
|
|
303
|
+
typically *slows things down*. See
|
|
304
|
+
[docs/llama-server-setup.md](docs/llama-server-setup.md) for the
|
|
305
|
+
full set of gotchas.
|
|
306
|
+
|
|
307
|
+
If your backend runs in **Kubernetes** or behind an **SSH tunnel**,
|
|
308
|
+
the CLI can launch the port-forward / tunnel command for you and
|
|
309
|
+
clean it up on exit — see the `PRE_RUN_COMMAND` block in
|
|
310
|
+
[`config/backend.env.example`](config/backend.env.example). Useful
|
|
311
|
+
to avoid the `kubectl port-forward -n ollama svc/ollama 11434:11434 &`
|
|
312
|
+
ritual before every run.
|
|
313
|
+
|
|
314
|
+
## Quick start — cloud LLM (faster, costs money)
|
|
315
|
+
|
|
316
|
+
```bash
|
|
317
|
+
# Install the SDK for whichever provider you're using
|
|
318
|
+
pip install -e '.[claude]' # for Claude
|
|
319
|
+
# or
|
|
320
|
+
pip install -e '.[openai]' # for real OpenAI
|
|
321
|
+
|
|
322
|
+
# Configure via the env file
|
|
323
|
+
cp config/backend.env.example config/backend.env
|
|
324
|
+
```
|
|
325
|
+
|
|
326
|
+
Then edit `config/backend.env` and uncomment the **Anthropic Claude**
|
|
327
|
+
section (set `ANTHROPIC_API_KEY` + `CLAUDE_MODEL`) **or** the
|
|
328
|
+
**OpenAI-compatible** section pointed at real OpenAI (set
|
|
329
|
+
`OPENAI_API_KEY` to your real key + `OPENAI_MODEL=gpt-4o-mini` or
|
|
330
|
+
`gpt-4.1-mini`). Leave `OPENAI_BASE_URL` at the default (or unset) to
|
|
331
|
+
hit `api.openai.com`.
|
|
332
|
+
|
|
333
|
+
## Set up OAuth + run
|
|
334
|
+
|
|
335
|
+
```bash
|
|
336
|
+
# 1. One-time: create Google OAuth credentials (see docs/oauth-setup.md)
|
|
337
|
+
# Save the JSON as config/credentials.json.
|
|
338
|
+
|
|
339
|
+
# 2. Configure your label catalog and classification rules
|
|
340
|
+
cp config/labels.example.yaml config/labels.yaml
|
|
341
|
+
cp config/rules.example.md config/rules.md
|
|
342
|
+
cp config/whitelist.example.txt config/whitelist.txt
|
|
343
|
+
# Edit each to match your priorities.
|
|
344
|
+
|
|
345
|
+
# 3. Authorize the tool against your Gmail account (browser pops once)
|
|
346
|
+
python -m gmail_cleanup auth
|
|
347
|
+
|
|
348
|
+
# 4. Dry-run on a small sample to sanity-check
|
|
349
|
+
python -m gmail_cleanup classify \
|
|
350
|
+
--query "older_than:90d -has:userlabels" \
|
|
351
|
+
--limit 100 --dry-run
|
|
352
|
+
# Review dry-run.log — every decision is recorded.
|
|
353
|
+
|
|
354
|
+
# 5. If it looks good, run for real
|
|
355
|
+
python -m gmail_cleanup classify \
|
|
356
|
+
--query "older_than:90d -has:userlabels" \
|
|
357
|
+
--apply
|
|
358
|
+
```
|
|
359
|
+
|
|
360
|
+
For a tee'd console log alongside the per-decision JSONL, pass
|
|
361
|
+
`--console-log dry-run.console.log` (handy for long runs you want
|
|
362
|
+
to inspect after the fact).
|
|
363
|
+
|
|
364
|
+
Every subcommand has built-in help — `python -m gmail_cleanup
|
|
365
|
+
--help` lists subcommands, and `python -m gmail_cleanup <subcommand>
|
|
366
|
+
--help` shows all options for that subcommand:
|
|
367
|
+
|
|
368
|
+
```bash
|
|
369
|
+
python -m gmail_cleanup --help # auth | classify | apply-log | relabel
|
|
370
|
+
python -m gmail_cleanup classify --help # --query, --apply, --dry-run, --concurrency, ...
|
|
371
|
+
```
|
|
372
|
+
|
|
373
|
+
### Recovering from a transient backend failure: `--retry-errors`
|
|
374
|
+
|
|
375
|
+
If a Ollama / llama.cpp / OpenAI hiccup mass-errored a batch of
|
|
376
|
+
threads, you'll see `action: "error"` rows in `dry-run.log`. Re-run
|
|
377
|
+
classify with `--retry-errors` to re-classify only those threads
|
|
378
|
+
(the resume set is the union of non-errored IDs):
|
|
379
|
+
|
|
380
|
+
```bash
|
|
381
|
+
python -m gmail_cleanup classify \
|
|
382
|
+
--query "older_than:90d -has:userlabels" \
|
|
383
|
+
--retry-errors --dry-run
|
|
384
|
+
```
|
|
385
|
+
|
|
386
|
+
The 312k-thread reference run finished with **0 final errors** after
|
|
387
|
+
a single retry pass over 809 errored threads — see
|
|
388
|
+
[docs/results.md](docs/results.md#retry-pass).
|
|
389
|
+
|
|
390
|
+
## Applying decisions from a dry-run: the `apply-log` pass
|
|
391
|
+
|
|
392
|
+
After a dry-run produces `dry-run.log`, you can replay its decisions
|
|
393
|
+
to Gmail without re-running the LLM. Useful when:
|
|
394
|
+
|
|
395
|
+
- You want to audit `dry-run.log` first, then commit the result
|
|
396
|
+
later (or on a different machine) — no need to keep the GPU around.
|
|
397
|
+
- Your apply session got interrupted; `state-applied.json`
|
|
398
|
+
remembers what's done and the next invocation picks up where it
|
|
399
|
+
left off.
|
|
400
|
+
- You re-ran classify on the same threads, the rules improved, and
|
|
401
|
+
you want to apply only the *latest* decision per thread.
|
|
402
|
+
|
|
403
|
+
```bash
|
|
404
|
+
# Preview what apply-log will do without touching Gmail
|
|
405
|
+
python -m gmail_cleanup apply-log --dry-run
|
|
406
|
+
|
|
407
|
+
# Actually apply (mutates Gmail; resumable via state-applied.json)
|
|
408
|
+
python -m gmail_cleanup apply-log --apply
|
|
409
|
+
|
|
410
|
+
# See every option
|
|
411
|
+
python -m gmail_cleanup apply-log --help
|
|
412
|
+
```
|
|
413
|
+
|
|
414
|
+
Key properties:
|
|
415
|
+
|
|
416
|
+
- **No LLM call.** It just reads the JSONL log and translates each
|
|
417
|
+
row into a Gmail batch HTTP request — trash for `action: "trash"`,
|
|
418
|
+
`threads.modify(addLabelIds=...)` for `action: "keep"` + label.
|
|
419
|
+
- **Latest decision wins.** Multiple log lines with the same thread
|
|
420
|
+
id collapse to the most recent one. Re-running classify and then
|
|
421
|
+
apply-log is safe.
|
|
422
|
+
- **Resumable** via `state-applied.json` — successful IDs are
|
|
423
|
+
checkpointed after every batch.
|
|
424
|
+
- **Robust to 429s** — Gmail's per-user concurrent ceiling
|
|
425
|
+
(~3.3 ops/sec) is hit easily; the subcommand retries 429ed
|
|
426
|
+
requests with exponential backoff within each batch.
|
|
427
|
+
|
|
428
|
+
## Reorganizing later: the `relabel` pass
|
|
429
|
+
|
|
430
|
+
After a big classification run you'll often want to *add* label
|
|
431
|
+
categories — you notice 1,000 emails landed in `Receipts` that are
|
|
432
|
+
really travel itineraries, so you add a `Travel` label. The `relabel`
|
|
433
|
+
subcommand reorganizes **already-kept** mail against your updated
|
|
434
|
+
`config/labels.yaml` without redoing any keep-vs-trash decisions:
|
|
435
|
+
|
|
436
|
+
```bash
|
|
437
|
+
# Add the new categories to config/labels.yaml first, then:
|
|
438
|
+
|
|
439
|
+
# Dry-run — proposes label changes, writes relabel.log, touches nothing
|
|
440
|
+
python -m gmail_cleanup relabel --input-log dry-run.log --dry-run
|
|
441
|
+
|
|
442
|
+
# Review relabel.log — each line shows old_label → new_label + changed flag
|
|
443
|
+
|
|
444
|
+
# Apply — moves the Gmail label for every email whose label changed
|
|
445
|
+
python -m gmail_cleanup relabel --input-log applied.log --apply
|
|
446
|
+
```
|
|
447
|
+
|
|
448
|
+
Key properties:
|
|
449
|
+
|
|
450
|
+
- **It cannot trash anything.** `relabel` only ever assigns a label.
|
|
451
|
+
The LLM is never asked to decide keep-vs-trash; emails that were
|
|
452
|
+
kept stay kept.
|
|
453
|
+
- **It reads from a decision log** (`dry-run.log` or `applied.log`),
|
|
454
|
+
taking only the `keep` rows. Trash and error rows are ignored.
|
|
455
|
+
- **On `--apply`** it does a single `threads.modify` per changed
|
|
456
|
+
email: add the new label, remove the old one. Emails whose label
|
|
457
|
+
didn't change get no API call.
|
|
458
|
+
- **Resumable** via its own `relabel-state.json` (separate from the
|
|
459
|
+
classify checkpoint), so Ctrl+C and re-run is safe.
|
|
460
|
+
- **`--refetch-snippets`** re-pulls each email's snippet from Gmail
|
|
461
|
+
for richer context — slower (one API call per email) but produces
|
|
462
|
+
noticeably better labels than sender + subject alone.
|
|
463
|
+
|
|
464
|
+
## Configuration
|
|
465
|
+
|
|
466
|
+
### `config/labels.yaml`
|
|
467
|
+
|
|
468
|
+
Maps Gmail labels to importance categories. The agent uses existing
|
|
469
|
+
labels first, then creates new ones from the `auto_create` list as
|
|
470
|
+
needed.
|
|
471
|
+
|
|
472
|
+
```yaml
|
|
473
|
+
existing:
|
|
474
|
+
- Taxes
|
|
475
|
+
- Filed
|
|
476
|
+
- Notes
|
|
477
|
+
auto_create:
|
|
478
|
+
Family: # personal correspondence
|
|
479
|
+
Receipts: # invoices, order confirmations
|
|
480
|
+
Medical: # healthcare providers
|
|
481
|
+
School: # educational institutions
|
|
482
|
+
Sports: # league registrations, schedules
|
|
483
|
+
Organizations: # community / professional groups
|
|
484
|
+
Government: # government offices, tax authorities
|
|
485
|
+
Registrations: # event / account / program confirmations
|
|
486
|
+
```
|
|
487
|
+
|
|
488
|
+
### `config/rules.md`
|
|
489
|
+
|
|
490
|
+
Free-form classification rules in plain English. The prompt builder
|
|
491
|
+
embeds it verbatim, so write to the LLM in plain language: *"emails
|
|
492
|
+
from any school or daycare are always keep"*, *"discount codes alone
|
|
493
|
+
are trash unless from a vendor I already buy from"*, etc.
|
|
494
|
+
|
|
495
|
+
### `config/whitelist.txt`
|
|
496
|
+
|
|
497
|
+
One sender per line. Anything matching is always kept; the LLM never
|
|
498
|
+
sees these and cannot override.
|
|
499
|
+
|
|
500
|
+
```
|
|
501
|
+
@yourdomain.com
|
|
502
|
+
mom@example.com
|
|
503
|
+
school.edu
|
|
504
|
+
```
|
|
505
|
+
|
|
506
|
+
## Safety
|
|
507
|
+
|
|
508
|
+
- **Trash, not permanent delete.** Gmail keeps trashed mail for 30
|
|
509
|
+
days; recovery is one click in the UI.
|
|
510
|
+
- **Sender whitelist always wins over LLM judgment.**
|
|
511
|
+
- **`--dry-run` is the default for `classify`.** You must explicitly
|
|
512
|
+
pass `--apply` for any state change.
|
|
513
|
+
- **Every run produces a `.log` file.** Loss-of-mail is auditable.
|
|
514
|
+
- **The script never modifies INBOX label state on already-labeled
|
|
515
|
+
threads** — your existing organization is preserved.
|
|
516
|
+
- **OAuth scope is `gmail.modify`** — enough to trash and label, not
|
|
517
|
+
enough to permanently delete or send mail.
|
|
518
|
+
|
|
519
|
+
## Privacy
|
|
520
|
+
|
|
521
|
+
See [docs/privacy.md](docs/privacy.md) for what data goes where.
|
|
522
|
+
|
|
523
|
+
Short version:
|
|
524
|
+
|
|
525
|
+
- **Local backends (Ollama, LM Studio):** no email content leaves
|
|
526
|
+
your machine. The script makes Gmail API calls (Google sees what
|
|
527
|
+
it always sees) and posts prompts to your local LLM server.
|
|
528
|
+
- **Cloud backends (Claude, OpenAI):** sender + subject + snippet
|
|
529
|
+
for each email are sent to the chosen API. No full bodies unless
|
|
530
|
+
you opt in with `--include-body`. No attachments. No headers
|
|
531
|
+
beyond From/Subject/Date.
|
|
532
|
+
|
|
533
|
+
## Model selection
|
|
534
|
+
|
|
535
|
+
For local backends, the script uses each provider's JSON-output mode
|
|
536
|
+
to make sure the LLM returns parseable decisions:
|
|
537
|
+
|
|
538
|
+
- Ollama: `format: "json"`
|
|
539
|
+
- OpenAI-compatible (LM Studio, llama.cpp, vLLM, OpenAI):
|
|
540
|
+
`response_format: {type: "json_object"}` (toggleable via
|
|
541
|
+
`OPENAI_JSON_MODE=0` if your server doesn't support it).
|
|
542
|
+
|
|
543
|
+
| Model | Backend | Quality | Speed (concurrency 4) |
|
|
544
|
+
|---|---|---|---|
|
|
545
|
+
| `qwen3.6:35b-a3b` (IQ3) | Ollama | excellent | ~80 emails/min on RX 9070 XT + RX 9060 XT |
|
|
546
|
+
| `qwen3:8b` | Ollama | good | ~120 emails/min on a 12 GB GPU |
|
|
547
|
+
| `qwen2.5-7b-instruct` | LM Studio | good | similar to above |
|
|
548
|
+
| `claude-haiku-4-5` | Claude | excellent | ~3,000–5,000 emails/min |
|
|
549
|
+
| `gpt-4o-mini` | OpenAI | good | ~2,000–4,000 emails/min |
|
|
550
|
+
|
|
551
|
+
Smaller models still work but produce more "missing decision" retries
|
|
552
|
+
and more `error`-marked rows in the audit log. Anything ≥7B with
|
|
553
|
+
strong JSON-output discipline should be fine.
|
|
554
|
+
|
|
555
|
+
## Architecture
|
|
556
|
+
|
|
557
|
+
```
|
|
558
|
+
gmail_cleanup/
|
|
559
|
+
├── __main__.py dotenv + pre-run hook + dispatch to cli.main
|
|
560
|
+
├── cli.py Click subcommands (auth, classify, relabel, apply-log)
|
|
561
|
+
├── gmail_client.py thin wrapper over googleapiclient with retry
|
|
562
|
+
├── prompt.py builds the LLM prompt from rules.md + labels.yaml
|
|
563
|
+
├── applylog.py apply-log subcommand (Gmail batch HTTP, 429 retry)
|
|
564
|
+
├── portforward.py optional PRE_RUN_COMMAND hook (kubectl / SSH tunnel)
|
|
565
|
+
└── backends/
|
|
566
|
+
├── ollama.py local Ollama server (HTTP JSON-mode)
|
|
567
|
+
├── openai.py OpenAI-compatible (LM Studio, llama.cpp, real OpenAI)
|
|
568
|
+
└── claude.py Anthropic Claude
|
|
569
|
+
```
|
|
570
|
+
|
|
571
|
+
The pipeline:
|
|
572
|
+
|
|
573
|
+
1. `gmail_client.search_threads()` paginates `threads.list` with the
|
|
574
|
+
user's query. Resume-mode skips IDs already in `state.json` before
|
|
575
|
+
making the per-thread metadata fetch.
|
|
576
|
+
2. For each new thread, `messages.get(format=metadata)` retrieves
|
|
577
|
+
From / Subject / Date.
|
|
578
|
+
3. Threads are batched (20 each), passed to `prompt.build_prompt()`,
|
|
579
|
+
and sent to the configured backend's `classify_batch()`.
|
|
580
|
+
4. The JSON response is validated; missing decisions are retried up
|
|
581
|
+
to 2× per batch.
|
|
582
|
+
5. Decisions are logged to `dry-run.log` (or `applied.log` with
|
|
583
|
+
`--apply`), and threads are trashed or labeled.
|
|
584
|
+
6. `state.json` is checkpointed after every batch.
|
|
585
|
+
|
|
586
|
+
A ThreadPoolExecutor runs `--concurrency` batches in parallel. The
|
|
587
|
+
LLM client is shared across workers via a connection pool.
|
|
588
|
+
|
|
589
|
+
## Limitations
|
|
590
|
+
|
|
591
|
+
- **Threads, not individual messages.** The unit of decision is the
|
|
592
|
+
thread. If a thread has a useful reply buried in promotional noise,
|
|
593
|
+
the snippet-based classifier may still call it trash. Use
|
|
594
|
+
`--include-body` for higher-stakes runs.
|
|
595
|
+
- **English-only rules out of the box.** The prompt is English; the
|
|
596
|
+
model handles multilingual senders fine but your `rules.md` should
|
|
597
|
+
be English unless your model is multilingual-tuned.
|
|
598
|
+
- **No spam-folder integration.** The script only looks at the
|
|
599
|
+
mailbox you point it at. It won't read or rescue from Spam.
|
|
600
|
+
- **Sequential thread enumeration.** Gmail's `threads.list` is
|
|
601
|
+
single-threaded; resume scans of huge state files still take a few
|
|
602
|
+
minutes of pure list-pagination time.
|
|
603
|
+
|
|
604
|
+
## License
|
|
605
|
+
|
|
606
|
+
MIT — see [LICENSE](LICENSE).
|