mail-parser 4.6.4__tar.gz → 4.7.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mail_parser-4.6.4/CLAUDE.md → mail_parser-4.7.0/AGENTS.md +109 -18
- mail_parser-4.7.0/CLAUDE.md +5 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/PKG-INFO +57 -1
- {mail_parser-4.6.4 → mail_parser-4.7.0}/README.md +56 -0
- mail_parser-4.7.0/src/mailparser/addresses.py +291 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/src/mailparser/core.py +42 -13
- {mail_parser-4.6.4 → mail_parser-4.7.0}/src/mailparser/utils.py +18 -153
- {mail_parser-4.6.4 → mail_parser-4.7.0}/src/mailparser/version.py +1 -1
- mail_parser-4.7.0/tests/test_address_recovery.py +289 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/uv.lock +827 -651
- {mail_parser-4.6.4 → mail_parser-4.7.0}/.claude/agents/security-reviewer.md +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/.github/FUNDING.yml +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/.github/ISSUE_TEMPLATE/bug_report.md +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/.github/ISSUE_TEMPLATE/feature_request.md +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/.github/copilot-instructions.md +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/.github/instructions/containerization-docker-best-practices.instructions.md +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/.github/instructions/github-actions-ci-cd-best-practices.instructions.md +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/.github/instructions/markdown.instructions.md +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/.github/instructions/python.instructions.md +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/.github/workflows/main.yml +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/.gitignore +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/.markdownlint.json +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/.pre-commit-config.yaml +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/Dockerfile +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/LICENSE.txt +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/Makefile +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/NOTICE.txt +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/docker-compose.yml +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/docs/images/Bitcoin SpamScope.jpg +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/pyproject.toml +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/src/mailparser/__init__.py +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/src/mailparser/__main__.py +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/src/mailparser/const.py +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/src/mailparser/exceptions.py +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_malformed_1 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_malformed_2 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_malformed_3 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_outlook_1 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_1 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_10 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_11 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_12 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_13 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_14 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_15 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_16 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_17 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_18 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_19 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_2 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_3 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_4 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_5 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_6 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_7 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_8 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_9 +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/test_improved_received_patterns.py +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/test_mail_parser.py +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/test_main.py +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/test_received_corpus.py +0 -0
- {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/test_utils.py +0 -0
|
@@ -1,6 +1,8 @@
|
|
|
1
|
-
#
|
|
1
|
+
# AGENTS.md
|
|
2
2
|
|
|
3
|
-
This file
|
|
3
|
+
This file is the shared source of repository instructions for Codex, Claude Code,
|
|
4
|
+
and other coding agents. It applies throughout this repository. Keep shared
|
|
5
|
+
instructions here; `CLAUDE.md` imports this file for Claude Code.
|
|
4
6
|
|
|
5
7
|
## Engineering Principles
|
|
6
8
|
|
|
@@ -13,6 +15,71 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co
|
|
|
13
15
|
- Every third-party import must have an explicit entry in `pyproject.toml`. Use extras when the
|
|
14
16
|
runtime feature requires them (e.g. `elasticsearch[async]`).
|
|
15
17
|
|
|
18
|
+
## RFC 5322 compliance and forensic recovery
|
|
19
|
+
|
|
20
|
+
For **every code change**, consult [RFC 5322](https://www.rfc-editor.org/info/rfc5322/)
|
|
21
|
+
and check the affected behavior against its applicable syntax and semantics.
|
|
22
|
+
Record the sections consulted and the compliance checks performed in the change
|
|
23
|
+
summary; if the change has no message-format impact, explicitly explain why.
|
|
24
|
+
Check applicable errata and updates linked by the RFC Editor, and related
|
|
25
|
+
standards when the change concerns SMTP trace fields or MIME.
|
|
26
|
+
|
|
27
|
+
Distinguish syntax permitted for newly generated messages (sections 2 and 3)
|
|
28
|
+
from obsolete syntax that receivers must still parse (section 4). Consult
|
|
29
|
+
section 5 for security considerations. Do not label an accepted obsolete form
|
|
30
|
+
invalid solely because it must not be generated in new messages.
|
|
31
|
+
|
|
32
|
+
This is a forensic parser: malformed input can deliberately exploit differences
|
|
33
|
+
between parsers to evade security tools. For nonconforming messages:
|
|
34
|
+
|
|
35
|
+
- Recover as much content as can be interpreted safely, preserving suspicious
|
|
36
|
+
display text, addresses, headers, and payload evidence. Do not silently discard
|
|
37
|
+
or normalize away data that could expose an evasion attempt.
|
|
38
|
+
- Surface violations through `has_defects` and the relevant structured defect
|
|
39
|
+
metadata. For address headers, preserve `address_header_defects` entries with
|
|
40
|
+
`reason`, `raw`, `recovered`, `candidates`, `header`, and `occurrence`.
|
|
41
|
+
Successful recovery must not erase the defect or imply RFC compliance.
|
|
42
|
+
- Keep ambiguous or incomplete evidence distinguishable from selected values.
|
|
43
|
+
Never promote an address found in display text or a comment over a complete
|
|
44
|
+
structural mailbox. Preserve occurrence context for repeated headers.
|
|
45
|
+
- Report observable defects and conflicting interpretations; malformed syntax
|
|
46
|
+
alone does not prove malicious intent.
|
|
47
|
+
- Keep existing resource, path, subprocess, and trust-boundary safeguards.
|
|
48
|
+
Recovery must not introduce crashes, unbounded amplification, unsafe side
|
|
49
|
+
effects, or fabricated attribution. Use controlled errors when safe recovery
|
|
50
|
+
is impossible; the fail-fast principle must not become blanket rejection of
|
|
51
|
+
recoverable malformed mail.
|
|
52
|
+
|
|
53
|
+
For parsing changes, cover conforming input, applicable obsolete syntax, and
|
|
54
|
+
malformed/adversarial variants with regression tests. Assert both recovered
|
|
55
|
+
content and defect metadata, including cases where no unambiguous recovery is
|
|
56
|
+
possible. Check that valid neighboring fields or list members remain intact.
|
|
57
|
+
|
|
58
|
+
Example of required recovery behavior (address syntax: sections 3.2.3, 3.2.5,
|
|
59
|
+
and 3.4). Preserve the misleading unquoted display name as evidence, select the
|
|
60
|
+
address inside angle brackets, and expose the invalid display name:
|
|
61
|
+
|
|
62
|
+
```python
|
|
63
|
+
import mailparser
|
|
64
|
+
|
|
65
|
+
mail = mailparser.parse_from_string(
|
|
66
|
+
"From: billing@trusted.example < billing@vendor.example >\r\n\r\n"
|
|
67
|
+
)
|
|
68
|
+
assert mail.from_ == [("billing@trusted.example", "billing@vendor.example")]
|
|
69
|
+
assert mail.has_defects is True
|
|
70
|
+
assert mail.address_header_defects == [{
|
|
71
|
+
"reason": "invalid-display-name",
|
|
72
|
+
"raw": "billing@trusted.example < billing@vendor.example >",
|
|
73
|
+
"recovered": True,
|
|
74
|
+
"candidates": [{
|
|
75
|
+
"display_name": "billing@trusted.example",
|
|
76
|
+
"address": "billing@vendor.example",
|
|
77
|
+
}],
|
|
78
|
+
"header": "from",
|
|
79
|
+
"occurrence": 0,
|
|
80
|
+
}]
|
|
81
|
+
```
|
|
82
|
+
|
|
16
83
|
## Python Style
|
|
17
84
|
|
|
18
85
|
- Follow PEP 8. Use 4 spaces, never tabs.
|
|
@@ -27,22 +94,26 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co
|
|
|
27
94
|
```bash
|
|
28
95
|
uv sync # install all dependencies
|
|
29
96
|
uv run pytest # run all tests
|
|
30
|
-
uv run pytest tests/test_mail_parser.py
|
|
97
|
+
uv run pytest tests/test_mail_parser.py -k test_valid_mail # single test
|
|
31
98
|
uv run ruff check . # lint
|
|
32
99
|
uv run ruff format . # format
|
|
100
|
+
uv run pre-commit run --files AGENTS.md CLAUDE.md README.md # selected files
|
|
101
|
+
make pre-commit # all tracked files
|
|
33
102
|
make check # lint + tests
|
|
34
103
|
make build # clean + build wheel/sdist
|
|
35
104
|
```
|
|
36
105
|
|
|
37
106
|
## Architecture
|
|
38
107
|
|
|
39
|
-
**Package layout**: `src/mailparser/` (src layout,
|
|
108
|
+
**Package layout**: `src/mailparser/` (src layout, Python 3.9+, stdlib-only core;
|
|
109
|
+
optional Outlook dependencies are installed with `uv sync --extra outlook`).
|
|
40
110
|
|
|
41
111
|
### Module responsibilities
|
|
42
112
|
|
|
43
113
|
| Module | Role |
|
|
44
114
|
| --------------- | --------------------------------------------------------------------- |
|
|
45
115
|
| `core.py` | `MailParser` class + `parse_from_*` factory functions |
|
|
116
|
+
| `addresses.py` | Forensic address scanner and recovery diagnostics |
|
|
46
117
|
| `utils.py` | Stateless parsing helpers (address, received headers, date, encoding) |
|
|
47
118
|
| `const.py` | Compiled regexes, header sets, clause splitter |
|
|
48
119
|
| `exceptions.py` | Custom exception hierarchy |
|
|
@@ -80,11 +151,17 @@ parse. `Message.get_all()` is a linear scan, so the one call per distinct name p
|
|
|
80
151
|
`_make_mail()` cost O(distinct × total) on attacker-chosen names. Look up via the index inside any
|
|
81
152
|
loop over header names.
|
|
82
153
|
|
|
83
|
-
**
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
154
|
+
**Forensic address parsing**: `addresses.py` scans structural delimiters in a
|
|
155
|
+
single pass, respecting quotes, escapes, nested comments and domain literals.
|
|
156
|
+
`utils.get_addresses()` adapts input and delegates to it. Complete angle-addrs
|
|
157
|
+
have precedence over display text; incomplete/ambiguous items are evidence,
|
|
158
|
+
not selected mailboxes. Never search display names/comments for bare addresses
|
|
159
|
+
or reintroduce the old overlapping fallback regex (ReDoS). Parsing a malformed
|
|
160
|
+
list member must not change the interpretation of its neighbours. `core.py`
|
|
161
|
+
caches address results and recovery diagnostics once per parse, examining every
|
|
162
|
+
header occurrence while retaining first-occurrence semantics for address
|
|
163
|
+
properties. `address_header_defects` is reserved computed metadata; literal
|
|
164
|
+
wire headers with that name remain accessible through `headers`/`message`.
|
|
88
165
|
|
|
89
166
|
**Received header parsing**: `utils.receiveds_parsing()` tokenizes on RFC 5321 clause keywords
|
|
90
167
|
(`from`, `by`, `via`, `with`, `id`, `for`, `envelope-from`) using `const._CLAUSE_SPLITTER`.
|
|
@@ -116,8 +193,10 @@ cases — three rounds of that each reopened a hole the previous one closed.
|
|
|
116
193
|
RFC violations. `EPILOGUE_DEFECTS` triggers special epilogue extraction to recover hidden payloads
|
|
117
194
|
in malformed boundaries.
|
|
118
195
|
|
|
119
|
-
**Outlook support**: `parse_from_file_msg()`
|
|
120
|
-
(`
|
|
196
|
+
**Outlook support**: `parse_from_file_msg()` prefers the optional `extract-msg`
|
|
197
|
+
Python backend (`uv sync --extra outlook`). When unavailable, it falls back to
|
|
198
|
+
the deprecated system `msgconvert` Perl tool (`libemail-outlook-message-perl`)
|
|
199
|
+
to convert `.msg` → `.eml`, then parses normally.
|
|
121
200
|
|
|
122
201
|
**Partial vs full mail**: `_make_mail(complete=True)` includes all headers found in the message;
|
|
123
202
|
`complete=False` restricts to `const.ADDRESSES_HEADERS | const.OTHERS_PARTS` (the "main" headers).
|
|
@@ -136,12 +215,24 @@ attribute access.
|
|
|
136
215
|
|
|
137
216
|
After every change:
|
|
138
217
|
|
|
139
|
-
1.
|
|
218
|
+
1. For code changes, perform the RFC 5322 review described above before
|
|
219
|
+
implementation and verify the resulting behavior against it before reporting
|
|
220
|
+
done. Include the RFC sections, recovery/defect checks, and any remaining
|
|
221
|
+
compliance gaps in the change summary.
|
|
222
|
+
1. Add/update unittests for behavior changes.
|
|
140
223
|
1. Update README.md if the change affects usage, API, or setup.
|
|
141
|
-
1.
|
|
224
|
+
1. Run `uv run pre-commit run --files <changed-files>`; include new files and
|
|
225
|
+
fix all reported issues before proceeding. Use `make pre-commit` for all
|
|
226
|
+
tracked files.
|
|
142
227
|
1. Run full test suite; fix all failures before reporting done.
|
|
143
|
-
1. Run a security review of the change
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
228
|
+
1. Run a security review of the change using the procedure in
|
|
229
|
+
[`.claude/agents/security-reviewer.md`](.claude/agents/security-reviewer.md).
|
|
230
|
+
In Claude Code, use the `security-reviewer` sub-agent. In Codex or another
|
|
231
|
+
agent, give the Markdown body of that file to a review sub-agent when
|
|
232
|
+
available, or follow it directly when delegation is unavailable. The YAML
|
|
233
|
+
front matter configures Claude Code only; it does not select tools or models
|
|
234
|
+
in Codex. All parsed input is attacker-controlled, so any change to parsing,
|
|
235
|
+
regexes, subprocess, temp files, or attachment handling must be reviewed.
|
|
236
|
+
Reproduce and fix every High/Medium finding (with a regression test) before
|
|
237
|
+
reporting done. For documentation-only changes, check that security
|
|
238
|
+
invariants and review requirements remain intact.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: mail-parser
|
|
3
|
-
Version: 4.
|
|
3
|
+
Version: 4.7.0
|
|
4
4
|
Summary: A tool that parses emails by enhancing the Python standard library, extracting all details into a comprehensive object.
|
|
5
5
|
Project-URL: Homepage, https://github.com/SpamScope/mail-parser
|
|
6
6
|
Project-URL: Source, https://github.com/SpamScope/mail-parser
|
|
@@ -298,6 +298,49 @@ mail.to_raw # Original "To:" header string as it appears in the email
|
|
|
298
298
|
The command-line tool outputs parsed emails in JSON format by default for easy integration with
|
|
299
299
|
other tools and pipelines.
|
|
300
300
|
|
|
301
|
+
### Address recovery and ambiguous headers
|
|
302
|
+
|
|
303
|
+
Address properties such as `from_`, `to`, and `reply_to` retain their list of
|
|
304
|
+
`(display_name, address)` tuples. Mailbox selection respects quoted strings,
|
|
305
|
+
escapes, nested comments, groups, and folding whitespace. For example:
|
|
306
|
+
|
|
307
|
+
```python
|
|
308
|
+
mail = mailparser.parse_from_string(
|
|
309
|
+
"From: billing@trusted.example < billing@vendor.example >\r\n\r\n"
|
|
310
|
+
)
|
|
311
|
+
mail.from_
|
|
312
|
+
# [('billing@trusted.example', 'billing@vendor.example')]
|
|
313
|
+
mail.has_defects
|
|
314
|
+
# True
|
|
315
|
+
mail.address_header_defects[0]["reason"]
|
|
316
|
+
# 'invalid-display-name'
|
|
317
|
+
```
|
|
318
|
+
|
|
319
|
+
RFC 5322 permits whitespace around the mailbox inside `<...>`; the unquoted
|
|
320
|
+
`@` in this example's display name requires forensic recovery. A complete,
|
|
321
|
+
unambiguous angle-address takes precedence over its display name. Addresses
|
|
322
|
+
inside comments or quoted display names are never mailbox candidates.
|
|
323
|
+
|
|
324
|
+
`address_header_defects` exposes recovery and ambiguity evidence. Each entry
|
|
325
|
+
contains `header`, zero-based `occurrence`, the original `raw` list item,
|
|
326
|
+
`reason`, `recovered`, and `candidates` (display-name/address dictionaries).
|
|
327
|
+
It is included in `mail`, `mail_partial`, and their JSON output when nonempty,
|
|
328
|
+
and also sets `has_defects` and the `AddressHeaderDefect` category. Diagnostics
|
|
329
|
+
are computed once per parse, including repeated header occurrences; address
|
|
330
|
+
properties retain their existing first-occurrence semantics.
|
|
331
|
+
|
|
332
|
+
Ambiguous or incomplete items are omitted from address tuples, with their
|
|
333
|
+
raw evidence and any structurally identified candidates retained in diagnostics.
|
|
334
|
+
Valid neighbouring items are parsed independently. Consumers making filtering
|
|
335
|
+
or attribution decisions should inspect these diagnostics: an empty address
|
|
336
|
+
list does not establish that the original header contained no addresses.
|
|
337
|
+
`structural-recovery` means the stdlib's interpretation could not be used; it
|
|
338
|
+
is not by itself proof of an RFC violation. This recovery policy is not a full
|
|
339
|
+
RFC validator or a guarantee of how every mail client displays malformed mail.
|
|
340
|
+
The complete original header values remain available through `from_raw`,
|
|
341
|
+
`to_raw`, etc.; literal headers colliding with computed metadata remain in
|
|
342
|
+
`headers` and `message.get_all(...)`.
|
|
343
|
+
|
|
301
344
|
## Defects and Their Critical Role in Email Security
|
|
302
345
|
|
|
303
346
|
Email structural defects are not merely technical curiosities—they represent **potential security
|
|
@@ -574,3 +617,16 @@ The default configuration includes:
|
|
|
574
617
|
|
|
575
618
|
Customize the `docker-compose.yml` file to adjust mount points, command-line options, or
|
|
576
619
|
environment variables for your specific use case.
|
|
620
|
+
|
|
621
|
+
# Working with coding agents
|
|
622
|
+
|
|
623
|
+
[AGENTS.md](AGENTS.md) is the shared source of development commands, coding
|
|
624
|
+
conventions, architecture notes, and security review requirements.
|
|
625
|
+
[Codex reads it automatically](https://developers.openai.com/codex/guides/agents-md).
|
|
626
|
+
[CLAUDE.md](CLAUDE.md) imports the same file using
|
|
627
|
+
[Claude Code's import syntax](https://code.claude.com/docs/en/memory#import-additional-files),
|
|
628
|
+
so updates to shared guidance belong in `AGENTS.md`.
|
|
629
|
+
|
|
630
|
+
The existing [security reviewer](.claude/agents/security-reviewer.md) remains
|
|
631
|
+
available as a Claude Code sub-agent. Codex follows the same review procedure
|
|
632
|
+
as described in `AGENTS.md`.
|
|
@@ -266,6 +266,49 @@ mail.to_raw # Original "To:" header string as it appears in the email
|
|
|
266
266
|
The command-line tool outputs parsed emails in JSON format by default for easy integration with
|
|
267
267
|
other tools and pipelines.
|
|
268
268
|
|
|
269
|
+
### Address recovery and ambiguous headers
|
|
270
|
+
|
|
271
|
+
Address properties such as `from_`, `to`, and `reply_to` retain their list of
|
|
272
|
+
`(display_name, address)` tuples. Mailbox selection respects quoted strings,
|
|
273
|
+
escapes, nested comments, groups, and folding whitespace. For example:
|
|
274
|
+
|
|
275
|
+
```python
|
|
276
|
+
mail = mailparser.parse_from_string(
|
|
277
|
+
"From: billing@trusted.example < billing@vendor.example >\r\n\r\n"
|
|
278
|
+
)
|
|
279
|
+
mail.from_
|
|
280
|
+
# [('billing@trusted.example', 'billing@vendor.example')]
|
|
281
|
+
mail.has_defects
|
|
282
|
+
# True
|
|
283
|
+
mail.address_header_defects[0]["reason"]
|
|
284
|
+
# 'invalid-display-name'
|
|
285
|
+
```
|
|
286
|
+
|
|
287
|
+
RFC 5322 permits whitespace around the mailbox inside `<...>`; the unquoted
|
|
288
|
+
`@` in this example's display name requires forensic recovery. A complete,
|
|
289
|
+
unambiguous angle-address takes precedence over its display name. Addresses
|
|
290
|
+
inside comments or quoted display names are never mailbox candidates.
|
|
291
|
+
|
|
292
|
+
`address_header_defects` exposes recovery and ambiguity evidence. Each entry
|
|
293
|
+
contains `header`, zero-based `occurrence`, the original `raw` list item,
|
|
294
|
+
`reason`, `recovered`, and `candidates` (display-name/address dictionaries).
|
|
295
|
+
It is included in `mail`, `mail_partial`, and their JSON output when nonempty,
|
|
296
|
+
and also sets `has_defects` and the `AddressHeaderDefect` category. Diagnostics
|
|
297
|
+
are computed once per parse, including repeated header occurrences; address
|
|
298
|
+
properties retain their existing first-occurrence semantics.
|
|
299
|
+
|
|
300
|
+
Ambiguous or incomplete items are omitted from address tuples, with their
|
|
301
|
+
raw evidence and any structurally identified candidates retained in diagnostics.
|
|
302
|
+
Valid neighbouring items are parsed independently. Consumers making filtering
|
|
303
|
+
or attribution decisions should inspect these diagnostics: an empty address
|
|
304
|
+
list does not establish that the original header contained no addresses.
|
|
305
|
+
`structural-recovery` means the stdlib's interpretation could not be used; it
|
|
306
|
+
is not by itself proof of an RFC violation. This recovery policy is not a full
|
|
307
|
+
RFC validator or a guarantee of how every mail client displays malformed mail.
|
|
308
|
+
The complete original header values remain available through `from_raw`,
|
|
309
|
+
`to_raw`, etc.; literal headers colliding with computed metadata remain in
|
|
310
|
+
`headers` and `message.get_all(...)`.
|
|
311
|
+
|
|
269
312
|
## Defects and Their Critical Role in Email Security
|
|
270
313
|
|
|
271
314
|
Email structural defects are not merely technical curiosities—they represent **potential security
|
|
@@ -542,3 +585,16 @@ The default configuration includes:
|
|
|
542
585
|
|
|
543
586
|
Customize the `docker-compose.yml` file to adjust mount points, command-line options, or
|
|
544
587
|
environment variables for your specific use case.
|
|
588
|
+
|
|
589
|
+
# Working with coding agents
|
|
590
|
+
|
|
591
|
+
[AGENTS.md](AGENTS.md) is the shared source of development commands, coding
|
|
592
|
+
conventions, architecture notes, and security review requirements.
|
|
593
|
+
[Codex reads it automatically](https://developers.openai.com/codex/guides/agents-md).
|
|
594
|
+
[CLAUDE.md](CLAUDE.md) imports the same file using
|
|
595
|
+
[Claude Code's import syntax](https://code.claude.com/docs/en/memory#import-additional-files),
|
|
596
|
+
so updates to shared guidance belong in `AGENTS.md`.
|
|
597
|
+
|
|
598
|
+
The existing [security reviewer](.claude/agents/security-reviewer.md) remains
|
|
599
|
+
available as a Claude Code sub-agent. Codex follows the same review procedure
|
|
600
|
+
as described in `AGENTS.md`.
|
|
@@ -0,0 +1,291 @@
|
|
|
1
|
+
"""Structural address recovery for untrusted RFC 5322 header values.
|
|
2
|
+
|
|
3
|
+
The scanner owns lexical boundaries and checks complete addr-specs against
|
|
4
|
+
bounded grammar patterns. The stdlib supplies conventional display names for
|
|
5
|
+
structurally consistent items. Recovery never searches a display name or
|
|
6
|
+
comment for a convenient email substring.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import re
|
|
12
|
+
from dataclasses import dataclass, field
|
|
13
|
+
|
|
14
|
+
_FOLD = re.compile(r"\r?\n[ \t]+")
|
|
15
|
+
# Anchored grammar checks: atom, quote and delimiter character sets do not
|
|
16
|
+
# overlap. In particular, do not search for bare addresses in arbitrary text
|
|
17
|
+
# or hand long malformed word sequences to headerregistry's quadratic parser.
|
|
18
|
+
_ATOM = r"[a-zA-Z0-9!#$%&'*+/=?^_`{|}~-]+"
|
|
19
|
+
_QUOTED = r'"(?:[^"\\\r\n]|\\[^\r\n])*"'
|
|
20
|
+
_WORD = rf"(?:{_ATOM}|{_QUOTED})"
|
|
21
|
+
_LOCAL = rf"{_WORD}(?:[ \t]*\.[ \t]*{_WORD})*"
|
|
22
|
+
_DOMAIN = rf"{_ATOM}(?:[ \t]*\.[ \t]*{_ATOM})*"
|
|
23
|
+
_LITERAL = r"\[(?:[^\[\]\\\r\n]|\\[^\r\n])*\]"
|
|
24
|
+
_ADDR_SPEC = re.compile(rf"{_LOCAL}[ \t]*@[ \t]*(?:{_DOMAIN}|{_LITERAL})")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass
|
|
28
|
+
class _Item:
|
|
29
|
+
raw: str
|
|
30
|
+
text: str
|
|
31
|
+
angles: list[tuple[int, int]]
|
|
32
|
+
problems: list[str] = field(default_factory=list)
|
|
33
|
+
comment_depth: int = 0
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _scan(raw: str) -> list[_Item]:
|
|
37
|
+
"""Split only at structural delimiters; bound work per input character."""
|
|
38
|
+
items: list[_Item] = []
|
|
39
|
+
buf: list[str] = []
|
|
40
|
+
angles: list[tuple[int, int]] = []
|
|
41
|
+
problems: list[str] = []
|
|
42
|
+
start = 0
|
|
43
|
+
angle = None
|
|
44
|
+
comment = max_comment = 0
|
|
45
|
+
quote = literal = escaped = group = False
|
|
46
|
+
|
|
47
|
+
def emit(end):
|
|
48
|
+
nonlocal buf, angles, problems, start, max_comment
|
|
49
|
+
text = "".join(buf)
|
|
50
|
+
if text.strip() or problems:
|
|
51
|
+
items.append(_Item(raw[start:end], text, angles, problems, max_comment))
|
|
52
|
+
buf, angles, problems = [], [], []
|
|
53
|
+
start = end + 1
|
|
54
|
+
max_comment = 0
|
|
55
|
+
|
|
56
|
+
for i, ch in enumerate(raw):
|
|
57
|
+
if escaped:
|
|
58
|
+
if not comment:
|
|
59
|
+
buf.append(ch)
|
|
60
|
+
escaped = False
|
|
61
|
+
continue
|
|
62
|
+
if ch == "\\" and (quote or comment or literal):
|
|
63
|
+
escaped = True
|
|
64
|
+
if not comment:
|
|
65
|
+
buf.append(ch)
|
|
66
|
+
continue
|
|
67
|
+
if comment:
|
|
68
|
+
if ch == "(":
|
|
69
|
+
comment += 1
|
|
70
|
+
max_comment = max(max_comment, comment)
|
|
71
|
+
elif ch == ")":
|
|
72
|
+
comment -= 1
|
|
73
|
+
continue
|
|
74
|
+
if quote:
|
|
75
|
+
buf.append(ch)
|
|
76
|
+
if ch == '"':
|
|
77
|
+
quote = False
|
|
78
|
+
continue
|
|
79
|
+
if literal:
|
|
80
|
+
buf.append(ch)
|
|
81
|
+
if ch == "]":
|
|
82
|
+
literal = False
|
|
83
|
+
continue
|
|
84
|
+
if ch == "(":
|
|
85
|
+
comment = 1
|
|
86
|
+
max_comment = max(max_comment, 1)
|
|
87
|
+
buf.append(" ")
|
|
88
|
+
elif ch == '"':
|
|
89
|
+
quote = True
|
|
90
|
+
buf.append(ch)
|
|
91
|
+
elif ch == "[":
|
|
92
|
+
literal = True
|
|
93
|
+
buf.append(ch)
|
|
94
|
+
elif ch == "<":
|
|
95
|
+
if angle is not None:
|
|
96
|
+
problems.append("nested-angle")
|
|
97
|
+
else:
|
|
98
|
+
angle = len(buf)
|
|
99
|
+
buf.append(ch)
|
|
100
|
+
elif ch == ">":
|
|
101
|
+
if angle is None:
|
|
102
|
+
problems.append("unmatched-angle")
|
|
103
|
+
else:
|
|
104
|
+
angles.append((angle, len(buf)))
|
|
105
|
+
angle = None
|
|
106
|
+
buf.append(ch)
|
|
107
|
+
elif angle is None and ch == ",":
|
|
108
|
+
emit(i)
|
|
109
|
+
elif angle is None and ch == ":":
|
|
110
|
+
if group or angles:
|
|
111
|
+
problems.append("invalid-group-boundary")
|
|
112
|
+
buf.append(ch)
|
|
113
|
+
else:
|
|
114
|
+
group = True
|
|
115
|
+
# Preserve malformed labels as diagnostics, never mailboxes.
|
|
116
|
+
if not _valid_phrase("".join(buf)):
|
|
117
|
+
problems.append("invalid-group-name")
|
|
118
|
+
emit(i)
|
|
119
|
+
buf = []
|
|
120
|
+
start = i + 1
|
|
121
|
+
elif angle is None and ch == ";":
|
|
122
|
+
if not group:
|
|
123
|
+
problems.append("unmatched-group")
|
|
124
|
+
emit(i)
|
|
125
|
+
group = False
|
|
126
|
+
else:
|
|
127
|
+
if ch in ")]":
|
|
128
|
+
problems.append("unmatched-delimiter")
|
|
129
|
+
buf.append(ch)
|
|
130
|
+
|
|
131
|
+
if quote or comment or literal or angle is not None or escaped:
|
|
132
|
+
problems.append("unclosed-delimiter")
|
|
133
|
+
if group:
|
|
134
|
+
problems.append("unclosed-group")
|
|
135
|
+
emit(len(raw))
|
|
136
|
+
return items
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _mailbox(value: str) -> str | None:
|
|
140
|
+
"""Validate the whole addr-spec, including quoted local parts and CFWS."""
|
|
141
|
+
value = _FOLD.sub(" ", value).strip()
|
|
142
|
+
# RFC 5322 section 4.4: obsolete source routes precede the addr-spec.
|
|
143
|
+
if value.startswith(("@", ",")):
|
|
144
|
+
literal = escaped = False
|
|
145
|
+
domains = []
|
|
146
|
+
start = 0
|
|
147
|
+
for i, ch in enumerate(value):
|
|
148
|
+
if escaped:
|
|
149
|
+
escaped = False
|
|
150
|
+
elif ch == "\\" and literal:
|
|
151
|
+
escaped = True
|
|
152
|
+
elif ch == "[":
|
|
153
|
+
literal = True
|
|
154
|
+
elif ch == "]":
|
|
155
|
+
literal = False
|
|
156
|
+
elif not literal and ch in ",:":
|
|
157
|
+
domain = value[start:i].strip()
|
|
158
|
+
if domain:
|
|
159
|
+
domains.append(domain)
|
|
160
|
+
start = i + 1
|
|
161
|
+
if ch == ":":
|
|
162
|
+
break
|
|
163
|
+
else:
|
|
164
|
+
return None
|
|
165
|
+
if not domains or any(
|
|
166
|
+
not domain.startswith("@") or not _ADDR_SPEC.fullmatch("route" + domain)
|
|
167
|
+
for domain in domains
|
|
168
|
+
):
|
|
169
|
+
return None
|
|
170
|
+
value = value[start:].strip()
|
|
171
|
+
if not _ADDR_SPEC.fullmatch(value):
|
|
172
|
+
return None
|
|
173
|
+
# CFWS is not part of an atom. Retain spaces and escapes inside quoted
|
|
174
|
+
# local parts and domain literals, but remove token-separating FWS.
|
|
175
|
+
normalized = []
|
|
176
|
+
quoted = literal = escaped = False
|
|
177
|
+
for ch in value:
|
|
178
|
+
if escaped:
|
|
179
|
+
escaped = False
|
|
180
|
+
elif ch == "\\" and (quoted or literal):
|
|
181
|
+
escaped = True
|
|
182
|
+
elif ch == '"' and not literal:
|
|
183
|
+
quoted = not quoted
|
|
184
|
+
elif ch == "[" and not quoted:
|
|
185
|
+
literal = True
|
|
186
|
+
elif ch == "]" and literal:
|
|
187
|
+
literal = False
|
|
188
|
+
if ch not in " \t" or quoted or literal:
|
|
189
|
+
normalized.append(ch)
|
|
190
|
+
return "".join(normalized)
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _valid_phrase(value: str) -> bool:
|
|
194
|
+
"""Check label specials outside quoted strings (period is obs-phrase)."""
|
|
195
|
+
quoted = escaped = False
|
|
196
|
+
for ch in _FOLD.sub(" ", value).strip():
|
|
197
|
+
if ord(ch) < 32 and ch != "\t":
|
|
198
|
+
return False
|
|
199
|
+
if escaped:
|
|
200
|
+
escaped = False
|
|
201
|
+
elif ch == "\\" and quoted:
|
|
202
|
+
escaped = True
|
|
203
|
+
elif ch == '"':
|
|
204
|
+
quoted = not quoted
|
|
205
|
+
elif not quoted and ch in "@<>[]:;,\\":
|
|
206
|
+
return False
|
|
207
|
+
return bool(value.strip()) and not (quoted or escaped)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def _name(value: str) -> str:
|
|
211
|
+
"""Recover a display phrase, unquoting quoted pairs without regexes."""
|
|
212
|
+
result = []
|
|
213
|
+
quoted = escaped = False
|
|
214
|
+
for ch in _FOLD.sub(" ", value).strip():
|
|
215
|
+
if escaped:
|
|
216
|
+
result.append(ch)
|
|
217
|
+
escaped = False
|
|
218
|
+
elif ch == "\\" and quoted:
|
|
219
|
+
escaped = True
|
|
220
|
+
elif ch == '"':
|
|
221
|
+
quoted = not quoted
|
|
222
|
+
else:
|
|
223
|
+
result.append(ch)
|
|
224
|
+
return "".join(result).strip()
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def parse_address_header(raw: str, strict_parser):
|
|
228
|
+
"""Return mailbox tuples and evidence for recovery or ambiguity.
|
|
229
|
+
|
|
230
|
+
``strict_parser`` is the runtime-compatible email.utils adapter. Each
|
|
231
|
+
list member is parsed independently so a malformed member cannot change
|
|
232
|
+
the interpretation of a neighbouring, valid quoted name.
|
|
233
|
+
"""
|
|
234
|
+
results = []
|
|
235
|
+
diagnostics = []
|
|
236
|
+
for item in _scan(raw):
|
|
237
|
+
candidates = []
|
|
238
|
+
# Derive the prefix once: repeating it for every angle-addr would
|
|
239
|
+
# copy quadratically much text on a long ambiguous item.
|
|
240
|
+
name = _name(item.text[: item.angles[0][0]]) if item.angles else ""
|
|
241
|
+
for index, (left, right) in enumerate(item.angles):
|
|
242
|
+
address = _mailbox(item.text[left + 1 : right])
|
|
243
|
+
if address:
|
|
244
|
+
candidates.append((name if index == 0 else "", address))
|
|
245
|
+
reason = None
|
|
246
|
+
selected = []
|
|
247
|
+
if item.problems:
|
|
248
|
+
reason = ",".join(sorted(set(item.problems)))
|
|
249
|
+
elif item.angles:
|
|
250
|
+
left, right = item.angles[0]
|
|
251
|
+
if len(item.angles) != 1 or item.text[right + 1 :].strip():
|
|
252
|
+
reason = "ambiguous-angle-address"
|
|
253
|
+
elif not candidates:
|
|
254
|
+
reason = "invalid-angle-address"
|
|
255
|
+
else:
|
|
256
|
+
selected = candidates
|
|
257
|
+
if item.text[:left].strip() and not _valid_phrase(item.text[:left]):
|
|
258
|
+
reason = "invalid-display-name"
|
|
259
|
+
else:
|
|
260
|
+
address = _mailbox(item.text)
|
|
261
|
+
if address:
|
|
262
|
+
selected = [("", address)]
|
|
263
|
+
candidates = selected
|
|
264
|
+
else:
|
|
265
|
+
reason = "invalid-or-ambiguous-address"
|
|
266
|
+
|
|
267
|
+
if selected:
|
|
268
|
+
# Deep comments are stripped by the iterative scanner before
|
|
269
|
+
# reaching the stdlib's recursive comment parser.
|
|
270
|
+
source = item.raw if item.comment_depth < 16 else item.text
|
|
271
|
+
parsed = strict_parser([source])
|
|
272
|
+
if [addr for _, addr in parsed if addr] == [a for _, a in selected]:
|
|
273
|
+
selected = [(name, addr) for name, addr in parsed if addr]
|
|
274
|
+
else:
|
|
275
|
+
# Parsing can reject legal CFWS too. This records recovery,
|
|
276
|
+
# not a claim that every rejection proves RFC noncompliance.
|
|
277
|
+
reason = reason or "structural-recovery"
|
|
278
|
+
results.extend(selected)
|
|
279
|
+
if reason:
|
|
280
|
+
diagnostics.append(
|
|
281
|
+
{
|
|
282
|
+
"reason": reason,
|
|
283
|
+
"raw": item.raw,
|
|
284
|
+
"recovered": bool(selected),
|
|
285
|
+
"candidates": [
|
|
286
|
+
{"display_name": name, "address": addr}
|
|
287
|
+
for name, addr in candidates
|
|
288
|
+
],
|
|
289
|
+
}
|
|
290
|
+
)
|
|
291
|
+
return results, diagnostics
|