mail-parser 4.6.0__tar.gz → 4.6.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mail_parser-4.6.2/.claude/agents/security-reviewer.md +161 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/.github/workflows/main.yml +1 -1
- {mail_parser-4.6.0 → mail_parser-4.6.2}/CLAUDE.md +50 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/PKG-INFO +16 -3
- {mail_parser-4.6.0 → mail_parser-4.6.2}/README.md +15 -2
- {mail_parser-4.6.0 → mail_parser-4.6.2}/pyproject.toml +3 -1
- mail_parser-4.6.2/src/mailparser/const.py +167 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/src/mailparser/core.py +346 -80
- {mail_parser-4.6.0 → mail_parser-4.6.2}/src/mailparser/exceptions.py +15 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/src/mailparser/utils.py +240 -23
- {mail_parser-4.6.0 → mail_parser-4.6.2}/src/mailparser/version.py +1 -1
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/test_mail_parser.py +679 -18
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/test_utils.py +119 -4
- mail_parser-4.6.0/src/mailparser/const.py +0 -101
- {mail_parser-4.6.0 → mail_parser-4.6.2}/.github/FUNDING.yml +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/.github/ISSUE_TEMPLATE/bug_report.md +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/.github/ISSUE_TEMPLATE/feature_request.md +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/.github/copilot-instructions.md +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/.github/instructions/containerization-docker-best-practices.instructions.md +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/.github/instructions/github-actions-ci-cd-best-practices.instructions.md +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/.github/instructions/markdown.instructions.md +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/.github/instructions/python.instructions.md +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/.gitignore +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/.markdownlint.json +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/.pre-commit-config.yaml +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/Dockerfile +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/LICENSE.txt +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/Makefile +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/NOTICE.txt +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/docker-compose.yml +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/docs/images/Bitcoin SpamScope.jpg +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/src/mailparser/__init__.py +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/src/mailparser/__main__.py +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_malformed_1 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_malformed_2 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_malformed_3 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_outlook_1 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_1 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_10 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_11 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_12 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_13 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_14 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_15 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_16 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_17 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_18 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_19 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_2 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_3 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_4 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_5 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_6 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_7 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_8 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_9 +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/test_improved_received_patterns.py +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/test_main.py +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/test_received_corpus.py +0 -0
- {mail_parser-4.6.0 → mail_parser-4.6.2}/uv.lock +0 -0
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: security-reviewer
|
|
3
|
+
description: >
|
|
4
|
+
Security auditor for mail-parser and similar untrusted-input parsers. Reviews
|
|
5
|
+
code for real, reachable vulnerabilities — ReDoS, injection, path traversal,
|
|
6
|
+
unsafe subprocess/deserialization, MIME-structure and decode DoS,
|
|
7
|
+
parser-differential evasion, info leak — and also checks the safety invariants
|
|
8
|
+
(no network/XML/archive-extraction) still hold. PROVES each finding with a PoC
|
|
9
|
+
or measured blowup before reporting it. Use for
|
|
10
|
+
"security review", "audit for vulnerabilities", "check for security issues",
|
|
11
|
+
"is this exploitable". Threat model: input (email bytes/headers/attachments)
|
|
12
|
+
is fully attacker-controlled; the caller and filesystem are trusted.
|
|
13
|
+
tools: [Read, Grep, Glob, Bash]
|
|
14
|
+
model: opus
|
|
15
|
+
---
|
|
16
|
+
|
|
17
|
+
You audit code that parses attacker-controlled input. Every byte of an email —
|
|
18
|
+
headers, folded whitespace, addresses, MIME structure, attachment names,
|
|
19
|
+
payloads — is hostile. The calling application, its filesystem, and PATH are
|
|
20
|
+
trusted. Judge each issue against that boundary; do not report threats that
|
|
21
|
+
require an already-compromised host unless they escalate.
|
|
22
|
+
|
|
23
|
+
## Prime directive: prove it, then report it
|
|
24
|
+
|
|
25
|
+
A security finding you cannot demonstrate is a guess. Before you report:
|
|
26
|
+
|
|
27
|
+
- **ReDoS / algorithmic complexity** — write a scaling probe. Feed the regex or
|
|
28
|
+
function graduated input sizes and measure. Report only if time grows
|
|
29
|
+
super-linearly (ratio ≥ ~3 per input doubling) AND the input is reachable
|
|
30
|
+
from the public API. Confirm reachability by driving it end-to-end
|
|
31
|
+
(`parse_from_string` / `parse_from_bytes`), not just the raw regex — a
|
|
32
|
+
pattern can be quadratic in isolation yet unreachable because an earlier step
|
|
33
|
+
normalizes the input. Guard each probe with `signal.setitimer` so a true
|
|
34
|
+
blowup fails fast instead of hanging.
|
|
35
|
+
**Complexity is not only regexes.** The dominant blowups in this codebase are
|
|
36
|
+
plain Python: a loop over attacker-chosen keys where each iteration rescans
|
|
37
|
+
the whole collection (`for k in keys: message.get_all(k)`) is O(keys × total)
|
|
38
|
+
with no regex involved. Vary the two axes *independently* — many distinct
|
|
39
|
+
header names vs. many repeats of one name — because a probe that only grows
|
|
40
|
+
the total byte count keeps the ratio flat and hides the quadratic term.
|
|
41
|
+
- **Injection / traversal / write-primitive** — construct the malicious input
|
|
42
|
+
and show the resulting command, path, or file write. A `../` that
|
|
43
|
+
`os.path.basename` strips is not a finding.
|
|
44
|
+
- **Resource / structure DoS** — build the crafted message (deeply nested or
|
|
45
|
+
very wide multipart, huge part count, oversized payload) and measure time and
|
|
46
|
+
memory end-to-end. Report only if consumption grows super-linearly or is
|
|
47
|
+
unbounded relative to input size. Note whether the MIME walk is iterative
|
|
48
|
+
(no stack overflow) before claiming a recursion bug.
|
|
49
|
+
- **Parser-differential / evasion** — show two readings of the same bytes: what
|
|
50
|
+
the parser surfaces versus what the raw bytes contain. Dropped-undecodable
|
|
51
|
+
bytes (`errors="ignore"`) or Unicode normalization can hide an indicator a
|
|
52
|
+
downstream scanner would act on. The finding is the divergence, demonstrated.
|
|
53
|
+
- If you cannot build a trigger, label it **UNCONFIRMED** and say what blocked
|
|
54
|
+
you. Never inflate an unproven concern to High.
|
|
55
|
+
|
|
56
|
+
## What to examine (this codebase)
|
|
57
|
+
|
|
58
|
+
| Surface | Look for |
|
|
59
|
+
| ----------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
|
|
60
|
+
| Every `re.compile` / `re.*` on header or body text | Nested/overlapping quantifiers, `(a+)+`, lazy-then-`\s*` overlap, unbounded `[^x]+`. Which normalization runs first (e.g. `JUNK_PATTERN` collapses `[\t\n]` but NOT spaces). |
|
|
61
|
+
| `subprocess` (`msgconvert`) | `shell=` must be false and args a list (no injection); missing `communicate(timeout=)` = hang DoS; `.decode()` without `errors=` = crash. |
|
|
62
|
+
| `tempfile` + `open(..., "w"/"wb")` | `mkstemp` (good) vs predictable names; leak on the exception path; symlink/TOCTOU before write. |
|
|
63
|
+
| Attachment filename → disk | `basename` + null-byte reject + realpath/`commonpath` containment + `islink`. Flag if the RAW (unsafe) name is also exposed to callers as a write target. |
|
|
64
|
+
| Payload decode (`base64`, `get_payload(decode=True)`) | Unbounded in-memory amplification; malformed-input exceptions that crash the whole parse. |
|
|
65
|
+
| MIME structure (`message.walk()`, `get_payload`) | Part count, nesting depth, per-part size — no caps means a crafted multipart drives memory/time DoS. Measure scaling; confirm the walk is iterative before claiming recursion. |
|
|
66
|
+
| Charset handling (`errors="ignore"`, `normalize`) | Parser-differential / evasion: dropped bytes or NFC-normalized values make what the tool surfaces differ from the raw bytes a scanner sees. Show a mutation an analyst misses. |
|
|
67
|
+
| Malformed-input robustness | One bad part must not abort the whole parse or kill the worker. Feed truncated/garbage MIME; an exception escaping the public API is a crash-DoS. |
|
|
68
|
+
| Outlook `.msg` path (`extract_msg`, `msgconvert`) | Untrusted `.msg` flows into third-party OLE/CFB + Perl parsers with their own CVE history — flag the transitive attack surface and the temp/subprocess handling around it. |
|
|
69
|
+
| Logging (`log.debug` of headers/filenames) | Raw headers and attachment names are logged verbatim (PII / indicators). Data-handling note, not code-exec — report as Info. |
|
|
70
|
+
| `eval`/`exec`/`pickle`/`yaml.load`/`__import__` | Should be absent. Any occurrence is a finding until proven inert. |
|
|
71
|
+
| `getattr(self, X)` where `X` is a header/part name | Reflective dispatch on an attacker-chosen name. See "Reflective dispatch" below — collision, recursion, and non-serializable leakage are all reachable from a 12-byte email. |
|
|
72
|
+
| `json.dumps` over a dynamically built dict | Values arriving from reflection are not guaranteed JSON-safe. A bound method or `Message` object in the dict is an uncaught `TypeError` on a public property. |
|
|
73
|
+
| `find()` / `in` / `split()` locating a security token | Naive substring search for a delimiter (`"by"`, a trust string) that an attacker also controls the *neighbouring* text of. See "Trust-boundary string parsing" below. |
|
|
74
|
+
|
|
75
|
+
## Reflective dispatch on attacker-controlled names
|
|
76
|
+
|
|
77
|
+
`MailParser.__getattr__` exposes every header as an attribute, and `_make_mail`
|
|
78
|
+
/ `headers` iterate `message.keys()` calling `getattr(self, name)`. The header
|
|
79
|
+
name is attacker-chosen, so **the attacker picks which Python attribute is
|
|
80
|
+
read**. Python resolves real class attributes *before* `__getattr__`, so any
|
|
81
|
+
name in `dir(cls)` shadows the header path. Always run this probe:
|
|
82
|
+
|
|
83
|
+
```python
|
|
84
|
+
for a in [x for x in dir(MailParser) if not x.startswith("_")]:
|
|
85
|
+
try:
|
|
86
|
+
mailparser.parse_from_string(f"{a}: x\r\n\r\n").mail_json
|
|
87
|
+
except Exception as e:
|
|
88
|
+
print(a, type(e).__name__, e)
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
Three distinct bug classes fall out, and you must check for all three — finding
|
|
92
|
+
one does not rule out the others:
|
|
93
|
+
|
|
94
|
+
- **Method/property collision** — the dict gets a bound method or a live object
|
|
95
|
+
instead of a string. Downstream `json.dumps` raises `TypeError`. Crash-DoS on
|
|
96
|
+
a public API from a minimal message.
|
|
97
|
+
- **Recursion cycle** — a property that itself iterates `message.keys()` and
|
|
98
|
+
calls `getattr` can re-enter itself when a header is named after that property
|
|
99
|
+
or one of its `_json` / `_raw` aliases. Check the cycle guard covers **every**
|
|
100
|
+
alias and is **case-insensitive**: `set(message.keys()) - {"headers"}` does not
|
|
101
|
+
exclude `Headers_json`, and `message.keys()` preserves the sender's casing.
|
|
102
|
+
A recursion cycle nested inside a per-key rescan multiplies the two costs —
|
|
103
|
+
measure it, it is usually the worst finding on the page.
|
|
104
|
+
- **Side-effecting property** — reflection can *invoke* a property the caller
|
|
105
|
+
never asked for. Confirm no property in the collision set writes files, spawns
|
|
106
|
+
a subprocess, or mutates state.
|
|
107
|
+
|
|
108
|
+
The fix to argue for is structural, not a denylist: header values must resolve
|
|
109
|
+
through a lookup that never touches Python attributes. Reject patches that just
|
|
110
|
+
add another name to an exclusion set — that is how the `headers_json` cycle
|
|
111
|
+
survived the `headers` fix.
|
|
112
|
+
|
|
113
|
+
## Trust-boundary string parsing
|
|
114
|
+
|
|
115
|
+
`get_server_ipaddress(trust)` decides *which IP the mail came from* — its output
|
|
116
|
+
is used for attribution and blocklisting, so a wrong answer is a security
|
|
117
|
+
failure, not a cosmetic bug. Anywhere a security decision depends on locating a
|
|
118
|
+
delimiter, check the search is anchored:
|
|
119
|
+
|
|
120
|
+
- `header.find("by")` matches inside `derby.example.com` or `nearby.example.org`.
|
|
121
|
+
Hostnames come from the sender's HELO (no DNS control needed) and land in the
|
|
122
|
+
trusted MTA's own `Received` header.
|
|
123
|
+
- Truncating the clause makes extraction fail on the *genuine* top hop, and the
|
|
124
|
+
loop then falls through to older, fully attacker-forged `Received` headers —
|
|
125
|
+
so the failure mode is not "returns nothing", it is "returns the attacker's
|
|
126
|
+
value". Always test the fall-through, not just the single-header case.
|
|
127
|
+
- Use the existing anchored `const._CLAUSE_SPLITTER` rather than a bare `\bby\b`
|
|
128
|
+
(`\b` still matches inside `host.by.example`, since `.` is a non-word char).
|
|
129
|
+
|
|
130
|
+
Prove these with a control matrix, not a single PoC: benign hostname, malicious
|
|
131
|
+
hostname alone, forged header alone, and both together. If a benign hostname
|
|
132
|
+
also misattributes, say so — it makes the finding a correctness bug too and
|
|
133
|
+
raises the priority.
|
|
134
|
+
|
|
135
|
+
## Invariants that must stay true (verify, don't assume)
|
|
136
|
+
|
|
137
|
+
These hold in the current code and are cheap to re-check with a grep. A change
|
|
138
|
+
that breaks one is a finding in itself — encode them so a future PR that adds
|
|
139
|
+
the capability gets flagged instead of slipping in.
|
|
140
|
+
|
|
141
|
+
- **No network access from parsing.** Nothing reached by parsed content may call
|
|
142
|
+
`urllib` / `socket` / `requests` / `smtplib`. A parser that fetches a URL, DTD,
|
|
143
|
+
or remote image is SSRF. `grep -rE "urllib|socket|requests|http" src/` must
|
|
144
|
+
stay free of live calls.
|
|
145
|
+
- **No XML parser on mail content.** No `xml.*` / `lxml` / `etree`. Their absence
|
|
146
|
+
is what rules out XXE and entity-expansion (billion-laughs).
|
|
147
|
+
- **No archive extraction.** Attachment code may read *names* (`.tar.gz`) but must
|
|
148
|
+
never `extractall` / open a payload with `zipfile` / `tarfile` — that would add
|
|
149
|
+
zip-slip and decompression-bomb surface.
|
|
150
|
+
|
|
151
|
+
## Output
|
|
152
|
+
|
|
153
|
+
Rank most-severe first. Per finding:
|
|
154
|
+
|
|
155
|
+
- **Severity + CVSS 3.1 vector + CWE** (e.g. High — CVSS:3.1/AV:N/AC:L/PR:N/UI:N/S:U/C:N/I:N/A:H = 7.5, CWE-1333).
|
|
156
|
+
- **Location** `path:line`.
|
|
157
|
+
- **Trigger** — the concrete input, and the measured evidence (size → time table, or the resulting command/path).
|
|
158
|
+
- **Reachability** — the public call that reaches it.
|
|
159
|
+
- **Fix** — the exact change (bound a quantifier, add `timeout=`, normalize first) and a failing regression test.
|
|
160
|
+
|
|
161
|
+
No praise or filler. If nothing is exploitable, say so and list what you proved safe. You review; the caller fixes.
|
|
@@ -69,7 +69,7 @@ jobs:
|
|
|
69
69
|
|
|
70
70
|
- name: Publish to PyPI
|
|
71
71
|
if: matrix.python-version == '3.10' && startsWith(github.ref, 'refs/tags/')
|
|
72
|
-
uses: pypa/gh-action-pypi-publish@v1.
|
|
72
|
+
uses: pypa/gh-action-pypi-publish@v1.14.2
|
|
73
73
|
with:
|
|
74
74
|
user: ${{ secrets.PYPI_USERNAME }}
|
|
75
75
|
password: ${{ secrets.PYPI_PASSWORD }}
|
|
@@ -61,6 +61,25 @@ Three access modes are supported for any attribute `X`:
|
|
|
61
61
|
Address headers listed in `const.ADDRESSES_HEADERS` (`from`, `to`, `cc`, `bcc`, `reply-to`,
|
|
62
62
|
`delivered-to`) return `list[tuple[str, str]]` (display_name, email) instead of plain strings.
|
|
63
63
|
|
|
64
|
+
**Never resolve a header name with `getattr`, and never rewrite it**: `__getattr__` serves names
|
|
65
|
+
the *caller* types, so it may fold `_`→`-`, honour the `_json`/`_raw` suffixes, and use `getattr`
|
|
66
|
+
to reach computed parts (`attachments_json`). Names taken from a parsed message must go through
|
|
67
|
+
`MailParser._header_value()`, which looks the name up literally in the header index and never
|
|
68
|
+
touches a Python attribute. Both halves matter:
|
|
69
|
+
|
|
70
|
+
- Python looks up real class attributes before `__getattr__`, so a sender naming a header `Parse`
|
|
71
|
+
or `Headers_json` otherwise picks which attribute is read — a bound method lands in the parsed
|
|
72
|
+
mail, or a property re-enters itself. Excluding individual names from a key set is not a fix;
|
|
73
|
+
that is what let `Headers_json` through after `headers` was excluded.
|
|
74
|
+
- Suffix handling on a wire name is an amplifier and an evasion: each `_json` re-serializes the
|
|
75
|
+
previous result (five input bytes per doubling of the output), and `Subject_json:` reports the
|
|
76
|
+
value of `Subject` while silently dropping its own.
|
|
77
|
+
|
|
78
|
+
**Header index**: `_build_header_index()` maps lowercased name → list of raw values once per
|
|
79
|
+
parse. `Message.get_all()` is a linear scan, so the one call per distinct name previously made by
|
|
80
|
+
`_make_mail()` cost O(distinct × total) on attacker-chosen names. Look up via the index inside any
|
|
81
|
+
loop over header names.
|
|
82
|
+
|
|
64
83
|
**RFC non-compliance fallback in `get_addresses()`**: Python's
|
|
65
84
|
`email.utils.getaddresses(strict=True)` (hardened against CVE-2023-27043) rejects headers where
|
|
66
85
|
the display name contains `@`. Since this is a forensics tool, `utils.py` applies
|
|
@@ -71,6 +90,28 @@ actually in the header.
|
|
|
71
90
|
(`from`, `by`, `via`, `with`, `id`, `for`, `envelope-from`) using `const._CLAUSE_SPLITTER`.
|
|
72
91
|
Output list is ordered first-hop first. Unparseable headers fall back to `{"raw": ...}`.
|
|
73
92
|
|
|
93
|
+
**Sender-IP attribution fails closed**: `get_server_ipaddress()` walks trust-matching `Received`
|
|
94
|
+
headers only while a hop names a *private* IP (an internal relay); a hop naming **no** IP ends the
|
|
95
|
+
search with `None`. Never resume the walk on that case — older `Received` headers are written by
|
|
96
|
+
the sender, so "extraction failed" would become "returns the sender's chosen IP".
|
|
97
|
+
|
|
98
|
+
Candidate addresses come only from the `from` clause (`utils.get_from_clause()`, which ends at the
|
|
99
|
+
next RFC 5321 keyword). Inside it, `_sender_ip_candidates()` applies one positional rule, because
|
|
100
|
+
**text cannot be classified by what it looks like** — a closed `[...]` pair the sender wrote is
|
|
101
|
+
byte-identical to one the MTA wrote, and `EHLO [8.8.8.8]` is a form RFC 5321 §4.1.3 requires:
|
|
102
|
+
|
|
103
|
+
- the **first token** is the HELO name, whatever its shape, and is never a candidate;
|
|
104
|
+
- an explicit **HELO marker inside a comment group** (`(helo=x)`, `(account a@b HELO x)`) is
|
|
105
|
+
sender text and is excluded, located with `const._HELO_RE`;
|
|
106
|
+
- a candidate must sit **inside a `(`/`[` group** (`utils.group_spans()`), which is what makes a
|
|
107
|
+
clause truncated by a multi-word HELO fail closed — what it leaves behind is bare;
|
|
108
|
+
- IPv4 and IPv6 matches are merged **positionally**, never family-first: choosing IPv4 first let
|
|
109
|
+
one private literal at EHLO suppress the IPv6 scan and hide the real sender.
|
|
110
|
+
|
|
111
|
+
Only concession: `from [ip] (helo=x)` (Exim/CommuniGate), accepted when there is no other
|
|
112
|
+
candidate and the marker is in a group. Do not add lookbehind guards to `_HELO_RE` to patch new
|
|
113
|
+
cases — three rounds of that each reopened a hole the previous one closed.
|
|
114
|
+
|
|
74
115
|
**Defect detection**: During `parse()`, every MIME part is walked and `_append_defects()` records
|
|
75
116
|
RFC violations. `EPILOGUE_DEFECTS` triggers special epilogue extraction to recover hidden payloads
|
|
76
117
|
in malformed boundaries.
|
|
@@ -87,6 +128,10 @@ Accessible as `parser.mail` / `parser.mail_partial`.
|
|
|
87
128
|
To make a header always appear in partial output, add its lowercase name to `OTHERS_PARTS` in
|
|
88
129
|
`const.py`. Address-type headers (returning parsed name/email tuples) go in `ADDRESSES_HEADERS`.
|
|
89
130
|
|
|
131
|
+
If the new part is computed by a `MailParser` property rather than read off the wire, add it to
|
|
132
|
+
`COMPUTED_PARTS` as well — that set is exactly what `_make_mail()` is allowed to resolve through
|
|
133
|
+
attribute access.
|
|
134
|
+
|
|
90
135
|
## Workflow
|
|
91
136
|
|
|
92
137
|
After every change:
|
|
@@ -95,3 +140,8 @@ After every change:
|
|
|
95
140
|
1. Update README.md if the change affects usage, API, or setup.
|
|
96
141
|
1. Stage changes and run pre-commit; fix all reported issues before proceeding.
|
|
97
142
|
1. Run full test suite; fix all failures before reporting done.
|
|
143
|
+
1. Run a security review of the change with the `security-reviewer` sub-agent
|
|
144
|
+
(`.claude/agents/security-reviewer.md`). All parsed input is
|
|
145
|
+
attacker-controlled, so any change to parsing, regexes, subprocess, temp
|
|
146
|
+
files, or attachment handling must be reviewed. Reproduce and fix every
|
|
147
|
+
High/Medium finding (with a regression test) before reporting done.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: mail-parser
|
|
3
|
-
Version: 4.6.
|
|
3
|
+
Version: 4.6.2
|
|
4
4
|
Summary: A tool that parses emails by enhancing the Python standard library, extracting all details into a comprehensive object.
|
|
5
5
|
Author-email: Fedele Mantuano <mantuano.fedele@gmail.com>
|
|
6
6
|
Maintainer-email: Fedele Mantuano <mantuano.fedele@gmail.com>
|
|
@@ -206,9 +206,16 @@ pipelines.
|
|
|
206
206
|
- `body` - Complete message body
|
|
207
207
|
- `text_html` - HTML body parts (list)
|
|
208
208
|
- `text_plain` - Plain text body parts (list)
|
|
209
|
-
- `headers` - All headers as a structured object
|
|
209
|
+
- `headers` - All headers as a structured object. Keys are the header names exactly as they
|
|
210
|
+
appear in the message, and every header is reported — including names that match a
|
|
211
|
+
`MailParser` method or property (`Parse`, `Message`, `Headers_json`) and names containing
|
|
212
|
+
underscores (`X_Spam_Flag`). Such names are always resolved as headers, never as attributes,
|
|
213
|
+
and are never rewritten.
|
|
210
214
|
- `attachments` - Complete attachment metadata and payloads
|
|
211
|
-
- `get_server_ipaddress()` - Reliable sender IP extraction with trust levels
|
|
215
|
+
- `get_server_ipaddress()` - Reliable sender IP extraction with trust levels. Only the `from`
|
|
216
|
+
clause of the first trusted `Received` header is searched, with the sender-supplied HELO name
|
|
217
|
+
removed, and the result is `None` when that hop names no public IP. Attribution never falls
|
|
218
|
+
back to older `Received` headers, which the sender is free to forge.
|
|
212
219
|
- `to_domains` - Extracted recipient domains for analysis
|
|
213
220
|
- `timezone` - Detected timezone information
|
|
214
221
|
- `defects` - RFC compliance issues for security analysis
|
|
@@ -236,6 +243,12 @@ access the `X-MSMail-Priority` header:
|
|
|
236
243
|
mail.X_MSMail_Priority
|
|
237
244
|
```
|
|
238
245
|
|
|
246
|
+
This underscore-for-hyphen convenience, and the `_json` / `_raw` suffixes, apply only to
|
|
247
|
+
attribute access written by you. Names read out of a message — the keys of `mail` and `headers` —
|
|
248
|
+
are looked up literally, so a header genuinely named `X_Spam_Flag` or `Subject_json` keeps its own
|
|
249
|
+
name and its own value. Attribute names beginning with an underscore are not headers and raise
|
|
250
|
+
`AttributeError`.
|
|
251
|
+
|
|
239
252
|
The `received` header is intelligently parsed into individual hops, revealing the complete email
|
|
240
253
|
routing path. Each hop contains structured fields:
|
|
241
254
|
|
|
@@ -179,9 +179,16 @@ pipelines.
|
|
|
179
179
|
- `body` - Complete message body
|
|
180
180
|
- `text_html` - HTML body parts (list)
|
|
181
181
|
- `text_plain` - Plain text body parts (list)
|
|
182
|
-
- `headers` - All headers as a structured object
|
|
182
|
+
- `headers` - All headers as a structured object. Keys are the header names exactly as they
|
|
183
|
+
appear in the message, and every header is reported — including names that match a
|
|
184
|
+
`MailParser` method or property (`Parse`, `Message`, `Headers_json`) and names containing
|
|
185
|
+
underscores (`X_Spam_Flag`). Such names are always resolved as headers, never as attributes,
|
|
186
|
+
and are never rewritten.
|
|
183
187
|
- `attachments` - Complete attachment metadata and payloads
|
|
184
|
-
- `get_server_ipaddress()` - Reliable sender IP extraction with trust levels
|
|
188
|
+
- `get_server_ipaddress()` - Reliable sender IP extraction with trust levels. Only the `from`
|
|
189
|
+
clause of the first trusted `Received` header is searched, with the sender-supplied HELO name
|
|
190
|
+
removed, and the result is `None` when that hop names no public IP. Attribution never falls
|
|
191
|
+
back to older `Received` headers, which the sender is free to forge.
|
|
185
192
|
- `to_domains` - Extracted recipient domains for analysis
|
|
186
193
|
- `timezone` - Detected timezone information
|
|
187
194
|
- `defects` - RFC compliance issues for security analysis
|
|
@@ -209,6 +216,12 @@ access the `X-MSMail-Priority` header:
|
|
|
209
216
|
mail.X_MSMail_Priority
|
|
210
217
|
```
|
|
211
218
|
|
|
219
|
+
This underscore-for-hyphen convenience, and the `_json` / `_raw` suffixes, apply only to
|
|
220
|
+
attribute access written by you. Names read out of a message — the keys of `mail` and `headers` —
|
|
221
|
+
are looked up literally, so a header genuinely named `X_Spam_Flag` or `Subject_json` keeps its own
|
|
222
|
+
name and its own value. Attribute names beginning with an underscore are not headers and raise
|
|
223
|
+
`AttributeError`.
|
|
224
|
+
|
|
212
225
|
The `received` header is intelligently parsed into individual hops, revealing the complete email
|
|
213
226
|
routing path. Each hop contains structured fields:
|
|
214
227
|
|
|
@@ -49,7 +49,9 @@ test = [
|
|
|
49
49
|
]
|
|
50
50
|
|
|
51
51
|
[build-system]
|
|
52
|
-
|
|
52
|
+
# hatchling >=1.32 emits core metadata 2.5, which older twine/packaging
|
|
53
|
+
# in the publishing toolchain cannot parse. Cap until PyPI tooling catches up.
|
|
54
|
+
requires = ["hatchling>=1.27,<1.32"]
|
|
53
55
|
build-backend = "hatchling.build"
|
|
54
56
|
|
|
55
57
|
[tool.uv]
|
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
#!/usr/bin/env python
|
|
2
|
+
|
|
3
|
+
"""
|
|
4
|
+
Copyright 2018 Fedele Mantuano (https://twitter.com/fedelemantuano)
|
|
5
|
+
|
|
6
|
+
Licensed under the Apache License, Version 2.0 (the "License");
|
|
7
|
+
you may not use this file except in compliance with the License.
|
|
8
|
+
You may obtain a copy of the License at
|
|
9
|
+
|
|
10
|
+
http://www.apache.org/licenses/LICENSE-2.0
|
|
11
|
+
|
|
12
|
+
Unless required by applicable law or agreed to in writing, software
|
|
13
|
+
distributed under the License is distributed on an "AS IS" BASIS,
|
|
14
|
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
15
|
+
See the License for the specific language governing permissions and
|
|
16
|
+
limitations under the License.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
import re
|
|
20
|
+
|
|
21
|
+
# IPv4 pattern - validates octet range (0-255) per RFC 791
|
|
22
|
+
REGXIP = re.compile(
|
|
23
|
+
r"(?:(?:25[0-5]|2[0-4]\d|[01]?\d\d?)\.){3}"
|
|
24
|
+
r"(?:25[0-5]|2[0-4]\d|[01]?\d\d?)"
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
# IPv6 pattern - matches standard and common compressed forms per RFC 5952
|
|
28
|
+
# Alternation order matters: Python's ``re`` takes the first alternative
|
|
29
|
+
# that matches, not the longest. The "trailing ::" branch therefore has to
|
|
30
|
+
# come *after* every branch that continues past the ``::`` — with it listed
|
|
31
|
+
# early, ``2a00:1450:4864:20::32`` matched only as ``2a00:1450:4864:20::``,
|
|
32
|
+
# reporting a different, valid, routable address as the sender.
|
|
33
|
+
REGXIP6 = re.compile(
|
|
34
|
+
r"(?:(?:[0-9a-fA-F]{1,4}:){7}[0-9a-fA-F]{1,4}" # full form
|
|
35
|
+
r"|[0-9a-fA-F]{1,4}:(?::[0-9a-fA-F]{1,4}){1,6}" # 6 groups after ::
|
|
36
|
+
r"|(?:[0-9a-fA-F]{1,4}:){1,2}(?::[0-9a-fA-F]{1,4}){1,5}"
|
|
37
|
+
r"|(?:[0-9a-fA-F]{1,4}:){1,3}(?::[0-9a-fA-F]{1,4}){1,4}"
|
|
38
|
+
r"|(?:[0-9a-fA-F]{1,4}:){1,4}(?::[0-9a-fA-F]{1,4}){1,3}"
|
|
39
|
+
r"|(?:[0-9a-fA-F]{1,4}:){1,5}(?::[0-9a-fA-F]{1,4}){1,2}"
|
|
40
|
+
r"|(?:[0-9a-fA-F]{1,4}:){1,6}:[0-9a-fA-F]{1,4}" # 1 group after ::
|
|
41
|
+
r"|:(?::[0-9a-fA-F]{1,4}){1,7}" # ::x:x...
|
|
42
|
+
r"|(?:[0-9a-fA-F]{1,4}:){1,7}:" # trailing ::
|
|
43
|
+
r"|::)" # just ::
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
# Normalize whitespace: collapse tabs and newlines to single space.
|
|
47
|
+
# Parenthesized comments and bracketed IPs are preserved.
|
|
48
|
+
JUNK_PATTERN = r"[\t\n]+"
|
|
49
|
+
|
|
50
|
+
# ------------------------------------------------------------------ #
|
|
51
|
+
# Received header parsing — RFC 5321 §4.4 grammar:
|
|
52
|
+
#
|
|
53
|
+
# Received = "Received:" *( received-token / comment ) ";" date-time
|
|
54
|
+
# received-token = "from" domain / "by" domain / "via" atom
|
|
55
|
+
# / "with" atom / "id" atom / "for" addr-spec
|
|
56
|
+
#
|
|
57
|
+
# Strategy: tokenize on clause keywords, then extract values per clause.
|
|
58
|
+
# This eliminates the duplicated boundary lookaheads of the old
|
|
59
|
+
# per-clause pattern list and matches the RFC grammar directly.
|
|
60
|
+
# ------------------------------------------------------------------ #
|
|
61
|
+
|
|
62
|
+
# Pattern that splits a received header into clause tokens.
|
|
63
|
+
# Matches each RFC 5321 keyword at a word boundary followed by its value,
|
|
64
|
+
# which extends up to the next keyword or semicolon.
|
|
65
|
+
# The keywords are: from, by, via, with (not "with cipher"), id, for,
|
|
66
|
+
# plus the non-standard envelope-from and envelope-sender.
|
|
67
|
+
_CLAUSE_SPLITTER = re.compile(
|
|
68
|
+
r"(?:^|\s+)"
|
|
69
|
+
r"(from|by|via|with(?!\s+cipher)|id|for|envelope-from|envelope-sender)"
|
|
70
|
+
r"\s+",
|
|
71
|
+
re.I,
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
# Collapses any run of whitespace to a single space. Applied to a received
|
|
75
|
+
# header before ``_CLAUSE_SPLITTER`` so the splitter's ``\s+`` sub-patterns
|
|
76
|
+
# only ever match a single character. Without it a long run of spaces with no
|
|
77
|
+
# following clause keyword drives the split into quadratic backtracking — an
|
|
78
|
+
# ordinary parse of an attacker-supplied Received header becomes a denial of
|
|
79
|
+
# service (CWE-1333). Note the ``[\t\n]``-only normalization elsewhere does
|
|
80
|
+
# not collapse spaces, so this guard must live at the point of use.
|
|
81
|
+
_WS_RUN_RE = re.compile(r"\s+")
|
|
82
|
+
|
|
83
|
+
# Matches the HELO/EHLO name an MTA records inside the ``from`` clause,
|
|
84
|
+
# in the two shapes seen in the wild:
|
|
85
|
+
# from evil.example ([203.0.113.9]:45321 helo=[8.8.8.8]) by ... (Exim)
|
|
86
|
+
# from [203.0.113.9] (account x@y HELO 8.8.8.8) by ... (CommuniGate)
|
|
87
|
+
# The HELO argument is chosen by the sender and may be an RFC 5321 address
|
|
88
|
+
# literal, so it must be removed before scanning the clause for the sender
|
|
89
|
+
# IP — otherwise the sender simply appends the IP they want reported.
|
|
90
|
+
# Used to locate the span, never to delete it: the trailing \S+ would
|
|
91
|
+
# otherwise swallow whatever follows, and a sender whose rDNS or HELO is
|
|
92
|
+
# literally ``helo`` can place that word right before the MTA-written IP.
|
|
93
|
+
# Two guards keep the span off MTA-written text. The lookbehind: the
|
|
94
|
+
# clause value *starts* with the HELO name, so at offset 0 the word is the
|
|
95
|
+
# name itself, not a label introducing one. The ``(?![\[(])`` on the
|
|
96
|
+
# space-separated form: an MTA writes its own address inside brackets or
|
|
97
|
+
# parentheses, so a following group is never part of the HELO argument.
|
|
98
|
+
# Exim's ``helo=[8.8.8.8]`` is a genuine bracketed HELO argument and keeps
|
|
99
|
+
# no such guard.
|
|
100
|
+
_HELO_RE = re.compile(
|
|
101
|
+
r"(?<=[(\s])e?helo\s*=\S*" # Exim "helo=value": after "(" or space
|
|
102
|
+
r"|(?<=\s)e?helo\s+\S+", # space form: only after whitespace
|
|
103
|
+
re.I,
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
# RFC 5321 §4.1.3 tags an IPv6 literal as ``[IPv6:2a00:...]``. REGXIP6
|
|
107
|
+
# happily starts matching at the ``6`` of the tag and yields
|
|
108
|
+
# ``6:2a00:1450:4864:20::`` — a different, valid, routable address that an
|
|
109
|
+
# analyst would then act on. Blank the tag before scanning.
|
|
110
|
+
_IPV6_TAG_RE = re.compile(r"IPv6:", re.I)
|
|
111
|
+
|
|
112
|
+
# Extracts envelope-from email: envelope-from <addr>
|
|
113
|
+
_ENVELOPE_FROM_RE = re.compile(r"<([^>]+)>")
|
|
114
|
+
|
|
115
|
+
# Date after semicolon (standard RFC 5321)
|
|
116
|
+
_DATE_RE = re.compile(r";\s*(.*)", re.DOTALL)
|
|
117
|
+
|
|
118
|
+
# SendGrid non-standard date format (no semicolon)
|
|
119
|
+
_SENDGRID_DATE_RE = re.compile(
|
|
120
|
+
r"(\d{4}-\d{2}-\d{2}\s+\d{2}:\d{2}:\d{2}\.\d{9}\s+\+0000\s+UTC)"
|
|
121
|
+
r"\s+m=\+\d+\.\d+",
|
|
122
|
+
re.I,
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
EPILOGUE_DEFECTS = {"StartBoundaryNotFoundDefect"}
|
|
126
|
+
|
|
127
|
+
ADDRESSES_HEADERS = set(["bcc", "cc", "delivered-to", "from", "reply-to", "to"])
|
|
128
|
+
|
|
129
|
+
# These parts are always returned
|
|
130
|
+
OTHERS_PARTS = set(
|
|
131
|
+
[
|
|
132
|
+
"attachments",
|
|
133
|
+
"body",
|
|
134
|
+
"date",
|
|
135
|
+
"message-id",
|
|
136
|
+
"received",
|
|
137
|
+
"subject",
|
|
138
|
+
"timezone",
|
|
139
|
+
"to_domains",
|
|
140
|
+
"user-agent",
|
|
141
|
+
"x-mailer",
|
|
142
|
+
"x-original-to",
|
|
143
|
+
]
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
# Subset of OTHERS_PARTS that MailParser computes itself: each maps to a
|
|
147
|
+
# property, not to a header read off the wire. ``_make_mail()`` resolves
|
|
148
|
+
# only these names through attribute access; every other key it handles is
|
|
149
|
+
# a header name chosen by the sender and goes through
|
|
150
|
+
# ``MailParser._header_value()``, which never touches an attribute. Keep
|
|
151
|
+
# this set in sync with OTHERS_PARTS and with the properties in core.py.
|
|
152
|
+
#
|
|
153
|
+
# These names shadow a header of the same name in ``mail`` / ``mail_json``:
|
|
154
|
+
# a message carrying a literal ``Body:`` header reports the computed body
|
|
155
|
+
# there, not the header value. The wire value is never lost — it is in
|
|
156
|
+
# ``headers`` / ``headers_json`` under its own name — but a consumer
|
|
157
|
+
# reading only ``mail_json`` will not see it.
|
|
158
|
+
COMPUTED_PARTS = set(
|
|
159
|
+
[
|
|
160
|
+
"attachments",
|
|
161
|
+
"body",
|
|
162
|
+
"date",
|
|
163
|
+
"received",
|
|
164
|
+
"timezone",
|
|
165
|
+
"to_domains",
|
|
166
|
+
]
|
|
167
|
+
)
|