mail-parser 4.6.0__tar.gz → 4.6.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. mail_parser-4.6.2/.claude/agents/security-reviewer.md +161 -0
  2. {mail_parser-4.6.0 → mail_parser-4.6.2}/.github/workflows/main.yml +1 -1
  3. {mail_parser-4.6.0 → mail_parser-4.6.2}/CLAUDE.md +50 -0
  4. {mail_parser-4.6.0 → mail_parser-4.6.2}/PKG-INFO +16 -3
  5. {mail_parser-4.6.0 → mail_parser-4.6.2}/README.md +15 -2
  6. {mail_parser-4.6.0 → mail_parser-4.6.2}/pyproject.toml +3 -1
  7. mail_parser-4.6.2/src/mailparser/const.py +167 -0
  8. {mail_parser-4.6.0 → mail_parser-4.6.2}/src/mailparser/core.py +346 -80
  9. {mail_parser-4.6.0 → mail_parser-4.6.2}/src/mailparser/exceptions.py +15 -0
  10. {mail_parser-4.6.0 → mail_parser-4.6.2}/src/mailparser/utils.py +240 -23
  11. {mail_parser-4.6.0 → mail_parser-4.6.2}/src/mailparser/version.py +1 -1
  12. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/test_mail_parser.py +679 -18
  13. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/test_utils.py +119 -4
  14. mail_parser-4.6.0/src/mailparser/const.py +0 -101
  15. {mail_parser-4.6.0 → mail_parser-4.6.2}/.github/FUNDING.yml +0 -0
  16. {mail_parser-4.6.0 → mail_parser-4.6.2}/.github/ISSUE_TEMPLATE/bug_report.md +0 -0
  17. {mail_parser-4.6.0 → mail_parser-4.6.2}/.github/ISSUE_TEMPLATE/feature_request.md +0 -0
  18. {mail_parser-4.6.0 → mail_parser-4.6.2}/.github/copilot-instructions.md +0 -0
  19. {mail_parser-4.6.0 → mail_parser-4.6.2}/.github/instructions/containerization-docker-best-practices.instructions.md +0 -0
  20. {mail_parser-4.6.0 → mail_parser-4.6.2}/.github/instructions/github-actions-ci-cd-best-practices.instructions.md +0 -0
  21. {mail_parser-4.6.0 → mail_parser-4.6.2}/.github/instructions/markdown.instructions.md +0 -0
  22. {mail_parser-4.6.0 → mail_parser-4.6.2}/.github/instructions/python.instructions.md +0 -0
  23. {mail_parser-4.6.0 → mail_parser-4.6.2}/.gitignore +0 -0
  24. {mail_parser-4.6.0 → mail_parser-4.6.2}/.markdownlint.json +0 -0
  25. {mail_parser-4.6.0 → mail_parser-4.6.2}/.pre-commit-config.yaml +0 -0
  26. {mail_parser-4.6.0 → mail_parser-4.6.2}/Dockerfile +0 -0
  27. {mail_parser-4.6.0 → mail_parser-4.6.2}/LICENSE.txt +0 -0
  28. {mail_parser-4.6.0 → mail_parser-4.6.2}/Makefile +0 -0
  29. {mail_parser-4.6.0 → mail_parser-4.6.2}/NOTICE.txt +0 -0
  30. {mail_parser-4.6.0 → mail_parser-4.6.2}/docker-compose.yml +0 -0
  31. {mail_parser-4.6.0 → mail_parser-4.6.2}/docs/images/Bitcoin SpamScope.jpg +0 -0
  32. {mail_parser-4.6.0 → mail_parser-4.6.2}/src/mailparser/__init__.py +0 -0
  33. {mail_parser-4.6.0 → mail_parser-4.6.2}/src/mailparser/__main__.py +0 -0
  34. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_malformed_1 +0 -0
  35. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_malformed_2 +0 -0
  36. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_malformed_3 +0 -0
  37. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_outlook_1 +0 -0
  38. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_1 +0 -0
  39. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_10 +0 -0
  40. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_11 +0 -0
  41. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_12 +0 -0
  42. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_13 +0 -0
  43. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_14 +0 -0
  44. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_15 +0 -0
  45. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_16 +0 -0
  46. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_17 +0 -0
  47. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_18 +0 -0
  48. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_19 +0 -0
  49. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_2 +0 -0
  50. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_3 +0 -0
  51. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_4 +0 -0
  52. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_5 +0 -0
  53. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_6 +0 -0
  54. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_7 +0 -0
  55. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_8 +0 -0
  56. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/mails/mail_test_9 +0 -0
  57. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/test_improved_received_patterns.py +0 -0
  58. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/test_main.py +0 -0
  59. {mail_parser-4.6.0 → mail_parser-4.6.2}/tests/test_received_corpus.py +0 -0
  60. {mail_parser-4.6.0 → mail_parser-4.6.2}/uv.lock +0 -0
@@ -0,0 +1,161 @@
1
+ ---
2
+ name: security-reviewer
3
+ description: >
4
+ Security auditor for mail-parser and similar untrusted-input parsers. Reviews
5
+ code for real, reachable vulnerabilities — ReDoS, injection, path traversal,
6
+ unsafe subprocess/deserialization, MIME-structure and decode DoS,
7
+ parser-differential evasion, info leak — and also checks the safety invariants
8
+ (no network/XML/archive-extraction) still hold. PROVES each finding with a PoC
9
+ or measured blowup before reporting it. Use for
10
+ "security review", "audit for vulnerabilities", "check for security issues",
11
+ "is this exploitable". Threat model: input (email bytes/headers/attachments)
12
+ is fully attacker-controlled; the caller and filesystem are trusted.
13
+ tools: [Read, Grep, Glob, Bash]
14
+ model: opus
15
+ ---
16
+
17
+ You audit code that parses attacker-controlled input. Every byte of an email —
18
+ headers, folded whitespace, addresses, MIME structure, attachment names,
19
+ payloads — is hostile. The calling application, its filesystem, and PATH are
20
+ trusted. Judge each issue against that boundary; do not report threats that
21
+ require an already-compromised host unless they escalate.
22
+
23
+ ## Prime directive: prove it, then report it
24
+
25
+ A security finding you cannot demonstrate is a guess. Before you report:
26
+
27
+ - **ReDoS / algorithmic complexity** — write a scaling probe. Feed the regex or
28
+ function graduated input sizes and measure. Report only if time grows
29
+ super-linearly (ratio ≥ ~3 per input doubling) AND the input is reachable
30
+ from the public API. Confirm reachability by driving it end-to-end
31
+ (`parse_from_string` / `parse_from_bytes`), not just the raw regex — a
32
+ pattern can be quadratic in isolation yet unreachable because an earlier step
33
+ normalizes the input. Guard each probe with `signal.setitimer` so a true
34
+ blowup fails fast instead of hanging.
35
+ **Complexity is not only regexes.** The dominant blowups in this codebase are
36
+ plain Python: a loop over attacker-chosen keys where each iteration rescans
37
+ the whole collection (`for k in keys: message.get_all(k)`) is O(keys × total)
38
+ with no regex involved. Vary the two axes *independently* — many distinct
39
+ header names vs. many repeats of one name — because a probe that only grows
40
+ the total byte count keeps the ratio flat and hides the quadratic term.
41
+ - **Injection / traversal / write-primitive** — construct the malicious input
42
+ and show the resulting command, path, or file write. A `../` that
43
+ `os.path.basename` strips is not a finding.
44
+ - **Resource / structure DoS** — build the crafted message (deeply nested or
45
+ very wide multipart, huge part count, oversized payload) and measure time and
46
+ memory end-to-end. Report only if consumption grows super-linearly or is
47
+ unbounded relative to input size. Note whether the MIME walk is iterative
48
+ (no stack overflow) before claiming a recursion bug.
49
+ - **Parser-differential / evasion** — show two readings of the same bytes: what
50
+ the parser surfaces versus what the raw bytes contain. Dropped-undecodable
51
+ bytes (`errors="ignore"`) or Unicode normalization can hide an indicator a
52
+ downstream scanner would act on. The finding is the divergence, demonstrated.
53
+ - If you cannot build a trigger, label it **UNCONFIRMED** and say what blocked
54
+ you. Never inflate an unproven concern to High.
55
+
56
+ ## What to examine (this codebase)
57
+
58
+ | Surface | Look for |
59
+ | ----------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
60
+ | Every `re.compile` / `re.*` on header or body text | Nested/overlapping quantifiers, `(a+)+`, lazy-then-`\s*` overlap, unbounded `[^x]+`. Which normalization runs first (e.g. `JUNK_PATTERN` collapses `[\t\n]` but NOT spaces). |
61
+ | `subprocess` (`msgconvert`) | `shell=` must be false and args a list (no injection); missing `communicate(timeout=)` = hang DoS; `.decode()` without `errors=` = crash. |
62
+ | `tempfile` + `open(..., "w"/"wb")` | `mkstemp` (good) vs predictable names; leak on the exception path; symlink/TOCTOU before write. |
63
+ | Attachment filename → disk | `basename` + null-byte reject + realpath/`commonpath` containment + `islink`. Flag if the RAW (unsafe) name is also exposed to callers as a write target. |
64
+ | Payload decode (`base64`, `get_payload(decode=True)`) | Unbounded in-memory amplification; malformed-input exceptions that crash the whole parse. |
65
+ | MIME structure (`message.walk()`, `get_payload`) | Part count, nesting depth, per-part size — no caps means a crafted multipart drives memory/time DoS. Measure scaling; confirm the walk is iterative before claiming recursion. |
66
+ | Charset handling (`errors="ignore"`, `normalize`) | Parser-differential / evasion: dropped bytes or NFC-normalized values make what the tool surfaces differ from the raw bytes a scanner sees. Show a mutation an analyst misses. |
67
+ | Malformed-input robustness | One bad part must not abort the whole parse or kill the worker. Feed truncated/garbage MIME; an exception escaping the public API is a crash-DoS. |
68
+ | Outlook `.msg` path (`extract_msg`, `msgconvert`) | Untrusted `.msg` flows into third-party OLE/CFB + Perl parsers with their own CVE history — flag the transitive attack surface and the temp/subprocess handling around it. |
69
+ | Logging (`log.debug` of headers/filenames) | Raw headers and attachment names are logged verbatim (PII / indicators). Data-handling note, not code-exec — report as Info. |
70
+ | `eval`/`exec`/`pickle`/`yaml.load`/`__import__` | Should be absent. Any occurrence is a finding until proven inert. |
71
+ | `getattr(self, X)` where `X` is a header/part name | Reflective dispatch on an attacker-chosen name. See "Reflective dispatch" below — collision, recursion, and non-serializable leakage are all reachable from a 12-byte email. |
72
+ | `json.dumps` over a dynamically built dict | Values arriving from reflection are not guaranteed JSON-safe. A bound method or `Message` object in the dict is an uncaught `TypeError` on a public property. |
73
+ | `find()` / `in` / `split()` locating a security token | Naive substring search for a delimiter (`"by"`, a trust string) that an attacker also controls the *neighbouring* text of. See "Trust-boundary string parsing" below. |
74
+
75
+ ## Reflective dispatch on attacker-controlled names
76
+
77
+ `MailParser.__getattr__` exposes every header as an attribute, and `_make_mail`
78
+ / `headers` iterate `message.keys()` calling `getattr(self, name)`. The header
79
+ name is attacker-chosen, so **the attacker picks which Python attribute is
80
+ read**. Python resolves real class attributes *before* `__getattr__`, so any
81
+ name in `dir(cls)` shadows the header path. Always run this probe:
82
+
83
+ ```python
84
+ for a in [x for x in dir(MailParser) if not x.startswith("_")]:
85
+ try:
86
+ mailparser.parse_from_string(f"{a}: x\r\n\r\n").mail_json
87
+ except Exception as e:
88
+ print(a, type(e).__name__, e)
89
+ ```
90
+
91
+ Three distinct bug classes fall out, and you must check for all three — finding
92
+ one does not rule out the others:
93
+
94
+ - **Method/property collision** — the dict gets a bound method or a live object
95
+ instead of a string. Downstream `json.dumps` raises `TypeError`. Crash-DoS on
96
+ a public API from a minimal message.
97
+ - **Recursion cycle** — a property that itself iterates `message.keys()` and
98
+ calls `getattr` can re-enter itself when a header is named after that property
99
+ or one of its `_json` / `_raw` aliases. Check the cycle guard covers **every**
100
+ alias and is **case-insensitive**: `set(message.keys()) - {"headers"}` does not
101
+ exclude `Headers_json`, and `message.keys()` preserves the sender's casing.
102
+ A recursion cycle nested inside a per-key rescan multiplies the two costs —
103
+ measure it, it is usually the worst finding on the page.
104
+ - **Side-effecting property** — reflection can *invoke* a property the caller
105
+ never asked for. Confirm no property in the collision set writes files, spawns
106
+ a subprocess, or mutates state.
107
+
108
+ The fix to argue for is structural, not a denylist: header values must resolve
109
+ through a lookup that never touches Python attributes. Reject patches that just
110
+ add another name to an exclusion set — that is how the `headers_json` cycle
111
+ survived the `headers` fix.
112
+
113
+ ## Trust-boundary string parsing
114
+
115
+ `get_server_ipaddress(trust)` decides *which IP the mail came from* — its output
116
+ is used for attribution and blocklisting, so a wrong answer is a security
117
+ failure, not a cosmetic bug. Anywhere a security decision depends on locating a
118
+ delimiter, check the search is anchored:
119
+
120
+ - `header.find("by")` matches inside `derby.example.com` or `nearby.example.org`.
121
+ Hostnames come from the sender's HELO (no DNS control needed) and land in the
122
+ trusted MTA's own `Received` header.
123
+ - Truncating the clause makes extraction fail on the *genuine* top hop, and the
124
+ loop then falls through to older, fully attacker-forged `Received` headers —
125
+ so the failure mode is not "returns nothing", it is "returns the attacker's
126
+ value". Always test the fall-through, not just the single-header case.
127
+ - Use the existing anchored `const._CLAUSE_SPLITTER` rather than a bare `\bby\b`
128
+ (`\b` still matches inside `host.by.example`, since `.` is a non-word char).
129
+
130
+ Prove these with a control matrix, not a single PoC: benign hostname, malicious
131
+ hostname alone, forged header alone, and both together. If a benign hostname
132
+ also misattributes, say so — it makes the finding a correctness bug too and
133
+ raises the priority.
134
+
135
+ ## Invariants that must stay true (verify, don't assume)
136
+
137
+ These hold in the current code and are cheap to re-check with a grep. A change
138
+ that breaks one is a finding in itself — encode them so a future PR that adds
139
+ the capability gets flagged instead of slipping in.
140
+
141
+ - **No network access from parsing.** Nothing reached by parsed content may call
142
+ `urllib` / `socket` / `requests` / `smtplib`. A parser that fetches a URL, DTD,
143
+ or remote image is SSRF. `grep -rE "urllib|socket|requests|http" src/` must
144
+ stay free of live calls.
145
+ - **No XML parser on mail content.** No `xml.*` / `lxml` / `etree`. Their absence
146
+ is what rules out XXE and entity-expansion (billion-laughs).
147
+ - **No archive extraction.** Attachment code may read *names* (`.tar.gz`) but must
148
+ never `extractall` / open a payload with `zipfile` / `tarfile` — that would add
149
+ zip-slip and decompression-bomb surface.
150
+
151
+ ## Output
152
+
153
+ Rank most-severe first. Per finding:
154
+
155
+ - **Severity + CVSS 3.1 vector + CWE** (e.g. High — CVSS:3.1/AV:N/AC:L/PR:N/UI:N/S:U/C:N/I:N/A:H = 7.5, CWE-1333).
156
+ - **Location** `path:line`.
157
+ - **Trigger** — the concrete input, and the measured evidence (size → time table, or the resulting command/path).
158
+ - **Reachability** — the public call that reaches it.
159
+ - **Fix** — the exact change (bound a quantifier, add `timeout=`, normalize first) and a failing regression test.
160
+
161
+ No praise or filler. If nothing is exploitable, say so and list what you proved safe. You review; the caller fixes.
@@ -69,7 +69,7 @@ jobs:
69
69
 
70
70
  - name: Publish to PyPI
71
71
  if: matrix.python-version == '3.10' && startsWith(github.ref, 'refs/tags/')
72
- uses: pypa/gh-action-pypi-publish@v1.13.0
72
+ uses: pypa/gh-action-pypi-publish@v1.14.2
73
73
  with:
74
74
  user: ${{ secrets.PYPI_USERNAME }}
75
75
  password: ${{ secrets.PYPI_PASSWORD }}
@@ -61,6 +61,25 @@ Three access modes are supported for any attribute `X`:
61
61
  Address headers listed in `const.ADDRESSES_HEADERS` (`from`, `to`, `cc`, `bcc`, `reply-to`,
62
62
  `delivered-to`) return `list[tuple[str, str]]` (display_name, email) instead of plain strings.
63
63
 
64
+ **Never resolve a header name with `getattr`, and never rewrite it**: `__getattr__` serves names
65
+ the *caller* types, so it may fold `_`→`-`, honour the `_json`/`_raw` suffixes, and use `getattr`
66
+ to reach computed parts (`attachments_json`). Names taken from a parsed message must go through
67
+ `MailParser._header_value()`, which looks the name up literally in the header index and never
68
+ touches a Python attribute. Both halves matter:
69
+
70
+ - Python looks up real class attributes before `__getattr__`, so a sender naming a header `Parse`
71
+ or `Headers_json` otherwise picks which attribute is read — a bound method lands in the parsed
72
+ mail, or a property re-enters itself. Excluding individual names from a key set is not a fix;
73
+ that is what let `Headers_json` through after `headers` was excluded.
74
+ - Suffix handling on a wire name is an amplifier and an evasion: each `_json` re-serializes the
75
+ previous result (five input bytes per doubling of the output), and `Subject_json:` reports the
76
+ value of `Subject` while silently dropping its own.
77
+
78
+ **Header index**: `_build_header_index()` maps lowercased name → list of raw values once per
79
+ parse. `Message.get_all()` is a linear scan, so the one call per distinct name previously made by
80
+ `_make_mail()` cost O(distinct × total) on attacker-chosen names. Look up via the index inside any
81
+ loop over header names.
82
+
64
83
  **RFC non-compliance fallback in `get_addresses()`**: Python's
65
84
  `email.utils.getaddresses(strict=True)` (hardened against CVE-2023-27043) rejects headers where
66
85
  the display name contains `@`. Since this is a forensics tool, `utils.py` applies
@@ -71,6 +90,28 @@ actually in the header.
71
90
  (`from`, `by`, `via`, `with`, `id`, `for`, `envelope-from`) using `const._CLAUSE_SPLITTER`.
72
91
  Output list is ordered first-hop first. Unparseable headers fall back to `{"raw": ...}`.
73
92
 
93
+ **Sender-IP attribution fails closed**: `get_server_ipaddress()` walks trust-matching `Received`
94
+ headers only while a hop names a *private* IP (an internal relay); a hop naming **no** IP ends the
95
+ search with `None`. Never resume the walk on that case — older `Received` headers are written by
96
+ the sender, so "extraction failed" would become "returns the sender's chosen IP".
97
+
98
+ Candidate addresses come only from the `from` clause (`utils.get_from_clause()`, which ends at the
99
+ next RFC 5321 keyword). Inside it, `_sender_ip_candidates()` applies one positional rule, because
100
+ **text cannot be classified by what it looks like** — a closed `[...]` pair the sender wrote is
101
+ byte-identical to one the MTA wrote, and `EHLO [8.8.8.8]` is a form RFC 5321 §4.1.3 requires:
102
+
103
+ - the **first token** is the HELO name, whatever its shape, and is never a candidate;
104
+ - an explicit **HELO marker inside a comment group** (`(helo=x)`, `(account a@b HELO x)`) is
105
+ sender text and is excluded, located with `const._HELO_RE`;
106
+ - a candidate must sit **inside a `(`/`[` group** (`utils.group_spans()`), which is what makes a
107
+ clause truncated by a multi-word HELO fail closed — what it leaves behind is bare;
108
+ - IPv4 and IPv6 matches are merged **positionally**, never family-first: choosing IPv4 first let
109
+ one private literal at EHLO suppress the IPv6 scan and hide the real sender.
110
+
111
+ Only concession: `from [ip] (helo=x)` (Exim/CommuniGate), accepted when there is no other
112
+ candidate and the marker is in a group. Do not add lookbehind guards to `_HELO_RE` to patch new
113
+ cases — three rounds of that each reopened a hole the previous one closed.
114
+
74
115
  **Defect detection**: During `parse()`, every MIME part is walked and `_append_defects()` records
75
116
  RFC violations. `EPILOGUE_DEFECTS` triggers special epilogue extraction to recover hidden payloads
76
117
  in malformed boundaries.
@@ -87,6 +128,10 @@ Accessible as `parser.mail` / `parser.mail_partial`.
87
128
  To make a header always appear in partial output, add its lowercase name to `OTHERS_PARTS` in
88
129
  `const.py`. Address-type headers (returning parsed name/email tuples) go in `ADDRESSES_HEADERS`.
89
130
 
131
+ If the new part is computed by a `MailParser` property rather than read off the wire, add it to
132
+ `COMPUTED_PARTS` as well — that set is exactly what `_make_mail()` is allowed to resolve through
133
+ attribute access.
134
+
90
135
  ## Workflow
91
136
 
92
137
  After every change:
@@ -95,3 +140,8 @@ After every change:
95
140
  1. Update README.md if the change affects usage, API, or setup.
96
141
  1. Stage changes and run pre-commit; fix all reported issues before proceeding.
97
142
  1. Run full test suite; fix all failures before reporting done.
143
+ 1. Run a security review of the change with the `security-reviewer` sub-agent
144
+ (`.claude/agents/security-reviewer.md`). All parsed input is
145
+ attacker-controlled, so any change to parsing, regexes, subprocess, temp
146
+ files, or attachment handling must be reviewed. Reproduce and fix every
147
+ High/Medium finding (with a regression test) before reporting done.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: mail-parser
3
- Version: 4.6.0
3
+ Version: 4.6.2
4
4
  Summary: A tool that parses emails by enhancing the Python standard library, extracting all details into a comprehensive object.
5
5
  Author-email: Fedele Mantuano <mantuano.fedele@gmail.com>
6
6
  Maintainer-email: Fedele Mantuano <mantuano.fedele@gmail.com>
@@ -206,9 +206,16 @@ pipelines.
206
206
  - `body` - Complete message body
207
207
  - `text_html` - HTML body parts (list)
208
208
  - `text_plain` - Plain text body parts (list)
209
- - `headers` - All headers as a structured object
209
+ - `headers` - All headers as a structured object. Keys are the header names exactly as they
210
+ appear in the message, and every header is reported — including names that match a
211
+ `MailParser` method or property (`Parse`, `Message`, `Headers_json`) and names containing
212
+ underscores (`X_Spam_Flag`). Such names are always resolved as headers, never as attributes,
213
+ and are never rewritten.
210
214
  - `attachments` - Complete attachment metadata and payloads
211
- - `get_server_ipaddress()` - Reliable sender IP extraction with trust levels
215
+ - `get_server_ipaddress()` - Reliable sender IP extraction with trust levels. Only the `from`
216
+ clause of the first trusted `Received` header is searched, with the sender-supplied HELO name
217
+ removed, and the result is `None` when that hop names no public IP. Attribution never falls
218
+ back to older `Received` headers, which the sender is free to forge.
212
219
  - `to_domains` - Extracted recipient domains for analysis
213
220
  - `timezone` - Detected timezone information
214
221
  - `defects` - RFC compliance issues for security analysis
@@ -236,6 +243,12 @@ access the `X-MSMail-Priority` header:
236
243
  mail.X_MSMail_Priority
237
244
  ```
238
245
 
246
+ This underscore-for-hyphen convenience, and the `_json` / `_raw` suffixes, apply only to
247
+ attribute access written by you. Names read out of a message — the keys of `mail` and `headers` —
248
+ are looked up literally, so a header genuinely named `X_Spam_Flag` or `Subject_json` keeps its own
249
+ name and its own value. Attribute names beginning with an underscore are not headers and raise
250
+ `AttributeError`.
251
+
239
252
  The `received` header is intelligently parsed into individual hops, revealing the complete email
240
253
  routing path. Each hop contains structured fields:
241
254
 
@@ -179,9 +179,16 @@ pipelines.
179
179
  - `body` - Complete message body
180
180
  - `text_html` - HTML body parts (list)
181
181
  - `text_plain` - Plain text body parts (list)
182
- - `headers` - All headers as a structured object
182
+ - `headers` - All headers as a structured object. Keys are the header names exactly as they
183
+ appear in the message, and every header is reported — including names that match a
184
+ `MailParser` method or property (`Parse`, `Message`, `Headers_json`) and names containing
185
+ underscores (`X_Spam_Flag`). Such names are always resolved as headers, never as attributes,
186
+ and are never rewritten.
183
187
  - `attachments` - Complete attachment metadata and payloads
184
- - `get_server_ipaddress()` - Reliable sender IP extraction with trust levels
188
+ - `get_server_ipaddress()` - Reliable sender IP extraction with trust levels. Only the `from`
189
+ clause of the first trusted `Received` header is searched, with the sender-supplied HELO name
190
+ removed, and the result is `None` when that hop names no public IP. Attribution never falls
191
+ back to older `Received` headers, which the sender is free to forge.
185
192
  - `to_domains` - Extracted recipient domains for analysis
186
193
  - `timezone` - Detected timezone information
187
194
  - `defects` - RFC compliance issues for security analysis
@@ -209,6 +216,12 @@ access the `X-MSMail-Priority` header:
209
216
  mail.X_MSMail_Priority
210
217
  ```
211
218
 
219
+ This underscore-for-hyphen convenience, and the `_json` / `_raw` suffixes, apply only to
220
+ attribute access written by you. Names read out of a message — the keys of `mail` and `headers` —
221
+ are looked up literally, so a header genuinely named `X_Spam_Flag` or `Subject_json` keeps its own
222
+ name and its own value. Attribute names beginning with an underscore are not headers and raise
223
+ `AttributeError`.
224
+
212
225
  The `received` header is intelligently parsed into individual hops, revealing the complete email
213
226
  routing path. Each hop contains structured fields:
214
227
 
@@ -49,7 +49,9 @@ test = [
49
49
  ]
50
50
 
51
51
  [build-system]
52
- requires = ["hatchling"]
52
+ # hatchling >=1.32 emits core metadata 2.5, which older twine/packaging
53
+ # in the publishing toolchain cannot parse. Cap until PyPI tooling catches up.
54
+ requires = ["hatchling>=1.27,<1.32"]
53
55
  build-backend = "hatchling.build"
54
56
 
55
57
  [tool.uv]
@@ -0,0 +1,167 @@
1
+ #!/usr/bin/env python
2
+
3
+ """
4
+ Copyright 2018 Fedele Mantuano (https://twitter.com/fedelemantuano)
5
+
6
+ Licensed under the Apache License, Version 2.0 (the "License");
7
+ you may not use this file except in compliance with the License.
8
+ You may obtain a copy of the License at
9
+
10
+ http://www.apache.org/licenses/LICENSE-2.0
11
+
12
+ Unless required by applicable law or agreed to in writing, software
13
+ distributed under the License is distributed on an "AS IS" BASIS,
14
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
15
+ See the License for the specific language governing permissions and
16
+ limitations under the License.
17
+ """
18
+
19
+ import re
20
+
21
+ # IPv4 pattern - validates octet range (0-255) per RFC 791
22
+ REGXIP = re.compile(
23
+ r"(?:(?:25[0-5]|2[0-4]\d|[01]?\d\d?)\.){3}"
24
+ r"(?:25[0-5]|2[0-4]\d|[01]?\d\d?)"
25
+ )
26
+
27
+ # IPv6 pattern - matches standard and common compressed forms per RFC 5952
28
+ # Alternation order matters: Python's ``re`` takes the first alternative
29
+ # that matches, not the longest. The "trailing ::" branch therefore has to
30
+ # come *after* every branch that continues past the ``::`` — with it listed
31
+ # early, ``2a00:1450:4864:20::32`` matched only as ``2a00:1450:4864:20::``,
32
+ # reporting a different, valid, routable address as the sender.
33
+ REGXIP6 = re.compile(
34
+ r"(?:(?:[0-9a-fA-F]{1,4}:){7}[0-9a-fA-F]{1,4}" # full form
35
+ r"|[0-9a-fA-F]{1,4}:(?::[0-9a-fA-F]{1,4}){1,6}" # 6 groups after ::
36
+ r"|(?:[0-9a-fA-F]{1,4}:){1,2}(?::[0-9a-fA-F]{1,4}){1,5}"
37
+ r"|(?:[0-9a-fA-F]{1,4}:){1,3}(?::[0-9a-fA-F]{1,4}){1,4}"
38
+ r"|(?:[0-9a-fA-F]{1,4}:){1,4}(?::[0-9a-fA-F]{1,4}){1,3}"
39
+ r"|(?:[0-9a-fA-F]{1,4}:){1,5}(?::[0-9a-fA-F]{1,4}){1,2}"
40
+ r"|(?:[0-9a-fA-F]{1,4}:){1,6}:[0-9a-fA-F]{1,4}" # 1 group after ::
41
+ r"|:(?::[0-9a-fA-F]{1,4}){1,7}" # ::x:x...
42
+ r"|(?:[0-9a-fA-F]{1,4}:){1,7}:" # trailing ::
43
+ r"|::)" # just ::
44
+ )
45
+
46
+ # Normalize whitespace: collapse tabs and newlines to single space.
47
+ # Parenthesized comments and bracketed IPs are preserved.
48
+ JUNK_PATTERN = r"[\t\n]+"
49
+
50
+ # ------------------------------------------------------------------ #
51
+ # Received header parsing — RFC 5321 §4.4 grammar:
52
+ #
53
+ # Received = "Received:" *( received-token / comment ) ";" date-time
54
+ # received-token = "from" domain / "by" domain / "via" atom
55
+ # / "with" atom / "id" atom / "for" addr-spec
56
+ #
57
+ # Strategy: tokenize on clause keywords, then extract values per clause.
58
+ # This eliminates the duplicated boundary lookaheads of the old
59
+ # per-clause pattern list and matches the RFC grammar directly.
60
+ # ------------------------------------------------------------------ #
61
+
62
+ # Pattern that splits a received header into clause tokens.
63
+ # Matches each RFC 5321 keyword at a word boundary followed by its value,
64
+ # which extends up to the next keyword or semicolon.
65
+ # The keywords are: from, by, via, with (not "with cipher"), id, for,
66
+ # plus the non-standard envelope-from and envelope-sender.
67
+ _CLAUSE_SPLITTER = re.compile(
68
+ r"(?:^|\s+)"
69
+ r"(from|by|via|with(?!\s+cipher)|id|for|envelope-from|envelope-sender)"
70
+ r"\s+",
71
+ re.I,
72
+ )
73
+
74
+ # Collapses any run of whitespace to a single space. Applied to a received
75
+ # header before ``_CLAUSE_SPLITTER`` so the splitter's ``\s+`` sub-patterns
76
+ # only ever match a single character. Without it a long run of spaces with no
77
+ # following clause keyword drives the split into quadratic backtracking — an
78
+ # ordinary parse of an attacker-supplied Received header becomes a denial of
79
+ # service (CWE-1333). Note the ``[\t\n]``-only normalization elsewhere does
80
+ # not collapse spaces, so this guard must live at the point of use.
81
+ _WS_RUN_RE = re.compile(r"\s+")
82
+
83
+ # Matches the HELO/EHLO name an MTA records inside the ``from`` clause,
84
+ # in the two shapes seen in the wild:
85
+ # from evil.example ([203.0.113.9]:45321 helo=[8.8.8.8]) by ... (Exim)
86
+ # from [203.0.113.9] (account x@y HELO 8.8.8.8) by ... (CommuniGate)
87
+ # The HELO argument is chosen by the sender and may be an RFC 5321 address
88
+ # literal, so it must be removed before scanning the clause for the sender
89
+ # IP — otherwise the sender simply appends the IP they want reported.
90
+ # Used to locate the span, never to delete it: the trailing \S+ would
91
+ # otherwise swallow whatever follows, and a sender whose rDNS or HELO is
92
+ # literally ``helo`` can place that word right before the MTA-written IP.
93
+ # Two guards keep the span off MTA-written text. The lookbehind: the
94
+ # clause value *starts* with the HELO name, so at offset 0 the word is the
95
+ # name itself, not a label introducing one. The ``(?![\[(])`` on the
96
+ # space-separated form: an MTA writes its own address inside brackets or
97
+ # parentheses, so a following group is never part of the HELO argument.
98
+ # Exim's ``helo=[8.8.8.8]`` is a genuine bracketed HELO argument and keeps
99
+ # no such guard.
100
+ _HELO_RE = re.compile(
101
+ r"(?<=[(\s])e?helo\s*=\S*" # Exim "helo=value": after "(" or space
102
+ r"|(?<=\s)e?helo\s+\S+", # space form: only after whitespace
103
+ re.I,
104
+ )
105
+
106
+ # RFC 5321 §4.1.3 tags an IPv6 literal as ``[IPv6:2a00:...]``. REGXIP6
107
+ # happily starts matching at the ``6`` of the tag and yields
108
+ # ``6:2a00:1450:4864:20::`` — a different, valid, routable address that an
109
+ # analyst would then act on. Blank the tag before scanning.
110
+ _IPV6_TAG_RE = re.compile(r"IPv6:", re.I)
111
+
112
+ # Extracts envelope-from email: envelope-from <addr>
113
+ _ENVELOPE_FROM_RE = re.compile(r"<([^>]+)>")
114
+
115
+ # Date after semicolon (standard RFC 5321)
116
+ _DATE_RE = re.compile(r";\s*(.*)", re.DOTALL)
117
+
118
+ # SendGrid non-standard date format (no semicolon)
119
+ _SENDGRID_DATE_RE = re.compile(
120
+ r"(\d{4}-\d{2}-\d{2}\s+\d{2}:\d{2}:\d{2}\.\d{9}\s+\+0000\s+UTC)"
121
+ r"\s+m=\+\d+\.\d+",
122
+ re.I,
123
+ )
124
+
125
+ EPILOGUE_DEFECTS = {"StartBoundaryNotFoundDefect"}
126
+
127
+ ADDRESSES_HEADERS = set(["bcc", "cc", "delivered-to", "from", "reply-to", "to"])
128
+
129
+ # These parts are always returned
130
+ OTHERS_PARTS = set(
131
+ [
132
+ "attachments",
133
+ "body",
134
+ "date",
135
+ "message-id",
136
+ "received",
137
+ "subject",
138
+ "timezone",
139
+ "to_domains",
140
+ "user-agent",
141
+ "x-mailer",
142
+ "x-original-to",
143
+ ]
144
+ )
145
+
146
+ # Subset of OTHERS_PARTS that MailParser computes itself: each maps to a
147
+ # property, not to a header read off the wire. ``_make_mail()`` resolves
148
+ # only these names through attribute access; every other key it handles is
149
+ # a header name chosen by the sender and goes through
150
+ # ``MailParser._header_value()``, which never touches an attribute. Keep
151
+ # this set in sync with OTHERS_PARTS and with the properties in core.py.
152
+ #
153
+ # These names shadow a header of the same name in ``mail`` / ``mail_json``:
154
+ # a message carrying a literal ``Body:`` header reports the computed body
155
+ # there, not the header value. The wire value is never lost — it is in
156
+ # ``headers`` / ``headers_json`` under its own name — but a consumer
157
+ # reading only ``mail_json`` will not see it.
158
+ COMPUTED_PARTS = set(
159
+ [
160
+ "attachments",
161
+ "body",
162
+ "date",
163
+ "received",
164
+ "timezone",
165
+ "to_domains",
166
+ ]
167
+ )