mail-parser 4.6.4__tar.gz → 4.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. mail_parser-4.6.4/CLAUDE.md → mail_parser-4.7.0/AGENTS.md +109 -18
  2. mail_parser-4.7.0/CLAUDE.md +5 -0
  3. {mail_parser-4.6.4 → mail_parser-4.7.0}/PKG-INFO +57 -1
  4. {mail_parser-4.6.4 → mail_parser-4.7.0}/README.md +56 -0
  5. mail_parser-4.7.0/src/mailparser/addresses.py +291 -0
  6. {mail_parser-4.6.4 → mail_parser-4.7.0}/src/mailparser/core.py +42 -13
  7. {mail_parser-4.6.4 → mail_parser-4.7.0}/src/mailparser/utils.py +18 -153
  8. {mail_parser-4.6.4 → mail_parser-4.7.0}/src/mailparser/version.py +1 -1
  9. mail_parser-4.7.0/tests/test_address_recovery.py +289 -0
  10. {mail_parser-4.6.4 → mail_parser-4.7.0}/uv.lock +827 -651
  11. {mail_parser-4.6.4 → mail_parser-4.7.0}/.claude/agents/security-reviewer.md +0 -0
  12. {mail_parser-4.6.4 → mail_parser-4.7.0}/.github/FUNDING.yml +0 -0
  13. {mail_parser-4.6.4 → mail_parser-4.7.0}/.github/ISSUE_TEMPLATE/bug_report.md +0 -0
  14. {mail_parser-4.6.4 → mail_parser-4.7.0}/.github/ISSUE_TEMPLATE/feature_request.md +0 -0
  15. {mail_parser-4.6.4 → mail_parser-4.7.0}/.github/copilot-instructions.md +0 -0
  16. {mail_parser-4.6.4 → mail_parser-4.7.0}/.github/instructions/containerization-docker-best-practices.instructions.md +0 -0
  17. {mail_parser-4.6.4 → mail_parser-4.7.0}/.github/instructions/github-actions-ci-cd-best-practices.instructions.md +0 -0
  18. {mail_parser-4.6.4 → mail_parser-4.7.0}/.github/instructions/markdown.instructions.md +0 -0
  19. {mail_parser-4.6.4 → mail_parser-4.7.0}/.github/instructions/python.instructions.md +0 -0
  20. {mail_parser-4.6.4 → mail_parser-4.7.0}/.github/workflows/main.yml +0 -0
  21. {mail_parser-4.6.4 → mail_parser-4.7.0}/.gitignore +0 -0
  22. {mail_parser-4.6.4 → mail_parser-4.7.0}/.markdownlint.json +0 -0
  23. {mail_parser-4.6.4 → mail_parser-4.7.0}/.pre-commit-config.yaml +0 -0
  24. {mail_parser-4.6.4 → mail_parser-4.7.0}/Dockerfile +0 -0
  25. {mail_parser-4.6.4 → mail_parser-4.7.0}/LICENSE.txt +0 -0
  26. {mail_parser-4.6.4 → mail_parser-4.7.0}/Makefile +0 -0
  27. {mail_parser-4.6.4 → mail_parser-4.7.0}/NOTICE.txt +0 -0
  28. {mail_parser-4.6.4 → mail_parser-4.7.0}/docker-compose.yml +0 -0
  29. {mail_parser-4.6.4 → mail_parser-4.7.0}/docs/images/Bitcoin SpamScope.jpg +0 -0
  30. {mail_parser-4.6.4 → mail_parser-4.7.0}/pyproject.toml +0 -0
  31. {mail_parser-4.6.4 → mail_parser-4.7.0}/src/mailparser/__init__.py +0 -0
  32. {mail_parser-4.6.4 → mail_parser-4.7.0}/src/mailparser/__main__.py +0 -0
  33. {mail_parser-4.6.4 → mail_parser-4.7.0}/src/mailparser/const.py +0 -0
  34. {mail_parser-4.6.4 → mail_parser-4.7.0}/src/mailparser/exceptions.py +0 -0
  35. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_malformed_1 +0 -0
  36. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_malformed_2 +0 -0
  37. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_malformed_3 +0 -0
  38. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_outlook_1 +0 -0
  39. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_1 +0 -0
  40. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_10 +0 -0
  41. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_11 +0 -0
  42. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_12 +0 -0
  43. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_13 +0 -0
  44. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_14 +0 -0
  45. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_15 +0 -0
  46. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_16 +0 -0
  47. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_17 +0 -0
  48. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_18 +0 -0
  49. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_19 +0 -0
  50. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_2 +0 -0
  51. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_3 +0 -0
  52. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_4 +0 -0
  53. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_5 +0 -0
  54. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_6 +0 -0
  55. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_7 +0 -0
  56. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_8 +0 -0
  57. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/mails/mail_test_9 +0 -0
  58. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/test_improved_received_patterns.py +0 -0
  59. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/test_mail_parser.py +0 -0
  60. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/test_main.py +0 -0
  61. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/test_received_corpus.py +0 -0
  62. {mail_parser-4.6.4 → mail_parser-4.7.0}/tests/test_utils.py +0 -0
@@ -1,6 +1,8 @@
1
- # CLAUDE.md
1
+ # AGENTS.md
2
2
 
3
- This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository.
3
+ This file is the shared source of repository instructions for Codex, Claude Code,
4
+ and other coding agents. It applies throughout this repository. Keep shared
5
+ instructions here; `CLAUDE.md` imports this file for Claude Code.
4
6
 
5
7
  ## Engineering Principles
6
8
 
@@ -13,6 +15,71 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co
13
15
  - Every third-party import must have an explicit entry in `pyproject.toml`. Use extras when the
14
16
  runtime feature requires them (e.g. `elasticsearch[async]`).
15
17
 
18
+ ## RFC 5322 compliance and forensic recovery
19
+
20
+ For **every code change**, consult [RFC 5322](https://www.rfc-editor.org/info/rfc5322/)
21
+ and check the affected behavior against its applicable syntax and semantics.
22
+ Record the sections consulted and the compliance checks performed in the change
23
+ summary; if the change has no message-format impact, explicitly explain why.
24
+ Check applicable errata and updates linked by the RFC Editor, and related
25
+ standards when the change concerns SMTP trace fields or MIME.
26
+
27
+ Distinguish syntax permitted for newly generated messages (sections 2 and 3)
28
+ from obsolete syntax that receivers must still parse (section 4). Consult
29
+ section 5 for security considerations. Do not label an accepted obsolete form
30
+ invalid solely because it must not be generated in new messages.
31
+
32
+ This is a forensic parser: malformed input can deliberately exploit differences
33
+ between parsers to evade security tools. For nonconforming messages:
34
+
35
+ - Recover as much content as can be interpreted safely, preserving suspicious
36
+ display text, addresses, headers, and payload evidence. Do not silently discard
37
+ or normalize away data that could expose an evasion attempt.
38
+ - Surface violations through `has_defects` and the relevant structured defect
39
+ metadata. For address headers, preserve `address_header_defects` entries with
40
+ `reason`, `raw`, `recovered`, `candidates`, `header`, and `occurrence`.
41
+ Successful recovery must not erase the defect or imply RFC compliance.
42
+ - Keep ambiguous or incomplete evidence distinguishable from selected values.
43
+ Never promote an address found in display text or a comment over a complete
44
+ structural mailbox. Preserve occurrence context for repeated headers.
45
+ - Report observable defects and conflicting interpretations; malformed syntax
46
+ alone does not prove malicious intent.
47
+ - Keep existing resource, path, subprocess, and trust-boundary safeguards.
48
+ Recovery must not introduce crashes, unbounded amplification, unsafe side
49
+ effects, or fabricated attribution. Use controlled errors when safe recovery
50
+ is impossible; the fail-fast principle must not become blanket rejection of
51
+ recoverable malformed mail.
52
+
53
+ For parsing changes, cover conforming input, applicable obsolete syntax, and
54
+ malformed/adversarial variants with regression tests. Assert both recovered
55
+ content and defect metadata, including cases where no unambiguous recovery is
56
+ possible. Check that valid neighboring fields or list members remain intact.
57
+
58
+ Example of required recovery behavior (address syntax: sections 3.2.3, 3.2.5,
59
+ and 3.4). Preserve the misleading unquoted display name as evidence, select the
60
+ address inside angle brackets, and expose the invalid display name:
61
+
62
+ ```python
63
+ import mailparser
64
+
65
+ mail = mailparser.parse_from_string(
66
+ "From: billing@trusted.example < billing@vendor.example >\r\n\r\n"
67
+ )
68
+ assert mail.from_ == [("billing@trusted.example", "billing@vendor.example")]
69
+ assert mail.has_defects is True
70
+ assert mail.address_header_defects == [{
71
+ "reason": "invalid-display-name",
72
+ "raw": "billing@trusted.example < billing@vendor.example >",
73
+ "recovered": True,
74
+ "candidates": [{
75
+ "display_name": "billing@trusted.example",
76
+ "address": "billing@vendor.example",
77
+ }],
78
+ "header": "from",
79
+ "occurrence": 0,
80
+ }]
81
+ ```
82
+
16
83
  ## Python Style
17
84
 
18
85
  - Follow PEP 8. Use 4 spaces, never tabs.
@@ -27,22 +94,26 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co
27
94
  ```bash
28
95
  uv sync # install all dependencies
29
96
  uv run pytest # run all tests
30
- uv run pytest tests/test_mail_parser.py::TestMailParser::test_name # single test
97
+ uv run pytest tests/test_mail_parser.py -k test_valid_mail # single test
31
98
  uv run ruff check . # lint
32
99
  uv run ruff format . # format
100
+ uv run pre-commit run --files AGENTS.md CLAUDE.md README.md # selected files
101
+ make pre-commit # all tracked files
33
102
  make check # lint + tests
34
103
  make build # clean + build wheel/sdist
35
104
  ```
36
105
 
37
106
  ## Architecture
38
107
 
39
- **Package layout**: `src/mailparser/` (src layout, no runtime deps — stdlib only).
108
+ **Package layout**: `src/mailparser/` (src layout, Python 3.9+, stdlib-only core;
109
+ optional Outlook dependencies are installed with `uv sync --extra outlook`).
40
110
 
41
111
  ### Module responsibilities
42
112
 
43
113
  | Module | Role |
44
114
  | --------------- | --------------------------------------------------------------------- |
45
115
  | `core.py` | `MailParser` class + `parse_from_*` factory functions |
116
+ | `addresses.py` | Forensic address scanner and recovery diagnostics |
46
117
  | `utils.py` | Stateless parsing helpers (address, received headers, date, encoding) |
47
118
  | `const.py` | Compiled regexes, header sets, clause splitter |
48
119
  | `exceptions.py` | Custom exception hierarchy |
@@ -80,11 +151,17 @@ parse. `Message.get_all()` is a linear scan, so the one call per distinct name p
80
151
  `_make_mail()` cost O(distinct × total) on attacker-chosen names. Look up via the index inside any
81
152
  loop over header names.
82
153
 
83
- **RFC non-compliance fallback in `get_addresses()`**: Python's
84
- `email.utils.getaddresses(strict=True)` (hardened against CVE-2023-27043) rejects headers where
85
- the display name contains `@`. Since this is a forensics tool, `utils.py` applies
86
- `_ADDR_FALLBACK_RE` whenever strict parsing returns empty results — always surfacing what is
87
- actually in the header.
154
+ **Forensic address parsing**: `addresses.py` scans structural delimiters in a
155
+ single pass, respecting quotes, escapes, nested comments and domain literals.
156
+ `utils.get_addresses()` adapts input and delegates to it. Complete angle-addrs
157
+ have precedence over display text; incomplete/ambiguous items are evidence,
158
+ not selected mailboxes. Never search display names/comments for bare addresses
159
+ or reintroduce the old overlapping fallback regex (ReDoS). Parsing a malformed
160
+ list member must not change the interpretation of its neighbours. `core.py`
161
+ caches address results and recovery diagnostics once per parse, examining every
162
+ header occurrence while retaining first-occurrence semantics for address
163
+ properties. `address_header_defects` is reserved computed metadata; literal
164
+ wire headers with that name remain accessible through `headers`/`message`.
88
165
 
89
166
  **Received header parsing**: `utils.receiveds_parsing()` tokenizes on RFC 5321 clause keywords
90
167
  (`from`, `by`, `via`, `with`, `id`, `for`, `envelope-from`) using `const._CLAUSE_SPLITTER`.
@@ -116,8 +193,10 @@ cases — three rounds of that each reopened a hole the previous one closed.
116
193
  RFC violations. `EPILOGUE_DEFECTS` triggers special epilogue extraction to recover hidden payloads
117
194
  in malformed boundaries.
118
195
 
119
- **Outlook support**: `parse_from_file_msg()` shells out to the system `msgconvert` Perl tool
120
- (`libemail-outlook-message-perl`) to convert `.msg` → `.eml`, then parses normally.
196
+ **Outlook support**: `parse_from_file_msg()` prefers the optional `extract-msg`
197
+ Python backend (`uv sync --extra outlook`). When unavailable, it falls back to
198
+ the deprecated system `msgconvert` Perl tool (`libemail-outlook-message-perl`)
199
+ to convert `.msg` → `.eml`, then parses normally.
121
200
 
122
201
  **Partial vs full mail**: `_make_mail(complete=True)` includes all headers found in the message;
123
202
  `complete=False` restricts to `const.ADDRESSES_HEADERS | const.OTHERS_PARTS` (the "main" headers).
@@ -136,12 +215,24 @@ attribute access.
136
215
 
137
216
  After every change:
138
217
 
139
- 1. Add/update unittests to cover the change.
218
+ 1. For code changes, perform the RFC 5322 review described above before
219
+ implementation and verify the resulting behavior against it before reporting
220
+ done. Include the RFC sections, recovery/defect checks, and any remaining
221
+ compliance gaps in the change summary.
222
+ 1. Add/update unittests for behavior changes.
140
223
  1. Update README.md if the change affects usage, API, or setup.
141
- 1. Stage changes and run pre-commit; fix all reported issues before proceeding.
224
+ 1. Run `uv run pre-commit run --files <changed-files>`; include new files and
225
+ fix all reported issues before proceeding. Use `make pre-commit` for all
226
+ tracked files.
142
227
  1. Run full test suite; fix all failures before reporting done.
143
- 1. Run a security review of the change with the `security-reviewer` sub-agent
144
- (`.claude/agents/security-reviewer.md`). All parsed input is
145
- attacker-controlled, so any change to parsing, regexes, subprocess, temp
146
- files, or attachment handling must be reviewed. Reproduce and fix every
147
- High/Medium finding (with a regression test) before reporting done.
228
+ 1. Run a security review of the change using the procedure in
229
+ [`.claude/agents/security-reviewer.md`](.claude/agents/security-reviewer.md).
230
+ In Claude Code, use the `security-reviewer` sub-agent. In Codex or another
231
+ agent, give the Markdown body of that file to a review sub-agent when
232
+ available, or follow it directly when delegation is unavailable. The YAML
233
+ front matter configures Claude Code only; it does not select tools or models
234
+ in Codex. All parsed input is attacker-controlled, so any change to parsing,
235
+ regexes, subprocess, temp files, or attachment handling must be reviewed.
236
+ Reproduce and fix every High/Medium finding (with a regression test) before
237
+ reporting done. For documentation-only changes, check that security
238
+ invariants and review requirements remain intact.
@@ -0,0 +1,5 @@
1
+ # Claude Code instructions
2
+
3
+ Shared repository instructions are maintained in `AGENTS.md`.
4
+
5
+ @AGENTS.md
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: mail-parser
3
- Version: 4.6.4
3
+ Version: 4.7.0
4
4
  Summary: A tool that parses emails by enhancing the Python standard library, extracting all details into a comprehensive object.
5
5
  Project-URL: Homepage, https://github.com/SpamScope/mail-parser
6
6
  Project-URL: Source, https://github.com/SpamScope/mail-parser
@@ -298,6 +298,49 @@ mail.to_raw # Original "To:" header string as it appears in the email
298
298
  The command-line tool outputs parsed emails in JSON format by default for easy integration with
299
299
  other tools and pipelines.
300
300
 
301
+ ### Address recovery and ambiguous headers
302
+
303
+ Address properties such as `from_`, `to`, and `reply_to` retain their list of
304
+ `(display_name, address)` tuples. Mailbox selection respects quoted strings,
305
+ escapes, nested comments, groups, and folding whitespace. For example:
306
+
307
+ ```python
308
+ mail = mailparser.parse_from_string(
309
+ "From: billing@trusted.example < billing@vendor.example >\r\n\r\n"
310
+ )
311
+ mail.from_
312
+ # [('billing@trusted.example', 'billing@vendor.example')]
313
+ mail.has_defects
314
+ # True
315
+ mail.address_header_defects[0]["reason"]
316
+ # 'invalid-display-name'
317
+ ```
318
+
319
+ RFC 5322 permits whitespace around the mailbox inside `<...>`; the unquoted
320
+ `@` in this example's display name requires forensic recovery. A complete,
321
+ unambiguous angle-address takes precedence over its display name. Addresses
322
+ inside comments or quoted display names are never mailbox candidates.
323
+
324
+ `address_header_defects` exposes recovery and ambiguity evidence. Each entry
325
+ contains `header`, zero-based `occurrence`, the original `raw` list item,
326
+ `reason`, `recovered`, and `candidates` (display-name/address dictionaries).
327
+ It is included in `mail`, `mail_partial`, and their JSON output when nonempty,
328
+ and also sets `has_defects` and the `AddressHeaderDefect` category. Diagnostics
329
+ are computed once per parse, including repeated header occurrences; address
330
+ properties retain their existing first-occurrence semantics.
331
+
332
+ Ambiguous or incomplete items are omitted from address tuples, with their
333
+ raw evidence and any structurally identified candidates retained in diagnostics.
334
+ Valid neighbouring items are parsed independently. Consumers making filtering
335
+ or attribution decisions should inspect these diagnostics: an empty address
336
+ list does not establish that the original header contained no addresses.
337
+ `structural-recovery` means the stdlib's interpretation could not be used; it
338
+ is not by itself proof of an RFC violation. This recovery policy is not a full
339
+ RFC validator or a guarantee of how every mail client displays malformed mail.
340
+ The complete original header values remain available through `from_raw`,
341
+ `to_raw`, etc.; literal headers colliding with computed metadata remain in
342
+ `headers` and `message.get_all(...)`.
343
+
301
344
  ## Defects and Their Critical Role in Email Security
302
345
 
303
346
  Email structural defects are not merely technical curiosities—they represent **potential security
@@ -574,3 +617,16 @@ The default configuration includes:
574
617
 
575
618
  Customize the `docker-compose.yml` file to adjust mount points, command-line options, or
576
619
  environment variables for your specific use case.
620
+
621
+ # Working with coding agents
622
+
623
+ [AGENTS.md](AGENTS.md) is the shared source of development commands, coding
624
+ conventions, architecture notes, and security review requirements.
625
+ [Codex reads it automatically](https://developers.openai.com/codex/guides/agents-md).
626
+ [CLAUDE.md](CLAUDE.md) imports the same file using
627
+ [Claude Code's import syntax](https://code.claude.com/docs/en/memory#import-additional-files),
628
+ so updates to shared guidance belong in `AGENTS.md`.
629
+
630
+ The existing [security reviewer](.claude/agents/security-reviewer.md) remains
631
+ available as a Claude Code sub-agent. Codex follows the same review procedure
632
+ as described in `AGENTS.md`.
@@ -266,6 +266,49 @@ mail.to_raw # Original "To:" header string as it appears in the email
266
266
  The command-line tool outputs parsed emails in JSON format by default for easy integration with
267
267
  other tools and pipelines.
268
268
 
269
+ ### Address recovery and ambiguous headers
270
+
271
+ Address properties such as `from_`, `to`, and `reply_to` retain their list of
272
+ `(display_name, address)` tuples. Mailbox selection respects quoted strings,
273
+ escapes, nested comments, groups, and folding whitespace. For example:
274
+
275
+ ```python
276
+ mail = mailparser.parse_from_string(
277
+ "From: billing@trusted.example < billing@vendor.example >\r\n\r\n"
278
+ )
279
+ mail.from_
280
+ # [('billing@trusted.example', 'billing@vendor.example')]
281
+ mail.has_defects
282
+ # True
283
+ mail.address_header_defects[0]["reason"]
284
+ # 'invalid-display-name'
285
+ ```
286
+
287
+ RFC 5322 permits whitespace around the mailbox inside `<...>`; the unquoted
288
+ `@` in this example's display name requires forensic recovery. A complete,
289
+ unambiguous angle-address takes precedence over its display name. Addresses
290
+ inside comments or quoted display names are never mailbox candidates.
291
+
292
+ `address_header_defects` exposes recovery and ambiguity evidence. Each entry
293
+ contains `header`, zero-based `occurrence`, the original `raw` list item,
294
+ `reason`, `recovered`, and `candidates` (display-name/address dictionaries).
295
+ It is included in `mail`, `mail_partial`, and their JSON output when nonempty,
296
+ and also sets `has_defects` and the `AddressHeaderDefect` category. Diagnostics
297
+ are computed once per parse, including repeated header occurrences; address
298
+ properties retain their existing first-occurrence semantics.
299
+
300
+ Ambiguous or incomplete items are omitted from address tuples, with their
301
+ raw evidence and any structurally identified candidates retained in diagnostics.
302
+ Valid neighbouring items are parsed independently. Consumers making filtering
303
+ or attribution decisions should inspect these diagnostics: an empty address
304
+ list does not establish that the original header contained no addresses.
305
+ `structural-recovery` means the stdlib's interpretation could not be used; it
306
+ is not by itself proof of an RFC violation. This recovery policy is not a full
307
+ RFC validator or a guarantee of how every mail client displays malformed mail.
308
+ The complete original header values remain available through `from_raw`,
309
+ `to_raw`, etc.; literal headers colliding with computed metadata remain in
310
+ `headers` and `message.get_all(...)`.
311
+
269
312
  ## Defects and Their Critical Role in Email Security
270
313
 
271
314
  Email structural defects are not merely technical curiosities—they represent **potential security
@@ -542,3 +585,16 @@ The default configuration includes:
542
585
 
543
586
  Customize the `docker-compose.yml` file to adjust mount points, command-line options, or
544
587
  environment variables for your specific use case.
588
+
589
+ # Working with coding agents
590
+
591
+ [AGENTS.md](AGENTS.md) is the shared source of development commands, coding
592
+ conventions, architecture notes, and security review requirements.
593
+ [Codex reads it automatically](https://developers.openai.com/codex/guides/agents-md).
594
+ [CLAUDE.md](CLAUDE.md) imports the same file using
595
+ [Claude Code's import syntax](https://code.claude.com/docs/en/memory#import-additional-files),
596
+ so updates to shared guidance belong in `AGENTS.md`.
597
+
598
+ The existing [security reviewer](.claude/agents/security-reviewer.md) remains
599
+ available as a Claude Code sub-agent. Codex follows the same review procedure
600
+ as described in `AGENTS.md`.
@@ -0,0 +1,291 @@
1
+ """Structural address recovery for untrusted RFC 5322 header values.
2
+
3
+ The scanner owns lexical boundaries and checks complete addr-specs against
4
+ bounded grammar patterns. The stdlib supplies conventional display names for
5
+ structurally consistent items. Recovery never searches a display name or
6
+ comment for a convenient email substring.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import re
12
+ from dataclasses import dataclass, field
13
+
14
+ _FOLD = re.compile(r"\r?\n[ \t]+")
15
+ # Anchored grammar checks: atom, quote and delimiter character sets do not
16
+ # overlap. In particular, do not search for bare addresses in arbitrary text
17
+ # or hand long malformed word sequences to headerregistry's quadratic parser.
18
+ _ATOM = r"[a-zA-Z0-9!#$%&'*+/=?^_`{|}~-]+"
19
+ _QUOTED = r'"(?:[^"\\\r\n]|\\[^\r\n])*"'
20
+ _WORD = rf"(?:{_ATOM}|{_QUOTED})"
21
+ _LOCAL = rf"{_WORD}(?:[ \t]*\.[ \t]*{_WORD})*"
22
+ _DOMAIN = rf"{_ATOM}(?:[ \t]*\.[ \t]*{_ATOM})*"
23
+ _LITERAL = r"\[(?:[^\[\]\\\r\n]|\\[^\r\n])*\]"
24
+ _ADDR_SPEC = re.compile(rf"{_LOCAL}[ \t]*@[ \t]*(?:{_DOMAIN}|{_LITERAL})")
25
+
26
+
27
+ @dataclass
28
+ class _Item:
29
+ raw: str
30
+ text: str
31
+ angles: list[tuple[int, int]]
32
+ problems: list[str] = field(default_factory=list)
33
+ comment_depth: int = 0
34
+
35
+
36
+ def _scan(raw: str) -> list[_Item]:
37
+ """Split only at structural delimiters; bound work per input character."""
38
+ items: list[_Item] = []
39
+ buf: list[str] = []
40
+ angles: list[tuple[int, int]] = []
41
+ problems: list[str] = []
42
+ start = 0
43
+ angle = None
44
+ comment = max_comment = 0
45
+ quote = literal = escaped = group = False
46
+
47
+ def emit(end):
48
+ nonlocal buf, angles, problems, start, max_comment
49
+ text = "".join(buf)
50
+ if text.strip() or problems:
51
+ items.append(_Item(raw[start:end], text, angles, problems, max_comment))
52
+ buf, angles, problems = [], [], []
53
+ start = end + 1
54
+ max_comment = 0
55
+
56
+ for i, ch in enumerate(raw):
57
+ if escaped:
58
+ if not comment:
59
+ buf.append(ch)
60
+ escaped = False
61
+ continue
62
+ if ch == "\\" and (quote or comment or literal):
63
+ escaped = True
64
+ if not comment:
65
+ buf.append(ch)
66
+ continue
67
+ if comment:
68
+ if ch == "(":
69
+ comment += 1
70
+ max_comment = max(max_comment, comment)
71
+ elif ch == ")":
72
+ comment -= 1
73
+ continue
74
+ if quote:
75
+ buf.append(ch)
76
+ if ch == '"':
77
+ quote = False
78
+ continue
79
+ if literal:
80
+ buf.append(ch)
81
+ if ch == "]":
82
+ literal = False
83
+ continue
84
+ if ch == "(":
85
+ comment = 1
86
+ max_comment = max(max_comment, 1)
87
+ buf.append(" ")
88
+ elif ch == '"':
89
+ quote = True
90
+ buf.append(ch)
91
+ elif ch == "[":
92
+ literal = True
93
+ buf.append(ch)
94
+ elif ch == "<":
95
+ if angle is not None:
96
+ problems.append("nested-angle")
97
+ else:
98
+ angle = len(buf)
99
+ buf.append(ch)
100
+ elif ch == ">":
101
+ if angle is None:
102
+ problems.append("unmatched-angle")
103
+ else:
104
+ angles.append((angle, len(buf)))
105
+ angle = None
106
+ buf.append(ch)
107
+ elif angle is None and ch == ",":
108
+ emit(i)
109
+ elif angle is None and ch == ":":
110
+ if group or angles:
111
+ problems.append("invalid-group-boundary")
112
+ buf.append(ch)
113
+ else:
114
+ group = True
115
+ # Preserve malformed labels as diagnostics, never mailboxes.
116
+ if not _valid_phrase("".join(buf)):
117
+ problems.append("invalid-group-name")
118
+ emit(i)
119
+ buf = []
120
+ start = i + 1
121
+ elif angle is None and ch == ";":
122
+ if not group:
123
+ problems.append("unmatched-group")
124
+ emit(i)
125
+ group = False
126
+ else:
127
+ if ch in ")]":
128
+ problems.append("unmatched-delimiter")
129
+ buf.append(ch)
130
+
131
+ if quote or comment or literal or angle is not None or escaped:
132
+ problems.append("unclosed-delimiter")
133
+ if group:
134
+ problems.append("unclosed-group")
135
+ emit(len(raw))
136
+ return items
137
+
138
+
139
+ def _mailbox(value: str) -> str | None:
140
+ """Validate the whole addr-spec, including quoted local parts and CFWS."""
141
+ value = _FOLD.sub(" ", value).strip()
142
+ # RFC 5322 section 4.4: obsolete source routes precede the addr-spec.
143
+ if value.startswith(("@", ",")):
144
+ literal = escaped = False
145
+ domains = []
146
+ start = 0
147
+ for i, ch in enumerate(value):
148
+ if escaped:
149
+ escaped = False
150
+ elif ch == "\\" and literal:
151
+ escaped = True
152
+ elif ch == "[":
153
+ literal = True
154
+ elif ch == "]":
155
+ literal = False
156
+ elif not literal and ch in ",:":
157
+ domain = value[start:i].strip()
158
+ if domain:
159
+ domains.append(domain)
160
+ start = i + 1
161
+ if ch == ":":
162
+ break
163
+ else:
164
+ return None
165
+ if not domains or any(
166
+ not domain.startswith("@") or not _ADDR_SPEC.fullmatch("route" + domain)
167
+ for domain in domains
168
+ ):
169
+ return None
170
+ value = value[start:].strip()
171
+ if not _ADDR_SPEC.fullmatch(value):
172
+ return None
173
+ # CFWS is not part of an atom. Retain spaces and escapes inside quoted
174
+ # local parts and domain literals, but remove token-separating FWS.
175
+ normalized = []
176
+ quoted = literal = escaped = False
177
+ for ch in value:
178
+ if escaped:
179
+ escaped = False
180
+ elif ch == "\\" and (quoted or literal):
181
+ escaped = True
182
+ elif ch == '"' and not literal:
183
+ quoted = not quoted
184
+ elif ch == "[" and not quoted:
185
+ literal = True
186
+ elif ch == "]" and literal:
187
+ literal = False
188
+ if ch not in " \t" or quoted or literal:
189
+ normalized.append(ch)
190
+ return "".join(normalized)
191
+
192
+
193
+ def _valid_phrase(value: str) -> bool:
194
+ """Check label specials outside quoted strings (period is obs-phrase)."""
195
+ quoted = escaped = False
196
+ for ch in _FOLD.sub(" ", value).strip():
197
+ if ord(ch) < 32 and ch != "\t":
198
+ return False
199
+ if escaped:
200
+ escaped = False
201
+ elif ch == "\\" and quoted:
202
+ escaped = True
203
+ elif ch == '"':
204
+ quoted = not quoted
205
+ elif not quoted and ch in "@<>[]:;,\\":
206
+ return False
207
+ return bool(value.strip()) and not (quoted or escaped)
208
+
209
+
210
+ def _name(value: str) -> str:
211
+ """Recover a display phrase, unquoting quoted pairs without regexes."""
212
+ result = []
213
+ quoted = escaped = False
214
+ for ch in _FOLD.sub(" ", value).strip():
215
+ if escaped:
216
+ result.append(ch)
217
+ escaped = False
218
+ elif ch == "\\" and quoted:
219
+ escaped = True
220
+ elif ch == '"':
221
+ quoted = not quoted
222
+ else:
223
+ result.append(ch)
224
+ return "".join(result).strip()
225
+
226
+
227
+ def parse_address_header(raw: str, strict_parser):
228
+ """Return mailbox tuples and evidence for recovery or ambiguity.
229
+
230
+ ``strict_parser`` is the runtime-compatible email.utils adapter. Each
231
+ list member is parsed independently so a malformed member cannot change
232
+ the interpretation of a neighbouring, valid quoted name.
233
+ """
234
+ results = []
235
+ diagnostics = []
236
+ for item in _scan(raw):
237
+ candidates = []
238
+ # Derive the prefix once: repeating it for every angle-addr would
239
+ # copy quadratically much text on a long ambiguous item.
240
+ name = _name(item.text[: item.angles[0][0]]) if item.angles else ""
241
+ for index, (left, right) in enumerate(item.angles):
242
+ address = _mailbox(item.text[left + 1 : right])
243
+ if address:
244
+ candidates.append((name if index == 0 else "", address))
245
+ reason = None
246
+ selected = []
247
+ if item.problems:
248
+ reason = ",".join(sorted(set(item.problems)))
249
+ elif item.angles:
250
+ left, right = item.angles[0]
251
+ if len(item.angles) != 1 or item.text[right + 1 :].strip():
252
+ reason = "ambiguous-angle-address"
253
+ elif not candidates:
254
+ reason = "invalid-angle-address"
255
+ else:
256
+ selected = candidates
257
+ if item.text[:left].strip() and not _valid_phrase(item.text[:left]):
258
+ reason = "invalid-display-name"
259
+ else:
260
+ address = _mailbox(item.text)
261
+ if address:
262
+ selected = [("", address)]
263
+ candidates = selected
264
+ else:
265
+ reason = "invalid-or-ambiguous-address"
266
+
267
+ if selected:
268
+ # Deep comments are stripped by the iterative scanner before
269
+ # reaching the stdlib's recursive comment parser.
270
+ source = item.raw if item.comment_depth < 16 else item.text
271
+ parsed = strict_parser([source])
272
+ if [addr for _, addr in parsed if addr] == [a for _, a in selected]:
273
+ selected = [(name, addr) for name, addr in parsed if addr]
274
+ else:
275
+ # Parsing can reject legal CFWS too. This records recovery,
276
+ # not a claim that every rejection proves RFC noncompliance.
277
+ reason = reason or "structural-recovery"
278
+ results.extend(selected)
279
+ if reason:
280
+ diagnostics.append(
281
+ {
282
+ "reason": reason,
283
+ "raw": item.raw,
284
+ "recovered": bool(selected),
285
+ "candidates": [
286
+ {"display_name": name, "address": addr}
287
+ for name, addr in candidates
288
+ ],
289
+ }
290
+ )
291
+ return results, diagnostics