spot-sdk-python 2.0.0b1__tar.gz → 2.0.0b2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/PKG-INFO +2 -2
  2. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/README.md +1 -1
  3. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/pyproject.toml +1 -1
  4. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/__init__.py +38 -4
  5. spot_sdk_python-2.0.0b2/spot_sdk/_json.py +46 -0
  6. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/email.py +5 -3
  7. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/llm.py +4 -13
  8. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/mime.py +461 -118
  9. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/app.py +0 -0
  10. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/clients.py +0 -0
  11. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/config.py +0 -0
  12. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/errors.py +0 -0
  13. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/knowledge.py +0 -0
  14. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/knowledge_tags.py +0 -0
  15. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/logging.py +0 -0
  16. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/manifest.py +0 -0
  17. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/orchestrator.py +0 -0
  18. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/py.typed +0 -0
  19. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/retriever.py +0 -0
  20. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/settings.py +0 -0
  21. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/signals.py +0 -0
  22. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/testing/README.md +0 -0
  23. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/testing/__init__.py +0 -0
  24. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/testing/contract.py +0 -0
  25. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/testing/fake_knowledge_client.py +0 -0
  26. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/testing/fake_llm.py +0 -0
  27. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/testing/fake_spot.py +0 -0
  28. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/testing/views.py +0 -0
  29. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/verdict.py +0 -0
  30. {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/workflow.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: spot-sdk-python
3
- Version: 2.0.0b1
3
+ Version: 2.0.0b2
4
4
  Summary: SPOT plugin SDK: the plugin contract, the plugin runtime, its clients and test helpers
5
5
  License: Apache-2.0
6
6
  Author: SPOT Project
@@ -49,7 +49,7 @@ plugins share:
49
49
  ## Install
50
50
 
51
51
  ```bash
52
- pip install spot-sdk-python==2.0.0b1
52
+ pip install spot-sdk-python==2.0.0b2
53
53
  ```
54
54
 
55
55
  Python 3.11 to 3.15. Version 2 implements plugin contract 2, for
@@ -26,7 +26,7 @@ plugins share:
26
26
  ## Install
27
27
 
28
28
  ```bash
29
- pip install spot-sdk-python==2.0.0b1
29
+ pip install spot-sdk-python==2.0.0b2
30
30
  ```
31
31
 
32
32
  Python 3.11 to 3.15. Version 2 implements plugin contract 2, for
@@ -4,7 +4,7 @@ build-backend = "poetry.core.masonry.api"
4
4
 
5
5
  [tool.poetry]
6
6
  name = "spot-sdk-python"
7
- version = "2.0.0b1"
7
+ version = "2.0.0b2"
8
8
  description = "SPOT plugin SDK: the plugin contract, the plugin runtime, its clients and test helpers"
9
9
  authors = ["SPOT Project <spot@sonn.lu>"]
10
10
  license = "Apache-2.0"
@@ -1,13 +1,12 @@
1
1
  """SPOT SDK: the plugin contract v2, the plugin runtime and its clients."""
2
2
 
3
- from .app import SyncState, create_plugin_app
4
- from .clients import SpotAPIError, SpotClient
3
+ from importlib import import_module
4
+ from typing import TYPE_CHECKING, Any
5
+
5
6
  from .config import ConfigOption, ConfigStatus, ConfigUpdateRequest
6
7
  from .email import Address, Anomaly, AuthResult, EmailView, Envelope, Header, Part, Url
7
8
  from .errors import ErrorResponse
8
- from .knowledge import KnowledgeClient, KnowledgeDocument, chunk_text, content_hash
9
9
  from .knowledge_tags import KnowledgeTag
10
- from .llm import LLMClient, LLMOutputError, LLMToolCall, LLMToolCalls, LLMToolRound
11
10
  from .logging import configure_logging, get_logger, get_logging_config
12
11
  from .manifest import (
13
12
  CONTRACT_VERSION,
@@ -42,6 +41,41 @@ from .workflow import (
42
41
  WorkflowStage,
43
42
  )
44
43
 
44
+ if TYPE_CHECKING:
45
+ from .app import SyncState, create_plugin_app
46
+ from .clients import SpotAPIError, SpotClient
47
+ from .knowledge import KnowledgeClient, KnowledgeDocument, chunk_text, content_hash
48
+ from .llm import LLMClient, LLMOutputError, LLMToolCall, LLMToolCalls, LLMToolRound
49
+
50
+ # The modules that import FastAPI, Starlette or httpx load on first use, so importing
51
+ # `spot_sdk.mime` or the models stays light (PEP 562).
52
+ _LAZY = {
53
+ **dict.fromkeys(("SyncState", "create_plugin_app"), ".app"),
54
+ **dict.fromkeys(("SpotAPIError", "SpotClient"), ".clients"),
55
+ **dict.fromkeys(
56
+ ("KnowledgeClient", "KnowledgeDocument", "chunk_text", "content_hash"),
57
+ ".knowledge",
58
+ ),
59
+ **dict.fromkeys(
60
+ ("LLMClient", "LLMOutputError", "LLMToolCall", "LLMToolCalls", "LLMToolRound"),
61
+ ".llm",
62
+ ),
63
+ }
64
+
65
+
66
+ def __getattr__(name: str) -> Any:
67
+ module = _LAZY.get(name)
68
+ if module is None:
69
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
70
+ value = getattr(import_module(module, __name__), name)
71
+ globals()[name] = value
72
+ return value
73
+
74
+
75
+ def __dir__() -> list[str]:
76
+ return sorted({*globals(), *_LAZY})
77
+
78
+
45
79
  __all__ = [
46
80
  # Email view
47
81
  "Address",
@@ -0,0 +1,46 @@
1
+ """Strings measured and cut by the size of their JSON.
2
+
3
+ The view and the LLM client's rendering are sent as JSON, where escapes make
4
+ a string larger than its text: a control character takes six bytes
5
+ (``\\u0001``). ``json.dumps`` with ``ensure_ascii=False`` and pydantic's
6
+ ``model_dump_json`` escape strings alike, so sizes here are theirs.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import json
12
+
13
+ _MAX_CHARACTER_BYTES = 6
14
+ """The most bytes one character takes: a control character's escape."""
15
+
16
+
17
+ def json_size(text: str) -> int:
18
+ """The UTF-8 bytes of ``text``'s JSON string between its quotes, escapes
19
+ included; a lone surrogate (possible in a string built in Python) counts
20
+ as three."""
21
+ escaped = json.dumps(text, ensure_ascii=False)[1:-1]
22
+ return len(escaped.encode("utf-8", "surrogatepass"))
23
+
24
+
25
+ def cut_json(text: str, max_bytes: int) -> tuple[str, bool]:
26
+ """The longest start of ``text`` whose JSON string takes at most
27
+ ``max_bytes`` bytes between its quotes (``json_size``), and whether that
28
+ cut anything.
29
+
30
+ A character is kept or dropped whole, so neither an escape nor a
31
+ character's UTF-8 bytes are split. The text is measured in chunks that
32
+ surely fit, a sixth of the room left each, then character by character:
33
+ the work follows ``max_bytes``, whatever the text's length.
34
+ """
35
+ if len(text) * _MAX_CHARACTER_BYTES <= max_bytes:
36
+ return text, False # it surely fits
37
+ kept = 0
38
+ room = max_bytes
39
+ while kept < len(text):
40
+ chunk = text[kept : kept + max(room // _MAX_CHARACTER_BYTES, 1)]
41
+ used = json_size(chunk)
42
+ if used > room: # one character, which does not fit
43
+ break
44
+ kept += len(chunk)
45
+ room -= used
46
+ return text[:kept], kept < len(text)
@@ -88,11 +88,12 @@ class Anomaly(BaseModel):
88
88
  code: str = Field(description="One of the spot_sdk.mime anomaly codes")
89
89
  location: str | None = Field(
90
90
  default=None,
91
- description='"part:1.2", "header:Subject@1", "param:Content-Type.boundary@1.2", ...',
91
+ description='"part:1.2", "header:Subject@1", "param:Content-Type.boundary@1.2",'
92
+ ' ..., "envelope", "source": at most 200 bytes of JSON',
92
93
  )
93
94
  detail: str | None = Field(
94
95
  default=None,
95
- description="The values concerned, as JSON strings: at most 200 characters",
96
+ description="The values concerned, as JSON strings: at most 200 bytes of JSON",
96
97
  )
97
98
 
98
99
 
@@ -118,7 +119,8 @@ class EmailView(BaseModel):
118
119
  sha256: str = Field(description="Of the raw message")
119
120
  envelope: Envelope | None = None
120
121
  headers: list[Header] = Field(
121
- description="Every header, in order, X-SPOT-* removed"
122
+ description="The headers in order, within their budget, X-SPOT-* removed;"
123
+ " those left out are noted as header_limit"
122
124
  )
123
125
  subject: str = ""
124
126
  from_: Address | None = Field(default=None, alias="from")
@@ -35,7 +35,6 @@ from __future__ import annotations
35
35
 
36
36
  import json
37
37
  import secrets
38
- from bisect import bisect_right
39
38
  from collections.abc import Mapping, Sequence
40
39
  from datetime import UTC, datetime
41
40
  from typing import Any, Self, TypeVar, overload
@@ -51,6 +50,7 @@ from pydantic import (
51
50
  model_validator,
52
51
  )
53
52
 
53
+ from ._json import cut_json
54
54
  from .clients import json_body
55
55
  from .email import EmailView
56
56
 
@@ -417,18 +417,9 @@ def _cap(value: Any) -> tuple[Any, bool]:
417
417
 
418
418
 
419
419
  def _cap_text(text: str) -> tuple[str, bool]:
420
- """The longest start of ``text`` whose JSON form, as ``json.dumps`` writes
421
- it with escapes, takes at most ``_STRING_BYTES`` UTF-8 bytes between its
422
- quotes, and whether that cut anything. A character is kept or dropped
423
- whole, a lone surrogate counting as three bytes. The work is bounded
424
- whatever the text's length: no character takes less than a byte."""
425
- head = text[:_STRING_BYTES]
426
- kept = bisect_right(
427
- range(1, len(head) + 1),
428
- _STRING_BYTES,
429
- key=lambda end: _size(json.dumps(head[:end], ensure_ascii=False)) - 2,
430
- )
431
- return head[:kept], kept < len(text)
420
+ """``text`` cut to ``_STRING_BYTES`` of JSON (``cut_json``), and whether
421
+ that cut anything."""
422
+ return cut_json(text, _STRING_BYTES)
432
423
 
433
424
 
434
425
  _IDENTITY_FIELDS = ("from", "reply_to", "subject")
@@ -21,7 +21,9 @@ Anyone can send a message, so the parser never fails on its content:
21
21
  header values are cut at ``max_text_bytes``, and at most ``MAX_URLS`` URLs
22
22
  are kept. Each bound hit is an anomaly; the ones that leave content out set
23
23
  ``EmailView.truncated``. At most ``MAX_ANOMALIES`` anomalies are recorded,
24
- ``MAX_ANOMALIES_PER_CODE`` of each code, but always the first of each.
24
+ ``MAX_ANOMALIES_PER_CODE`` of each code, but always the first of each;
25
+ - the view is bounded: its JSON takes at most ``MAX_VIEW_BYTES``, the sum of
26
+ the budgets of its lists and long strings (see :func:`parse`).
25
27
 
26
28
  The standard library builds the MIME tree with the ``compat32`` policy, whose
27
29
  headers are plain strings: the ``default`` policy parses each header it reads
@@ -41,7 +43,7 @@ import json
41
43
  import math
42
44
  import re
43
45
  from array import array
44
- from collections import Counter
46
+ from collections import Counter, defaultdict
45
47
  from collections.abc import Collection, Iterable, Iterator
46
48
  from dataclasses import dataclass
47
49
  from datetime import datetime
@@ -49,13 +51,16 @@ from email.message import Message
49
51
  from email.policy import Compat32, Policy, compat32
50
52
  from email.utils import decode_params, getaddresses, parsedate_to_datetime, unquote
51
53
  from html import unescape as unescape_html
52
- from itertools import chain, islice
53
- from typing import Any, Literal, NamedTuple, TypeAlias, cast
54
+ from itertools import chain
55
+ from typing import Any, Literal, NamedTuple, TypeAlias, TypeVar, cast
54
56
  from urllib.parse import urlsplit
55
57
  from uuid import UUID
56
58
 
59
+ from pydantic import BaseModel
60
+ from pydantic_core import to_json
57
61
  from selectolax.lexbor import LexborHTMLParser, LexborNode
58
62
 
63
+ from ._json import cut_json, json_size
59
64
  from .email import (
60
65
  Address,
61
66
  Anomaly,
@@ -77,6 +82,33 @@ MAX_BOUNDARY_LENGTH = 1_000
77
82
  MAX_ANOMALIES = 1_000
78
83
  MAX_ANOMALIES_PER_CODE = 100
79
84
  MAX_URLS = 10_000
85
+ # The view's budgets, in bytes of its JSON as model_dump_json(by_alias=True)
86
+ # writes it: UTF-8, escapes included. A list keeps its first entries within
87
+ # its budget (the headers each one that fits), a string its start; each cut is
88
+ # an anomaly.
89
+ MAX_TEXT_BYTES = 1 << 20 # the text, the HTML and the visible text, each
90
+ MAX_HEADERS_BYTES = 1 << 20
91
+ MAX_PARTS_BYTES = 1 << 20
92
+ MAX_URLS_BYTES = 1 << 20
93
+ _URL_BYTES = 8 << 10 # each URL
94
+ _SUBJECT_BYTES = 16 << 10
95
+ # Each other string of an entry or a field: a part's, a URL's host and anchor
96
+ # text, an address's, a message id, a reference, an authentication result's,
97
+ # the envelope's, the source.
98
+ _STRING_BYTES = 1 << 10
99
+ # Each list of a field: reply_to, to, cc, references, authentication, and the
100
+ # envelope's rcpt_to.
101
+ _LIST_BYTES = 64 << 10
102
+ # An anomaly's location and detail, each; the detail reads that many
103
+ # characters of each value.
104
+ _MAX_DETAIL = 200
105
+ # The fields read from the headers, the envelope and the source: the subject,
106
+ # six lists, and ten strings (the sender's three, two message ids, the
107
+ # envelope's four, the source).
108
+ MAX_FIELDS_BYTES = _SUBJECT_BYTES + 6 * _LIST_BYTES + 10 * _STRING_BYTES
109
+ # The keys, the punctuation and the fields of fixed size: the id, the dates,
110
+ # the sizes, the digest and truncated.
111
+ _FIXED_BYTES = 1 << 10
80
112
  # The HTML budget, checked before the tree is built (_html_budget). Lexbor, as
81
113
  # the HTML standard, never caps the tree, and some of its work is quadratic:
82
114
  # each tag or comment makes nodes to walk; a tag may walk the whole stack of
@@ -124,11 +156,19 @@ boundaries: mail clients may split it either way."""
124
156
  DEPTH_LIMIT = "depth_limit"
125
157
  """Nesting goes deeper than MAX_DEPTH: the deeper parts are not read."""
126
158
  HEADER_LIMIT = "header_limit"
127
- """A part has more than MAX_HEADERS headers: the later ones are not read."""
159
+ """A part has more than MAX_HEADERS headers: the later ones are not kept, but
160
+ the view's fields are still read from them. Or a header does not fit the room
161
+ left in MAX_HEADERS_BYTES: it is left out of the view's headers, but the
162
+ fields read from it are not."""
128
163
  PART_LIMIT = "part_limit"
129
- """The message has more than MAX_PARTS parts: the later ones are not read."""
164
+ """The message has more than MAX_PARTS parts: the later ones are not read. Or
165
+ its parts take more than MAX_PARTS_BYTES: the later ones are left out of the
166
+ view's parts, but they are read."""
130
167
  TEXT_TRUNCATED = "text_truncated"
131
- """The text, the HTML or a header value was cut at max_text_bytes."""
168
+ """A string was cut at its budget: the text, the HTML or the visible text at
169
+ max_text_bytes of JSON (noted at their part), a header value at max_text_bytes
170
+ bytes or a field read from it (at the header), a string of a part (at the
171
+ part), the envelope or the source."""
132
172
  HEADER_DUPLICATE = "header_duplicate"
133
173
  """A part has more than one Content-Type, Content-Disposition,
134
174
  Content-Transfer-Encoding or MIME-Version header: the first one counts."""
@@ -150,7 +190,9 @@ HTML_STYLE_UNRESOLVED = "html_style_unresolved"
150
190
  known when rendering (var(), env(), attr(), a calc() that is not a literal):
151
191
  the text was kept visible. The detail names the properties."""
152
192
  URL_LIMIT = "url_limit"
153
- """The message has more than MAX_URLS URLs: the later ones are not recorded."""
193
+ """The message has more than MAX_URLS URLs, or they take more than
194
+ MAX_URLS_BYTES: the later ones are not recorded. Or a URL was cut at 8 KiB of
195
+ JSON, or its host or anchor text at 1 KiB."""
154
196
  ANOMALY_LIMIT = "anomaly_limit"
155
197
  """More than MAX_ANOMALIES anomalies were met: the later ones are not recorded."""
156
198
  ANOMALY_CODES = (
@@ -175,6 +217,30 @@ ANOMALY_CODES = (
175
217
  )
176
218
  """Every anomaly code, in the order above."""
177
219
 
220
+ # The anomalies, and the one that says more were met: each with the longest
221
+ # code, a location and a detail, and a comma.
222
+ MAX_ANOMALIES_BYTES = (MAX_ANOMALIES + 1) * (
223
+ len(
224
+ Anomaly(
225
+ code=max(ANOMALY_CODES, key=len), location="", detail=""
226
+ ).model_dump_json()
227
+ )
228
+ + 2 * _MAX_DETAIL
229
+ + 1
230
+ ) + 1
231
+ # The most bytes the view's JSON takes, whatever the message, with
232
+ # max_text_bytes at most its default: the sum of the budgets, about 6.84 MiB.
233
+ # Core refuses a view over 8 MiB.
234
+ MAX_VIEW_BYTES = (
235
+ _FIXED_BYTES
236
+ + 3 * MAX_TEXT_BYTES
237
+ + MAX_HEADERS_BYTES
238
+ + MAX_PARTS_BYTES
239
+ + MAX_URLS_BYTES
240
+ + MAX_FIELDS_BYTES
241
+ + MAX_ANOMALIES_BYTES
242
+ )
243
+
178
244
  _ENCODED_WORD = re.compile(r"=\?([^?\s]+)\?([bq])\?([^?\s]*)\?=", re.IGNORECASE)
179
245
  _URL = re.compile(r"\b(?:https?://|www\.)[^\s<>\"'`]+", re.IGNORECASE)
180
246
  _URL_TRAILER = ".,;:!?)]}'\""
@@ -205,7 +271,10 @@ _BASE64_RUN = re.compile(rb"([A-Za-z0-9+/]+)|(=+)")
205
271
  _STRUCTURAL_HEADERS = frozenset(
206
272
  {"content-type", "content-disposition", "content-transfer-encoding", "mime-version"}
207
273
  )
208
- _MAX_DETAIL = 200
274
+ # The headers the view reads a field from: the first of each name.
275
+ _FIELD_HEADERS = frozenset(
276
+ "subject from reply-to to cc date message-id in-reply-to references".split()
277
+ )
209
278
  _DISPOSITIONS: dict[str | None, Literal["inline", "attachment"]] = {
210
279
  "inline": "inline",
211
280
  "attachment": "attachment",
@@ -422,21 +491,56 @@ def parse(
422
491
  received_at: datetime,
423
492
  envelope: Envelope | None = None,
424
493
  trusted_authserv_ids: Collection[str] = (),
425
- max_text_bytes: int = 1 << 20,
494
+ max_text_bytes: int = MAX_TEXT_BYTES,
426
495
  ) -> EmailView:
427
496
  """Parse a raw RFC 5322 message into the email view.
428
497
 
429
498
  ``trusted_authserv_ids`` are the authserv-ids of our own MTAs: only their
430
499
  ``Authentication-Results`` count. ``max_text_bytes`` caps the text, the
431
- HTML and each header value.
500
+ HTML and the visible text, each in bytes of its JSON (escapes included),
501
+ and each header value in bytes. The text and the HTML are read from at
502
+ most that many bytes of their part, and their URLs and the visible text
503
+ from all of what was read, not from the cut the view stores.
504
+
505
+ The view's JSON (``model_dump_json(by_alias=True)``, in UTF-8) takes at
506
+ most ``MAX_VIEW_BYTES``, about 6.84 MiB, whatever the message, with
507
+ ``max_text_bytes`` at most its default. Each list and each long string has
508
+ a budget, in bytes of its JSON, escapes included:
509
+
510
+ - the text, the HTML and the visible text: ``MAX_TEXT_BYTES`` (1 MiB) each;
511
+ - the headers, the parts and the URLs: ``MAX_HEADERS_BYTES``,
512
+ ``MAX_PARTS_BYTES`` and ``MAX_URLS_BYTES`` (1 MiB each); each URL 8 KiB,
513
+ and each other string of a part or a URL 1 KiB;
514
+ - the fields read from the headers, the envelope and the source:
515
+ ``MAX_FIELDS_BYTES`` (410 KiB). The subject takes 16 KiB, each other
516
+ string 1 KiB, and each list (``reply_to``, ``to``, ``cc``,
517
+ ``references``, ``authentication``, the envelope's ``rcpt_to``) 64 KiB;
518
+ - the anomalies: ``MAX_ANOMALIES_BYTES`` (about 450 KiB), each location
519
+ and detail 200 bytes;
520
+ - the keys and the fields of fixed size: 1 KiB.
521
+
522
+ A list keeps its first entries, but the headers each one that fits the
523
+ room left; a string keeps its start, and no value is altered. The fields
524
+ read from the headers come from the first header of each name in the
525
+ whole header block, past ``MAX_HEADERS`` too, whatever headers the list
526
+ keeps. Each cut sets ``truncated`` and is an anomaly: ``text_truncated``
527
+ for a string, ``header_limit``, ``part_limit`` or ``url_limit`` for a
528
+ list (a URL cut is ``url_limit`` too).
432
529
  """
433
530
  root = _read(raw)
434
531
  nodes, left_out = _walk(root, budget=len(raw))
435
532
  _note_ambiguous_boundaries(nodes)
436
533
  fields, cut_headers = _fields(root, max_text_bytes)
437
- first: dict[str, str] = {}
438
- for name, value in fields:
439
- first.setdefault(name.lower(), value)
534
+ read, cut_from = _header_fields(
535
+ root.firsts,
536
+ fields,
537
+ {authserv_id.lower() for authserv_id in trusted_authserv_ids},
538
+ max_text_bytes,
539
+ )
540
+ given = {"source": _Cutter(), "envelope": _Cutter()}
541
+ source = given["source"](source)
542
+ if envelope is not None:
543
+ envelope = _envelope(envelope, given["envelope"])
440
544
 
441
545
  parts: list[Part] = []
442
546
  bodies: dict[str, _Body] = {}
@@ -451,44 +555,68 @@ def parse(
451
555
  and part.disposition != "attachment"
452
556
  and not node.enclosed
453
557
  ):
454
- content = _decode_text(data[:max_text_bytes], part.charset)
558
+ # The charset as written: a cut one may name another codec.
559
+ charset = _param(node.part, "charset")
560
+ content = _decode_text(data[:max_text_bytes], charset)
455
561
  bodies[part.content_type] = _Body(part.part_id, node.part, content)
456
562
  if len(data) > max_text_bytes:
457
- node.part.note(TEXT_TRUNCATED)
563
+ node.part.note_cut()
458
564
 
565
+ # The URLs and the visible text are read from the whole decoded bodies;
566
+ # the view stores each body cut to its budget.
459
567
  text = bodies.get("text/plain")
460
568
  html = bodies.get("text/html")
461
- found = [_header_urls(fields)] # the URLs, built as they are kept
569
+ cut_urls = _Cutter()
570
+ found = [_header_urls(fields, cut_urls)] # the URLs, built as they are kept
462
571
  if text:
463
- found.append(_urls_in(text.content, "text", text.part_id))
464
- visible_text = text.content if text and text.content.strip() else ""
572
+ found.append(_urls_in(text.content, "text", text.part_id, cut_urls))
573
+ shown = text if text and text.content.strip() else None # the visible text's
574
+ visible_text = shown.content if shown else ""
465
575
  if html:
466
576
  reader = _read_html(html.content)
467
- found.append(_html_urls(reader.links, html.part_id))
577
+ found.append(_html_urls(reader.links, html.part_id, cut_urls))
468
578
  html_text = reader.text
469
- found.append(_urls_in(html_text, "html_text", html.part_id))
470
- visible_text = visible_text or html_text
579
+ found.append(_urls_in(html_text, "html_text", html.part_id, cut_urls))
580
+ if shown is None:
581
+ shown, visible_text = html, html_text
471
582
  if reader.hidden_text:
472
583
  html.part.note(HTML_HIDDEN_TEXT)
473
584
  if reader.limit:
474
585
  html.part.note(HTML_LIMIT, None, reader.limit)
475
586
  if reader.unresolved:
476
587
  html.part.note(HTML_STYLE_UNRESOLVED, None, *sorted(reader.unresolved))
477
- # Each note with the id of its part: the header ones are the message's.
478
- notes = [(_Note(TEXT_TRUNCATED, subject), "1") for subject in cut_headers]
588
+ stored_text = _cut_body(text, text.content, max_text_bytes) if text else None
589
+ stored_html = _cut_body(html, html.content, max_text_bytes) if html else None
590
+ if shown:
591
+ visible_text = _cut_body(shown, visible_text, max_text_bytes)
592
+ # Each note with the id of its part: the header ones are the message's,
593
+ # once per header; the source and the envelope are no part's.
594
+ notes: list[tuple[_Note, str | None]] = [
595
+ (_Note(TEXT_TRUNCATED, subject), "1")
596
+ for subject in dict.fromkeys(cut_headers + cut_from)
597
+ ]
598
+ notes += [
599
+ (_Note(TEXT_TRUNCATED, name), None) for name, cut in given.items() if cut.cut
600
+ ]
479
601
  notes += [(note, node.part_id) for node in nodes for note in node.part.notes]
602
+ parts, left_out_part = _listed(parts, MAX_PARTS_BYTES)
603
+ if left_out_part is not None:
604
+ notes.append((_Note(PART_LIMIT), left_out_part.part_id))
480
605
  if left_out:
481
606
  notes.append((_Note(PART_LIMIT), left_out))
482
- urls = list(islice(chain.from_iterable(found), MAX_URLS + 1))
483
- if len(urls) > MAX_URLS:
484
- urls.pop()
607
+ urls, left_out_url = _listed(
608
+ chain.from_iterable(found), MAX_URLS_BYTES, limit=MAX_URLS
609
+ )
610
+ if left_out_url is not None or cut_urls.cut:
485
611
  notes.append((_Note(URL_LIMIT), "1"))
612
+ headers, header_left_out = _headers(fields)
613
+ if header_left_out:
614
+ notes.append((_Note(HEADER_LIMIT), "1"))
486
615
  kept = _kept(notes)
487
616
  anomalies = [_anomaly(note, part_id) for note, part_id in kept]
488
617
  if len(kept) < len(notes):
489
618
  anomalies.append(Anomaly(code=ANOMALY_LIMIT))
490
619
 
491
- senders = _addresses(first.get("from", ""))
492
620
  return EmailView(
493
621
  id=email_id,
494
622
  received_at=received_at,
@@ -496,26 +624,20 @@ def parse(
496
624
  size_bytes=len(raw),
497
625
  sha256=hashlib.sha256(raw).hexdigest(),
498
626
  envelope=envelope,
499
- headers=[
500
- Header(name=name, value=_decode_words(value))
501
- for name, value in fields
502
- if not name.lower().startswith("x-spot-")
503
- ],
504
- subject=_decode_words(first.get("subject", "")),
505
- from_=senders[0] if senders else None,
506
- reply_to=_addresses(first.get("reply-to", "")),
507
- to=_addresses(first.get("to", "")),
508
- cc=_addresses(first.get("cc", "")),
509
- date=_date(first.get("date", "")),
510
- message_id=first.get("message-id", "").strip() or None,
511
- in_reply_to=first.get("in-reply-to", "").strip() or None,
512
- references=_MESSAGE_ID.findall(first.get("references", "")),
513
- authentication=_authentication(
514
- fields, {authserv_id.lower() for authserv_id in trusted_authserv_ids}
515
- ),
627
+ headers=headers,
628
+ subject=read.subject,
629
+ from_=read.from_,
630
+ reply_to=read.reply_to,
631
+ to=read.to,
632
+ cc=read.cc,
633
+ date=read.date,
634
+ message_id=read.message_id,
635
+ in_reply_to=read.in_reply_to,
636
+ references=read.references,
637
+ authentication=read.authentication,
516
638
  parts=parts,
517
- text=text.content if text else None,
518
- html=html.content if html else None,
639
+ text=stored_text,
640
+ html=stored_html,
519
641
  visible_text=visible_text,
520
642
  urls=urls,
521
643
  truncated=any(note.code in _LIMIT_CODES for note, _ in notes),
@@ -536,6 +658,86 @@ def part_bytes(raw: bytes, part_id: str) -> tuple[bytes, str]:
536
658
  raise PartNotFoundError(part_id)
537
659
 
538
660
 
661
+ # ----------------------------------------------------------------------- #
662
+ # The view's budgets
663
+ # ----------------------------------------------------------------------- #
664
+
665
+
666
+ _Entry = TypeVar("_Entry", bound=BaseModel | str)
667
+
668
+
669
+ def _listed(
670
+ entries: Iterable[_Entry], budget: int, limit: int | None = None
671
+ ) -> tuple[list[_Entry], _Entry | None]:
672
+ """The first ``entries``, at most ``limit``, whose JSON array takes at
673
+ most ``budget`` bytes; and the first one left out, if any. Each entry is
674
+ measured as it comes, so the work stops at the budget."""
675
+ kept: list[_Entry] = []
676
+ size = 1 # "[", then each entry and the "," or "]" after it
677
+ for entry in entries:
678
+ size += _size(entry) + 1
679
+ if size > budget or len(kept) == limit:
680
+ return kept, entry
681
+ kept.append(entry)
682
+ return kept, None
683
+
684
+
685
+ def _headers(fields: list[tuple[str, str]]) -> tuple[list[Header], bool]:
686
+ """The headers in order, without ``X-SPOT-*``, each kept if it fits the
687
+ room left in ``MAX_HEADERS_BYTES``; and whether one was left out.
688
+
689
+ A header that does not fit is left out alone, so one long header does not
690
+ hide the ones after it. A character takes a byte at least: a header whose
691
+ name and decoded value are longer than the room left is left out without
692
+ measuring it.
693
+ """
694
+ headers: list[Header] = []
695
+ left_out = False
696
+ room = MAX_HEADERS_BYTES - 1 # "[", then each entry and the "," or "]" after it
697
+ for name, value in fields:
698
+ if name.lower().startswith("x-spot-"):
699
+ continue
700
+ value = _decode_words(value)
701
+ if len(name) + len(value) < room:
702
+ header = Header(name=name, value=value)
703
+ size = _size(header) + 1
704
+ if size <= room:
705
+ headers.append(header)
706
+ room -= size
707
+ continue
708
+ left_out = True
709
+ return headers, left_out
710
+
711
+
712
+ class _Cutter:
713
+ """Cuts strings to their budget, and remembers whether it cut one."""
714
+
715
+ def __init__(self) -> None:
716
+ self.cut = False
717
+
718
+ def __call__(self, text: str, max_bytes: int = _STRING_BYTES) -> str:
719
+ text, cut = cut_json(text, max_bytes)
720
+ self.cut = self.cut or cut
721
+ return text
722
+
723
+ def optional(self, text: str | None) -> str | None:
724
+ return None if text is None else self(text)
725
+
726
+ def listed(self, entries: Iterable[_Entry]) -> list[_Entry]:
727
+ """The first entries within ``_LIST_BYTES``; a list cut short counts as
728
+ a cut."""
729
+ kept, left_out = _listed(entries, _LIST_BYTES)
730
+ self.cut = self.cut or left_out is not None
731
+ return kept
732
+
733
+
734
+ def _size(value: BaseModel | str) -> int:
735
+ """The bytes of a value's JSON, as the view's."""
736
+ if isinstance(value, str):
737
+ return json_size(value) + 2 # and its quotes
738
+ return len(to_json(value, by_alias=True))
739
+
740
+
539
741
  # ----------------------------------------------------------------------- #
540
742
  # The MIME tree
541
743
  # ----------------------------------------------------------------------- #
@@ -556,14 +758,24 @@ class _Part(Message):
556
758
  candidates: list[str] | None = None # usable boundaries, preferred first
557
759
  raw_body: str | None = None # as read, raw bytes as surrogate escapes
558
760
  headers_dropped = False
761
+ content_cut = False
559
762
 
560
763
  def __init__(self, policy: Policy[Any] = compat32) -> None:
561
764
  super().__init__(policy)
562
765
  self.notes: list[_Note] = []
766
+ # By lower-cased name, the first header of each _FIELD_HEADERS name as
767
+ # read (name, raw value): past MAX_HEADERS too.
768
+ self.firsts: dict[str, tuple[str, str]] = {}
563
769
 
564
770
  def note(self, code: str, subject: str | None = None, *values: str) -> None:
565
771
  self.notes.append(_Note(code, subject, values))
566
772
 
773
+ def note_cut(self) -> None:
774
+ """Note, once, that content read from this part was cut at its budget."""
775
+ if not self.content_cut:
776
+ self.content_cut = True
777
+ self.note(TEXT_TRUNCATED)
778
+
567
779
  def attach(self, payload: Message | str) -> None:
568
780
  if isinstance(payload, _Part):
569
781
  payload.depth = self.depth + 1
@@ -576,6 +788,9 @@ class _Part(Message):
576
788
  super().set_payload(payload, charset)
577
789
 
578
790
  def set_raw(self, name: str, value: str) -> None:
791
+ key = name.lower()
792
+ if key in _FIELD_HEADERS:
793
+ self.firsts.setdefault(key, (name, value))
579
794
  if len(self) < MAX_HEADERS:
580
795
  super().set_raw(name, value)
581
796
  elif not self.headers_dropped:
@@ -662,6 +877,15 @@ class _Body:
662
877
  content: str
663
878
 
664
879
 
880
+ def _cut_body(body: _Body, content: str, max_bytes: int) -> str:
881
+ """``content`` read from a body, cut to ``max_bytes`` of JSON; a cut is
882
+ noted on the body's part."""
883
+ content, cut = cut_json(content, max_bytes)
884
+ if cut:
885
+ body.part.note_cut()
886
+ return content
887
+
888
+
665
889
  def _read(raw: bytes) -> _Part:
666
890
  return cast(_Part, email.message_from_bytes(raw, policy=_POLICY))
667
891
 
@@ -832,22 +1056,27 @@ def _decode_base64(text: str) -> tuple[bytes, str]:
832
1056
 
833
1057
 
834
1058
  def _describe(node: _Node, data: bytes) -> Part:
1059
+ """A part as the view lists it, each string cut to ``_STRING_BYTES``."""
835
1060
  part = node.part
836
1061
  charset = _param(part, "charset")
837
1062
  filename = _param(part, "filename", "content-disposition")
838
1063
  name = _param(part, "name")
839
1064
  if filename and name and filename != name:
840
1065
  part.note(PARAM_CONFLICT, "param:Content-Disposition.filename", filename, name)
841
- return Part(
1066
+ cut = _Cutter()
1067
+ described = Part(
842
1068
  part_id=node.part_id,
843
- content_type=_content_type(part),
844
- charset=charset.lower() if charset else None,
1069
+ content_type=cut(_content_type(part)),
1070
+ charset=cut.optional(charset.lower() if charset else None),
845
1071
  disposition=_DISPOSITIONS.get(part.get_content_disposition()),
846
- filename=filename or name,
847
- content_id=_clean(part.get("content-id") or "").strip() or None,
1072
+ filename=cut.optional(filename or name),
1073
+ content_id=cut.optional(_clean(part.get("content-id") or "").strip() or None),
848
1074
  size_bytes=len(data),
849
1075
  sha256=None if part.is_multipart() else hashlib.sha256(data).hexdigest(),
850
1076
  )
1077
+ if cut.cut:
1078
+ part.note_cut()
1079
+ return described
851
1080
 
852
1081
 
853
1082
  def _content_type(part: _Part) -> str:
@@ -1081,15 +1310,23 @@ def _fields(message: _Part, limit: int) -> tuple[list[tuple[str, str]], list[str
1081
1310
  for name, raw_value in message.items():
1082
1311
  key = name.lower()
1083
1312
  seen[key] += 1
1084
- value = _clean(raw_value)
1085
- data = value.encode()
1086
- if len(data) > limit:
1087
- value = data[:limit].decode(errors="ignore")
1313
+ value, was_cut = _header_value(raw_value, limit)
1314
+ if was_cut:
1088
1315
  cut.append(_header_subject(name, seen[key]))
1089
1316
  fields.append((name, value))
1090
1317
  return fields, cut
1091
1318
 
1092
1319
 
1320
+ def _header_value(raw_value: str, limit: int) -> tuple[str, bool]:
1321
+ """A header value as the view reads it, cut at ``limit`` bytes; and
1322
+ whether it was cut."""
1323
+ value = _clean(raw_value)
1324
+ data = value.encode()
1325
+ if len(data) > limit:
1326
+ return data[:limit].decode(errors="ignore"), True
1327
+ return value, False
1328
+
1329
+
1093
1330
  def _note_repeated_headers(part: _Part) -> None:
1094
1331
  """Note each structural header a part gives again: the first one counts."""
1095
1332
  seen: Counter[str] = Counter()
@@ -1109,7 +1346,9 @@ def _header_subject(name: str, occurrence: int) -> str:
1109
1346
  return f"header:{name}[{occurrence}]" if occurrence > 1 else f"header:{name}"
1110
1347
 
1111
1348
 
1112
- def _kept(notes: list[tuple[_Note, str]]) -> list[tuple[_Note, str]]:
1349
+ def _kept(
1350
+ notes: list[tuple[_Note, str | None]],
1351
+ ) -> list[tuple[_Note, str | None]]:
1113
1352
  """The notes recorded, in order: the first of each code, then the others
1114
1353
  while their code has fewer than ``MAX_ANOMALIES_PER_CODE``, and all fewer
1115
1354
  than ``MAX_ANOMALIES``. Cheap, common notes do not push out rare ones."""
@@ -1118,7 +1357,7 @@ def _kept(notes: list[tuple[_Note, str]]) -> list[tuple[_Note, str]]:
1118
1357
  first.setdefault(note.code, index)
1119
1358
  room = MAX_ANOMALIES - len(first) # for the notes after the first of a code
1120
1359
  counts: Counter[str] = Counter()
1121
- kept: list[tuple[_Note, str]] = []
1360
+ kept: list[tuple[_Note, str | None]] = []
1122
1361
  for index, (note, part_id) in enumerate(notes):
1123
1362
  if index != first[note.code]:
1124
1363
  if not room or counts[note.code] >= MAX_ANOMALIES_PER_CODE:
@@ -1129,18 +1368,20 @@ def _kept(notes: list[tuple[_Note, str]]) -> list[tuple[_Note, str]]:
1129
1368
  return kept
1130
1369
 
1131
1370
 
1132
- def _anomaly(note: _Note, part_id: str) -> Anomaly:
1371
+ def _anomaly(note: _Note, part_id: str | None) -> Anomaly:
1133
1372
  """An anomaly as the view carries it: its values JSON-quoted in the detail."""
1134
1373
  detail = ", ".join(
1135
1374
  json.dumps(_clean(value[:_MAX_DETAIL]), ensure_ascii=False)
1136
1375
  for value in note.values
1137
1376
  )
1377
+ if part_id is None:
1378
+ location = note.subject
1379
+ else:
1380
+ location = f"{note.subject}@{part_id}" if note.subject else f"part:{part_id}"
1138
1381
  return Anomaly(
1139
1382
  code=note.code,
1140
- location=_clean(
1141
- f"{note.subject}@{part_id}" if note.subject else f"part:{part_id}"
1142
- ),
1143
- detail=detail[:_MAX_DETAIL] or None,
1383
+ location=cut_json(_clean(location), _MAX_DETAIL)[0] if location else None,
1384
+ detail=cut_json(detail, _MAX_DETAIL)[0] or None,
1144
1385
  )
1145
1386
 
1146
1387
 
@@ -1159,20 +1400,21 @@ def _domain(host: str) -> str:
1159
1400
  return host
1160
1401
 
1161
1402
 
1162
- def _addresses(value: str) -> list[Address]:
1163
- """The mailboxes of an address header; a value that does not parse is kept whole."""
1403
+ def _addresses(value: str, cut: _Cutter) -> Iterator[Address]:
1404
+ """The mailboxes of an address header, each string cut to its budget; a
1405
+ value that does not parse is kept whole."""
1164
1406
  mailboxes = getaddresses([value])
1165
1407
  if not mailboxes or not all("@" in address for _, address in mailboxes):
1166
1408
  value = value.strip()
1167
- return [Address(address=value)] if value else []
1168
- return [
1169
- Address(
1170
- display_name=_decode_words(name),
1171
- address=address,
1172
- domain=_domain(address.rpartition("@")[2]),
1409
+ if value:
1410
+ yield Address(address=cut(value))
1411
+ return
1412
+ for name, address in mailboxes:
1413
+ yield Address(
1414
+ display_name=cut(_decode_words(name)),
1415
+ address=cut(address),
1416
+ domain=cut(_domain(address.rpartition("@")[2])),
1173
1417
  )
1174
- for name, address in mailboxes
1175
- ]
1176
1418
 
1177
1419
 
1178
1420
  def _date(value: str) -> datetime | None:
@@ -1182,6 +1424,86 @@ def _date(value: str) -> datetime | None:
1182
1424
  return None
1183
1425
 
1184
1426
 
1427
+ class _HeaderFields(NamedTuple):
1428
+ """The view's fields read from the headers."""
1429
+
1430
+ subject: str
1431
+ from_: Address | None
1432
+ reply_to: list[Address]
1433
+ to: list[Address]
1434
+ cc: list[Address]
1435
+ date: datetime | None
1436
+ message_id: str | None
1437
+ in_reply_to: str | None
1438
+ references: list[str]
1439
+ authentication: list[AuthResult]
1440
+
1441
+
1442
+ def _header_fields(
1443
+ firsts: dict[str, tuple[str, str]],
1444
+ fields: list[tuple[str, str]],
1445
+ trusted: set[str],
1446
+ limit: int,
1447
+ ) -> tuple[_HeaderFields, list[str]]:
1448
+ """The view's fields read from the first header of each name (``firsts``,
1449
+ the whole header block's, each value cut at ``limit`` bytes), and the
1450
+ authentication results from the trusted headers on top: within their
1451
+ budgets. Each string is cut to ``_STRING_BYTES``, the subject to
1452
+ ``_SUBJECT_BYTES``, and each list to ``_LIST_BYTES``.
1453
+
1454
+ Also the headers a field was cut from, as ``header:`` subjects.
1455
+ """
1456
+ # By lower-cased name: the name as written, and the value.
1457
+ first: dict[str, tuple[str, str]] = {}
1458
+ cuts: defaultdict[str, _Cutter] = defaultdict(_Cutter) # by header
1459
+ for key, (name, raw_value) in firsts.items():
1460
+ value, cut = _header_value(_POLICY.header_fetch_parse(name, raw_value), limit)
1461
+ first[key] = (name, value)
1462
+ cuts[key].cut = cut
1463
+
1464
+ def header(name: str) -> str:
1465
+ return first.get(name, ("", ""))[1]
1466
+
1467
+ def addresses(name: str) -> list[Address]:
1468
+ return cuts[name].listed(_addresses(header(name), cuts[name]))
1469
+
1470
+ def message_id(name: str) -> str | None:
1471
+ return cuts[name](header(name).strip()) or None
1472
+
1473
+ authentication, authentication_cut = _authentication(fields, trusted)
1474
+ read = _HeaderFields(
1475
+ subject=cuts["subject"](_decode_words(header("subject")), _SUBJECT_BYTES),
1476
+ from_=next(_addresses(header("from"), cuts["from"]), None),
1477
+ reply_to=addresses("reply-to"),
1478
+ to=addresses("to"),
1479
+ cc=addresses("cc"),
1480
+ date=_date(header("date")),
1481
+ message_id=message_id("message-id"),
1482
+ in_reply_to=message_id("in-reply-to"),
1483
+ references=cuts["references"].listed(
1484
+ cuts["references"](reference.group())
1485
+ for reference in _MESSAGE_ID.finditer(header("references"))
1486
+ ),
1487
+ authentication=authentication,
1488
+ )
1489
+ cut_from = [
1490
+ _header_subject(first[name][0], 1) for name, cut in cuts.items() if cut.cut
1491
+ ]
1492
+ return read, cut_from + ([authentication_cut] if authentication_cut else [])
1493
+
1494
+
1495
+ def _envelope(envelope: Envelope, cut: _Cutter) -> Envelope:
1496
+ """The SMTP envelope within its budgets: each string cut to
1497
+ ``_STRING_BYTES``, ``rcpt_to`` to ``_LIST_BYTES``."""
1498
+ return Envelope(
1499
+ mail_from=cut.optional(envelope.mail_from),
1500
+ rcpt_to=cut.listed(cut(recipient) for recipient in envelope.rcpt_to),
1501
+ client_address=cut.optional(envelope.client_address),
1502
+ client_name=cut.optional(envelope.client_name),
1503
+ helo=cut.optional(envelope.helo),
1504
+ )
1505
+
1506
+
1185
1507
  # ----------------------------------------------------------------------- #
1186
1508
  # Authentication-Results (RFC 8601)
1187
1509
  # ----------------------------------------------------------------------- #
@@ -1189,46 +1511,68 @@ def _date(value: str) -> datetime | None:
1189
1511
 
1190
1512
  def _authentication(
1191
1513
  fields: list[tuple[str, str]], trusted: set[str]
1192
- ) -> list[AuthResult]:
1193
- """The results of the trusted headers above the first untrusted one.
1514
+ ) -> tuple[list[AuthResult], str | None]:
1515
+ """The results of the trusted headers above the first untrusted one,
1516
+ within ``_LIST_BYTES``, each string cut to ``_STRING_BYTES``; and the
1517
+ header the first cut is in, as a ``header:`` subject, if any.
1194
1518
 
1195
1519
  Our MTAs add their headers on top, so only that run is theirs: a trusted
1196
1520
  authserv-id below an untrusted one was written by someone else.
1197
1521
  """
1198
1522
  results: list[AuthResult] = []
1523
+ first_cut: str | None = None
1524
+ size = 1 # as _listed counts
1525
+ for subject, result, cut in _trusted_results(fields, trusted):
1526
+ size += _size(result) + 1
1527
+ if size > _LIST_BYTES:
1528
+ return results, first_cut or subject
1529
+ results.append(result)
1530
+ if cut and first_cut is None:
1531
+ first_cut = subject
1532
+ return results, first_cut
1533
+
1534
+
1535
+ def _trusted_results(
1536
+ fields: list[tuple[str, str]], trusted: set[str]
1537
+ ) -> Iterator[tuple[str, AuthResult, bool]]:
1538
+ """Each result of the trusted headers on top, with the header it is in (a
1539
+ ``header:`` subject) and whether one of its strings was cut."""
1540
+ occurrence = 0
1199
1541
  for name, value in fields:
1200
1542
  if name.lower() != "authentication-results":
1201
1543
  continue
1544
+ occurrence += 1
1202
1545
  head, *resinfos = _segments(value)
1203
1546
  words = head.split()
1204
1547
  authserv_id = _unquote(words[0]).lower() if words else ""
1205
1548
  if authserv_id not in trusted:
1206
- break
1207
- results += _auth_results(authserv_id, resinfos)
1208
- return results
1549
+ return
1550
+ for result, cut in _auth_results(authserv_id, resinfos):
1551
+ yield _header_subject(name, occurrence), result, cut
1209
1552
 
1210
1553
 
1211
- def _auth_results(authserv_id: str, resinfos: list[str]) -> list[AuthResult]:
1212
- """Each resinfo's ``method=result``, with its ``ptype.property=value`` pairs."""
1213
- results: list[AuthResult] = []
1554
+ def _auth_results(
1555
+ authserv_id: str, resinfos: list[str]
1556
+ ) -> Iterator[tuple[AuthResult, bool]]:
1557
+ """Each resinfo's ``method=result``, with its ``ptype.property=value``
1558
+ pairs, each string cut to ``_STRING_BYTES``; and whether one was."""
1214
1559
  for resinfo in resinfos:
1215
1560
  pairs = _AUTH_PAIR.findall(resinfo)
1216
1561
  if not pairs: # "none": no result
1217
1562
  continue
1218
1563
  (method, result), *properties = pairs
1219
- results.append(
1220
- AuthResult(
1221
- authserv_id=authserv_id,
1222
- method=method.partition("/")[0].lower(),
1223
- result=_unquote(result).lower(),
1224
- properties={
1225
- key.lower(): _unquote(pvalue)
1226
- for key, pvalue in properties
1227
- if "." in key
1228
- },
1229
- )
1564
+ cut = _Cutter()
1565
+ read = AuthResult(
1566
+ authserv_id=cut(authserv_id),
1567
+ method=cut(method.partition("/")[0].lower()),
1568
+ result=cut(_unquote(result).lower()),
1569
+ properties={
1570
+ cut(key.lower()): cut(_unquote(pvalue))
1571
+ for key, pvalue in properties
1572
+ if "." in key
1573
+ },
1230
1574
  )
1231
- return results
1575
+ yield read, cut.cut
1232
1576
 
1233
1577
 
1234
1578
  def _segments(value: str) -> list[str]:
@@ -1299,22 +1643,33 @@ def _host(url: str) -> str:
1299
1643
  return _domain(host) if host else ""
1300
1644
 
1301
1645
 
1646
+ def _url(
1647
+ url: str,
1648
+ found_in: Literal["text", "html_href", "html_text", "header"],
1649
+ part_id: str | None,
1650
+ cut: _Cutter,
1651
+ anchor_text: str | None = None,
1652
+ ) -> Url:
1653
+ """A URL as the view records it: cut to ``_URL_BYTES``, its host (of the
1654
+ whole URL) and anchor text to ``_STRING_BYTES``."""
1655
+ return Url(
1656
+ url=cut(url, _URL_BYTES),
1657
+ host=cut(_host(url)),
1658
+ found_in=found_in,
1659
+ part_id=part_id,
1660
+ anchor_text=cut.optional(anchor_text),
1661
+ )
1662
+
1663
+
1302
1664
  def _urls_in(
1303
- text: str, found_in: Literal["text", "html_text"], part_id: str
1665
+ text: str, found_in: Literal["text", "html_text"], part_id: str, cut: _Cutter
1304
1666
  ) -> Iterator[Url]:
1305
1667
  """The ``http(s)://`` and ``www.`` URLs written in a text."""
1306
1668
  for match in _URL.finditer(text):
1307
- url = match.group().rstrip(_URL_TRAILER)
1308
- yield Url(
1309
- url=url,
1310
- host=_host(url),
1311
- found_in=found_in,
1312
- part_id=part_id,
1313
- anchor_text=None,
1314
- )
1669
+ yield _url(match.group().rstrip(_URL_TRAILER), found_in, part_id, cut)
1315
1670
 
1316
1671
 
1317
- def _header_urls(fields: list[tuple[str, str]]) -> Iterator[Url]:
1672
+ def _header_urls(fields: list[tuple[str, str]], cut: _Cutter) -> Iterator[Url]:
1318
1673
  """The URLs of the List-Unsubscribe headers (RFC 2369)."""
1319
1674
  for name, value in fields:
1320
1675
  if name.lower() != "list-unsubscribe":
@@ -1322,25 +1677,13 @@ def _header_urls(fields: list[tuple[str, str]]) -> Iterator[Url]:
1322
1677
  for item in _ANGLE_BRACKETS.findall(value):
1323
1678
  url = "".join(item.split())
1324
1679
  if url:
1325
- yield Url(
1326
- url=url,
1327
- host=_host(url),
1328
- found_in="header",
1329
- part_id=None,
1330
- anchor_text=None,
1331
- )
1680
+ yield _url(url, "header", None, cut)
1332
1681
 
1333
1682
 
1334
- def _html_urls(links: list[_Link], part_id: str) -> Iterator[Url]:
1683
+ def _html_urls(links: list[_Link], part_id: str, cut: _Cutter) -> Iterator[Url]:
1335
1684
  """The URLs of an HTML part's attributes."""
1336
1685
  for link in links:
1337
- yield Url(
1338
- url=link.url,
1339
- host=_host(link.url),
1340
- found_in="html_href",
1341
- part_id=part_id,
1342
- anchor_text=link.anchor_text,
1343
- )
1686
+ yield _url(link.url, "html_href", part_id, cut, link.anchor_text)
1344
1687
 
1345
1688
 
1346
1689
  @dataclass(slots=True)