spot-sdk-python 2.0.0b1__tar.gz → 2.0.0b2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/PKG-INFO +2 -2
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/README.md +1 -1
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/pyproject.toml +1 -1
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/__init__.py +38 -4
- spot_sdk_python-2.0.0b2/spot_sdk/_json.py +46 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/email.py +5 -3
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/llm.py +4 -13
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/mime.py +461 -118
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/app.py +0 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/clients.py +0 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/config.py +0 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/errors.py +0 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/knowledge.py +0 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/knowledge_tags.py +0 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/logging.py +0 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/manifest.py +0 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/orchestrator.py +0 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/py.typed +0 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/retriever.py +0 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/settings.py +0 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/signals.py +0 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/testing/README.md +0 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/testing/__init__.py +0 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/testing/contract.py +0 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/testing/fake_knowledge_client.py +0 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/testing/fake_llm.py +0 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/testing/fake_spot.py +0 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/testing/views.py +0 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/verdict.py +0 -0
- {spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/workflow.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: spot-sdk-python
|
|
3
|
-
Version: 2.0.
|
|
3
|
+
Version: 2.0.0b2
|
|
4
4
|
Summary: SPOT plugin SDK: the plugin contract, the plugin runtime, its clients and test helpers
|
|
5
5
|
License: Apache-2.0
|
|
6
6
|
Author: SPOT Project
|
|
@@ -49,7 +49,7 @@ plugins share:
|
|
|
49
49
|
## Install
|
|
50
50
|
|
|
51
51
|
```bash
|
|
52
|
-
pip install spot-sdk-python==2.0.
|
|
52
|
+
pip install spot-sdk-python==2.0.0b2
|
|
53
53
|
```
|
|
54
54
|
|
|
55
55
|
Python 3.11 to 3.15. Version 2 implements plugin contract 2, for
|
|
@@ -4,7 +4,7 @@ build-backend = "poetry.core.masonry.api"
|
|
|
4
4
|
|
|
5
5
|
[tool.poetry]
|
|
6
6
|
name = "spot-sdk-python"
|
|
7
|
-
version = "2.0.
|
|
7
|
+
version = "2.0.0b2"
|
|
8
8
|
description = "SPOT plugin SDK: the plugin contract, the plugin runtime, its clients and test helpers"
|
|
9
9
|
authors = ["SPOT Project <spot@sonn.lu>"]
|
|
10
10
|
license = "Apache-2.0"
|
|
@@ -1,13 +1,12 @@
|
|
|
1
1
|
"""SPOT SDK: the plugin contract v2, the plugin runtime and its clients."""
|
|
2
2
|
|
|
3
|
-
from
|
|
4
|
-
from
|
|
3
|
+
from importlib import import_module
|
|
4
|
+
from typing import TYPE_CHECKING, Any
|
|
5
|
+
|
|
5
6
|
from .config import ConfigOption, ConfigStatus, ConfigUpdateRequest
|
|
6
7
|
from .email import Address, Anomaly, AuthResult, EmailView, Envelope, Header, Part, Url
|
|
7
8
|
from .errors import ErrorResponse
|
|
8
|
-
from .knowledge import KnowledgeClient, KnowledgeDocument, chunk_text, content_hash
|
|
9
9
|
from .knowledge_tags import KnowledgeTag
|
|
10
|
-
from .llm import LLMClient, LLMOutputError, LLMToolCall, LLMToolCalls, LLMToolRound
|
|
11
10
|
from .logging import configure_logging, get_logger, get_logging_config
|
|
12
11
|
from .manifest import (
|
|
13
12
|
CONTRACT_VERSION,
|
|
@@ -42,6 +41,41 @@ from .workflow import (
|
|
|
42
41
|
WorkflowStage,
|
|
43
42
|
)
|
|
44
43
|
|
|
44
|
+
if TYPE_CHECKING:
|
|
45
|
+
from .app import SyncState, create_plugin_app
|
|
46
|
+
from .clients import SpotAPIError, SpotClient
|
|
47
|
+
from .knowledge import KnowledgeClient, KnowledgeDocument, chunk_text, content_hash
|
|
48
|
+
from .llm import LLMClient, LLMOutputError, LLMToolCall, LLMToolCalls, LLMToolRound
|
|
49
|
+
|
|
50
|
+
# The modules that import FastAPI, Starlette or httpx load on first use, so importing
|
|
51
|
+
# `spot_sdk.mime` or the models stays light (PEP 562).
|
|
52
|
+
_LAZY = {
|
|
53
|
+
**dict.fromkeys(("SyncState", "create_plugin_app"), ".app"),
|
|
54
|
+
**dict.fromkeys(("SpotAPIError", "SpotClient"), ".clients"),
|
|
55
|
+
**dict.fromkeys(
|
|
56
|
+
("KnowledgeClient", "KnowledgeDocument", "chunk_text", "content_hash"),
|
|
57
|
+
".knowledge",
|
|
58
|
+
),
|
|
59
|
+
**dict.fromkeys(
|
|
60
|
+
("LLMClient", "LLMOutputError", "LLMToolCall", "LLMToolCalls", "LLMToolRound"),
|
|
61
|
+
".llm",
|
|
62
|
+
),
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def __getattr__(name: str) -> Any:
|
|
67
|
+
module = _LAZY.get(name)
|
|
68
|
+
if module is None:
|
|
69
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
70
|
+
value = getattr(import_module(module, __name__), name)
|
|
71
|
+
globals()[name] = value
|
|
72
|
+
return value
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def __dir__() -> list[str]:
|
|
76
|
+
return sorted({*globals(), *_LAZY})
|
|
77
|
+
|
|
78
|
+
|
|
45
79
|
__all__ = [
|
|
46
80
|
# Email view
|
|
47
81
|
"Address",
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
"""Strings measured and cut by the size of their JSON.
|
|
2
|
+
|
|
3
|
+
The view and the LLM client's rendering are sent as JSON, where escapes make
|
|
4
|
+
a string larger than its text: a control character takes six bytes
|
|
5
|
+
(``\\u0001``). ``json.dumps`` with ``ensure_ascii=False`` and pydantic's
|
|
6
|
+
``model_dump_json`` escape strings alike, so sizes here are theirs.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import json
|
|
12
|
+
|
|
13
|
+
_MAX_CHARACTER_BYTES = 6
|
|
14
|
+
"""The most bytes one character takes: a control character's escape."""
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def json_size(text: str) -> int:
|
|
18
|
+
"""The UTF-8 bytes of ``text``'s JSON string between its quotes, escapes
|
|
19
|
+
included; a lone surrogate (possible in a string built in Python) counts
|
|
20
|
+
as three."""
|
|
21
|
+
escaped = json.dumps(text, ensure_ascii=False)[1:-1]
|
|
22
|
+
return len(escaped.encode("utf-8", "surrogatepass"))
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def cut_json(text: str, max_bytes: int) -> tuple[str, bool]:
|
|
26
|
+
"""The longest start of ``text`` whose JSON string takes at most
|
|
27
|
+
``max_bytes`` bytes between its quotes (``json_size``), and whether that
|
|
28
|
+
cut anything.
|
|
29
|
+
|
|
30
|
+
A character is kept or dropped whole, so neither an escape nor a
|
|
31
|
+
character's UTF-8 bytes are split. The text is measured in chunks that
|
|
32
|
+
surely fit, a sixth of the room left each, then character by character:
|
|
33
|
+
the work follows ``max_bytes``, whatever the text's length.
|
|
34
|
+
"""
|
|
35
|
+
if len(text) * _MAX_CHARACTER_BYTES <= max_bytes:
|
|
36
|
+
return text, False # it surely fits
|
|
37
|
+
kept = 0
|
|
38
|
+
room = max_bytes
|
|
39
|
+
while kept < len(text):
|
|
40
|
+
chunk = text[kept : kept + max(room // _MAX_CHARACTER_BYTES, 1)]
|
|
41
|
+
used = json_size(chunk)
|
|
42
|
+
if used > room: # one character, which does not fit
|
|
43
|
+
break
|
|
44
|
+
kept += len(chunk)
|
|
45
|
+
room -= used
|
|
46
|
+
return text[:kept], kept < len(text)
|
|
@@ -88,11 +88,12 @@ class Anomaly(BaseModel):
|
|
|
88
88
|
code: str = Field(description="One of the spot_sdk.mime anomaly codes")
|
|
89
89
|
location: str | None = Field(
|
|
90
90
|
default=None,
|
|
91
|
-
description='"part:1.2", "header:Subject@1", "param:Content-Type.boundary@1.2",
|
|
91
|
+
description='"part:1.2", "header:Subject@1", "param:Content-Type.boundary@1.2",'
|
|
92
|
+
' ..., "envelope", "source": at most 200 bytes of JSON',
|
|
92
93
|
)
|
|
93
94
|
detail: str | None = Field(
|
|
94
95
|
default=None,
|
|
95
|
-
description="The values concerned, as JSON strings: at most 200
|
|
96
|
+
description="The values concerned, as JSON strings: at most 200 bytes of JSON",
|
|
96
97
|
)
|
|
97
98
|
|
|
98
99
|
|
|
@@ -118,7 +119,8 @@ class EmailView(BaseModel):
|
|
|
118
119
|
sha256: str = Field(description="Of the raw message")
|
|
119
120
|
envelope: Envelope | None = None
|
|
120
121
|
headers: list[Header] = Field(
|
|
121
|
-
description="
|
|
122
|
+
description="The headers in order, within their budget, X-SPOT-* removed;"
|
|
123
|
+
" those left out are noted as header_limit"
|
|
122
124
|
)
|
|
123
125
|
subject: str = ""
|
|
124
126
|
from_: Address | None = Field(default=None, alias="from")
|
|
@@ -35,7 +35,6 @@ from __future__ import annotations
|
|
|
35
35
|
|
|
36
36
|
import json
|
|
37
37
|
import secrets
|
|
38
|
-
from bisect import bisect_right
|
|
39
38
|
from collections.abc import Mapping, Sequence
|
|
40
39
|
from datetime import UTC, datetime
|
|
41
40
|
from typing import Any, Self, TypeVar, overload
|
|
@@ -51,6 +50,7 @@ from pydantic import (
|
|
|
51
50
|
model_validator,
|
|
52
51
|
)
|
|
53
52
|
|
|
53
|
+
from ._json import cut_json
|
|
54
54
|
from .clients import json_body
|
|
55
55
|
from .email import EmailView
|
|
56
56
|
|
|
@@ -417,18 +417,9 @@ def _cap(value: Any) -> tuple[Any, bool]:
|
|
|
417
417
|
|
|
418
418
|
|
|
419
419
|
def _cap_text(text: str) -> tuple[str, bool]:
|
|
420
|
-
"""
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
whole, a lone surrogate counting as three bytes. The work is bounded
|
|
424
|
-
whatever the text's length: no character takes less than a byte."""
|
|
425
|
-
head = text[:_STRING_BYTES]
|
|
426
|
-
kept = bisect_right(
|
|
427
|
-
range(1, len(head) + 1),
|
|
428
|
-
_STRING_BYTES,
|
|
429
|
-
key=lambda end: _size(json.dumps(head[:end], ensure_ascii=False)) - 2,
|
|
430
|
-
)
|
|
431
|
-
return head[:kept], kept < len(text)
|
|
420
|
+
"""``text`` cut to ``_STRING_BYTES`` of JSON (``cut_json``), and whether
|
|
421
|
+
that cut anything."""
|
|
422
|
+
return cut_json(text, _STRING_BYTES)
|
|
432
423
|
|
|
433
424
|
|
|
434
425
|
_IDENTITY_FIELDS = ("from", "reply_to", "subject")
|
|
@@ -21,7 +21,9 @@ Anyone can send a message, so the parser never fails on its content:
|
|
|
21
21
|
header values are cut at ``max_text_bytes``, and at most ``MAX_URLS`` URLs
|
|
22
22
|
are kept. Each bound hit is an anomaly; the ones that leave content out set
|
|
23
23
|
``EmailView.truncated``. At most ``MAX_ANOMALIES`` anomalies are recorded,
|
|
24
|
-
``MAX_ANOMALIES_PER_CODE`` of each code, but always the first of each
|
|
24
|
+
``MAX_ANOMALIES_PER_CODE`` of each code, but always the first of each;
|
|
25
|
+
- the view is bounded: its JSON takes at most ``MAX_VIEW_BYTES``, the sum of
|
|
26
|
+
the budgets of its lists and long strings (see :func:`parse`).
|
|
25
27
|
|
|
26
28
|
The standard library builds the MIME tree with the ``compat32`` policy, whose
|
|
27
29
|
headers are plain strings: the ``default`` policy parses each header it reads
|
|
@@ -41,7 +43,7 @@ import json
|
|
|
41
43
|
import math
|
|
42
44
|
import re
|
|
43
45
|
from array import array
|
|
44
|
-
from collections import Counter
|
|
46
|
+
from collections import Counter, defaultdict
|
|
45
47
|
from collections.abc import Collection, Iterable, Iterator
|
|
46
48
|
from dataclasses import dataclass
|
|
47
49
|
from datetime import datetime
|
|
@@ -49,13 +51,16 @@ from email.message import Message
|
|
|
49
51
|
from email.policy import Compat32, Policy, compat32
|
|
50
52
|
from email.utils import decode_params, getaddresses, parsedate_to_datetime, unquote
|
|
51
53
|
from html import unescape as unescape_html
|
|
52
|
-
from itertools import chain
|
|
53
|
-
from typing import Any, Literal, NamedTuple, TypeAlias, cast
|
|
54
|
+
from itertools import chain
|
|
55
|
+
from typing import Any, Literal, NamedTuple, TypeAlias, TypeVar, cast
|
|
54
56
|
from urllib.parse import urlsplit
|
|
55
57
|
from uuid import UUID
|
|
56
58
|
|
|
59
|
+
from pydantic import BaseModel
|
|
60
|
+
from pydantic_core import to_json
|
|
57
61
|
from selectolax.lexbor import LexborHTMLParser, LexborNode
|
|
58
62
|
|
|
63
|
+
from ._json import cut_json, json_size
|
|
59
64
|
from .email import (
|
|
60
65
|
Address,
|
|
61
66
|
Anomaly,
|
|
@@ -77,6 +82,33 @@ MAX_BOUNDARY_LENGTH = 1_000
|
|
|
77
82
|
MAX_ANOMALIES = 1_000
|
|
78
83
|
MAX_ANOMALIES_PER_CODE = 100
|
|
79
84
|
MAX_URLS = 10_000
|
|
85
|
+
# The view's budgets, in bytes of its JSON as model_dump_json(by_alias=True)
|
|
86
|
+
# writes it: UTF-8, escapes included. A list keeps its first entries within
|
|
87
|
+
# its budget (the headers each one that fits), a string its start; each cut is
|
|
88
|
+
# an anomaly.
|
|
89
|
+
MAX_TEXT_BYTES = 1 << 20 # the text, the HTML and the visible text, each
|
|
90
|
+
MAX_HEADERS_BYTES = 1 << 20
|
|
91
|
+
MAX_PARTS_BYTES = 1 << 20
|
|
92
|
+
MAX_URLS_BYTES = 1 << 20
|
|
93
|
+
_URL_BYTES = 8 << 10 # each URL
|
|
94
|
+
_SUBJECT_BYTES = 16 << 10
|
|
95
|
+
# Each other string of an entry or a field: a part's, a URL's host and anchor
|
|
96
|
+
# text, an address's, a message id, a reference, an authentication result's,
|
|
97
|
+
# the envelope's, the source.
|
|
98
|
+
_STRING_BYTES = 1 << 10
|
|
99
|
+
# Each list of a field: reply_to, to, cc, references, authentication, and the
|
|
100
|
+
# envelope's rcpt_to.
|
|
101
|
+
_LIST_BYTES = 64 << 10
|
|
102
|
+
# An anomaly's location and detail, each; the detail reads that many
|
|
103
|
+
# characters of each value.
|
|
104
|
+
_MAX_DETAIL = 200
|
|
105
|
+
# The fields read from the headers, the envelope and the source: the subject,
|
|
106
|
+
# six lists, and ten strings (the sender's three, two message ids, the
|
|
107
|
+
# envelope's four, the source).
|
|
108
|
+
MAX_FIELDS_BYTES = _SUBJECT_BYTES + 6 * _LIST_BYTES + 10 * _STRING_BYTES
|
|
109
|
+
# The keys, the punctuation and the fields of fixed size: the id, the dates,
|
|
110
|
+
# the sizes, the digest and truncated.
|
|
111
|
+
_FIXED_BYTES = 1 << 10
|
|
80
112
|
# The HTML budget, checked before the tree is built (_html_budget). Lexbor, as
|
|
81
113
|
# the HTML standard, never caps the tree, and some of its work is quadratic:
|
|
82
114
|
# each tag or comment makes nodes to walk; a tag may walk the whole stack of
|
|
@@ -124,11 +156,19 @@ boundaries: mail clients may split it either way."""
|
|
|
124
156
|
DEPTH_LIMIT = "depth_limit"
|
|
125
157
|
"""Nesting goes deeper than MAX_DEPTH: the deeper parts are not read."""
|
|
126
158
|
HEADER_LIMIT = "header_limit"
|
|
127
|
-
"""A part has more than MAX_HEADERS headers: the later ones are not
|
|
159
|
+
"""A part has more than MAX_HEADERS headers: the later ones are not kept, but
|
|
160
|
+
the view's fields are still read from them. Or a header does not fit the room
|
|
161
|
+
left in MAX_HEADERS_BYTES: it is left out of the view's headers, but the
|
|
162
|
+
fields read from it are not."""
|
|
128
163
|
PART_LIMIT = "part_limit"
|
|
129
|
-
"""The message has more than MAX_PARTS parts: the later ones are not read.
|
|
164
|
+
"""The message has more than MAX_PARTS parts: the later ones are not read. Or
|
|
165
|
+
its parts take more than MAX_PARTS_BYTES: the later ones are left out of the
|
|
166
|
+
view's parts, but they are read."""
|
|
130
167
|
TEXT_TRUNCATED = "text_truncated"
|
|
131
|
-
"""
|
|
168
|
+
"""A string was cut at its budget: the text, the HTML or the visible text at
|
|
169
|
+
max_text_bytes of JSON (noted at their part), a header value at max_text_bytes
|
|
170
|
+
bytes or a field read from it (at the header), a string of a part (at the
|
|
171
|
+
part), the envelope or the source."""
|
|
132
172
|
HEADER_DUPLICATE = "header_duplicate"
|
|
133
173
|
"""A part has more than one Content-Type, Content-Disposition,
|
|
134
174
|
Content-Transfer-Encoding or MIME-Version header: the first one counts."""
|
|
@@ -150,7 +190,9 @@ HTML_STYLE_UNRESOLVED = "html_style_unresolved"
|
|
|
150
190
|
known when rendering (var(), env(), attr(), a calc() that is not a literal):
|
|
151
191
|
the text was kept visible. The detail names the properties."""
|
|
152
192
|
URL_LIMIT = "url_limit"
|
|
153
|
-
"""The message has more than MAX_URLS URLs
|
|
193
|
+
"""The message has more than MAX_URLS URLs, or they take more than
|
|
194
|
+
MAX_URLS_BYTES: the later ones are not recorded. Or a URL was cut at 8 KiB of
|
|
195
|
+
JSON, or its host or anchor text at 1 KiB."""
|
|
154
196
|
ANOMALY_LIMIT = "anomaly_limit"
|
|
155
197
|
"""More than MAX_ANOMALIES anomalies were met: the later ones are not recorded."""
|
|
156
198
|
ANOMALY_CODES = (
|
|
@@ -175,6 +217,30 @@ ANOMALY_CODES = (
|
|
|
175
217
|
)
|
|
176
218
|
"""Every anomaly code, in the order above."""
|
|
177
219
|
|
|
220
|
+
# The anomalies, and the one that says more were met: each with the longest
|
|
221
|
+
# code, a location and a detail, and a comma.
|
|
222
|
+
MAX_ANOMALIES_BYTES = (MAX_ANOMALIES + 1) * (
|
|
223
|
+
len(
|
|
224
|
+
Anomaly(
|
|
225
|
+
code=max(ANOMALY_CODES, key=len), location="", detail=""
|
|
226
|
+
).model_dump_json()
|
|
227
|
+
)
|
|
228
|
+
+ 2 * _MAX_DETAIL
|
|
229
|
+
+ 1
|
|
230
|
+
) + 1
|
|
231
|
+
# The most bytes the view's JSON takes, whatever the message, with
|
|
232
|
+
# max_text_bytes at most its default: the sum of the budgets, about 6.84 MiB.
|
|
233
|
+
# Core refuses a view over 8 MiB.
|
|
234
|
+
MAX_VIEW_BYTES = (
|
|
235
|
+
_FIXED_BYTES
|
|
236
|
+
+ 3 * MAX_TEXT_BYTES
|
|
237
|
+
+ MAX_HEADERS_BYTES
|
|
238
|
+
+ MAX_PARTS_BYTES
|
|
239
|
+
+ MAX_URLS_BYTES
|
|
240
|
+
+ MAX_FIELDS_BYTES
|
|
241
|
+
+ MAX_ANOMALIES_BYTES
|
|
242
|
+
)
|
|
243
|
+
|
|
178
244
|
_ENCODED_WORD = re.compile(r"=\?([^?\s]+)\?([bq])\?([^?\s]*)\?=", re.IGNORECASE)
|
|
179
245
|
_URL = re.compile(r"\b(?:https?://|www\.)[^\s<>\"'`]+", re.IGNORECASE)
|
|
180
246
|
_URL_TRAILER = ".,;:!?)]}'\""
|
|
@@ -205,7 +271,10 @@ _BASE64_RUN = re.compile(rb"([A-Za-z0-9+/]+)|(=+)")
|
|
|
205
271
|
_STRUCTURAL_HEADERS = frozenset(
|
|
206
272
|
{"content-type", "content-disposition", "content-transfer-encoding", "mime-version"}
|
|
207
273
|
)
|
|
208
|
-
|
|
274
|
+
# The headers the view reads a field from: the first of each name.
|
|
275
|
+
_FIELD_HEADERS = frozenset(
|
|
276
|
+
"subject from reply-to to cc date message-id in-reply-to references".split()
|
|
277
|
+
)
|
|
209
278
|
_DISPOSITIONS: dict[str | None, Literal["inline", "attachment"]] = {
|
|
210
279
|
"inline": "inline",
|
|
211
280
|
"attachment": "attachment",
|
|
@@ -422,21 +491,56 @@ def parse(
|
|
|
422
491
|
received_at: datetime,
|
|
423
492
|
envelope: Envelope | None = None,
|
|
424
493
|
trusted_authserv_ids: Collection[str] = (),
|
|
425
|
-
max_text_bytes: int =
|
|
494
|
+
max_text_bytes: int = MAX_TEXT_BYTES,
|
|
426
495
|
) -> EmailView:
|
|
427
496
|
"""Parse a raw RFC 5322 message into the email view.
|
|
428
497
|
|
|
429
498
|
``trusted_authserv_ids`` are the authserv-ids of our own MTAs: only their
|
|
430
499
|
``Authentication-Results`` count. ``max_text_bytes`` caps the text, the
|
|
431
|
-
HTML and each
|
|
500
|
+
HTML and the visible text, each in bytes of its JSON (escapes included),
|
|
501
|
+
and each header value in bytes. The text and the HTML are read from at
|
|
502
|
+
most that many bytes of their part, and their URLs and the visible text
|
|
503
|
+
from all of what was read, not from the cut the view stores.
|
|
504
|
+
|
|
505
|
+
The view's JSON (``model_dump_json(by_alias=True)``, in UTF-8) takes at
|
|
506
|
+
most ``MAX_VIEW_BYTES``, about 6.84 MiB, whatever the message, with
|
|
507
|
+
``max_text_bytes`` at most its default. Each list and each long string has
|
|
508
|
+
a budget, in bytes of its JSON, escapes included:
|
|
509
|
+
|
|
510
|
+
- the text, the HTML and the visible text: ``MAX_TEXT_BYTES`` (1 MiB) each;
|
|
511
|
+
- the headers, the parts and the URLs: ``MAX_HEADERS_BYTES``,
|
|
512
|
+
``MAX_PARTS_BYTES`` and ``MAX_URLS_BYTES`` (1 MiB each); each URL 8 KiB,
|
|
513
|
+
and each other string of a part or a URL 1 KiB;
|
|
514
|
+
- the fields read from the headers, the envelope and the source:
|
|
515
|
+
``MAX_FIELDS_BYTES`` (410 KiB). The subject takes 16 KiB, each other
|
|
516
|
+
string 1 KiB, and each list (``reply_to``, ``to``, ``cc``,
|
|
517
|
+
``references``, ``authentication``, the envelope's ``rcpt_to``) 64 KiB;
|
|
518
|
+
- the anomalies: ``MAX_ANOMALIES_BYTES`` (about 450 KiB), each location
|
|
519
|
+
and detail 200 bytes;
|
|
520
|
+
- the keys and the fields of fixed size: 1 KiB.
|
|
521
|
+
|
|
522
|
+
A list keeps its first entries, but the headers each one that fits the
|
|
523
|
+
room left; a string keeps its start, and no value is altered. The fields
|
|
524
|
+
read from the headers come from the first header of each name in the
|
|
525
|
+
whole header block, past ``MAX_HEADERS`` too, whatever headers the list
|
|
526
|
+
keeps. Each cut sets ``truncated`` and is an anomaly: ``text_truncated``
|
|
527
|
+
for a string, ``header_limit``, ``part_limit`` or ``url_limit`` for a
|
|
528
|
+
list (a URL cut is ``url_limit`` too).
|
|
432
529
|
"""
|
|
433
530
|
root = _read(raw)
|
|
434
531
|
nodes, left_out = _walk(root, budget=len(raw))
|
|
435
532
|
_note_ambiguous_boundaries(nodes)
|
|
436
533
|
fields, cut_headers = _fields(root, max_text_bytes)
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
534
|
+
read, cut_from = _header_fields(
|
|
535
|
+
root.firsts,
|
|
536
|
+
fields,
|
|
537
|
+
{authserv_id.lower() for authserv_id in trusted_authserv_ids},
|
|
538
|
+
max_text_bytes,
|
|
539
|
+
)
|
|
540
|
+
given = {"source": _Cutter(), "envelope": _Cutter()}
|
|
541
|
+
source = given["source"](source)
|
|
542
|
+
if envelope is not None:
|
|
543
|
+
envelope = _envelope(envelope, given["envelope"])
|
|
440
544
|
|
|
441
545
|
parts: list[Part] = []
|
|
442
546
|
bodies: dict[str, _Body] = {}
|
|
@@ -451,44 +555,68 @@ def parse(
|
|
|
451
555
|
and part.disposition != "attachment"
|
|
452
556
|
and not node.enclosed
|
|
453
557
|
):
|
|
454
|
-
|
|
558
|
+
# The charset as written: a cut one may name another codec.
|
|
559
|
+
charset = _param(node.part, "charset")
|
|
560
|
+
content = _decode_text(data[:max_text_bytes], charset)
|
|
455
561
|
bodies[part.content_type] = _Body(part.part_id, node.part, content)
|
|
456
562
|
if len(data) > max_text_bytes:
|
|
457
|
-
node.part.
|
|
563
|
+
node.part.note_cut()
|
|
458
564
|
|
|
565
|
+
# The URLs and the visible text are read from the whole decoded bodies;
|
|
566
|
+
# the view stores each body cut to its budget.
|
|
459
567
|
text = bodies.get("text/plain")
|
|
460
568
|
html = bodies.get("text/html")
|
|
461
|
-
|
|
569
|
+
cut_urls = _Cutter()
|
|
570
|
+
found = [_header_urls(fields, cut_urls)] # the URLs, built as they are kept
|
|
462
571
|
if text:
|
|
463
|
-
found.append(_urls_in(text.content, "text", text.part_id))
|
|
464
|
-
|
|
572
|
+
found.append(_urls_in(text.content, "text", text.part_id, cut_urls))
|
|
573
|
+
shown = text if text and text.content.strip() else None # the visible text's
|
|
574
|
+
visible_text = shown.content if shown else ""
|
|
465
575
|
if html:
|
|
466
576
|
reader = _read_html(html.content)
|
|
467
|
-
found.append(_html_urls(reader.links, html.part_id))
|
|
577
|
+
found.append(_html_urls(reader.links, html.part_id, cut_urls))
|
|
468
578
|
html_text = reader.text
|
|
469
|
-
found.append(_urls_in(html_text, "html_text", html.part_id))
|
|
470
|
-
|
|
579
|
+
found.append(_urls_in(html_text, "html_text", html.part_id, cut_urls))
|
|
580
|
+
if shown is None:
|
|
581
|
+
shown, visible_text = html, html_text
|
|
471
582
|
if reader.hidden_text:
|
|
472
583
|
html.part.note(HTML_HIDDEN_TEXT)
|
|
473
584
|
if reader.limit:
|
|
474
585
|
html.part.note(HTML_LIMIT, None, reader.limit)
|
|
475
586
|
if reader.unresolved:
|
|
476
587
|
html.part.note(HTML_STYLE_UNRESOLVED, None, *sorted(reader.unresolved))
|
|
477
|
-
|
|
478
|
-
|
|
588
|
+
stored_text = _cut_body(text, text.content, max_text_bytes) if text else None
|
|
589
|
+
stored_html = _cut_body(html, html.content, max_text_bytes) if html else None
|
|
590
|
+
if shown:
|
|
591
|
+
visible_text = _cut_body(shown, visible_text, max_text_bytes)
|
|
592
|
+
# Each note with the id of its part: the header ones are the message's,
|
|
593
|
+
# once per header; the source and the envelope are no part's.
|
|
594
|
+
notes: list[tuple[_Note, str | None]] = [
|
|
595
|
+
(_Note(TEXT_TRUNCATED, subject), "1")
|
|
596
|
+
for subject in dict.fromkeys(cut_headers + cut_from)
|
|
597
|
+
]
|
|
598
|
+
notes += [
|
|
599
|
+
(_Note(TEXT_TRUNCATED, name), None) for name, cut in given.items() if cut.cut
|
|
600
|
+
]
|
|
479
601
|
notes += [(note, node.part_id) for node in nodes for note in node.part.notes]
|
|
602
|
+
parts, left_out_part = _listed(parts, MAX_PARTS_BYTES)
|
|
603
|
+
if left_out_part is not None:
|
|
604
|
+
notes.append((_Note(PART_LIMIT), left_out_part.part_id))
|
|
480
605
|
if left_out:
|
|
481
606
|
notes.append((_Note(PART_LIMIT), left_out))
|
|
482
|
-
urls =
|
|
483
|
-
|
|
484
|
-
|
|
607
|
+
urls, left_out_url = _listed(
|
|
608
|
+
chain.from_iterable(found), MAX_URLS_BYTES, limit=MAX_URLS
|
|
609
|
+
)
|
|
610
|
+
if left_out_url is not None or cut_urls.cut:
|
|
485
611
|
notes.append((_Note(URL_LIMIT), "1"))
|
|
612
|
+
headers, header_left_out = _headers(fields)
|
|
613
|
+
if header_left_out:
|
|
614
|
+
notes.append((_Note(HEADER_LIMIT), "1"))
|
|
486
615
|
kept = _kept(notes)
|
|
487
616
|
anomalies = [_anomaly(note, part_id) for note, part_id in kept]
|
|
488
617
|
if len(kept) < len(notes):
|
|
489
618
|
anomalies.append(Anomaly(code=ANOMALY_LIMIT))
|
|
490
619
|
|
|
491
|
-
senders = _addresses(first.get("from", ""))
|
|
492
620
|
return EmailView(
|
|
493
621
|
id=email_id,
|
|
494
622
|
received_at=received_at,
|
|
@@ -496,26 +624,20 @@ def parse(
|
|
|
496
624
|
size_bytes=len(raw),
|
|
497
625
|
sha256=hashlib.sha256(raw).hexdigest(),
|
|
498
626
|
envelope=envelope,
|
|
499
|
-
headers=
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
message_id=first.get("message-id", "").strip() or None,
|
|
511
|
-
in_reply_to=first.get("in-reply-to", "").strip() or None,
|
|
512
|
-
references=_MESSAGE_ID.findall(first.get("references", "")),
|
|
513
|
-
authentication=_authentication(
|
|
514
|
-
fields, {authserv_id.lower() for authserv_id in trusted_authserv_ids}
|
|
515
|
-
),
|
|
627
|
+
headers=headers,
|
|
628
|
+
subject=read.subject,
|
|
629
|
+
from_=read.from_,
|
|
630
|
+
reply_to=read.reply_to,
|
|
631
|
+
to=read.to,
|
|
632
|
+
cc=read.cc,
|
|
633
|
+
date=read.date,
|
|
634
|
+
message_id=read.message_id,
|
|
635
|
+
in_reply_to=read.in_reply_to,
|
|
636
|
+
references=read.references,
|
|
637
|
+
authentication=read.authentication,
|
|
516
638
|
parts=parts,
|
|
517
|
-
text=
|
|
518
|
-
html=
|
|
639
|
+
text=stored_text,
|
|
640
|
+
html=stored_html,
|
|
519
641
|
visible_text=visible_text,
|
|
520
642
|
urls=urls,
|
|
521
643
|
truncated=any(note.code in _LIMIT_CODES for note, _ in notes),
|
|
@@ -536,6 +658,86 @@ def part_bytes(raw: bytes, part_id: str) -> tuple[bytes, str]:
|
|
|
536
658
|
raise PartNotFoundError(part_id)
|
|
537
659
|
|
|
538
660
|
|
|
661
|
+
# ----------------------------------------------------------------------- #
|
|
662
|
+
# The view's budgets
|
|
663
|
+
# ----------------------------------------------------------------------- #
|
|
664
|
+
|
|
665
|
+
|
|
666
|
+
_Entry = TypeVar("_Entry", bound=BaseModel | str)
|
|
667
|
+
|
|
668
|
+
|
|
669
|
+
def _listed(
|
|
670
|
+
entries: Iterable[_Entry], budget: int, limit: int | None = None
|
|
671
|
+
) -> tuple[list[_Entry], _Entry | None]:
|
|
672
|
+
"""The first ``entries``, at most ``limit``, whose JSON array takes at
|
|
673
|
+
most ``budget`` bytes; and the first one left out, if any. Each entry is
|
|
674
|
+
measured as it comes, so the work stops at the budget."""
|
|
675
|
+
kept: list[_Entry] = []
|
|
676
|
+
size = 1 # "[", then each entry and the "," or "]" after it
|
|
677
|
+
for entry in entries:
|
|
678
|
+
size += _size(entry) + 1
|
|
679
|
+
if size > budget or len(kept) == limit:
|
|
680
|
+
return kept, entry
|
|
681
|
+
kept.append(entry)
|
|
682
|
+
return kept, None
|
|
683
|
+
|
|
684
|
+
|
|
685
|
+
def _headers(fields: list[tuple[str, str]]) -> tuple[list[Header], bool]:
|
|
686
|
+
"""The headers in order, without ``X-SPOT-*``, each kept if it fits the
|
|
687
|
+
room left in ``MAX_HEADERS_BYTES``; and whether one was left out.
|
|
688
|
+
|
|
689
|
+
A header that does not fit is left out alone, so one long header does not
|
|
690
|
+
hide the ones after it. A character takes a byte at least: a header whose
|
|
691
|
+
name and decoded value are longer than the room left is left out without
|
|
692
|
+
measuring it.
|
|
693
|
+
"""
|
|
694
|
+
headers: list[Header] = []
|
|
695
|
+
left_out = False
|
|
696
|
+
room = MAX_HEADERS_BYTES - 1 # "[", then each entry and the "," or "]" after it
|
|
697
|
+
for name, value in fields:
|
|
698
|
+
if name.lower().startswith("x-spot-"):
|
|
699
|
+
continue
|
|
700
|
+
value = _decode_words(value)
|
|
701
|
+
if len(name) + len(value) < room:
|
|
702
|
+
header = Header(name=name, value=value)
|
|
703
|
+
size = _size(header) + 1
|
|
704
|
+
if size <= room:
|
|
705
|
+
headers.append(header)
|
|
706
|
+
room -= size
|
|
707
|
+
continue
|
|
708
|
+
left_out = True
|
|
709
|
+
return headers, left_out
|
|
710
|
+
|
|
711
|
+
|
|
712
|
+
class _Cutter:
|
|
713
|
+
"""Cuts strings to their budget, and remembers whether it cut one."""
|
|
714
|
+
|
|
715
|
+
def __init__(self) -> None:
|
|
716
|
+
self.cut = False
|
|
717
|
+
|
|
718
|
+
def __call__(self, text: str, max_bytes: int = _STRING_BYTES) -> str:
|
|
719
|
+
text, cut = cut_json(text, max_bytes)
|
|
720
|
+
self.cut = self.cut or cut
|
|
721
|
+
return text
|
|
722
|
+
|
|
723
|
+
def optional(self, text: str | None) -> str | None:
|
|
724
|
+
return None if text is None else self(text)
|
|
725
|
+
|
|
726
|
+
def listed(self, entries: Iterable[_Entry]) -> list[_Entry]:
|
|
727
|
+
"""The first entries within ``_LIST_BYTES``; a list cut short counts as
|
|
728
|
+
a cut."""
|
|
729
|
+
kept, left_out = _listed(entries, _LIST_BYTES)
|
|
730
|
+
self.cut = self.cut or left_out is not None
|
|
731
|
+
return kept
|
|
732
|
+
|
|
733
|
+
|
|
734
|
+
def _size(value: BaseModel | str) -> int:
|
|
735
|
+
"""The bytes of a value's JSON, as the view's."""
|
|
736
|
+
if isinstance(value, str):
|
|
737
|
+
return json_size(value) + 2 # and its quotes
|
|
738
|
+
return len(to_json(value, by_alias=True))
|
|
739
|
+
|
|
740
|
+
|
|
539
741
|
# ----------------------------------------------------------------------- #
|
|
540
742
|
# The MIME tree
|
|
541
743
|
# ----------------------------------------------------------------------- #
|
|
@@ -556,14 +758,24 @@ class _Part(Message):
|
|
|
556
758
|
candidates: list[str] | None = None # usable boundaries, preferred first
|
|
557
759
|
raw_body: str | None = None # as read, raw bytes as surrogate escapes
|
|
558
760
|
headers_dropped = False
|
|
761
|
+
content_cut = False
|
|
559
762
|
|
|
560
763
|
def __init__(self, policy: Policy[Any] = compat32) -> None:
|
|
561
764
|
super().__init__(policy)
|
|
562
765
|
self.notes: list[_Note] = []
|
|
766
|
+
# By lower-cased name, the first header of each _FIELD_HEADERS name as
|
|
767
|
+
# read (name, raw value): past MAX_HEADERS too.
|
|
768
|
+
self.firsts: dict[str, tuple[str, str]] = {}
|
|
563
769
|
|
|
564
770
|
def note(self, code: str, subject: str | None = None, *values: str) -> None:
|
|
565
771
|
self.notes.append(_Note(code, subject, values))
|
|
566
772
|
|
|
773
|
+
def note_cut(self) -> None:
|
|
774
|
+
"""Note, once, that content read from this part was cut at its budget."""
|
|
775
|
+
if not self.content_cut:
|
|
776
|
+
self.content_cut = True
|
|
777
|
+
self.note(TEXT_TRUNCATED)
|
|
778
|
+
|
|
567
779
|
def attach(self, payload: Message | str) -> None:
|
|
568
780
|
if isinstance(payload, _Part):
|
|
569
781
|
payload.depth = self.depth + 1
|
|
@@ -576,6 +788,9 @@ class _Part(Message):
|
|
|
576
788
|
super().set_payload(payload, charset)
|
|
577
789
|
|
|
578
790
|
def set_raw(self, name: str, value: str) -> None:
|
|
791
|
+
key = name.lower()
|
|
792
|
+
if key in _FIELD_HEADERS:
|
|
793
|
+
self.firsts.setdefault(key, (name, value))
|
|
579
794
|
if len(self) < MAX_HEADERS:
|
|
580
795
|
super().set_raw(name, value)
|
|
581
796
|
elif not self.headers_dropped:
|
|
@@ -662,6 +877,15 @@ class _Body:
|
|
|
662
877
|
content: str
|
|
663
878
|
|
|
664
879
|
|
|
880
|
+
def _cut_body(body: _Body, content: str, max_bytes: int) -> str:
|
|
881
|
+
"""``content`` read from a body, cut to ``max_bytes`` of JSON; a cut is
|
|
882
|
+
noted on the body's part."""
|
|
883
|
+
content, cut = cut_json(content, max_bytes)
|
|
884
|
+
if cut:
|
|
885
|
+
body.part.note_cut()
|
|
886
|
+
return content
|
|
887
|
+
|
|
888
|
+
|
|
665
889
|
def _read(raw: bytes) -> _Part:
|
|
666
890
|
return cast(_Part, email.message_from_bytes(raw, policy=_POLICY))
|
|
667
891
|
|
|
@@ -832,22 +1056,27 @@ def _decode_base64(text: str) -> tuple[bytes, str]:
|
|
|
832
1056
|
|
|
833
1057
|
|
|
834
1058
|
def _describe(node: _Node, data: bytes) -> Part:
|
|
1059
|
+
"""A part as the view lists it, each string cut to ``_STRING_BYTES``."""
|
|
835
1060
|
part = node.part
|
|
836
1061
|
charset = _param(part, "charset")
|
|
837
1062
|
filename = _param(part, "filename", "content-disposition")
|
|
838
1063
|
name = _param(part, "name")
|
|
839
1064
|
if filename and name and filename != name:
|
|
840
1065
|
part.note(PARAM_CONFLICT, "param:Content-Disposition.filename", filename, name)
|
|
841
|
-
|
|
1066
|
+
cut = _Cutter()
|
|
1067
|
+
described = Part(
|
|
842
1068
|
part_id=node.part_id,
|
|
843
|
-
content_type=_content_type(part),
|
|
844
|
-
charset=charset.lower() if charset else None,
|
|
1069
|
+
content_type=cut(_content_type(part)),
|
|
1070
|
+
charset=cut.optional(charset.lower() if charset else None),
|
|
845
1071
|
disposition=_DISPOSITIONS.get(part.get_content_disposition()),
|
|
846
|
-
filename=filename or name,
|
|
847
|
-
content_id=_clean(part.get("content-id") or "").strip() or None,
|
|
1072
|
+
filename=cut.optional(filename or name),
|
|
1073
|
+
content_id=cut.optional(_clean(part.get("content-id") or "").strip() or None),
|
|
848
1074
|
size_bytes=len(data),
|
|
849
1075
|
sha256=None if part.is_multipart() else hashlib.sha256(data).hexdigest(),
|
|
850
1076
|
)
|
|
1077
|
+
if cut.cut:
|
|
1078
|
+
part.note_cut()
|
|
1079
|
+
return described
|
|
851
1080
|
|
|
852
1081
|
|
|
853
1082
|
def _content_type(part: _Part) -> str:
|
|
@@ -1081,15 +1310,23 @@ def _fields(message: _Part, limit: int) -> tuple[list[tuple[str, str]], list[str
|
|
|
1081
1310
|
for name, raw_value in message.items():
|
|
1082
1311
|
key = name.lower()
|
|
1083
1312
|
seen[key] += 1
|
|
1084
|
-
value =
|
|
1085
|
-
|
|
1086
|
-
if len(data) > limit:
|
|
1087
|
-
value = data[:limit].decode(errors="ignore")
|
|
1313
|
+
value, was_cut = _header_value(raw_value, limit)
|
|
1314
|
+
if was_cut:
|
|
1088
1315
|
cut.append(_header_subject(name, seen[key]))
|
|
1089
1316
|
fields.append((name, value))
|
|
1090
1317
|
return fields, cut
|
|
1091
1318
|
|
|
1092
1319
|
|
|
1320
|
+
def _header_value(raw_value: str, limit: int) -> tuple[str, bool]:
|
|
1321
|
+
"""A header value as the view reads it, cut at ``limit`` bytes; and
|
|
1322
|
+
whether it was cut."""
|
|
1323
|
+
value = _clean(raw_value)
|
|
1324
|
+
data = value.encode()
|
|
1325
|
+
if len(data) > limit:
|
|
1326
|
+
return data[:limit].decode(errors="ignore"), True
|
|
1327
|
+
return value, False
|
|
1328
|
+
|
|
1329
|
+
|
|
1093
1330
|
def _note_repeated_headers(part: _Part) -> None:
|
|
1094
1331
|
"""Note each structural header a part gives again: the first one counts."""
|
|
1095
1332
|
seen: Counter[str] = Counter()
|
|
@@ -1109,7 +1346,9 @@ def _header_subject(name: str, occurrence: int) -> str:
|
|
|
1109
1346
|
return f"header:{name}[{occurrence}]" if occurrence > 1 else f"header:{name}"
|
|
1110
1347
|
|
|
1111
1348
|
|
|
1112
|
-
def _kept(
|
|
1349
|
+
def _kept(
|
|
1350
|
+
notes: list[tuple[_Note, str | None]],
|
|
1351
|
+
) -> list[tuple[_Note, str | None]]:
|
|
1113
1352
|
"""The notes recorded, in order: the first of each code, then the others
|
|
1114
1353
|
while their code has fewer than ``MAX_ANOMALIES_PER_CODE``, and all fewer
|
|
1115
1354
|
than ``MAX_ANOMALIES``. Cheap, common notes do not push out rare ones."""
|
|
@@ -1118,7 +1357,7 @@ def _kept(notes: list[tuple[_Note, str]]) -> list[tuple[_Note, str]]:
|
|
|
1118
1357
|
first.setdefault(note.code, index)
|
|
1119
1358
|
room = MAX_ANOMALIES - len(first) # for the notes after the first of a code
|
|
1120
1359
|
counts: Counter[str] = Counter()
|
|
1121
|
-
kept: list[tuple[_Note, str]] = []
|
|
1360
|
+
kept: list[tuple[_Note, str | None]] = []
|
|
1122
1361
|
for index, (note, part_id) in enumerate(notes):
|
|
1123
1362
|
if index != first[note.code]:
|
|
1124
1363
|
if not room or counts[note.code] >= MAX_ANOMALIES_PER_CODE:
|
|
@@ -1129,18 +1368,20 @@ def _kept(notes: list[tuple[_Note, str]]) -> list[tuple[_Note, str]]:
|
|
|
1129
1368
|
return kept
|
|
1130
1369
|
|
|
1131
1370
|
|
|
1132
|
-
def _anomaly(note: _Note, part_id: str) -> Anomaly:
|
|
1371
|
+
def _anomaly(note: _Note, part_id: str | None) -> Anomaly:
|
|
1133
1372
|
"""An anomaly as the view carries it: its values JSON-quoted in the detail."""
|
|
1134
1373
|
detail = ", ".join(
|
|
1135
1374
|
json.dumps(_clean(value[:_MAX_DETAIL]), ensure_ascii=False)
|
|
1136
1375
|
for value in note.values
|
|
1137
1376
|
)
|
|
1377
|
+
if part_id is None:
|
|
1378
|
+
location = note.subject
|
|
1379
|
+
else:
|
|
1380
|
+
location = f"{note.subject}@{part_id}" if note.subject else f"part:{part_id}"
|
|
1138
1381
|
return Anomaly(
|
|
1139
1382
|
code=note.code,
|
|
1140
|
-
location=_clean(
|
|
1141
|
-
|
|
1142
|
-
),
|
|
1143
|
-
detail=detail[:_MAX_DETAIL] or None,
|
|
1383
|
+
location=cut_json(_clean(location), _MAX_DETAIL)[0] if location else None,
|
|
1384
|
+
detail=cut_json(detail, _MAX_DETAIL)[0] or None,
|
|
1144
1385
|
)
|
|
1145
1386
|
|
|
1146
1387
|
|
|
@@ -1159,20 +1400,21 @@ def _domain(host: str) -> str:
|
|
|
1159
1400
|
return host
|
|
1160
1401
|
|
|
1161
1402
|
|
|
1162
|
-
def _addresses(value: str) ->
|
|
1163
|
-
"""The mailboxes of an address header
|
|
1403
|
+
def _addresses(value: str, cut: _Cutter) -> Iterator[Address]:
|
|
1404
|
+
"""The mailboxes of an address header, each string cut to its budget; a
|
|
1405
|
+
value that does not parse is kept whole."""
|
|
1164
1406
|
mailboxes = getaddresses([value])
|
|
1165
1407
|
if not mailboxes or not all("@" in address for _, address in mailboxes):
|
|
1166
1408
|
value = value.strip()
|
|
1167
|
-
|
|
1168
|
-
|
|
1169
|
-
|
|
1170
|
-
|
|
1171
|
-
|
|
1172
|
-
|
|
1409
|
+
if value:
|
|
1410
|
+
yield Address(address=cut(value))
|
|
1411
|
+
return
|
|
1412
|
+
for name, address in mailboxes:
|
|
1413
|
+
yield Address(
|
|
1414
|
+
display_name=cut(_decode_words(name)),
|
|
1415
|
+
address=cut(address),
|
|
1416
|
+
domain=cut(_domain(address.rpartition("@")[2])),
|
|
1173
1417
|
)
|
|
1174
|
-
for name, address in mailboxes
|
|
1175
|
-
]
|
|
1176
1418
|
|
|
1177
1419
|
|
|
1178
1420
|
def _date(value: str) -> datetime | None:
|
|
@@ -1182,6 +1424,86 @@ def _date(value: str) -> datetime | None:
|
|
|
1182
1424
|
return None
|
|
1183
1425
|
|
|
1184
1426
|
|
|
1427
|
+
class _HeaderFields(NamedTuple):
|
|
1428
|
+
"""The view's fields read from the headers."""
|
|
1429
|
+
|
|
1430
|
+
subject: str
|
|
1431
|
+
from_: Address | None
|
|
1432
|
+
reply_to: list[Address]
|
|
1433
|
+
to: list[Address]
|
|
1434
|
+
cc: list[Address]
|
|
1435
|
+
date: datetime | None
|
|
1436
|
+
message_id: str | None
|
|
1437
|
+
in_reply_to: str | None
|
|
1438
|
+
references: list[str]
|
|
1439
|
+
authentication: list[AuthResult]
|
|
1440
|
+
|
|
1441
|
+
|
|
1442
|
+
def _header_fields(
|
|
1443
|
+
firsts: dict[str, tuple[str, str]],
|
|
1444
|
+
fields: list[tuple[str, str]],
|
|
1445
|
+
trusted: set[str],
|
|
1446
|
+
limit: int,
|
|
1447
|
+
) -> tuple[_HeaderFields, list[str]]:
|
|
1448
|
+
"""The view's fields read from the first header of each name (``firsts``,
|
|
1449
|
+
the whole header block's, each value cut at ``limit`` bytes), and the
|
|
1450
|
+
authentication results from the trusted headers on top: within their
|
|
1451
|
+
budgets. Each string is cut to ``_STRING_BYTES``, the subject to
|
|
1452
|
+
``_SUBJECT_BYTES``, and each list to ``_LIST_BYTES``.
|
|
1453
|
+
|
|
1454
|
+
Also the headers a field was cut from, as ``header:`` subjects.
|
|
1455
|
+
"""
|
|
1456
|
+
# By lower-cased name: the name as written, and the value.
|
|
1457
|
+
first: dict[str, tuple[str, str]] = {}
|
|
1458
|
+
cuts: defaultdict[str, _Cutter] = defaultdict(_Cutter) # by header
|
|
1459
|
+
for key, (name, raw_value) in firsts.items():
|
|
1460
|
+
value, cut = _header_value(_POLICY.header_fetch_parse(name, raw_value), limit)
|
|
1461
|
+
first[key] = (name, value)
|
|
1462
|
+
cuts[key].cut = cut
|
|
1463
|
+
|
|
1464
|
+
def header(name: str) -> str:
|
|
1465
|
+
return first.get(name, ("", ""))[1]
|
|
1466
|
+
|
|
1467
|
+
def addresses(name: str) -> list[Address]:
|
|
1468
|
+
return cuts[name].listed(_addresses(header(name), cuts[name]))
|
|
1469
|
+
|
|
1470
|
+
def message_id(name: str) -> str | None:
|
|
1471
|
+
return cuts[name](header(name).strip()) or None
|
|
1472
|
+
|
|
1473
|
+
authentication, authentication_cut = _authentication(fields, trusted)
|
|
1474
|
+
read = _HeaderFields(
|
|
1475
|
+
subject=cuts["subject"](_decode_words(header("subject")), _SUBJECT_BYTES),
|
|
1476
|
+
from_=next(_addresses(header("from"), cuts["from"]), None),
|
|
1477
|
+
reply_to=addresses("reply-to"),
|
|
1478
|
+
to=addresses("to"),
|
|
1479
|
+
cc=addresses("cc"),
|
|
1480
|
+
date=_date(header("date")),
|
|
1481
|
+
message_id=message_id("message-id"),
|
|
1482
|
+
in_reply_to=message_id("in-reply-to"),
|
|
1483
|
+
references=cuts["references"].listed(
|
|
1484
|
+
cuts["references"](reference.group())
|
|
1485
|
+
for reference in _MESSAGE_ID.finditer(header("references"))
|
|
1486
|
+
),
|
|
1487
|
+
authentication=authentication,
|
|
1488
|
+
)
|
|
1489
|
+
cut_from = [
|
|
1490
|
+
_header_subject(first[name][0], 1) for name, cut in cuts.items() if cut.cut
|
|
1491
|
+
]
|
|
1492
|
+
return read, cut_from + ([authentication_cut] if authentication_cut else [])
|
|
1493
|
+
|
|
1494
|
+
|
|
1495
|
+
def _envelope(envelope: Envelope, cut: _Cutter) -> Envelope:
|
|
1496
|
+
"""The SMTP envelope within its budgets: each string cut to
|
|
1497
|
+
``_STRING_BYTES``, ``rcpt_to`` to ``_LIST_BYTES``."""
|
|
1498
|
+
return Envelope(
|
|
1499
|
+
mail_from=cut.optional(envelope.mail_from),
|
|
1500
|
+
rcpt_to=cut.listed(cut(recipient) for recipient in envelope.rcpt_to),
|
|
1501
|
+
client_address=cut.optional(envelope.client_address),
|
|
1502
|
+
client_name=cut.optional(envelope.client_name),
|
|
1503
|
+
helo=cut.optional(envelope.helo),
|
|
1504
|
+
)
|
|
1505
|
+
|
|
1506
|
+
|
|
1185
1507
|
# ----------------------------------------------------------------------- #
|
|
1186
1508
|
# Authentication-Results (RFC 8601)
|
|
1187
1509
|
# ----------------------------------------------------------------------- #
|
|
@@ -1189,46 +1511,68 @@ def _date(value: str) -> datetime | None:
|
|
|
1189
1511
|
|
|
1190
1512
|
def _authentication(
|
|
1191
1513
|
fields: list[tuple[str, str]], trusted: set[str]
|
|
1192
|
-
) -> list[AuthResult]:
|
|
1193
|
-
"""The results of the trusted headers above the first untrusted one
|
|
1514
|
+
) -> tuple[list[AuthResult], str | None]:
|
|
1515
|
+
"""The results of the trusted headers above the first untrusted one,
|
|
1516
|
+
within ``_LIST_BYTES``, each string cut to ``_STRING_BYTES``; and the
|
|
1517
|
+
header the first cut is in, as a ``header:`` subject, if any.
|
|
1194
1518
|
|
|
1195
1519
|
Our MTAs add their headers on top, so only that run is theirs: a trusted
|
|
1196
1520
|
authserv-id below an untrusted one was written by someone else.
|
|
1197
1521
|
"""
|
|
1198
1522
|
results: list[AuthResult] = []
|
|
1523
|
+
first_cut: str | None = None
|
|
1524
|
+
size = 1 # as _listed counts
|
|
1525
|
+
for subject, result, cut in _trusted_results(fields, trusted):
|
|
1526
|
+
size += _size(result) + 1
|
|
1527
|
+
if size > _LIST_BYTES:
|
|
1528
|
+
return results, first_cut or subject
|
|
1529
|
+
results.append(result)
|
|
1530
|
+
if cut and first_cut is None:
|
|
1531
|
+
first_cut = subject
|
|
1532
|
+
return results, first_cut
|
|
1533
|
+
|
|
1534
|
+
|
|
1535
|
+
def _trusted_results(
|
|
1536
|
+
fields: list[tuple[str, str]], trusted: set[str]
|
|
1537
|
+
) -> Iterator[tuple[str, AuthResult, bool]]:
|
|
1538
|
+
"""Each result of the trusted headers on top, with the header it is in (a
|
|
1539
|
+
``header:`` subject) and whether one of its strings was cut."""
|
|
1540
|
+
occurrence = 0
|
|
1199
1541
|
for name, value in fields:
|
|
1200
1542
|
if name.lower() != "authentication-results":
|
|
1201
1543
|
continue
|
|
1544
|
+
occurrence += 1
|
|
1202
1545
|
head, *resinfos = _segments(value)
|
|
1203
1546
|
words = head.split()
|
|
1204
1547
|
authserv_id = _unquote(words[0]).lower() if words else ""
|
|
1205
1548
|
if authserv_id not in trusted:
|
|
1206
|
-
|
|
1207
|
-
|
|
1208
|
-
|
|
1549
|
+
return
|
|
1550
|
+
for result, cut in _auth_results(authserv_id, resinfos):
|
|
1551
|
+
yield _header_subject(name, occurrence), result, cut
|
|
1209
1552
|
|
|
1210
1553
|
|
|
1211
|
-
def _auth_results(
|
|
1212
|
-
|
|
1213
|
-
|
|
1554
|
+
def _auth_results(
|
|
1555
|
+
authserv_id: str, resinfos: list[str]
|
|
1556
|
+
) -> Iterator[tuple[AuthResult, bool]]:
|
|
1557
|
+
"""Each resinfo's ``method=result``, with its ``ptype.property=value``
|
|
1558
|
+
pairs, each string cut to ``_STRING_BYTES``; and whether one was."""
|
|
1214
1559
|
for resinfo in resinfos:
|
|
1215
1560
|
pairs = _AUTH_PAIR.findall(resinfo)
|
|
1216
1561
|
if not pairs: # "none": no result
|
|
1217
1562
|
continue
|
|
1218
1563
|
(method, result), *properties = pairs
|
|
1219
|
-
|
|
1220
|
-
|
|
1221
|
-
|
|
1222
|
-
|
|
1223
|
-
|
|
1224
|
-
|
|
1225
|
-
|
|
1226
|
-
|
|
1227
|
-
|
|
1228
|
-
|
|
1229
|
-
)
|
|
1564
|
+
cut = _Cutter()
|
|
1565
|
+
read = AuthResult(
|
|
1566
|
+
authserv_id=cut(authserv_id),
|
|
1567
|
+
method=cut(method.partition("/")[0].lower()),
|
|
1568
|
+
result=cut(_unquote(result).lower()),
|
|
1569
|
+
properties={
|
|
1570
|
+
cut(key.lower()): cut(_unquote(pvalue))
|
|
1571
|
+
for key, pvalue in properties
|
|
1572
|
+
if "." in key
|
|
1573
|
+
},
|
|
1230
1574
|
)
|
|
1231
|
-
|
|
1575
|
+
yield read, cut.cut
|
|
1232
1576
|
|
|
1233
1577
|
|
|
1234
1578
|
def _segments(value: str) -> list[str]:
|
|
@@ -1299,22 +1643,33 @@ def _host(url: str) -> str:
|
|
|
1299
1643
|
return _domain(host) if host else ""
|
|
1300
1644
|
|
|
1301
1645
|
|
|
1646
|
+
def _url(
|
|
1647
|
+
url: str,
|
|
1648
|
+
found_in: Literal["text", "html_href", "html_text", "header"],
|
|
1649
|
+
part_id: str | None,
|
|
1650
|
+
cut: _Cutter,
|
|
1651
|
+
anchor_text: str | None = None,
|
|
1652
|
+
) -> Url:
|
|
1653
|
+
"""A URL as the view records it: cut to ``_URL_BYTES``, its host (of the
|
|
1654
|
+
whole URL) and anchor text to ``_STRING_BYTES``."""
|
|
1655
|
+
return Url(
|
|
1656
|
+
url=cut(url, _URL_BYTES),
|
|
1657
|
+
host=cut(_host(url)),
|
|
1658
|
+
found_in=found_in,
|
|
1659
|
+
part_id=part_id,
|
|
1660
|
+
anchor_text=cut.optional(anchor_text),
|
|
1661
|
+
)
|
|
1662
|
+
|
|
1663
|
+
|
|
1302
1664
|
def _urls_in(
|
|
1303
|
-
text: str, found_in: Literal["text", "html_text"], part_id: str
|
|
1665
|
+
text: str, found_in: Literal["text", "html_text"], part_id: str, cut: _Cutter
|
|
1304
1666
|
) -> Iterator[Url]:
|
|
1305
1667
|
"""The ``http(s)://`` and ``www.`` URLs written in a text."""
|
|
1306
1668
|
for match in _URL.finditer(text):
|
|
1307
|
-
|
|
1308
|
-
yield Url(
|
|
1309
|
-
url=url,
|
|
1310
|
-
host=_host(url),
|
|
1311
|
-
found_in=found_in,
|
|
1312
|
-
part_id=part_id,
|
|
1313
|
-
anchor_text=None,
|
|
1314
|
-
)
|
|
1669
|
+
yield _url(match.group().rstrip(_URL_TRAILER), found_in, part_id, cut)
|
|
1315
1670
|
|
|
1316
1671
|
|
|
1317
|
-
def _header_urls(fields: list[tuple[str, str]]) -> Iterator[Url]:
|
|
1672
|
+
def _header_urls(fields: list[tuple[str, str]], cut: _Cutter) -> Iterator[Url]:
|
|
1318
1673
|
"""The URLs of the List-Unsubscribe headers (RFC 2369)."""
|
|
1319
1674
|
for name, value in fields:
|
|
1320
1675
|
if name.lower() != "list-unsubscribe":
|
|
@@ -1322,25 +1677,13 @@ def _header_urls(fields: list[tuple[str, str]]) -> Iterator[Url]:
|
|
|
1322
1677
|
for item in _ANGLE_BRACKETS.findall(value):
|
|
1323
1678
|
url = "".join(item.split())
|
|
1324
1679
|
if url:
|
|
1325
|
-
yield
|
|
1326
|
-
url=url,
|
|
1327
|
-
host=_host(url),
|
|
1328
|
-
found_in="header",
|
|
1329
|
-
part_id=None,
|
|
1330
|
-
anchor_text=None,
|
|
1331
|
-
)
|
|
1680
|
+
yield _url(url, "header", None, cut)
|
|
1332
1681
|
|
|
1333
1682
|
|
|
1334
|
-
def _html_urls(links: list[_Link], part_id: str) -> Iterator[Url]:
|
|
1683
|
+
def _html_urls(links: list[_Link], part_id: str, cut: _Cutter) -> Iterator[Url]:
|
|
1335
1684
|
"""The URLs of an HTML part's attributes."""
|
|
1336
1685
|
for link in links:
|
|
1337
|
-
yield
|
|
1338
|
-
url=link.url,
|
|
1339
|
-
host=_host(link.url),
|
|
1340
|
-
found_in="html_href",
|
|
1341
|
-
part_id=part_id,
|
|
1342
|
-
anchor_text=link.anchor_text,
|
|
1343
|
-
)
|
|
1686
|
+
yield _url(link.url, "html_href", part_id, cut, link.anchor_text)
|
|
1344
1687
|
|
|
1345
1688
|
|
|
1346
1689
|
@dataclass(slots=True)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{spot_sdk_python-2.0.0b1 → spot_sdk_python-2.0.0b2}/spot_sdk/testing/fake_knowledge_client.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|