contextbase-plugin-microsoft-mail 0.0.0a1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,194 @@
1
+ from __future__ import annotations
2
+
3
+ from collections.abc import Iterable, Iterator, Mapping
4
+ from typing import Any
5
+
6
+ from shared_plugins.microsoft_graph import graph_object_to_payload
7
+
8
+ from .ctx import (
9
+ AttachmentContentRow,
10
+ MailFolderRow,
11
+ MessageRow,
12
+ )
13
+
14
+
15
+ def _payload(item: object) -> dict[str, Any]:
16
+ return graph_object_to_payload(item)
17
+
18
+
19
+ def _list_value(payload: Mapping[str, Any], key: str) -> list[Any]:
20
+ value = payload.get(key)
21
+ if value is None:
22
+ return []
23
+ if not isinstance(value, list):
24
+ raise TypeError(
25
+ f"translator: field {key!r} must be a list, got {type(value).__name__}"
26
+ )
27
+ return list(value)
28
+
29
+
30
+ def _dict_value(payload: Mapping[str, Any], key: str) -> dict[str, Any] | None:
31
+ value = payload.get(key)
32
+ if value is None:
33
+ return None
34
+ if not isinstance(value, dict):
35
+ raise TypeError(
36
+ f"translator: field {key!r} must be a dict, got {type(value).__name__}"
37
+ )
38
+ return dict(value)
39
+
40
+
41
+ def mail_folder_rows_to_ctx_models(
42
+ binding_id: str,
43
+ rows: Iterable[object],
44
+ ) -> Iterator[MailFolderRow]:
45
+ """Translate Graph delta mail-folder rows to MailFolderRow instances.
46
+
47
+ `@removed` rows from Graph carry only `id` + the `@removed` marker; they
48
+ become tombstones (`ctx_deleted=True`) so dlt's `hard_delete` deletes the
49
+ matching row at merge time. Live rows pass through with full field set.
50
+ """
51
+ for row in rows:
52
+ payload = _payload(row)
53
+ additional_data = _dict_value(payload, "additional_data") or {}
54
+ if "@removed" in additional_data or "@removed" in payload:
55
+ yield MailFolderRow(
56
+ ctx_binding_id=binding_id,
57
+ id=payload.get("id"),
58
+ ctx_deleted=True,
59
+ )
60
+ continue
61
+
62
+ yield MailFolderRow(
63
+ ctx_binding_id=binding_id,
64
+ id=payload.get("id"),
65
+ odata_type=payload.get("@odata.type"),
66
+ additional_data=additional_data,
67
+ child_folder_count=payload.get("childFolderCount"),
68
+ child_folders=_list_value(payload, "childFolders"),
69
+ display_name=payload.get("displayName"),
70
+ is_hidden=payload.get("isHidden"),
71
+ message_rules=_list_value(payload, "messageRules"),
72
+ messages=_list_value(payload, "messages"),
73
+ multi_value_extended_properties=_list_value(
74
+ payload,
75
+ "multiValueExtendedProperties",
76
+ ),
77
+ parent_folder_id=payload.get("parentFolderId"),
78
+ single_value_extended_properties=_list_value(
79
+ payload,
80
+ "singleValueExtendedProperties",
81
+ ),
82
+ total_item_count=payload.get("totalItemCount"),
83
+ unread_item_count=payload.get("unreadItemCount"),
84
+ )
85
+
86
+
87
+ def message_rows_to_ctx_models(
88
+ binding_id: str,
89
+ rows: Iterable[object],
90
+ *,
91
+ folder_id: str | None = None,
92
+ ) -> Iterator[MessageRow]:
93
+ """Translate Graph delta message rows to MessageRow instances.
94
+
95
+ `folder_id` is the folder whose delta produced these rows. It's used to
96
+ populate `parent_folder_id` on tombstone rows (`@removed` entries from Graph
97
+ contain only `id` + `@removed` and don't carry the folder).
98
+ """
99
+ for row in rows:
100
+ payload = _payload(row)
101
+ additional_data = _dict_value(payload, "additional_data") or {}
102
+ if "@removed" in additional_data or "@removed" in payload:
103
+ if folder_id is None:
104
+ raise RuntimeError(
105
+ "@removed message row received without folder_id context "
106
+ f"message_id={payload.get('id')!r}"
107
+ )
108
+ yield MessageRow(
109
+ ctx_binding_id=binding_id,
110
+ id=payload.get("id"),
111
+ parent_folder_id=folder_id,
112
+ ctx_deleted=True,
113
+ )
114
+ continue
115
+
116
+ yield MessageRow(
117
+ ctx_binding_id=binding_id,
118
+ ctx_source_updated_at=payload.get("lastModifiedDateTime"),
119
+ id=payload.get("id"),
120
+ odata_type=payload.get("@odata.type"),
121
+ etag=payload.get("@odata.etag"),
122
+ additional_data=additional_data,
123
+ attachments=_list_value(payload, "attachments"),
124
+ bcc_recipients=_list_value(payload, "bccRecipients"),
125
+ body=_dict_value(payload, "body"),
126
+ body_preview=payload.get("bodyPreview"),
127
+ categories=_list_value(payload, "categories"),
128
+ cc_recipients=_list_value(payload, "ccRecipients"),
129
+ change_key=payload.get("changeKey"),
130
+ conversation_id=payload.get("conversationId"),
131
+ conversation_index=payload.get("conversationIndex"),
132
+ created_date_time=payload.get("createdDateTime"),
133
+ extensions=_list_value(payload, "extensions"),
134
+ flag=_dict_value(payload, "flag"),
135
+ from_=_dict_value(payload, "from"),
136
+ has_attachments=payload.get("hasAttachments"),
137
+ importance=payload.get("importance"),
138
+ inference_classification=payload.get("inferenceClassification"),
139
+ internet_message_headers=_list_value(
140
+ payload,
141
+ "internetMessageHeaders",
142
+ ),
143
+ internet_message_id=payload.get("internetMessageId"),
144
+ is_delivery_receipt_requested=payload.get("isDeliveryReceiptRequested"),
145
+ is_draft=payload.get("isDraft"),
146
+ is_read=payload.get("isRead"),
147
+ is_read_receipt_requested=payload.get("isReadReceiptRequested"),
148
+ last_modified_date_time=payload.get("lastModifiedDateTime"),
149
+ multi_value_extended_properties=_list_value(
150
+ payload,
151
+ "multiValueExtendedProperties",
152
+ ),
153
+ parent_folder_id=payload.get("parentFolderId"),
154
+ received_date_time=payload.get("receivedDateTime"),
155
+ reply_to=_list_value(payload, "replyTo"),
156
+ sender=_dict_value(payload, "sender"),
157
+ sent_date_time=payload.get("sentDateTime"),
158
+ single_value_extended_properties=_list_value(
159
+ payload,
160
+ "singleValueExtendedProperties",
161
+ ),
162
+ subject=payload.get("subject"),
163
+ to_recipients=_list_value(payload, "toRecipients"),
164
+ unique_body=_dict_value(payload, "uniqueBody"),
165
+ web_link=payload.get("webLink"),
166
+ )
167
+
168
+
169
+ def attachment_content_row_from_graph_payload(
170
+ *,
171
+ binding_id: str,
172
+ message_id: str,
173
+ attachment_payload: Mapping[str, Any],
174
+ file_path: str,
175
+ ) -> AttachmentContentRow:
176
+ """Build an AttachmentContentRow from a Graph attachment object and a
177
+ locally-materialized file path."""
178
+ last_modified = attachment_payload.get("lastModifiedDateTime")
179
+ return AttachmentContentRow(
180
+ ctx_binding_id=binding_id,
181
+ ctx_source_updated_at=last_modified,
182
+ message_id=message_id,
183
+ attachment_id=attachment_payload.get("id"),
184
+ odata_type=attachment_payload.get("@odata.type"),
185
+ media_content_type=attachment_payload.get("@odata.mediaContentType"),
186
+ name=attachment_payload.get("name"),
187
+ content_type=attachment_payload.get("contentType"),
188
+ size=attachment_payload.get("size"),
189
+ is_inline=attachment_payload.get("isInline"),
190
+ content_id=attachment_payload.get("contentId"),
191
+ content_location=attachment_payload.get("contentLocation"),
192
+ last_modified_date_time=last_modified,
193
+ file_path=file_path,
194
+ )
@@ -0,0 +1,26 @@
1
+ {
2
+ "auth": {
3
+ "provider_id": "microsoft",
4
+ "scopes": ["Mail.ReadWrite", "Mail.Send"],
5
+ "type": "oauth"
6
+ },
7
+ "mcpServers": {
8
+ "microsoft_mail": {
9
+ "enabledTools": {
10
+ "create-draft-email": {},
11
+ "create-reply-draft": {},
12
+ "get-mail-message": { "autoAllow": true },
13
+ "list-mail-folder-messages": { "autoAllow": true },
14
+ "list-mail-folders": { "autoAllow": true },
15
+ "list-mail-messages": { "autoAllow": true },
16
+ "reply-mail-message": {},
17
+ "send-draft-message": {},
18
+ "send-mail": {}
19
+ },
20
+ "type": "http",
21
+ "url": "https://mcp-microsoft-365.usecontextlayer.com/mcp"
22
+ }
23
+ },
24
+ "mode": "dagster",
25
+ "plugin_id": "microsoft_mail"
26
+ }
@@ -0,0 +1 @@
1
+ """DLT sources for Microsoft Mail."""
@@ -0,0 +1,392 @@
1
+ from __future__ import annotations
2
+
3
+ import logging
4
+ import time
5
+ import uuid
6
+ from collections.abc import Iterator, Mapping
7
+ from dataclasses import dataclass
8
+ from typing import Any, Literal, Self
9
+
10
+ import dlt
11
+ from dlt.destinations.sql_client import SqlClientBase
12
+ from pydantic import model_validator
13
+ from shared_plugins.dlt import destination_has_table
14
+ from shared_plugins.microsoft_graph import graph_object_to_payload
15
+ from shared_plugins.models import IdStr, StrictModel
16
+ from shared_plugins.naming import (
17
+ dlt_resource_name,
18
+ dlt_source_name,
19
+ plugin_id_from_module,
20
+ )
21
+ from shared_plugins.resources import ctx_dlt_resource
22
+
23
+ from ..models.ctx import (
24
+ ATTACHMENT_CONTENT_COLUMN_DESCRIPTIONS,
25
+ AttachmentContentRow,
26
+ )
27
+ from ..utils.attachments import (
28
+ FILE_ATTACHMENT_ODATA_TYPE,
29
+ KNOWN_ATTACHMENT_ODATA_TYPES,
30
+ materialize_attachment_payloads,
31
+ )
32
+ from ..utils.client import SyncGraphMailClient
33
+
34
+ PLUGIN_ID = plugin_id_from_module(__file__)
35
+ JOB = "attachment_content"
36
+ SOURCE_NAME = dlt_source_name(PLUGIN_ID, JOB)
37
+ LOGGER = logging.getLogger(__name__)
38
+ DEFAULT_CANDIDATE_LIMIT = 200
39
+ ATTACHMENT_PREFER_HEADER = 'IdType="ImmutableId"'
40
+
41
+
42
+ @dataclass(frozen=True)
43
+ class Candidate:
44
+ action: str # 'materialize' or 'orphan'
45
+ message_id: str
46
+ attachments: list[dict[str, Any]] | None = None # only for 'materialize'
47
+
48
+
49
+ class AttachmentCandidateProjection(StrictModel):
50
+ action: Literal["materialize", "orphan"]
51
+ message_id: IdStr
52
+ attachments: list[dict[str, Any]] | None = None
53
+
54
+ @model_validator(mode="after")
55
+ def _validate_payload_for_action(self) -> Self:
56
+ if self.action == "materialize" and self.attachments is None:
57
+ raise ValueError(
58
+ "materialize attachment candidate requires attachments payload"
59
+ )
60
+ if self.action == "orphan" and self.attachments is not None:
61
+ raise ValueError(
62
+ "orphan attachment candidate must not include attachments payload"
63
+ )
64
+ return self
65
+
66
+ def to_candidate(self) -> Candidate:
67
+ return Candidate(
68
+ action=self.action,
69
+ message_id=self.message_id,
70
+ attachments=(
71
+ [dict(attachment) for attachment in self.attachments]
72
+ if self.attachments is not None
73
+ else None
74
+ ),
75
+ )
76
+
77
+
78
+ def parse_attachment_candidate(row: Mapping[str, Any]) -> Candidate:
79
+ return AttachmentCandidateProjection.model_validate(dict(row)).to_candidate()
80
+
81
+
82
+ def build_materialize_arm_query() -> str:
83
+ return """
84
+ WITH latest_message AS (
85
+ SELECT DISTINCT ON (_ctx_binding_id, id)
86
+ _ctx_binding_id,
87
+ id,
88
+ attachments
89
+ FROM messages
90
+ WHERE _ctx_binding_id = %s
91
+ ORDER BY _ctx_binding_id, id, last_modified_date_time DESC NULLS LAST
92
+ )
93
+ SELECT
94
+ 'materialize'::text AS action,
95
+ m.id AS message_id,
96
+ m.attachments AS attachments
97
+ FROM latest_message AS m
98
+ WHERE (
99
+ SELECT count(*)
100
+ FROM jsonb_array_elements(COALESCE(m.attachments, '[]'::jsonb)) AS a
101
+ WHERE COALESCE(a->>'@odata.type', '') NOT IN (
102
+ '#microsoft.graph.referenceAttachment',
103
+ '#microsoft.graph.itemAttachment'
104
+ )
105
+ ) <> (
106
+ SELECT count(*)
107
+ FROM attachment_content AS c
108
+ WHERE c._ctx_binding_id = m._ctx_binding_id
109
+ AND c.message_id = m.id
110
+ )
111
+ """.strip()
112
+
113
+
114
+ def build_orphan_arm_query() -> str:
115
+ return """
116
+ SELECT
117
+ 'orphan'::text AS action,
118
+ c.message_id,
119
+ NULL::jsonb AS attachments
120
+ FROM attachment_content AS c
121
+ LEFT JOIN messages AS m
122
+ ON m._ctx_binding_id = c._ctx_binding_id
123
+ AND m.id = c.message_id
124
+ WHERE c._ctx_binding_id = %s
125
+ AND m.id IS NULL
126
+ GROUP BY c.message_id
127
+ """.strip()
128
+
129
+
130
+ def build_bootstrap_materialize_arm_query() -> str:
131
+ """Materialize-arm variant for the case where `attachment_content` table
132
+ doesn't exist yet (first run after a clean reseed). Hardcodes the
133
+ existing-count subquery to 0 instead of selecting from `attachment_content`,
134
+ so the candidate query doesn't fail on a missing relation.
135
+ """
136
+ return """
137
+ WITH latest_message AS (
138
+ SELECT DISTINCT ON (_ctx_binding_id, id)
139
+ _ctx_binding_id,
140
+ id,
141
+ attachments
142
+ FROM messages
143
+ WHERE _ctx_binding_id = %s
144
+ ORDER BY _ctx_binding_id, id, last_modified_date_time DESC NULLS LAST
145
+ )
146
+ SELECT
147
+ 'materialize'::text AS action,
148
+ m.id AS message_id,
149
+ m.attachments AS attachments
150
+ FROM latest_message AS m
151
+ WHERE (
152
+ SELECT count(*)
153
+ FROM jsonb_array_elements(m.attachments) AS a
154
+ WHERE COALESCE(a->>'@odata.type', '') NOT IN (
155
+ '#microsoft.graph.referenceAttachment',
156
+ '#microsoft.graph.itemAttachment'
157
+ )
158
+ ) > 0
159
+ """.strip()
160
+
161
+
162
+ def build_combined_candidate_query(*, limit: int) -> str:
163
+ materialize = build_materialize_arm_query()
164
+ orphan = build_orphan_arm_query()
165
+ return f"""
166
+ WITH candidates AS (
167
+ {materialize}
168
+ UNION ALL
169
+ {orphan}
170
+ )
171
+ SELECT action, message_id, attachments
172
+ FROM candidates
173
+ ORDER BY message_id ASC
174
+ LIMIT %s
175
+ """.strip()
176
+
177
+
178
+ def iter_candidates(
179
+ sql_client: SqlClientBase[Any],
180
+ *,
181
+ binding_id: str,
182
+ limit: int,
183
+ ) -> list[Candidate]:
184
+ if not destination_has_table(sql_client, "messages"):
185
+ raise RuntimeError(
186
+ "messages table does not exist yet; run microsoft-mail-dlt-sync first"
187
+ )
188
+
189
+ if not destination_has_table(sql_client, "attachment_content"):
190
+ # Bootstrap: orphan arm has no rows AND we can't reference
191
+ # attachment_content yet. Use the bootstrap materialize-arm SQL.
192
+ query = (
193
+ build_bootstrap_materialize_arm_query()
194
+ + "\nORDER BY message_id ASC\nLIMIT %s"
195
+ )
196
+ params = (binding_id, limit)
197
+ else:
198
+ query = build_combined_candidate_query(limit=limit)
199
+ params = (binding_id, binding_id, limit)
200
+
201
+ candidates: list[Candidate] = []
202
+ with sql_client.execute_query(query, *params) as cursor:
203
+ if cursor.description is None:
204
+ return candidates
205
+ columns = [c[0] for c in cursor.description]
206
+ for row in cursor.fetchall():
207
+ raw = dict(zip(columns, row))
208
+ candidates.append(parse_attachment_candidate(raw))
209
+ return candidates
210
+
211
+
212
+ def _materializable_attachments(
213
+ attachments: list[dict[str, Any]],
214
+ ) -> list[dict[str, Any]]:
215
+ """Filter to file attachments. Reference and item attachments produce no
216
+ row in `attachment_content`. An unknown `@odata.type` is treated as a loud
217
+ error per spec §8 — Graph's attachment subtype set is closed."""
218
+ materializable: list[dict[str, Any]] = []
219
+ for a in attachments:
220
+ odata_type = a.get("@odata.type")
221
+ if odata_type not in KNOWN_ATTACHMENT_ODATA_TYPES:
222
+ raise RuntimeError(
223
+ f"unknown attachment @odata.type {odata_type!r}; "
224
+ f"expected one of {sorted(KNOWN_ATTACHMENT_ODATA_TYPES)}"
225
+ )
226
+ if odata_type == FILE_ATTACHMENT_ODATA_TYPE:
227
+ materializable.append(a)
228
+ return materializable
229
+
230
+
231
+ def fetch_and_emit_for_message(
232
+ *,
233
+ binding_id: str,
234
+ client: SyncGraphMailClient,
235
+ candidate: Candidate,
236
+ ) -> Iterator[AttachmentContentRow]:
237
+ if candidate.attachments is None:
238
+ raise RuntimeError(
239
+ f"materialize candidate without attachments payload message_id={candidate.message_id}"
240
+ )
241
+
242
+ full_payloads: list[dict[str, Any]] = []
243
+ for raw_attachment in _materializable_attachments(candidate.attachments):
244
+ attachment_id = raw_attachment.get("id")
245
+ if not isinstance(attachment_id, str):
246
+ raise RuntimeError(
247
+ "attachment in messages.attachments has no id "
248
+ f"message_id={candidate.message_id} payload_keys={sorted(raw_attachment)}"
249
+ )
250
+ full = client.get_attachment_full(
251
+ message_id=candidate.message_id,
252
+ attachment_id=attachment_id,
253
+ prefer_header=ATTACHMENT_PREFER_HEADER,
254
+ )
255
+ # `FileAttachment.content_bytes` holds the decoded file bytes and the
256
+ # Kiota JSON writer re-encodes them as one base64 string — requires
257
+ # microsoft-kiota-serialization-json>=1.11.7 (floor declared in
258
+ # shared_plugins), where upstream fixed get_bytes_value to base64-decode
259
+ # on deserialization. Older kiota left the wire base64 string undecoded,
260
+ # which double-encoded through the writer and silently corrupted files.
261
+ full_payloads.append(graph_object_to_payload(full))
262
+
263
+ rows = materialize_attachment_payloads(
264
+ binding_id=binding_id,
265
+ message_id=candidate.message_id,
266
+ attachment_payloads=full_payloads,
267
+ )
268
+ yield from rows
269
+
270
+
271
+ TOMBSTONE_ATTACHMENT_ID_PREFIX = "_ctx_tombstone:"
272
+
273
+
274
+ def _generate_tombstone_attachment_id() -> str:
275
+ """Build a unique sentinel `attachment_id` for a tombstone row.
276
+
277
+ The prefix marks intent; the random uuid4 suffix prevents any conceivable
278
+ collision with a real Graph attachment id and guarantees uniqueness
279
+ across multiple tombstones in the same merge.
280
+ """
281
+ return f"{TOMBSTONE_ATTACHMENT_ID_PREFIX}{uuid.uuid4()}"
282
+
283
+
284
+ def emit_orphan_tombstone(
285
+ *,
286
+ binding_id: str,
287
+ candidate: Candidate,
288
+ ) -> AttachmentContentRow:
289
+ """Build a tombstone row that dlt will use to delete all attachment_content
290
+ rows matching merge_key=(_ctx_binding_id, message_id).
291
+
292
+ The PK includes `attachment_id` (Postgres requires PK columns NOT NULL),
293
+ so the tombstone carries a sentinel id with the form
294
+ `_ctx_tombstone:<uuid4>`. dlt's `hard_delete` on `_ctx_deleted` deletes
295
+ the sentinel-bearing row at the end of the merge, so the sentinel never
296
+ persists in the destination."""
297
+ return AttachmentContentRow(
298
+ ctx_binding_id=binding_id,
299
+ message_id=candidate.message_id,
300
+ attachment_id=_generate_tombstone_attachment_id(),
301
+ ctx_deleted=True,
302
+ )
303
+
304
+
305
+ @dlt.source(name=SOURCE_NAME)
306
+ def microsoft_mail_attachment_source(
307
+ binding_id: str,
308
+ *,
309
+ client: SyncGraphMailClient,
310
+ ) -> tuple[Any, ...]:
311
+ @ctx_dlt_resource(
312
+ name=dlt_resource_name("attachment_content"),
313
+ write_disposition={"disposition": "merge", "strategy": "delete-insert"},
314
+ primary_key=("_ctx_binding_id", "message_id", "attachment_id"),
315
+ merge_key=("_ctx_binding_id", "message_id"),
316
+ columns={
317
+ **ATTACHMENT_CONTENT_COLUMN_DESCRIPTIONS,
318
+ # file_path is required on live rows but NULL on tombstones
319
+ # (ctx_deleted=True). dlt infers schema from first-observed data,
320
+ # so we declare it nullable upfront to match the pydantic Optional
321
+ # type. attachment_id stays NOT NULL because it's part of the PK
322
+ # — tombstones carry a sentinel value (TOMBSTONE_ATTACHMENT_ID).
323
+ "file_path": {"nullable": True},
324
+ "_ctx_deleted": {"data_type": "bool", "hard_delete": True},
325
+ },
326
+ )
327
+ def attachment_content() -> Iterator[AttachmentContentRow]:
328
+ # `schema_name=SOURCE_NAME` avoids dlt's `Schema(pipeline_name)`
329
+ # fallback inside `_get_schema_or_create`. The pipeline_name for a
330
+ # per-binding attachment_content pipeline routinely exceeds dlt's
331
+ # 64-char Schema-name limit, and the source schema is what we
332
+ # actually want here. `run_dlt_pipeline` pre-registers an empty
333
+ # schema with this name on cold start so the lookup succeeds.
334
+ with dlt.current.pipeline().sql_client(
335
+ schema_name=SOURCE_NAME,
336
+ ) as sql_client:
337
+ candidates = iter_candidates(
338
+ sql_client,
339
+ binding_id=binding_id,
340
+ limit=DEFAULT_CANDIDATE_LIMIT,
341
+ )
342
+
343
+ run_started = time.monotonic()
344
+ materialized_messages = 0
345
+ materialized_rows = 0
346
+ orphaned_messages = 0
347
+
348
+ for candidate in candidates:
349
+ if candidate.action == "orphan":
350
+ orphaned_messages += 1
351
+ yield emit_orphan_tombstone(binding_id=binding_id, candidate=candidate)
352
+ continue
353
+
354
+ if candidate.action != "materialize":
355
+ raise RuntimeError(f"unknown candidate action {candidate.action!r}")
356
+
357
+ # Pre-filter: when the materialize arm fires for a message
358
+ # whose materializable count is zero (e.g. all file attachments
359
+ # were removed but reference attachments remain, or the
360
+ # attachments array is empty entirely), there's nothing to
361
+ # fetch. Emit a tombstone instead — dlt's delete-insert merge
362
+ # is a no-op when zero rows are emitted for a merge_key, so
363
+ # leftover rows would persist without the tombstone.
364
+ materializable = _materializable_attachments(candidate.attachments or [])
365
+ if not materializable:
366
+ orphaned_messages += 1
367
+ yield emit_orphan_tombstone(binding_id=binding_id, candidate=candidate)
368
+ continue
369
+
370
+ rows = list(
371
+ fetch_and_emit_for_message(
372
+ binding_id=binding_id,
373
+ client=client,
374
+ candidate=candidate,
375
+ )
376
+ )
377
+ materialized_messages += 1
378
+ materialized_rows += len(rows)
379
+ yield from rows
380
+
381
+ elapsed = time.monotonic() - run_started
382
+ LOGGER.info(
383
+ "microsoft_mail.attachment_content.run_complete "
384
+ "candidates=%d materialized_messages=%d rows=%d orphaned_messages=%d elapsed=%.3fs",
385
+ len(candidates),
386
+ materialized_messages,
387
+ materialized_rows,
388
+ orphaned_messages,
389
+ elapsed,
390
+ )
391
+
392
+ return (attachment_content,)