pubkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pubkit/core/browser.py ADDED
@@ -0,0 +1,426 @@
1
+ # Copyright 2026 The pubkit Authors
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """BrowserAdapter — driving a rich-text editor that was never meant to be driven.
4
+
5
+ Everything in this module was learned the hard way against Medium's editor.
6
+ Five things matter, and they generalise to Substack, Ghost's editor, LinkedIn
7
+ articles and anything else built on contenteditable:
8
+
9
+ 1. **Replace by pasting over a selection, never by deleting everything.**
10
+ `execCommand('delete')` across the whole document destroys the editor's
11
+ internal scaffolding; afterwards pastes land nowhere and the only recovery
12
+ is a reload.
13
+
14
+ 2. **Trust nothing until it survives a reload.** A delete that took 355
15
+ paragraphs to 189 came back as 339. A link href set directly in the DOM
16
+ reverted completely. The editor keeps its own model and syncs deltas.
17
+
18
+ 3. **Images cannot travel inside pasted HTML.** `data:` URIs are stripped and
19
+ so are ordinary `https://` URLs. The only path that works is a synthetic
20
+ `ClipboardEvent` carrying a real `File` in its `DataTransfer`, which makes
21
+ the editor run its own upload.
22
+
23
+ 4. **Getting bytes into the page is the actual hard problem.** An injected
24
+ `<input type=file>` plus the automation layer's file-upload primitive
25
+ solves it without ever opening a native picker.
26
+
27
+ 5. **Anchors shift.** Every insert renumbers the document. Re-resolve by text
28
+ after every mutation, never cache an index.
29
+ """
30
+ from __future__ import annotations
31
+
32
+ import asyncio
33
+ import logging
34
+ from collections.abc import Sequence
35
+ from dataclasses import dataclass
36
+ from pathlib import Path
37
+ from typing import Any
38
+
39
+ from .adapter import BaseAdapter, Fingerprint, VerificationFailed
40
+ from .anchors import AnchorNotFound, find_anchor, similarity
41
+ from .transport import ROLLING_HASH_JS
42
+
43
+ log = logging.getLogger(__name__)
44
+
45
+
46
+ class DocumentReplaceMismatch(RuntimeError):
47
+ """Paste-replace produced the wrong paragraph count — almost always means
48
+ the selection covered part of the document (failure B1)."""
49
+
50
+
51
+ @dataclass
52
+ class EditorSelectors:
53
+ """Everything platform-specific about an editor, in one place.
54
+
55
+ A new browser platform is mostly this dataclass plus a publish click.
56
+ """
57
+
58
+ editable: str
59
+ #: All content roots. Getting this wrong — grabbing only the first — is what
60
+ #: made a paste insert instead of replace (failure B1).
61
+ content_roots: str
62
+ block: str
63
+ figure: str
64
+ figure_img: str
65
+ link: str
66
+ publish_button: str
67
+ tag_input: str | None = None
68
+ tag_chip: str | None = None
69
+ confirm_publish: str | None = None
70
+
71
+
72
+ UPLOAD_INPUT_ID = "__pubkit_upload"
73
+
74
+ _JS_HELPERS = (
75
+ ROLLING_HASH_JS
76
+ + """
77
+ window.__pk = window.__pk || {};
78
+
79
+ window.__pk.roots = function (sel) {
80
+ return Array.from(document.querySelectorAll(sel));
81
+ };
82
+
83
+ window.__pk.blocks = function (rootSel, blockSel) {
84
+ // Across ALL roots, not just the first. This single detail is the
85
+ // difference between replacing a document and duplicating it.
86
+ const out = [];
87
+ for (const root of window.__pk.roots(rootSel)) {
88
+ out.push(...Array.from(root.querySelectorAll(blockSel)));
89
+ }
90
+ return out;
91
+ };
92
+
93
+ window.__pk.texts = function (rootSel, blockSel) {
94
+ return window.__pk.blocks(rootSel, blockSel).map(e => (e.innerText || '').trim());
95
+ };
96
+
97
+ window.__pk.selectAll = function (rootSel, blockSel) {
98
+ const g = window.__pk.blocks(rootSel, blockSel);
99
+ if (!g.length) return 0;
100
+ const r = document.createRange();
101
+ r.setStartBefore(g[0]);
102
+ r.setEndAfter(g[g.length - 1]);
103
+ const s = getSelection();
104
+ s.removeAllRanges();
105
+ s.addRange(r);
106
+ return g.length;
107
+ };
108
+
109
+ window.__pk.caretAtEndOf = function (rootSel, blockSel, index) {
110
+ const g = window.__pk.blocks(rootSel, blockSel);
111
+ if (index < 0 || index >= g.length) return false;
112
+ const r = document.createRange();
113
+ r.selectNodeContents(g[index]);
114
+ r.collapse(false);
115
+ const s = getSelection();
116
+ s.removeAllRanges();
117
+ s.addRange(r);
118
+ return true;
119
+ };
120
+
121
+ window.__pk.pasteHtml = function (editableSel, html) {
122
+ const ed = document.querySelector(editableSel);
123
+ ed.focus();
124
+ const dt = new DataTransfer();
125
+ dt.setData('text/html', html);
126
+ ed.dispatchEvent(new ClipboardEvent('paste', {
127
+ clipboardData: dt, bubbles: true, cancelable: true,
128
+ }));
129
+ return true;
130
+ };
131
+
132
+ window.__pk.pasteFile = function (editableSel, fileIndex) {
133
+ // The ONLY image path that survives. Remote URLs and data: URIs are
134
+ // stripped from pasted HTML; a real File in DataTransfer.items makes the
135
+ // editor run its own uploader and put the result on its CDN.
136
+ const input = document.getElementById('%UPLOAD_ID%');
137
+ if (!input || !input.files || !input.files[fileIndex]) return 'nofile';
138
+ const ed = document.querySelector(editableSel);
139
+ ed.focus();
140
+ const dt = new DataTransfer();
141
+ dt.items.add(input.files[fileIndex]);
142
+ ed.dispatchEvent(new ClipboardEvent('paste', {
143
+ clipboardData: dt, bubbles: true, cancelable: true,
144
+ }));
145
+ return 'ok';
146
+ };
147
+
148
+ window.__pk.ensureUploadInput = function () {
149
+ let i = document.getElementById('%UPLOAD_ID%');
150
+ if (!i) {
151
+ i = document.createElement('input');
152
+ i.type = 'file';
153
+ i.id = '%UPLOAD_ID%';
154
+ i.multiple = true;
155
+ // Visible enough to be a real element, invisible enough not to matter.
156
+ i.style.cssText =
157
+ 'position:fixed;top:0;left:0;z-index:2147483647;opacity:0.01;width:80px;height:24px';
158
+ document.body.appendChild(i);
159
+ }
160
+ return true;
161
+ };
162
+
163
+ window.__pk.stage = function (key, index, data) {
164
+ sessionStorage.setItem(key + ':' + index, data);
165
+ return window.__pkHash(data);
166
+ };
167
+
168
+ window.__pk.stagedHash = function (key) {
169
+ const v = sessionStorage.getItem(key);
170
+ return v === null ? null : window.__pkHash(v);
171
+ };
172
+
173
+ window.__pk.assemble = function (key, count) {
174
+ let out = '';
175
+ for (let i = 0; i < count; i++) out += sessionStorage.getItem(key + ':' + i) || '';
176
+ sessionStorage.setItem(key, out);
177
+ for (let i = 0; i < count; i++) sessionStorage.removeItem(key + ':' + i);
178
+ return window.__pkHash(out);
179
+ };
180
+
181
+ window.__pk.inflate = async function (key) {
182
+ // Single gzip member only; concatenated members throw here even though
183
+ // Python's gzip.decompress accepts them.
184
+ const b64 = sessionStorage.getItem(key);
185
+ const bin = atob(b64);
186
+ const a = new Uint8Array(bin.length);
187
+ for (let i = 0; i < bin.length; i++) a[i] = bin.charCodeAt(i);
188
+ const ds = new DecompressionStream('gzip');
189
+ const w = ds.writable.getWriter();
190
+ w.write(a); w.close();
191
+ const rd = ds.readable.getReader();
192
+ const parts = []; let n = 0;
193
+ for (;;) { const { done, value } = await rd.read(); if (done) break; parts.push(value); n += value.length; }
194
+ const o = new Uint8Array(n); let k = 0;
195
+ for (const p of parts) { o.set(p, k); k += p.length; }
196
+ return new TextDecoder().decode(o);
197
+ };
198
+
199
+ window.__pk.fingerprint = function (rootSel, blockSel, figSel, linkSel, markerRe) {
200
+ const roots = window.__pk.roots(rootSel);
201
+ const text = roots.map(r => r.innerText).join('\\n');
202
+ const blocks = window.__pk.blocks(rootSel, blockSel);
203
+ return {
204
+ words: text.split(/\\s+/).filter(Boolean).length,
205
+ headings: blocks.filter(e => /^H[1-6]$/.test(e.tagName)).map(e => e.innerText.trim().slice(0, 48)),
206
+ images: roots.reduce((n, r) => n + r.querySelectorAll(figSel).length, 0),
207
+ links: roots.reduce((n, r) => n + r.querySelectorAll(linkSel).length, 0),
208
+ markers: (text.match(new RegExp(markerRe, 'g')) || []).length,
209
+ };
210
+ };
211
+ """.replace("%UPLOAD_ID%", UPLOAD_INPUT_ID)
212
+ )
213
+
214
+
215
+ class BrowserAdapter(BaseAdapter):
216
+ """Base class for editor-driven platforms.
217
+
218
+ Subclasses supply `selectors`, a login URL, a draft URL builder and a
219
+ publish routine. Everything below is platform-agnostic.
220
+ """
221
+
222
+ selectors: EditorSelectors
223
+ login_url: str = ""
224
+ marker_regex: str = r"\\[\\[\\s*IMAGE"
225
+
226
+ def __init__(self) -> None:
227
+ super().__init__()
228
+ self._page: Any = None
229
+
230
+ # ------------------------------------------------------------------ page
231
+ async def _eval(self, script: str, *args) -> Any:
232
+ return await self._page.evaluate(script, *args)
233
+
234
+ async def install_helpers(self) -> None:
235
+ await self._eval(_JS_HELPERS)
236
+ await self._eval("window.__pk.ensureUploadInput()")
237
+
238
+ # ------------------------------------------------------- document replace
239
+ async def replace_document(self, html: str, *, expect_blocks: int | None = None) -> int:
240
+ """Replace the entire document body by pasting over a full selection.
241
+
242
+ Deliberately *not* implemented as delete-then-paste. Deleting the whole
243
+ block range breaks the editor badly enough that a reload is the only
244
+ way back (failure B2).
245
+ """
246
+ s = self.selectors
247
+ before = await self._eval(
248
+ "([r,b]) => window.__pk.selectAll(r,b)", [s.content_roots, s.block]
249
+ )
250
+ if not before:
251
+ raise DocumentReplaceMismatch("no blocks found — selectors are wrong or page not ready")
252
+
253
+ await self._eval("([e,h]) => window.__pk.pasteHtml(e,h)", [s.editable, html])
254
+ await asyncio.sleep(2.5)
255
+
256
+ after = await self._eval(
257
+ "([r,b]) => window.__pk.blocks(r,b).length", [s.content_roots, s.block]
258
+ )
259
+ if expect_blocks is not None and after != expect_blocks:
260
+ raise DocumentReplaceMismatch(
261
+ f"expected {expect_blocks} blocks after replace, got {after} "
262
+ f"(was {before}). If after ≈ before + expected, the selection "
263
+ f"covered only one content root — check `content_roots`."
264
+ )
265
+ log.info("replaced document: %d blocks → %d blocks", before, after)
266
+ return after
267
+
268
+ # ----------------------------------------------------------- anchor lookup
269
+ async def block_texts(self) -> list[str]:
270
+ s = self.selectors
271
+ return await self._eval("([r,b]) => window.__pk.texts(r,b)", [s.content_roots, s.block])
272
+
273
+ async def resolve_anchor(self, target: str, *, used: set[int] | None = None) -> int:
274
+ texts = await self.block_texts()
275
+ idx = find_anchor(texts, target, used=used)
276
+ if idx < 0:
277
+ scored = sorted(((similarity(t, target), t) for t in texts), reverse=True)
278
+ best = scored[0] if scored else (0.0, None)
279
+ raise AnchorNotFound(target, best[1], best[0])
280
+ return idx
281
+
282
+ # ------------------------------------------------------------------ media
283
+ async def insert_image_at(self, caption: str, file_index: int, *, used: set[int] | None = None) -> None:
284
+ """Place an image directly above the paragraph that carries `caption`.
285
+
286
+ The figure lands *before* the caret's paragraph, so anchoring on the
287
+ caption gives the right visual result for free (failure B7). Anchors are
288
+ re-resolved every time because every insert shifts the indices.
289
+ """
290
+ s = self.selectors
291
+ idx = await self.resolve_anchor(caption, used=used)
292
+ ok = await self._eval(
293
+ "([r,b,i]) => window.__pk.caretAtEndOf(r,b,i)", [s.content_roots, s.block, idx]
294
+ )
295
+ if not ok:
296
+ raise AnchorNotFound(caption, None, 0.0)
297
+
298
+ before = await self._count_figures()
299
+ res = await self._eval("([e,i]) => window.__pk.pasteFile(e,i)", [s.editable, file_index])
300
+ if res == "nofile":
301
+ raise RuntimeError(
302
+ f"no staged file at index {file_index}; call stage_files() first"
303
+ )
304
+ await self._wait_for_figure(before + 1)
305
+
306
+ async def _count_figures(self) -> int:
307
+ s = self.selectors
308
+ return await self._eval(
309
+ "([r,f]) => window.__pk.roots(r).reduce((n,x)=>n+x.querySelectorAll(f).length,0)",
310
+ [s.content_roots, s.figure],
311
+ )
312
+
313
+ async def _wait_for_figure(self, want: int, timeout: float = 45.0) -> None:
314
+ """Wait for the editor's own uploader to finish.
315
+
316
+ Polls for the figure to exist *and* for its src to point at the
317
+ platform's CDN rather than a blob: URL — a blob means the upload has
318
+ not completed and publishing now would produce a broken image.
319
+ """
320
+ s = self.selectors
321
+ deadline = asyncio.get_event_loop().time() + timeout
322
+ while asyncio.get_event_loop().time() < deadline:
323
+ state = await self._eval(
324
+ "([r,f,i]) => { const roots = window.__pk.roots(r);"
325
+ " const figs = roots.flatMap(x => Array.from(x.querySelectorAll(f)));"
326
+ " return { n: figs.length, blobs: figs.filter(g => { const im = g.querySelector(i);"
327
+ " return !im || !im.src || im.src.startsWith('blob:') || im.src.startsWith('data:'); }).length }; }",
328
+ [s.content_roots, s.figure, s.figure_img],
329
+ )
330
+ if state["n"] >= want and state["blobs"] == 0:
331
+ return
332
+ await asyncio.sleep(1.0)
333
+ raise VerificationFailed(
334
+ f"image upload did not settle within {timeout:.0f}s "
335
+ "(figure missing, or src still a blob: URL — the platform has not stored it)"
336
+ )
337
+
338
+ async def stage_files(self, paths: Sequence[Path], upload) -> None:
339
+ """Hand real bytes to the page.
340
+
341
+ `upload` is the automation layer's file-upload primitive, which sets
342
+ `input.files` on an element. We inject the input ourselves so no native
343
+ file picker is ever opened — a picker is undriveable and blocks the
344
+ whole session.
345
+ """
346
+ await self.install_helpers()
347
+ await upload(f"#{UPLOAD_INPUT_ID}", [str(p) for p in paths])
348
+ n = await self._eval(
349
+ f"() => {{ const i = document.getElementById('{UPLOAD_INPUT_ID}');"
350
+ " return i && i.files ? i.files.length : 0; }"
351
+ )
352
+ if n != len(paths):
353
+ raise RuntimeError(f"staged {n} files but expected {len(paths)}")
354
+
355
+ # ------------------------------------------------------------ verification
356
+ async def fingerprint(self) -> Fingerprint:
357
+ s = self.selectors
358
+ raw = await self._eval(
359
+ "([r,b,f,l,m]) => window.__pk.fingerprint(r,b,f,l,m)",
360
+ [s.content_roots, s.block, s.figure, s.link, self.marker_regex],
361
+ )
362
+ return Fingerprint(**raw)
363
+
364
+ async def verify_after_reload(self, expected: Fingerprint, reload_fn) -> Fingerprint:
365
+ """Reload, then compare. The only honest verification (failure B3).
366
+
367
+ An editor will happily show you a change it has not persisted. Asking
368
+ the server what it actually has is the difference between "published
369
+ correctly" and "published, and broken for every reader".
370
+ """
371
+ await reload_fn()
372
+ await asyncio.sleep(3.0)
373
+ await self.install_helpers()
374
+ actual = await self.fingerprint()
375
+ actual.assert_matches(expected)
376
+ return actual
377
+
378
+ # -------------------------------------------------------------------- tags
379
+ async def apply_tags(self, tags: list[str], strategy: str) -> int:
380
+ """Commit tags, then *check* that chips actually appeared.
381
+
382
+ Typing `a,b,c` into a framework-controlled tag field once produced a
383
+ single 40-character invalid tag. Publishing without tags is a small
384
+ loss; publishing with one garbage tag is worse (failure B9).
385
+ """
386
+ s = self.selectors
387
+ if not s.tag_input or not tags:
388
+ return 0
389
+ await self._page.click(s.tag_input)
390
+ if strategy == "comma":
391
+ for t in tags:
392
+ await self._page.type(s.tag_input, t + ",", delay=40)
393
+ await asyncio.sleep(0.4)
394
+ elif strategy == "enter":
395
+ for t in tags:
396
+ await self._page.type(s.tag_input, t, delay=40)
397
+ await self._page.keyboard.press("Enter")
398
+ await asyncio.sleep(0.4)
399
+ else: # native_setter — drive the framework's own value setter
400
+ for t in tags:
401
+ await self._eval(
402
+ "([sel,val]) => { const el = document.querySelector(sel);"
403
+ " const proto = Object.getPrototypeOf(el);"
404
+ " const d = Object.getOwnPropertyDescriptor(proto, 'value');"
405
+ " d.set.call(el, val);"
406
+ " el.dispatchEvent(new Event('input', { bubbles: true }));"
407
+ " el.dispatchEvent(new KeyboardEvent('keydown', { key: 'Enter', bubbles: true })); }",
408
+ [s.tag_input, t],
409
+ )
410
+ await asyncio.sleep(0.4)
411
+
412
+ if not s.tag_chip:
413
+ return len(tags)
414
+ chips = await self._eval(f"() => document.querySelectorAll({s.tag_chip!r}).length")
415
+ if chips == 0:
416
+ log.warning(
417
+ "tag strategy %r committed no chips; publishing without tags "
418
+ "rather than risking one malformed tag",
419
+ strategy,
420
+ )
421
+ await self._eval(
422
+ "(sel) => { const el = document.querySelector(sel); if (el) { el.value=''; "
423
+ "el.dispatchEvent(new Event('input',{bubbles:true})); } }",
424
+ s.tag_input,
425
+ )
426
+ return chips
@@ -0,0 +1,256 @@
1
+ # Copyright 2026 The pubkit Authors
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Capability declarations and the planner that negotiates against them.
4
+
5
+ The planner's job is to turn "here is my content" plus "here is what this
6
+ platform can do" into an explicit, printable list of *degradations*. Surprises
7
+ at publish time are the enemy; `pubkit plan` exists so there are none.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ from enum import Enum
12
+ from typing import Literal
13
+
14
+ from pydantic import BaseModel, Field
15
+
16
+ from .ir import Document, Table
17
+
18
+
19
+ class TagSpec(BaseModel):
20
+ model_config = {"extra": "forbid"}
21
+
22
+ max_count: int = 5
23
+ max_len: int = 25
24
+ #: How the platform's tag widget actually commits a tag (failure B9).
25
+ strategy: Literal["comma", "enter", "native_setter", "api"] = "api"
26
+
27
+
28
+ class Capabilities(BaseModel):
29
+ model_config = {"extra": "forbid"}
30
+
31
+ tables: bool = True
32
+ code_blocks: Literal["fenced", "highlighted", "plain", "none"] = "fenced"
33
+ inline_html: bool = False
34
+ animated_gif: bool = True
35
+ headings: int = 6
36
+ max_body_chars: int | None = None
37
+ image_upload: Literal["api", "browser_paste", "none"] = "api"
38
+ canonical_url: bool = False
39
+ tags: TagSpec | None = None
40
+ scheduling: bool = False
41
+ threads: bool = False
42
+ #: Whether drafts can be created and revisited, or publishing is one-shot.
43
+ drafts: bool = True
44
+
45
+
46
+ class DegradationKind(str, Enum):
47
+ TABLE_TO_IMAGE = "table_to_image"
48
+ TABLE_TO_LIST = "table_to_list"
49
+ HTML_STRIPPED = "html_stripped"
50
+ CODE_FLATTENED = "code_flattened"
51
+ HEADINGS_CLAMPED = "headings_clamped"
52
+ SPLIT_INTO_THREAD = "split_into_thread"
53
+ GIF_TO_STILL = "gif_to_still"
54
+ TAGS_TRIMMED = "tags_trimmed"
55
+ IMAGES_VIA_BROWSER = "images_via_browser"
56
+ BODY_TRUNCATED = "body_truncated"
57
+
58
+
59
+ class Degradation(BaseModel):
60
+ model_config = {"extra": "forbid"}
61
+
62
+ kind: DegradationKind
63
+ detail: str
64
+ count: int = 1
65
+ #: True when the degradation loses information the author may care about.
66
+ lossy: bool = False
67
+
68
+
69
+ class PublishPlan(BaseModel):
70
+ model_config = {"extra": "forbid"}
71
+
72
+ document_id: str
73
+ content_id: str
74
+ platform: str
75
+ degradations: list[Degradation] = Field(default_factory=list)
76
+ #: Assets the renderer must generate before publishing (e.g. table PNGs).
77
+ generated_assets: list[str] = Field(default_factory=list)
78
+ estimated_body_chars: int = 0
79
+ blocking: list[str] = Field(default_factory=list)
80
+
81
+ @property
82
+ def ok(self) -> bool:
83
+ return not self.blocking
84
+
85
+ def human(self) -> str:
86
+ lines = [f"{self.platform}: {self.document_id} ({self.content_id[:12]})"]
87
+ if not self.degradations:
88
+ lines.append(" no degradation — native fidelity")
89
+ for d in self.degradations:
90
+ mark = "!" if d.lossy else "·"
91
+ lines.append(f" {mark} {d.kind.value:<20} {d.detail}")
92
+ for b in self.blocking:
93
+ lines.append(f" ✗ BLOCKING {b}")
94
+ return "\n".join(lines)
95
+
96
+
97
+ def plain_text_length(doc: Document) -> int:
98
+ """Characters a reader actually sees."""
99
+ total = len(doc.title) + len(doc.subtitle or "")
100
+ for b in doc.blocks:
101
+ text = getattr(b, "text", None)
102
+ if text:
103
+ total += len(text)
104
+ items = getattr(b, "items", None)
105
+ if items:
106
+ total += sum(len(i) for i in items)
107
+ if isinstance(b, Table):
108
+ total += sum(len(c) for c in b.header)
109
+ total += sum(len(c) for row in b.rows for c in row)
110
+ return total
111
+
112
+
113
+ def plan(doc: Document, platform: str, caps: Capabilities, adapter: object | None = None) -> PublishPlan:
114
+ """Intersect a document with a platform's capabilities.
115
+
116
+ Static capabilities cannot express everything: X in promo mode never
117
+ overflows 280 chars because it does not serialise the body at all. An
118
+ adapter may therefore refine its own plan — the last word belongs to the
119
+ code that will actually do the work.
120
+ """
121
+ p = PublishPlan(document_id=doc.id, content_id=doc.content_id, platform=platform)
122
+
123
+ # --- tables -----------------------------------------------------------
124
+ tables = doc.tables
125
+ if tables and not caps.tables:
126
+ if caps.image_upload != "none":
127
+ p.degradations.append(
128
+ Degradation(
129
+ kind=DegradationKind.TABLE_TO_IMAGE,
130
+ detail=f"{len(tables)} table(s) rendered as images — platform has no table support",
131
+ count=len(tables),
132
+ )
133
+ )
134
+ p.generated_assets.extend(f"table-{i:02d}" for i in range(len(tables)))
135
+ else:
136
+ p.degradations.append(
137
+ Degradation(
138
+ kind=DegradationKind.TABLE_TO_LIST,
139
+ detail=f"{len(tables)} table(s) flattened to lists",
140
+ count=len(tables),
141
+ lossy=True,
142
+ )
143
+ )
144
+
145
+ # --- figures ----------------------------------------------------------
146
+ figures = doc.figures
147
+ if figures:
148
+ if caps.image_upload == "none":
149
+ p.blocking.append(f"{len(figures)} figure(s) but platform accepts no images")
150
+ elif caps.image_upload == "browser_paste":
151
+ p.degradations.append(
152
+ Degradation(
153
+ kind=DegradationKind.IMAGES_VIA_BROWSER,
154
+ detail=(
155
+ f"{len(figures)} image(s) uploaded through the editor's own paste path "
156
+ "— remote URLs and data: URIs are stripped by this platform"
157
+ ),
158
+ count=len(figures),
159
+ )
160
+ )
161
+ if not caps.animated_gif:
162
+ gifs = [f for f in figures if doc.assets[f.asset_id].animated]
163
+ if gifs:
164
+ p.degradations.append(
165
+ Degradation(
166
+ kind=DegradationKind.GIF_TO_STILL,
167
+ detail=f"{len(gifs)} animation(s) reduced to a still frame",
168
+ count=len(gifs),
169
+ lossy=True,
170
+ )
171
+ )
172
+
173
+ # --- code -------------------------------------------------------------
174
+ if caps.code_blocks in ("plain", "none"):
175
+ n = sum(1 for b in doc.blocks if getattr(b, "type", None) == "code")
176
+ if n:
177
+ p.degradations.append(
178
+ Degradation(
179
+ kind=DegradationKind.CODE_FLATTENED,
180
+ detail=f"{n} code block(s) lose syntax highlighting",
181
+ count=n,
182
+ )
183
+ )
184
+
185
+ # --- headings ---------------------------------------------------------
186
+ deep = [b for b in doc.blocks if getattr(b, "type", None) == "heading" and b.level > caps.headings]
187
+ if deep:
188
+ p.degradations.append(
189
+ Degradation(
190
+ kind=DegradationKind.HEADINGS_CLAMPED,
191
+ detail=f"{len(deep)} heading(s) clamped to h{caps.headings}",
192
+ count=len(deep),
193
+ )
194
+ )
195
+
196
+ # --- length -----------------------------------------------------------
197
+ # Plain-text length, not the canonical JSON: JSON includes keys, quoting
198
+ # and escapes, which inflated a 3,300-word article into an estimated
199
+ # 94-tweet thread. An estimate that is wrong by 4x is worse than none.
200
+ approx = plain_text_length(doc)
201
+ p.estimated_body_chars = approx
202
+ if caps.max_body_chars and approx > caps.max_body_chars:
203
+ if caps.threads:
204
+ parts = -(-approx // caps.max_body_chars)
205
+ p.degradations.append(
206
+ Degradation(
207
+ kind=DegradationKind.SPLIT_INTO_THREAD,
208
+ detail=f"body split into ~{parts} posts",
209
+ count=parts,
210
+ )
211
+ )
212
+ else:
213
+ p.blocking.append(
214
+ f"body is {approx} chars, platform limit is {caps.max_body_chars} "
215
+ "and it has no thread support"
216
+ )
217
+
218
+ # --- tags -------------------------------------------------------------
219
+ if doc.tags:
220
+ if caps.tags is None:
221
+ p.degradations.append(
222
+ Degradation(
223
+ kind=DegradationKind.TAGS_TRIMMED,
224
+ detail=f"{len(doc.tags)} tag(s) dropped — platform has no tags",
225
+ count=len(doc.tags),
226
+ )
227
+ )
228
+ else:
229
+ over_len = [t for t in doc.tags if len(t) > caps.tags.max_len]
230
+ over_count = max(0, len(doc.tags) - caps.tags.max_count)
231
+ if over_len or over_count:
232
+ p.degradations.append(
233
+ Degradation(
234
+ kind=DegradationKind.TAGS_TRIMMED,
235
+ detail=(
236
+ f"keeping {min(len(doc.tags), caps.tags.max_count)} of {len(doc.tags)}"
237
+ + (f"; {len(over_len)} exceed {caps.tags.max_len} chars" if over_len else "")
238
+ ),
239
+ count=len(doc.tags),
240
+ )
241
+ )
242
+
243
+ if adapter is not None and hasattr(adapter, "refine_plan"):
244
+ p = adapter.refine_plan(doc, p)
245
+ return p
246
+
247
+
248
+ def select_tags(tags: list[str], spec: TagSpec | None) -> list[str]:
249
+ """Trim a tag list to what a platform will actually accept.
250
+
251
+ Silently sending an over-long tag is how one publish attempt ended up with a
252
+ single 40-character invalid tag instead of five good ones (failure B9).
253
+ """
254
+ if spec is None:
255
+ return []
256
+ return [t for t in tags if len(t) <= spec.max_len][: spec.max_count]