pubkit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pubkit/__init__.py +14 -0
- pubkit/__main__.py +7 -0
- pubkit/adapters/__init__.py +2 -0
- pubkit/adapters/api_base.py +67 -0
- pubkit/adapters/devto.py +195 -0
- pubkit/adapters/medium.py +184 -0
- pubkit/adapters/substack.py +136 -0
- pubkit/adapters/x.py +216 -0
- pubkit/browserctl.py +78 -0
- pubkit/cli.py +306 -0
- pubkit/core/__init__.py +2 -0
- pubkit/core/adapter.py +206 -0
- pubkit/core/anchors.py +89 -0
- pubkit/core/auth.py +209 -0
- pubkit/core/browser.py +426 -0
- pubkit/core/capabilities.py +256 -0
- pubkit/core/checks.py +344 -0
- pubkit/core/ir.py +240 -0
- pubkit/core/loader.py +278 -0
- pubkit/core/runner.py +268 -0
- pubkit/core/state.py +213 -0
- pubkit/core/transport.py +170 -0
- pubkit/py.typed +0 -0
- pubkit/registry.py +81 -0
- pubkit/render/__init__.py +2 -0
- pubkit/render/html.py +221 -0
- pubkit/scaffold.py +271 -0
- pubkit/workflows/__init__.py +2 -0
- pubkit/workflows/airflow.py +115 -0
- pubkit-0.1.0.dist-info/METADATA +291 -0
- pubkit-0.1.0.dist-info/RECORD +35 -0
- pubkit-0.1.0.dist-info/WHEEL +4 -0
- pubkit-0.1.0.dist-info/entry_points.txt +2 -0
- pubkit-0.1.0.dist-info/licenses/LICENSE +202 -0
- pubkit-0.1.0.dist-info/licenses/NOTICE +7 -0
pubkit/core/browser.py
ADDED
|
@@ -0,0 +1,426 @@
|
|
|
1
|
+
# Copyright 2026 The pubkit Authors
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
"""BrowserAdapter — driving a rich-text editor that was never meant to be driven.
|
|
4
|
+
|
|
5
|
+
Everything in this module was learned the hard way against Medium's editor.
|
|
6
|
+
Five things matter, and they generalise to Substack, Ghost's editor, LinkedIn
|
|
7
|
+
articles and anything else built on contenteditable:
|
|
8
|
+
|
|
9
|
+
1. **Replace by pasting over a selection, never by deleting everything.**
|
|
10
|
+
`execCommand('delete')` across the whole document destroys the editor's
|
|
11
|
+
internal scaffolding; afterwards pastes land nowhere and the only recovery
|
|
12
|
+
is a reload.
|
|
13
|
+
|
|
14
|
+
2. **Trust nothing until it survives a reload.** A delete that took 355
|
|
15
|
+
paragraphs to 189 came back as 339. A link href set directly in the DOM
|
|
16
|
+
reverted completely. The editor keeps its own model and syncs deltas.
|
|
17
|
+
|
|
18
|
+
3. **Images cannot travel inside pasted HTML.** `data:` URIs are stripped and
|
|
19
|
+
so are ordinary `https://` URLs. The only path that works is a synthetic
|
|
20
|
+
`ClipboardEvent` carrying a real `File` in its `DataTransfer`, which makes
|
|
21
|
+
the editor run its own upload.
|
|
22
|
+
|
|
23
|
+
4. **Getting bytes into the page is the actual hard problem.** An injected
|
|
24
|
+
`<input type=file>` plus the automation layer's file-upload primitive
|
|
25
|
+
solves it without ever opening a native picker.
|
|
26
|
+
|
|
27
|
+
5. **Anchors shift.** Every insert renumbers the document. Re-resolve by text
|
|
28
|
+
after every mutation, never cache an index.
|
|
29
|
+
"""
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
import asyncio
|
|
33
|
+
import logging
|
|
34
|
+
from collections.abc import Sequence
|
|
35
|
+
from dataclasses import dataclass
|
|
36
|
+
from pathlib import Path
|
|
37
|
+
from typing import Any
|
|
38
|
+
|
|
39
|
+
from .adapter import BaseAdapter, Fingerprint, VerificationFailed
|
|
40
|
+
from .anchors import AnchorNotFound, find_anchor, similarity
|
|
41
|
+
from .transport import ROLLING_HASH_JS
|
|
42
|
+
|
|
43
|
+
log = logging.getLogger(__name__)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class DocumentReplaceMismatch(RuntimeError):
|
|
47
|
+
"""Paste-replace produced the wrong paragraph count — almost always means
|
|
48
|
+
the selection covered part of the document (failure B1)."""
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@dataclass
|
|
52
|
+
class EditorSelectors:
|
|
53
|
+
"""Everything platform-specific about an editor, in one place.
|
|
54
|
+
|
|
55
|
+
A new browser platform is mostly this dataclass plus a publish click.
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
editable: str
|
|
59
|
+
#: All content roots. Getting this wrong — grabbing only the first — is what
|
|
60
|
+
#: made a paste insert instead of replace (failure B1).
|
|
61
|
+
content_roots: str
|
|
62
|
+
block: str
|
|
63
|
+
figure: str
|
|
64
|
+
figure_img: str
|
|
65
|
+
link: str
|
|
66
|
+
publish_button: str
|
|
67
|
+
tag_input: str | None = None
|
|
68
|
+
tag_chip: str | None = None
|
|
69
|
+
confirm_publish: str | None = None
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
UPLOAD_INPUT_ID = "__pubkit_upload"
|
|
73
|
+
|
|
74
|
+
_JS_HELPERS = (
|
|
75
|
+
ROLLING_HASH_JS
|
|
76
|
+
+ """
|
|
77
|
+
window.__pk = window.__pk || {};
|
|
78
|
+
|
|
79
|
+
window.__pk.roots = function (sel) {
|
|
80
|
+
return Array.from(document.querySelectorAll(sel));
|
|
81
|
+
};
|
|
82
|
+
|
|
83
|
+
window.__pk.blocks = function (rootSel, blockSel) {
|
|
84
|
+
// Across ALL roots, not just the first. This single detail is the
|
|
85
|
+
// difference between replacing a document and duplicating it.
|
|
86
|
+
const out = [];
|
|
87
|
+
for (const root of window.__pk.roots(rootSel)) {
|
|
88
|
+
out.push(...Array.from(root.querySelectorAll(blockSel)));
|
|
89
|
+
}
|
|
90
|
+
return out;
|
|
91
|
+
};
|
|
92
|
+
|
|
93
|
+
window.__pk.texts = function (rootSel, blockSel) {
|
|
94
|
+
return window.__pk.blocks(rootSel, blockSel).map(e => (e.innerText || '').trim());
|
|
95
|
+
};
|
|
96
|
+
|
|
97
|
+
window.__pk.selectAll = function (rootSel, blockSel) {
|
|
98
|
+
const g = window.__pk.blocks(rootSel, blockSel);
|
|
99
|
+
if (!g.length) return 0;
|
|
100
|
+
const r = document.createRange();
|
|
101
|
+
r.setStartBefore(g[0]);
|
|
102
|
+
r.setEndAfter(g[g.length - 1]);
|
|
103
|
+
const s = getSelection();
|
|
104
|
+
s.removeAllRanges();
|
|
105
|
+
s.addRange(r);
|
|
106
|
+
return g.length;
|
|
107
|
+
};
|
|
108
|
+
|
|
109
|
+
window.__pk.caretAtEndOf = function (rootSel, blockSel, index) {
|
|
110
|
+
const g = window.__pk.blocks(rootSel, blockSel);
|
|
111
|
+
if (index < 0 || index >= g.length) return false;
|
|
112
|
+
const r = document.createRange();
|
|
113
|
+
r.selectNodeContents(g[index]);
|
|
114
|
+
r.collapse(false);
|
|
115
|
+
const s = getSelection();
|
|
116
|
+
s.removeAllRanges();
|
|
117
|
+
s.addRange(r);
|
|
118
|
+
return true;
|
|
119
|
+
};
|
|
120
|
+
|
|
121
|
+
window.__pk.pasteHtml = function (editableSel, html) {
|
|
122
|
+
const ed = document.querySelector(editableSel);
|
|
123
|
+
ed.focus();
|
|
124
|
+
const dt = new DataTransfer();
|
|
125
|
+
dt.setData('text/html', html);
|
|
126
|
+
ed.dispatchEvent(new ClipboardEvent('paste', {
|
|
127
|
+
clipboardData: dt, bubbles: true, cancelable: true,
|
|
128
|
+
}));
|
|
129
|
+
return true;
|
|
130
|
+
};
|
|
131
|
+
|
|
132
|
+
window.__pk.pasteFile = function (editableSel, fileIndex) {
|
|
133
|
+
// The ONLY image path that survives. Remote URLs and data: URIs are
|
|
134
|
+
// stripped from pasted HTML; a real File in DataTransfer.items makes the
|
|
135
|
+
// editor run its own uploader and put the result on its CDN.
|
|
136
|
+
const input = document.getElementById('%UPLOAD_ID%');
|
|
137
|
+
if (!input || !input.files || !input.files[fileIndex]) return 'nofile';
|
|
138
|
+
const ed = document.querySelector(editableSel);
|
|
139
|
+
ed.focus();
|
|
140
|
+
const dt = new DataTransfer();
|
|
141
|
+
dt.items.add(input.files[fileIndex]);
|
|
142
|
+
ed.dispatchEvent(new ClipboardEvent('paste', {
|
|
143
|
+
clipboardData: dt, bubbles: true, cancelable: true,
|
|
144
|
+
}));
|
|
145
|
+
return 'ok';
|
|
146
|
+
};
|
|
147
|
+
|
|
148
|
+
window.__pk.ensureUploadInput = function () {
|
|
149
|
+
let i = document.getElementById('%UPLOAD_ID%');
|
|
150
|
+
if (!i) {
|
|
151
|
+
i = document.createElement('input');
|
|
152
|
+
i.type = 'file';
|
|
153
|
+
i.id = '%UPLOAD_ID%';
|
|
154
|
+
i.multiple = true;
|
|
155
|
+
// Visible enough to be a real element, invisible enough not to matter.
|
|
156
|
+
i.style.cssText =
|
|
157
|
+
'position:fixed;top:0;left:0;z-index:2147483647;opacity:0.01;width:80px;height:24px';
|
|
158
|
+
document.body.appendChild(i);
|
|
159
|
+
}
|
|
160
|
+
return true;
|
|
161
|
+
};
|
|
162
|
+
|
|
163
|
+
window.__pk.stage = function (key, index, data) {
|
|
164
|
+
sessionStorage.setItem(key + ':' + index, data);
|
|
165
|
+
return window.__pkHash(data);
|
|
166
|
+
};
|
|
167
|
+
|
|
168
|
+
window.__pk.stagedHash = function (key) {
|
|
169
|
+
const v = sessionStorage.getItem(key);
|
|
170
|
+
return v === null ? null : window.__pkHash(v);
|
|
171
|
+
};
|
|
172
|
+
|
|
173
|
+
window.__pk.assemble = function (key, count) {
|
|
174
|
+
let out = '';
|
|
175
|
+
for (let i = 0; i < count; i++) out += sessionStorage.getItem(key + ':' + i) || '';
|
|
176
|
+
sessionStorage.setItem(key, out);
|
|
177
|
+
for (let i = 0; i < count; i++) sessionStorage.removeItem(key + ':' + i);
|
|
178
|
+
return window.__pkHash(out);
|
|
179
|
+
};
|
|
180
|
+
|
|
181
|
+
window.__pk.inflate = async function (key) {
|
|
182
|
+
// Single gzip member only; concatenated members throw here even though
|
|
183
|
+
// Python's gzip.decompress accepts them.
|
|
184
|
+
const b64 = sessionStorage.getItem(key);
|
|
185
|
+
const bin = atob(b64);
|
|
186
|
+
const a = new Uint8Array(bin.length);
|
|
187
|
+
for (let i = 0; i < bin.length; i++) a[i] = bin.charCodeAt(i);
|
|
188
|
+
const ds = new DecompressionStream('gzip');
|
|
189
|
+
const w = ds.writable.getWriter();
|
|
190
|
+
w.write(a); w.close();
|
|
191
|
+
const rd = ds.readable.getReader();
|
|
192
|
+
const parts = []; let n = 0;
|
|
193
|
+
for (;;) { const { done, value } = await rd.read(); if (done) break; parts.push(value); n += value.length; }
|
|
194
|
+
const o = new Uint8Array(n); let k = 0;
|
|
195
|
+
for (const p of parts) { o.set(p, k); k += p.length; }
|
|
196
|
+
return new TextDecoder().decode(o);
|
|
197
|
+
};
|
|
198
|
+
|
|
199
|
+
window.__pk.fingerprint = function (rootSel, blockSel, figSel, linkSel, markerRe) {
|
|
200
|
+
const roots = window.__pk.roots(rootSel);
|
|
201
|
+
const text = roots.map(r => r.innerText).join('\\n');
|
|
202
|
+
const blocks = window.__pk.blocks(rootSel, blockSel);
|
|
203
|
+
return {
|
|
204
|
+
words: text.split(/\\s+/).filter(Boolean).length,
|
|
205
|
+
headings: blocks.filter(e => /^H[1-6]$/.test(e.tagName)).map(e => e.innerText.trim().slice(0, 48)),
|
|
206
|
+
images: roots.reduce((n, r) => n + r.querySelectorAll(figSel).length, 0),
|
|
207
|
+
links: roots.reduce((n, r) => n + r.querySelectorAll(linkSel).length, 0),
|
|
208
|
+
markers: (text.match(new RegExp(markerRe, 'g')) || []).length,
|
|
209
|
+
};
|
|
210
|
+
};
|
|
211
|
+
""".replace("%UPLOAD_ID%", UPLOAD_INPUT_ID)
|
|
212
|
+
)
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
class BrowserAdapter(BaseAdapter):
|
|
216
|
+
"""Base class for editor-driven platforms.
|
|
217
|
+
|
|
218
|
+
Subclasses supply `selectors`, a login URL, a draft URL builder and a
|
|
219
|
+
publish routine. Everything below is platform-agnostic.
|
|
220
|
+
"""
|
|
221
|
+
|
|
222
|
+
selectors: EditorSelectors
|
|
223
|
+
login_url: str = ""
|
|
224
|
+
marker_regex: str = r"\\[\\[\\s*IMAGE"
|
|
225
|
+
|
|
226
|
+
def __init__(self) -> None:
|
|
227
|
+
super().__init__()
|
|
228
|
+
self._page: Any = None
|
|
229
|
+
|
|
230
|
+
# ------------------------------------------------------------------ page
|
|
231
|
+
async def _eval(self, script: str, *args) -> Any:
|
|
232
|
+
return await self._page.evaluate(script, *args)
|
|
233
|
+
|
|
234
|
+
async def install_helpers(self) -> None:
|
|
235
|
+
await self._eval(_JS_HELPERS)
|
|
236
|
+
await self._eval("window.__pk.ensureUploadInput()")
|
|
237
|
+
|
|
238
|
+
# ------------------------------------------------------- document replace
|
|
239
|
+
async def replace_document(self, html: str, *, expect_blocks: int | None = None) -> int:
|
|
240
|
+
"""Replace the entire document body by pasting over a full selection.
|
|
241
|
+
|
|
242
|
+
Deliberately *not* implemented as delete-then-paste. Deleting the whole
|
|
243
|
+
block range breaks the editor badly enough that a reload is the only
|
|
244
|
+
way back (failure B2).
|
|
245
|
+
"""
|
|
246
|
+
s = self.selectors
|
|
247
|
+
before = await self._eval(
|
|
248
|
+
"([r,b]) => window.__pk.selectAll(r,b)", [s.content_roots, s.block]
|
|
249
|
+
)
|
|
250
|
+
if not before:
|
|
251
|
+
raise DocumentReplaceMismatch("no blocks found — selectors are wrong or page not ready")
|
|
252
|
+
|
|
253
|
+
await self._eval("([e,h]) => window.__pk.pasteHtml(e,h)", [s.editable, html])
|
|
254
|
+
await asyncio.sleep(2.5)
|
|
255
|
+
|
|
256
|
+
after = await self._eval(
|
|
257
|
+
"([r,b]) => window.__pk.blocks(r,b).length", [s.content_roots, s.block]
|
|
258
|
+
)
|
|
259
|
+
if expect_blocks is not None and after != expect_blocks:
|
|
260
|
+
raise DocumentReplaceMismatch(
|
|
261
|
+
f"expected {expect_blocks} blocks after replace, got {after} "
|
|
262
|
+
f"(was {before}). If after ≈ before + expected, the selection "
|
|
263
|
+
f"covered only one content root — check `content_roots`."
|
|
264
|
+
)
|
|
265
|
+
log.info("replaced document: %d blocks → %d blocks", before, after)
|
|
266
|
+
return after
|
|
267
|
+
|
|
268
|
+
# ----------------------------------------------------------- anchor lookup
|
|
269
|
+
async def block_texts(self) -> list[str]:
|
|
270
|
+
s = self.selectors
|
|
271
|
+
return await self._eval("([r,b]) => window.__pk.texts(r,b)", [s.content_roots, s.block])
|
|
272
|
+
|
|
273
|
+
async def resolve_anchor(self, target: str, *, used: set[int] | None = None) -> int:
|
|
274
|
+
texts = await self.block_texts()
|
|
275
|
+
idx = find_anchor(texts, target, used=used)
|
|
276
|
+
if idx < 0:
|
|
277
|
+
scored = sorted(((similarity(t, target), t) for t in texts), reverse=True)
|
|
278
|
+
best = scored[0] if scored else (0.0, None)
|
|
279
|
+
raise AnchorNotFound(target, best[1], best[0])
|
|
280
|
+
return idx
|
|
281
|
+
|
|
282
|
+
# ------------------------------------------------------------------ media
|
|
283
|
+
async def insert_image_at(self, caption: str, file_index: int, *, used: set[int] | None = None) -> None:
|
|
284
|
+
"""Place an image directly above the paragraph that carries `caption`.
|
|
285
|
+
|
|
286
|
+
The figure lands *before* the caret's paragraph, so anchoring on the
|
|
287
|
+
caption gives the right visual result for free (failure B7). Anchors are
|
|
288
|
+
re-resolved every time because every insert shifts the indices.
|
|
289
|
+
"""
|
|
290
|
+
s = self.selectors
|
|
291
|
+
idx = await self.resolve_anchor(caption, used=used)
|
|
292
|
+
ok = await self._eval(
|
|
293
|
+
"([r,b,i]) => window.__pk.caretAtEndOf(r,b,i)", [s.content_roots, s.block, idx]
|
|
294
|
+
)
|
|
295
|
+
if not ok:
|
|
296
|
+
raise AnchorNotFound(caption, None, 0.0)
|
|
297
|
+
|
|
298
|
+
before = await self._count_figures()
|
|
299
|
+
res = await self._eval("([e,i]) => window.__pk.pasteFile(e,i)", [s.editable, file_index])
|
|
300
|
+
if res == "nofile":
|
|
301
|
+
raise RuntimeError(
|
|
302
|
+
f"no staged file at index {file_index}; call stage_files() first"
|
|
303
|
+
)
|
|
304
|
+
await self._wait_for_figure(before + 1)
|
|
305
|
+
|
|
306
|
+
async def _count_figures(self) -> int:
|
|
307
|
+
s = self.selectors
|
|
308
|
+
return await self._eval(
|
|
309
|
+
"([r,f]) => window.__pk.roots(r).reduce((n,x)=>n+x.querySelectorAll(f).length,0)",
|
|
310
|
+
[s.content_roots, s.figure],
|
|
311
|
+
)
|
|
312
|
+
|
|
313
|
+
async def _wait_for_figure(self, want: int, timeout: float = 45.0) -> None:
|
|
314
|
+
"""Wait for the editor's own uploader to finish.
|
|
315
|
+
|
|
316
|
+
Polls for the figure to exist *and* for its src to point at the
|
|
317
|
+
platform's CDN rather than a blob: URL — a blob means the upload has
|
|
318
|
+
not completed and publishing now would produce a broken image.
|
|
319
|
+
"""
|
|
320
|
+
s = self.selectors
|
|
321
|
+
deadline = asyncio.get_event_loop().time() + timeout
|
|
322
|
+
while asyncio.get_event_loop().time() < deadline:
|
|
323
|
+
state = await self._eval(
|
|
324
|
+
"([r,f,i]) => { const roots = window.__pk.roots(r);"
|
|
325
|
+
" const figs = roots.flatMap(x => Array.from(x.querySelectorAll(f)));"
|
|
326
|
+
" return { n: figs.length, blobs: figs.filter(g => { const im = g.querySelector(i);"
|
|
327
|
+
" return !im || !im.src || im.src.startsWith('blob:') || im.src.startsWith('data:'); }).length }; }",
|
|
328
|
+
[s.content_roots, s.figure, s.figure_img],
|
|
329
|
+
)
|
|
330
|
+
if state["n"] >= want and state["blobs"] == 0:
|
|
331
|
+
return
|
|
332
|
+
await asyncio.sleep(1.0)
|
|
333
|
+
raise VerificationFailed(
|
|
334
|
+
f"image upload did not settle within {timeout:.0f}s "
|
|
335
|
+
"(figure missing, or src still a blob: URL — the platform has not stored it)"
|
|
336
|
+
)
|
|
337
|
+
|
|
338
|
+
async def stage_files(self, paths: Sequence[Path], upload) -> None:
|
|
339
|
+
"""Hand real bytes to the page.
|
|
340
|
+
|
|
341
|
+
`upload` is the automation layer's file-upload primitive, which sets
|
|
342
|
+
`input.files` on an element. We inject the input ourselves so no native
|
|
343
|
+
file picker is ever opened — a picker is undriveable and blocks the
|
|
344
|
+
whole session.
|
|
345
|
+
"""
|
|
346
|
+
await self.install_helpers()
|
|
347
|
+
await upload(f"#{UPLOAD_INPUT_ID}", [str(p) for p in paths])
|
|
348
|
+
n = await self._eval(
|
|
349
|
+
f"() => {{ const i = document.getElementById('{UPLOAD_INPUT_ID}');"
|
|
350
|
+
" return i && i.files ? i.files.length : 0; }"
|
|
351
|
+
)
|
|
352
|
+
if n != len(paths):
|
|
353
|
+
raise RuntimeError(f"staged {n} files but expected {len(paths)}")
|
|
354
|
+
|
|
355
|
+
# ------------------------------------------------------------ verification
|
|
356
|
+
async def fingerprint(self) -> Fingerprint:
|
|
357
|
+
s = self.selectors
|
|
358
|
+
raw = await self._eval(
|
|
359
|
+
"([r,b,f,l,m]) => window.__pk.fingerprint(r,b,f,l,m)",
|
|
360
|
+
[s.content_roots, s.block, s.figure, s.link, self.marker_regex],
|
|
361
|
+
)
|
|
362
|
+
return Fingerprint(**raw)
|
|
363
|
+
|
|
364
|
+
async def verify_after_reload(self, expected: Fingerprint, reload_fn) -> Fingerprint:
|
|
365
|
+
"""Reload, then compare. The only honest verification (failure B3).
|
|
366
|
+
|
|
367
|
+
An editor will happily show you a change it has not persisted. Asking
|
|
368
|
+
the server what it actually has is the difference between "published
|
|
369
|
+
correctly" and "published, and broken for every reader".
|
|
370
|
+
"""
|
|
371
|
+
await reload_fn()
|
|
372
|
+
await asyncio.sleep(3.0)
|
|
373
|
+
await self.install_helpers()
|
|
374
|
+
actual = await self.fingerprint()
|
|
375
|
+
actual.assert_matches(expected)
|
|
376
|
+
return actual
|
|
377
|
+
|
|
378
|
+
# -------------------------------------------------------------------- tags
|
|
379
|
+
async def apply_tags(self, tags: list[str], strategy: str) -> int:
|
|
380
|
+
"""Commit tags, then *check* that chips actually appeared.
|
|
381
|
+
|
|
382
|
+
Typing `a,b,c` into a framework-controlled tag field once produced a
|
|
383
|
+
single 40-character invalid tag. Publishing without tags is a small
|
|
384
|
+
loss; publishing with one garbage tag is worse (failure B9).
|
|
385
|
+
"""
|
|
386
|
+
s = self.selectors
|
|
387
|
+
if not s.tag_input or not tags:
|
|
388
|
+
return 0
|
|
389
|
+
await self._page.click(s.tag_input)
|
|
390
|
+
if strategy == "comma":
|
|
391
|
+
for t in tags:
|
|
392
|
+
await self._page.type(s.tag_input, t + ",", delay=40)
|
|
393
|
+
await asyncio.sleep(0.4)
|
|
394
|
+
elif strategy == "enter":
|
|
395
|
+
for t in tags:
|
|
396
|
+
await self._page.type(s.tag_input, t, delay=40)
|
|
397
|
+
await self._page.keyboard.press("Enter")
|
|
398
|
+
await asyncio.sleep(0.4)
|
|
399
|
+
else: # native_setter — drive the framework's own value setter
|
|
400
|
+
for t in tags:
|
|
401
|
+
await self._eval(
|
|
402
|
+
"([sel,val]) => { const el = document.querySelector(sel);"
|
|
403
|
+
" const proto = Object.getPrototypeOf(el);"
|
|
404
|
+
" const d = Object.getOwnPropertyDescriptor(proto, 'value');"
|
|
405
|
+
" d.set.call(el, val);"
|
|
406
|
+
" el.dispatchEvent(new Event('input', { bubbles: true }));"
|
|
407
|
+
" el.dispatchEvent(new KeyboardEvent('keydown', { key: 'Enter', bubbles: true })); }",
|
|
408
|
+
[s.tag_input, t],
|
|
409
|
+
)
|
|
410
|
+
await asyncio.sleep(0.4)
|
|
411
|
+
|
|
412
|
+
if not s.tag_chip:
|
|
413
|
+
return len(tags)
|
|
414
|
+
chips = await self._eval(f"() => document.querySelectorAll({s.tag_chip!r}).length")
|
|
415
|
+
if chips == 0:
|
|
416
|
+
log.warning(
|
|
417
|
+
"tag strategy %r committed no chips; publishing without tags "
|
|
418
|
+
"rather than risking one malformed tag",
|
|
419
|
+
strategy,
|
|
420
|
+
)
|
|
421
|
+
await self._eval(
|
|
422
|
+
"(sel) => { const el = document.querySelector(sel); if (el) { el.value=''; "
|
|
423
|
+
"el.dispatchEvent(new Event('input',{bubbles:true})); } }",
|
|
424
|
+
s.tag_input,
|
|
425
|
+
)
|
|
426
|
+
return chips
|
|
@@ -0,0 +1,256 @@
|
|
|
1
|
+
# Copyright 2026 The pubkit Authors
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
"""Capability declarations and the planner that negotiates against them.
|
|
4
|
+
|
|
5
|
+
The planner's job is to turn "here is my content" plus "here is what this
|
|
6
|
+
platform can do" into an explicit, printable list of *degradations*. Surprises
|
|
7
|
+
at publish time are the enemy; `pubkit plan` exists so there are none.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from enum import Enum
|
|
12
|
+
from typing import Literal
|
|
13
|
+
|
|
14
|
+
from pydantic import BaseModel, Field
|
|
15
|
+
|
|
16
|
+
from .ir import Document, Table
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class TagSpec(BaseModel):
|
|
20
|
+
model_config = {"extra": "forbid"}
|
|
21
|
+
|
|
22
|
+
max_count: int = 5
|
|
23
|
+
max_len: int = 25
|
|
24
|
+
#: How the platform's tag widget actually commits a tag (failure B9).
|
|
25
|
+
strategy: Literal["comma", "enter", "native_setter", "api"] = "api"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class Capabilities(BaseModel):
|
|
29
|
+
model_config = {"extra": "forbid"}
|
|
30
|
+
|
|
31
|
+
tables: bool = True
|
|
32
|
+
code_blocks: Literal["fenced", "highlighted", "plain", "none"] = "fenced"
|
|
33
|
+
inline_html: bool = False
|
|
34
|
+
animated_gif: bool = True
|
|
35
|
+
headings: int = 6
|
|
36
|
+
max_body_chars: int | None = None
|
|
37
|
+
image_upload: Literal["api", "browser_paste", "none"] = "api"
|
|
38
|
+
canonical_url: bool = False
|
|
39
|
+
tags: TagSpec | None = None
|
|
40
|
+
scheduling: bool = False
|
|
41
|
+
threads: bool = False
|
|
42
|
+
#: Whether drafts can be created and revisited, or publishing is one-shot.
|
|
43
|
+
drafts: bool = True
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class DegradationKind(str, Enum):
|
|
47
|
+
TABLE_TO_IMAGE = "table_to_image"
|
|
48
|
+
TABLE_TO_LIST = "table_to_list"
|
|
49
|
+
HTML_STRIPPED = "html_stripped"
|
|
50
|
+
CODE_FLATTENED = "code_flattened"
|
|
51
|
+
HEADINGS_CLAMPED = "headings_clamped"
|
|
52
|
+
SPLIT_INTO_THREAD = "split_into_thread"
|
|
53
|
+
GIF_TO_STILL = "gif_to_still"
|
|
54
|
+
TAGS_TRIMMED = "tags_trimmed"
|
|
55
|
+
IMAGES_VIA_BROWSER = "images_via_browser"
|
|
56
|
+
BODY_TRUNCATED = "body_truncated"
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class Degradation(BaseModel):
|
|
60
|
+
model_config = {"extra": "forbid"}
|
|
61
|
+
|
|
62
|
+
kind: DegradationKind
|
|
63
|
+
detail: str
|
|
64
|
+
count: int = 1
|
|
65
|
+
#: True when the degradation loses information the author may care about.
|
|
66
|
+
lossy: bool = False
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
class PublishPlan(BaseModel):
|
|
70
|
+
model_config = {"extra": "forbid"}
|
|
71
|
+
|
|
72
|
+
document_id: str
|
|
73
|
+
content_id: str
|
|
74
|
+
platform: str
|
|
75
|
+
degradations: list[Degradation] = Field(default_factory=list)
|
|
76
|
+
#: Assets the renderer must generate before publishing (e.g. table PNGs).
|
|
77
|
+
generated_assets: list[str] = Field(default_factory=list)
|
|
78
|
+
estimated_body_chars: int = 0
|
|
79
|
+
blocking: list[str] = Field(default_factory=list)
|
|
80
|
+
|
|
81
|
+
@property
|
|
82
|
+
def ok(self) -> bool:
|
|
83
|
+
return not self.blocking
|
|
84
|
+
|
|
85
|
+
def human(self) -> str:
|
|
86
|
+
lines = [f"{self.platform}: {self.document_id} ({self.content_id[:12]})"]
|
|
87
|
+
if not self.degradations:
|
|
88
|
+
lines.append(" no degradation — native fidelity")
|
|
89
|
+
for d in self.degradations:
|
|
90
|
+
mark = "!" if d.lossy else "·"
|
|
91
|
+
lines.append(f" {mark} {d.kind.value:<20} {d.detail}")
|
|
92
|
+
for b in self.blocking:
|
|
93
|
+
lines.append(f" ✗ BLOCKING {b}")
|
|
94
|
+
return "\n".join(lines)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def plain_text_length(doc: Document) -> int:
|
|
98
|
+
"""Characters a reader actually sees."""
|
|
99
|
+
total = len(doc.title) + len(doc.subtitle or "")
|
|
100
|
+
for b in doc.blocks:
|
|
101
|
+
text = getattr(b, "text", None)
|
|
102
|
+
if text:
|
|
103
|
+
total += len(text)
|
|
104
|
+
items = getattr(b, "items", None)
|
|
105
|
+
if items:
|
|
106
|
+
total += sum(len(i) for i in items)
|
|
107
|
+
if isinstance(b, Table):
|
|
108
|
+
total += sum(len(c) for c in b.header)
|
|
109
|
+
total += sum(len(c) for row in b.rows for c in row)
|
|
110
|
+
return total
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def plan(doc: Document, platform: str, caps: Capabilities, adapter: object | None = None) -> PublishPlan:
|
|
114
|
+
"""Intersect a document with a platform's capabilities.
|
|
115
|
+
|
|
116
|
+
Static capabilities cannot express everything: X in promo mode never
|
|
117
|
+
overflows 280 chars because it does not serialise the body at all. An
|
|
118
|
+
adapter may therefore refine its own plan — the last word belongs to the
|
|
119
|
+
code that will actually do the work.
|
|
120
|
+
"""
|
|
121
|
+
p = PublishPlan(document_id=doc.id, content_id=doc.content_id, platform=platform)
|
|
122
|
+
|
|
123
|
+
# --- tables -----------------------------------------------------------
|
|
124
|
+
tables = doc.tables
|
|
125
|
+
if tables and not caps.tables:
|
|
126
|
+
if caps.image_upload != "none":
|
|
127
|
+
p.degradations.append(
|
|
128
|
+
Degradation(
|
|
129
|
+
kind=DegradationKind.TABLE_TO_IMAGE,
|
|
130
|
+
detail=f"{len(tables)} table(s) rendered as images — platform has no table support",
|
|
131
|
+
count=len(tables),
|
|
132
|
+
)
|
|
133
|
+
)
|
|
134
|
+
p.generated_assets.extend(f"table-{i:02d}" for i in range(len(tables)))
|
|
135
|
+
else:
|
|
136
|
+
p.degradations.append(
|
|
137
|
+
Degradation(
|
|
138
|
+
kind=DegradationKind.TABLE_TO_LIST,
|
|
139
|
+
detail=f"{len(tables)} table(s) flattened to lists",
|
|
140
|
+
count=len(tables),
|
|
141
|
+
lossy=True,
|
|
142
|
+
)
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
# --- figures ----------------------------------------------------------
|
|
146
|
+
figures = doc.figures
|
|
147
|
+
if figures:
|
|
148
|
+
if caps.image_upload == "none":
|
|
149
|
+
p.blocking.append(f"{len(figures)} figure(s) but platform accepts no images")
|
|
150
|
+
elif caps.image_upload == "browser_paste":
|
|
151
|
+
p.degradations.append(
|
|
152
|
+
Degradation(
|
|
153
|
+
kind=DegradationKind.IMAGES_VIA_BROWSER,
|
|
154
|
+
detail=(
|
|
155
|
+
f"{len(figures)} image(s) uploaded through the editor's own paste path "
|
|
156
|
+
"— remote URLs and data: URIs are stripped by this platform"
|
|
157
|
+
),
|
|
158
|
+
count=len(figures),
|
|
159
|
+
)
|
|
160
|
+
)
|
|
161
|
+
if not caps.animated_gif:
|
|
162
|
+
gifs = [f for f in figures if doc.assets[f.asset_id].animated]
|
|
163
|
+
if gifs:
|
|
164
|
+
p.degradations.append(
|
|
165
|
+
Degradation(
|
|
166
|
+
kind=DegradationKind.GIF_TO_STILL,
|
|
167
|
+
detail=f"{len(gifs)} animation(s) reduced to a still frame",
|
|
168
|
+
count=len(gifs),
|
|
169
|
+
lossy=True,
|
|
170
|
+
)
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
# --- code -------------------------------------------------------------
|
|
174
|
+
if caps.code_blocks in ("plain", "none"):
|
|
175
|
+
n = sum(1 for b in doc.blocks if getattr(b, "type", None) == "code")
|
|
176
|
+
if n:
|
|
177
|
+
p.degradations.append(
|
|
178
|
+
Degradation(
|
|
179
|
+
kind=DegradationKind.CODE_FLATTENED,
|
|
180
|
+
detail=f"{n} code block(s) lose syntax highlighting",
|
|
181
|
+
count=n,
|
|
182
|
+
)
|
|
183
|
+
)
|
|
184
|
+
|
|
185
|
+
# --- headings ---------------------------------------------------------
|
|
186
|
+
deep = [b for b in doc.blocks if getattr(b, "type", None) == "heading" and b.level > caps.headings]
|
|
187
|
+
if deep:
|
|
188
|
+
p.degradations.append(
|
|
189
|
+
Degradation(
|
|
190
|
+
kind=DegradationKind.HEADINGS_CLAMPED,
|
|
191
|
+
detail=f"{len(deep)} heading(s) clamped to h{caps.headings}",
|
|
192
|
+
count=len(deep),
|
|
193
|
+
)
|
|
194
|
+
)
|
|
195
|
+
|
|
196
|
+
# --- length -----------------------------------------------------------
|
|
197
|
+
# Plain-text length, not the canonical JSON: JSON includes keys, quoting
|
|
198
|
+
# and escapes, which inflated a 3,300-word article into an estimated
|
|
199
|
+
# 94-tweet thread. An estimate that is wrong by 4x is worse than none.
|
|
200
|
+
approx = plain_text_length(doc)
|
|
201
|
+
p.estimated_body_chars = approx
|
|
202
|
+
if caps.max_body_chars and approx > caps.max_body_chars:
|
|
203
|
+
if caps.threads:
|
|
204
|
+
parts = -(-approx // caps.max_body_chars)
|
|
205
|
+
p.degradations.append(
|
|
206
|
+
Degradation(
|
|
207
|
+
kind=DegradationKind.SPLIT_INTO_THREAD,
|
|
208
|
+
detail=f"body split into ~{parts} posts",
|
|
209
|
+
count=parts,
|
|
210
|
+
)
|
|
211
|
+
)
|
|
212
|
+
else:
|
|
213
|
+
p.blocking.append(
|
|
214
|
+
f"body is {approx} chars, platform limit is {caps.max_body_chars} "
|
|
215
|
+
"and it has no thread support"
|
|
216
|
+
)
|
|
217
|
+
|
|
218
|
+
# --- tags -------------------------------------------------------------
|
|
219
|
+
if doc.tags:
|
|
220
|
+
if caps.tags is None:
|
|
221
|
+
p.degradations.append(
|
|
222
|
+
Degradation(
|
|
223
|
+
kind=DegradationKind.TAGS_TRIMMED,
|
|
224
|
+
detail=f"{len(doc.tags)} tag(s) dropped — platform has no tags",
|
|
225
|
+
count=len(doc.tags),
|
|
226
|
+
)
|
|
227
|
+
)
|
|
228
|
+
else:
|
|
229
|
+
over_len = [t for t in doc.tags if len(t) > caps.tags.max_len]
|
|
230
|
+
over_count = max(0, len(doc.tags) - caps.tags.max_count)
|
|
231
|
+
if over_len or over_count:
|
|
232
|
+
p.degradations.append(
|
|
233
|
+
Degradation(
|
|
234
|
+
kind=DegradationKind.TAGS_TRIMMED,
|
|
235
|
+
detail=(
|
|
236
|
+
f"keeping {min(len(doc.tags), caps.tags.max_count)} of {len(doc.tags)}"
|
|
237
|
+
+ (f"; {len(over_len)} exceed {caps.tags.max_len} chars" if over_len else "")
|
|
238
|
+
),
|
|
239
|
+
count=len(doc.tags),
|
|
240
|
+
)
|
|
241
|
+
)
|
|
242
|
+
|
|
243
|
+
if adapter is not None and hasattr(adapter, "refine_plan"):
|
|
244
|
+
p = adapter.refine_plan(doc, p)
|
|
245
|
+
return p
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def select_tags(tags: list[str], spec: TagSpec | None) -> list[str]:
|
|
249
|
+
"""Trim a tag list to what a platform will actually accept.
|
|
250
|
+
|
|
251
|
+
Silently sending an over-long tag is how one publish attempt ended up with a
|
|
252
|
+
single 40-character invalid tag instead of five good ones (failure B9).
|
|
253
|
+
"""
|
|
254
|
+
if spec is None:
|
|
255
|
+
return []
|
|
256
|
+
return [t for t in tags if len(t) <= spec.max_len][: spec.max_count]
|