pubkit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pubkit/__init__.py +14 -0
- pubkit/__main__.py +7 -0
- pubkit/adapters/__init__.py +2 -0
- pubkit/adapters/api_base.py +67 -0
- pubkit/adapters/devto.py +195 -0
- pubkit/adapters/medium.py +184 -0
- pubkit/adapters/substack.py +136 -0
- pubkit/adapters/x.py +216 -0
- pubkit/browserctl.py +78 -0
- pubkit/cli.py +306 -0
- pubkit/core/__init__.py +2 -0
- pubkit/core/adapter.py +206 -0
- pubkit/core/anchors.py +89 -0
- pubkit/core/auth.py +209 -0
- pubkit/core/browser.py +426 -0
- pubkit/core/capabilities.py +256 -0
- pubkit/core/checks.py +344 -0
- pubkit/core/ir.py +240 -0
- pubkit/core/loader.py +278 -0
- pubkit/core/runner.py +268 -0
- pubkit/core/state.py +213 -0
- pubkit/core/transport.py +170 -0
- pubkit/py.typed +0 -0
- pubkit/registry.py +81 -0
- pubkit/render/__init__.py +2 -0
- pubkit/render/html.py +221 -0
- pubkit/scaffold.py +271 -0
- pubkit/workflows/__init__.py +2 -0
- pubkit/workflows/airflow.py +115 -0
- pubkit-0.1.0.dist-info/METADATA +291 -0
- pubkit-0.1.0.dist-info/RECORD +35 -0
- pubkit-0.1.0.dist-info/WHEEL +4 -0
- pubkit-0.1.0.dist-info/entry_points.txt +2 -0
- pubkit-0.1.0.dist-info/licenses/LICENSE +202 -0
- pubkit-0.1.0.dist-info/licenses/NOTICE +7 -0
pubkit/core/state.py
ADDED
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
# Copyright 2026 The pubkit Authors
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
"""Idempotency and resumption.
|
|
4
|
+
|
|
5
|
+
A publishing run is a distributed transaction across platforms you do not
|
|
6
|
+
control, over links that drop. During the run that motivated this tool the
|
|
7
|
+
automation bridge disconnected eight times and the browser extension died
|
|
8
|
+
mid-publish. The only sane response is to make every step resumable and every
|
|
9
|
+
step safe to re-enter.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
import sqlite3
|
|
15
|
+
import time
|
|
16
|
+
from contextlib import contextmanager
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
from enum import Enum
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
SCHEMA = """
|
|
22
|
+
CREATE TABLE IF NOT EXISTS runs (
|
|
23
|
+
run_id TEXT PRIMARY KEY,
|
|
24
|
+
plan_hash TEXT NOT NULL,
|
|
25
|
+
created_at REAL NOT NULL,
|
|
26
|
+
status TEXT NOT NULL
|
|
27
|
+
);
|
|
28
|
+
CREATE TABLE IF NOT EXISTS steps (
|
|
29
|
+
run_id TEXT NOT NULL,
|
|
30
|
+
document_id TEXT NOT NULL,
|
|
31
|
+
platform TEXT NOT NULL,
|
|
32
|
+
step TEXT NOT NULL,
|
|
33
|
+
status TEXT NOT NULL,
|
|
34
|
+
content_id TEXT,
|
|
35
|
+
fingerprint TEXT,
|
|
36
|
+
remote_ref TEXT,
|
|
37
|
+
error TEXT,
|
|
38
|
+
updated_at REAL NOT NULL,
|
|
39
|
+
PRIMARY KEY (run_id, document_id, platform, step)
|
|
40
|
+
);
|
|
41
|
+
CREATE TABLE IF NOT EXISTS remotes (
|
|
42
|
+
document_id TEXT NOT NULL,
|
|
43
|
+
platform TEXT NOT NULL,
|
|
44
|
+
content_id TEXT NOT NULL,
|
|
45
|
+
remote_ref TEXT NOT NULL,
|
|
46
|
+
url TEXT,
|
|
47
|
+
published INTEGER NOT NULL DEFAULT 0,
|
|
48
|
+
updated_at REAL NOT NULL,
|
|
49
|
+
PRIMARY KEY (document_id, platform)
|
|
50
|
+
);
|
|
51
|
+
CREATE TABLE IF NOT EXISTS channel_tuning (
|
|
52
|
+
channel TEXT PRIMARY KEY,
|
|
53
|
+
chunk_size INTEGER NOT NULL,
|
|
54
|
+
updated_at REAL NOT NULL
|
|
55
|
+
);
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class Step(str, Enum):
|
|
60
|
+
AUTH = "auth"
|
|
61
|
+
DRAFT = "draft"
|
|
62
|
+
CONTENT = "content"
|
|
63
|
+
MEDIA = "media"
|
|
64
|
+
VERIFY = "verify"
|
|
65
|
+
PUBLISH = "publish"
|
|
66
|
+
|
|
67
|
+
@classmethod
|
|
68
|
+
def order(cls) -> list[Step]:
|
|
69
|
+
return [cls.AUTH, cls.DRAFT, cls.CONTENT, cls.MEDIA, cls.VERIFY, cls.PUBLISH]
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
class Status(str, Enum):
|
|
73
|
+
PENDING = "pending"
|
|
74
|
+
RUNNING = "running"
|
|
75
|
+
DONE = "done"
|
|
76
|
+
FAILED = "failed"
|
|
77
|
+
SKIPPED = "skipped"
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
@dataclass
|
|
81
|
+
class RemoteRecord:
|
|
82
|
+
document_id: str
|
|
83
|
+
platform: str
|
|
84
|
+
content_id: str
|
|
85
|
+
remote_ref: str
|
|
86
|
+
url: str | None
|
|
87
|
+
published: bool
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
class StateStore:
|
|
91
|
+
def __init__(self, path: Path | str = ".pubkit/state.sqlite") -> None:
|
|
92
|
+
self.path = Path(path)
|
|
93
|
+
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
94
|
+
self._conn = sqlite3.connect(self.path)
|
|
95
|
+
self._conn.row_factory = sqlite3.Row
|
|
96
|
+
self._conn.executescript(SCHEMA)
|
|
97
|
+
self._conn.commit()
|
|
98
|
+
|
|
99
|
+
# ------------------------------------------------------------------ runs
|
|
100
|
+
def start_run(self, run_id: str, plan_hash: str) -> None:
|
|
101
|
+
self._conn.execute(
|
|
102
|
+
"INSERT OR REPLACE INTO runs VALUES (?,?,?,?)",
|
|
103
|
+
(run_id, plan_hash, time.time(), Status.RUNNING.value),
|
|
104
|
+
)
|
|
105
|
+
self._conn.commit()
|
|
106
|
+
|
|
107
|
+
def finish_run(self, run_id: str, status: Status) -> None:
|
|
108
|
+
self._conn.execute("UPDATE runs SET status=? WHERE run_id=?", (status.value, run_id))
|
|
109
|
+
self._conn.commit()
|
|
110
|
+
|
|
111
|
+
# ----------------------------------------------------------------- steps
|
|
112
|
+
def step_status(self, run_id: str, doc: str, platform: str, step: Step) -> Status:
|
|
113
|
+
row = self._conn.execute(
|
|
114
|
+
"SELECT status FROM steps WHERE run_id=? AND document_id=? AND platform=? AND step=?",
|
|
115
|
+
(run_id, doc, platform, step.value),
|
|
116
|
+
).fetchone()
|
|
117
|
+
return Status(row["status"]) if row else Status.PENDING
|
|
118
|
+
|
|
119
|
+
def record_step(
|
|
120
|
+
self,
|
|
121
|
+
run_id: str,
|
|
122
|
+
doc: str,
|
|
123
|
+
platform: str,
|
|
124
|
+
step: Step,
|
|
125
|
+
status: Status,
|
|
126
|
+
*,
|
|
127
|
+
content_id: str | None = None,
|
|
128
|
+
fingerprint: dict | None = None,
|
|
129
|
+
remote_ref: str | None = None,
|
|
130
|
+
error: str | None = None,
|
|
131
|
+
) -> None:
|
|
132
|
+
self._conn.execute(
|
|
133
|
+
"INSERT OR REPLACE INTO steps VALUES (?,?,?,?,?,?,?,?,?,?)",
|
|
134
|
+
(
|
|
135
|
+
run_id,
|
|
136
|
+
doc,
|
|
137
|
+
platform,
|
|
138
|
+
step.value,
|
|
139
|
+
status.value,
|
|
140
|
+
content_id,
|
|
141
|
+
json.dumps(fingerprint) if fingerprint else None,
|
|
142
|
+
remote_ref,
|
|
143
|
+
error,
|
|
144
|
+
time.time(),
|
|
145
|
+
),
|
|
146
|
+
)
|
|
147
|
+
self._conn.commit()
|
|
148
|
+
|
|
149
|
+
def resume_point(self, run_id: str, doc: str, platform: str) -> Step:
|
|
150
|
+
"""First step that is not `done`. Re-running picks up exactly here."""
|
|
151
|
+
for step in Step.order():
|
|
152
|
+
if self.step_status(run_id, doc, platform, step) is not Status.DONE:
|
|
153
|
+
return step
|
|
154
|
+
return Step.PUBLISH
|
|
155
|
+
|
|
156
|
+
# --------------------------------------------------------------- remotes
|
|
157
|
+
def remember_remote(
|
|
158
|
+
self, doc: str, platform: str, content_id: str, ref: str, url: str | None = None, published: bool = False
|
|
159
|
+
) -> None:
|
|
160
|
+
self._conn.execute(
|
|
161
|
+
"INSERT OR REPLACE INTO remotes VALUES (?,?,?,?,?,?,?)",
|
|
162
|
+
(doc, platform, content_id, ref, url, int(published), time.time()),
|
|
163
|
+
)
|
|
164
|
+
self._conn.commit()
|
|
165
|
+
|
|
166
|
+
def remote(self, doc: str, platform: str) -> RemoteRecord | None:
|
|
167
|
+
row = self._conn.execute(
|
|
168
|
+
"SELECT * FROM remotes WHERE document_id=? AND platform=?", (doc, platform)
|
|
169
|
+
).fetchone()
|
|
170
|
+
if not row:
|
|
171
|
+
return None
|
|
172
|
+
return RemoteRecord(
|
|
173
|
+
document_id=row["document_id"],
|
|
174
|
+
platform=row["platform"],
|
|
175
|
+
content_id=row["content_id"],
|
|
176
|
+
remote_ref=row["remote_ref"],
|
|
177
|
+
url=row["url"],
|
|
178
|
+
published=bool(row["published"]),
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
def needs_update(self, doc: str, platform: str, content_id: str) -> bool:
|
|
182
|
+
"""False when the remote already holds exactly this content.
|
|
183
|
+
|
|
184
|
+
This is what stops a re-run creating a second draft — the mistake that
|
|
185
|
+
turns a retry into a mess someone has to clean up by hand (failure D2).
|
|
186
|
+
"""
|
|
187
|
+
rec = self.remote(doc, platform)
|
|
188
|
+
return rec is None or rec.content_id != content_id
|
|
189
|
+
|
|
190
|
+
# -------------------------------------------------------- channel tuning
|
|
191
|
+
def tuned_chunk_size(self, channel: str) -> int | None:
|
|
192
|
+
row = self._conn.execute(
|
|
193
|
+
"SELECT chunk_size FROM channel_tuning WHERE channel=?", (channel,)
|
|
194
|
+
).fetchone()
|
|
195
|
+
return row["chunk_size"] if row else None
|
|
196
|
+
|
|
197
|
+
def save_chunk_size(self, channel: str, size: int) -> None:
|
|
198
|
+
self._conn.execute(
|
|
199
|
+
"INSERT OR REPLACE INTO channel_tuning VALUES (?,?,?)", (channel, size, time.time())
|
|
200
|
+
)
|
|
201
|
+
self._conn.commit()
|
|
202
|
+
|
|
203
|
+
@contextmanager
|
|
204
|
+
def transaction(self):
|
|
205
|
+
try:
|
|
206
|
+
yield self._conn
|
|
207
|
+
self._conn.commit()
|
|
208
|
+
except Exception:
|
|
209
|
+
self._conn.rollback()
|
|
210
|
+
raise
|
|
211
|
+
|
|
212
|
+
def close(self) -> None:
|
|
213
|
+
self._conn.close()
|
pubkit/core/transport.py
ADDED
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
# Copyright 2026 The pubkit Authors
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
"""Verified chunked transport.
|
|
4
|
+
|
|
5
|
+
Written because a 12,368-character payload once arrived with exactly one byte
|
|
6
|
+
mutated (`h` → `g`) and nothing anywhere reported an error. gzip failed three
|
|
7
|
+
steps later with `invalid distance too far back`, which is a spectacularly
|
|
8
|
+
unhelpful way to learn your channel is lossy.
|
|
9
|
+
|
|
10
|
+
The rules this module enforces:
|
|
11
|
+
|
|
12
|
+
A1 every chunk is hash-verified end to end
|
|
13
|
+
A2 the per-call size ceiling is *probed*, never assumed
|
|
14
|
+
A3 staging survives navigation, and re-pushing is a no-op
|
|
15
|
+
A4 exactly one gzip member per payload
|
|
16
|
+
"""
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import base64
|
|
20
|
+
import gzip
|
|
21
|
+
import io
|
|
22
|
+
import zlib
|
|
23
|
+
from collections.abc import Awaitable, Callable
|
|
24
|
+
from dataclasses import dataclass, field
|
|
25
|
+
from typing import Protocol
|
|
26
|
+
|
|
27
|
+
# The rolling hash below is deliberately trivial and identical on both sides:
|
|
28
|
+
# a 32-bit FNV-ish accumulation that a page can compute in three lines of JS
|
|
29
|
+
# without pulling in a crypto library. It is an integrity check against a lossy
|
|
30
|
+
# channel, not a security primitive.
|
|
31
|
+
_MASK = 0xFFFFFFFF
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def rolling_hash(s: str) -> str:
|
|
35
|
+
h = 0
|
|
36
|
+
for ch in s:
|
|
37
|
+
h = (h * 31 + ord(ch)) & _MASK
|
|
38
|
+
return format(h, "x")
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
#: The JavaScript twin of :func:`rolling_hash`. Kept next to it on purpose —
|
|
42
|
+
#: if one changes and the other does not, verification breaks silently, which
|
|
43
|
+
#: is the exact failure this module exists to prevent.
|
|
44
|
+
ROLLING_HASH_JS = """
|
|
45
|
+
window.__pkHash = function (s) {
|
|
46
|
+
let h = 0;
|
|
47
|
+
for (let i = 0; i < s.length; i++) { h = (h * 31 + s.charCodeAt(i)) >>> 0; }
|
|
48
|
+
return h.toString(16);
|
|
49
|
+
};
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def encode_payload(text: str) -> str:
|
|
54
|
+
"""gzip + base64, guaranteed single-member (failure A4).
|
|
55
|
+
|
|
56
|
+
`gzip.compress` emits one member; concatenating two independently
|
|
57
|
+
compressed halves does not, and `DecompressionStream('gzip')` rejects that
|
|
58
|
+
with `Junk found after end of compressed data` while Python's
|
|
59
|
+
`gzip.decompress` happily accepts it. Divergent behaviour between the two
|
|
60
|
+
ends of a pipe is how a bug hides for an hour.
|
|
61
|
+
"""
|
|
62
|
+
buf = io.BytesIO()
|
|
63
|
+
with gzip.GzipFile(fileobj=buf, mode="wb", mtime=0) as fh:
|
|
64
|
+
fh.write(text.encode("utf-8"))
|
|
65
|
+
raw = buf.getvalue()
|
|
66
|
+
assert raw.count(b"\x1f\x8b") >= 1, "not gzip"
|
|
67
|
+
return base64.b64encode(raw).decode("ascii")
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def decode_payload(b64: str) -> str:
|
|
71
|
+
return gzip.decompress(base64.b64decode(b64)).decode("utf-8")
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
class Channel(Protocol):
|
|
75
|
+
"""Anything that can move a string to the far side and run code there."""
|
|
76
|
+
|
|
77
|
+
async def push_chunk(self, key: str, index: int, data: str) -> None: ...
|
|
78
|
+
async def remote_hash(self, key: str) -> str | None: ...
|
|
79
|
+
async def remote_length(self, key: str) -> int: ...
|
|
80
|
+
async def assemble(self, key: str, count: int) -> None: ...
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@dataclass
|
|
84
|
+
class ChunkedTransport:
|
|
85
|
+
"""Move a large payload across a channel with an unknown, undocumented
|
|
86
|
+
per-call size ceiling, and prove it arrived intact."""
|
|
87
|
+
|
|
88
|
+
channel: Channel
|
|
89
|
+
#: Starting guess. The real ceiling is discovered, not trusted.
|
|
90
|
+
chunk_size: int = 5_600
|
|
91
|
+
max_chunk: int = 45_000
|
|
92
|
+
min_chunk: int = 1_024
|
|
93
|
+
probed: bool = False
|
|
94
|
+
_stats: dict = field(default_factory=dict)
|
|
95
|
+
|
|
96
|
+
async def probe_ceiling(self, probe: Callable[[int], Awaitable[bool]]) -> int:
|
|
97
|
+
"""Binary-search the largest chunk the channel accepts (failure A2).
|
|
98
|
+
|
|
99
|
+
Called once per channel; the result belongs in the state store so the
|
|
100
|
+
next run starts from the known-good value.
|
|
101
|
+
"""
|
|
102
|
+
lo, hi = self.min_chunk, self.max_chunk
|
|
103
|
+
best = self.min_chunk
|
|
104
|
+
while lo <= hi:
|
|
105
|
+
mid = (lo + hi) // 2
|
|
106
|
+
if await probe(mid):
|
|
107
|
+
best, lo = mid, mid + 1
|
|
108
|
+
else:
|
|
109
|
+
hi = mid - 1
|
|
110
|
+
# Back off 20%: the ceiling is not always stable under load, and a
|
|
111
|
+
# chunk that fails mid-run costs a retry plus a reassembly.
|
|
112
|
+
self.chunk_size = max(self.min_chunk, int(best * 0.8))
|
|
113
|
+
self.probed = True
|
|
114
|
+
return self.chunk_size
|
|
115
|
+
|
|
116
|
+
async def send(self, key: str, text: str, *, compress: bool = True) -> str:
|
|
117
|
+
"""Push `text` under `key`, verify it, and return the payload hash.
|
|
118
|
+
|
|
119
|
+
Idempotent: if the far side already holds a payload with the right hash
|
|
120
|
+
the whole thing is a no-op, which is what makes a dropped bridge cost
|
|
121
|
+
one step rather than one run (failure D1).
|
|
122
|
+
"""
|
|
123
|
+
payload = encode_payload(text) if compress else text
|
|
124
|
+
want = rolling_hash(payload)
|
|
125
|
+
|
|
126
|
+
existing = await self.channel.remote_hash(key)
|
|
127
|
+
if existing == want:
|
|
128
|
+
self._stats["skipped"] = self._stats.get("skipped", 0) + 1
|
|
129
|
+
return want
|
|
130
|
+
|
|
131
|
+
chunks = [payload[i : i + self.chunk_size] for i in range(0, len(payload), self.chunk_size)]
|
|
132
|
+
for i, chunk in enumerate(chunks):
|
|
133
|
+
await self._push_verified(key, i, chunk)
|
|
134
|
+
|
|
135
|
+
await self.channel.assemble(key, len(chunks))
|
|
136
|
+
|
|
137
|
+
got = await self.channel.remote_hash(key)
|
|
138
|
+
if got != want:
|
|
139
|
+
remote_len = await self.channel.remote_length(key)
|
|
140
|
+
raise TransportCorruption(
|
|
141
|
+
f"payload {key!r} corrupted in transit: "
|
|
142
|
+
f"sent {len(payload)} chars hash={want}, "
|
|
143
|
+
f"remote has {remote_len} chars hash={got}"
|
|
144
|
+
)
|
|
145
|
+
self._stats["sent"] = self._stats.get("sent", 0) + len(chunks)
|
|
146
|
+
return want
|
|
147
|
+
|
|
148
|
+
async def _push_verified(self, key: str, index: int, chunk: str, *, attempts: int = 3) -> None:
|
|
149
|
+
want = rolling_hash(chunk)
|
|
150
|
+
last: str | None = None
|
|
151
|
+
for _ in range(attempts):
|
|
152
|
+
await self.channel.push_chunk(key, index, chunk)
|
|
153
|
+
got = await self.channel.remote_hash(f"{key}:{index}")
|
|
154
|
+
if got == want:
|
|
155
|
+
return
|
|
156
|
+
last = got
|
|
157
|
+
raise TransportCorruption(
|
|
158
|
+
f"chunk {index} of {key!r} failed {attempts} verification attempts "
|
|
159
|
+
f"(want {want}, got {last}); channel is lossy at {len(chunk)} chars"
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
class TransportCorruption(RuntimeError):
|
|
164
|
+
"""The channel altered the payload. Never retry blindly past this."""
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def crc(text: str) -> int:
|
|
168
|
+
"""A second, independent checksum. Used where a payload crosses two hops
|
|
169
|
+
and you want to know *which* hop ate it."""
|
|
170
|
+
return zlib.crc32(text.encode("utf-8")) & _MASK
|
pubkit/py.typed
ADDED
|
File without changes
|
pubkit/registry.py
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
# Copyright 2026 The pubkit Authors
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
"""Adapter registry and plugin discovery.
|
|
4
|
+
|
|
5
|
+
Built-ins are registered here; third-party adapters are discovered through the
|
|
6
|
+
`pubkit.adapters` entry point, so adding a platform never requires touching
|
|
7
|
+
this repository.
|
|
8
|
+
|
|
9
|
+
# in your own package's pyproject.toml
|
|
10
|
+
[project.entry-points."pubkit.adapters"]
|
|
11
|
+
ghost = "my_pubkit_ghost:GhostAdapter"
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import logging
|
|
16
|
+
from collections.abc import Callable
|
|
17
|
+
from importlib.metadata import entry_points
|
|
18
|
+
|
|
19
|
+
log = logging.getLogger(__name__)
|
|
20
|
+
|
|
21
|
+
_BUILTIN: dict[str, Callable[..., object]] = {}
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def register(name: str, factory: Callable[..., object]) -> None:
|
|
25
|
+
_BUILTIN[name] = factory
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _load_builtins() -> None:
|
|
29
|
+
if _BUILTIN:
|
|
30
|
+
return
|
|
31
|
+
from .adapters.devto import DevToAdapter, HashnodeAdapter
|
|
32
|
+
from .adapters.medium import MediumAdapter
|
|
33
|
+
from .adapters.substack import SubstackAdapter
|
|
34
|
+
from .adapters.x import XAdapter
|
|
35
|
+
|
|
36
|
+
register("medium", MediumAdapter)
|
|
37
|
+
register("substack", SubstackAdapter)
|
|
38
|
+
register("x", XAdapter)
|
|
39
|
+
register("twitter", XAdapter)
|
|
40
|
+
register("devto", DevToAdapter)
|
|
41
|
+
register("hashnode", HashnodeAdapter)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _plugins() -> dict[str, Callable[..., object]]:
|
|
45
|
+
found: dict[str, Callable[..., object]] = {}
|
|
46
|
+
try:
|
|
47
|
+
eps = entry_points(group="pubkit.adapters")
|
|
48
|
+
except TypeError: # pragma: no cover - older importlib
|
|
49
|
+
eps = entry_points().get("pubkit.adapters", []) # type: ignore[assignment]
|
|
50
|
+
for ep in eps:
|
|
51
|
+
try:
|
|
52
|
+
found[ep.name] = ep.load()
|
|
53
|
+
except Exception: # noqa: BLE001
|
|
54
|
+
log.exception("could not load adapter plugin %s", ep.name)
|
|
55
|
+
return found
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def build_adapter(name: str, **kwargs):
|
|
59
|
+
_load_builtins()
|
|
60
|
+
factories = {**_BUILTIN, **_plugins()}
|
|
61
|
+
if name not in factories:
|
|
62
|
+
raise KeyError(f"unknown platform {name!r}; available: {', '.join(sorted(factories))}")
|
|
63
|
+
factory = factories[name]
|
|
64
|
+
if name == "substack" and "publication" not in kwargs:
|
|
65
|
+
import os
|
|
66
|
+
|
|
67
|
+
kwargs["publication"] = os.environ.get("PUBKIT_SUBSTACK_URL", "https://example.substack.com")
|
|
68
|
+
return factory(**kwargs)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def list_adapters() -> list[tuple[str, object]]:
|
|
72
|
+
_load_builtins()
|
|
73
|
+
out = []
|
|
74
|
+
for name in sorted({**_BUILTIN, **_plugins()}):
|
|
75
|
+
if name == "twitter":
|
|
76
|
+
continue
|
|
77
|
+
try:
|
|
78
|
+
out.append((name, build_adapter(name)))
|
|
79
|
+
except Exception: # noqa: BLE001
|
|
80
|
+
continue
|
|
81
|
+
return out
|
pubkit/render/html.py
ADDED
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
# Copyright 2026 The pubkit Authors
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
"""Renderers: IR → what a platform will actually accept."""
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import html as _html
|
|
7
|
+
import re
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
|
|
10
|
+
from ..core.ir import (
|
|
11
|
+
Callout,
|
|
12
|
+
Code,
|
|
13
|
+
Document,
|
|
14
|
+
Embed,
|
|
15
|
+
Figure,
|
|
16
|
+
Heading,
|
|
17
|
+
ListBlock,
|
|
18
|
+
Paragraph,
|
|
19
|
+
Quote,
|
|
20
|
+
Rule,
|
|
21
|
+
Table,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
_INLINE = [
|
|
25
|
+
(re.compile(r"\*\*(.+?)\*\*", re.S), r"<strong>\1</strong>"),
|
|
26
|
+
(re.compile(r"(?<!\*)\*([^*]+?)\*(?!\*)"), r"<em>\1</em>"),
|
|
27
|
+
(re.compile(r"`([^`]+?)`"), r"<code>\1</code>"),
|
|
28
|
+
(re.compile(r"\[([^\]]+?)\]\(([^)]+?)\)"), r'<a href="\2">\1</a>'),
|
|
29
|
+
]
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def inline(text: str) -> str:
|
|
33
|
+
out = _html.escape(text, quote=False)
|
|
34
|
+
for pattern, repl in _INLINE:
|
|
35
|
+
out = pattern.sub(repl, out)
|
|
36
|
+
return out
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass
|
|
40
|
+
class EditorHtmlRenderer:
|
|
41
|
+
"""HTML shaped for a contenteditable editor, not for the web.
|
|
42
|
+
|
|
43
|
+
Two deliberate constraints:
|
|
44
|
+
|
|
45
|
+
* **No images.** Figures render as a caption paragraph only. The image
|
|
46
|
+
arrives later through the editor's own upload path, because both `data:`
|
|
47
|
+
URIs and remote URLs are stripped from pasted HTML (failure B5).
|
|
48
|
+
The caption doubles as the insertion anchor.
|
|
49
|
+
|
|
50
|
+
* **Tables are pre-rendered to images upstream.** By the time a table gets
|
|
51
|
+
here it is already a Figure, or the planner decided the platform supports
|
|
52
|
+
tables and a different renderer is in use.
|
|
53
|
+
|
|
54
|
+
Heading levels are remapped because editors reserve h1/h2 for title and
|
|
55
|
+
subtitle; article headings start at h3.
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
heading_offset: int = 2
|
|
59
|
+
max_heading: int = 4
|
|
60
|
+
|
|
61
|
+
def render(self, doc: Document) -> str:
|
|
62
|
+
parts: list[str] = [f"<h3>{inline(doc.title)}</h3>"]
|
|
63
|
+
if doc.subtitle:
|
|
64
|
+
parts.append(f"<h4>{inline(doc.subtitle)}</h4>")
|
|
65
|
+
for block in doc.blocks:
|
|
66
|
+
parts.append(self.block(block, doc))
|
|
67
|
+
return "".join(p for p in parts if p)
|
|
68
|
+
|
|
69
|
+
def block(self, b, doc: Document) -> str:
|
|
70
|
+
if isinstance(b, Heading):
|
|
71
|
+
lvl = min(self.max_heading, b.level + self.heading_offset)
|
|
72
|
+
prefix = f"{b.number}. " if b.number else ""
|
|
73
|
+
return f"<h{lvl}>{inline(prefix + b.text)}</h{lvl}>"
|
|
74
|
+
if isinstance(b, Paragraph):
|
|
75
|
+
return f"<p>{inline(b.text)}</p>"
|
|
76
|
+
if isinstance(b, Code):
|
|
77
|
+
return f"<pre><code>{_html.escape(b.text)}</code></pre>"
|
|
78
|
+
if isinstance(b, Quote):
|
|
79
|
+
attr = f"<br><em>— {inline(b.attribution)}</em>" if b.attribution else ""
|
|
80
|
+
return f"<blockquote>{inline(b.text)}{attr}</blockquote>"
|
|
81
|
+
if isinstance(b, ListBlock):
|
|
82
|
+
tag = "ol" if b.ordered else "ul"
|
|
83
|
+
items = "".join(f"<li>{inline(i)}</li>" for i in b.items)
|
|
84
|
+
return f"<{tag}>{items}</{tag}>"
|
|
85
|
+
if isinstance(b, Callout):
|
|
86
|
+
title = f"<strong>{inline(b.title)}</strong><br>" if b.title else ""
|
|
87
|
+
return f"<blockquote>{title}{inline(b.text)}</blockquote>"
|
|
88
|
+
if isinstance(b, Rule):
|
|
89
|
+
return "<hr>"
|
|
90
|
+
if isinstance(b, Embed):
|
|
91
|
+
return f'<p><a href="{_html.escape(b.url, quote=True)}">{_html.escape(b.url)}</a></p>'
|
|
92
|
+
if isinstance(b, Figure):
|
|
93
|
+
# Caption only. It is both the visible caption and the anchor the
|
|
94
|
+
# image will be inserted above.
|
|
95
|
+
cap = b.caption or (doc.assets[b.asset_id].alt if b.asset_id in doc.assets else "")
|
|
96
|
+
return f"<p><em>{inline(cap)}</em></p>" if cap else ""
|
|
97
|
+
if isinstance(b, Table):
|
|
98
|
+
head = "".join(f"<th>{inline(h)}</th>" for h in b.header)
|
|
99
|
+
rows = "".join(
|
|
100
|
+
"<tr>" + "".join(f"<td>{inline(c)}</td>" for c in row) + "</tr>" for row in b.rows
|
|
101
|
+
)
|
|
102
|
+
cap = f"<caption>{inline(b.caption)}</caption>" if b.caption else ""
|
|
103
|
+
return f"<table>{cap}<thead><tr>{head}</tr></thead><tbody>{rows}</tbody></table>"
|
|
104
|
+
return ""
|
|
105
|
+
|
|
106
|
+
def caption_anchors(self, doc: Document) -> list[tuple[str, str]]:
|
|
107
|
+
"""(asset_id, caption) in document order — the insertion plan."""
|
|
108
|
+
out = []
|
|
109
|
+
for b in doc.blocks:
|
|
110
|
+
if isinstance(b, Figure):
|
|
111
|
+
cap = b.caption or (doc.assets[b.asset_id].alt if b.asset_id in doc.assets else "")
|
|
112
|
+
if cap:
|
|
113
|
+
out.append((b.asset_id, cap))
|
|
114
|
+
return out
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
@dataclass
|
|
118
|
+
class MarkdownRenderer:
|
|
119
|
+
"""Plain Markdown for API platforms (Dev.to, Hashnode, Ghost, static sites)."""
|
|
120
|
+
|
|
121
|
+
image_url_for: object = None # Callable[[str], str] | None
|
|
122
|
+
|
|
123
|
+
def render(self, doc: Document) -> str:
|
|
124
|
+
lines: list[str] = []
|
|
125
|
+
for b in doc.blocks:
|
|
126
|
+
lines.append(self.block(b, doc))
|
|
127
|
+
return "\n\n".join(x for x in lines if x).strip() + "\n"
|
|
128
|
+
|
|
129
|
+
def block(self, b, doc: Document) -> str:
|
|
130
|
+
if isinstance(b, Heading):
|
|
131
|
+
prefix = f"{b.number}. " if b.number else ""
|
|
132
|
+
return "#" * min(6, b.level + 1) + " " + prefix + b.text
|
|
133
|
+
if isinstance(b, Paragraph):
|
|
134
|
+
return b.text
|
|
135
|
+
if isinstance(b, Code):
|
|
136
|
+
return f"```{b.language}\n{b.text}\n```"
|
|
137
|
+
if isinstance(b, Quote):
|
|
138
|
+
body = "\n".join("> " + ln for ln in b.text.splitlines())
|
|
139
|
+
return body + (f"\n>\n> — {b.attribution}" if b.attribution else "")
|
|
140
|
+
if isinstance(b, ListBlock):
|
|
141
|
+
return "\n".join(
|
|
142
|
+
(f"{i+1}. " if b.ordered else "- ") + item for i, item in enumerate(b.items)
|
|
143
|
+
)
|
|
144
|
+
if isinstance(b, Callout):
|
|
145
|
+
title = f"**{b.title}**\n>\n" if b.title else ""
|
|
146
|
+
return "> " + title.replace("\n", "\n> ") + b.text
|
|
147
|
+
if isinstance(b, Rule):
|
|
148
|
+
return "---"
|
|
149
|
+
if isinstance(b, Embed):
|
|
150
|
+
return b.url
|
|
151
|
+
if isinstance(b, Figure):
|
|
152
|
+
asset = doc.assets.get(b.asset_id)
|
|
153
|
+
alt = b.alt or (asset.alt if asset else "")
|
|
154
|
+
url = self.image_url_for(b.asset_id) if self.image_url_for else (str(asset.path) if asset else "")
|
|
155
|
+
cap = f"\n*{b.caption}*" if b.caption else ""
|
|
156
|
+
return f"{cap}"
|
|
157
|
+
if isinstance(b, Table):
|
|
158
|
+
align = b.align or ["l"] * len(b.header)
|
|
159
|
+
sep = {"l": ":---", "c": ":---:", "r": "---:"}
|
|
160
|
+
head = "| " + " | ".join(b.header) + " |"
|
|
161
|
+
rule = "| " + " | ".join(sep.get(a, ":---") for a in align) + " |"
|
|
162
|
+
rows = "\n".join("| " + " | ".join(r) + " |" for r in b.rows)
|
|
163
|
+
cap = f"\n*{b.caption}*" if b.caption else ""
|
|
164
|
+
return f"{head}\n{rule}\n{rows}{cap}"
|
|
165
|
+
return ""
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
MD_STRIP = [
|
|
169
|
+
(re.compile(r"\*\*(.+?)\*\*", re.S), r"\1"),
|
|
170
|
+
(re.compile(r"(?<!\*)\*([^*]+?)\*(?!\*)"), r"\1"),
|
|
171
|
+
(re.compile(r"`([^`]+?)`"), r"\1"),
|
|
172
|
+
(re.compile(r"\[([^\]]+?)\]\(([^)]+?)\)"), r"\1"),
|
|
173
|
+
(re.compile(r"<!--.*?-->", re.S), ""),
|
|
174
|
+
]
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def plain(text: str) -> str:
|
|
178
|
+
"""Strip inline markdown.
|
|
179
|
+
|
|
180
|
+
Platforms that take plain text render `**bold**` literally, which reads as
|
|
181
|
+
a formatting bug to every reader who sees it.
|
|
182
|
+
"""
|
|
183
|
+
for pattern, repl in MD_STRIP:
|
|
184
|
+
text = pattern.sub(repl, text)
|
|
185
|
+
return re.sub(r"\s+", " ", text).strip()
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
@dataclass
|
|
189
|
+
class ThreadRenderer:
|
|
190
|
+
"""Split a document into a numbered thread for X/Bluesky/Mastodon."""
|
|
191
|
+
|
|
192
|
+
limit: int = 280
|
|
193
|
+
reserve: int = 12 # room for " (3/11)"
|
|
194
|
+
|
|
195
|
+
def render(self, doc: Document) -> list[str]:
|
|
196
|
+
chunks: list[str] = []
|
|
197
|
+
budget = self.limit - self.reserve
|
|
198
|
+
buf = ""
|
|
199
|
+
for b in doc.blocks:
|
|
200
|
+
if isinstance(b, (Heading, Rule, Figure, Table)):
|
|
201
|
+
if buf:
|
|
202
|
+
chunks.append(buf.strip())
|
|
203
|
+
buf = ""
|
|
204
|
+
if isinstance(b, Heading):
|
|
205
|
+
buf = b.text.strip() + "\n\n"
|
|
206
|
+
continue
|
|
207
|
+
text = getattr(b, "text", "") or (
|
|
208
|
+
"\n".join(b.items) if isinstance(b, ListBlock) else ""
|
|
209
|
+
)
|
|
210
|
+
for sentence in re.split(r"(?<=[.!?])\s+", plain(text)):
|
|
211
|
+
if not sentence:
|
|
212
|
+
continue
|
|
213
|
+
if len(buf) + len(sentence) + 1 > budget:
|
|
214
|
+
chunks.append(buf.strip())
|
|
215
|
+
buf = ""
|
|
216
|
+
buf += sentence + " "
|
|
217
|
+
if buf.strip():
|
|
218
|
+
chunks.append(buf.strip())
|
|
219
|
+
|
|
220
|
+
total = len(chunks)
|
|
221
|
+
return [f"{c} ({i+1}/{total})" if total > 1 else c for i, c in enumerate(chunks)]
|