pubkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pubkit/core/state.py ADDED
@@ -0,0 +1,213 @@
1
+ # Copyright 2026 The pubkit Authors
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Idempotency and resumption.
4
+
5
+ A publishing run is a distributed transaction across platforms you do not
6
+ control, over links that drop. During the run that motivated this tool the
7
+ automation bridge disconnected eight times and the browser extension died
8
+ mid-publish. The only sane response is to make every step resumable and every
9
+ step safe to re-enter.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ import json
14
+ import sqlite3
15
+ import time
16
+ from contextlib import contextmanager
17
+ from dataclasses import dataclass
18
+ from enum import Enum
19
+ from pathlib import Path
20
+
21
+ SCHEMA = """
22
+ CREATE TABLE IF NOT EXISTS runs (
23
+ run_id TEXT PRIMARY KEY,
24
+ plan_hash TEXT NOT NULL,
25
+ created_at REAL NOT NULL,
26
+ status TEXT NOT NULL
27
+ );
28
+ CREATE TABLE IF NOT EXISTS steps (
29
+ run_id TEXT NOT NULL,
30
+ document_id TEXT NOT NULL,
31
+ platform TEXT NOT NULL,
32
+ step TEXT NOT NULL,
33
+ status TEXT NOT NULL,
34
+ content_id TEXT,
35
+ fingerprint TEXT,
36
+ remote_ref TEXT,
37
+ error TEXT,
38
+ updated_at REAL NOT NULL,
39
+ PRIMARY KEY (run_id, document_id, platform, step)
40
+ );
41
+ CREATE TABLE IF NOT EXISTS remotes (
42
+ document_id TEXT NOT NULL,
43
+ platform TEXT NOT NULL,
44
+ content_id TEXT NOT NULL,
45
+ remote_ref TEXT NOT NULL,
46
+ url TEXT,
47
+ published INTEGER NOT NULL DEFAULT 0,
48
+ updated_at REAL NOT NULL,
49
+ PRIMARY KEY (document_id, platform)
50
+ );
51
+ CREATE TABLE IF NOT EXISTS channel_tuning (
52
+ channel TEXT PRIMARY KEY,
53
+ chunk_size INTEGER NOT NULL,
54
+ updated_at REAL NOT NULL
55
+ );
56
+ """
57
+
58
+
59
+ class Step(str, Enum):
60
+ AUTH = "auth"
61
+ DRAFT = "draft"
62
+ CONTENT = "content"
63
+ MEDIA = "media"
64
+ VERIFY = "verify"
65
+ PUBLISH = "publish"
66
+
67
+ @classmethod
68
+ def order(cls) -> list[Step]:
69
+ return [cls.AUTH, cls.DRAFT, cls.CONTENT, cls.MEDIA, cls.VERIFY, cls.PUBLISH]
70
+
71
+
72
+ class Status(str, Enum):
73
+ PENDING = "pending"
74
+ RUNNING = "running"
75
+ DONE = "done"
76
+ FAILED = "failed"
77
+ SKIPPED = "skipped"
78
+
79
+
80
+ @dataclass
81
+ class RemoteRecord:
82
+ document_id: str
83
+ platform: str
84
+ content_id: str
85
+ remote_ref: str
86
+ url: str | None
87
+ published: bool
88
+
89
+
90
+ class StateStore:
91
+ def __init__(self, path: Path | str = ".pubkit/state.sqlite") -> None:
92
+ self.path = Path(path)
93
+ self.path.parent.mkdir(parents=True, exist_ok=True)
94
+ self._conn = sqlite3.connect(self.path)
95
+ self._conn.row_factory = sqlite3.Row
96
+ self._conn.executescript(SCHEMA)
97
+ self._conn.commit()
98
+
99
+ # ------------------------------------------------------------------ runs
100
+ def start_run(self, run_id: str, plan_hash: str) -> None:
101
+ self._conn.execute(
102
+ "INSERT OR REPLACE INTO runs VALUES (?,?,?,?)",
103
+ (run_id, plan_hash, time.time(), Status.RUNNING.value),
104
+ )
105
+ self._conn.commit()
106
+
107
+ def finish_run(self, run_id: str, status: Status) -> None:
108
+ self._conn.execute("UPDATE runs SET status=? WHERE run_id=?", (status.value, run_id))
109
+ self._conn.commit()
110
+
111
+ # ----------------------------------------------------------------- steps
112
+ def step_status(self, run_id: str, doc: str, platform: str, step: Step) -> Status:
113
+ row = self._conn.execute(
114
+ "SELECT status FROM steps WHERE run_id=? AND document_id=? AND platform=? AND step=?",
115
+ (run_id, doc, platform, step.value),
116
+ ).fetchone()
117
+ return Status(row["status"]) if row else Status.PENDING
118
+
119
+ def record_step(
120
+ self,
121
+ run_id: str,
122
+ doc: str,
123
+ platform: str,
124
+ step: Step,
125
+ status: Status,
126
+ *,
127
+ content_id: str | None = None,
128
+ fingerprint: dict | None = None,
129
+ remote_ref: str | None = None,
130
+ error: str | None = None,
131
+ ) -> None:
132
+ self._conn.execute(
133
+ "INSERT OR REPLACE INTO steps VALUES (?,?,?,?,?,?,?,?,?,?)",
134
+ (
135
+ run_id,
136
+ doc,
137
+ platform,
138
+ step.value,
139
+ status.value,
140
+ content_id,
141
+ json.dumps(fingerprint) if fingerprint else None,
142
+ remote_ref,
143
+ error,
144
+ time.time(),
145
+ ),
146
+ )
147
+ self._conn.commit()
148
+
149
+ def resume_point(self, run_id: str, doc: str, platform: str) -> Step:
150
+ """First step that is not `done`. Re-running picks up exactly here."""
151
+ for step in Step.order():
152
+ if self.step_status(run_id, doc, platform, step) is not Status.DONE:
153
+ return step
154
+ return Step.PUBLISH
155
+
156
+ # --------------------------------------------------------------- remotes
157
+ def remember_remote(
158
+ self, doc: str, platform: str, content_id: str, ref: str, url: str | None = None, published: bool = False
159
+ ) -> None:
160
+ self._conn.execute(
161
+ "INSERT OR REPLACE INTO remotes VALUES (?,?,?,?,?,?,?)",
162
+ (doc, platform, content_id, ref, url, int(published), time.time()),
163
+ )
164
+ self._conn.commit()
165
+
166
+ def remote(self, doc: str, platform: str) -> RemoteRecord | None:
167
+ row = self._conn.execute(
168
+ "SELECT * FROM remotes WHERE document_id=? AND platform=?", (doc, platform)
169
+ ).fetchone()
170
+ if not row:
171
+ return None
172
+ return RemoteRecord(
173
+ document_id=row["document_id"],
174
+ platform=row["platform"],
175
+ content_id=row["content_id"],
176
+ remote_ref=row["remote_ref"],
177
+ url=row["url"],
178
+ published=bool(row["published"]),
179
+ )
180
+
181
+ def needs_update(self, doc: str, platform: str, content_id: str) -> bool:
182
+ """False when the remote already holds exactly this content.
183
+
184
+ This is what stops a re-run creating a second draft — the mistake that
185
+ turns a retry into a mess someone has to clean up by hand (failure D2).
186
+ """
187
+ rec = self.remote(doc, platform)
188
+ return rec is None or rec.content_id != content_id
189
+
190
+ # -------------------------------------------------------- channel tuning
191
+ def tuned_chunk_size(self, channel: str) -> int | None:
192
+ row = self._conn.execute(
193
+ "SELECT chunk_size FROM channel_tuning WHERE channel=?", (channel,)
194
+ ).fetchone()
195
+ return row["chunk_size"] if row else None
196
+
197
+ def save_chunk_size(self, channel: str, size: int) -> None:
198
+ self._conn.execute(
199
+ "INSERT OR REPLACE INTO channel_tuning VALUES (?,?,?)", (channel, size, time.time())
200
+ )
201
+ self._conn.commit()
202
+
203
+ @contextmanager
204
+ def transaction(self):
205
+ try:
206
+ yield self._conn
207
+ self._conn.commit()
208
+ except Exception:
209
+ self._conn.rollback()
210
+ raise
211
+
212
+ def close(self) -> None:
213
+ self._conn.close()
@@ -0,0 +1,170 @@
1
+ # Copyright 2026 The pubkit Authors
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Verified chunked transport.
4
+
5
+ Written because a 12,368-character payload once arrived with exactly one byte
6
+ mutated (`h` → `g`) and nothing anywhere reported an error. gzip failed three
7
+ steps later with `invalid distance too far back`, which is a spectacularly
8
+ unhelpful way to learn your channel is lossy.
9
+
10
+ The rules this module enforces:
11
+
12
+ A1 every chunk is hash-verified end to end
13
+ A2 the per-call size ceiling is *probed*, never assumed
14
+ A3 staging survives navigation, and re-pushing is a no-op
15
+ A4 exactly one gzip member per payload
16
+ """
17
+ from __future__ import annotations
18
+
19
+ import base64
20
+ import gzip
21
+ import io
22
+ import zlib
23
+ from collections.abc import Awaitable, Callable
24
+ from dataclasses import dataclass, field
25
+ from typing import Protocol
26
+
27
+ # The rolling hash below is deliberately trivial and identical on both sides:
28
+ # a 32-bit FNV-ish accumulation that a page can compute in three lines of JS
29
+ # without pulling in a crypto library. It is an integrity check against a lossy
30
+ # channel, not a security primitive.
31
+ _MASK = 0xFFFFFFFF
32
+
33
+
34
+ def rolling_hash(s: str) -> str:
35
+ h = 0
36
+ for ch in s:
37
+ h = (h * 31 + ord(ch)) & _MASK
38
+ return format(h, "x")
39
+
40
+
41
+ #: The JavaScript twin of :func:`rolling_hash`. Kept next to it on purpose —
42
+ #: if one changes and the other does not, verification breaks silently, which
43
+ #: is the exact failure this module exists to prevent.
44
+ ROLLING_HASH_JS = """
45
+ window.__pkHash = function (s) {
46
+ let h = 0;
47
+ for (let i = 0; i < s.length; i++) { h = (h * 31 + s.charCodeAt(i)) >>> 0; }
48
+ return h.toString(16);
49
+ };
50
+ """
51
+
52
+
53
+ def encode_payload(text: str) -> str:
54
+ """gzip + base64, guaranteed single-member (failure A4).
55
+
56
+ `gzip.compress` emits one member; concatenating two independently
57
+ compressed halves does not, and `DecompressionStream('gzip')` rejects that
58
+ with `Junk found after end of compressed data` while Python's
59
+ `gzip.decompress` happily accepts it. Divergent behaviour between the two
60
+ ends of a pipe is how a bug hides for an hour.
61
+ """
62
+ buf = io.BytesIO()
63
+ with gzip.GzipFile(fileobj=buf, mode="wb", mtime=0) as fh:
64
+ fh.write(text.encode("utf-8"))
65
+ raw = buf.getvalue()
66
+ assert raw.count(b"\x1f\x8b") >= 1, "not gzip"
67
+ return base64.b64encode(raw).decode("ascii")
68
+
69
+
70
+ def decode_payload(b64: str) -> str:
71
+ return gzip.decompress(base64.b64decode(b64)).decode("utf-8")
72
+
73
+
74
+ class Channel(Protocol):
75
+ """Anything that can move a string to the far side and run code there."""
76
+
77
+ async def push_chunk(self, key: str, index: int, data: str) -> None: ...
78
+ async def remote_hash(self, key: str) -> str | None: ...
79
+ async def remote_length(self, key: str) -> int: ...
80
+ async def assemble(self, key: str, count: int) -> None: ...
81
+
82
+
83
+ @dataclass
84
+ class ChunkedTransport:
85
+ """Move a large payload across a channel with an unknown, undocumented
86
+ per-call size ceiling, and prove it arrived intact."""
87
+
88
+ channel: Channel
89
+ #: Starting guess. The real ceiling is discovered, not trusted.
90
+ chunk_size: int = 5_600
91
+ max_chunk: int = 45_000
92
+ min_chunk: int = 1_024
93
+ probed: bool = False
94
+ _stats: dict = field(default_factory=dict)
95
+
96
+ async def probe_ceiling(self, probe: Callable[[int], Awaitable[bool]]) -> int:
97
+ """Binary-search the largest chunk the channel accepts (failure A2).
98
+
99
+ Called once per channel; the result belongs in the state store so the
100
+ next run starts from the known-good value.
101
+ """
102
+ lo, hi = self.min_chunk, self.max_chunk
103
+ best = self.min_chunk
104
+ while lo <= hi:
105
+ mid = (lo + hi) // 2
106
+ if await probe(mid):
107
+ best, lo = mid, mid + 1
108
+ else:
109
+ hi = mid - 1
110
+ # Back off 20%: the ceiling is not always stable under load, and a
111
+ # chunk that fails mid-run costs a retry plus a reassembly.
112
+ self.chunk_size = max(self.min_chunk, int(best * 0.8))
113
+ self.probed = True
114
+ return self.chunk_size
115
+
116
+ async def send(self, key: str, text: str, *, compress: bool = True) -> str:
117
+ """Push `text` under `key`, verify it, and return the payload hash.
118
+
119
+ Idempotent: if the far side already holds a payload with the right hash
120
+ the whole thing is a no-op, which is what makes a dropped bridge cost
121
+ one step rather than one run (failure D1).
122
+ """
123
+ payload = encode_payload(text) if compress else text
124
+ want = rolling_hash(payload)
125
+
126
+ existing = await self.channel.remote_hash(key)
127
+ if existing == want:
128
+ self._stats["skipped"] = self._stats.get("skipped", 0) + 1
129
+ return want
130
+
131
+ chunks = [payload[i : i + self.chunk_size] for i in range(0, len(payload), self.chunk_size)]
132
+ for i, chunk in enumerate(chunks):
133
+ await self._push_verified(key, i, chunk)
134
+
135
+ await self.channel.assemble(key, len(chunks))
136
+
137
+ got = await self.channel.remote_hash(key)
138
+ if got != want:
139
+ remote_len = await self.channel.remote_length(key)
140
+ raise TransportCorruption(
141
+ f"payload {key!r} corrupted in transit: "
142
+ f"sent {len(payload)} chars hash={want}, "
143
+ f"remote has {remote_len} chars hash={got}"
144
+ )
145
+ self._stats["sent"] = self._stats.get("sent", 0) + len(chunks)
146
+ return want
147
+
148
+ async def _push_verified(self, key: str, index: int, chunk: str, *, attempts: int = 3) -> None:
149
+ want = rolling_hash(chunk)
150
+ last: str | None = None
151
+ for _ in range(attempts):
152
+ await self.channel.push_chunk(key, index, chunk)
153
+ got = await self.channel.remote_hash(f"{key}:{index}")
154
+ if got == want:
155
+ return
156
+ last = got
157
+ raise TransportCorruption(
158
+ f"chunk {index} of {key!r} failed {attempts} verification attempts "
159
+ f"(want {want}, got {last}); channel is lossy at {len(chunk)} chars"
160
+ )
161
+
162
+
163
+ class TransportCorruption(RuntimeError):
164
+ """The channel altered the payload. Never retry blindly past this."""
165
+
166
+
167
+ def crc(text: str) -> int:
168
+ """A second, independent checksum. Used where a payload crosses two hops
169
+ and you want to know *which* hop ate it."""
170
+ return zlib.crc32(text.encode("utf-8")) & _MASK
pubkit/py.typed ADDED
File without changes
pubkit/registry.py ADDED
@@ -0,0 +1,81 @@
1
+ # Copyright 2026 The pubkit Authors
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Adapter registry and plugin discovery.
4
+
5
+ Built-ins are registered here; third-party adapters are discovered through the
6
+ `pubkit.adapters` entry point, so adding a platform never requires touching
7
+ this repository.
8
+
9
+ # in your own package's pyproject.toml
10
+ [project.entry-points."pubkit.adapters"]
11
+ ghost = "my_pubkit_ghost:GhostAdapter"
12
+ """
13
+ from __future__ import annotations
14
+
15
+ import logging
16
+ from collections.abc import Callable
17
+ from importlib.metadata import entry_points
18
+
19
+ log = logging.getLogger(__name__)
20
+
21
+ _BUILTIN: dict[str, Callable[..., object]] = {}
22
+
23
+
24
+ def register(name: str, factory: Callable[..., object]) -> None:
25
+ _BUILTIN[name] = factory
26
+
27
+
28
+ def _load_builtins() -> None:
29
+ if _BUILTIN:
30
+ return
31
+ from .adapters.devto import DevToAdapter, HashnodeAdapter
32
+ from .adapters.medium import MediumAdapter
33
+ from .adapters.substack import SubstackAdapter
34
+ from .adapters.x import XAdapter
35
+
36
+ register("medium", MediumAdapter)
37
+ register("substack", SubstackAdapter)
38
+ register("x", XAdapter)
39
+ register("twitter", XAdapter)
40
+ register("devto", DevToAdapter)
41
+ register("hashnode", HashnodeAdapter)
42
+
43
+
44
+ def _plugins() -> dict[str, Callable[..., object]]:
45
+ found: dict[str, Callable[..., object]] = {}
46
+ try:
47
+ eps = entry_points(group="pubkit.adapters")
48
+ except TypeError: # pragma: no cover - older importlib
49
+ eps = entry_points().get("pubkit.adapters", []) # type: ignore[assignment]
50
+ for ep in eps:
51
+ try:
52
+ found[ep.name] = ep.load()
53
+ except Exception: # noqa: BLE001
54
+ log.exception("could not load adapter plugin %s", ep.name)
55
+ return found
56
+
57
+
58
+ def build_adapter(name: str, **kwargs):
59
+ _load_builtins()
60
+ factories = {**_BUILTIN, **_plugins()}
61
+ if name not in factories:
62
+ raise KeyError(f"unknown platform {name!r}; available: {', '.join(sorted(factories))}")
63
+ factory = factories[name]
64
+ if name == "substack" and "publication" not in kwargs:
65
+ import os
66
+
67
+ kwargs["publication"] = os.environ.get("PUBKIT_SUBSTACK_URL", "https://example.substack.com")
68
+ return factory(**kwargs)
69
+
70
+
71
+ def list_adapters() -> list[tuple[str, object]]:
72
+ _load_builtins()
73
+ out = []
74
+ for name in sorted({**_BUILTIN, **_plugins()}):
75
+ if name == "twitter":
76
+ continue
77
+ try:
78
+ out.append((name, build_adapter(name)))
79
+ except Exception: # noqa: BLE001
80
+ continue
81
+ return out
@@ -0,0 +1,2 @@
1
+ # Copyright 2026 The pubkit Authors
2
+ # SPDX-License-Identifier: Apache-2.0
pubkit/render/html.py ADDED
@@ -0,0 +1,221 @@
1
+ # Copyright 2026 The pubkit Authors
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Renderers: IR → what a platform will actually accept."""
4
+ from __future__ import annotations
5
+
6
+ import html as _html
7
+ import re
8
+ from dataclasses import dataclass
9
+
10
+ from ..core.ir import (
11
+ Callout,
12
+ Code,
13
+ Document,
14
+ Embed,
15
+ Figure,
16
+ Heading,
17
+ ListBlock,
18
+ Paragraph,
19
+ Quote,
20
+ Rule,
21
+ Table,
22
+ )
23
+
24
+ _INLINE = [
25
+ (re.compile(r"\*\*(.+?)\*\*", re.S), r"<strong>\1</strong>"),
26
+ (re.compile(r"(?<!\*)\*([^*]+?)\*(?!\*)"), r"<em>\1</em>"),
27
+ (re.compile(r"`([^`]+?)`"), r"<code>\1</code>"),
28
+ (re.compile(r"\[([^\]]+?)\]\(([^)]+?)\)"), r'<a href="\2">\1</a>'),
29
+ ]
30
+
31
+
32
+ def inline(text: str) -> str:
33
+ out = _html.escape(text, quote=False)
34
+ for pattern, repl in _INLINE:
35
+ out = pattern.sub(repl, out)
36
+ return out
37
+
38
+
39
+ @dataclass
40
+ class EditorHtmlRenderer:
41
+ """HTML shaped for a contenteditable editor, not for the web.
42
+
43
+ Two deliberate constraints:
44
+
45
+ * **No images.** Figures render as a caption paragraph only. The image
46
+ arrives later through the editor's own upload path, because both `data:`
47
+ URIs and remote URLs are stripped from pasted HTML (failure B5).
48
+ The caption doubles as the insertion anchor.
49
+
50
+ * **Tables are pre-rendered to images upstream.** By the time a table gets
51
+ here it is already a Figure, or the planner decided the platform supports
52
+ tables and a different renderer is in use.
53
+
54
+ Heading levels are remapped because editors reserve h1/h2 for title and
55
+ subtitle; article headings start at h3.
56
+ """
57
+
58
+ heading_offset: int = 2
59
+ max_heading: int = 4
60
+
61
+ def render(self, doc: Document) -> str:
62
+ parts: list[str] = [f"<h3>{inline(doc.title)}</h3>"]
63
+ if doc.subtitle:
64
+ parts.append(f"<h4>{inline(doc.subtitle)}</h4>")
65
+ for block in doc.blocks:
66
+ parts.append(self.block(block, doc))
67
+ return "".join(p for p in parts if p)
68
+
69
+ def block(self, b, doc: Document) -> str:
70
+ if isinstance(b, Heading):
71
+ lvl = min(self.max_heading, b.level + self.heading_offset)
72
+ prefix = f"{b.number}. " if b.number else ""
73
+ return f"<h{lvl}>{inline(prefix + b.text)}</h{lvl}>"
74
+ if isinstance(b, Paragraph):
75
+ return f"<p>{inline(b.text)}</p>"
76
+ if isinstance(b, Code):
77
+ return f"<pre><code>{_html.escape(b.text)}</code></pre>"
78
+ if isinstance(b, Quote):
79
+ attr = f"<br><em>— {inline(b.attribution)}</em>" if b.attribution else ""
80
+ return f"<blockquote>{inline(b.text)}{attr}</blockquote>"
81
+ if isinstance(b, ListBlock):
82
+ tag = "ol" if b.ordered else "ul"
83
+ items = "".join(f"<li>{inline(i)}</li>" for i in b.items)
84
+ return f"<{tag}>{items}</{tag}>"
85
+ if isinstance(b, Callout):
86
+ title = f"<strong>{inline(b.title)}</strong><br>" if b.title else ""
87
+ return f"<blockquote>{title}{inline(b.text)}</blockquote>"
88
+ if isinstance(b, Rule):
89
+ return "<hr>"
90
+ if isinstance(b, Embed):
91
+ return f'<p><a href="{_html.escape(b.url, quote=True)}">{_html.escape(b.url)}</a></p>'
92
+ if isinstance(b, Figure):
93
+ # Caption only. It is both the visible caption and the anchor the
94
+ # image will be inserted above.
95
+ cap = b.caption or (doc.assets[b.asset_id].alt if b.asset_id in doc.assets else "")
96
+ return f"<p><em>{inline(cap)}</em></p>" if cap else ""
97
+ if isinstance(b, Table):
98
+ head = "".join(f"<th>{inline(h)}</th>" for h in b.header)
99
+ rows = "".join(
100
+ "<tr>" + "".join(f"<td>{inline(c)}</td>" for c in row) + "</tr>" for row in b.rows
101
+ )
102
+ cap = f"<caption>{inline(b.caption)}</caption>" if b.caption else ""
103
+ return f"<table>{cap}<thead><tr>{head}</tr></thead><tbody>{rows}</tbody></table>"
104
+ return ""
105
+
106
+ def caption_anchors(self, doc: Document) -> list[tuple[str, str]]:
107
+ """(asset_id, caption) in document order — the insertion plan."""
108
+ out = []
109
+ for b in doc.blocks:
110
+ if isinstance(b, Figure):
111
+ cap = b.caption or (doc.assets[b.asset_id].alt if b.asset_id in doc.assets else "")
112
+ if cap:
113
+ out.append((b.asset_id, cap))
114
+ return out
115
+
116
+
117
+ @dataclass
118
+ class MarkdownRenderer:
119
+ """Plain Markdown for API platforms (Dev.to, Hashnode, Ghost, static sites)."""
120
+
121
+ image_url_for: object = None # Callable[[str], str] | None
122
+
123
+ def render(self, doc: Document) -> str:
124
+ lines: list[str] = []
125
+ for b in doc.blocks:
126
+ lines.append(self.block(b, doc))
127
+ return "\n\n".join(x for x in lines if x).strip() + "\n"
128
+
129
+ def block(self, b, doc: Document) -> str:
130
+ if isinstance(b, Heading):
131
+ prefix = f"{b.number}. " if b.number else ""
132
+ return "#" * min(6, b.level + 1) + " " + prefix + b.text
133
+ if isinstance(b, Paragraph):
134
+ return b.text
135
+ if isinstance(b, Code):
136
+ return f"```{b.language}\n{b.text}\n```"
137
+ if isinstance(b, Quote):
138
+ body = "\n".join("> " + ln for ln in b.text.splitlines())
139
+ return body + (f"\n>\n> — {b.attribution}" if b.attribution else "")
140
+ if isinstance(b, ListBlock):
141
+ return "\n".join(
142
+ (f"{i+1}. " if b.ordered else "- ") + item for i, item in enumerate(b.items)
143
+ )
144
+ if isinstance(b, Callout):
145
+ title = f"**{b.title}**\n>\n" if b.title else ""
146
+ return "> " + title.replace("\n", "\n> ") + b.text
147
+ if isinstance(b, Rule):
148
+ return "---"
149
+ if isinstance(b, Embed):
150
+ return b.url
151
+ if isinstance(b, Figure):
152
+ asset = doc.assets.get(b.asset_id)
153
+ alt = b.alt or (asset.alt if asset else "")
154
+ url = self.image_url_for(b.asset_id) if self.image_url_for else (str(asset.path) if asset else "")
155
+ cap = f"\n*{b.caption}*" if b.caption else ""
156
+ return f"![{alt}]({url}){cap}"
157
+ if isinstance(b, Table):
158
+ align = b.align or ["l"] * len(b.header)
159
+ sep = {"l": ":---", "c": ":---:", "r": "---:"}
160
+ head = "| " + " | ".join(b.header) + " |"
161
+ rule = "| " + " | ".join(sep.get(a, ":---") for a in align) + " |"
162
+ rows = "\n".join("| " + " | ".join(r) + " |" for r in b.rows)
163
+ cap = f"\n*{b.caption}*" if b.caption else ""
164
+ return f"{head}\n{rule}\n{rows}{cap}"
165
+ return ""
166
+
167
+
168
+ MD_STRIP = [
169
+ (re.compile(r"\*\*(.+?)\*\*", re.S), r"\1"),
170
+ (re.compile(r"(?<!\*)\*([^*]+?)\*(?!\*)"), r"\1"),
171
+ (re.compile(r"`([^`]+?)`"), r"\1"),
172
+ (re.compile(r"\[([^\]]+?)\]\(([^)]+?)\)"), r"\1"),
173
+ (re.compile(r"<!--.*?-->", re.S), ""),
174
+ ]
175
+
176
+
177
+ def plain(text: str) -> str:
178
+ """Strip inline markdown.
179
+
180
+ Platforms that take plain text render `**bold**` literally, which reads as
181
+ a formatting bug to every reader who sees it.
182
+ """
183
+ for pattern, repl in MD_STRIP:
184
+ text = pattern.sub(repl, text)
185
+ return re.sub(r"\s+", " ", text).strip()
186
+
187
+
188
+ @dataclass
189
+ class ThreadRenderer:
190
+ """Split a document into a numbered thread for X/Bluesky/Mastodon."""
191
+
192
+ limit: int = 280
193
+ reserve: int = 12 # room for " (3/11)"
194
+
195
+ def render(self, doc: Document) -> list[str]:
196
+ chunks: list[str] = []
197
+ budget = self.limit - self.reserve
198
+ buf = ""
199
+ for b in doc.blocks:
200
+ if isinstance(b, (Heading, Rule, Figure, Table)):
201
+ if buf:
202
+ chunks.append(buf.strip())
203
+ buf = ""
204
+ if isinstance(b, Heading):
205
+ buf = b.text.strip() + "\n\n"
206
+ continue
207
+ text = getattr(b, "text", "") or (
208
+ "\n".join(b.items) if isinstance(b, ListBlock) else ""
209
+ )
210
+ for sentence in re.split(r"(?<=[.!?])\s+", plain(text)):
211
+ if not sentence:
212
+ continue
213
+ if len(buf) + len(sentence) + 1 > budget:
214
+ chunks.append(buf.strip())
215
+ buf = ""
216
+ buf += sentence + " "
217
+ if buf.strip():
218
+ chunks.append(buf.strip())
219
+
220
+ total = len(chunks)
221
+ return [f"{c} ({i+1}/{total})" if total > 1 else c for i, c in enumerate(chunks)]