pubkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pubkit/core/loader.py ADDED
@@ -0,0 +1,278 @@
1
+ # Copyright 2026 The pubkit Authors
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Markdown + front-matter → Post IR."""
4
+ from __future__ import annotations
5
+
6
+ import re
7
+ from pathlib import Path
8
+
9
+ import yaml
10
+
11
+ from .ir import (
12
+ Asset,
13
+ Budget,
14
+ Callout,
15
+ Code,
16
+ Document,
17
+ Embed,
18
+ Figure,
19
+ Heading,
20
+ ListBlock,
21
+ Paragraph,
22
+ Quote,
23
+ Rule,
24
+ Series,
25
+ SeriesRef,
26
+ Table,
27
+ )
28
+
29
+
30
+ class FrontMatterError(ValueError):
31
+ """Bad YAML front-matter, reported in terms an author can act on."""
32
+
33
+
34
+ FM_RE = re.compile(r"^---\n(.*?)\n---\n", re.S)
35
+ FIG_RE = re.compile(r"^!\[(?P<alt>[^\]]*)\]\((?P<src>[^)\s]+)\)(?:\s*\"(?P<cap>[^\"]*)\")?\s*$")
36
+ CALLOUT_RE = re.compile(r"^>\s*\[!(?P<style>note|warning|tip|key)\]\s*(?P<title>.*)$", re.I)
37
+
38
+
39
+ def _table(lines: list[str], caption: str | None) -> Table:
40
+ cells = [[c.strip() for c in ln.strip().strip("|").split("|")] for ln in lines]
41
+ header, sep, *rows = cells
42
+ align = []
43
+ for s in sep:
44
+ align.append("c" if s.startswith(":") and s.endswith(":") else "r" if s.endswith(":") else "l")
45
+ return Table(header=header, rows=rows, align=align, caption=caption)
46
+
47
+
48
+ def parse_markdown(text: str, *, base: Path) -> tuple[dict, list, dict[str, Asset]]:
49
+ meta: dict = {}
50
+ m = FM_RE.match(text)
51
+ if m:
52
+ try:
53
+ meta = yaml.safe_load(m.group(1)) or {}
54
+ except yaml.YAMLError as exc:
55
+ # Titles routinely contain a colon ("Part 1: The Hardware"), which
56
+ # is invalid unquoted YAML. Say so plainly instead of surfacing a
57
+ # parser traceback at the author.
58
+ raise FrontMatterError(
59
+ f"invalid front-matter: {exc}\n"
60
+ "Tip: quote any value containing a colon, e.g.\n"
61
+ ' title: "Inside AI Infrastructure, Part 1: The Hardware"'
62
+ ) from exc
63
+ if not isinstance(meta, dict):
64
+ raise FrontMatterError("front-matter must be a mapping of keys to values")
65
+ text = text[m.end() :]
66
+
67
+ blocks: list = []
68
+ assets: dict[str, Asset] = {}
69
+ lines = text.split("\n")
70
+ i = 0
71
+ fig_n = 0
72
+
73
+ while i < len(lines):
74
+ line = lines[i]
75
+ stripped = line.strip()
76
+
77
+ if not stripped:
78
+ i += 1
79
+ continue
80
+
81
+ # fenced code
82
+ if stripped.startswith("```"):
83
+ lang = stripped[3:].strip()
84
+ i += 1
85
+ buf = []
86
+ while i < len(lines) and not lines[i].strip().startswith("```"):
87
+ buf.append(lines[i])
88
+ i += 1
89
+ i += 1
90
+ blocks.append(Code(language=lang, text="\n".join(buf)))
91
+ continue
92
+
93
+ # horizontal rule
94
+ if re.fullmatch(r"(\*\s*){3,}|(-\s*){3,}|(_\s*){3,}", stripped):
95
+ blocks.append(Rule())
96
+ i += 1
97
+ continue
98
+
99
+ # heading
100
+ if stripped.startswith("#"):
101
+ level = len(stripped) - len(stripped.lstrip("#"))
102
+ blocks.append(Heading(level=level, text=stripped[level:].strip()))
103
+ i += 1
104
+ continue
105
+
106
+ # table
107
+ if stripped.startswith("|") and i + 1 < len(lines) and re.match(r"^\|[\s:\-|]+\|$", lines[i + 1].strip()):
108
+ buf = []
109
+ while i < len(lines) and lines[i].strip().startswith("|"):
110
+ buf.append(lines[i])
111
+ i += 1
112
+ cap = None
113
+ if i < len(lines) and lines[i].strip().startswith("*") and lines[i].strip().endswith("*"):
114
+ cap = lines[i].strip().strip("*")
115
+ i += 1
116
+ blocks.append(_table(buf, cap))
117
+ continue
118
+
119
+ # figure
120
+ fm = FIG_RE.match(stripped)
121
+ if fm:
122
+ fig_n += 1
123
+ aid = f"fig-{fig_n:02d}"
124
+ src = fm.group("src")
125
+ assets[aid] = Asset(id=aid, path=(base / src).resolve(), alt=fm.group("alt") or "")
126
+ cap = fm.group("cap")
127
+ if cap is None and i + 1 < len(lines):
128
+ nxt = lines[i + 1].strip()
129
+ if nxt.startswith("*") and nxt.endswith("*") and len(nxt) > 2:
130
+ cap = nxt.strip("*")
131
+ i += 1
132
+ blocks.append(Figure(asset_id=aid, caption=cap, alt=fm.group("alt") or ""))
133
+ i += 1
134
+ continue
135
+
136
+ # callout
137
+ cm = CALLOUT_RE.match(stripped)
138
+ if cm:
139
+ i += 1
140
+ buf = []
141
+ while i < len(lines) and lines[i].strip().startswith(">"):
142
+ buf.append(lines[i].strip().lstrip(">").strip())
143
+ i += 1
144
+ blocks.append(
145
+ Callout(style=cm.group("style").lower(), title=cm.group("title") or None, text=" ".join(buf))
146
+ )
147
+ continue
148
+
149
+ # quote
150
+ if stripped.startswith(">"):
151
+ buf = []
152
+ while i < len(lines) and lines[i].strip().startswith(">"):
153
+ buf.append(lines[i].strip().lstrip(">").strip())
154
+ i += 1
155
+ blocks.append(Quote(text=" ".join(b for b in buf if b)))
156
+ continue
157
+
158
+ # list
159
+ if re.match(r"^([-*+]|\d+\.)\s+", stripped):
160
+ ordered = bool(re.match(r"^\d+\.", stripped))
161
+ items = []
162
+ while i < len(lines) and re.match(r"^([-*+]|\d+\.)\s+", lines[i].strip()):
163
+ items.append(re.sub(r"^([-*+]|\d+\.)\s+", "", lines[i].strip()))
164
+ i += 1
165
+ blocks.append(ListBlock(ordered=ordered, items=items))
166
+ continue
167
+
168
+ # embed on its own line
169
+ if re.fullmatch(r"https?://\S+", stripped):
170
+ kind = (
171
+ "youtube" if "youtu" in stripped
172
+ else "gist" if "gist.github" in stripped
173
+ else "tweet" if ("twitter.com" in stripped or "x.com" in stripped)
174
+ else "generic"
175
+ )
176
+ blocks.append(Embed(url=stripped, kind=kind))
177
+ i += 1
178
+ continue
179
+
180
+ # paragraph
181
+ buf = []
182
+ while i < len(lines) and lines[i].strip() and not lines[i].strip().startswith(("#", ">", "|", "```")):
183
+ if FIG_RE.match(lines[i].strip()):
184
+ break
185
+ buf.append(lines[i].strip())
186
+ i += 1
187
+ if buf:
188
+ blocks.append(Paragraph(text=" ".join(buf)))
189
+
190
+ return meta, blocks, assets
191
+
192
+
193
+ def load_document(path: Path) -> Document:
194
+ meta, blocks, assets = parse_markdown(path.read_text(encoding="utf-8"), base=path.parent)
195
+ series = None
196
+ if "series" in meta:
197
+ s = meta["series"]
198
+ series = SeriesRef(id=s["id"], index=int(s["index"]), of=int(s["of"]))
199
+ budget = None
200
+ if "budget" in meta:
201
+ b = meta["budget"]
202
+ budget = Budget(words=b.get("words"), tolerance=float(b.get("tolerance", 0.15)))
203
+
204
+ doc = Document(
205
+ id=meta.get("id") or path.stem,
206
+ title=meta.get("title") or "Untitled",
207
+ subtitle=meta.get("subtitle"),
208
+ tags=list(meta.get("tags") or []),
209
+ canonical_url=meta.get("canonical_url"),
210
+ blocks=blocks,
211
+ assets=assets,
212
+ series=series,
213
+ budget=budget,
214
+ platform_overrides=meta.get("platforms") or {},
215
+ )
216
+ number_sections(doc)
217
+ return doc
218
+
219
+
220
+ def number_sections(doc: Document, start: int = 1) -> None:
221
+ """Generate section numbers.
222
+
223
+ Hand-numbered sections are how "Part 1" ended up meaning both a series part
224
+ and a section heading in the same document, breaking seven cross-references
225
+ when one article became three (failure C2). Numbering is generated here and
226
+ nowhere else.
227
+ """
228
+ major = start - 1
229
+ minor = 0
230
+ for b in doc.blocks:
231
+ if isinstance(b, Heading):
232
+ if b.level == 1:
233
+ major += 1
234
+ minor = 0
235
+ b.number = str(major)
236
+ elif b.level == 2 and major > 0:
237
+ minor += 1
238
+ b.number = f"{major}.{minor}"
239
+ else:
240
+ b.number = None
241
+
242
+
243
+ def load_series(paths: list[Path]) -> Series:
244
+ docs = [load_document(p) for p in paths]
245
+ docs.sort(key=lambda d: d.series.index if d.series else 0)
246
+ # Continue section numbering across the series so chapter 4 follows 3.
247
+ offset = 1
248
+ for d in docs:
249
+ number_sections(d, start=offset)
250
+ offset += sum(1 for b in d.blocks if isinstance(b, Heading) and b.level == 1)
251
+ sid = docs[0].series.id if docs[0].series else "series"
252
+ return Series(id=sid, title=docs[0].title.split(",")[0], documents=docs)
253
+
254
+
255
+ def resolve_series_links(series: Series) -> None:
256
+ """Phase 2 of the two-phase publish: fill in `${series.partN.url}`.
257
+
258
+ Until every draft exists, a cross-link has nothing to point at. Publishing
259
+ with the placeholder still in place is how `URL-PART-2` nearly reached a
260
+ live post (failure C3).
261
+ """
262
+ urls = {
263
+ d.series.index: d.series.sibling_urls.get(d.series.index) or ""
264
+ for d in series.documents
265
+ if d.series
266
+ }
267
+ for d in series.documents:
268
+ if d.series:
269
+ urls.update(d.series.sibling_urls)
270
+ for d in series.documents:
271
+ for b in d.blocks:
272
+ text = getattr(b, "text", None)
273
+ if not text:
274
+ continue
275
+ for idx, url in urls.items():
276
+ if url:
277
+ text = text.replace(f"${{series.part{idx}.url}}", url)
278
+ b.text = text
pubkit/core/runner.py ADDED
@@ -0,0 +1,268 @@
1
+ # Copyright 2026 The pubkit Authors
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """The runner: idempotent, resumable, gated, fan-out-safe."""
4
+ from __future__ import annotations
5
+
6
+ import hashlib
7
+ import logging
8
+ import time
9
+ from collections.abc import Callable, Sequence
10
+ from dataclasses import dataclass, field
11
+
12
+ from .adapter import Adapter, Context, PublishedRef, VerificationFailed, with_retry
13
+ from .auth import CredentialError
14
+ from .capabilities import PublishPlan
15
+ from .capabilities import plan as make_plan
16
+ from .checks import CheckResult, run_checks
17
+ from .ir import Document, Series
18
+ from .state import StateStore, Status, Step
19
+
20
+ log = logging.getLogger(__name__)
21
+
22
+ HOOKS = ("pre_check", "post_plan", "pre_publish", "post_publish", "on_error")
23
+
24
+
25
+ @dataclass
26
+ class LegResult:
27
+ document_id: str
28
+ platform: str
29
+ status: str
30
+ url: str | None = None
31
+ error: str | None = None
32
+ resumed_from: str | None = None
33
+
34
+
35
+ @dataclass
36
+ class RunReport:
37
+ run_id: str
38
+ legs: list[LegResult] = field(default_factory=list)
39
+
40
+ @property
41
+ def ok(self) -> bool:
42
+ return all(leg.status in ("published", "skipped", "drafted") for leg in self.legs)
43
+
44
+ def human(self) -> str:
45
+ rows = []
46
+ for leg in self.legs:
47
+ mark = {"published": "✓", "drafted": "◔", "skipped": "–", "failed": "✗"}.get(leg.status, "?")
48
+ tail = leg.url or leg.error or ""
49
+ resumed = f" (resumed at {leg.resumed_from})" if leg.resumed_from else ""
50
+ rows.append(f" {mark} {leg.platform:<10} {leg.document_id:<28} {tail}{resumed}")
51
+ return "\n".join(rows)
52
+
53
+
54
+ class Pipeline:
55
+ """Load → check → plan → publish, with every step resumable.
56
+
57
+ The shape is dictated by two things that happen constantly in practice: the
58
+ channel drops mid-run, and you re-run the same command afterwards. If a
59
+ re-run is not safe, nobody will re-run it, and they will finish the job by
60
+ hand at 1am.
61
+ """
62
+
63
+ def __init__(
64
+ self,
65
+ state: StateStore | None = None,
66
+ hooks: dict[str, list[Callable]] | None = None,
67
+ ) -> None:
68
+ self.state = state or StateStore()
69
+ self.hooks: dict[str, list[Callable]] = {h: [] for h in HOOKS}
70
+ for name, fns in (hooks or {}).items():
71
+ self.hooks.setdefault(name, []).extend(fns)
72
+
73
+ def on(self, event: str, fn: Callable) -> None:
74
+ self.hooks.setdefault(event, []).append(fn)
75
+
76
+ def _fire(self, event: str, **kw) -> None:
77
+ for fn in self.hooks.get(event, []):
78
+ try:
79
+ result = fn(**kw)
80
+ if event == "pre_publish" and result is False:
81
+ raise PermissionError("a pre_publish hook vetoed this publish")
82
+ except PermissionError:
83
+ raise
84
+ except Exception: # noqa: BLE001 - a bad hook must not break a run
85
+ log.exception("hook %s failed", event)
86
+
87
+ # ------------------------------------------------------------------ plan
88
+ def check(self, doc: Document, ctx: dict | None = None) -> CheckResult:
89
+ self._fire("pre_check", doc=doc)
90
+ return run_checks(doc, ctx)
91
+
92
+ def plan(self, docs: Sequence[Document], adapters: Sequence[Adapter]) -> list[PublishPlan]:
93
+ plans = [make_plan(d, a.name, a.capabilities, a) for d in docs for a in adapters]
94
+ self._fire("post_plan", plans=plans)
95
+ return plans
96
+
97
+ @staticmethod
98
+ def plan_hash(plans: Sequence[PublishPlan]) -> str:
99
+ """Pin the plan to its content.
100
+
101
+ `--confirm` approves a *specific* plan. If the content changes between
102
+ planning and applying, the hash changes and the run stops — which is the
103
+ difference between reviewing what ships and reviewing something else
104
+ (failure D3).
105
+ """
106
+ payload = "|".join(f"{p.platform}:{p.document_id}:{p.content_id}" for p in sorted(
107
+ plans, key=lambda x: (x.platform, x.document_id)
108
+ ))
109
+ return hashlib.blake2b(payload.encode(), digest_size=12).hexdigest()
110
+
111
+ # --------------------------------------------------------------- publish
112
+ async def run_leg(
113
+ self,
114
+ doc: Document,
115
+ adapter: Adapter,
116
+ plan_: PublishPlan,
117
+ ctx: Context,
118
+ run_id: str,
119
+ *,
120
+ publish: bool,
121
+ ) -> LegResult:
122
+ pf, did = adapter.name, doc.id
123
+ resume = self.state.resume_point(run_id, did, pf)
124
+
125
+ if not self.state.needs_update(did, pf, doc.content_id) and not publish:
126
+ log.info("%s/%s: content unchanged, nothing to do", pf, did)
127
+ rec = self.state.remote(did, pf)
128
+ return LegResult(did, pf, "skipped", url=rec.url if rec else None)
129
+
130
+ rec = self.state.remote(did, pf)
131
+ if rec:
132
+ ctx.options["remote_ref"] = rec.remote_ref
133
+
134
+ try:
135
+ order = Step.order()
136
+ start = order.index(resume)
137
+
138
+ for step in order[start:]:
139
+ if step is Step.PUBLISH and not publish:
140
+ break
141
+ self.state.record_step(run_id, did, pf, step, Status.RUNNING)
142
+
143
+ if step is Step.AUTH:
144
+ await with_retry(lambda: adapter.authenticate(ctx))
145
+
146
+ elif step is Step.DRAFT:
147
+ ref = await with_retry(lambda: adapter.ensure_draft(doc, ctx))
148
+ ctx.options["remote_ref"] = ref.id
149
+ self._ref = ref
150
+ self.state.remember_remote(did, pf, doc.content_id, ref.id, ref.url)
151
+
152
+ elif step is Step.CONTENT:
153
+ await with_retry(lambda: adapter.push_content(doc, plan_, self._ref, ctx))
154
+
155
+ elif step is Step.MEDIA:
156
+ await with_retry(lambda: adapter.push_media(doc, plan_, self._ref, ctx))
157
+
158
+ elif step is Step.VERIFY:
159
+ fp = await adapter.verify(doc, plan_, self._ref, ctx)
160
+ self.state.record_step(
161
+ run_id, did, pf, step, Status.DONE, content_id=doc.content_id, fingerprint=fp.__dict__
162
+ )
163
+ continue
164
+
165
+ elif step is Step.PUBLISH:
166
+ self._fire("pre_publish", doc=doc, platform=pf, plan=plan_)
167
+ pub: PublishedRef = await adapter.publish(doc, self._ref, ctx)
168
+ self.state.remember_remote(did, pf, doc.content_id, pub.id, pub.url, published=True)
169
+ self.state.record_step(run_id, did, pf, step, Status.DONE, remote_ref=pub.id)
170
+ self._fire("post_publish", doc=doc, platform=pf, url=pub.url)
171
+ return LegResult(did, pf, "published", url=pub.url,
172
+ resumed_from=resume.value if start else None)
173
+
174
+ self.state.record_step(run_id, did, pf, step, Status.DONE, content_id=doc.content_id)
175
+
176
+ rec = self.state.remote(did, pf)
177
+ return LegResult(did, pf, "drafted", url=rec.url if rec else None,
178
+ resumed_from=resume.value if start else None)
179
+
180
+ except VerificationFailed as exc:
181
+ # The most important failure to surface loudly: the remote does not
182
+ # contain what we think it does. Never proceed to publish.
183
+ self.state.record_step(run_id, did, pf, Step.VERIFY, Status.FAILED, error=str(exc))
184
+ self._fire("on_error", doc=doc, platform=pf, error=exc)
185
+ log.error("%s/%s: verification failed — NOT publishing: %s", pf, did, exc)
186
+ return LegResult(did, pf, "failed", error=f"verification: {exc}")
187
+
188
+ except (CredentialError, PermissionError) as exc:
189
+ # Expected, actionable, and not worth a stack trace: the message
190
+ # already tells the user exactly what to run.
191
+ self.state.record_step(run_id, did, pf, resume, Status.FAILED, error=str(exc))
192
+ self._fire("on_error", doc=doc, platform=pf, error=exc)
193
+ log.error("%s/%s: %s", pf, did, exc)
194
+ return LegResult(did, pf, "failed", error=str(exc))
195
+
196
+ except Exception as exc: # noqa: BLE001
197
+ self.state.record_step(run_id, did, pf, resume, Status.FAILED, error=str(exc))
198
+ self._fire("on_error", doc=doc, platform=pf, error=exc)
199
+ log.exception("%s/%s failed", pf, did)
200
+ return LegResult(did, pf, "failed", error=str(exc))
201
+
202
+ async def run(
203
+ self,
204
+ docs: Sequence[Document],
205
+ adapters: Sequence[Adapter],
206
+ ctx_for: Callable[[str], Context],
207
+ *,
208
+ publish: bool = False,
209
+ run_id: str | None = None,
210
+ ) -> RunReport:
211
+ plans = self.plan(docs, adapters)
212
+ blocking = [p for p in plans if not p.ok]
213
+ if blocking:
214
+ raise RuntimeError(
215
+ "plan is blocked:\n" + "\n".join(p.human() for p in blocking)
216
+ )
217
+
218
+ run_id = run_id or f"run-{int(time.time())}"
219
+ self.state.start_run(run_id, self.plan_hash(plans))
220
+ report = RunReport(run_id=run_id)
221
+
222
+ by_key = {(p.document_id, p.platform): p for p in plans}
223
+ # Sequential on purpose. Browser adapters share one browser, and
224
+ # hammering five platforms at once turns one rate-limit into five.
225
+ for doc in docs:
226
+ for adapter in adapters:
227
+ ctx = ctx_for(adapter.name)
228
+ leg = await self.run_leg(
229
+ doc, adapter, by_key[(doc.id, adapter.name)], ctx, run_id, publish=publish
230
+ )
231
+ report.legs.append(leg)
232
+
233
+ self.state.finish_run(run_id, Status.DONE if report.ok else Status.FAILED)
234
+ return report
235
+
236
+ async def run_series(
237
+ self,
238
+ series: Series,
239
+ adapters: Sequence[Adapter],
240
+ ctx_for: Callable[[str], Context],
241
+ *,
242
+ publish: bool = False,
243
+ ) -> RunReport:
244
+ """Two-phase publish, the only correct way to ship mutually-linked docs.
245
+
246
+ Phase 1 creates every draft and collects permalinks; phase 2 substitutes
247
+ them into the cross-links and pushes final content. Doing it in one pass
248
+ means the first document links to a URL that does not exist yet.
249
+ """
250
+ from .loader import resolve_series_links
251
+
252
+ log.info("phase 1/2: creating drafts to collect permalinks")
253
+ phase1 = await self.run(series.documents, adapters, ctx_for, publish=False)
254
+
255
+ for leg in phase1.legs:
256
+ if leg.url:
257
+ doc = next(d for d in series.documents if d.id == leg.document_id)
258
+ if doc.series:
259
+ for other in series.documents:
260
+ if other.series:
261
+ other.series.sibling_urls[doc.series.index] = leg.url
262
+
263
+ resolve_series_links(series)
264
+
265
+ log.info("phase 2/2: pushing final content with resolved cross-links")
266
+ phase2 = await self.run(series.documents, adapters, ctx_for, publish=publish)
267
+ phase2.legs = phase2.legs or phase1.legs
268
+ return phase2