pubkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pubkit/scaffold.py ADDED
@@ -0,0 +1,271 @@
1
+ # Copyright 2026 The pubkit Authors
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """`pubkit init` and `pubkit doctor`.
4
+
5
+ Adoption is mostly a first-five-minutes problem. `init` gives someone a content
6
+ repo that already validates and already has a working CI pipeline; `doctor`
7
+ answers "why isn't this working" without them having to read the source.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import importlib.util
12
+ import os
13
+ import platform
14
+ import sys
15
+ from dataclasses import dataclass
16
+ from pathlib import Path
17
+
18
+ SAMPLE_POST = '''---
19
+ id: hello-world
20
+ title: "The thing I learned the hard way"
21
+ subtitle: A one-line promise of what the reader gets.
22
+ tags: [Engineering, Writing]
23
+ canonical_url: ""
24
+ # Uncomment to have the build check your length:
25
+ # budget: {words: 1200, tolerance: 0.15}
26
+ ---
27
+
28
+ # What this is about
29
+
30
+ Open with the specific, surprising fact. Not a preamble — the fact.
31
+
32
+ Numbers you will reuse can be declared once and checked by the build:
33
+
34
+ <!-- pubkit:define fast = 900 -->
35
+ <!-- pubkit:define slow = 128 -->
36
+
37
+ The fast path is 7× wider than the slow one. <!-- pubkit:assert 7x = fast/slow ±10% -->
38
+
39
+ ## A section with evidence
40
+
41
+ | Approach | Throughput |
42
+ | :--- | ---: |
43
+ | Naive | 128 MB/s |
44
+ | Tuned | 900 MB/s |
45
+ *Measured on the same hardware, same dataset*
46
+
47
+ > [!key] The rule
48
+ > State the rule you want the reader to remember, once, in its own block.
49
+
50
+ ```python
51
+ def example() -> None:
52
+ print("code blocks survive to every platform that supports them")
53
+ ```
54
+
55
+ Close with what changes for the reader tomorrow.
56
+ '''
57
+
58
+ SAMPLE_SERIES_NOTE = '''---
59
+ id: part-1
60
+ title: "My Series, Part 1: The Setup"
61
+ subtitle: What part one establishes.
62
+ tags: [Engineering]
63
+ series: {id: my-series, index: 1, of: 2}
64
+ ---
65
+
66
+ # The first idea
67
+
68
+ Cross-links resolve themselves during a two-phase publish:
69
+
70
+ Continued in [Part 2](${series.part2.url}).
71
+ '''
72
+
73
+ CONFIG = """# pubkit.toml — optional. Everything here can also be a CLI flag.
74
+ [defaults]
75
+ platforms = ["medium", "devto"]
76
+
77
+ [medium]
78
+ # Medium has no tables; pubkit renders yours as styled images automatically.
79
+
80
+ [devto]
81
+ # Dev.to wants images at public URLs. Point this at wherever you host them.
82
+ # asset_base_url = "https://raw.githubusercontent.com/you/blog/main/"
83
+
84
+ [x]
85
+ mode = "promo" # "promo" (hook + claims + link) or "full"
86
+ max_posts = 12
87
+ """
88
+
89
+ CI = """name: publish
90
+ on:
91
+ pull_request:
92
+ paths: ["content/**"]
93
+ push:
94
+ branches: [main]
95
+ paths: ["content/**"]
96
+
97
+ jobs:
98
+ plan:
99
+ if: github.event_name == 'pull_request'
100
+ runs-on: ubuntu-latest
101
+ steps:
102
+ - uses: actions/checkout@v4
103
+ - uses: actions/setup-python@v5
104
+ with: {python-version: "3.12"}
105
+ - run: pip install "pubkit[all]"
106
+ - run: pubkit validate content/ --strict
107
+ - name: Plan
108
+ run: |
109
+ echo '## Publishing plan' >> $GITHUB_STEP_SUMMARY
110
+ echo '```' >> $GITHUB_STEP_SUMMARY
111
+ pubkit plan content/ --to medium,devto >> $GITHUB_STEP_SUMMARY
112
+ echo '```' >> $GITHUB_STEP_SUMMARY
113
+
114
+ publish:
115
+ if: github.ref == 'refs/heads/main'
116
+ runs-on: ubuntu-latest
117
+ environment: production # add a required reviewer here for a human gate
118
+ steps:
119
+ - uses: actions/checkout@v4
120
+ - uses: actions/setup-python@v5
121
+ with: {python-version: "3.12"}
122
+ - run: pip install "pubkit[all]"
123
+ - run: pubkit publish content/ --to devto --confirm
124
+ env:
125
+ PUBKIT_DEVTO_TOKEN: ${{ secrets.DEVTO_TOKEN }}
126
+ """
127
+
128
+ README = """# Content
129
+
130
+ Written once here, published everywhere by [pubkit](https://github.com/arunsingh/pubkit).
131
+
132
+ ```bash
133
+ pubkit validate content/ # check before anything leaves your laptop
134
+ pubkit plan content/ --to medium # see exactly what the platform will get
135
+ pubkit publish content/ --to medium # draft
136
+ pubkit publish content/ --to medium --confirm # live
137
+ ```
138
+
139
+ `.pubkit/` holds the state store and your encrypted sessions. It is gitignored,
140
+ and it is what makes a re-run resume instead of duplicating your drafts.
141
+ """
142
+
143
+
144
+ def init_repo(root: Path, *, series: bool = False, force: bool = False) -> list[Path]:
145
+ """Scaffold a content repository that validates on the first try."""
146
+ written: list[Path] = []
147
+
148
+ def write(rel: str, text: str) -> None:
149
+ p = root / rel
150
+ if p.exists() and not force:
151
+ return
152
+ p.parent.mkdir(parents=True, exist_ok=True)
153
+ p.write_text(text, encoding="utf-8")
154
+ written.append(p)
155
+
156
+ write("content/hello-world.md", SAMPLE_POST)
157
+ if series:
158
+ write("content/series/part-1.md", SAMPLE_SERIES_NOTE)
159
+ write(
160
+ "content/series/part-2.md",
161
+ SAMPLE_SERIES_NOTE.replace("part-1", "part-2")
162
+ .replace("index: 1", "index: 2")
163
+ .replace("Part 1: The Setup", "Part 2: The Payoff")
164
+ .replace("What part one establishes.", "What part two delivers.")
165
+ .replace("Continued in [Part 2](${series.part2.url}).", "Back to [Part 1](${series.part1.url})."),
166
+ )
167
+ write("content/img/.gitkeep", "")
168
+ write("pubkit.toml", CONFIG)
169
+ write(".github/workflows/publish.yml", CI)
170
+ write("README.md", README)
171
+ write(".gitignore", ".pubkit/\n__pycache__/\n")
172
+ return written
173
+
174
+
175
+ # ---------------------------------------------------------------------------
176
+ # doctor
177
+ # ---------------------------------------------------------------------------
178
+ @dataclass
179
+ class Finding:
180
+ ok: bool
181
+ label: str
182
+ detail: str
183
+ fix: str = ""
184
+
185
+
186
+ def _has(mod: str) -> bool:
187
+ return importlib.util.find_spec(mod) is not None
188
+
189
+
190
+ def diagnose() -> list[Finding]:
191
+ """Answer 'why isn't this working' before anyone has to read the source."""
192
+ out: list[Finding] = []
193
+
194
+ v = sys.version_info
195
+ out.append(
196
+ Finding(
197
+ v >= (3, 11),
198
+ "python",
199
+ f"{v.major}.{v.minor}.{v.micro} on {platform.system()}",
200
+ "pubkit needs Python 3.11+",
201
+ )
202
+ )
203
+
204
+ out.append(
205
+ Finding(
206
+ _has("playwright"),
207
+ "playwright",
208
+ "installed" if _has("playwright") else "missing",
209
+ 'pip install "pubkit[browser]" && playwright install chromium '
210
+ "— required for Medium and Substack only",
211
+ )
212
+ )
213
+
214
+ if _has("playwright"):
215
+ from pathlib import Path as _P
216
+
217
+ cache = _P.home() / (
218
+ "Library/Caches/ms-playwright" if platform.system() == "Darwin" else ".cache/ms-playwright"
219
+ )
220
+ found = cache.exists() and any(cache.glob("chromium*"))
221
+ out.append(
222
+ Finding(found, "chromium", "downloaded" if found else "not downloaded", "playwright install chromium")
223
+ )
224
+
225
+ out.append(
226
+ Finding(
227
+ _has("keyring"),
228
+ "keyring",
229
+ "available — tokens go to the OS keychain" if _has("keyring") else "missing",
230
+ 'pip install "pubkit[keychain]", or set PUBKIT_VAULT_KEY for an encrypted file vault',
231
+ )
232
+ )
233
+ out.append(
234
+ Finding(
235
+ _has("cryptography"),
236
+ "cryptography",
237
+ "available — sessions encrypted at rest" if _has("cryptography") else "missing",
238
+ 'pip install "pubkit[keychain]" — without it, saved sessions are stored in plain text',
239
+ )
240
+ )
241
+
242
+ state = Path(".pubkit/state.sqlite")
243
+ out.append(
244
+ Finding(
245
+ True,
246
+ "state store",
247
+ f"{state} ({state.stat().st_size} bytes)" if state.exists() else "not created yet (normal)",
248
+ )
249
+ )
250
+
251
+ from .core.auth import SessionStore, TokenStore
252
+ from .registry import list_adapters
253
+
254
+ tokens, sessions = TokenStore(), SessionStore()
255
+ for name, _ in list_adapters():
256
+ has_tok = bool(tokens.get(name)) or bool(tokens.get(name, "bearer"))
257
+ has_sess = sessions.exists(name)
258
+ out.append(
259
+ Finding(
260
+ has_tok or has_sess,
261
+ f"auth:{name}",
262
+ "token" if has_tok else "session" if has_sess else "not configured",
263
+ f"pubkit auth login {name}",
264
+ )
265
+ )
266
+
267
+ env = [k for k in os.environ if k.startswith("PUBKIT_")]
268
+ if env:
269
+ out.append(Finding(True, "env", ", ".join(sorted(env))))
270
+
271
+ return out
@@ -0,0 +1,2 @@
1
+ # Copyright 2026 The pubkit Authors
2
+ # SPDX-License-Identifier: Apache-2.0
@@ -0,0 +1,115 @@
1
+ # Copyright 2026 The pubkit Authors
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Airflow integration.
4
+
5
+ One task per (document, platform) rather than one task for the whole fan-out.
6
+ That is the entire trick: retries, alerting, SLA misses and partial reruns
7
+ become Airflow's problem, which it is much better at than a bespoke loop.
8
+
9
+ with DAG("blog", schedule="0 9 * * TUE") as dag:
10
+ validate = PubkitValidateOperator(task_id="validate", path="content/series/")
11
+ fanout = PubkitPublishOperator.expand_fanout(
12
+ path="content/series/",
13
+ platforms=["medium", "devto", "x"],
14
+ confirm=True,
15
+ )
16
+ validate >> fanout
17
+ """
18
+ from __future__ import annotations
19
+
20
+ import asyncio
21
+ from collections.abc import Sequence
22
+ from pathlib import Path
23
+ from typing import Any
24
+
25
+ try: # pragma: no cover - optional dependency
26
+ from airflow.exceptions import AirflowFailException
27
+ from airflow.models import BaseOperator
28
+ except Exception: # pragma: no cover
29
+ BaseOperator = object # type: ignore[assignment,misc]
30
+
31
+ class AirflowFailException(RuntimeError): # type: ignore[no-redef]
32
+ pass
33
+
34
+
35
+ class PubkitValidateOperator(BaseOperator): # type: ignore[misc]
36
+ """Run the check pipeline. Fails the DAG before anything reaches a platform."""
37
+
38
+ template_fields = ("path",)
39
+
40
+ def __init__(self, path: str, strict: bool = False, **kw: Any) -> None:
41
+ super().__init__(**kw)
42
+ self.path, self.strict = path, strict
43
+
44
+ def execute(self, context) -> dict:
45
+ from ..core.checks import run_checks
46
+ from ..core.loader import load_document, load_series
47
+
48
+ p = Path(self.path)
49
+ docs = load_series(sorted(p.glob("*.md"))).documents if p.is_dir() else [load_document(p)]
50
+ problems = []
51
+ for doc in docs:
52
+ res = run_checks(doc)
53
+ for f in res.findings:
54
+ self.log.info("%s", f)
55
+ if res.errors or (self.strict and res.findings):
56
+ problems.append(doc.id)
57
+ if problems:
58
+ raise AirflowFailException(f"checks failed for: {', '.join(problems)}")
59
+ return {"documents": [d.id for d in docs]}
60
+
61
+
62
+ class PubkitPublishOperator(BaseOperator): # type: ignore[misc]
63
+ """Publish one document to one platform.
64
+
65
+ Idempotent by construction: the state store means an Airflow retry resumes
66
+ at the failed step rather than starting over, and never creates a second
67
+ draft.
68
+ """
69
+
70
+ template_fields = ("path", "platform")
71
+
72
+ def __init__(
73
+ self,
74
+ path: str,
75
+ platform: str,
76
+ document_id: str | None = None,
77
+ confirm: bool = False,
78
+ state_path: str = ".pubkit/state.sqlite",
79
+ **kw: Any,
80
+ ) -> None:
81
+ super().__init__(**kw)
82
+ self.path, self.platform = path, platform
83
+ self.document_id, self.confirm, self.state_path = document_id, confirm, state_path
84
+
85
+ def execute(self, context) -> dict:
86
+ from ..core.adapter import Context
87
+ from ..core.auth import SessionStore, TokenStore
88
+ from ..core.loader import load_document, load_series
89
+ from ..core.runner import Pipeline
90
+ from ..core.state import StateStore
91
+ from ..registry import build_adapter
92
+
93
+ p = Path(self.path)
94
+ docs = load_series(sorted(p.glob("*.md"))).documents if p.is_dir() else [load_document(p)]
95
+ if self.document_id:
96
+ docs = [d for d in docs if d.id == self.document_id]
97
+
98
+ tokens, sessions = TokenStore(), SessionStore()
99
+ adapter = build_adapter(self.platform)
100
+ pipe = Pipeline(StateStore(self.state_path))
101
+
102
+ def ctx_for(_name: str) -> Context:
103
+ return Context(platform=self.platform, tokens=tokens, sessions=sessions, confirm=self.confirm)
104
+
105
+ # A stable run_id per DAG run is what makes an Airflow retry resume
106
+ # instead of restart.
107
+ run_id = f"airflow-{context['run_id']}"
108
+ report = asyncio.run(pipe.run(docs, [adapter], ctx_for, publish=self.confirm, run_id=run_id))
109
+ if not report.ok:
110
+ raise AirflowFailException(report.human())
111
+ return {"legs": [leg.__dict__ for leg in report.legs]}
112
+
113
+ @classmethod
114
+ def expand_fanout(cls, path: str, platforms: Sequence[str], **kw: Any) -> list:
115
+ return [cls(task_id=f"publish_{p}", path=path, platform=p, **kw) for p in platforms]
@@ -0,0 +1,291 @@
1
+ Metadata-Version: 2.5
2
+ Name: pubkit
3
+ Version: 0.1.0
4
+ Summary: Publish one source to many platforms, safely. Medium, Substack, X, Dev.to, Hashnode.
5
+ Project-URL: Homepage, https://github.com/arunsingh/pubkit
6
+ Project-URL: Issues, https://github.com/arunsingh/pubkit/issues
7
+ Project-URL: Documentation, https://github.com/arunsingh/pubkit#readme
8
+ Project-URL: Changelog, https://github.com/arunsingh/pubkit/blob/main/CHANGELOG.md
9
+ Project-URL: Failure modes, https://github.com/arunsingh/pubkit/blob/main/docs/FAILURE-MODES.md
10
+ Author: Arun Singh
11
+ License-Expression: Apache-2.0
12
+ License-File: LICENSE
13
+ License-File: NOTICE
14
+ Keywords: automation,content,devto,hashnode,medium,publishing,substack,twitter,x
15
+ Classifier: Development Status :: 4 - Beta
16
+ Classifier: Intended Audience :: Developers
17
+ Classifier: License :: OSI Approved :: Apache Software License
18
+ Classifier: Operating System :: OS Independent
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Topic :: Internet :: WWW/HTTP :: Dynamic Content
22
+ Classifier: Topic :: Text Processing :: Markup
23
+ Classifier: Typing :: Typed
24
+ Requires-Python: >=3.11
25
+ Requires-Dist: httpx>=0.27
26
+ Requires-Dist: pydantic>=2.6
27
+ Requires-Dist: pyyaml>=6.0
28
+ Requires-Dist: rich>=13.7
29
+ Requires-Dist: typer>=0.12
30
+ Provides-Extra: airflow
31
+ Requires-Dist: apache-airflow>=2.9; extra == 'airflow'
32
+ Provides-Extra: all
33
+ Requires-Dist: cryptography>=42.0; extra == 'all'
34
+ Requires-Dist: keyring>=25.0; extra == 'all'
35
+ Requires-Dist: pillow>=10.0; extra == 'all'
36
+ Requires-Dist: playwright>=1.44; extra == 'all'
37
+ Provides-Extra: browser
38
+ Requires-Dist: playwright>=1.44; extra == 'browser'
39
+ Provides-Extra: dev
40
+ Requires-Dist: mypy>=1.10; extra == 'dev'
41
+ Requires-Dist: pytest-asyncio>=0.23; extra == 'dev'
42
+ Requires-Dist: pytest>=8.0; extra == 'dev'
43
+ Requires-Dist: ruff>=0.4; extra == 'dev'
44
+ Provides-Extra: keychain
45
+ Requires-Dist: cryptography>=42.0; extra == 'keychain'
46
+ Requires-Dist: keyring>=25.0; extra == 'keychain'
47
+ Provides-Extra: render
48
+ Requires-Dist: pillow>=10.0; extra == 'render'
49
+ Description-Content-Type: text/markdown
50
+
51
+ # pubkit
52
+
53
+ **Publish one source to many platforms, safely.**
54
+ Medium · Substack · X · Dev.to · Hashnode — and whatever you add next.
55
+
56
+ [![PyPI](https://img.shields.io/pypi/v/pubkit.svg)](https://pypi.org/project/pubkit/)
57
+ [![Python](https://img.shields.io/pypi/pyversions/pubkit.svg)](https://pypi.org/project/pubkit/)
58
+ [![License](https://img.shields.io/badge/license-Apache%202.0-blue.svg)](LICENSE)
59
+ [![CI](https://github.com/arunsingh/pubkit/actions/workflows/ci.yml/badge.svg)](https://github.com/arunsingh/pubkit/actions/workflows/ci.yml)
60
+
61
+ ---
62
+
63
+ ## Why this exists
64
+
65
+ I published a 15,000-word, three-part illustrated series to Medium. It took far
66
+ longer than writing it, and not for any reason I could have predicted.
67
+
68
+ A 12,368-character payload arrived with exactly **one byte** changed, and
69
+ nothing reported an error. Pasting a document replaced 2 of 169 paragraphs and
70
+ silently appended the rest, producing a scrambled duplicate. A delete that
71
+ visibly worked was back after a reload. Link hrefs set in the DOM reverted every
72
+ time. Every image — `data:` URI and `https://` URL alike — was stripped out of
73
+ pasted HTML, and it took eight dead ends to find the one insertion path that
74
+ works. A scripted check of 32 numeric claims found 5 that were wrong, including
75
+ two ratios transposed in an edit that no amount of re-reading had caught. And
76
+ the automation bridge dropped eight times, so everything had to be re-runnable.
77
+
78
+ None of that is Medium being unusual. It is what automating *any* rich-text
79
+ editor looks like. pubkit is that knowledge, extracted and made reusable.
80
+
81
+ [`docs/FAILURE-MODES.md`](docs/FAILURE-MODES.md) catalogues all of it — every
82
+ entry names the component that answers it.
83
+
84
+ ## Install
85
+
86
+ ```bash
87
+ pip install "pubkit[all]"
88
+ playwright install chromium # Medium and Substack only
89
+ pubkit doctor # confirms your environment before you start
90
+ ```
91
+
92
+ Thirty seconds to something real:
93
+
94
+ ```bash
95
+ pubkit init my-blog && cd my-blog
96
+ pubkit validate content/
97
+ pubkit plan content/ --to medium,devto
98
+ ```
99
+
100
+ ## Use it
101
+
102
+ ```bash
103
+ # 0. Scaffold a content repo that validates on the first try.
104
+ pubkit init my-blog && cd my-blog
105
+
106
+ # 1. Check the content. No network, no browser, no side effects.
107
+ pubkit validate content/series/
108
+
109
+ # 2. See exactly what each platform will get — including every degradation.
110
+ pubkit plan content/series/ --to medium,devto,x
111
+
112
+ # 3. Sign in. pubkit never accepts a password.
113
+ pubkit auth login medium # opens a window; you sign in; it keeps the session
114
+ pubkit auth login devto # API token → OS keychain
115
+
116
+ # 4. Drafts everywhere. Safe to re-run; a dropped connection costs one step.
117
+ pubkit publish content/series/ --to medium,devto,x
118
+
119
+ # 5. Go live.
120
+ pubkit publish content/series/ --to medium,devto,x --confirm
121
+ ```
122
+
123
+ `plan` output on a real series:
124
+
125
+ ```
126
+ medium: part-1 (bf9abfe78ce4)
127
+ · images_via_browser 4 image(s) uploaded through the editor's own paste path
128
+ — remote URLs and data: URIs are stripped by this platform
129
+ · code_flattened 4 code block(s) lose syntax highlighting
130
+
131
+ devto: part-1 (bf9abfe78ce4)
132
+ · tags_trimmed keeping 4 of 5
133
+
134
+ x: part-1 (bf9abfe78ce4)
135
+ · split_into_thread promo thread: 12 posts — hook, key claims, link to the full article
136
+ · tags_trimmed 5 tag(s) dropped — platform has no tags
137
+ ```
138
+
139
+ Nothing is a surprise at publish time. That is the whole design goal.
140
+
141
+ ## Write once
142
+
143
+ ````markdown
144
+ ---
145
+ id: part-1
146
+ title: "Inside AI Infrastructure, Part 1: The Hardware"
147
+ subtitle: Why a GPU is fast, and why it turns out to be a memory problem.
148
+ tags: [AI Infrastructure, GPU, SRE]
149
+ series: {id: inside-ai, index: 1, of: 3}
150
+ budget: {words: 3500}
151
+ ---
152
+
153
+ # What is behind the box you type into
154
+
155
+ <!-- pubkit:define nvlink = 900 -->
156
+ <!-- pubkit:define pcie5 = 128 -->
157
+
158
+ NVLink is 7× wider than PCIe Gen5. <!-- pubkit:assert 7x = nvlink/pcie5 ±10% -->
159
+
160
+ | Model | FP16 size |
161
+ | :--- | ---: |
162
+ | 70B | 140 GB |
163
+ *What the weights actually cost*
164
+
165
+ ![memory layout](img/vram.gif)
166
+ *Every token costs one full read of the model out of VRAM.*
167
+
168
+ Continued in [Part 2](${series.part2.url}).
169
+ ````
170
+
171
+ That one file becomes: a Medium draft with the table rendered as a styled image
172
+ and the GIF uploaded through the editor's own path; a Dev.to article with a
173
+ native table and a fenced code block; a 12-post promo thread on X with a link
174
+ back. The cross-link resolves itself during a two-phase publish.
175
+
176
+ ## What it actually guards against
177
+
178
+ | Guard | What it prevents |
179
+ |---|---|
180
+ | Hash-verified chunked transport | a silently corrupted payload |
181
+ | Probed chunk ceiling | undocumented per-call size limits |
182
+ | `replace_document()` with a block-count assert | a paste that appends instead of replacing |
183
+ | Reload-then-fingerprint verification | edits the editor showed you but never saved |
184
+ | Caption-anchored `File` paste | images stripped out of pasted HTML |
185
+ | Unicode-folding anchor resolution | an em dash breaking a text match |
186
+ | Tag-chip post-condition | one 40-character invalid tag instead of five good ones |
187
+ | `pubkit:assert` arithmetic | a transposed ratio reaching print |
188
+ | Two-phase series publish | `URL-PART-2` going live as a link |
189
+ | Content-addressed state store | a re-run creating a second draft |
190
+ | Plan hash + `--confirm` | publishing something other than what you reviewed |
191
+
192
+ ## Architecture in one paragraph
193
+
194
+ A canonical **Post IR** sits between your source and every platform. Adapters
195
+ declare **capabilities**; the planner intersects the two and emits explicit
196
+ **degradations**. A **runner** executes six steps per (document, platform) —
197
+ auth, draft, content, media, verify, publish — recording each in a sqlite state
198
+ store so a re-run resumes rather than restarts. Browser adapters inherit a DOM
199
+ toolkit that encodes everything above; API adapters inherit rate limiting and
200
+ retries. [`docs/ARCHITECTURE.md`](docs/ARCHITECTURE.md) has the diagram.
201
+
202
+ ## Add a platform
203
+
204
+ ```python
205
+ from pubkit.core.browser import BrowserAdapter, EditorSelectors
206
+ from pubkit.core.capabilities import Capabilities
207
+
208
+ class GhostAdapter(BrowserAdapter):
209
+ name = "ghost"
210
+ capabilities = Capabilities(tables=True, image_upload="browser_paste")
211
+ selectors = EditorSelectors(
212
+ editable=".koenig-editor__editor",
213
+ content_roots=".koenig-editor__editor", # ALL roots, not the first
214
+ block="[data-kg='editor'] > *",
215
+ figure="figure", figure_img="figure img", link="a",
216
+ publish_button=".gh-publishmenu-trigger",
217
+ )
218
+ async def ensure_draft(self, doc, ctx): ...
219
+ async def publish(self, doc, ref, ctx): ...
220
+ ```
221
+
222
+ Register it from your own package — no fork required:
223
+
224
+ ```toml
225
+ [project.entry-points."pubkit.adapters"]
226
+ ghost = "my_pubkit_ghost:GhostAdapter"
227
+ ```
228
+
229
+ Five extension points, all first-class: `pubkit.adapters`, `pubkit.renderers`,
230
+ `pubkit.checks`, `pubkit.hooks`, and per-platform `transforms:` in config.
231
+
232
+ ## Fit it into a workflow
233
+
234
+ **GitHub Actions** — plan on every PR, publish on merge:
235
+
236
+ ```yaml
237
+ - run: pubkit validate content/ --strict
238
+ - run: pubkit plan content/ --to medium,devto >> $GITHUB_STEP_SUMMARY
239
+ - run: pubkit publish content/ --to medium,devto --confirm
240
+ if: github.ref == 'refs/heads/main'
241
+ env:
242
+ PUBKIT_VAULT_KEY: ${{ secrets.PUBKIT_VAULT_KEY }}
243
+ ```
244
+
245
+ **Airflow** — one task per (document, platform), so retries and alerting belong
246
+ to Airflow:
247
+
248
+ ```python
249
+ validate = PubkitValidateOperator(task_id="validate", path="content/series/")
250
+ validate >> PubkitPublishOperator.expand_fanout(
251
+ path="content/series/", platforms=["medium", "devto", "x"], confirm=True,
252
+ )
253
+ ```
254
+
255
+ **Library** — `from pubkit import Pipeline` for everything else.
256
+
257
+ ## Security
258
+
259
+ - **pubkit never accepts a password.** Browser platforms use an interactive
260
+ login you perform; pubkit persists only the session, encrypted at rest.
261
+ - API tokens live in the OS keychain, or an encrypted vault for CI.
262
+ - Secrets are stripped from logs by a filter, not by discipline.
263
+ - Adapters receive credentials for their own platform and nothing else.
264
+ - Every public write needs `--confirm`, against a hash-pinned plan.
265
+
266
+ ## Status
267
+
268
+ v0.1. The core, the checks, the planner and the state machine are tested (32
269
+ tests) and exercised end-to-end against a real published series in
270
+ `examples/inside-ai-infra/`. The browser adapters encode techniques verified by
271
+ hand against live editors; selectors are the part most likely to need a patch
272
+ when a platform ships a redesign, which is exactly why they are isolated in one
273
+ dataclass per adapter.
274
+
275
+ ## Docs
276
+
277
+ | | |
278
+ |---|---|
279
+ | [QUICKSTART.md](docs/QUICKSTART.md) | five minutes, one platform |
280
+ | [FAILURE-MODES.md](docs/FAILURE-MODES.md) | the 21 ways publishing quietly goes wrong |
281
+ | [ARCHITECTURE.md](docs/ARCHITECTURE.md) | how it fits together |
282
+ | [ADAPTERS.md](docs/ADAPTERS.md) | add a platform |
283
+ | [SECURITY.md](SECURITY.md) | threat model and credential handling |
284
+
285
+ Contributions welcome — especially new adapters and new checks. The one hard
286
+ rule: a bug fix adds a row to `docs/FAILURE-MODES.md` and a test named after the
287
+ failure. See [CONTRIBUTING.md](CONTRIBUTING.md).
288
+
289
+ ## Licence
290
+
291
+ Apache 2.0. See [LICENSE](LICENSE) and [NOTICE](NOTICE).