pubkit 0.1.0__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. pubkit-0.2.0/CHANGELOG.md +97 -0
  2. {pubkit-0.1.0 → pubkit-0.2.0}/PKG-INFO +51 -3
  3. {pubkit-0.1.0 → pubkit-0.2.0}/README.md +50 -2
  4. {pubkit-0.1.0 → pubkit-0.2.0}/pyproject.toml +1 -1
  5. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/__init__.py +1 -1
  6. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/adapters/medium.py +1 -1
  7. pubkit-0.2.0/src/pubkit/browserctl.py +213 -0
  8. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/cli.py +38 -9
  9. pubkit-0.2.0/src/pubkit/core/assets.py +96 -0
  10. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/browser.py +1 -1
  11. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/runner.py +7 -0
  12. pubkit-0.2.0/src/pubkit/render/tables.py +199 -0
  13. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/scaffold.py +27 -7
  14. pubkit-0.2.0/tests/fixtures/fake_editor.html +119 -0
  15. pubkit-0.2.0/tests/test_browser.py +182 -0
  16. {pubkit-0.1.0 → pubkit-0.2.0}/tests/test_core.py +129 -0
  17. pubkit-0.1.0/CHANGELOG.md +0 -45
  18. pubkit-0.1.0/src/pubkit/browserctl.py +0 -78
  19. {pubkit-0.1.0 → pubkit-0.2.0}/.gitignore +0 -0
  20. {pubkit-0.1.0 → pubkit-0.2.0}/LICENSE +0 -0
  21. {pubkit-0.1.0 → pubkit-0.2.0}/NOTICE +0 -0
  22. {pubkit-0.1.0 → pubkit-0.2.0}/docs/ADAPTERS.md +0 -0
  23. {pubkit-0.1.0 → pubkit-0.2.0}/docs/ARCHITECTURE.md +0 -0
  24. {pubkit-0.1.0 → pubkit-0.2.0}/docs/FAILURE-MODES.md +0 -0
  25. {pubkit-0.1.0 → pubkit-0.2.0}/docs/QUICKSTART.md +0 -0
  26. {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part1-01-model-sizes.png +0 -0
  27. {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part1-02-cpu-vs-gpu-ANIMATED.gif +0 -0
  28. {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part1-03-starving-cores-ANIMATED.gif +0 -0
  29. {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part1-04-gpu-comparison.png +0 -0
  30. {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part2-01-one-token-one-read-ANIMATED.gif +0 -0
  31. {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part2-02-decode-ceilings.png +0 -0
  32. {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part2-03-prefill-vs-decode.png +0 -0
  33. {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part2-04-kv-cache-cost.png +0 -0
  34. {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part2-05-batching-ANIMATED.gif +0 -0
  35. {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part2-06-vram-budget.png +0 -0
  36. {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part2-07-throughput-latency.png +0 -0
  37. {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part3-01-interconnect.png +0 -0
  38. {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part3-02-llmd-request-path.png +0 -0
  39. {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part3-03-llmd-objects.png +0 -0
  40. {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part3-04-slos.png +0 -0
  41. {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/part-1.md +0 -0
  42. {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/part-2.md +0 -0
  43. {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/part-3.md +0 -0
  44. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/__main__.py +0 -0
  45. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/adapters/__init__.py +0 -0
  46. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/adapters/api_base.py +0 -0
  47. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/adapters/devto.py +0 -0
  48. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/adapters/substack.py +0 -0
  49. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/adapters/x.py +0 -0
  50. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/__init__.py +0 -0
  51. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/adapter.py +0 -0
  52. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/anchors.py +0 -0
  53. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/auth.py +0 -0
  54. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/capabilities.py +0 -0
  55. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/checks.py +0 -0
  56. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/ir.py +0 -0
  57. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/loader.py +0 -0
  58. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/state.py +0 -0
  59. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/transport.py +0 -0
  60. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/py.typed +0 -0
  61. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/registry.py +0 -0
  62. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/render/__init__.py +0 -0
  63. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/render/html.py +0 -0
  64. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/workflows/__init__.py +0 -0
  65. {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/workflows/airflow.py +0 -0
@@ -0,0 +1,97 @@
1
+ # Changelog
2
+
3
+ Format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
4
+ Versioning is [semantic](https://semver.org/); adapters may change behaviour on
5
+ a minor bump when a platform changes underneath them.
6
+
7
+ ## [0.2.0] — 2026-09-11
8
+
9
+ The release that makes `pubkit publish --to medium --confirm` actually work
10
+ unattended. 0.1.x could plan a Medium publish; it could not perform one.
11
+
12
+ ### Added
13
+
14
+ - **Browser lifecycle** (`browserctl.BrowserPool`, `attached()`). This was the
15
+ missing seam: the runner called `adapter.authenticate()` on a `MediumAdapter`
16
+ whose `_page` was still `None`. Now one browser serves the whole run, one
17
+ context per platform so a Medium session is never presented to Substack, and
18
+ sessions are refreshed on the way out because cookies rotate. API-only runs
19
+ never launch Chromium at all.
20
+ - **Table rendering** (`render.tables`, `core.assets.materialise`). The planner
21
+ said `table_to_image` and listed the asset ids; nothing produced the files.
22
+ Tables now render to quantised PNGs at 2x, cached on cell content so a re-run
23
+ neither re-renders nor re-uploads an unchanged table. The source IR is never
24
+ mutated — the same document keeps native tables on Dev.to in the same run.
25
+ - `pubkit auth verify <platform>` — check a session before a publish depends on
26
+ it, rather than discovering it expired halfway through.
27
+ - **Browser integration tests** against a deliberately hostile fixture editor
28
+ that reproduces multiple content roots, image stripping, figures landing
29
+ above the caret, and uploads that sit on a `blob:` URL before resolving.
30
+ They skip cleanly where Chromium is unavailable.
31
+
32
+ ### Fixed
33
+
34
+ - `marker_regex` was double-escaped, so every `fingerprint()` call threw
35
+ `SyntaxError: Invalid regular expression` inside the page. Verification was
36
+ therefore broken on every browser adapter. Found by the new integration
37
+ tests on their first run.
38
+ - `auth login` routed on a capability flag rather than the platform list, so
39
+ `pubkit auth login substack` took the API-token path and prompted for a token
40
+ that does not exist.
41
+
42
+ ## [0.1.1] — 2026-09-11
43
+
44
+ ### Added
45
+
46
+ - Docker image on the Playwright base (`ghcr.io/arunsingh/pubkit`), so browser
47
+ adapters work in CI without hand-installing Chromium's system libraries.
48
+ - `action.yml` — the repository is now a reusable GitHub Action.
49
+ - `CITATION.cff`, and a Homebrew formula template under `packaging/`.
50
+
51
+ ### Fixed
52
+
53
+ - `pubkit doctor` ignored `PLAYWRIGHT_BROWSERS_PATH` and reported Chromium as
54
+ missing on managed images (CI runners, devcontainers, the Playwright Docker
55
+ image) where it was in fact installed — sending people off to fix a problem
56
+ they did not have. Found by running the published package in exactly such an
57
+ environment.
58
+
59
+ ## [0.1.0] — 2026-09-11
60
+
61
+ First release. Extracted from a real, painful publishing run: a 15,000-word
62
+ illustrated three-part series to Medium that surfaced 21 distinct failure modes,
63
+ all catalogued in `docs/FAILURE-MODES.md`.
64
+
65
+ ### Added
66
+
67
+ - **Post IR** — a canonical, platform-agnostic document model with a
68
+ content-addressed id used as the idempotency key everywhere.
69
+ - **Capability negotiation** — adapters declare what they support; the planner
70
+ emits explicit, printable degradations. `pubkit plan` shows them before
71
+ anything is sent.
72
+ - **Check pipeline** — placeholders, numeric assertions, cross-references,
73
+ number drift, assets, budget, structure. Pluggable via `pubkit.checks`.
74
+ - **Verified chunked transport** — per-chunk rolling-hash verification, probed
75
+ size ceilings, single-member gzip.
76
+ - **Browser adapter toolkit** — paste-based document replace with a block-count
77
+ assertion, reload-then-fingerprint verification, caption-anchored `File`
78
+ image insertion, Unicode-folding anchor resolution, tag-chip post-conditions.
79
+ - **Adapters** — Medium, Substack (browser); X, Dev.to, Hashnode (API).
80
+ - **Runner** — six resumable steps per (document, platform), sqlite state store,
81
+ two-phase publish for series, hash-pinned `--confirm` gate, per-platform rate
82
+ limiting and jittered retry.
83
+ - **CLI** — `init`, `validate`, `plan`, `publish`, `status`, `platforms`,
84
+ `doctor`, `auth login|logout|list`.
85
+ - **Workflows** — GitHub Actions examples and Airflow operators.
86
+ - **Extension points** — `pubkit.adapters`, `pubkit.renderers`, `pubkit.checks`,
87
+ `pubkit.hooks`, plus per-platform transforms.
88
+
89
+ ### Known limitations
90
+
91
+ - Browser adapter selectors will need patching when a platform redesigns its
92
+ editor. They are isolated in one `EditorSelectors` dataclass per adapter
93
+ precisely so that is a small change.
94
+ - Table→image rendering requires the `render` extra; without it, tables degrade
95
+ to lists on platforms that lack table support.
96
+ - X media upload uses the v1.1 endpoint, which is what the v2 API still
97
+ requires.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: pubkit
3
- Version: 0.1.0
3
+ Version: 0.2.0
4
4
  Summary: Publish one source to many platforms, safely. Medium, Substack, X, Dev.to, Hashnode.
5
5
  Project-URL: Homepage, https://github.com/arunsingh/pubkit
6
6
  Project-URL: Issues, https://github.com/arunsingh/pubkit/issues
@@ -109,9 +109,10 @@ pubkit validate content/series/
109
109
  # 2. See exactly what each platform will get — including every degradation.
110
110
  pubkit plan content/series/ --to medium,devto,x
111
111
 
112
- # 3. Sign in. pubkit never accepts a password.
113
- pubkit auth login medium # opens a window; you sign in; it keeps the session
112
+ # 3. Sign in, once. pubkit never accepts a password.
113
+ pubkit auth login medium # a window opens; you sign in; it keeps the session
114
114
  pubkit auth login devto # API token → OS keychain
115
+ pubkit auth verify medium # confirm before a publish depends on it
115
116
 
116
117
  # 4. Drafts everywhere. Safe to re-run; a dropped connection costs one step.
117
118
  pubkit publish content/series/ --to medium,devto,x
@@ -138,6 +139,31 @@ x: part-1 (bf9abfe78ce4)
138
139
 
139
140
  Nothing is a surprise at publish time. That is the whole design goal.
140
141
 
142
+ ## Sign in once, then it runs on its own
143
+
144
+ ```bash
145
+ pubkit auth login medium
146
+ ```
147
+
148
+ A real browser window opens at Medium's login page. You sign in — password
149
+ manager, MFA, device confirmation, whatever it asks for. pubkit watches for the
150
+ post-login URL, saves the session encrypted, and closes the window.
151
+
152
+ From then on, unattended:
153
+
154
+ ```bash
155
+ pubkit publish content/ --to medium,devto,x --confirm
156
+ ```
157
+
158
+ One browser serves the whole run, one context per platform so a Medium session
159
+ is never presented to Substack, and sessions refresh on the way out because
160
+ cookies rotate. Publishing only to API platforms never launches Chromium at all.
161
+
162
+ pubkit sees cookies. It never sees, types or stores a password — which is both
163
+ the right security posture and the only thing that works, since platforms
164
+ increasingly gate login behind challenges an automation layer has no business
165
+ trying to defeat.
166
+
141
167
  ## Write once
142
168
 
143
169
  ````markdown
@@ -229,6 +255,28 @@ ghost = "my_pubkit_ghost:GhostAdapter"
229
255
  Five extension points, all first-class: `pubkit.adapters`, `pubkit.renderers`,
230
256
  `pubkit.checks`, `pubkit.hooks`, and per-platform `transforms:` in config.
231
257
 
258
+ ## Install it however you like
259
+
260
+ ```bash
261
+ pip install "pubkit[all]" # the normal way
262
+ pipx install "pubkit[all]" # isolated CLI
263
+ uv tool install "pubkit[all]" # same, faster
264
+ docker run --rm -v "$PWD:/work" ghcr.io/arunsingh/pubkit validate content/
265
+ ```
266
+
267
+ The Docker image ships Chromium and its system libraries, which is the part
268
+ nobody wants to install on a CI runner by hand.
269
+
270
+ As a GitHub Action, no install step at all:
271
+
272
+ ```yaml
273
+ - uses: arunsingh/pubkit@v1
274
+ with:
275
+ command: plan
276
+ path: content/
277
+ platforms: medium,devto
278
+ ```
279
+
232
280
  ## Fit it into a workflow
233
281
 
234
282
  **GitHub Actions** — plan on every PR, publish on merge:
@@ -59,9 +59,10 @@ pubkit validate content/series/
59
59
  # 2. See exactly what each platform will get — including every degradation.
60
60
  pubkit plan content/series/ --to medium,devto,x
61
61
 
62
- # 3. Sign in. pubkit never accepts a password.
63
- pubkit auth login medium # opens a window; you sign in; it keeps the session
62
+ # 3. Sign in, once. pubkit never accepts a password.
63
+ pubkit auth login medium # a window opens; you sign in; it keeps the session
64
64
  pubkit auth login devto # API token → OS keychain
65
+ pubkit auth verify medium # confirm before a publish depends on it
65
66
 
66
67
  # 4. Drafts everywhere. Safe to re-run; a dropped connection costs one step.
67
68
  pubkit publish content/series/ --to medium,devto,x
@@ -88,6 +89,31 @@ x: part-1 (bf9abfe78ce4)
88
89
 
89
90
  Nothing is a surprise at publish time. That is the whole design goal.
90
91
 
92
+ ## Sign in once, then it runs on its own
93
+
94
+ ```bash
95
+ pubkit auth login medium
96
+ ```
97
+
98
+ A real browser window opens at Medium's login page. You sign in — password
99
+ manager, MFA, device confirmation, whatever it asks for. pubkit watches for the
100
+ post-login URL, saves the session encrypted, and closes the window.
101
+
102
+ From then on, unattended:
103
+
104
+ ```bash
105
+ pubkit publish content/ --to medium,devto,x --confirm
106
+ ```
107
+
108
+ One browser serves the whole run, one context per platform so a Medium session
109
+ is never presented to Substack, and sessions refresh on the way out because
110
+ cookies rotate. Publishing only to API platforms never launches Chromium at all.
111
+
112
+ pubkit sees cookies. It never sees, types or stores a password — which is both
113
+ the right security posture and the only thing that works, since platforms
114
+ increasingly gate login behind challenges an automation layer has no business
115
+ trying to defeat.
116
+
91
117
  ## Write once
92
118
 
93
119
  ````markdown
@@ -179,6 +205,28 @@ ghost = "my_pubkit_ghost:GhostAdapter"
179
205
  Five extension points, all first-class: `pubkit.adapters`, `pubkit.renderers`,
180
206
  `pubkit.checks`, `pubkit.hooks`, and per-platform `transforms:` in config.
181
207
 
208
+ ## Install it however you like
209
+
210
+ ```bash
211
+ pip install "pubkit[all]" # the normal way
212
+ pipx install "pubkit[all]" # isolated CLI
213
+ uv tool install "pubkit[all]" # same, faster
214
+ docker run --rm -v "$PWD:/work" ghcr.io/arunsingh/pubkit validate content/
215
+ ```
216
+
217
+ The Docker image ships Chromium and its system libraries, which is the part
218
+ nobody wants to install on a CI runner by hand.
219
+
220
+ As a GitHub Action, no install step at all:
221
+
222
+ ```yaml
223
+ - uses: arunsingh/pubkit@v1
224
+ with:
225
+ command: plan
226
+ path: content/
227
+ platforms: medium,devto
228
+ ```
229
+
182
230
  ## Fit it into a workflow
183
231
 
184
232
  **GitHub Actions** — plan on every PR, publish on merge:
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "pubkit"
7
- version = "0.1.0"
7
+ version = "0.2.0"
8
8
  description = "Publish one source to many platforms, safely. Medium, Substack, X, Dev.to, Hashnode."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.11"
@@ -6,7 +6,7 @@ from .core.loader import load_document, load_series # noqa: F401
6
6
  from .core.runner import Pipeline, RunReport # noqa: F401
7
7
  from .registry import build_adapter, list_adapters, register # noqa: F401
8
8
 
9
- __version__ = "0.1.0"
9
+ __version__ = "0.2.0"
10
10
  __all__ = [
11
11
  "Document", "Series", "Asset", "Figure", "Table", "Heading", "Paragraph", "Code",
12
12
  "Pipeline", "RunReport", "load_document", "load_series",
@@ -68,7 +68,7 @@ class MediumAdapter(BrowserAdapter):
68
68
 
69
69
  login_url = "https://medium.com/m/signin"
70
70
  new_story_url = "https://medium.com/new-story"
71
- marker_regex = r"\\[\\[\\s*IMAGE"
71
+ marker_regex = r"\[\[\s*IMAGE"
72
72
 
73
73
  def __init__(self, page=None, upload=None) -> None:
74
74
  super().__init__()
@@ -0,0 +1,213 @@
1
+ # Copyright 2026 The pubkit Authors
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Browser lifecycle.
4
+
5
+ Browser adapters need a live Playwright page and a way to hand real bytes to a
6
+ file input. Nothing else in the pipeline should have to know that, so this
7
+ module owns it: one browser for the whole run, one context per platform, and
8
+ sessions refreshed on the way out.
9
+
10
+ The login flow is deliberately human-in-the-loop:
11
+
12
+ pubkit auth login medium
13
+ → a *visible* window opens at the platform's login page
14
+ → you sign in: password manager, MFA, device confirmation, whatever
15
+ → pubkit waits until it sees you are through, saves the session, closes
16
+
17
+ pubkit sees cookies. It never sees, types or stores a password. That is not
18
+ only the right security posture, it is the only thing that works: platforms
19
+ increasingly gate login behind challenges an automation layer has no business
20
+ trying to defeat.
21
+ """
22
+ from __future__ import annotations
23
+
24
+ import asyncio
25
+ import logging
26
+ from collections.abc import Sequence
27
+ from contextlib import AsyncExitStack, asynccontextmanager
28
+
29
+ from .core.auth import CredentialError, SessionStore
30
+
31
+ log = logging.getLogger(__name__)
32
+
33
+ #: (where to send the user, how to tell they made it)
34
+ LOGIN_FLOWS: dict[str, tuple[str, str]] = {
35
+ "medium": ("https://medium.com/m/signin", "medium.com/me/"),
36
+ "substack": ("https://substack.com/sign-in", "substack.com/home"),
37
+ }
38
+
39
+ #: Platforms whose adapters need a page injected.
40
+ BROWSER_PLATFORMS = set(LOGIN_FLOWS)
41
+
42
+
43
+ def _uploader(page):
44
+ """Hand real bytes to a file input.
45
+
46
+ Never click a file input to open a picker — a native dialog cannot be
47
+ driven and it blocks the whole session. `set_input_files` sets the files
48
+ directly, which is why `BrowserAdapter` injects its own hidden input.
49
+ """
50
+
51
+ async def upload(selector: str, paths: Sequence[str]) -> None:
52
+ await page.set_input_files(selector, list(paths))
53
+
54
+ return upload
55
+
56
+
57
+ class BrowserPool:
58
+ """One browser for the run; one context per platform.
59
+
60
+ Separate contexts matter: each platform gets only its own cookies, so a
61
+ Medium session is never presented to Substack.
62
+ """
63
+
64
+ def __init__(self, sessions: SessionStore, *, headless: bool = True, slow_mo: int = 0) -> None:
65
+ self.sessions = sessions
66
+ self.headless = headless
67
+ self.slow_mo = slow_mo
68
+ self._stack = AsyncExitStack()
69
+ self._pw = None
70
+ self._browser = None
71
+ self._contexts: dict[str, object] = {}
72
+
73
+ async def __aenter__(self) -> BrowserPool:
74
+ from playwright.async_api import async_playwright
75
+
76
+ self._pw = await self._stack.enter_async_context(async_playwright())
77
+ self._browser = await self._pw.chromium.launch(headless=self.headless, slow_mo=self.slow_mo)
78
+ return self
79
+
80
+ async def __aexit__(self, *exc) -> None:
81
+ for platform, ctx in self._contexts.items():
82
+ # Cookies rotate. Refreshing on the way out is what stops a session
83
+ # silently expiring between runs.
84
+ try:
85
+ self.sessions.save(platform, await ctx.storage_state())
86
+ except Exception: # noqa: BLE001
87
+ log.debug("could not refresh the %s session", platform)
88
+ try:
89
+ await ctx.close()
90
+ except Exception: # noqa: BLE001
91
+ pass
92
+ if self._browser:
93
+ await self._browser.close()
94
+ await self._stack.aclose()
95
+
96
+ async def page_for(self, platform: str):
97
+ """A page carrying `platform`'s saved session, plus its uploader."""
98
+ if platform in self._contexts:
99
+ ctx = self._contexts[platform]
100
+ return ctx.pages[0], _uploader(ctx.pages[0])
101
+
102
+ state = self.sessions.load(platform)
103
+ if state is None:
104
+ raise CredentialError(
105
+ f"no saved {platform} session.\n"
106
+ f" Run: pubkit auth login {platform}\n"
107
+ f" A browser window opens, you sign in yourself, and pubkit keeps\n"
108
+ f" only the session — it never asks for your password."
109
+ )
110
+ ctx = await self._browser.new_context(
111
+ storage_state=state, viewport={"width": 1440, "height": 900}
112
+ )
113
+ page = await ctx.new_page()
114
+ self._contexts[platform] = ctx
115
+ return page, _uploader(page)
116
+
117
+
118
+ @asynccontextmanager
119
+ async def attached(adapters: Sequence, sessions: SessionStore, *, headless: bool = True):
120
+ """Yield `adapters` with browser ones wired to a live page.
121
+
122
+ This is the seam that was missing: the runner calls `adapter.authenticate()`
123
+ and friends, but a `MediumAdapter` built by the registry has `_page = None`
124
+ until something puts a page in it. That something is here.
125
+
126
+ API adapters pass through untouched, and no browser is launched at all if
127
+ none of the adapters need one — publishing to Dev.to should not pay for
128
+ Chromium.
129
+ """
130
+ needs_browser = [a for a in adapters if getattr(a, "name", None) in BROWSER_PLATFORMS]
131
+ if not needs_browser:
132
+ yield list(adapters)
133
+ return
134
+
135
+ # Check every session BEFORE a browser exists. A missing login is the most
136
+ # common reason a publish cannot start, and spending a Chromium launch to
137
+ # then say "run pubkit auth login" is both slow and the wrong order — it
138
+ # also leaves a browser process to clean up on the failure path.
139
+ missing = [a.name for a in needs_browser if sessions.load(a.name) is None]
140
+ if missing:
141
+ raise CredentialError(
142
+ "no saved session for: " + ", ".join(missing) + ".\n"
143
+ + "\n".join(f" Run: pubkit auth login {p}" for p in missing)
144
+ + "\n A browser window opens, you sign in yourself, and pubkit keeps\n"
145
+ " only the session — it never asks for your password."
146
+ )
147
+
148
+ async with BrowserPool(sessions, headless=headless) as pool:
149
+ for adapter in needs_browser:
150
+ page, upload = await pool.page_for(adapter.name)
151
+ adapter._page = page
152
+ adapter._upload = upload
153
+ log.debug("attached a browser page to %s", adapter.name)
154
+ yield list(adapters)
155
+
156
+
157
+ async def interactive_login(platform: str, sessions: SessionStore, timeout: float = 300.0) -> None:
158
+ """Open a visible window, wait for the human, save the session."""
159
+ from playwright.async_api import async_playwright
160
+
161
+ login_url, success_marker = LOGIN_FLOWS.get(
162
+ platform, (f"https://{platform}.com/login", f"{platform}.com")
163
+ )
164
+
165
+ async with async_playwright() as pw:
166
+ browser = await pw.chromium.launch(headless=False)
167
+ ctx = await browser.new_context(viewport={"width": 1280, "height": 900})
168
+ page = await ctx.new_page()
169
+ await page.goto(login_url)
170
+
171
+ print(f"\n A browser window is open at {login_url}")
172
+ print(" Sign in there — password manager, MFA, all of it.")
173
+ print(f" Waiting up to {timeout / 60:.0f} minutes. Close this with Ctrl-C to cancel.\n")
174
+
175
+ deadline = asyncio.get_event_loop().time() + timeout
176
+ while asyncio.get_event_loop().time() < deadline:
177
+ try:
178
+ url = page.url
179
+ except Exception as exc: # window closed by the user
180
+ await browser.close()
181
+ raise CredentialError(
182
+ f"{platform} login window was closed before sign-in completed"
183
+ ) from exc
184
+
185
+ if success_marker in url:
186
+ await asyncio.sleep(2) # let the last auth cookie land
187
+ sessions.save(platform, await ctx.storage_state())
188
+ await browser.close()
189
+ return
190
+ await asyncio.sleep(1.5)
191
+
192
+ await browser.close()
193
+ raise TimeoutError(
194
+ f"no {platform} sign-in detected within {timeout / 60:.0f} minutes. "
195
+ f"pubkit watches for a URL containing {success_marker!r}."
196
+ )
197
+
198
+
199
+ async def verify_session(platform: str, sessions: SessionStore, *, headless: bool = True) -> bool:
200
+ """Is the saved session still good? Cheaper to find out now than mid-publish."""
201
+ _, success_marker = LOGIN_FLOWS.get(platform, ("", platform))
202
+ probe = {
203
+ "medium": "https://medium.com/me/stories/drafts",
204
+ "substack": "https://substack.com/home",
205
+ }.get(platform)
206
+ if probe is None or sessions.load(platform) is None:
207
+ return False
208
+
209
+ async with BrowserPool(sessions, headless=headless) as pool:
210
+ page, _ = await pool.page_for(platform)
211
+ await page.goto(probe, wait_until="domcontentloaded")
212
+ await asyncio.sleep(1.5)
213
+ return "signin" not in page.url and "sign-in" not in page.url
@@ -147,11 +147,17 @@ def publish(
147
147
  )
148
148
 
149
149
  pipe = Pipeline(StateStore())
150
+ live = confirm and not draft_only
150
151
 
151
152
  async def go():
152
- if hasattr(target, "documents"):
153
- return await pipe.run_series(target, adapters, ctx_for, publish=confirm and not draft_only)
154
- return await pipe.run([target], adapters, ctx_for, publish=confirm and not draft_only)
153
+ # Browser adapters get a live page here and nowhere else. API-only runs
154
+ # never launch Chromium.
155
+ from .browserctl import attached
156
+
157
+ async with attached(adapters, sessions, headless=headless) as ready:
158
+ if hasattr(target, "documents"):
159
+ return await pipe.run_series(target, ready, ctx_for, publish=live)
160
+ return await pipe.run([target], ready, ctx_for, publish=live)
155
161
 
156
162
  report = asyncio.run(go())
157
163
  console.print(f"\n[bold]{report.run_id}[/]")
@@ -254,10 +260,12 @@ def auth_login(
254
260
  Browser platforms: a real browser window opens and you sign in yourself —
255
261
  pubkit stores only the resulting session, never a password.
256
262
  """
263
+ from .browserctl import BROWSER_PLATFORMS
264
+
257
265
  tokens = TokenStore()
258
- adapter = build_adapter(platform)
266
+ build_adapter(platform) # fail fast on an unknown platform
259
267
 
260
- if adapter.capabilities.image_upload != "browser_paste" and adapter.name not in ("substack",):
268
+ if platform not in BROWSER_PLATFORMS:
261
269
  if token is None:
262
270
  token = typer.prompt(f"{platform} API token", hide_input=True)
263
271
  tokens.set(platform, token.strip())
@@ -266,12 +274,15 @@ def auth_login(
266
274
 
267
275
  from .browserctl import interactive_login
268
276
 
277
+ try:
278
+ asyncio.run(interactive_login(platform, SessionStore()))
279
+ except Exception as exc: # noqa: BLE001
280
+ console.print(f"[red]{exc}[/]")
281
+ raise typer.Exit(1) from exc
269
282
  console.print(
270
- f"Opening a browser window for {platform}. Sign in there — including MFA — "
271
- "and pubkit will save the session when it sees you are logged in."
283
+ f"[green]saved {platform} session (encrypted)[/]\n"
284
+ f"Verify any time with: pubkit auth verify {platform}"
272
285
  )
273
- asyncio.run(interactive_login(platform, SessionStore()))
274
- console.print(f"[green]saved {platform} session (encrypted)[/]")
275
286
 
276
287
 
277
288
  @auth_app.command("logout")
@@ -281,6 +292,24 @@ def auth_logout(platform: str = typer.Argument(...)):
281
292
  console.print(f"[green]forgot all {platform} credentials[/]")
282
293
 
283
294
 
295
+ @auth_app.command("verify")
296
+ def auth_verify(platform: str = typer.Argument(...)):
297
+ """Check a saved session is still valid — cheaper now than mid-publish."""
298
+ from .browserctl import BROWSER_PLATFORMS, verify_session
299
+
300
+ if platform not in BROWSER_PLATFORMS:
301
+ tok = TokenStore().get(platform) or TokenStore().get(platform, "bearer")
302
+ console.print(f"[green]{platform}: token present[/]" if tok else f"[red]{platform}: no token[/]")
303
+ raise typer.Exit(0 if tok else 1)
304
+
305
+ ok = asyncio.run(verify_session(platform, SessionStore()))
306
+ if ok:
307
+ console.print(f"[green]{platform}: session valid[/]")
308
+ else:
309
+ console.print(f"[red]{platform}: session missing or expired[/] — run `pubkit auth login {platform}`")
310
+ raise typer.Exit(0 if ok else 1)
311
+
312
+
284
313
  @auth_app.command("list")
285
314
  def auth_list():
286
315
  tokens, sessions = TokenStore(), SessionStore()
@@ -0,0 +1,96 @@
1
+ # Copyright 2026 The pubkit Authors
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Materialise the assets a plan promised.
4
+
5
+ The planner says "8 tables → images" and lists the asset ids. Something has to
6
+ actually produce those files, and it has to happen per-platform, because the
7
+ same document keeps native tables on Dev.to and needs images on Medium.
8
+
9
+ This runs between planning and publishing, and is cached on content: a table
10
+ whose cells have not changed keeps its rendered file, so a re-run does not
11
+ re-render or re-upload it.
12
+ """
13
+ from __future__ import annotations
14
+
15
+ import logging
16
+ from pathlib import Path
17
+
18
+ from ..render.tables import RenderUnavailable, TableStyle, render_table, table_asset_id
19
+ from .capabilities import Capabilities
20
+ from .ir import Asset, Document, Figure, Table
21
+
22
+ log = logging.getLogger(__name__)
23
+
24
+
25
+ def materialise(
26
+ doc: Document,
27
+ caps: Capabilities,
28
+ workdir: Path,
29
+ *,
30
+ style: TableStyle | None = None,
31
+ ) -> Document:
32
+ """Return a copy of `doc` adapted to what `caps` can actually render.
33
+
34
+ Tables become Figures on platforms without table support; everything else is
35
+ left alone. The original document is never mutated — the same IR has to
36
+ serve every platform in the same run.
37
+ """
38
+ if caps.tables or not doc.tables:
39
+ return doc
40
+
41
+ if caps.image_upload == "none":
42
+ log.warning(
43
+ "%s supports neither tables nor images; %d table(s) will degrade to lists",
44
+ getattr(caps, "name", "platform"),
45
+ len(doc.tables),
46
+ )
47
+ return doc
48
+
49
+ out_dir = workdir / "rendered"
50
+ new_blocks: list = []
51
+ new_assets = dict(doc.assets)
52
+ rendered = 0
53
+ reused = 0
54
+ n = 0
55
+
56
+ for block in doc.blocks:
57
+ if not isinstance(block, Table):
58
+ new_blocks.append(block)
59
+ continue
60
+
61
+ n += 1
62
+ aid = table_asset_id(block, n)
63
+ path = out_dir / f"{aid}.png"
64
+
65
+ if not path.exists():
66
+ try:
67
+ render_table(block, path, style)
68
+ rendered += 1
69
+ except RenderUnavailable:
70
+ # Better a readable list than a crash at publish time.
71
+ log.warning("Pillow missing — table %d degrades to a list", n)
72
+ from .ir import ListBlock
73
+
74
+ new_blocks.append(
75
+ ListBlock(
76
+ ordered=False,
77
+ items=[
78
+ " · ".join(f"{h}: {c}" for h, c in zip(block.header, row, strict=False))
79
+ for row in block.rows
80
+ ],
81
+ )
82
+ )
83
+ continue
84
+ else:
85
+ reused += 1
86
+
87
+ new_assets[aid] = Asset(
88
+ id=aid, path=path, alt=block.caption or "table", generated_from=f"table[{n}]"
89
+ )
90
+ new_blocks.append(Figure(asset_id=aid, caption=block.caption, alt=block.caption or "table"))
91
+
92
+ if rendered or reused:
93
+ log.info("tables → images: %d rendered, %d reused from cache", rendered, reused)
94
+
95
+ adapted = doc.model_copy(update={"blocks": new_blocks, "assets": new_assets})
96
+ return adapted
@@ -221,7 +221,7 @@ class BrowserAdapter(BaseAdapter):
221
221
 
222
222
  selectors: EditorSelectors
223
223
  login_url: str = ""
224
- marker_regex: str = r"\\[\\[\\s*IMAGE"
224
+ marker_regex: str = r"\[\[\s*IMAGE" # a JS RegExp source string, not a Python pattern
225
225
 
226
226
  def __init__(self) -> None:
227
227
  super().__init__()