pubkit 0.1.0__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pubkit-0.2.0/CHANGELOG.md +97 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/PKG-INFO +51 -3
- {pubkit-0.1.0 → pubkit-0.2.0}/README.md +50 -2
- {pubkit-0.1.0 → pubkit-0.2.0}/pyproject.toml +1 -1
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/__init__.py +1 -1
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/adapters/medium.py +1 -1
- pubkit-0.2.0/src/pubkit/browserctl.py +213 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/cli.py +38 -9
- pubkit-0.2.0/src/pubkit/core/assets.py +96 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/browser.py +1 -1
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/runner.py +7 -0
- pubkit-0.2.0/src/pubkit/render/tables.py +199 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/scaffold.py +27 -7
- pubkit-0.2.0/tests/fixtures/fake_editor.html +119 -0
- pubkit-0.2.0/tests/test_browser.py +182 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/tests/test_core.py +129 -0
- pubkit-0.1.0/CHANGELOG.md +0 -45
- pubkit-0.1.0/src/pubkit/browserctl.py +0 -78
- {pubkit-0.1.0 → pubkit-0.2.0}/.gitignore +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/LICENSE +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/NOTICE +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/docs/ADAPTERS.md +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/docs/ARCHITECTURE.md +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/docs/FAILURE-MODES.md +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/docs/QUICKSTART.md +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part1-01-model-sizes.png +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part1-02-cpu-vs-gpu-ANIMATED.gif +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part1-03-starving-cores-ANIMATED.gif +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part1-04-gpu-comparison.png +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part2-01-one-token-one-read-ANIMATED.gif +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part2-02-decode-ceilings.png +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part2-03-prefill-vs-decode.png +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part2-04-kv-cache-cost.png +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part2-05-batching-ANIMATED.gif +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part2-06-vram-budget.png +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part2-07-throughput-latency.png +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part3-01-interconnect.png +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part3-02-llmd-request-path.png +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part3-03-llmd-objects.png +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/img/part3-04-slos.png +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/part-1.md +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/part-2.md +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/examples/inside-ai-infra/part-3.md +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/__main__.py +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/adapters/__init__.py +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/adapters/api_base.py +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/adapters/devto.py +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/adapters/substack.py +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/adapters/x.py +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/__init__.py +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/adapter.py +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/anchors.py +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/auth.py +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/capabilities.py +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/checks.py +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/ir.py +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/loader.py +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/state.py +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/core/transport.py +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/py.typed +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/registry.py +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/render/__init__.py +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/render/html.py +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/workflows/__init__.py +0 -0
- {pubkit-0.1.0 → pubkit-0.2.0}/src/pubkit/workflows/airflow.py +0 -0
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
Format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
|
|
4
|
+
Versioning is [semantic](https://semver.org/); adapters may change behaviour on
|
|
5
|
+
a minor bump when a platform changes underneath them.
|
|
6
|
+
|
|
7
|
+
## [0.2.0] — 2026-09-11
|
|
8
|
+
|
|
9
|
+
The release that makes `pubkit publish --to medium --confirm` actually work
|
|
10
|
+
unattended. 0.1.x could plan a Medium publish; it could not perform one.
|
|
11
|
+
|
|
12
|
+
### Added
|
|
13
|
+
|
|
14
|
+
- **Browser lifecycle** (`browserctl.BrowserPool`, `attached()`). This was the
|
|
15
|
+
missing seam: the runner called `adapter.authenticate()` on a `MediumAdapter`
|
|
16
|
+
whose `_page` was still `None`. Now one browser serves the whole run, one
|
|
17
|
+
context per platform so a Medium session is never presented to Substack, and
|
|
18
|
+
sessions are refreshed on the way out because cookies rotate. API-only runs
|
|
19
|
+
never launch Chromium at all.
|
|
20
|
+
- **Table rendering** (`render.tables`, `core.assets.materialise`). The planner
|
|
21
|
+
said `table_to_image` and listed the asset ids; nothing produced the files.
|
|
22
|
+
Tables now render to quantised PNGs at 2x, cached on cell content so a re-run
|
|
23
|
+
neither re-renders nor re-uploads an unchanged table. The source IR is never
|
|
24
|
+
mutated — the same document keeps native tables on Dev.to in the same run.
|
|
25
|
+
- `pubkit auth verify <platform>` — check a session before a publish depends on
|
|
26
|
+
it, rather than discovering it expired halfway through.
|
|
27
|
+
- **Browser integration tests** against a deliberately hostile fixture editor
|
|
28
|
+
that reproduces multiple content roots, image stripping, figures landing
|
|
29
|
+
above the caret, and uploads that sit on a `blob:` URL before resolving.
|
|
30
|
+
They skip cleanly where Chromium is unavailable.
|
|
31
|
+
|
|
32
|
+
### Fixed
|
|
33
|
+
|
|
34
|
+
- `marker_regex` was double-escaped, so every `fingerprint()` call threw
|
|
35
|
+
`SyntaxError: Invalid regular expression` inside the page. Verification was
|
|
36
|
+
therefore broken on every browser adapter. Found by the new integration
|
|
37
|
+
tests on their first run.
|
|
38
|
+
- `auth login` routed on a capability flag rather than the platform list, so
|
|
39
|
+
`pubkit auth login substack` took the API-token path and prompted for a token
|
|
40
|
+
that does not exist.
|
|
41
|
+
|
|
42
|
+
## [0.1.1] — 2026-09-11
|
|
43
|
+
|
|
44
|
+
### Added
|
|
45
|
+
|
|
46
|
+
- Docker image on the Playwright base (`ghcr.io/arunsingh/pubkit`), so browser
|
|
47
|
+
adapters work in CI without hand-installing Chromium's system libraries.
|
|
48
|
+
- `action.yml` — the repository is now a reusable GitHub Action.
|
|
49
|
+
- `CITATION.cff`, and a Homebrew formula template under `packaging/`.
|
|
50
|
+
|
|
51
|
+
### Fixed
|
|
52
|
+
|
|
53
|
+
- `pubkit doctor` ignored `PLAYWRIGHT_BROWSERS_PATH` and reported Chromium as
|
|
54
|
+
missing on managed images (CI runners, devcontainers, the Playwright Docker
|
|
55
|
+
image) where it was in fact installed — sending people off to fix a problem
|
|
56
|
+
they did not have. Found by running the published package in exactly such an
|
|
57
|
+
environment.
|
|
58
|
+
|
|
59
|
+
## [0.1.0] — 2026-09-11
|
|
60
|
+
|
|
61
|
+
First release. Extracted from a real, painful publishing run: a 15,000-word
|
|
62
|
+
illustrated three-part series to Medium that surfaced 21 distinct failure modes,
|
|
63
|
+
all catalogued in `docs/FAILURE-MODES.md`.
|
|
64
|
+
|
|
65
|
+
### Added
|
|
66
|
+
|
|
67
|
+
- **Post IR** — a canonical, platform-agnostic document model with a
|
|
68
|
+
content-addressed id used as the idempotency key everywhere.
|
|
69
|
+
- **Capability negotiation** — adapters declare what they support; the planner
|
|
70
|
+
emits explicit, printable degradations. `pubkit plan` shows them before
|
|
71
|
+
anything is sent.
|
|
72
|
+
- **Check pipeline** — placeholders, numeric assertions, cross-references,
|
|
73
|
+
number drift, assets, budget, structure. Pluggable via `pubkit.checks`.
|
|
74
|
+
- **Verified chunked transport** — per-chunk rolling-hash verification, probed
|
|
75
|
+
size ceilings, single-member gzip.
|
|
76
|
+
- **Browser adapter toolkit** — paste-based document replace with a block-count
|
|
77
|
+
assertion, reload-then-fingerprint verification, caption-anchored `File`
|
|
78
|
+
image insertion, Unicode-folding anchor resolution, tag-chip post-conditions.
|
|
79
|
+
- **Adapters** — Medium, Substack (browser); X, Dev.to, Hashnode (API).
|
|
80
|
+
- **Runner** — six resumable steps per (document, platform), sqlite state store,
|
|
81
|
+
two-phase publish for series, hash-pinned `--confirm` gate, per-platform rate
|
|
82
|
+
limiting and jittered retry.
|
|
83
|
+
- **CLI** — `init`, `validate`, `plan`, `publish`, `status`, `platforms`,
|
|
84
|
+
`doctor`, `auth login|logout|list`.
|
|
85
|
+
- **Workflows** — GitHub Actions examples and Airflow operators.
|
|
86
|
+
- **Extension points** — `pubkit.adapters`, `pubkit.renderers`, `pubkit.checks`,
|
|
87
|
+
`pubkit.hooks`, plus per-platform transforms.
|
|
88
|
+
|
|
89
|
+
### Known limitations
|
|
90
|
+
|
|
91
|
+
- Browser adapter selectors will need patching when a platform redesigns its
|
|
92
|
+
editor. They are isolated in one `EditorSelectors` dataclass per adapter
|
|
93
|
+
precisely so that is a small change.
|
|
94
|
+
- Table→image rendering requires the `render` extra; without it, tables degrade
|
|
95
|
+
to lists on platforms that lack table support.
|
|
96
|
+
- X media upload uses the v1.1 endpoint, which is what the v2 API still
|
|
97
|
+
requires.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: pubkit
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: Publish one source to many platforms, safely. Medium, Substack, X, Dev.to, Hashnode.
|
|
5
5
|
Project-URL: Homepage, https://github.com/arunsingh/pubkit
|
|
6
6
|
Project-URL: Issues, https://github.com/arunsingh/pubkit/issues
|
|
@@ -109,9 +109,10 @@ pubkit validate content/series/
|
|
|
109
109
|
# 2. See exactly what each platform will get — including every degradation.
|
|
110
110
|
pubkit plan content/series/ --to medium,devto,x
|
|
111
111
|
|
|
112
|
-
# 3. Sign in. pubkit never accepts a password.
|
|
113
|
-
pubkit auth login medium #
|
|
112
|
+
# 3. Sign in, once. pubkit never accepts a password.
|
|
113
|
+
pubkit auth login medium # a window opens; you sign in; it keeps the session
|
|
114
114
|
pubkit auth login devto # API token → OS keychain
|
|
115
|
+
pubkit auth verify medium # confirm before a publish depends on it
|
|
115
116
|
|
|
116
117
|
# 4. Drafts everywhere. Safe to re-run; a dropped connection costs one step.
|
|
117
118
|
pubkit publish content/series/ --to medium,devto,x
|
|
@@ -138,6 +139,31 @@ x: part-1 (bf9abfe78ce4)
|
|
|
138
139
|
|
|
139
140
|
Nothing is a surprise at publish time. That is the whole design goal.
|
|
140
141
|
|
|
142
|
+
## Sign in once, then it runs on its own
|
|
143
|
+
|
|
144
|
+
```bash
|
|
145
|
+
pubkit auth login medium
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
A real browser window opens at Medium's login page. You sign in — password
|
|
149
|
+
manager, MFA, device confirmation, whatever it asks for. pubkit watches for the
|
|
150
|
+
post-login URL, saves the session encrypted, and closes the window.
|
|
151
|
+
|
|
152
|
+
From then on, unattended:
|
|
153
|
+
|
|
154
|
+
```bash
|
|
155
|
+
pubkit publish content/ --to medium,devto,x --confirm
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
One browser serves the whole run, one context per platform so a Medium session
|
|
159
|
+
is never presented to Substack, and sessions refresh on the way out because
|
|
160
|
+
cookies rotate. Publishing only to API platforms never launches Chromium at all.
|
|
161
|
+
|
|
162
|
+
pubkit sees cookies. It never sees, types or stores a password — which is both
|
|
163
|
+
the right security posture and the only thing that works, since platforms
|
|
164
|
+
increasingly gate login behind challenges an automation layer has no business
|
|
165
|
+
trying to defeat.
|
|
166
|
+
|
|
141
167
|
## Write once
|
|
142
168
|
|
|
143
169
|
````markdown
|
|
@@ -229,6 +255,28 @@ ghost = "my_pubkit_ghost:GhostAdapter"
|
|
|
229
255
|
Five extension points, all first-class: `pubkit.adapters`, `pubkit.renderers`,
|
|
230
256
|
`pubkit.checks`, `pubkit.hooks`, and per-platform `transforms:` in config.
|
|
231
257
|
|
|
258
|
+
## Install it however you like
|
|
259
|
+
|
|
260
|
+
```bash
|
|
261
|
+
pip install "pubkit[all]" # the normal way
|
|
262
|
+
pipx install "pubkit[all]" # isolated CLI
|
|
263
|
+
uv tool install "pubkit[all]" # same, faster
|
|
264
|
+
docker run --rm -v "$PWD:/work" ghcr.io/arunsingh/pubkit validate content/
|
|
265
|
+
```
|
|
266
|
+
|
|
267
|
+
The Docker image ships Chromium and its system libraries, which is the part
|
|
268
|
+
nobody wants to install on a CI runner by hand.
|
|
269
|
+
|
|
270
|
+
As a GitHub Action, no install step at all:
|
|
271
|
+
|
|
272
|
+
```yaml
|
|
273
|
+
- uses: arunsingh/pubkit@v1
|
|
274
|
+
with:
|
|
275
|
+
command: plan
|
|
276
|
+
path: content/
|
|
277
|
+
platforms: medium,devto
|
|
278
|
+
```
|
|
279
|
+
|
|
232
280
|
## Fit it into a workflow
|
|
233
281
|
|
|
234
282
|
**GitHub Actions** — plan on every PR, publish on merge:
|
|
@@ -59,9 +59,10 @@ pubkit validate content/series/
|
|
|
59
59
|
# 2. See exactly what each platform will get — including every degradation.
|
|
60
60
|
pubkit plan content/series/ --to medium,devto,x
|
|
61
61
|
|
|
62
|
-
# 3. Sign in. pubkit never accepts a password.
|
|
63
|
-
pubkit auth login medium #
|
|
62
|
+
# 3. Sign in, once. pubkit never accepts a password.
|
|
63
|
+
pubkit auth login medium # a window opens; you sign in; it keeps the session
|
|
64
64
|
pubkit auth login devto # API token → OS keychain
|
|
65
|
+
pubkit auth verify medium # confirm before a publish depends on it
|
|
65
66
|
|
|
66
67
|
# 4. Drafts everywhere. Safe to re-run; a dropped connection costs one step.
|
|
67
68
|
pubkit publish content/series/ --to medium,devto,x
|
|
@@ -88,6 +89,31 @@ x: part-1 (bf9abfe78ce4)
|
|
|
88
89
|
|
|
89
90
|
Nothing is a surprise at publish time. That is the whole design goal.
|
|
90
91
|
|
|
92
|
+
## Sign in once, then it runs on its own
|
|
93
|
+
|
|
94
|
+
```bash
|
|
95
|
+
pubkit auth login medium
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
A real browser window opens at Medium's login page. You sign in — password
|
|
99
|
+
manager, MFA, device confirmation, whatever it asks for. pubkit watches for the
|
|
100
|
+
post-login URL, saves the session encrypted, and closes the window.
|
|
101
|
+
|
|
102
|
+
From then on, unattended:
|
|
103
|
+
|
|
104
|
+
```bash
|
|
105
|
+
pubkit publish content/ --to medium,devto,x --confirm
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
One browser serves the whole run, one context per platform so a Medium session
|
|
109
|
+
is never presented to Substack, and sessions refresh on the way out because
|
|
110
|
+
cookies rotate. Publishing only to API platforms never launches Chromium at all.
|
|
111
|
+
|
|
112
|
+
pubkit sees cookies. It never sees, types or stores a password — which is both
|
|
113
|
+
the right security posture and the only thing that works, since platforms
|
|
114
|
+
increasingly gate login behind challenges an automation layer has no business
|
|
115
|
+
trying to defeat.
|
|
116
|
+
|
|
91
117
|
## Write once
|
|
92
118
|
|
|
93
119
|
````markdown
|
|
@@ -179,6 +205,28 @@ ghost = "my_pubkit_ghost:GhostAdapter"
|
|
|
179
205
|
Five extension points, all first-class: `pubkit.adapters`, `pubkit.renderers`,
|
|
180
206
|
`pubkit.checks`, `pubkit.hooks`, and per-platform `transforms:` in config.
|
|
181
207
|
|
|
208
|
+
## Install it however you like
|
|
209
|
+
|
|
210
|
+
```bash
|
|
211
|
+
pip install "pubkit[all]" # the normal way
|
|
212
|
+
pipx install "pubkit[all]" # isolated CLI
|
|
213
|
+
uv tool install "pubkit[all]" # same, faster
|
|
214
|
+
docker run --rm -v "$PWD:/work" ghcr.io/arunsingh/pubkit validate content/
|
|
215
|
+
```
|
|
216
|
+
|
|
217
|
+
The Docker image ships Chromium and its system libraries, which is the part
|
|
218
|
+
nobody wants to install on a CI runner by hand.
|
|
219
|
+
|
|
220
|
+
As a GitHub Action, no install step at all:
|
|
221
|
+
|
|
222
|
+
```yaml
|
|
223
|
+
- uses: arunsingh/pubkit@v1
|
|
224
|
+
with:
|
|
225
|
+
command: plan
|
|
226
|
+
path: content/
|
|
227
|
+
platforms: medium,devto
|
|
228
|
+
```
|
|
229
|
+
|
|
182
230
|
## Fit it into a workflow
|
|
183
231
|
|
|
184
232
|
**GitHub Actions** — plan on every PR, publish on merge:
|
|
@@ -6,7 +6,7 @@ from .core.loader import load_document, load_series # noqa: F401
|
|
|
6
6
|
from .core.runner import Pipeline, RunReport # noqa: F401
|
|
7
7
|
from .registry import build_adapter, list_adapters, register # noqa: F401
|
|
8
8
|
|
|
9
|
-
__version__ = "0.
|
|
9
|
+
__version__ = "0.2.0"
|
|
10
10
|
__all__ = [
|
|
11
11
|
"Document", "Series", "Asset", "Figure", "Table", "Heading", "Paragraph", "Code",
|
|
12
12
|
"Pipeline", "RunReport", "load_document", "load_series",
|
|
@@ -68,7 +68,7 @@ class MediumAdapter(BrowserAdapter):
|
|
|
68
68
|
|
|
69
69
|
login_url = "https://medium.com/m/signin"
|
|
70
70
|
new_story_url = "https://medium.com/new-story"
|
|
71
|
-
marker_regex = r"
|
|
71
|
+
marker_regex = r"\[\[\s*IMAGE"
|
|
72
72
|
|
|
73
73
|
def __init__(self, page=None, upload=None) -> None:
|
|
74
74
|
super().__init__()
|
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
# Copyright 2026 The pubkit Authors
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
"""Browser lifecycle.
|
|
4
|
+
|
|
5
|
+
Browser adapters need a live Playwright page and a way to hand real bytes to a
|
|
6
|
+
file input. Nothing else in the pipeline should have to know that, so this
|
|
7
|
+
module owns it: one browser for the whole run, one context per platform, and
|
|
8
|
+
sessions refreshed on the way out.
|
|
9
|
+
|
|
10
|
+
The login flow is deliberately human-in-the-loop:
|
|
11
|
+
|
|
12
|
+
pubkit auth login medium
|
|
13
|
+
→ a *visible* window opens at the platform's login page
|
|
14
|
+
→ you sign in: password manager, MFA, device confirmation, whatever
|
|
15
|
+
→ pubkit waits until it sees you are through, saves the session, closes
|
|
16
|
+
|
|
17
|
+
pubkit sees cookies. It never sees, types or stores a password. That is not
|
|
18
|
+
only the right security posture, it is the only thing that works: platforms
|
|
19
|
+
increasingly gate login behind challenges an automation layer has no business
|
|
20
|
+
trying to defeat.
|
|
21
|
+
"""
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import asyncio
|
|
25
|
+
import logging
|
|
26
|
+
from collections.abc import Sequence
|
|
27
|
+
from contextlib import AsyncExitStack, asynccontextmanager
|
|
28
|
+
|
|
29
|
+
from .core.auth import CredentialError, SessionStore
|
|
30
|
+
|
|
31
|
+
log = logging.getLogger(__name__)
|
|
32
|
+
|
|
33
|
+
#: (where to send the user, how to tell they made it)
|
|
34
|
+
LOGIN_FLOWS: dict[str, tuple[str, str]] = {
|
|
35
|
+
"medium": ("https://medium.com/m/signin", "medium.com/me/"),
|
|
36
|
+
"substack": ("https://substack.com/sign-in", "substack.com/home"),
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
#: Platforms whose adapters need a page injected.
|
|
40
|
+
BROWSER_PLATFORMS = set(LOGIN_FLOWS)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _uploader(page):
|
|
44
|
+
"""Hand real bytes to a file input.
|
|
45
|
+
|
|
46
|
+
Never click a file input to open a picker — a native dialog cannot be
|
|
47
|
+
driven and it blocks the whole session. `set_input_files` sets the files
|
|
48
|
+
directly, which is why `BrowserAdapter` injects its own hidden input.
|
|
49
|
+
"""
|
|
50
|
+
|
|
51
|
+
async def upload(selector: str, paths: Sequence[str]) -> None:
|
|
52
|
+
await page.set_input_files(selector, list(paths))
|
|
53
|
+
|
|
54
|
+
return upload
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class BrowserPool:
|
|
58
|
+
"""One browser for the run; one context per platform.
|
|
59
|
+
|
|
60
|
+
Separate contexts matter: each platform gets only its own cookies, so a
|
|
61
|
+
Medium session is never presented to Substack.
|
|
62
|
+
"""
|
|
63
|
+
|
|
64
|
+
def __init__(self, sessions: SessionStore, *, headless: bool = True, slow_mo: int = 0) -> None:
|
|
65
|
+
self.sessions = sessions
|
|
66
|
+
self.headless = headless
|
|
67
|
+
self.slow_mo = slow_mo
|
|
68
|
+
self._stack = AsyncExitStack()
|
|
69
|
+
self._pw = None
|
|
70
|
+
self._browser = None
|
|
71
|
+
self._contexts: dict[str, object] = {}
|
|
72
|
+
|
|
73
|
+
async def __aenter__(self) -> BrowserPool:
|
|
74
|
+
from playwright.async_api import async_playwright
|
|
75
|
+
|
|
76
|
+
self._pw = await self._stack.enter_async_context(async_playwright())
|
|
77
|
+
self._browser = await self._pw.chromium.launch(headless=self.headless, slow_mo=self.slow_mo)
|
|
78
|
+
return self
|
|
79
|
+
|
|
80
|
+
async def __aexit__(self, *exc) -> None:
|
|
81
|
+
for platform, ctx in self._contexts.items():
|
|
82
|
+
# Cookies rotate. Refreshing on the way out is what stops a session
|
|
83
|
+
# silently expiring between runs.
|
|
84
|
+
try:
|
|
85
|
+
self.sessions.save(platform, await ctx.storage_state())
|
|
86
|
+
except Exception: # noqa: BLE001
|
|
87
|
+
log.debug("could not refresh the %s session", platform)
|
|
88
|
+
try:
|
|
89
|
+
await ctx.close()
|
|
90
|
+
except Exception: # noqa: BLE001
|
|
91
|
+
pass
|
|
92
|
+
if self._browser:
|
|
93
|
+
await self._browser.close()
|
|
94
|
+
await self._stack.aclose()
|
|
95
|
+
|
|
96
|
+
async def page_for(self, platform: str):
|
|
97
|
+
"""A page carrying `platform`'s saved session, plus its uploader."""
|
|
98
|
+
if platform in self._contexts:
|
|
99
|
+
ctx = self._contexts[platform]
|
|
100
|
+
return ctx.pages[0], _uploader(ctx.pages[0])
|
|
101
|
+
|
|
102
|
+
state = self.sessions.load(platform)
|
|
103
|
+
if state is None:
|
|
104
|
+
raise CredentialError(
|
|
105
|
+
f"no saved {platform} session.\n"
|
|
106
|
+
f" Run: pubkit auth login {platform}\n"
|
|
107
|
+
f" A browser window opens, you sign in yourself, and pubkit keeps\n"
|
|
108
|
+
f" only the session — it never asks for your password."
|
|
109
|
+
)
|
|
110
|
+
ctx = await self._browser.new_context(
|
|
111
|
+
storage_state=state, viewport={"width": 1440, "height": 900}
|
|
112
|
+
)
|
|
113
|
+
page = await ctx.new_page()
|
|
114
|
+
self._contexts[platform] = ctx
|
|
115
|
+
return page, _uploader(page)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
@asynccontextmanager
|
|
119
|
+
async def attached(adapters: Sequence, sessions: SessionStore, *, headless: bool = True):
|
|
120
|
+
"""Yield `adapters` with browser ones wired to a live page.
|
|
121
|
+
|
|
122
|
+
This is the seam that was missing: the runner calls `adapter.authenticate()`
|
|
123
|
+
and friends, but a `MediumAdapter` built by the registry has `_page = None`
|
|
124
|
+
until something puts a page in it. That something is here.
|
|
125
|
+
|
|
126
|
+
API adapters pass through untouched, and no browser is launched at all if
|
|
127
|
+
none of the adapters need one — publishing to Dev.to should not pay for
|
|
128
|
+
Chromium.
|
|
129
|
+
"""
|
|
130
|
+
needs_browser = [a for a in adapters if getattr(a, "name", None) in BROWSER_PLATFORMS]
|
|
131
|
+
if not needs_browser:
|
|
132
|
+
yield list(adapters)
|
|
133
|
+
return
|
|
134
|
+
|
|
135
|
+
# Check every session BEFORE a browser exists. A missing login is the most
|
|
136
|
+
# common reason a publish cannot start, and spending a Chromium launch to
|
|
137
|
+
# then say "run pubkit auth login" is both slow and the wrong order — it
|
|
138
|
+
# also leaves a browser process to clean up on the failure path.
|
|
139
|
+
missing = [a.name for a in needs_browser if sessions.load(a.name) is None]
|
|
140
|
+
if missing:
|
|
141
|
+
raise CredentialError(
|
|
142
|
+
"no saved session for: " + ", ".join(missing) + ".\n"
|
|
143
|
+
+ "\n".join(f" Run: pubkit auth login {p}" for p in missing)
|
|
144
|
+
+ "\n A browser window opens, you sign in yourself, and pubkit keeps\n"
|
|
145
|
+
" only the session — it never asks for your password."
|
|
146
|
+
)
|
|
147
|
+
|
|
148
|
+
async with BrowserPool(sessions, headless=headless) as pool:
|
|
149
|
+
for adapter in needs_browser:
|
|
150
|
+
page, upload = await pool.page_for(adapter.name)
|
|
151
|
+
adapter._page = page
|
|
152
|
+
adapter._upload = upload
|
|
153
|
+
log.debug("attached a browser page to %s", adapter.name)
|
|
154
|
+
yield list(adapters)
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
async def interactive_login(platform: str, sessions: SessionStore, timeout: float = 300.0) -> None:
|
|
158
|
+
"""Open a visible window, wait for the human, save the session."""
|
|
159
|
+
from playwright.async_api import async_playwright
|
|
160
|
+
|
|
161
|
+
login_url, success_marker = LOGIN_FLOWS.get(
|
|
162
|
+
platform, (f"https://{platform}.com/login", f"{platform}.com")
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
async with async_playwright() as pw:
|
|
166
|
+
browser = await pw.chromium.launch(headless=False)
|
|
167
|
+
ctx = await browser.new_context(viewport={"width": 1280, "height": 900})
|
|
168
|
+
page = await ctx.new_page()
|
|
169
|
+
await page.goto(login_url)
|
|
170
|
+
|
|
171
|
+
print(f"\n A browser window is open at {login_url}")
|
|
172
|
+
print(" Sign in there — password manager, MFA, all of it.")
|
|
173
|
+
print(f" Waiting up to {timeout / 60:.0f} minutes. Close this with Ctrl-C to cancel.\n")
|
|
174
|
+
|
|
175
|
+
deadline = asyncio.get_event_loop().time() + timeout
|
|
176
|
+
while asyncio.get_event_loop().time() < deadline:
|
|
177
|
+
try:
|
|
178
|
+
url = page.url
|
|
179
|
+
except Exception as exc: # window closed by the user
|
|
180
|
+
await browser.close()
|
|
181
|
+
raise CredentialError(
|
|
182
|
+
f"{platform} login window was closed before sign-in completed"
|
|
183
|
+
) from exc
|
|
184
|
+
|
|
185
|
+
if success_marker in url:
|
|
186
|
+
await asyncio.sleep(2) # let the last auth cookie land
|
|
187
|
+
sessions.save(platform, await ctx.storage_state())
|
|
188
|
+
await browser.close()
|
|
189
|
+
return
|
|
190
|
+
await asyncio.sleep(1.5)
|
|
191
|
+
|
|
192
|
+
await browser.close()
|
|
193
|
+
raise TimeoutError(
|
|
194
|
+
f"no {platform} sign-in detected within {timeout / 60:.0f} minutes. "
|
|
195
|
+
f"pubkit watches for a URL containing {success_marker!r}."
|
|
196
|
+
)
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
async def verify_session(platform: str, sessions: SessionStore, *, headless: bool = True) -> bool:
|
|
200
|
+
"""Is the saved session still good? Cheaper to find out now than mid-publish."""
|
|
201
|
+
_, success_marker = LOGIN_FLOWS.get(platform, ("", platform))
|
|
202
|
+
probe = {
|
|
203
|
+
"medium": "https://medium.com/me/stories/drafts",
|
|
204
|
+
"substack": "https://substack.com/home",
|
|
205
|
+
}.get(platform)
|
|
206
|
+
if probe is None or sessions.load(platform) is None:
|
|
207
|
+
return False
|
|
208
|
+
|
|
209
|
+
async with BrowserPool(sessions, headless=headless) as pool:
|
|
210
|
+
page, _ = await pool.page_for(platform)
|
|
211
|
+
await page.goto(probe, wait_until="domcontentloaded")
|
|
212
|
+
await asyncio.sleep(1.5)
|
|
213
|
+
return "signin" not in page.url and "sign-in" not in page.url
|
|
@@ -147,11 +147,17 @@ def publish(
|
|
|
147
147
|
)
|
|
148
148
|
|
|
149
149
|
pipe = Pipeline(StateStore())
|
|
150
|
+
live = confirm and not draft_only
|
|
150
151
|
|
|
151
152
|
async def go():
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
153
|
+
# Browser adapters get a live page here and nowhere else. API-only runs
|
|
154
|
+
# never launch Chromium.
|
|
155
|
+
from .browserctl import attached
|
|
156
|
+
|
|
157
|
+
async with attached(adapters, sessions, headless=headless) as ready:
|
|
158
|
+
if hasattr(target, "documents"):
|
|
159
|
+
return await pipe.run_series(target, ready, ctx_for, publish=live)
|
|
160
|
+
return await pipe.run([target], ready, ctx_for, publish=live)
|
|
155
161
|
|
|
156
162
|
report = asyncio.run(go())
|
|
157
163
|
console.print(f"\n[bold]{report.run_id}[/]")
|
|
@@ -254,10 +260,12 @@ def auth_login(
|
|
|
254
260
|
Browser platforms: a real browser window opens and you sign in yourself —
|
|
255
261
|
pubkit stores only the resulting session, never a password.
|
|
256
262
|
"""
|
|
263
|
+
from .browserctl import BROWSER_PLATFORMS
|
|
264
|
+
|
|
257
265
|
tokens = TokenStore()
|
|
258
|
-
|
|
266
|
+
build_adapter(platform) # fail fast on an unknown platform
|
|
259
267
|
|
|
260
|
-
if
|
|
268
|
+
if platform not in BROWSER_PLATFORMS:
|
|
261
269
|
if token is None:
|
|
262
270
|
token = typer.prompt(f"{platform} API token", hide_input=True)
|
|
263
271
|
tokens.set(platform, token.strip())
|
|
@@ -266,12 +274,15 @@ def auth_login(
|
|
|
266
274
|
|
|
267
275
|
from .browserctl import interactive_login
|
|
268
276
|
|
|
277
|
+
try:
|
|
278
|
+
asyncio.run(interactive_login(platform, SessionStore()))
|
|
279
|
+
except Exception as exc: # noqa: BLE001
|
|
280
|
+
console.print(f"[red]{exc}[/]")
|
|
281
|
+
raise typer.Exit(1) from exc
|
|
269
282
|
console.print(
|
|
270
|
-
f"
|
|
271
|
-
"
|
|
283
|
+
f"[green]saved {platform} session (encrypted)[/]\n"
|
|
284
|
+
f"Verify any time with: pubkit auth verify {platform}"
|
|
272
285
|
)
|
|
273
|
-
asyncio.run(interactive_login(platform, SessionStore()))
|
|
274
|
-
console.print(f"[green]saved {platform} session (encrypted)[/]")
|
|
275
286
|
|
|
276
287
|
|
|
277
288
|
@auth_app.command("logout")
|
|
@@ -281,6 +292,24 @@ def auth_logout(platform: str = typer.Argument(...)):
|
|
|
281
292
|
console.print(f"[green]forgot all {platform} credentials[/]")
|
|
282
293
|
|
|
283
294
|
|
|
295
|
+
@auth_app.command("verify")
|
|
296
|
+
def auth_verify(platform: str = typer.Argument(...)):
|
|
297
|
+
"""Check a saved session is still valid — cheaper now than mid-publish."""
|
|
298
|
+
from .browserctl import BROWSER_PLATFORMS, verify_session
|
|
299
|
+
|
|
300
|
+
if platform not in BROWSER_PLATFORMS:
|
|
301
|
+
tok = TokenStore().get(platform) or TokenStore().get(platform, "bearer")
|
|
302
|
+
console.print(f"[green]{platform}: token present[/]" if tok else f"[red]{platform}: no token[/]")
|
|
303
|
+
raise typer.Exit(0 if tok else 1)
|
|
304
|
+
|
|
305
|
+
ok = asyncio.run(verify_session(platform, SessionStore()))
|
|
306
|
+
if ok:
|
|
307
|
+
console.print(f"[green]{platform}: session valid[/]")
|
|
308
|
+
else:
|
|
309
|
+
console.print(f"[red]{platform}: session missing or expired[/] — run `pubkit auth login {platform}`")
|
|
310
|
+
raise typer.Exit(0 if ok else 1)
|
|
311
|
+
|
|
312
|
+
|
|
284
313
|
@auth_app.command("list")
|
|
285
314
|
def auth_list():
|
|
286
315
|
tokens, sessions = TokenStore(), SessionStore()
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
# Copyright 2026 The pubkit Authors
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
"""Materialise the assets a plan promised.
|
|
4
|
+
|
|
5
|
+
The planner says "8 tables → images" and lists the asset ids. Something has to
|
|
6
|
+
actually produce those files, and it has to happen per-platform, because the
|
|
7
|
+
same document keeps native tables on Dev.to and needs images on Medium.
|
|
8
|
+
|
|
9
|
+
This runs between planning and publishing, and is cached on content: a table
|
|
10
|
+
whose cells have not changed keeps its rendered file, so a re-run does not
|
|
11
|
+
re-render or re-upload it.
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import logging
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
from ..render.tables import RenderUnavailable, TableStyle, render_table, table_asset_id
|
|
19
|
+
from .capabilities import Capabilities
|
|
20
|
+
from .ir import Asset, Document, Figure, Table
|
|
21
|
+
|
|
22
|
+
log = logging.getLogger(__name__)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def materialise(
|
|
26
|
+
doc: Document,
|
|
27
|
+
caps: Capabilities,
|
|
28
|
+
workdir: Path,
|
|
29
|
+
*,
|
|
30
|
+
style: TableStyle | None = None,
|
|
31
|
+
) -> Document:
|
|
32
|
+
"""Return a copy of `doc` adapted to what `caps` can actually render.
|
|
33
|
+
|
|
34
|
+
Tables become Figures on platforms without table support; everything else is
|
|
35
|
+
left alone. The original document is never mutated — the same IR has to
|
|
36
|
+
serve every platform in the same run.
|
|
37
|
+
"""
|
|
38
|
+
if caps.tables or not doc.tables:
|
|
39
|
+
return doc
|
|
40
|
+
|
|
41
|
+
if caps.image_upload == "none":
|
|
42
|
+
log.warning(
|
|
43
|
+
"%s supports neither tables nor images; %d table(s) will degrade to lists",
|
|
44
|
+
getattr(caps, "name", "platform"),
|
|
45
|
+
len(doc.tables),
|
|
46
|
+
)
|
|
47
|
+
return doc
|
|
48
|
+
|
|
49
|
+
out_dir = workdir / "rendered"
|
|
50
|
+
new_blocks: list = []
|
|
51
|
+
new_assets = dict(doc.assets)
|
|
52
|
+
rendered = 0
|
|
53
|
+
reused = 0
|
|
54
|
+
n = 0
|
|
55
|
+
|
|
56
|
+
for block in doc.blocks:
|
|
57
|
+
if not isinstance(block, Table):
|
|
58
|
+
new_blocks.append(block)
|
|
59
|
+
continue
|
|
60
|
+
|
|
61
|
+
n += 1
|
|
62
|
+
aid = table_asset_id(block, n)
|
|
63
|
+
path = out_dir / f"{aid}.png"
|
|
64
|
+
|
|
65
|
+
if not path.exists():
|
|
66
|
+
try:
|
|
67
|
+
render_table(block, path, style)
|
|
68
|
+
rendered += 1
|
|
69
|
+
except RenderUnavailable:
|
|
70
|
+
# Better a readable list than a crash at publish time.
|
|
71
|
+
log.warning("Pillow missing — table %d degrades to a list", n)
|
|
72
|
+
from .ir import ListBlock
|
|
73
|
+
|
|
74
|
+
new_blocks.append(
|
|
75
|
+
ListBlock(
|
|
76
|
+
ordered=False,
|
|
77
|
+
items=[
|
|
78
|
+
" · ".join(f"{h}: {c}" for h, c in zip(block.header, row, strict=False))
|
|
79
|
+
for row in block.rows
|
|
80
|
+
],
|
|
81
|
+
)
|
|
82
|
+
)
|
|
83
|
+
continue
|
|
84
|
+
else:
|
|
85
|
+
reused += 1
|
|
86
|
+
|
|
87
|
+
new_assets[aid] = Asset(
|
|
88
|
+
id=aid, path=path, alt=block.caption or "table", generated_from=f"table[{n}]"
|
|
89
|
+
)
|
|
90
|
+
new_blocks.append(Figure(asset_id=aid, caption=block.caption, alt=block.caption or "table"))
|
|
91
|
+
|
|
92
|
+
if rendered or reused:
|
|
93
|
+
log.info("tables → images: %d rendered, %d reused from cache", rendered, reused)
|
|
94
|
+
|
|
95
|
+
adapted = doc.model_copy(update={"blocks": new_blocks, "assets": new_assets})
|
|
96
|
+
return adapted
|
|
@@ -221,7 +221,7 @@ class BrowserAdapter(BaseAdapter):
|
|
|
221
221
|
|
|
222
222
|
selectors: EditorSelectors
|
|
223
223
|
login_url: str = ""
|
|
224
|
-
marker_regex: str = r"
|
|
224
|
+
marker_regex: str = r"\[\[\s*IMAGE" # a JS RegExp source string, not a Python pattern
|
|
225
225
|
|
|
226
226
|
def __init__(self) -> None:
|
|
227
227
|
super().__init__()
|