pubkit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pubkit/__init__.py +14 -0
- pubkit/__main__.py +7 -0
- pubkit/adapters/__init__.py +2 -0
- pubkit/adapters/api_base.py +67 -0
- pubkit/adapters/devto.py +195 -0
- pubkit/adapters/medium.py +184 -0
- pubkit/adapters/substack.py +136 -0
- pubkit/adapters/x.py +216 -0
- pubkit/browserctl.py +78 -0
- pubkit/cli.py +306 -0
- pubkit/core/__init__.py +2 -0
- pubkit/core/adapter.py +206 -0
- pubkit/core/anchors.py +89 -0
- pubkit/core/auth.py +209 -0
- pubkit/core/browser.py +426 -0
- pubkit/core/capabilities.py +256 -0
- pubkit/core/checks.py +344 -0
- pubkit/core/ir.py +240 -0
- pubkit/core/loader.py +278 -0
- pubkit/core/runner.py +268 -0
- pubkit/core/state.py +213 -0
- pubkit/core/transport.py +170 -0
- pubkit/py.typed +0 -0
- pubkit/registry.py +81 -0
- pubkit/render/__init__.py +2 -0
- pubkit/render/html.py +221 -0
- pubkit/scaffold.py +271 -0
- pubkit/workflows/__init__.py +2 -0
- pubkit/workflows/airflow.py +115 -0
- pubkit-0.1.0.dist-info/METADATA +291 -0
- pubkit-0.1.0.dist-info/RECORD +35 -0
- pubkit-0.1.0.dist-info/WHEEL +4 -0
- pubkit-0.1.0.dist-info/entry_points.txt +2 -0
- pubkit-0.1.0.dist-info/licenses/LICENSE +202 -0
- pubkit-0.1.0.dist-info/licenses/NOTICE +7 -0
pubkit/core/checks.py
ADDED
|
@@ -0,0 +1,344 @@
|
|
|
1
|
+
# Copyright 2026 The pubkit Authors
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
"""Pre-flight checks.
|
|
4
|
+
|
|
5
|
+
These exist because of a specific, humbling experience: a scripted pass over 32
|
|
6
|
+
numeric claims in a finished article found 5 that were wrong, including two
|
|
7
|
+
ratios that had been transposed during an edit. Every one of the 5 was real.
|
|
8
|
+
Prose does not have a type checker; this is the closest thing.
|
|
9
|
+
|
|
10
|
+
Checks are pluggable (`pubkit.checks` entry point). Ship your own house-style
|
|
11
|
+
rule and it runs in the same pipeline.
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import re
|
|
16
|
+
from collections.abc import Callable, Iterable
|
|
17
|
+
from dataclasses import dataclass, field
|
|
18
|
+
from enum import Enum
|
|
19
|
+
from typing import Protocol
|
|
20
|
+
|
|
21
|
+
from .ir import Callout, Document, Heading, ListBlock, Paragraph, Quote, Table
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class Severity(str, Enum):
|
|
25
|
+
ERROR = "error"
|
|
26
|
+
WARN = "warn"
|
|
27
|
+
INFO = "info"
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class Finding:
|
|
32
|
+
check: str
|
|
33
|
+
severity: Severity
|
|
34
|
+
message: str
|
|
35
|
+
where: str = ""
|
|
36
|
+
|
|
37
|
+
def __str__(self) -> str:
|
|
38
|
+
loc = f" [{self.where}]" if self.where else ""
|
|
39
|
+
return f"{self.severity.value.upper():5} {self.check}{loc}: {self.message}"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class CheckResult:
|
|
44
|
+
findings: list[Finding] = field(default_factory=list)
|
|
45
|
+
|
|
46
|
+
@property
|
|
47
|
+
def errors(self) -> list[Finding]:
|
|
48
|
+
return [f for f in self.findings if f.severity is Severity.ERROR]
|
|
49
|
+
|
|
50
|
+
@property
|
|
51
|
+
def ok(self) -> bool:
|
|
52
|
+
return not self.errors
|
|
53
|
+
|
|
54
|
+
def extend(self, other: CheckResult) -> None:
|
|
55
|
+
self.findings.extend(other.findings)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class Check(Protocol):
|
|
59
|
+
name: str
|
|
60
|
+
|
|
61
|
+
def __call__(self, doc: Document, ctx: dict) -> CheckResult: ...
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _text_blocks(doc: Document) -> Iterable[tuple[str, str]]:
|
|
65
|
+
for i, b in enumerate(doc.blocks):
|
|
66
|
+
if isinstance(b, (Paragraph, Quote, Callout)):
|
|
67
|
+
yield f"block[{i}]", b.text
|
|
68
|
+
elif isinstance(b, Heading):
|
|
69
|
+
yield f"heading[{i}]", b.text
|
|
70
|
+
elif isinstance(b, ListBlock):
|
|
71
|
+
for j, item in enumerate(b.items):
|
|
72
|
+
yield f"block[{i}].item[{j}]", item
|
|
73
|
+
elif isinstance(b, Table):
|
|
74
|
+
for r, row in enumerate(b.rows):
|
|
75
|
+
yield f"block[{i}].row[{r}]", " | ".join(row)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
# --------------------------------------------------------------------------
|
|
79
|
+
# placeholders (failure C3)
|
|
80
|
+
# --------------------------------------------------------------------------
|
|
81
|
+
PLACEHOLDER_RE = re.compile(r"(\$\{[^}]+\}|URL-PART-\d+|TODO|TK|FIXME|XXX|\[\[\s*IMAGE)", re.I)
|
|
82
|
+
SERIES_TOKEN_RE = re.compile(r"\$\{series\.part(\d+)\.url\}")
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def check_placeholders(doc: Document, ctx: dict) -> CheckResult:
|
|
86
|
+
"""No placeholder token may reach a published document.
|
|
87
|
+
|
|
88
|
+
`URL-PART-2` links once existed in a draft because part 2 had no URL yet —
|
|
89
|
+
the chicken-and-egg every multi-document publish hits. The fix is the
|
|
90
|
+
two-phase publish; this check is the guard rail that proves it ran.
|
|
91
|
+
"""
|
|
92
|
+
res = CheckResult()
|
|
93
|
+
allow: set[str] = set(ctx.get("allow_placeholders", ()))
|
|
94
|
+
# `${series.partN.url}` is *expected* to be unresolved before phase 1 of a
|
|
95
|
+
# two-phase publish — the sibling has no URL yet. It is only an error if it
|
|
96
|
+
# names a part that does not exist, or if it is still there at publish time
|
|
97
|
+
# (which is what `post_resolution` mode checks).
|
|
98
|
+
post = bool(ctx.get("post_resolution"))
|
|
99
|
+
for where, text in _text_blocks(doc):
|
|
100
|
+
for m in PLACEHOLDER_RE.finditer(text):
|
|
101
|
+
tok = m.group(0)
|
|
102
|
+
if tok in allow:
|
|
103
|
+
continue
|
|
104
|
+
sm = SERIES_TOKEN_RE.fullmatch(tok)
|
|
105
|
+
if sm and doc.series:
|
|
106
|
+
idx = int(sm.group(1))
|
|
107
|
+
if idx > doc.series.of:
|
|
108
|
+
res.findings.append(
|
|
109
|
+
Finding(
|
|
110
|
+
"placeholders",
|
|
111
|
+
Severity.ERROR,
|
|
112
|
+
f"{tok!r} names part {idx} but the series has {doc.series.of}",
|
|
113
|
+
where,
|
|
114
|
+
)
|
|
115
|
+
)
|
|
116
|
+
elif post:
|
|
117
|
+
res.findings.append(
|
|
118
|
+
Finding(
|
|
119
|
+
"placeholders",
|
|
120
|
+
Severity.ERROR,
|
|
121
|
+
f"{tok!r} survived link resolution — phase 1 did not record that URL",
|
|
122
|
+
where,
|
|
123
|
+
)
|
|
124
|
+
)
|
|
125
|
+
else:
|
|
126
|
+
res.findings.append(
|
|
127
|
+
Finding("placeholders", Severity.INFO, f"{tok} resolves during publish", where)
|
|
128
|
+
)
|
|
129
|
+
continue
|
|
130
|
+
res.findings.append(
|
|
131
|
+
Finding("placeholders", Severity.ERROR, f"unresolved placeholder {tok!r}", where)
|
|
132
|
+
)
|
|
133
|
+
return res
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
# --------------------------------------------------------------------------
|
|
137
|
+
# numeric consistency (failure C1)
|
|
138
|
+
# --------------------------------------------------------------------------
|
|
139
|
+
_NUM = r"(?:\d[\d,]*(?:\.\d+)?)"
|
|
140
|
+
RATIO_RE = re.compile(rf"({_NUM})\s*(?:×|x|times)\b", re.I)
|
|
141
|
+
DEFINE_RE = re.compile(rf"<!--\s*pubkit:define\s+(\w+)\s*=\s*({_NUM})\s*-->")
|
|
142
|
+
ASSERT_RE = re.compile(
|
|
143
|
+
rf"<!--\s*pubkit:assert\s+({_NUM})\s*(?:×|x)?\s*=\s*(\w+)\s*/\s*(\w+)\s*(?:±\s*({_NUM})%)?\s*-->"
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def check_numeric(doc: Document, ctx: dict) -> CheckResult:
|
|
148
|
+
"""Re-derive every ratio the author asserted from its declared inputs.
|
|
149
|
+
|
|
150
|
+
Authors annotate source values once:
|
|
151
|
+
|
|
152
|
+
<!-- pubkit:define nvlink = 900 -->
|
|
153
|
+
<!-- pubkit:define pcie5 = 128 -->
|
|
154
|
+
|
|
155
|
+
and then assert derived claims where they appear:
|
|
156
|
+
|
|
157
|
+
NVLink is 7× wider than PCIe Gen5. <!-- pubkit:assert 7x = nvlink/pcie5 ±10% -->
|
|
158
|
+
|
|
159
|
+
This is what catches a transposition. Two ratios were once swapped in an
|
|
160
|
+
edit — 18× and 7× traded places — and no amount of re-reading found it.
|
|
161
|
+
Arithmetic did, in under a second.
|
|
162
|
+
"""
|
|
163
|
+
res = CheckResult()
|
|
164
|
+
defines: dict[str, float] = {}
|
|
165
|
+
raw = "\n".join(t for _, t in _text_blocks(doc))
|
|
166
|
+
for name, val in DEFINE_RE.findall(raw):
|
|
167
|
+
defines[name] = float(val.replace(",", ""))
|
|
168
|
+
|
|
169
|
+
for where, text in _text_blocks(doc):
|
|
170
|
+
for claimed, num, den, tol in ASSERT_RE.findall(text):
|
|
171
|
+
if num not in defines or den not in defines:
|
|
172
|
+
res.findings.append(
|
|
173
|
+
Finding("numeric", Severity.ERROR, f"assert references undefined {num!r}/{den!r}", where)
|
|
174
|
+
)
|
|
175
|
+
continue
|
|
176
|
+
if defines[den] == 0:
|
|
177
|
+
res.findings.append(Finding("numeric", Severity.ERROR, f"{den!r} is zero", where))
|
|
178
|
+
continue
|
|
179
|
+
actual = defines[num] / defines[den]
|
|
180
|
+
want = float(claimed.replace(",", ""))
|
|
181
|
+
tolerance = float(tol) / 100 if tol else 0.10
|
|
182
|
+
if want == 0 or abs(actual - want) / max(abs(want), 1e-9) > tolerance:
|
|
183
|
+
res.findings.append(
|
|
184
|
+
Finding(
|
|
185
|
+
"numeric",
|
|
186
|
+
Severity.ERROR,
|
|
187
|
+
f"claimed {want:g}× but {num}/{den} = {actual:.3g}× "
|
|
188
|
+
f"({defines[num]:g}/{defines[den]:g}), outside ±{tolerance:.0%}",
|
|
189
|
+
where,
|
|
190
|
+
)
|
|
191
|
+
)
|
|
192
|
+
return res
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def check_number_drift(doc: Document, ctx: dict) -> CheckResult:
|
|
196
|
+
"""Flag the same quantity stated with two different values.
|
|
197
|
+
|
|
198
|
+
One section said 5,000 tok/s and another derived 5,150 from the same
|
|
199
|
+
assumptions; three cost figures downstream inherited the discrepancy.
|
|
200
|
+
"""
|
|
201
|
+
res = CheckResult()
|
|
202
|
+
pattern = re.compile(rf"({_NUM})\s*(tok/s|tokens/s|GB/s|TB/s|GB|ms|µs|%|\$/M)", re.I)
|
|
203
|
+
seen: dict[str, set[str]] = {}
|
|
204
|
+
for _where, text in _text_blocks(doc):
|
|
205
|
+
for val, unit in pattern.findall(text):
|
|
206
|
+
key = unit.lower()
|
|
207
|
+
seen.setdefault(key, set()).add(val.replace(",", ""))
|
|
208
|
+
for unit, values in seen.items():
|
|
209
|
+
if len(values) > 6: # a table of varied figures, not a drift signal
|
|
210
|
+
continue
|
|
211
|
+
floats = sorted(float(v) for v in values)
|
|
212
|
+
for a, b in zip(floats, floats[1:], strict=False):
|
|
213
|
+
if a and abs(b - a) / a < 0.08 and a != b:
|
|
214
|
+
res.findings.append(
|
|
215
|
+
Finding(
|
|
216
|
+
"number-drift",
|
|
217
|
+
Severity.WARN,
|
|
218
|
+
f"{a:g} and {b:g} {unit} differ by <8% — same quantity stated twice?",
|
|
219
|
+
)
|
|
220
|
+
)
|
|
221
|
+
return res
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
# --------------------------------------------------------------------------
|
|
225
|
+
# cross-references (failure C2)
|
|
226
|
+
# --------------------------------------------------------------------------
|
|
227
|
+
XREF_RE = re.compile(r"\b[Pp]art\s+(\d+)\b")
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def check_xrefs(doc: Document, ctx: dict) -> CheckResult:
|
|
231
|
+
""""Part N" must resolve, and must not collide with section numbering.
|
|
232
|
+
|
|
233
|
+
Splitting one article into three turned seven internal references into
|
|
234
|
+
lies, and made the phrase "Part 1" ambiguous between a series part and a
|
|
235
|
+
section. Sections are numbered by generator now; this check enforces that
|
|
236
|
+
"Part N" only ever means the series.
|
|
237
|
+
"""
|
|
238
|
+
res = CheckResult()
|
|
239
|
+
if doc.series is None:
|
|
240
|
+
for where, text in _text_blocks(doc):
|
|
241
|
+
if XREF_RE.search(text):
|
|
242
|
+
res.findings.append(
|
|
243
|
+
Finding("xref", Severity.WARN, "'Part N' used but document declares no series", where)
|
|
244
|
+
)
|
|
245
|
+
return res
|
|
246
|
+
|
|
247
|
+
for where, text in _text_blocks(doc):
|
|
248
|
+
for n in XREF_RE.findall(text):
|
|
249
|
+
idx = int(n)
|
|
250
|
+
if idx < 1 or idx > doc.series.of:
|
|
251
|
+
res.findings.append(
|
|
252
|
+
Finding(
|
|
253
|
+
"xref",
|
|
254
|
+
Severity.ERROR,
|
|
255
|
+
f"refers to Part {idx} but the series has {doc.series.of} parts",
|
|
256
|
+
where,
|
|
257
|
+
)
|
|
258
|
+
)
|
|
259
|
+
elif idx == doc.series.index and "this" not in text.lower():
|
|
260
|
+
res.findings.append(
|
|
261
|
+
Finding(
|
|
262
|
+
"xref",
|
|
263
|
+
Severity.WARN,
|
|
264
|
+
f"Part {idx} refers to itself — probably a leftover from a split",
|
|
265
|
+
where,
|
|
266
|
+
)
|
|
267
|
+
)
|
|
268
|
+
return res
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
# --------------------------------------------------------------------------
|
|
272
|
+
# assets, budget, structure
|
|
273
|
+
# --------------------------------------------------------------------------
|
|
274
|
+
def check_assets(doc: Document, ctx: dict) -> CheckResult:
|
|
275
|
+
res = CheckResult()
|
|
276
|
+
for i, fig in enumerate(doc.figures):
|
|
277
|
+
asset = doc.assets.get(fig.asset_id)
|
|
278
|
+
if asset is None:
|
|
279
|
+
res.findings.append(
|
|
280
|
+
Finding("assets", Severity.ERROR, f"figure references unknown asset {fig.asset_id!r}", f"figure[{i}]")
|
|
281
|
+
)
|
|
282
|
+
continue
|
|
283
|
+
if not asset.path.exists() and asset.generated_from is None:
|
|
284
|
+
res.findings.append(
|
|
285
|
+
Finding("assets", Severity.ERROR, f"asset file missing: {asset.path}", f"figure[{i}]")
|
|
286
|
+
)
|
|
287
|
+
if not (fig.alt or asset.alt):
|
|
288
|
+
res.findings.append(
|
|
289
|
+
Finding("assets", Severity.WARN, f"no alt text for {fig.asset_id!r}", f"figure[{i}]")
|
|
290
|
+
)
|
|
291
|
+
unused = set(doc.assets) - {f.asset_id for f in doc.figures}
|
|
292
|
+
for a in sorted(unused):
|
|
293
|
+
res.findings.append(Finding("assets", Severity.INFO, f"asset {a!r} declared but never used"))
|
|
294
|
+
return res
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def check_budget(doc: Document, ctx: dict) -> CheckResult:
|
|
298
|
+
res = CheckResult()
|
|
299
|
+
if not doc.budget or not doc.budget.words:
|
|
300
|
+
return res
|
|
301
|
+
target, tol = doc.budget.words, doc.budget.tolerance
|
|
302
|
+
actual = doc.word_count
|
|
303
|
+
if abs(actual - target) / target > tol:
|
|
304
|
+
res.findings.append(
|
|
305
|
+
Finding(
|
|
306
|
+
"budget",
|
|
307
|
+
Severity.WARN,
|
|
308
|
+
f"{actual:,} words against a target of {target:,} (±{tol:.0%})",
|
|
309
|
+
)
|
|
310
|
+
)
|
|
311
|
+
return res
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
def check_structure(doc: Document, ctx: dict) -> CheckResult:
|
|
315
|
+
res = CheckResult()
|
|
316
|
+
if not doc.title.strip():
|
|
317
|
+
res.findings.append(Finding("structure", Severity.ERROR, "empty title"))
|
|
318
|
+
levels = [b.level for b in doc.blocks if isinstance(b, Heading)]
|
|
319
|
+
for a, b in zip(levels, levels[1:], strict=False):
|
|
320
|
+
if b > a + 1:
|
|
321
|
+
res.findings.append(
|
|
322
|
+
Finding("structure", Severity.WARN, f"heading level jumps h{a} → h{b}")
|
|
323
|
+
)
|
|
324
|
+
break
|
|
325
|
+
return res
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
DEFAULT_CHECKS: list[tuple[str, Callable[[Document, dict], CheckResult]]] = [
|
|
329
|
+
("structure", check_structure),
|
|
330
|
+
("placeholders", check_placeholders),
|
|
331
|
+
("numeric", check_numeric),
|
|
332
|
+
("number-drift", check_number_drift),
|
|
333
|
+
("xref", check_xrefs),
|
|
334
|
+
("assets", check_assets),
|
|
335
|
+
("budget", check_budget),
|
|
336
|
+
]
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
def run_checks(doc: Document, ctx: dict | None = None, extra: list | None = None) -> CheckResult:
|
|
340
|
+
ctx = ctx or {}
|
|
341
|
+
out = CheckResult()
|
|
342
|
+
for _name, fn in DEFAULT_CHECKS + list(extra or []):
|
|
343
|
+
out.extend(fn(doc, ctx))
|
|
344
|
+
return out
|
pubkit/core/ir.py
ADDED
|
@@ -0,0 +1,240 @@
|
|
|
1
|
+
# Copyright 2026 The pubkit Authors
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
"""The canonical Post IR.
|
|
4
|
+
|
|
5
|
+
Everything upstream (loaders) produces this; everything downstream (checks,
|
|
6
|
+
planner, adapters) consumes it. No platform detail is allowed in this module —
|
|
7
|
+
that is the whole point of it existing.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import hashlib
|
|
12
|
+
import json
|
|
13
|
+
from enum import Enum
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Annotated, Literal
|
|
16
|
+
|
|
17
|
+
from pydantic import BaseModel, Field, field_validator
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class BlockType(str, Enum):
|
|
21
|
+
HEADING = "heading"
|
|
22
|
+
PARAGRAPH = "paragraph"
|
|
23
|
+
CODE = "code"
|
|
24
|
+
QUOTE = "quote"
|
|
25
|
+
LIST = "list"
|
|
26
|
+
TABLE = "table"
|
|
27
|
+
FIGURE = "figure"
|
|
28
|
+
EMBED = "embed"
|
|
29
|
+
RULE = "rule"
|
|
30
|
+
CALLOUT = "callout"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class _Block(BaseModel):
|
|
34
|
+
model_config = {"extra": "forbid"}
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class Heading(_Block):
|
|
38
|
+
type: Literal[BlockType.HEADING] = BlockType.HEADING
|
|
39
|
+
level: int = Field(ge=1, le=6)
|
|
40
|
+
text: str
|
|
41
|
+
#: Generated by the numbering pass. Never authored by hand (failure C2).
|
|
42
|
+
number: str | None = None
|
|
43
|
+
anchor: str | None = None
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class Paragraph(_Block):
|
|
47
|
+
type: Literal[BlockType.PARAGRAPH] = BlockType.PARAGRAPH
|
|
48
|
+
#: Inline markdown is permitted here; adapters degrade it as needed.
|
|
49
|
+
text: str
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class Code(_Block):
|
|
53
|
+
type: Literal[BlockType.CODE] = BlockType.CODE
|
|
54
|
+
language: str = ""
|
|
55
|
+
text: str
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class Quote(_Block):
|
|
59
|
+
type: Literal[BlockType.QUOTE] = BlockType.QUOTE
|
|
60
|
+
text: str
|
|
61
|
+
attribution: str | None = None
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class ListBlock(_Block):
|
|
65
|
+
type: Literal[BlockType.LIST] = BlockType.LIST
|
|
66
|
+
ordered: bool = False
|
|
67
|
+
items: list[str]
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class Table(_Block):
|
|
71
|
+
"""A *semantic* table.
|
|
72
|
+
|
|
73
|
+
Whether this ends up as a real table, a rendered PNG or a bulleted list is
|
|
74
|
+
the planner's decision, taken per platform. Keeping it semantic here is
|
|
75
|
+
what lets one source serve Medium (no tables) and Dev.to (tables) without
|
|
76
|
+
forking the content.
|
|
77
|
+
"""
|
|
78
|
+
|
|
79
|
+
type: Literal[BlockType.TABLE] = BlockType.TABLE
|
|
80
|
+
caption: str | None = None
|
|
81
|
+
header: list[str]
|
|
82
|
+
rows: list[list[str]]
|
|
83
|
+
#: Optional per-column alignment: "l" | "c" | "r"
|
|
84
|
+
align: list[str] | None = None
|
|
85
|
+
|
|
86
|
+
@field_validator("rows")
|
|
87
|
+
@classmethod
|
|
88
|
+
def _rectangular(cls, rows: list[list[str]], info) -> list[list[str]]:
|
|
89
|
+
header = info.data.get("header") or []
|
|
90
|
+
for i, row in enumerate(rows):
|
|
91
|
+
if len(row) != len(header):
|
|
92
|
+
raise ValueError(
|
|
93
|
+
f"row {i} has {len(row)} cells, header has {len(header)}"
|
|
94
|
+
)
|
|
95
|
+
return rows
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
class Figure(_Block):
|
|
99
|
+
type: Literal[BlockType.FIGURE] = BlockType.FIGURE
|
|
100
|
+
#: Reference into Document.assets. Never a filesystem path (failure B5/B6).
|
|
101
|
+
asset_id: str
|
|
102
|
+
caption: str | None = None
|
|
103
|
+
alt: str | None = None
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
class Embed(_Block):
|
|
107
|
+
type: Literal[BlockType.EMBED] = BlockType.EMBED
|
|
108
|
+
url: str
|
|
109
|
+
kind: Literal["youtube", "gist", "tweet", "codepen", "generic"] = "generic"
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
class Rule(_Block):
|
|
113
|
+
type: Literal[BlockType.RULE] = BlockType.RULE
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
class Callout(_Block):
|
|
117
|
+
type: Literal[BlockType.CALLOUT] = BlockType.CALLOUT
|
|
118
|
+
style: Literal["note", "warning", "tip", "key"] = "note"
|
|
119
|
+
title: str | None = None
|
|
120
|
+
text: str
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
Block = Annotated[
|
|
124
|
+
Heading | Paragraph | Code | Quote | ListBlock | Table | Figure | Embed | Rule | Callout,
|
|
125
|
+
Field(discriminator="type"),
|
|
126
|
+
]
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
class Asset(BaseModel):
|
|
130
|
+
model_config = {"extra": "forbid"}
|
|
131
|
+
|
|
132
|
+
id: str
|
|
133
|
+
path: Path
|
|
134
|
+
alt: str = ""
|
|
135
|
+
#: Set by the renderer when an asset is *generated* (e.g. a table PNG).
|
|
136
|
+
generated_from: str | None = None
|
|
137
|
+
|
|
138
|
+
@property
|
|
139
|
+
def mime(self) -> str:
|
|
140
|
+
return {
|
|
141
|
+
".png": "image/png",
|
|
142
|
+
".jpg": "image/jpeg",
|
|
143
|
+
".jpeg": "image/jpeg",
|
|
144
|
+
".gif": "image/gif",
|
|
145
|
+
".webp": "image/webp",
|
|
146
|
+
".svg": "image/svg+xml",
|
|
147
|
+
}.get(self.path.suffix.lower(), "application/octet-stream")
|
|
148
|
+
|
|
149
|
+
@property
|
|
150
|
+
def animated(self) -> bool:
|
|
151
|
+
return self.path.suffix.lower() == ".gif"
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
class SeriesRef(BaseModel):
|
|
155
|
+
model_config = {"extra": "forbid"}
|
|
156
|
+
|
|
157
|
+
id: str
|
|
158
|
+
index: int = Field(ge=1)
|
|
159
|
+
of: int = Field(ge=1)
|
|
160
|
+
#: Populated during phase 1 of a two-phase publish (failure C3).
|
|
161
|
+
sibling_urls: dict[int, str] = Field(default_factory=dict)
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
class Budget(BaseModel):
|
|
165
|
+
"""Length constraints, checked pre-flight (failure C4)."""
|
|
166
|
+
|
|
167
|
+
model_config = {"extra": "forbid"}
|
|
168
|
+
|
|
169
|
+
words: int | None = None
|
|
170
|
+
tolerance: float = 0.15
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
class Document(BaseModel):
|
|
174
|
+
model_config = {"extra": "forbid"}
|
|
175
|
+
|
|
176
|
+
id: str
|
|
177
|
+
title: str
|
|
178
|
+
subtitle: str | None = None
|
|
179
|
+
tags: list[str] = Field(default_factory=list)
|
|
180
|
+
canonical_url: str | None = None
|
|
181
|
+
blocks: list[Block] = Field(default_factory=list)
|
|
182
|
+
assets: dict[str, Asset] = Field(default_factory=dict)
|
|
183
|
+
series: SeriesRef | None = None
|
|
184
|
+
budget: Budget | None = None
|
|
185
|
+
#: Free-form, passed through to adapters that understand it.
|
|
186
|
+
platform_overrides: dict[str, dict] = Field(default_factory=dict)
|
|
187
|
+
|
|
188
|
+
# ---------------------------------------------------------------- helpers
|
|
189
|
+
@property
|
|
190
|
+
def word_count(self) -> int:
|
|
191
|
+
parts: list[str] = []
|
|
192
|
+
for b in self.blocks:
|
|
193
|
+
if isinstance(b, (Paragraph, Quote, Callout)):
|
|
194
|
+
parts.append(b.text)
|
|
195
|
+
elif isinstance(b, Heading):
|
|
196
|
+
parts.append(b.text)
|
|
197
|
+
elif isinstance(b, ListBlock):
|
|
198
|
+
parts.extend(b.items)
|
|
199
|
+
elif isinstance(b, Table):
|
|
200
|
+
parts.extend(b.header)
|
|
201
|
+
for r in b.rows:
|
|
202
|
+
parts.extend(r)
|
|
203
|
+
return len(" ".join(parts).split())
|
|
204
|
+
|
|
205
|
+
@property
|
|
206
|
+
def figures(self) -> list[Figure]:
|
|
207
|
+
return [b for b in self.blocks if isinstance(b, Figure)]
|
|
208
|
+
|
|
209
|
+
@property
|
|
210
|
+
def tables(self) -> list[Table]:
|
|
211
|
+
return [b for b in self.blocks if isinstance(b, Table)]
|
|
212
|
+
|
|
213
|
+
def canonical_json(self) -> str:
|
|
214
|
+
"""Stable serialisation used for the content id.
|
|
215
|
+
|
|
216
|
+
Sorted keys, no whitespace, UTF-8 preserved. Note `ensure_ascii=False`:
|
|
217
|
+
escaping `×` into `\\u00d7` is how a correction pipeline silently
|
|
218
|
+
stopped matching once already (failure C5).
|
|
219
|
+
"""
|
|
220
|
+
payload = self.model_dump(mode="json", exclude={"series"})
|
|
221
|
+
return json.dumps(payload, sort_keys=True, separators=(",", ":"), ensure_ascii=False)
|
|
222
|
+
|
|
223
|
+
@property
|
|
224
|
+
def content_id(self) -> str:
|
|
225
|
+
"""Idempotency key. Same content ⇒ same id ⇒ update, never re-create."""
|
|
226
|
+
return hashlib.blake2b(self.canonical_json().encode("utf-8"), digest_size=16).hexdigest()
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
class Series(BaseModel):
|
|
230
|
+
model_config = {"extra": "forbid"}
|
|
231
|
+
|
|
232
|
+
id: str
|
|
233
|
+
title: str
|
|
234
|
+
documents: list[Document]
|
|
235
|
+
|
|
236
|
+
def by_index(self, i: int) -> Document:
|
|
237
|
+
for d in self.documents:
|
|
238
|
+
if d.series and d.series.index == i:
|
|
239
|
+
return d
|
|
240
|
+
raise KeyError(f"no document at series index {i}")
|