pubkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pubkit/core/checks.py ADDED
@@ -0,0 +1,344 @@
1
+ # Copyright 2026 The pubkit Authors
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Pre-flight checks.
4
+
5
+ These exist because of a specific, humbling experience: a scripted pass over 32
6
+ numeric claims in a finished article found 5 that were wrong, including two
7
+ ratios that had been transposed during an edit. Every one of the 5 was real.
8
+ Prose does not have a type checker; this is the closest thing.
9
+
10
+ Checks are pluggable (`pubkit.checks` entry point). Ship your own house-style
11
+ rule and it runs in the same pipeline.
12
+ """
13
+ from __future__ import annotations
14
+
15
+ import re
16
+ from collections.abc import Callable, Iterable
17
+ from dataclasses import dataclass, field
18
+ from enum import Enum
19
+ from typing import Protocol
20
+
21
+ from .ir import Callout, Document, Heading, ListBlock, Paragraph, Quote, Table
22
+
23
+
24
+ class Severity(str, Enum):
25
+ ERROR = "error"
26
+ WARN = "warn"
27
+ INFO = "info"
28
+
29
+
30
+ @dataclass
31
+ class Finding:
32
+ check: str
33
+ severity: Severity
34
+ message: str
35
+ where: str = ""
36
+
37
+ def __str__(self) -> str:
38
+ loc = f" [{self.where}]" if self.where else ""
39
+ return f"{self.severity.value.upper():5} {self.check}{loc}: {self.message}"
40
+
41
+
42
+ @dataclass
43
+ class CheckResult:
44
+ findings: list[Finding] = field(default_factory=list)
45
+
46
+ @property
47
+ def errors(self) -> list[Finding]:
48
+ return [f for f in self.findings if f.severity is Severity.ERROR]
49
+
50
+ @property
51
+ def ok(self) -> bool:
52
+ return not self.errors
53
+
54
+ def extend(self, other: CheckResult) -> None:
55
+ self.findings.extend(other.findings)
56
+
57
+
58
+ class Check(Protocol):
59
+ name: str
60
+
61
+ def __call__(self, doc: Document, ctx: dict) -> CheckResult: ...
62
+
63
+
64
+ def _text_blocks(doc: Document) -> Iterable[tuple[str, str]]:
65
+ for i, b in enumerate(doc.blocks):
66
+ if isinstance(b, (Paragraph, Quote, Callout)):
67
+ yield f"block[{i}]", b.text
68
+ elif isinstance(b, Heading):
69
+ yield f"heading[{i}]", b.text
70
+ elif isinstance(b, ListBlock):
71
+ for j, item in enumerate(b.items):
72
+ yield f"block[{i}].item[{j}]", item
73
+ elif isinstance(b, Table):
74
+ for r, row in enumerate(b.rows):
75
+ yield f"block[{i}].row[{r}]", " | ".join(row)
76
+
77
+
78
+ # --------------------------------------------------------------------------
79
+ # placeholders (failure C3)
80
+ # --------------------------------------------------------------------------
81
+ PLACEHOLDER_RE = re.compile(r"(\$\{[^}]+\}|URL-PART-\d+|TODO|TK|FIXME|XXX|\[\[\s*IMAGE)", re.I)
82
+ SERIES_TOKEN_RE = re.compile(r"\$\{series\.part(\d+)\.url\}")
83
+
84
+
85
+ def check_placeholders(doc: Document, ctx: dict) -> CheckResult:
86
+ """No placeholder token may reach a published document.
87
+
88
+ `URL-PART-2` links once existed in a draft because part 2 had no URL yet —
89
+ the chicken-and-egg every multi-document publish hits. The fix is the
90
+ two-phase publish; this check is the guard rail that proves it ran.
91
+ """
92
+ res = CheckResult()
93
+ allow: set[str] = set(ctx.get("allow_placeholders", ()))
94
+ # `${series.partN.url}` is *expected* to be unresolved before phase 1 of a
95
+ # two-phase publish — the sibling has no URL yet. It is only an error if it
96
+ # names a part that does not exist, or if it is still there at publish time
97
+ # (which is what `post_resolution` mode checks).
98
+ post = bool(ctx.get("post_resolution"))
99
+ for where, text in _text_blocks(doc):
100
+ for m in PLACEHOLDER_RE.finditer(text):
101
+ tok = m.group(0)
102
+ if tok in allow:
103
+ continue
104
+ sm = SERIES_TOKEN_RE.fullmatch(tok)
105
+ if sm and doc.series:
106
+ idx = int(sm.group(1))
107
+ if idx > doc.series.of:
108
+ res.findings.append(
109
+ Finding(
110
+ "placeholders",
111
+ Severity.ERROR,
112
+ f"{tok!r} names part {idx} but the series has {doc.series.of}",
113
+ where,
114
+ )
115
+ )
116
+ elif post:
117
+ res.findings.append(
118
+ Finding(
119
+ "placeholders",
120
+ Severity.ERROR,
121
+ f"{tok!r} survived link resolution — phase 1 did not record that URL",
122
+ where,
123
+ )
124
+ )
125
+ else:
126
+ res.findings.append(
127
+ Finding("placeholders", Severity.INFO, f"{tok} resolves during publish", where)
128
+ )
129
+ continue
130
+ res.findings.append(
131
+ Finding("placeholders", Severity.ERROR, f"unresolved placeholder {tok!r}", where)
132
+ )
133
+ return res
134
+
135
+
136
+ # --------------------------------------------------------------------------
137
+ # numeric consistency (failure C1)
138
+ # --------------------------------------------------------------------------
139
+ _NUM = r"(?:\d[\d,]*(?:\.\d+)?)"
140
+ RATIO_RE = re.compile(rf"({_NUM})\s*(?:×|x|times)\b", re.I)
141
+ DEFINE_RE = re.compile(rf"<!--\s*pubkit:define\s+(\w+)\s*=\s*({_NUM})\s*-->")
142
+ ASSERT_RE = re.compile(
143
+ rf"<!--\s*pubkit:assert\s+({_NUM})\s*(?:×|x)?\s*=\s*(\w+)\s*/\s*(\w+)\s*(?:±\s*({_NUM})%)?\s*-->"
144
+ )
145
+
146
+
147
+ def check_numeric(doc: Document, ctx: dict) -> CheckResult:
148
+ """Re-derive every ratio the author asserted from its declared inputs.
149
+
150
+ Authors annotate source values once:
151
+
152
+ <!-- pubkit:define nvlink = 900 -->
153
+ <!-- pubkit:define pcie5 = 128 -->
154
+
155
+ and then assert derived claims where they appear:
156
+
157
+ NVLink is 7× wider than PCIe Gen5. <!-- pubkit:assert 7x = nvlink/pcie5 ±10% -->
158
+
159
+ This is what catches a transposition. Two ratios were once swapped in an
160
+ edit — 18× and 7× traded places — and no amount of re-reading found it.
161
+ Arithmetic did, in under a second.
162
+ """
163
+ res = CheckResult()
164
+ defines: dict[str, float] = {}
165
+ raw = "\n".join(t for _, t in _text_blocks(doc))
166
+ for name, val in DEFINE_RE.findall(raw):
167
+ defines[name] = float(val.replace(",", ""))
168
+
169
+ for where, text in _text_blocks(doc):
170
+ for claimed, num, den, tol in ASSERT_RE.findall(text):
171
+ if num not in defines or den not in defines:
172
+ res.findings.append(
173
+ Finding("numeric", Severity.ERROR, f"assert references undefined {num!r}/{den!r}", where)
174
+ )
175
+ continue
176
+ if defines[den] == 0:
177
+ res.findings.append(Finding("numeric", Severity.ERROR, f"{den!r} is zero", where))
178
+ continue
179
+ actual = defines[num] / defines[den]
180
+ want = float(claimed.replace(",", ""))
181
+ tolerance = float(tol) / 100 if tol else 0.10
182
+ if want == 0 or abs(actual - want) / max(abs(want), 1e-9) > tolerance:
183
+ res.findings.append(
184
+ Finding(
185
+ "numeric",
186
+ Severity.ERROR,
187
+ f"claimed {want:g}× but {num}/{den} = {actual:.3g}× "
188
+ f"({defines[num]:g}/{defines[den]:g}), outside ±{tolerance:.0%}",
189
+ where,
190
+ )
191
+ )
192
+ return res
193
+
194
+
195
+ def check_number_drift(doc: Document, ctx: dict) -> CheckResult:
196
+ """Flag the same quantity stated with two different values.
197
+
198
+ One section said 5,000 tok/s and another derived 5,150 from the same
199
+ assumptions; three cost figures downstream inherited the discrepancy.
200
+ """
201
+ res = CheckResult()
202
+ pattern = re.compile(rf"({_NUM})\s*(tok/s|tokens/s|GB/s|TB/s|GB|ms|µs|%|\$/M)", re.I)
203
+ seen: dict[str, set[str]] = {}
204
+ for _where, text in _text_blocks(doc):
205
+ for val, unit in pattern.findall(text):
206
+ key = unit.lower()
207
+ seen.setdefault(key, set()).add(val.replace(",", ""))
208
+ for unit, values in seen.items():
209
+ if len(values) > 6: # a table of varied figures, not a drift signal
210
+ continue
211
+ floats = sorted(float(v) for v in values)
212
+ for a, b in zip(floats, floats[1:], strict=False):
213
+ if a and abs(b - a) / a < 0.08 and a != b:
214
+ res.findings.append(
215
+ Finding(
216
+ "number-drift",
217
+ Severity.WARN,
218
+ f"{a:g} and {b:g} {unit} differ by <8% — same quantity stated twice?",
219
+ )
220
+ )
221
+ return res
222
+
223
+
224
+ # --------------------------------------------------------------------------
225
+ # cross-references (failure C2)
226
+ # --------------------------------------------------------------------------
227
+ XREF_RE = re.compile(r"\b[Pp]art\s+(\d+)\b")
228
+
229
+
230
+ def check_xrefs(doc: Document, ctx: dict) -> CheckResult:
231
+ """"Part N" must resolve, and must not collide with section numbering.
232
+
233
+ Splitting one article into three turned seven internal references into
234
+ lies, and made the phrase "Part 1" ambiguous between a series part and a
235
+ section. Sections are numbered by generator now; this check enforces that
236
+ "Part N" only ever means the series.
237
+ """
238
+ res = CheckResult()
239
+ if doc.series is None:
240
+ for where, text in _text_blocks(doc):
241
+ if XREF_RE.search(text):
242
+ res.findings.append(
243
+ Finding("xref", Severity.WARN, "'Part N' used but document declares no series", where)
244
+ )
245
+ return res
246
+
247
+ for where, text in _text_blocks(doc):
248
+ for n in XREF_RE.findall(text):
249
+ idx = int(n)
250
+ if idx < 1 or idx > doc.series.of:
251
+ res.findings.append(
252
+ Finding(
253
+ "xref",
254
+ Severity.ERROR,
255
+ f"refers to Part {idx} but the series has {doc.series.of} parts",
256
+ where,
257
+ )
258
+ )
259
+ elif idx == doc.series.index and "this" not in text.lower():
260
+ res.findings.append(
261
+ Finding(
262
+ "xref",
263
+ Severity.WARN,
264
+ f"Part {idx} refers to itself — probably a leftover from a split",
265
+ where,
266
+ )
267
+ )
268
+ return res
269
+
270
+
271
+ # --------------------------------------------------------------------------
272
+ # assets, budget, structure
273
+ # --------------------------------------------------------------------------
274
+ def check_assets(doc: Document, ctx: dict) -> CheckResult:
275
+ res = CheckResult()
276
+ for i, fig in enumerate(doc.figures):
277
+ asset = doc.assets.get(fig.asset_id)
278
+ if asset is None:
279
+ res.findings.append(
280
+ Finding("assets", Severity.ERROR, f"figure references unknown asset {fig.asset_id!r}", f"figure[{i}]")
281
+ )
282
+ continue
283
+ if not asset.path.exists() and asset.generated_from is None:
284
+ res.findings.append(
285
+ Finding("assets", Severity.ERROR, f"asset file missing: {asset.path}", f"figure[{i}]")
286
+ )
287
+ if not (fig.alt or asset.alt):
288
+ res.findings.append(
289
+ Finding("assets", Severity.WARN, f"no alt text for {fig.asset_id!r}", f"figure[{i}]")
290
+ )
291
+ unused = set(doc.assets) - {f.asset_id for f in doc.figures}
292
+ for a in sorted(unused):
293
+ res.findings.append(Finding("assets", Severity.INFO, f"asset {a!r} declared but never used"))
294
+ return res
295
+
296
+
297
+ def check_budget(doc: Document, ctx: dict) -> CheckResult:
298
+ res = CheckResult()
299
+ if not doc.budget or not doc.budget.words:
300
+ return res
301
+ target, tol = doc.budget.words, doc.budget.tolerance
302
+ actual = doc.word_count
303
+ if abs(actual - target) / target > tol:
304
+ res.findings.append(
305
+ Finding(
306
+ "budget",
307
+ Severity.WARN,
308
+ f"{actual:,} words against a target of {target:,} (±{tol:.0%})",
309
+ )
310
+ )
311
+ return res
312
+
313
+
314
+ def check_structure(doc: Document, ctx: dict) -> CheckResult:
315
+ res = CheckResult()
316
+ if not doc.title.strip():
317
+ res.findings.append(Finding("structure", Severity.ERROR, "empty title"))
318
+ levels = [b.level for b in doc.blocks if isinstance(b, Heading)]
319
+ for a, b in zip(levels, levels[1:], strict=False):
320
+ if b > a + 1:
321
+ res.findings.append(
322
+ Finding("structure", Severity.WARN, f"heading level jumps h{a} → h{b}")
323
+ )
324
+ break
325
+ return res
326
+
327
+
328
+ DEFAULT_CHECKS: list[tuple[str, Callable[[Document, dict], CheckResult]]] = [
329
+ ("structure", check_structure),
330
+ ("placeholders", check_placeholders),
331
+ ("numeric", check_numeric),
332
+ ("number-drift", check_number_drift),
333
+ ("xref", check_xrefs),
334
+ ("assets", check_assets),
335
+ ("budget", check_budget),
336
+ ]
337
+
338
+
339
+ def run_checks(doc: Document, ctx: dict | None = None, extra: list | None = None) -> CheckResult:
340
+ ctx = ctx or {}
341
+ out = CheckResult()
342
+ for _name, fn in DEFAULT_CHECKS + list(extra or []):
343
+ out.extend(fn(doc, ctx))
344
+ return out
pubkit/core/ir.py ADDED
@@ -0,0 +1,240 @@
1
+ # Copyright 2026 The pubkit Authors
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """The canonical Post IR.
4
+
5
+ Everything upstream (loaders) produces this; everything downstream (checks,
6
+ planner, adapters) consumes it. No platform detail is allowed in this module —
7
+ that is the whole point of it existing.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import hashlib
12
+ import json
13
+ from enum import Enum
14
+ from pathlib import Path
15
+ from typing import Annotated, Literal
16
+
17
+ from pydantic import BaseModel, Field, field_validator
18
+
19
+
20
+ class BlockType(str, Enum):
21
+ HEADING = "heading"
22
+ PARAGRAPH = "paragraph"
23
+ CODE = "code"
24
+ QUOTE = "quote"
25
+ LIST = "list"
26
+ TABLE = "table"
27
+ FIGURE = "figure"
28
+ EMBED = "embed"
29
+ RULE = "rule"
30
+ CALLOUT = "callout"
31
+
32
+
33
+ class _Block(BaseModel):
34
+ model_config = {"extra": "forbid"}
35
+
36
+
37
+ class Heading(_Block):
38
+ type: Literal[BlockType.HEADING] = BlockType.HEADING
39
+ level: int = Field(ge=1, le=6)
40
+ text: str
41
+ #: Generated by the numbering pass. Never authored by hand (failure C2).
42
+ number: str | None = None
43
+ anchor: str | None = None
44
+
45
+
46
+ class Paragraph(_Block):
47
+ type: Literal[BlockType.PARAGRAPH] = BlockType.PARAGRAPH
48
+ #: Inline markdown is permitted here; adapters degrade it as needed.
49
+ text: str
50
+
51
+
52
+ class Code(_Block):
53
+ type: Literal[BlockType.CODE] = BlockType.CODE
54
+ language: str = ""
55
+ text: str
56
+
57
+
58
+ class Quote(_Block):
59
+ type: Literal[BlockType.QUOTE] = BlockType.QUOTE
60
+ text: str
61
+ attribution: str | None = None
62
+
63
+
64
+ class ListBlock(_Block):
65
+ type: Literal[BlockType.LIST] = BlockType.LIST
66
+ ordered: bool = False
67
+ items: list[str]
68
+
69
+
70
+ class Table(_Block):
71
+ """A *semantic* table.
72
+
73
+ Whether this ends up as a real table, a rendered PNG or a bulleted list is
74
+ the planner's decision, taken per platform. Keeping it semantic here is
75
+ what lets one source serve Medium (no tables) and Dev.to (tables) without
76
+ forking the content.
77
+ """
78
+
79
+ type: Literal[BlockType.TABLE] = BlockType.TABLE
80
+ caption: str | None = None
81
+ header: list[str]
82
+ rows: list[list[str]]
83
+ #: Optional per-column alignment: "l" | "c" | "r"
84
+ align: list[str] | None = None
85
+
86
+ @field_validator("rows")
87
+ @classmethod
88
+ def _rectangular(cls, rows: list[list[str]], info) -> list[list[str]]:
89
+ header = info.data.get("header") or []
90
+ for i, row in enumerate(rows):
91
+ if len(row) != len(header):
92
+ raise ValueError(
93
+ f"row {i} has {len(row)} cells, header has {len(header)}"
94
+ )
95
+ return rows
96
+
97
+
98
+ class Figure(_Block):
99
+ type: Literal[BlockType.FIGURE] = BlockType.FIGURE
100
+ #: Reference into Document.assets. Never a filesystem path (failure B5/B6).
101
+ asset_id: str
102
+ caption: str | None = None
103
+ alt: str | None = None
104
+
105
+
106
+ class Embed(_Block):
107
+ type: Literal[BlockType.EMBED] = BlockType.EMBED
108
+ url: str
109
+ kind: Literal["youtube", "gist", "tweet", "codepen", "generic"] = "generic"
110
+
111
+
112
+ class Rule(_Block):
113
+ type: Literal[BlockType.RULE] = BlockType.RULE
114
+
115
+
116
+ class Callout(_Block):
117
+ type: Literal[BlockType.CALLOUT] = BlockType.CALLOUT
118
+ style: Literal["note", "warning", "tip", "key"] = "note"
119
+ title: str | None = None
120
+ text: str
121
+
122
+
123
+ Block = Annotated[
124
+ Heading | Paragraph | Code | Quote | ListBlock | Table | Figure | Embed | Rule | Callout,
125
+ Field(discriminator="type"),
126
+ ]
127
+
128
+
129
+ class Asset(BaseModel):
130
+ model_config = {"extra": "forbid"}
131
+
132
+ id: str
133
+ path: Path
134
+ alt: str = ""
135
+ #: Set by the renderer when an asset is *generated* (e.g. a table PNG).
136
+ generated_from: str | None = None
137
+
138
+ @property
139
+ def mime(self) -> str:
140
+ return {
141
+ ".png": "image/png",
142
+ ".jpg": "image/jpeg",
143
+ ".jpeg": "image/jpeg",
144
+ ".gif": "image/gif",
145
+ ".webp": "image/webp",
146
+ ".svg": "image/svg+xml",
147
+ }.get(self.path.suffix.lower(), "application/octet-stream")
148
+
149
+ @property
150
+ def animated(self) -> bool:
151
+ return self.path.suffix.lower() == ".gif"
152
+
153
+
154
+ class SeriesRef(BaseModel):
155
+ model_config = {"extra": "forbid"}
156
+
157
+ id: str
158
+ index: int = Field(ge=1)
159
+ of: int = Field(ge=1)
160
+ #: Populated during phase 1 of a two-phase publish (failure C3).
161
+ sibling_urls: dict[int, str] = Field(default_factory=dict)
162
+
163
+
164
+ class Budget(BaseModel):
165
+ """Length constraints, checked pre-flight (failure C4)."""
166
+
167
+ model_config = {"extra": "forbid"}
168
+
169
+ words: int | None = None
170
+ tolerance: float = 0.15
171
+
172
+
173
+ class Document(BaseModel):
174
+ model_config = {"extra": "forbid"}
175
+
176
+ id: str
177
+ title: str
178
+ subtitle: str | None = None
179
+ tags: list[str] = Field(default_factory=list)
180
+ canonical_url: str | None = None
181
+ blocks: list[Block] = Field(default_factory=list)
182
+ assets: dict[str, Asset] = Field(default_factory=dict)
183
+ series: SeriesRef | None = None
184
+ budget: Budget | None = None
185
+ #: Free-form, passed through to adapters that understand it.
186
+ platform_overrides: dict[str, dict] = Field(default_factory=dict)
187
+
188
+ # ---------------------------------------------------------------- helpers
189
+ @property
190
+ def word_count(self) -> int:
191
+ parts: list[str] = []
192
+ for b in self.blocks:
193
+ if isinstance(b, (Paragraph, Quote, Callout)):
194
+ parts.append(b.text)
195
+ elif isinstance(b, Heading):
196
+ parts.append(b.text)
197
+ elif isinstance(b, ListBlock):
198
+ parts.extend(b.items)
199
+ elif isinstance(b, Table):
200
+ parts.extend(b.header)
201
+ for r in b.rows:
202
+ parts.extend(r)
203
+ return len(" ".join(parts).split())
204
+
205
+ @property
206
+ def figures(self) -> list[Figure]:
207
+ return [b for b in self.blocks if isinstance(b, Figure)]
208
+
209
+ @property
210
+ def tables(self) -> list[Table]:
211
+ return [b for b in self.blocks if isinstance(b, Table)]
212
+
213
+ def canonical_json(self) -> str:
214
+ """Stable serialisation used for the content id.
215
+
216
+ Sorted keys, no whitespace, UTF-8 preserved. Note `ensure_ascii=False`:
217
+ escaping `×` into `\\u00d7` is how a correction pipeline silently
218
+ stopped matching once already (failure C5).
219
+ """
220
+ payload = self.model_dump(mode="json", exclude={"series"})
221
+ return json.dumps(payload, sort_keys=True, separators=(",", ":"), ensure_ascii=False)
222
+
223
+ @property
224
+ def content_id(self) -> str:
225
+ """Idempotency key. Same content ⇒ same id ⇒ update, never re-create."""
226
+ return hashlib.blake2b(self.canonical_json().encode("utf-8"), digest_size=16).hexdigest()
227
+
228
+
229
+ class Series(BaseModel):
230
+ model_config = {"extra": "forbid"}
231
+
232
+ id: str
233
+ title: str
234
+ documents: list[Document]
235
+
236
+ def by_index(self, i: int) -> Document:
237
+ for d in self.documents:
238
+ if d.series and d.series.index == i:
239
+ return d
240
+ raise KeyError(f"no document at series index {i}")