constant-docs 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- constant_docs/__init__.py +18 -0
- constant_docs/__main__.py +15 -0
- constant_docs/api.py +968 -0
- constant_docs/auto.py +239 -0
- constant_docs/checks.py +494 -0
- constant_docs/cli.py +1085 -0
- constant_docs/config.py +1158 -0
- constant_docs/coverage.py +165 -0
- constant_docs/decisions.py +303 -0
- constant_docs/document.py +438 -0
- constant_docs/fingerprint.py +27 -0
- constant_docs/globs.py +154 -0
- constant_docs/guides/quickstart.md +124 -0
- constant_docs/guides/readme.md +198 -0
- constant_docs/house-style.md +70 -0
- constant_docs/index.py +210 -0
- constant_docs/kinds.py +304 -0
- constant_docs/paths.py +246 -0
- constant_docs/prompts/architecture.md +61 -0
- constant_docs/prompts/cli-reference.md +40 -0
- constant_docs/prompts/config-reference.md +38 -0
- constant_docs/prompts/errors.md +50 -0
- constant_docs/prompts/log.md +22 -0
- constant_docs/prompts/module.md +39 -0
- constant_docs/prompts/spec.md +68 -0
- constant_docs/state.py +273 -0
- constant_docs-0.4.0.dist-info/METADATA +380 -0
- constant_docs-0.4.0.dist-info/RECORD +31 -0
- constant_docs-0.4.0.dist-info/WHEEL +4 -0
- constant_docs-0.4.0.dist-info/entry_points.txt +3 -0
- constant_docs-0.4.0.dist-info/licenses/LICENSE +15 -0
|
@@ -0,0 +1,438 @@
|
|
|
1
|
+
"""Read, write, and validate constant-docs documents.
|
|
2
|
+
|
|
3
|
+
Handles frontmatter parsing/serialization, required-heading checks,
|
|
4
|
+
atomic writes, and the blank-line convention between reserved (OKF) and
|
|
5
|
+
extension keys.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import os
|
|
11
|
+
import re
|
|
12
|
+
import tempfile
|
|
13
|
+
from dataclasses import dataclass, field
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
import yaml
|
|
18
|
+
|
|
19
|
+
# Order of reserved OKF keys in the serialized output
|
|
20
|
+
_RESERVED_KEYS = [
|
|
21
|
+
"type",
|
|
22
|
+
"title",
|
|
23
|
+
"description",
|
|
24
|
+
"tags",
|
|
25
|
+
"created",
|
|
26
|
+
"updated",
|
|
27
|
+
"timestamp",
|
|
28
|
+
]
|
|
29
|
+
|
|
30
|
+
# The `module` kind's sections, in the form configuration writes them.
|
|
31
|
+
#
|
|
32
|
+
# Sections are H2 because a document is a title plus sections, not a set of
|
|
33
|
+
# sections. The title lives in frontmatter, where the Open Knowledge Format
|
|
34
|
+
# already has a key for it, and a body carries no H1 at all. That is what lets
|
|
35
|
+
# an existing markdown document be adopted as a tracked one — the earlier
|
|
36
|
+
# H1-section model could describe nothing that was already written — and it is
|
|
37
|
+
# what a strict OKF consumer requires, reserving H1 for the title.
|
|
38
|
+
DEFAULT_SECTIONS = ["## Purpose", "## Correctness pillars", "## Known failure modes"]
|
|
39
|
+
|
|
40
|
+
_FENCE_RE = re.compile(r"^```.*?^```", re.MULTILINE | re.DOTALL)
|
|
41
|
+
_H1_RE = re.compile(r"^# ", re.MULTILINE)
|
|
42
|
+
|
|
43
|
+
_FM_RE = re.compile(r"^---\n(.*?)\n---\n?(.*)", re.DOTALL)
|
|
44
|
+
|
|
45
|
+
# The same block, written where a Markdown host renders nothing.
|
|
46
|
+
#
|
|
47
|
+
# A document under the docs root is part of an Open Knowledge Format bundle
|
|
48
|
+
# and carries YAML frontmatter, which is what makes the tree conformant. A
|
|
49
|
+
# document declared outside the root is already outside that bundle, and the
|
|
50
|
+
# hosts that render those files — GitHub renders a README on the repository
|
|
51
|
+
# landing page — show frontmatter as a table above the title. That is the
|
|
52
|
+
# first screenful of the most-read file in the repository spent on a hash.
|
|
53
|
+
#
|
|
54
|
+
# The block still lives in the file. The alternative is a sidecar holding the
|
|
55
|
+
# hash, and `hash-in-frontmatter` refused that: two files that must agree,
|
|
56
|
+
# with nothing keeping them in step.
|
|
57
|
+
_COMMENT_RE = re.compile(r"^<!--\s*constant-docs\n(.*?)\n-->\n?(.*)", re.DOTALL)
|
|
58
|
+
|
|
59
|
+
YAML_FRONTMATTER = "yaml"
|
|
60
|
+
COMMENT_FRONTMATTER = "comment"
|
|
61
|
+
FRONTMATTER_STYLES = (YAML_FRONTMATTER, COMMENT_FRONTMATTER)
|
|
62
|
+
|
|
63
|
+
# The single frontmatter key this tool owns. An OKF producer extension: a
|
|
64
|
+
# consumer that knows nothing about constant-docs ignores it, and nothing the
|
|
65
|
+
# host writes can collide with it. Defined here rather than in `api`, because
|
|
66
|
+
# `paths` and `index` read it too and a second spelling of it would let a
|
|
67
|
+
# document be fresh to one caller and malformed to another.
|
|
68
|
+
EXTENSION_KEY = "constant_docs"
|
|
69
|
+
|
|
70
|
+
# The same block under the names earlier versions wrote. Read, never written:
|
|
71
|
+
# a document carrying one of these is migrated on its next write, so a
|
|
72
|
+
# repository that upgrades mid-life ends up with one block rather than two.
|
|
73
|
+
LEGACY_EXTENSION_KEYS = ("module_docs",)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def extension_block(frontmatter: dict[str, Any]) -> dict[str, Any]:
|
|
77
|
+
"""Return this tool's frontmatter block, tolerating pre-0.2 spellings.
|
|
78
|
+
|
|
79
|
+
Returns an empty mapping when the document carries no recognisable block,
|
|
80
|
+
so callers can read a key off the result without a null check.
|
|
81
|
+
"""
|
|
82
|
+
for key in (EXTENSION_KEY, *LEGACY_EXTENSION_KEYS):
|
|
83
|
+
block = frontmatter.get(key)
|
|
84
|
+
if isinstance(block, dict):
|
|
85
|
+
return block
|
|
86
|
+
return {}
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
# The key inside our block that makes a document *ours*.
|
|
90
|
+
#
|
|
91
|
+
# `source_hash` is the tool's entire claim on a file: it says these contents
|
|
92
|
+
# were derived from these source files at this digest. A file without one was
|
|
93
|
+
# never tracked, so nothing about it can have gone stale — and a tool with no
|
|
94
|
+
# stake in a file has no business deleting it.
|
|
95
|
+
#
|
|
96
|
+
# One key rather than the full required set, because the two errors are not
|
|
97
|
+
# equally expensive. Declining to prune a document we did write leaves a stale
|
|
98
|
+
# file that `verify` still reports. Pruning one we did not write destroys
|
|
99
|
+
# somebody's work, silently, and having a `docs/` folder already is the normal
|
|
100
|
+
# way people adopt a documentation tool.
|
|
101
|
+
OWNERSHIP_KEY = "source_hash"
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def is_tracked(frontmatter: dict[str, Any]) -> bool:
|
|
105
|
+
"""Return whether this tool wrote the document *frontmatter* came from.
|
|
106
|
+
|
|
107
|
+
The read is deliberately narrow. The block must be ours — namespaced, or
|
|
108
|
+
under a spelling an earlier version of this tool wrote — and it must carry
|
|
109
|
+
a non-empty :data:`OWNERSHIP_KEY`. A top-level ``source_hash`` does not
|
|
110
|
+
count: the namespacing is the whole reason an OKF producer extension
|
|
111
|
+
cannot collide with its host, and honouring a bare key would give that
|
|
112
|
+
away for the one decision where being wrong costs a file.
|
|
113
|
+
"""
|
|
114
|
+
return bool(str(extension_block(frontmatter).get(OWNERSHIP_KEY, "") or "").strip())
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
class DocumentError(Exception):
|
|
118
|
+
"""A document could not be parsed or is structurally invalid."""
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
@dataclass
|
|
122
|
+
class Document:
|
|
123
|
+
"""An in-memory representation of a module document.
|
|
124
|
+
|
|
125
|
+
``frontmatter`` carries every YAML key from the parse (reserved,
|
|
126
|
+
required, and unknown). ``body`` is the markdown content after the
|
|
127
|
+
closing ``---``.
|
|
128
|
+
"""
|
|
129
|
+
|
|
130
|
+
frontmatter: dict[str, Any] = field(default_factory=dict)
|
|
131
|
+
body: str = ""
|
|
132
|
+
# The headings this document's kind requires. Carried on the document
|
|
133
|
+
# rather than passed to `serialize`, so a document cannot be written
|
|
134
|
+
# against one kind's rules and read back under another's.
|
|
135
|
+
sections: list[str] = field(default_factory=lambda: list(DEFAULT_SECTIONS))
|
|
136
|
+
# How the block is written: YAML frontmatter, or an HTML comment a
|
|
137
|
+
# Markdown host renders as nothing. Set from the kind, and carried on the
|
|
138
|
+
# document for the same reason `sections` is — so a document cannot be
|
|
139
|
+
# written in one form and read back expecting the other.
|
|
140
|
+
frontmatter_style: str = YAML_FRONTMATTER
|
|
141
|
+
|
|
142
|
+
# ------------------------------------------------------------------
|
|
143
|
+
# Serialization
|
|
144
|
+
# ------------------------------------------------------------------
|
|
145
|
+
|
|
146
|
+
def serialize(self) -> str:
|
|
147
|
+
"""Return the full document string (frontmatter + body).
|
|
148
|
+
|
|
149
|
+
Raises :exc:`DocumentError` if required headings are missing.
|
|
150
|
+
"""
|
|
151
|
+
# An H1 in the body duplicates the title, which lives in frontmatter —
|
|
152
|
+
# unless the frontmatter is a comment nobody renders, in which case the
|
|
153
|
+
# document has no visible title at all and needs one. The rule and its
|
|
154
|
+
# exception come from the same place: exactly one rendered H1.
|
|
155
|
+
if self.frontmatter_style != COMMENT_FRONTMATTER:
|
|
156
|
+
offending_h1 = find_h1(self.body)
|
|
157
|
+
if offending_h1 is not None:
|
|
158
|
+
raise DocumentError(
|
|
159
|
+
f"Body contains an H1 heading: {offending_h1!r}. H1 is the "
|
|
160
|
+
f"document's title and lives in frontmatter; sections start "
|
|
161
|
+
f"at H2."
|
|
162
|
+
)
|
|
163
|
+
if not has_required_headings(self.body, self.sections):
|
|
164
|
+
raise DocumentError(
|
|
165
|
+
"Body is missing one or more required headings: "
|
|
166
|
+
+ ", ".join(self.sections)
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
fm = dict(self.frontmatter)
|
|
170
|
+
|
|
171
|
+
# Extract reserved keys in order
|
|
172
|
+
reserved: dict[str, Any] = {}
|
|
173
|
+
for k in _RESERVED_KEYS:
|
|
174
|
+
if k in fm:
|
|
175
|
+
reserved[k] = fm.pop(k)
|
|
176
|
+
|
|
177
|
+
# Remaining keys: sort for deterministic output
|
|
178
|
+
extensions: dict[str, Any] = {}
|
|
179
|
+
for k in sorted(fm):
|
|
180
|
+
val = fm[k]
|
|
181
|
+
# AC 33 — always sort source_files list
|
|
182
|
+
if k == "source_files" and isinstance(val, list):
|
|
183
|
+
val = sorted(val)
|
|
184
|
+
extensions[k] = val
|
|
185
|
+
|
|
186
|
+
# Dump reserved block — always block style
|
|
187
|
+
reserved_yaml = yaml.dump(
|
|
188
|
+
reserved,
|
|
189
|
+
default_flow_style=False,
|
|
190
|
+
allow_unicode=True,
|
|
191
|
+
sort_keys=False,
|
|
192
|
+
)
|
|
193
|
+
|
|
194
|
+
# Dump extension block — always block style
|
|
195
|
+
ext_yaml = yaml.dump(
|
|
196
|
+
extensions,
|
|
197
|
+
default_flow_style=False,
|
|
198
|
+
allow_unicode=True,
|
|
199
|
+
sort_keys=False,
|
|
200
|
+
)
|
|
201
|
+
|
|
202
|
+
# Assemble: frontmatter with blank line separator, then body
|
|
203
|
+
if self.frontmatter_style == COMMENT_FRONTMATTER:
|
|
204
|
+
# One YAML document either way, so a reader that has the text can
|
|
205
|
+
# parse it with the same loader. Only the delimiters differ.
|
|
206
|
+
inner = f"{reserved_yaml}\n{ext_yaml}".rstrip("\n")
|
|
207
|
+
return f"<!-- constant-docs\n{inner}\n-->\n\n{self.body}"
|
|
208
|
+
parts = ["---\n", reserved_yaml, "\n", ext_yaml, "---\n", self.body]
|
|
209
|
+
return "".join(parts)
|
|
210
|
+
|
|
211
|
+
# ------------------------------------------------------------------
|
|
212
|
+
# Disk I/O
|
|
213
|
+
# ------------------------------------------------------------------
|
|
214
|
+
|
|
215
|
+
def save(self, path: str | Path) -> None:
|
|
216
|
+
"""Write this document to *path* atomically.
|
|
217
|
+
|
|
218
|
+
If an existing file at *path* already contains the same content,
|
|
219
|
+
the file is left untouched (byte-identical).
|
|
220
|
+
"""
|
|
221
|
+
content = self.serialize()
|
|
222
|
+
path = Path(path)
|
|
223
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
224
|
+
|
|
225
|
+
# Compare with existing content
|
|
226
|
+
if path.exists() and path.read_bytes() == content.encode():
|
|
227
|
+
return # byte-identical — no-op
|
|
228
|
+
|
|
229
|
+
# Atomic write via temp file + rename
|
|
230
|
+
fd, tmp_path = tempfile.mkstemp(
|
|
231
|
+
dir=str(path.parent),
|
|
232
|
+
prefix=f".{path.name}.tmp.",
|
|
233
|
+
)
|
|
234
|
+
try:
|
|
235
|
+
with os.fdopen(fd, "w", encoding="utf-8") as f:
|
|
236
|
+
f.write(content)
|
|
237
|
+
replace_preserving_mode(tmp_path, path)
|
|
238
|
+
except BaseException:
|
|
239
|
+
# Clean up the temp file on any error
|
|
240
|
+
if os.path.exists(tmp_path):
|
|
241
|
+
os.unlink(tmp_path)
|
|
242
|
+
raise
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
# ======================================================================
|
|
246
|
+
# Module-level helpers
|
|
247
|
+
# ======================================================================
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def replace_preserving_mode(tmp_path: str, target: Path) -> None:
|
|
251
|
+
"""Move *tmp_path* onto *target*, giving it a sensible mode first.
|
|
252
|
+
|
|
253
|
+
``tempfile.mkstemp`` creates at 0600 and ``os.replace`` carries that mode
|
|
254
|
+
across, so every file written this way came out owner-read-only and a
|
|
255
|
+
rewrite silently discarded whatever mode the file already had. Git records
|
|
256
|
+
only the exec bit, so the change is invisible in review and surfaces later
|
|
257
|
+
as a CI job or a docs server, running as another user, getting EACCES on
|
|
258
|
+
the documentation this tool exists to publish.
|
|
259
|
+
|
|
260
|
+
An existing file keeps its own mode. A new one gets what a plain create
|
|
261
|
+
would have given it, umask and all. Shared with `index` and `state`, which
|
|
262
|
+
write through the same temp-then-rename discipline and had the same bug.
|
|
263
|
+
"""
|
|
264
|
+
try:
|
|
265
|
+
mode = target.stat().st_mode & 0o777
|
|
266
|
+
except OSError:
|
|
267
|
+
umask = os.umask(0)
|
|
268
|
+
os.umask(umask)
|
|
269
|
+
mode = 0o666 & ~umask
|
|
270
|
+
try:
|
|
271
|
+
os.chmod(tmp_path, mode)
|
|
272
|
+
except OSError:
|
|
273
|
+
# A filesystem that will not take a mode is not a reason to lose the
|
|
274
|
+
# write. The content still lands; only the permissions are as given.
|
|
275
|
+
pass
|
|
276
|
+
os.replace(tmp_path, target)
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
class DocumentCache:
|
|
280
|
+
"""Documents parsed once per command, keyed by path.
|
|
281
|
+
|
|
282
|
+
One `verify` reads every document twice — once to compare its hash, once
|
|
283
|
+
to conformance-check it — and a third time for a checked kind. Each read is
|
|
284
|
+
a `read_text`, a DOTALL frontmatter regex and a `yaml.safe_load`.
|
|
285
|
+
|
|
286
|
+
Scoped to one call and passed down, never module-global: this tool writes
|
|
287
|
+
the files it reads, and a cache outliving a single command would hand back
|
|
288
|
+
the body from before the write. A failure is remembered too, so a file that
|
|
289
|
+
cannot be parsed is not re-parsed once per scanner.
|
|
290
|
+
"""
|
|
291
|
+
|
|
292
|
+
def __init__(self) -> None:
|
|
293
|
+
self._by_path: dict[Path, Document | Exception] = {}
|
|
294
|
+
|
|
295
|
+
def load(self, path: str | Path) -> Document:
|
|
296
|
+
"""Return the document at *path*, reading it at most once."""
|
|
297
|
+
key = Path(path)
|
|
298
|
+
hit = self._by_path.get(key)
|
|
299
|
+
if hit is None:
|
|
300
|
+
try:
|
|
301
|
+
hit = load(key)
|
|
302
|
+
except (OSError, DocumentError, ValueError, yaml.YAMLError) as e:
|
|
303
|
+
hit = e
|
|
304
|
+
self._by_path[key] = hit
|
|
305
|
+
if isinstance(hit, Exception):
|
|
306
|
+
raise hit
|
|
307
|
+
return hit
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
def load(path: str | Path) -> Document:
|
|
311
|
+
"""Read a document from disk and parse its frontmatter + body.
|
|
312
|
+
|
|
313
|
+
Raises :exc:`DocumentError` if the file is missing or malformed.
|
|
314
|
+
"""
|
|
315
|
+
path = Path(path)
|
|
316
|
+
if not path.exists():
|
|
317
|
+
raise DocumentError(f"Document not found: {path}")
|
|
318
|
+
|
|
319
|
+
text = path.read_text(encoding="utf-8")
|
|
320
|
+
return parse(text)
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def parse(text: str) -> Document:
|
|
324
|
+
"""Parse a document string into frontmatter and body.
|
|
325
|
+
|
|
326
|
+
Both written forms are accepted, and which one was found is recorded on
|
|
327
|
+
the document. A caller that reads a document, edits it and writes it back
|
|
328
|
+
must not silently convert it: the form is a property of the file, and
|
|
329
|
+
changing it would put a table on somebody's landing page.
|
|
330
|
+
"""
|
|
331
|
+
fm, body, style = _split(text)
|
|
332
|
+
return Document(frontmatter=fm, body=body, frontmatter_style=style)
|
|
333
|
+
|
|
334
|
+
|
|
335
|
+
def _parse_frontmatter_body(text: str) -> tuple[dict[str, Any], str]:
|
|
336
|
+
"""Split a document string into (frontmatter_dict, body_string).
|
|
337
|
+
|
|
338
|
+
The style-agnostic spelling, kept because most callers do not care which
|
|
339
|
+
delimiters a document used and should not have to say so.
|
|
340
|
+
"""
|
|
341
|
+
fm, body, _style = _split(text)
|
|
342
|
+
return fm, body
|
|
343
|
+
|
|
344
|
+
|
|
345
|
+
def _split(text: str) -> tuple[dict[str, Any], str, str]:
|
|
346
|
+
"""Split a document string into (frontmatter_dict, body, style)."""
|
|
347
|
+
style = YAML_FRONTMATTER
|
|
348
|
+
m = _FM_RE.match(text)
|
|
349
|
+
if m is None:
|
|
350
|
+
m = _COMMENT_RE.match(text)
|
|
351
|
+
style = COMMENT_FRONTMATTER
|
|
352
|
+
if not m:
|
|
353
|
+
raise DocumentError(
|
|
354
|
+
"Document must start with ``---`` delimited YAML frontmatter, or "
|
|
355
|
+
"a ``<!-- constant-docs`` comment block"
|
|
356
|
+
)
|
|
357
|
+
fm_text, body = m.group(1), m.group(2)
|
|
358
|
+
# Wrapped, because `load`'s docstring promises `DocumentError` for a
|
|
359
|
+
# malformed document and PyYAML's own exception is not one. Every caller
|
|
360
|
+
# that guards a document read had written its own tuple around that gap,
|
|
361
|
+
# and they disagreed: `plan`, `index.build` and the `mark` hook each let a
|
|
362
|
+
# scanner error escape, so one hand-edited document took down every
|
|
363
|
+
# command instead of being reported as the one file at fault.
|
|
364
|
+
try:
|
|
365
|
+
parsed = yaml.safe_load(fm_text)
|
|
366
|
+
except yaml.YAMLError as e:
|
|
367
|
+
raise DocumentError(f"Frontmatter is not valid YAML: {e}") from e
|
|
368
|
+
if not isinstance(parsed, dict):
|
|
369
|
+
raise DocumentError("Frontmatter must be a YAML mapping")
|
|
370
|
+
return parsed, body, style
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
def _without_fences(body: str) -> str:
|
|
374
|
+
"""Return *body* with fenced code blocks removed.
|
|
375
|
+
|
|
376
|
+
A heading inside a fence is an example, not a heading. Documents about
|
|
377
|
+
markdown tools are full of them.
|
|
378
|
+
"""
|
|
379
|
+
return _FENCE_RE.sub("", body)
|
|
380
|
+
|
|
381
|
+
|
|
382
|
+
def section_headings(body: str) -> set[str]:
|
|
383
|
+
"""Return the H2 section headings in *body*, each as ``"## Heading"``.
|
|
384
|
+
|
|
385
|
+
H3 and deeper are structure *within* a section — an entry in a log, a
|
|
386
|
+
sub-point in a table — and are not sections.
|
|
387
|
+
"""
|
|
388
|
+
return {
|
|
389
|
+
"## " + line.lstrip().removeprefix("## ").strip()
|
|
390
|
+
for line in _without_fences(body).splitlines()
|
|
391
|
+
if line.lstrip().startswith("## ") and not line.lstrip().startswith("### ")
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
def find_h1(body: str) -> str | None:
|
|
396
|
+
"""Return the first H1 line in *body*, or None.
|
|
397
|
+
|
|
398
|
+
A body must have none: the title is a frontmatter key. Returned rather
|
|
399
|
+
than asserted so the caller can name the offending line.
|
|
400
|
+
"""
|
|
401
|
+
match = _H1_RE.search(_without_fences(body))
|
|
402
|
+
if not match:
|
|
403
|
+
return None
|
|
404
|
+
line = _without_fences(body)[match.start() :].split("\n", 1)[0]
|
|
405
|
+
return line.strip()
|
|
406
|
+
|
|
407
|
+
|
|
408
|
+
def has_required_headings(body: str, sections: list[str] | None = None) -> bool:
|
|
409
|
+
"""Return ``True`` when *body* carries every heading in *sections*."""
|
|
410
|
+
required = DEFAULT_SECTIONS if sections is None else sections
|
|
411
|
+
present = section_headings(body)
|
|
412
|
+
return all(h in present for h in required)
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
def extract_description(body: str, section: str = "## Purpose") -> str:
|
|
416
|
+
"""Return the first sentence of *section*.
|
|
417
|
+
|
|
418
|
+
Falls back to the whole section when there is no sentence-ending
|
|
419
|
+
punctuation. Returns an empty string when the section is missing, is
|
|
420
|
+
empty, or opens with a heading rather than prose — which is what a log's
|
|
421
|
+
``# Entries`` does, and a heading is not a description.
|
|
422
|
+
"""
|
|
423
|
+
heading = re.escape(section.strip())
|
|
424
|
+
match = re.search(
|
|
425
|
+
rf"^{heading}\s*\n+(.*?)(?=\n#{{1,2}} |\Z)", body, re.MULTILINE | re.DOTALL
|
|
426
|
+
)
|
|
427
|
+
if not match:
|
|
428
|
+
return ""
|
|
429
|
+
text = match.group(1).strip()
|
|
430
|
+
if not text or text.startswith("#"):
|
|
431
|
+
return ""
|
|
432
|
+
|
|
433
|
+
# First sentence (up to the first sentence-ending period + space/end)
|
|
434
|
+
sentence_match = re.match(r"^(.*?[.!?])(?:\s|$)", text, re.DOTALL)
|
|
435
|
+
sentence = sentence_match.group(1) if sentence_match else text
|
|
436
|
+
# Collapsed to one line: the description is written into YAML and into a
|
|
437
|
+
# markdown list item in the index, and a list item ends at the line break.
|
|
438
|
+
return " ".join(sentence.split())
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""Canonical content hash for constant-docs.
|
|
2
|
+
|
|
3
|
+
Exactly one implementation of the digest lives here. Both ``api.apply``
|
|
4
|
+
and ``api.verify`` import it — if you find yourself writing the hashing
|
|
5
|
+
loop a second time you have already made the mistake.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import hashlib
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def compute(files: list[Path], root: Path = Path(".")) -> str:
|
|
15
|
+
"""Return a deterministic content hash covering *files*.
|
|
16
|
+
|
|
17
|
+
Format: ``sha256:<64 hex chars>``.
|
|
18
|
+
|
|
19
|
+
Paths are hashed as POSIX-relative strings (*root* is stripped from the
|
|
20
|
+
digest). File contents are read through *root* so the same file yields
|
|
21
|
+
the same hash regardless of where the repository is checked out.
|
|
22
|
+
"""
|
|
23
|
+
h = hashlib.sha256()
|
|
24
|
+
for p in sorted(files, key=lambda q: q.as_posix()):
|
|
25
|
+
h.update(p.as_posix().encode())
|
|
26
|
+
h.update(hashlib.sha256((root / p).read_bytes()).digest())
|
|
27
|
+
return "sha256:" + h.hexdigest()
|
constant_docs/globs.py
ADDED
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
"""Glob matching, in one implementation.
|
|
2
|
+
|
|
3
|
+
Two callers need to answer "does this path belong to this module": glob
|
|
4
|
+
resolution, which walks the tree and asks it of every file, and `mark`, which
|
|
5
|
+
is handed one path by a hook and must answer without touching the disk at all.
|
|
6
|
+
|
|
7
|
+
They use the same matcher for the same reason there is one digest: two
|
|
8
|
+
implementations that agree today will disagree eventually, and the symptom
|
|
9
|
+
would be a hook marking the wrong module dirty while `verify` insists the
|
|
10
|
+
right one is fine.
|
|
11
|
+
|
|
12
|
+
Semantics follow `pathlib.Path.glob`, with one addition. Brace alternatives
|
|
13
|
+
(`*.{ts,tsx}`) are expanded, because the specification documents them and
|
|
14
|
+
`Path.glob` silently matches nothing when given one.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import re
|
|
20
|
+
from functools import lru_cache
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def expand_braces(pattern: str) -> list[str]:
|
|
24
|
+
"""Expand `{a,b}` alternatives into separate patterns.
|
|
25
|
+
|
|
26
|
+
Nested braces expand too, outermost first. A pattern with no braces comes
|
|
27
|
+
back as a single-item list, so callers need no special case.
|
|
28
|
+
"""
|
|
29
|
+
start = pattern.find("{")
|
|
30
|
+
if start == -1:
|
|
31
|
+
return [pattern]
|
|
32
|
+
|
|
33
|
+
depth = 0
|
|
34
|
+
for i in range(start, len(pattern)):
|
|
35
|
+
if pattern[i] == "{":
|
|
36
|
+
depth += 1
|
|
37
|
+
elif pattern[i] == "}":
|
|
38
|
+
depth -= 1
|
|
39
|
+
if depth == 0:
|
|
40
|
+
end = i
|
|
41
|
+
break
|
|
42
|
+
else:
|
|
43
|
+
# Unbalanced: treat the brace as a literal rather than guessing.
|
|
44
|
+
return [pattern]
|
|
45
|
+
|
|
46
|
+
prefix, body, suffix = pattern[:start], pattern[start + 1 : end], pattern[end + 1 :]
|
|
47
|
+
|
|
48
|
+
alternatives: list[str] = []
|
|
49
|
+
depth = 0
|
|
50
|
+
current = ""
|
|
51
|
+
for ch in body:
|
|
52
|
+
if ch == "{":
|
|
53
|
+
depth += 1
|
|
54
|
+
elif ch == "}":
|
|
55
|
+
depth -= 1
|
|
56
|
+
if ch == "," and depth == 0:
|
|
57
|
+
alternatives.append(current)
|
|
58
|
+
current = ""
|
|
59
|
+
else:
|
|
60
|
+
current += ch
|
|
61
|
+
alternatives.append(current)
|
|
62
|
+
|
|
63
|
+
out: list[str] = []
|
|
64
|
+
for alt in alternatives:
|
|
65
|
+
out.extend(expand_braces(prefix + alt + suffix))
|
|
66
|
+
return out
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _translate(pattern: str) -> str:
|
|
70
|
+
"""Return a regex source matching *pattern* against a relative POSIX path.
|
|
71
|
+
|
|
72
|
+
`**` is only a recursive wildcard when it is a whole segment, which is
|
|
73
|
+
what `Path.glob` does; `a**b` is two ordinary stars.
|
|
74
|
+
"""
|
|
75
|
+
segments = pattern.split("/")
|
|
76
|
+
last = len(segments) - 1
|
|
77
|
+
parts: list[str] = []
|
|
78
|
+
for index, segment in enumerate(segments):
|
|
79
|
+
if segment == "**":
|
|
80
|
+
if index == last:
|
|
81
|
+
# Trailing `**` takes everything below, however deep.
|
|
82
|
+
parts.append(".*")
|
|
83
|
+
else:
|
|
84
|
+
# Zero or more whole segments. The group carries its own
|
|
85
|
+
# trailing separator, so `src/**/x` matches `src/x`.
|
|
86
|
+
parts.append("(?:[^/]+/)*")
|
|
87
|
+
continue
|
|
88
|
+
parts.append(_translate_segment(segment))
|
|
89
|
+
if index != last:
|
|
90
|
+
parts.append("/")
|
|
91
|
+
return f"(?s:{''.join(parts)})\\Z"
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _translate_segment(segment: str) -> str:
|
|
95
|
+
"""Translate one path segment: `*`, `?`, and character classes."""
|
|
96
|
+
out: list[str] = []
|
|
97
|
+
i = 0
|
|
98
|
+
while i < len(segment):
|
|
99
|
+
ch = segment[i]
|
|
100
|
+
if ch == "*":
|
|
101
|
+
out.append("[^/]*")
|
|
102
|
+
elif ch == "?":
|
|
103
|
+
out.append("[^/]")
|
|
104
|
+
elif ch == "[":
|
|
105
|
+
close = segment.find("]", i + 1)
|
|
106
|
+
if close == -1:
|
|
107
|
+
out.append(re.escape(ch))
|
|
108
|
+
else:
|
|
109
|
+
body = segment[i + 1 : close]
|
|
110
|
+
if body.startswith("!"):
|
|
111
|
+
body = "^" + body[1:]
|
|
112
|
+
out.append(f"[{body}]")
|
|
113
|
+
i = close + 1
|
|
114
|
+
continue
|
|
115
|
+
else:
|
|
116
|
+
out.append(re.escape(ch))
|
|
117
|
+
i += 1
|
|
118
|
+
return "".join(out)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
class GlobError(ValueError):
|
|
122
|
+
"""A configured glob cannot be translated into a matcher.
|
|
123
|
+
|
|
124
|
+
Its own type so the loader can name the module the pattern belongs to.
|
|
125
|
+
A bare `re.error` was neither a `ConfigError` nor anything else a handler
|
|
126
|
+
looked for, so a reversed range in one glob exited 1 — the drift code —
|
|
127
|
+
from every command, and made `mark` fail loudly on every file write.
|
|
128
|
+
"""
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
@lru_cache(maxsize=1024)
|
|
132
|
+
def _compiled(pattern: str) -> tuple[re.Pattern[str], ...]:
|
|
133
|
+
# A bracket expression is handed to the regex engine as written, so the
|
|
134
|
+
# engine is also what validates it. Caught here rather than at each of the
|
|
135
|
+
# three call sites, because this is the one place a pattern becomes a
|
|
136
|
+
# matcher.
|
|
137
|
+
try:
|
|
138
|
+
return tuple(re.compile(_translate(p)) for p in expand_braces(pattern))
|
|
139
|
+
except re.error as e:
|
|
140
|
+
raise GlobError(
|
|
141
|
+
f"Glob {pattern!r} is not a valid pattern: {e}. A bracket "
|
|
142
|
+
f"expression is used as written, so a reversed range such as "
|
|
143
|
+
f"'[z-a]' names nothing."
|
|
144
|
+
) from e
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def matches(pattern: str, rel_posix: str) -> bool:
|
|
148
|
+
"""True when *rel_posix* — a repository-relative POSIX path — matches.
|
|
149
|
+
|
|
150
|
+
Pure string work: this never reads the filesystem, which is what lets
|
|
151
|
+
`mark` run on every file write without becoming something a user turns
|
|
152
|
+
off.
|
|
153
|
+
"""
|
|
154
|
+
return any(rx.match(rel_posix) for rx in _compiled(pattern))
|