constant-docs 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
constant_docs/api.py ADDED
@@ -0,0 +1,968 @@
1
+ """Orchestration facade for constant-docs.
2
+
3
+ Public interface: ``plan``, ``apply``, ``verify``, ``prune``.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ from collections.abc import Iterable
9
+ from dataclasses import dataclass, field
10
+ from datetime import UTC, datetime
11
+ from pathlib import Path
12
+ from typing import Any, Literal
13
+
14
+ from constant_docs import state
15
+ from constant_docs.checks import (
16
+ CatalogueError,
17
+ DiagramError,
18
+ check_architecture,
19
+ check_catalogue,
20
+ check_errors,
21
+ )
22
+ from constant_docs.checks import check_body as check_architecture_body
23
+ from constant_docs.checks import sources_for as _sources_for
24
+ from constant_docs.config import MISSING_KEY, Config, ConfigError
25
+ from constant_docs.config import load as load_config
26
+ from constant_docs.config import repo_root as _repo_root
27
+ from constant_docs.decisions import DecisionLoss, check_rewrite
28
+ from constant_docs.document import (
29
+ EXTENSION_KEY,
30
+ LEGACY_EXTENSION_KEYS,
31
+ Document,
32
+ DocumentCache,
33
+ DocumentError,
34
+ extension_block,
35
+ extract_description,
36
+ has_required_headings,
37
+ )
38
+ from constant_docs.document import (
39
+ load as load_document,
40
+ )
41
+ from constant_docs.fingerprint import compute as compute_hash
42
+ from constant_docs.index import conformance_check
43
+ from constant_docs.kinds import (
44
+ ARCHITECTURE_CHECK,
45
+ ERRORS_CHECK,
46
+ Kind,
47
+ KindError,
48
+ resolve_prompt,
49
+ )
50
+ from constant_docs.paths import OrphanReport, doc_path_for
51
+ from constant_docs.paths import foreign as find_foreign
52
+ from constant_docs.paths import orphans as find_orphans
53
+
54
+ _PROMPT_PATH = Path(__file__).parent / "prompts" / "module.md"
55
+
56
+
57
+ # ---------------------------------------------------------------------------
58
+ # Data types
59
+ # ---------------------------------------------------------------------------
60
+
61
+
62
+ @dataclass(frozen=True)
63
+ class ModulePlan:
64
+ """Everything the harness needs to generate or regenerate a document."""
65
+
66
+ module: str
67
+ doc_path: Path
68
+ files: list[Path]
69
+ previous_body: str | None
70
+ previous_description: str | None
71
+ required_headings: list[str]
72
+ reason: Literal["changed", "missing"]
73
+ # A kind decides the shape of the write. `mode` is the one a caller must
74
+ # branch on: "replace" takes a whole body through `apply`, "append" takes
75
+ # a single entry through `append`, and passing one to the other is refused
76
+ # rather than silently accepted.
77
+ kind: str = "module"
78
+ mode: Literal["replace", "append"] = "replace"
79
+ prompt: str = ""
80
+
81
+
82
+ @dataclass(frozen=True)
83
+ class Plan:
84
+ """Result of a ``plan`` invocation."""
85
+
86
+ stale: list[ModulePlan]
87
+ orphans: list[OrphanReport]
88
+ fresh: list[str]
89
+ prompt: str
90
+ # Modules whose source could not be hashed — a file listed by the walk and
91
+ # unreadable now, or deleted between the two. One module's problem, carried
92
+ # rather than raised, so the rest of the report survives it.
93
+ unreadable: list[str] = field(default_factory=list)
94
+
95
+
96
+ @dataclass(frozen=True)
97
+ class PruneReport:
98
+ """What one ``prune`` deleted, and what it declined to touch.
99
+
100
+ Returned rather than printed, so a caller that is not the CLI can act on
101
+ the skipped list. It exists at all because the alternative — deleting and
102
+ saying only *"Pruned orphans."* — is how this command destroyed files it
103
+ had never written without anyone noticing.
104
+ """
105
+
106
+ deleted: list[Path]
107
+ # Markdown under the docs root carrying none of this tool's frontmatter.
108
+ # Somebody else's files: not orphans, not deleted, and named every time
109
+ # because silence reads as "there was nothing else".
110
+ skipped: list[Path]
111
+ # Orphans this run could not remove — a read-only directory, a permission.
112
+ # Carried rather than raised, for the same reason `deleted` exists at all.
113
+ failed: list[str] = field(default_factory=list)
114
+
115
+
116
+ @dataclass(frozen=True)
117
+ class VerifyReport:
118
+ """Result of a ``verify`` invocation."""
119
+
120
+ stale: list[str]
121
+ missing: list[str]
122
+ orphans: list[OrphanReport]
123
+ fresh: list[str]
124
+ conformance_issues: list[str] = field(default_factory=list)
125
+ # Problems inside a document's body that the tool can decide from facts it
126
+ # already holds — today, an architecture diagram naming a module the
127
+ # configuration does not have. Kept apart from `conformance_issues`, which
128
+ # are about a document's shape rather than its claims.
129
+ content_issues: list[str] = field(default_factory=list)
130
+ # Files no module covers and no exclusion names. Empty unless `verify` was
131
+ # asked for them: adopting the tool must not immediately fail a repository
132
+ # that has not finished mapping itself.
133
+ coverage_gaps: list[str] = field(default_factory=list)
134
+
135
+ @property
136
+ def exit_code(self) -> int:
137
+ """0 when clean, 1 on drift, a missing document, an orphan, or an issue."""
138
+ if (
139
+ self.stale
140
+ or self.missing
141
+ or self.orphans
142
+ or self.conformance_issues
143
+ or self.content_issues
144
+ or self.coverage_gaps
145
+ ):
146
+ return 1
147
+ return 0
148
+
149
+
150
+ # ---------------------------------------------------------------------------
151
+ # Helpers
152
+ # ---------------------------------------------------------------------------
153
+
154
+
155
+ def _load_prompt() -> str:
156
+ """Return the generation conventions text."""
157
+ return _PROMPT_PATH.read_text(encoding="utf-8")
158
+
159
+
160
+ def _modules_for_paths(cfg: Config, changed_paths: list[str] | None) -> set[str]:
161
+ """Resolve changed paths to the module keys they affect.
162
+
163
+ When *changed_paths* is ``None``, every configured module is in scope.
164
+
165
+ Takes the loaded `Config`, not the path it came from. Taking the path
166
+ meant this reloaded and re-resolved a configuration its only caller was
167
+ already holding — a second full walk of the repository, for a function
168
+ that reads one attribute.
169
+ """
170
+ if changed_paths is None:
171
+ return set(cfg.module_files.keys())
172
+
173
+ changed_set = {Path(p).as_posix() for p in changed_paths}
174
+ affected: set[str] = set()
175
+ for mod_key, files in cfg.module_files.items():
176
+ if any(f.as_posix() in changed_set for f in files):
177
+ affected.add(mod_key)
178
+ return affected
179
+
180
+
181
+ def _prompt_for(kind: Kind, repo_root: Path) -> str:
182
+ """Resolve a kind's prompt, reporting a missing file as configuration.
183
+
184
+ `docs/ERRORS.md` says a `KindError` reaches the caller wrapped as a
185
+ `ConfigError`. That was true only of the load-time wrap, and `load_kinds`
186
+ never touches the filesystem — so a kind naming a prompt file that is not
187
+ there loaded cleanly and then escaped from inside `plan`, past every
188
+ handler, as exit 1 with no JSON at all.
189
+ """
190
+ try:
191
+ return resolve_prompt(kind, repo_root)
192
+ except KindError as e:
193
+ raise ConfigError(str(e), key="kinds") from e
194
+
195
+
196
+ def _kind_for(cfg: Any, module_key: str) -> tuple[Any, Kind]:
197
+ """Return the (module entry, kind) pair for *module_key*.
198
+
199
+ Raises `ValueError` when the module is not configured, and `ConfigError`
200
+ when it names a kind that is not — though the loader has already refused
201
+ that, so the second is a belt-and-braces case.
202
+ """
203
+ entry = next((m for m in cfg.modules if m.key == module_key), None)
204
+ if entry is None:
205
+ raise ValueError(f"Module {module_key!r} is not in the configuration")
206
+ kind = cfg.kinds.get(entry.kind)
207
+ if kind is None:
208
+ raise ConfigError(
209
+ f"Module {module_key!r} names kind {entry.kind!r}, which is not declared",
210
+ key="modules",
211
+ )
212
+ return entry, kind
213
+
214
+
215
+ def _now_iso() -> str:
216
+ """Return an ISO-8601 timestamp string in UTC."""
217
+ return datetime.now(UTC).strftime("%Y-%m-%dT%H:%M:%SZ")
218
+
219
+
220
+ def _today() -> str:
221
+ """Return today's date as ``YYYY-MM-DD``, which is what an OKF date is."""
222
+ return datetime.now(UTC).strftime("%Y-%m-%d")
223
+
224
+
225
+ def _seed_tags(kind: Kind) -> list[str]:
226
+ """Return the one tag a fresh document starts with.
227
+
228
+ Derived from the kind's own `type`, lowercased and hyphenated, because
229
+ that is a fact the document already states rather than a taxonomy this
230
+ tool invented. A conformance checker wants `tags` present and non-empty;
231
+ a documentation set that wants real tags replaces this and keeps it, like
232
+ every other host key.
233
+ """
234
+ return ["-".join(kind.type.lower().split())]
235
+
236
+
237
+ def _retirements(existing_doc: Any, retired_now: list[str]) -> dict[str, str]:
238
+ """Merge this write's retirements into whatever the document already had.
239
+
240
+ Retirements accumulate and are never dropped. The record is the point: a
241
+ decision that was deliberately retired reads, in the diff, as a deliberate
242
+ act — whereas an absence reads as nothing at all, which is exactly what a
243
+ silent loss looks like.
244
+
245
+ Dated, because *when* a decision was retired is the question anybody
246
+ reading the record will actually have.
247
+ """
248
+ previous: dict[str, str] = {}
249
+ if existing_doc is not None:
250
+ raw = extension_block(existing_doc.frontmatter).get("retired_decisions")
251
+ if isinstance(raw, dict):
252
+ previous = {str(k): str(v) for k, v in raw.items()}
253
+ elif isinstance(raw, list):
254
+ # Tolerated, never written: a hand-edited document may carry the
255
+ # bare list a reader would naturally reach for.
256
+ previous = {str(k): "" for k in raw}
257
+ today = _today()
258
+ for name in retired_now:
259
+ previous.setdefault(name, today)
260
+ return dict(sorted(previous.items()))
261
+
262
+
263
+ def _title_for(module_key: str) -> str:
264
+ """Return a human-readable title seeded from a module key.
265
+
266
+ The last segment, separators turned into spaces. Capitalisation is added
267
+ only when the segment carries none of its own, so `api` becomes `Api` and
268
+ `SPEC` stays `SPEC`. This is a seed: the host owns `title` and whatever it
269
+ puts there survives every later write.
270
+ """
271
+ segment = module_key.rstrip("/").split("/")[-1]
272
+ words = segment.replace("_", " ").replace("-", " ").strip()
273
+ if not words:
274
+ return module_key
275
+ if words == words.lower():
276
+ return words[0].upper() + words[1:]
277
+ return words
278
+
279
+
280
+ # ---------------------------------------------------------------------------
281
+ # Public API
282
+ # ---------------------------------------------------------------------------
283
+
284
+
285
+ # Re-exported for callers that inspect frontmatter; `document` owns the
286
+ # definition, so there is exactly one spelling of it in the codebase.
287
+ CONSTANT_DOCS_KEY = EXTENSION_KEY
288
+ # Written at the top level before the producer-extension block existed. Dropped
289
+ # on write so a document migrated from the old format does not carry both.
290
+ # Safe to delete once no document predates the change.
291
+ _LEGACY_TOP_LEVEL_KEYS = frozenset(
292
+ {
293
+ "module",
294
+ "source_glob",
295
+ "source_files",
296
+ "source_hash",
297
+ "hash_method",
298
+ "hash_covers",
299
+ "timestamp",
300
+ "generator",
301
+ "generator_spec",
302
+ }
303
+ )
304
+
305
+
306
+ GENERATOR_SPEC = "https://github.com/ashborn-systems/constant-docs/blob/main/SPEC.md"
307
+
308
+
309
+ def _version() -> str:
310
+ """Resolved lazily: importing the package at module level would be circular.
311
+
312
+ Deferring to ``__version__`` rather than calling ``importlib.metadata``
313
+ again keeps one answer to "what version wrote this document", including
314
+ the fallback used by a checkout that was never installed.
315
+ """
316
+ from constant_docs import __version__
317
+
318
+ return __version__
319
+
320
+
321
+ def _resolve_config(config_path: str | Path | None) -> Path:
322
+ """Return the config path, discovering it from the working directory.
323
+
324
+ The specification's public API calls `verify()` and `apply(module, body=...)`
325
+ with no config argument, so a harness running inside a repository does not
326
+ have to locate the file itself. An explicit path always wins.
327
+ """
328
+ if config_path is not None:
329
+ return Path(config_path)
330
+ here = Path.cwd().resolve()
331
+ for folder in [here, *here.parents]:
332
+ candidate = folder / "constant-docs.yaml"
333
+ if candidate.is_file():
334
+ return candidate
335
+ raise ConfigError(
336
+ "no constant-docs.yaml found in the working directory or any parent; "
337
+ "pass config_path explicitly",
338
+ # The same key the explicit-path miss carries. Without it the hook
339
+ # commands cannot tell "this is not a constant-docs repository" from
340
+ # "this repository's config is broken", and a globally installed Stop
341
+ # hook starts talking in every repository that does not use the tool.
342
+ key=MISSING_KEY,
343
+ )
344
+
345
+
346
+ def plan(
347
+ config_path: str | Path | None = None,
348
+ changed_paths: list[str] | None = None,
349
+ modules: list[str] | None = None,
350
+ ) -> Plan:
351
+ """Inspect the repository and return the current state.
352
+
353
+ Returns a :class:`Plan` listing stale modules, orphans, fresh modules,
354
+ and the generation-prompt text.
355
+
356
+ *changed_paths* of ``None`` means consider every module. *modules* names
357
+ module keys directly, which is what `settle` has and saves it inventing a
358
+ path that resolves back to them.
359
+ """
360
+ resolved = _resolve_config(config_path)
361
+ return _plan(load_config(resolved), _repo_root(resolved), changed_paths, modules)
362
+
363
+
364
+ def _plan(
365
+ cfg: Config,
366
+ repo_root: Path,
367
+ changed_paths: list[str] | None = None,
368
+ modules: list[str] | None = None,
369
+ cache: DocumentCache | None = None,
370
+ ) -> Plan:
371
+ """The body of :func:`plan`, over a configuration already loaded.
372
+
373
+ Split out so `verify` and `settle` — both of which have the `Config` in
374
+ hand — stop handing a path back to be read again. One `verify` used to load
375
+ the configuration three times and walk the repository three times with it.
376
+ """
377
+ cache = cache or DocumentCache()
378
+ if modules is not None:
379
+ modules_in_scope = {m for m in modules if m in cfg.module_files}
380
+ else:
381
+ modules_in_scope = _modules_for_paths(cfg, changed_paths)
382
+ stale: list[ModulePlan] = []
383
+ fresh: list[str] = []
384
+ unreadable: list[str] = []
385
+
386
+ # Sorted, not set-ordered. `stale` and `fresh` are built in this order and
387
+ # only two of the three lists were sorted downstream, so `verify --json`
388
+ # and `plan --json` came back with a different byte sequence on every
389
+ # process — `PYTHONHASHSEED` randomises string hashing — which is exactly
390
+ # what the reproducibility comment in `verify` claims not to happen.
391
+ for mod_key in sorted(modules_in_scope):
392
+ _, kind = _kind_for(cfg, mod_key)
393
+ files = cfg.module_files[mod_key]
394
+ try:
395
+ current_hash = compute_hash(files, root=repo_root)
396
+ except OSError as e:
397
+ # One module's problem, not the whole command's — which is what
398
+ # `orphans` already does with the same call. A file deleted between
399
+ # the walk and the hash, or one the walk listed and cannot read,
400
+ # used to abort `verify` before it printed any JSON at all.
401
+ unreadable.append(f"{mod_key} — source could not be read: {e}")
402
+ continue
403
+
404
+ doc_file = doc_path_for(cfg, mod_key)
405
+ doc_abs = repo_root / doc_file
406
+ previous_body: str | None = None
407
+ previous_description: str | None = None
408
+ reason: Literal["changed", "missing"] = "missing"
409
+
410
+ if doc_abs.exists():
411
+ try:
412
+ existing = cache.load(doc_abs)
413
+ stored_hash = str(
414
+ extension_block(existing.frontmatter).get("source_hash", "")
415
+ )
416
+ previous_body = existing.body
417
+ previous_description = existing.frontmatter.get("description")
418
+ if stored_hash == current_hash:
419
+ fresh.append(mod_key)
420
+ continue
421
+ reason = "changed"
422
+ except (OSError, DocumentError):
423
+ # Malformed document → treat as missing
424
+ reason = "missing"
425
+
426
+ stale.append(
427
+ ModulePlan(
428
+ module=mod_key,
429
+ doc_path=doc_file,
430
+ files=files,
431
+ previous_body=previous_body,
432
+ previous_description=previous_description,
433
+ required_headings=list(kind.sections),
434
+ reason=reason,
435
+ kind=kind.name,
436
+ mode=kind.mode,
437
+ prompt=_prompt_for(kind, repo_root),
438
+ )
439
+ )
440
+
441
+ # Orphans
442
+ orphan_list = find_orphans(cfg, repo_root)
443
+
444
+ return Plan(
445
+ stale=stale,
446
+ orphans=orphan_list,
447
+ fresh=fresh,
448
+ prompt=_load_prompt(),
449
+ unreadable=unreadable,
450
+ )
451
+
452
+
453
+ def _write(
454
+ module_key: str,
455
+ body: str,
456
+ *,
457
+ config_path: Path,
458
+ repo_root: Path,
459
+ cfg: Any,
460
+ module_entry: Any,
461
+ kind: Kind,
462
+ retire: Iterable[str] = (),
463
+ ) -> None:
464
+ """Assemble frontmatter for *body* and write the document.
465
+
466
+ Shared by `apply` and `append`, because the frontmatter a document carries
467
+ is a property of the module and its kind, not of how the body was produced.
468
+
469
+ Every refusal happens before anything is written, so a rejected body
470
+ leaves the document on disk exactly as it was.
471
+ """
472
+ if not has_required_headings(body, kind.sections):
473
+ raise DocumentError(
474
+ f"Body is missing one or more required headings for kind "
475
+ f"{kind.name!r}: " + ", ".join(kind.sections)
476
+ )
477
+
478
+ doc_file = doc_path_for(cfg, module_key)
479
+ doc_abs = repo_root / doc_file
480
+ # Read once, used three times: to compare decisions, to carry retirements
481
+ # forward, and to preserve host frontmatter. A malformed document reads as
482
+ # absent here, which is what the rest of the tool already does with one.
483
+ existing_doc = None
484
+ unparsed_text: str | None = None
485
+ if doc_abs.exists():
486
+ try:
487
+ existing_doc = load_document(doc_abs)
488
+ except (OSError, DocumentError, ValueError):
489
+ existing_doc = None
490
+ # Kept, because a document that does not parse still has text and
491
+ # that text can record decisions. See the guard below.
492
+ try:
493
+ unparsed_text = doc_abs.read_text(encoding="utf-8")
494
+ except (OSError, UnicodeDecodeError):
495
+ unparsed_text = None
496
+
497
+ # A regeneration is free to rewrite every decision this document records
498
+ # and unable to drop one without saying so. Checked here rather than in
499
+ # `apply`, because this is the single point every write passes through —
500
+ # the same reason there is one digest and one glob matcher.
501
+ #
502
+ # An append cannot fail this: its body is the previous one plus a suffix,
503
+ # so nothing can go missing. That is a property worth having rather than a
504
+ # special case worth writing.
505
+ retired_now: list[str] = []
506
+ if existing_doc is not None:
507
+ retired_now = check_rewrite(existing_doc.body, body, retire)
508
+ elif unparsed_text is not None:
509
+ # A document that will not parse is still checked for what it records.
510
+ # Treating it as absent turned the guard off entirely, so a single
511
+ # stray leading newline — or a merge marker — was enough to drop every
512
+ # decision in the file without a word. A marker is an HTML comment in
513
+ # the body and frontmatter carries none, so reading the whole file as
514
+ # the previous body sees exactly the decisions that are there.
515
+ #
516
+ # The document is still regenerated from scratch afterwards, which is
517
+ # what the rest of the tool does with one it cannot read. What is no
518
+ # longer allowed is losing a decision on the way.
519
+ retired_now = check_rewrite(unparsed_text, body, retire)
520
+ elif list(retire):
521
+ raise DecisionLoss(
522
+ f"A retirement was declared for {module_key!r}, which has no "
523
+ f"document yet. There is nothing recorded to retire.",
524
+ sorted({r.strip().lower() for r in retire if r.strip()}),
525
+ )
526
+
527
+ # The one kind whose content is checked rather than only hashed. Refused
528
+ # here as well as reported by `verify`, so a diagram naming a module that
529
+ # does not exist is caught when it is written rather than a release later.
530
+ if kind.check == ARCHITECTURE_CHECK:
531
+ problems = check_architecture_body(body, [m.key for m in cfg.modules])
532
+ if problems:
533
+ raise DiagramError(
534
+ f"The architecture body for {module_key!r} was refused:\n"
535
+ + "\n".join(f" {p}" for p in problems)
536
+ )
537
+ elif kind.check == ERRORS_CHECK:
538
+ problems = check_catalogue(body, _sources_for(cfg, repo_root, module_key))
539
+ if problems:
540
+ raise CatalogueError(
541
+ f"The error catalogue for {module_key!r} was refused:\n"
542
+ + "\n".join(f" {p}" for p in problems)
543
+ )
544
+
545
+ files = cfg.module_files[module_key]
546
+ try:
547
+ current_hash = compute_hash(files, root=repo_root)
548
+ except OSError as e:
549
+ raise DocumentError(
550
+ f"Cannot hash the sources for {module_key!r}: {e}. Nothing was "
551
+ f"written: a document whose hash is unknown would report stale on "
552
+ f"the next run whatever it holds."
553
+ ) from e
554
+
555
+ # Everything this tool owns lives under one key, which is what the Open
556
+ # Knowledge Format calls a producer extension: consumers tolerate unknown
557
+ # keys, so a namespaced block cannot collide with a host schema. The tool
558
+ # writes CONSTANT_DOCS_KEY and nothing else — `type`, `title`, `tags` and
559
+ # dates belong to whoever owns the documentation set, not to us.
560
+ #
561
+ # Assembled first so a retirement can sit next to
562
+ # `kind` — what the document *is* — rather than after the generator, where
563
+ # a reader scanning a diff would miss it.
564
+ block: dict[str, Any] = {"module": module_key, "kind": kind.name}
565
+ # Retirements accumulate and are never dropped: the record is the whole
566
+ # point, and a retirement that left no trace would look exactly like the
567
+ # silent drop this guard exists to catch. A document that has retired
568
+ # nothing carries no such key, so the mechanism costs an ordinary document
569
+ # not one byte.
570
+ retired = _retirements(existing_doc, retired_now)
571
+ if retired:
572
+ block["retired_decisions"] = retired
573
+ block.update(
574
+ {
575
+ "source_glob": module_entry.globs[0]
576
+ if len(module_entry.globs) == 1
577
+ else module_entry.globs,
578
+ # Only when there is something to say. A document is read on
579
+ # its own, so the boundary rule has to be legible from the
580
+ # frontmatter alone — and "whatever `api` covers" is half of
581
+ # that rule when a document declares it.
582
+ **({"covers": list(module_entry.covers)} if module_entry.covers else {}),
583
+ "source_files": [f.as_posix() for f in sorted(files)],
584
+ "source_hash": current_hash,
585
+ "hash_method": "sha256-over-sorted-path-and-content",
586
+ "hash_covers": "source_files",
587
+ "timestamp": _now_iso(),
588
+ "generator": f"constant-docs/{_version()}",
589
+ "generator_spec": GENERATOR_SPEC,
590
+ }
591
+ )
592
+
593
+ # `type` is the only key the Open Knowledge Format requires; `title`,
594
+ # `description` and `tags` are recommended, and dated keys are what a real
595
+ # consumer checks for. All six are seeded here and none is ever
596
+ # overwritten: they belong to whoever owns the documentation set, and
597
+ # seeding is what makes the docs tree an OKF bundle rather than merely
598
+ # OKF-shaped. The tool's own "when did this body last change" lives on
599
+ # `timestamp` inside its block, so nothing here needs to move on a write.
600
+ fm: dict[str, Any] = {
601
+ "type": kind.type,
602
+ "title": _title_for(module_key),
603
+ # Criterion 31, generalised: the first sentence of the kind's opening
604
+ # section. A kind whose opening section holds headings rather than
605
+ # prose — a log's `# Entries` — yields nothing, and the existing
606
+ # description is kept below.
607
+ "description": extract_description(body, kind.sections[0]) or module_key,
608
+ "tags": _seed_tags(kind),
609
+ "created": _today(),
610
+ "updated": _today(),
611
+ CONSTANT_DOCS_KEY: block,
612
+ }
613
+
614
+ # Preserve every existing key the tool does not own. `type` and
615
+ # `description` are seeded above for a fresh document, then deferred to
616
+ # the host's values if the document already carries them.
617
+ if existing_doc is not None:
618
+ for k, v in existing_doc.frontmatter.items():
619
+ if (
620
+ k != CONSTANT_DOCS_KEY
621
+ and k not in LEGACY_EXTENSION_KEYS
622
+ and k not in _LEGACY_TOP_LEVEL_KEYS
623
+ ):
624
+ fm[k] = v
625
+
626
+ doc = Document(
627
+ frontmatter=fm,
628
+ body=body,
629
+ sections=list(kind.sections),
630
+ # From the kind, not from whatever the file happened to carry. A
631
+ # document is written in the form its kind declares, so changing the
632
+ # declaration converts every document of that kind on the next write
633
+ # rather than leaving the estate in two forms with nothing saying
634
+ # which is current.
635
+ frontmatter_style=kind.frontmatter,
636
+ )
637
+ doc.save(doc_abs)
638
+
639
+ # The write that satisfies a module is what clears it from the dirty set:
640
+ # precise, per module, and needing no cooperation from the caller. A
641
+ # partial generation clears only what it actually wrote.
642
+ state.discard(repo_root, module_key)
643
+
644
+
645
+ def apply(
646
+ module_key: str,
647
+ body: str,
648
+ config_path: str | Path | None = None,
649
+ retire: Iterable[str] = (),
650
+ ) -> None:
651
+ """Write *body* as the whole document body for *module_key*.
652
+
653
+ The body must carry every heading the module's kind requires; it is
654
+ checked before anything is written, so a rejected body leaves an existing
655
+ document untouched. Frontmatter is updated with the current source hash,
656
+ file list, and timestamp.
657
+
658
+ *retire* names decision identifiers this rewrite deliberately removes.
659
+ Without it, a body that drops a decision the previous one recorded is
660
+ refused: a rewrite is free to reword every decision and unable to drop one
661
+ silently. Declaring a retirement that did not happen is refused too, so
662
+ the parameter cannot become a way to switch the check off.
663
+
664
+ Refused for a module whose kind appends. Overloading one call to mean both
665
+ "replace this" and "add to this" is how a log gets silently destroyed by a
666
+ caller that passed the wrong shape.
667
+ """
668
+ config_path = _resolve_config(config_path)
669
+ repo_root = _repo_root(config_path)
670
+ cfg = load_config(config_path)
671
+ module_entry, kind = _kind_for(cfg, module_key)
672
+
673
+ if kind.mode == "append":
674
+ raise ValueError(
675
+ f"Module {module_key!r} is of kind {kind.name!r}, which appends. "
676
+ f"Use append() to add an entry; apply() would replace everything "
677
+ f"already written."
678
+ )
679
+
680
+ _write(
681
+ module_key,
682
+ body,
683
+ config_path=config_path,
684
+ repo_root=repo_root,
685
+ cfg=cfg,
686
+ module_entry=module_entry,
687
+ kind=kind,
688
+ retire=retire,
689
+ )
690
+
691
+
692
+ def append(
693
+ module_key: str,
694
+ entry: str,
695
+ title: str,
696
+ config_path: str | Path | None = None,
697
+ ) -> None:
698
+ """Add one entry to the document for *module_key*.
699
+
700
+ The entry goes under the kind's first section as an H3 whose text is
701
+ Everything already in the document is left exactly as it is: an
702
+ append that can modify history is a replace with extra steps.
703
+
704
+ The hash still moves on write, so "has anything changed since the last
705
+ entry" is the same comparison as everywhere else in the tool. That is why
706
+ a log can be stale at all, given its body is not a function of the source.
707
+
708
+ Refused for a module whose kind replaces.
709
+ """
710
+ config_path = _resolve_config(config_path)
711
+ repo_root = _repo_root(config_path)
712
+ cfg = load_config(config_path)
713
+ module_entry, kind = _kind_for(cfg, module_key)
714
+
715
+ if kind.mode != "append":
716
+ raise ValueError(
717
+ f"Module {module_key!r} is of kind {kind.name!r}, which replaces. "
718
+ f"Use apply() with the whole body."
719
+ )
720
+
721
+ title = title.strip()
722
+ if not title:
723
+ raise ValueError("An entry needs a title; it becomes the entry's H2 heading.")
724
+
725
+ doc_abs = repo_root / doc_path_for(cfg, module_key)
726
+ previous = ""
727
+ if doc_abs.exists():
728
+ try:
729
+ previous = load_document(doc_abs).body
730
+ except (OSError, DocumentError):
731
+ # A document that cannot be parsed must not be silently discarded
732
+ # by an append — that is the one write mode where the old text is
733
+ # the whole point.
734
+ raise DocumentError(
735
+ f"Cannot append to {doc_abs}: it exists but does not parse. "
736
+ f"Fix or remove it; appending would lose what it holds."
737
+ ) from None
738
+
739
+ body = _extend(previous, kind.sections, title, entry)
740
+
741
+ _write(
742
+ module_key,
743
+ body,
744
+ config_path=config_path,
745
+ repo_root=repo_root,
746
+ cfg=cfg,
747
+ module_entry=module_entry,
748
+ kind=kind,
749
+ )
750
+
751
+
752
+ def _section_end(body: str, heading: str) -> int | None:
753
+ """Return the line index just past the end of *heading*'s section.
754
+
755
+ None when the heading is not in *body*, which leaves the caller appending
756
+ at the end — the honest answer for a body that does not carry the section
757
+ its kind declares. Fenced blocks are tracked, because a log full of code
758
+ samples has `## ` inside them and a heading inside a fence is an example.
759
+ """
760
+ if not heading:
761
+ return None
762
+ wanted = heading.strip()
763
+ lines = body.split("\n")
764
+ fenced = False
765
+ start: int | None = None
766
+ for i, line in enumerate(lines):
767
+ # Both CommonMark fence characters. `~~~` is the usual escape when the
768
+ # sample itself contains backticks, which a documentation tool's own
769
+ # log is full of — and splicing an entry into the middle of one would
770
+ # lose the block rather than merely misplace the entry.
771
+ if line.lstrip().startswith(("```", "~~~")):
772
+ fenced = not fenced
773
+ continue
774
+ if fenced:
775
+ continue
776
+ if start is None:
777
+ if line.strip() == wanted:
778
+ start = i
779
+ continue
780
+ if line.startswith("## "):
781
+ end = i
782
+ # Blank lines before the next heading belong to the gap, so the
783
+ # entry lands under this section rather than adrift above the one
784
+ # that follows.
785
+ while end > start + 1 and not lines[end - 1].strip():
786
+ end -= 1
787
+ return end
788
+ return None if start is None else len(lines)
789
+
790
+
791
+ def _extend(previous: str, sections: list[str], title: str, entry: str) -> str:
792
+ """Return *previous* with one entry added to the end of its first section.
793
+
794
+ Every byte of *previous* survives in order, which is what makes "the first
795
+ entry is byte-identical after the second append" testable rather than a
796
+ matter of trust.
797
+ """
798
+ scaffold = previous.rstrip("\n")
799
+ if not scaffold:
800
+ # A new log still has to satisfy its kind's required headings.
801
+ scaffold = "\n\n".join(sections)
802
+ # An entry sits inside a section, and sections are H2, so an entry is H3.
803
+ block = f"### {title}\n\n{entry.strip()}\n"
804
+
805
+ # At the end of the kind's *first section*, not at the end of the body.
806
+ # For a single-section kind those are the same place, which is how
807
+ # appending to the end went unnoticed: give a log a second section and
808
+ # every entry filed itself under the last heading instead, leaving the
809
+ # first permanently empty and `extract_description` with nothing to find.
810
+ end = _section_end(scaffold, sections[0] if sections else "")
811
+ if end is None:
812
+ return f"{scaffold}\n\n{block}"
813
+ lines = scaffold.split("\n")
814
+ head = "\n".join(lines[:end]).rstrip("\n")
815
+ tail = "\n".join(lines[end:]).strip("\n")
816
+ if not tail:
817
+ return f"{head}\n\n{block}"
818
+ return f"{head}\n\n{block}\n{tail}\n"
819
+
820
+
821
+ def verify(
822
+ config_path: str | Path | None = None, coverage: bool = False
823
+ ) -> VerifyReport:
824
+ """Run ``plan`` on every module and return a drift report.
825
+
826
+ *coverage* folds uncovered source into the gate. Opt-in, because adopting
827
+ the tool on a repository that has not finished mapping itself would
828
+ otherwise fail on the first run — and a check that fails from the moment
829
+ it is installed is one somebody switches off.
830
+ """
831
+ resolved = _resolve_config(config_path)
832
+ root = _repo_root(resolved)
833
+ # Loaded once, and every document parsed once. `plan` used to be handed the
834
+ # path and load it again, `_modules_for_paths` a third time, and each of the
835
+ # scanners below re-read documents `plan` had already read — three walks of
836
+ # the repository and two parses of every document, per `verify`.
837
+ cfg = load_config(resolved)
838
+ cache = DocumentCache()
839
+ p = _plan(cfg, root, cache=cache)
840
+ # Sorted so two runs over the same repository produce byte-identical
841
+ # output. `_plan` iterates its module set in sorted order for the same
842
+ # reason: sorting two of the three lists here left `fresh` in string-hash
843
+ # order, which `PYTHONHASHSEED` randomises per process.
844
+ stale_names = sorted(m.module for m in p.stale if m.reason == "changed")
845
+ missing_names = sorted(m.module for m in p.stale if m.reason == "missing")
846
+
847
+ # AC 25 — conformance check on all docs under docs_root
848
+ conformance_issues = conformance_check(cfg, root, cache=cache)
849
+
850
+ return VerifyReport(
851
+ stale=stale_names,
852
+ missing=missing_names,
853
+ orphans=p.orphans,
854
+ fresh=p.fresh,
855
+ conformance_issues=conformance_issues,
856
+ content_issues=(
857
+ check_architecture(cfg, root, cache=cache)
858
+ + check_errors(cfg, root, cache=cache)
859
+ + sorted(p.unreadable)
860
+ ),
861
+ coverage_gaps=_coverage_gaps(cfg, root) if coverage else [],
862
+ )
863
+
864
+
865
+ def _coverage_gaps(cfg: Any, repo_root: Path) -> list[str]:
866
+ """Return one line per directory holding uncovered source.
867
+
868
+ By directory rather than by file: a gate that printed four hundred paths is
869
+ a gate nobody reads, and the decision a reader has to make is about
870
+ directories anyway.
871
+ """
872
+ from constant_docs.coverage import by_directory, uncovered_files
873
+
874
+ grouped = by_directory(uncovered_files(cfg, repo_root))
875
+ return [
876
+ f"{directory} — {len(files)} file{'' if len(files) == 1 else 's'} "
877
+ f"covered by no module"
878
+ for directory, files in grouped.items()
879
+ ]
880
+
881
+
882
+ def settle(config_path: str | Path | None = None) -> Plan:
883
+ """Plan over the dirty set, without clearing it.
884
+
885
+ Called at quiescence — the end of a turn — so the generator and the coding
886
+ loop never write the same file at once.
887
+
888
+ **The set is not cleared by being read.** If it cleared on emit and the
889
+ caller then failed to generate, the staleness would be lost and nothing
890
+ would ever notice again. A module is dropped here only when the plan finds
891
+ it *fresh*, which means the file was edited and edited back: dirty by the
892
+ mark, clean by the hash, and provably no work.
893
+ """
894
+ resolved = _resolve_config(config_path)
895
+ repo_root = _repo_root(resolved)
896
+ dirty = state.read(repo_root)
897
+ if not dirty.modules:
898
+ return Plan(stale=[], orphans=[], fresh=[], prompt=_load_prompt())
899
+
900
+ cfg = load_config(resolved)
901
+ # A module deleted from the configuration since it was marked is not work;
902
+ # it is a configuration change, and `verify` reports its document as an
903
+ # orphan.
904
+ known = [m for m in dirty.modules if m in cfg.module_files]
905
+ p = _plan(cfg, _repo_root(resolved), changed_paths=None, modules=known)
906
+
907
+ if p.fresh:
908
+ remaining = [m for m in dirty.modules if m not in set(p.fresh)]
909
+ if remaining != dirty.modules:
910
+ dirty.modules = remaining
911
+ state.write(repo_root, dirty)
912
+
913
+ return p
914
+
915
+
916
+ def prune(config_path: str | Path | None = None) -> PruneReport:
917
+ """Delete orphaned documents and the empty directories they leave.
918
+
919
+ **Only documents this tool wrote.** A markdown file under the docs root
920
+ carrying none of our frontmatter is somebody else's — a decision record, a
921
+ hand-off note, whatever was in the `docs/` folder before the tool arrived —
922
+ and it is returned in ``skipped`` rather than deleted. Having a `docs/`
923
+ folder already is the normal way people adopt a documentation tool, and
924
+ pointing it at that folder is the first thing they do.
925
+ """
926
+ config_path = _resolve_config(config_path)
927
+ repo_root = _repo_root(config_path)
928
+ cfg = load_config(config_path)
929
+ orphan_list = find_orphans(cfg, repo_root)
930
+ # Assembled before the deletions, not in the return statement after them.
931
+ # Building it afterwards meant a failure while scanning — one non-UTF-8
932
+ # file under the docs root was enough — landed with the orphans already
933
+ # unlinked and no report of which ones.
934
+ skipped = find_foreign(cfg, repo_root)
935
+ docs_root_abs = repo_root / cfg.docs_root
936
+
937
+ deleted: list[Path] = []
938
+ deleted_dirs: set[Path] = set()
939
+ failed: list[str] = []
940
+ for orphan in orphan_list:
941
+ # `OrphanReport.doc_path` is repository-relative, so the absolute one
942
+ # is built here, where a file is actually opened, and goes no further.
943
+ doc_abs = repo_root / orphan.doc_path
944
+ if doc_abs.exists():
945
+ try:
946
+ doc_abs.unlink()
947
+ except OSError as e:
948
+ # One file that will not go is not a reason to lose the record
949
+ # of the ones that did. Reported beside them rather than raised.
950
+ failed.append(f"{orphan.doc_path} — {e}")
951
+ continue
952
+ deleted.append(orphan.doc_path)
953
+ # Track the parent directory for cleanup
954
+ parent = doc_abs.parent
955
+ while parent != docs_root_abs and parent != parent.parent:
956
+ deleted_dirs.add(parent)
957
+ parent = parent.parent
958
+
959
+ # Remove empty directories (bottom-up). A directory still holding a file
960
+ # this tool did not write is not empty, so nothing here can take one.
961
+ for d in sorted(deleted_dirs, key=lambda p: len(p.parts), reverse=True):
962
+ try:
963
+ if d.exists() and not any(d.iterdir()):
964
+ d.rmdir()
965
+ except OSError:
966
+ pass
967
+
968
+ return PruneReport(deleted=deleted, skipped=skipped, failed=failed)