constant-docs 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,165 @@
1
+ """What the repository holds, and what covers it.
2
+
3
+ `verify` can only check what has been declared. It hashes the modules in the
4
+ configuration and says nothing at all about source nobody registered, so a
5
+ project can drift to a large fraction undocumented while every check passes.
6
+ That is a control reporting success while not looking, and it was the one place
7
+ this tool had it.
8
+
9
+ Coverage is a **set difference** — the repository's files, minus the union of
10
+ every module's resolved files, minus what configuration says is deliberately
11
+ uncovered. No model, no parsing, no judgement.
12
+
13
+ ## Exclusions are declared, not inferred
14
+
15
+ A repository has directories that legitimately want no document: fixtures,
16
+ vendored code, generated output. Those are named in configuration with a reason
17
+ each, in the idiom every other exemption here uses. An inferred exclusion is one
18
+ nobody decided, and it will outlive whatever made it sensible.
19
+
20
+ Two exclusions are *not* declared, because they are already refused everywhere
21
+ else: secrets, which must never be read at all, and derived output, which is
22
+ built from source the tool already hashes.
23
+
24
+ ## The judgement half
25
+
26
+ `init` is the other side of this and does not put a model in the package. The
27
+ tool gathers facts — the tree, the file counts, what is covered, what is
28
+ excluded — the harness proposes a module map, and `add` writes it after
29
+ validating it.
30
+
31
+ **A re-run never removes.** `init` is meant to be run again as a repository
32
+ grows, which makes periodic-plus-destructive the shape to avoid: dropping a
33
+ module orphans its document and `prune` then deletes it. So it is additive and
34
+ proposing only. Removing a module stays a hand edit, because it destroys a
35
+ document and that should cost somebody a decision.
36
+ """
37
+
38
+ from __future__ import annotations
39
+
40
+ from dataclasses import dataclass
41
+ from pathlib import Path
42
+ from typing import Any
43
+
44
+ from constant_docs.config import _walk
45
+ from constant_docs.globs import matches as glob_matches
46
+
47
+ # Files that are the tool's own furniture rather than a repository's source.
48
+ # Nothing would ever document them, and proposing them wastes the reader's
49
+ # attention on the one output where attention is the whole cost.
50
+ _FURNITURE = ("constant-docs.yaml",)
51
+
52
+
53
+ def _all_files(cfg: Any, repo_root: Path) -> list[str]:
54
+ """Every candidate file, repository-relative and POSIX-separated.
55
+
56
+ The same walk glob resolution uses, so secrets, build caches and the docs
57
+ root are already gone — a file that can never be hashed must never be
58
+ proposed for documenting either.
59
+ """
60
+ return _walk(repo_root.resolve(), cfg.docs_root)
61
+
62
+
63
+ def uncovered_files(
64
+ cfg: Any, repo_root: Path, all_files: list[str] | None = None
65
+ ) -> list[Path]:
66
+ """Return every file no module covers and no exclusion names, sorted.
67
+
68
+ *all_files* lets a caller that has already walked the tree hand the
69
+ result in. `facts` needs the same list for its directory counts, and
70
+ walking twice for one question made `init` — the command written to be
71
+ re-run as a repository grows — pay for three walks where one would do.
72
+ """
73
+ covered = {f.as_posix() for files in cfg.module_files.values() for f in files}
74
+ out: list[Path] = []
75
+ for rel in _all_files(cfg, repo_root) if all_files is None else all_files:
76
+ if rel in covered or rel in _FURNITURE:
77
+ continue
78
+ if any(glob_matches(pattern, rel) for pattern in cfg.uncovered):
79
+ continue
80
+ out.append(Path(rel))
81
+ return sorted(out)
82
+
83
+
84
+ def by_directory(files: list[Path]) -> dict[str, list[str]]:
85
+ """Group *files* by their parent directory.
86
+
87
+ Four hundred paths is a list; twelve directories is a decision. The output
88
+ of a coverage report is meant to be acted on, and a report nobody can act
89
+ on is one nobody runs twice.
90
+ """
91
+ grouped: dict[str, list[str]] = {}
92
+ for f in files:
93
+ parent = f.parent.as_posix()
94
+ grouped.setdefault(parent if parent else ".", []).append(f.as_posix())
95
+ return {k: sorted(v) for k, v in sorted(grouped.items())}
96
+
97
+
98
+ @dataclass(frozen=True)
99
+ class Directory:
100
+ """One directory of the repository, and what the configuration says of it."""
101
+
102
+ path: str
103
+ files: int
104
+ bytes: int
105
+ covered: bool
106
+
107
+
108
+ @dataclass(frozen=True)
109
+ class Facts:
110
+ """Everything `init` knows, for a harness that has to propose a module map.
111
+
112
+ Facts only. What a module *is* in somebody's code is a judgement, and this
113
+ package holds no model to make one.
114
+ """
115
+
116
+ directories: list[Directory]
117
+ uncovered: list[str]
118
+ excluded: dict[str, str]
119
+ empty_modules: list[str]
120
+
121
+
122
+ def facts(cfg: Any, repo_root: Path) -> Facts:
123
+ """Gather what a harness needs to propose a module map."""
124
+ covered = {f.as_posix() for files in cfg.module_files.values() for f in files}
125
+ # Walked once and used twice. Both the exclusion pass and the directory
126
+ # counts below need the same list.
127
+ all_files = _all_files(cfg, repo_root)
128
+ missing = uncovered_files(cfg, repo_root, all_files=all_files)
129
+
130
+ counts: dict[str, int] = {}
131
+ sizes: dict[str, int] = {}
132
+ covered_dirs: dict[str, bool] = {}
133
+ for rel in all_files:
134
+ if rel in _FURNITURE:
135
+ continue
136
+ parent = Path(rel).parent.as_posix() or "."
137
+ counts[parent] = counts.get(parent, 0) + 1
138
+ try:
139
+ sizes[parent] = sizes.get(parent, 0) + (repo_root / rel).stat().st_size
140
+ except OSError:
141
+ sizes.setdefault(parent, 0)
142
+ # A directory counts as covered when anything in it is, which is what a
143
+ # reader deciding where to look next actually wants to know.
144
+ covered_dirs[parent] = covered_dirs.get(parent, False) or rel in covered
145
+
146
+ return Facts(
147
+ directories=[
148
+ Directory(
149
+ path=path,
150
+ files=counts[path],
151
+ bytes=sizes.get(path, 0),
152
+ covered=covered_dirs.get(path, False),
153
+ )
154
+ for path in sorted(counts)
155
+ ],
156
+ uncovered=[p.as_posix() for p in missing],
157
+ excluded=dict(sorted(cfg.uncovered.items())),
158
+ # A module whose glob matches nothing. Reported, never removed: its
159
+ # document is an orphan, and `prune` deletes an orphan's document, so a
160
+ # periodic command that dropped the entry would delete prose on a
161
+ # schedule.
162
+ empty_modules=sorted(
163
+ key for key, files in cfg.module_files.items() if not files
164
+ ),
165
+ )
@@ -0,0 +1,303 @@
1
+ """The decisions a document records, and the guard against losing one.
2
+
3
+ ## Why this exists
4
+
5
+ `docs/SPEC.md` is a tracked document in replace mode: a model rewrites its
6
+ body whenever the source moves. The decisions it records — *the tool never
7
+ calls a model*, *hash contents rather than the interface*, *no source parsing*
8
+ — are choices the source cannot state for itself, and they are the reason the
9
+ tool is what it is.
10
+
11
+ What protected them before this module was a paragraph in the `spec` prompt
12
+ asking the model to edit rather than rewrite. That paragraph is the right
13
+ instruction and it is **not a control**. A regeneration that drops a decision,
14
+ keeps its required sections and updates its hash passes `verify` and reports
15
+ success. The failure mode is not a model behaving badly; it is that when one
16
+ does, nothing says so — and fifty regenerations later the leak is silent,
17
+ gradual, and invisible to every check the tool has.
18
+
19
+ ## Why detection rather than ownership
20
+
21
+ The obvious answer is to mark the specification human-owned and forbid the
22
+ tool to write it. That is the wrong one, and this project has the evidence:
23
+ the documents that rotted were exactly the human-versioned ones. Ownership is
24
+ not a mechanism. What makes an agent-maintained document safe is that **loss
25
+ is detectable**, so a regeneration is free to rewrite anything and unable to
26
+ drop something without saying so.
27
+
28
+ ## The marker
29
+
30
+ A decision carries a stable identifier, in the document rather than in a side
31
+ file — the same reasoning that put the hash in frontmatter rather than a lock
32
+ file. A side file is a second thing to keep in step, and the document travels
33
+ without it.
34
+
35
+ ```markdown
36
+ | <!-- decision: no-model-client --> Never call a model | The harness … |
37
+ ```
38
+
39
+ An HTML comment because it is the one syntax that works everywhere a decision
40
+ actually gets written — a table cell, a paragraph, a heading — and renders as
41
+ nothing, so the identifier costs the reader no attention. Put it at the start
42
+ of the decision: everything from the marker to the next marker or the next H2
43
+ is that decision's content, so a marker at the start means the whole decision
44
+ is covered by the emptiness check.
45
+
46
+ ## The limit, stated rather than discovered
47
+
48
+ This catches a decision being **deleted**. It cannot catch one being
49
+ **degraded** — reworded into something weaker that still says words. No check
50
+ this tool can make would, short of understanding the prose, and a check that
51
+ fought editing would make the tool useless for the thing it exists to do: a
52
+ body that rewords every decision while keeping every identifier is accepted,
53
+ deliberately.
54
+ """
55
+
56
+ from __future__ import annotations
57
+
58
+ import re
59
+ from collections.abc import Iterable
60
+
61
+ # `<!-- decision: some-slug -->`, tolerant of the whitespace a formatter or a
62
+ # model will introduce, strict about the slug: lowercase, digits and hyphens,
63
+ # so an identifier cannot drift by capitalisation while looking unchanged.
64
+ _MARKER_RE = re.compile(
65
+ r"<!--\s*decision\s*:\s*([a-z0-9][a-z0-9-]*)\s*-->",
66
+ re.IGNORECASE,
67
+ )
68
+
69
+ # A fenced block, and an inline code span. Both are stripped before anything
70
+ # is matched: a marker inside either is an example of the syntax, not a
71
+ # decision. This repository's own specification documents the marker, so the
72
+ # distinction is load-bearing rather than theoretical.
73
+ _FENCE_RE = re.compile(r"^```.*?^```", re.MULTILINE | re.DOTALL)
74
+ _CODE_SPAN_RE = re.compile(r"`[^`\n]*`")
75
+
76
+ # A decision's content runs to the next marker or the next section, whichever
77
+ # comes first. Without the section stop, the last decision in
78
+ # `## Design decisions` would swallow `## Acceptance criteria` and every later
79
+ # section, and emptying it would then be undetectable.
80
+ _SECTION_RE = re.compile(r"^## ", re.MULTILINE)
81
+
82
+
83
+ class DecisionError(Exception):
84
+ """A body's decisions are wrong in a way that refuses the write.
85
+
86
+ Its own family rather than a `DocumentError`, so this module depends on
87
+ nothing but the standard library — the same property that lets `globs` be
88
+ shared by the resolver and by `mark`.
89
+ """
90
+
91
+
92
+ class DuplicateDecision(DecisionError):
93
+ """Two decisions in one body share an identifier.
94
+
95
+ Refused rather than resolved, because the guard answers "is this decision
96
+ still here" by name. Two decisions under one name means dropping either
97
+ leaves the name present and the loss invisible, which is the exact failure
98
+ this module exists to prevent.
99
+ """
100
+
101
+
102
+ class DecisionLoss(DecisionError):
103
+ """A rewrite dropped or hollowed out a decision without declaring it.
104
+
105
+ Carries the identifiers rather than only a message, so a caller can act on
106
+ them — put them back, or declare the retirement — without parsing prose.
107
+ """
108
+
109
+ def __init__(self, message: str, decisions: list[str]) -> None:
110
+ super().__init__(message)
111
+ self.decisions = decisions
112
+
113
+
114
+ def _strip_examples(body: str) -> str:
115
+ """Return *body* with fenced blocks and inline code spans removed."""
116
+ return _CODE_SPAN_RE.sub("", _FENCE_RE.sub("", body))
117
+
118
+
119
+ def _spans(body: str) -> list[tuple[str, str]]:
120
+ """Return (identifier, content) in document order, duplicates included.
121
+
122
+ Content is everything from the end of a marker to the next marker or the
123
+ next H2, whichever comes first.
124
+ """
125
+ text = _strip_examples(body)
126
+ found = list(_MARKER_RE.finditer(text))
127
+ out: list[tuple[str, str]] = []
128
+ for index, match in enumerate(found):
129
+ start = match.end()
130
+ end = found[index + 1].start() if index + 1 < len(found) else len(text)
131
+ section = _SECTION_RE.search(text, start, end)
132
+ if section is not None:
133
+ end = section.start()
134
+ out.append((match.group(1).lower(), text[start:end]))
135
+ return out
136
+
137
+
138
+ def _first_wins(spans: list[tuple[str, str]]) -> dict[str, str]:
139
+ """Collapse duplicates by keeping the first, for comparison purposes.
140
+
141
+ `identifiers` refuses a duplicate outright. The comparison functions do
142
+ not, because the body they are comparing *against* is whatever is already
143
+ on disk — possibly written before this rule existed — and a document that
144
+ cannot be read is a document that can never be repaired.
145
+ """
146
+ out: dict[str, str] = {}
147
+ for name, content in spans:
148
+ out.setdefault(name, content)
149
+ return out
150
+
151
+
152
+ def _carries_content(text: str) -> bool:
153
+ """True when *text* holds anything a reader would call a decision.
154
+
155
+ Alphanumeric characters, and nothing else, because the alternative is a
156
+ length threshold and there is no honest number to pick. This is the line
157
+ between *deleted* and *present*: a table row gutted to `| | |` renders as
158
+ a row and records nothing, and a reworded decision is accepted whatever
159
+ its length.
160
+ """
161
+ return any(ch.isalnum() for ch in text)
162
+
163
+
164
+ def identifiers(body: str) -> dict[str, str]:
165
+ """Return every decision *body* records, as identifier to content.
166
+
167
+ Raises :exc:`DuplicateDecision` when one identifier appears twice.
168
+ """
169
+ spans = _spans(body)
170
+ repeated = duplicates(body)
171
+ if repeated:
172
+ raise DuplicateDecision(
173
+ "Two decisions share an identifier: "
174
+ + ", ".join(repeated)
175
+ + ". An identifier is how a decision is found again, so it has to "
176
+ "name exactly one."
177
+ )
178
+ return _first_wins(spans)
179
+
180
+
181
+ def _duplicates_in(spans: list[tuple[str, str]]) -> list[str]:
182
+ """Return every identifier *spans* uses more than once, sorted."""
183
+ seen: set[str] = set()
184
+ repeated: set[str] = set()
185
+ for name, _ in spans:
186
+ if name in seen:
187
+ repeated.add(name)
188
+ seen.add(name)
189
+ return sorted(repeated)
190
+
191
+
192
+ def duplicates(body: str) -> list[str]:
193
+ """Return every identifier *body* uses more than once, sorted."""
194
+ return _duplicates_in(_spans(body))
195
+
196
+
197
+ def lost(previous: str, current: str) -> list[str]:
198
+ """Return the identifiers *previous* recorded and *current* does not.
199
+
200
+ Sorted, and complete: every dropped identifier, not the first one found.
201
+ A refusal that names one of three is a refusal somebody satisfies three
202
+ times.
203
+ """
204
+ return _lost_in(_first_wins(_spans(previous)), _first_wins(_spans(current)))
205
+
206
+
207
+ def _lost_in(before: dict[str, str], after: dict[str, str]) -> list[str]:
208
+ """Return the identifiers *before* recorded and *after* does not."""
209
+ return sorted(set(before) - set(after))
210
+
211
+
212
+ def check_rewrite(previous: str, current: str, retire: Iterable[str] = ()) -> list[str]:
213
+ """Refuse *current* if it loses a decision *previous* recorded.
214
+
215
+ Returns the identifiers this write retires, sorted, for the caller to
216
+ record. Raises :exc:`DecisionLoss` naming every identifier at fault, and
217
+ :exc:`DuplicateDecision` when *current* uses one identifier twice.
218
+
219
+ A declared retirement is checked against what actually happened, both
220
+ ways. Declaring the removal of a decision the body still records is
221
+ refused, and so is declaring one that was never there. Without that,
222
+ `--retire` becomes a flag somebody pastes in to make the refusal go away,
223
+ which is the escape hatch defeating the control — the same failure as a
224
+ lint suppression applied to a whole file.
225
+ """
226
+ # Both bodies scanned once, here, rather than once inside each helper.
227
+ # `duplicates`, `lost` and `emptied` each re-derived the spans, and every
228
+ # `_spans` re-runs `_strip_examples` over the whole body — five passes and
229
+ # ten full-body substitutions for a document this repository's own SPEC
230
+ # makes 466 KB of scanning where 186 KB is the whole question.
231
+ before_raw = _spans(previous)
232
+ after_raw = _spans(current)
233
+ before_spans = _first_wins(before_raw)
234
+ after_spans = _first_wins(after_raw)
235
+
236
+ repeated = _duplicates_in(after_raw)
237
+ if repeated:
238
+ raise DuplicateDecision(
239
+ "Two decisions share an identifier: "
240
+ + ", ".join(repeated)
241
+ + ". An identifier is how a decision is found again, so it has to "
242
+ "name exactly one."
243
+ )
244
+
245
+ declared = {name.strip().lower() for name in retire if name.strip()}
246
+ gone = set(_lost_in(before_spans, after_spans)) | set(
247
+ _emptied_in(before_spans, after_spans)
248
+ )
249
+
250
+ undeclared = sorted(gone - declared)
251
+ if undeclared:
252
+ raise DecisionLoss(
253
+ "This body drops "
254
+ + ("a decision" if len(undeclared) == 1 else "decisions")
255
+ + " the previous one recorded: "
256
+ + ", ".join(undeclared)
257
+ + ". Put "
258
+ + ("it" if len(undeclared) == 1 else "them")
259
+ + " back, or retire "
260
+ + ("it" if len(undeclared) == 1 else "them")
261
+ + " deliberately so the retirement is recorded rather than being "
262
+ "an absence.",
263
+ undeclared,
264
+ )
265
+
266
+ spurious = sorted(declared - gone)
267
+ if spurious:
268
+ present = set(before_spans)
269
+ unknown = sorted(name for name in spurious if name not in present)
270
+ detail = (
271
+ f"never recorded: {', '.join(unknown)}"
272
+ if unknown
273
+ else f"still recorded by this body: {', '.join(spurious)}"
274
+ )
275
+ raise DecisionLoss(
276
+ f"A retirement was declared for a decision that is {detail}. A "
277
+ "retirement records something that happened; declaring one that "
278
+ "did not turns the guard into a flag somebody pastes in.",
279
+ spurious,
280
+ )
281
+
282
+ return sorted(declared)
283
+
284
+
285
+ def emptied(previous: str, current: str) -> list[str]:
286
+ """Return identifiers kept by *current* whose decision was hollowed out.
287
+
288
+ Only where *previous* carried content: the guard reports loss, not
289
+ pre-existing thinness, so adopting a document that already had an empty
290
+ decision does not become impossible.
291
+ """
292
+ return _emptied_in(_first_wins(_spans(previous)), _first_wins(_spans(current)))
293
+
294
+
295
+ def _emptied_in(before: dict[str, str], after: dict[str, str]) -> list[str]:
296
+ """Return the kept identifiers whose decision was hollowed out."""
297
+ return sorted(
298
+ name
299
+ for name, content in before.items()
300
+ if name in after
301
+ and _carries_content(content)
302
+ and not _carries_content(after[name])
303
+ )