adopt-knowledge 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- adopt_knowledge/__init__.py +246 -0
- adopt_knowledge/changes.py +378 -0
- adopt_knowledge/documents.py +321 -0
- adopt_knowledge/drafting.py +615 -0
- adopt_knowledge/gaps.py +219 -0
- adopt_knowledge/gitlog.py +294 -0
- adopt_knowledge/harvest.py +454 -0
- adopt_knowledge/ingest.py +334 -0
- adopt_knowledge/matchers.py +326 -0
- adopt_knowledge/ports.py +204 -0
- adopt_knowledge/py.typed +0 -0
- adopt_knowledge/review.py +535 -0
- adopt_knowledge-0.4.0.dist-info/METADATA +20 -0
- adopt_knowledge-0.4.0.dist-info/RECORD +17 -0
- adopt_knowledge-0.4.0.dist-info/WHEEL +4 -0
- adopt_knowledge-0.4.0.dist-info/licenses/LICENSE +201 -0
- adopt_knowledge-0.4.0.dist-info/licenses/NOTICE +39 -0
|
@@ -0,0 +1,321 @@
|
|
|
1
|
+
"""Files in, `Document` values out -- the ingest reader.
|
|
2
|
+
|
|
3
|
+
Markdown and plain text only, which is v6.1 §6 Build 2's v1 scope stated as a
|
|
4
|
+
capability rather than as an intention: nothing here can open a PDF or a DOCX,
|
|
5
|
+
so the richer-ingestion trigger has to fire before richer ingestion exists.
|
|
6
|
+
|
|
7
|
+
**Three heuristics, each overridable by frontmatter, each with a stated
|
|
8
|
+
default.** Title, kind and audience are guesses about a document a human wrote
|
|
9
|
+
for their own reasons, and a guess that cannot be corrected is a guess that
|
|
10
|
+
becomes wrong permanently. Frontmatter wins over the heuristic, always.
|
|
11
|
+
|
|
12
|
+
**The digest is over the body, not the file.** Frontmatter that changed while
|
|
13
|
+
the prose did not is not a new revision of the knowledge -- it is a new answer
|
|
14
|
+
to "who is this for", which lands on `audience_tag` instead. This is the same
|
|
15
|
+
reasoning H5 applies to identities one level up: what changed has to be the
|
|
16
|
+
thing the row is about.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
import hashlib
|
|
20
|
+
from collections.abc import Iterable, Iterator, Sequence
|
|
21
|
+
from dataclasses import dataclass
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
from typing import Any, Final
|
|
24
|
+
|
|
25
|
+
import yaml
|
|
26
|
+
|
|
27
|
+
from adopt_const import DETECT_MAX_FILES, MAP_MAX_FILE_BYTES
|
|
28
|
+
from adopt_detect import walk_files
|
|
29
|
+
from adopt_model._enums import ItemKind
|
|
30
|
+
from adopt_obs import AdoptError, ErrorCode
|
|
31
|
+
|
|
32
|
+
__all__ = [
|
|
33
|
+
"AUDIENCES",
|
|
34
|
+
"DEFAULT_AUDIENCE",
|
|
35
|
+
"DEFAULT_KIND",
|
|
36
|
+
"Document",
|
|
37
|
+
"body_digest",
|
|
38
|
+
"discover",
|
|
39
|
+
"read_document",
|
|
40
|
+
]
|
|
41
|
+
|
|
42
|
+
#: The audience vocabulary v6.1 §6 Build 4 names for pack assembly. `audience_tag`
|
|
43
|
+
#: holds free text -- the manifest declares no enum -- so the vocabulary lives
|
|
44
|
+
#: here, where the writer is, and `--audience` accepts anything: a firm with a
|
|
45
|
+
#: fifth audience should not have to patch the tool, and a typo is visible in
|
|
46
|
+
#: `adopt gaps` rather than silently unmatched.
|
|
47
|
+
AUDIENCES: Final[tuple[str, ...]] = ("technical", "client_ops", "end_user", "admin")
|
|
48
|
+
DEFAULT_AUDIENCE: Final[str] = "technical"
|
|
49
|
+
#: An ingested document describes how something is done. `rationale` is what
|
|
50
|
+
#: harvest mines (why it is done that way) and is deliberately not the default
|
|
51
|
+
#: here: labelling every README a decision record would fill the decision
|
|
52
|
+
#: appendix of every future pack with installation instructions.
|
|
53
|
+
DEFAULT_KIND: Final[ItemKind] = "procedure"
|
|
54
|
+
|
|
55
|
+
_SUFFIXES: Final[frozenset[str]] = frozenset({".md", ".markdown", ".mdown", ".txt"})
|
|
56
|
+
_FENCE: Final[str] = "---"
|
|
57
|
+
_DIGEST_PREFIX: Final[str] = "sha256"
|
|
58
|
+
|
|
59
|
+
#: Path fragments that name an audience more reliably than the prose does. Order
|
|
60
|
+
#: matters: the first hit wins, so the more specific fragments come first.
|
|
61
|
+
_AUDIENCE_HINTS: Final[tuple[tuple[str, str], ...]] = (
|
|
62
|
+
("runbook", "client_ops"),
|
|
63
|
+
("operations", "client_ops"),
|
|
64
|
+
("ops/", "client_ops"),
|
|
65
|
+
("admin", "admin"),
|
|
66
|
+
("install", "admin"),
|
|
67
|
+
("deploy", "admin"),
|
|
68
|
+
("user-guide", "end_user"),
|
|
69
|
+
("user_guide", "end_user"),
|
|
70
|
+
("guide", "end_user"),
|
|
71
|
+
("tutorial", "end_user"),
|
|
72
|
+
("getting-started", "end_user"),
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
@dataclass(frozen=True, slots=True)
|
|
77
|
+
class Document:
|
|
78
|
+
"""One ingestable document, resolved.
|
|
79
|
+
|
|
80
|
+
`path` is POSIX-relative to the scanned root and is what
|
|
81
|
+
`provenance.source_ref` records, so a bundle written on Windows and read on
|
|
82
|
+
Linux cites the same file.
|
|
83
|
+
"""
|
|
84
|
+
|
|
85
|
+
path: str
|
|
86
|
+
title: str
|
|
87
|
+
kind: ItemKind
|
|
88
|
+
audiences: tuple[str, ...]
|
|
89
|
+
body_md: str
|
|
90
|
+
digest: str
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def body_digest(body: str) -> str:
|
|
94
|
+
"""`sha256:<hex>` over the document body.
|
|
95
|
+
|
|
96
|
+
Newlines are normalised first. A checkout with `core.autocrlf=true` must
|
|
97
|
+
not present every document as changed -- the CRLF lesson the 0.3.1 release
|
|
98
|
+
paid for, applied where the next writer would meet it.
|
|
99
|
+
"""
|
|
100
|
+
normalised = body.replace("\r\n", "\n").replace("\r", "\n")
|
|
101
|
+
digest = hashlib.sha256(normalised.encode("utf-8")).hexdigest()
|
|
102
|
+
return f"{_DIGEST_PREFIX}:{digest}"
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def split_frontmatter(text: str) -> tuple[dict[str, Any], str]:
|
|
106
|
+
"""`(frontmatter, body)`. Malformed or absent frontmatter yields `({}, text)`.
|
|
107
|
+
|
|
108
|
+
A document whose frontmatter does not parse is ingested as prose rather than
|
|
109
|
+
refused: the YAML is an optional convenience, and refusing the file would
|
|
110
|
+
lose the knowledge in it over a typo in a field nobody required.
|
|
111
|
+
"""
|
|
112
|
+
if not text.startswith(_FENCE):
|
|
113
|
+
return {}, text
|
|
114
|
+
lines = text.splitlines(keepends=True)
|
|
115
|
+
for index in range(1, len(lines)):
|
|
116
|
+
if lines[index].strip() == _FENCE:
|
|
117
|
+
block = "".join(lines[1:index])
|
|
118
|
+
body = "".join(lines[index + 1 :])
|
|
119
|
+
try:
|
|
120
|
+
loaded = yaml.safe_load(block)
|
|
121
|
+
except yaml.YAMLError:
|
|
122
|
+
return {}, text
|
|
123
|
+
return (loaded if isinstance(loaded, dict) else {}), body
|
|
124
|
+
return {}, text
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def _title_of(frontmatter: dict[str, Any], body: str, path: Path) -> str:
|
|
128
|
+
declared = frontmatter.get("title")
|
|
129
|
+
if isinstance(declared, str) and declared.strip():
|
|
130
|
+
return declared.strip()
|
|
131
|
+
for line in body.splitlines():
|
|
132
|
+
stripped = line.strip()
|
|
133
|
+
if stripped.startswith("#"):
|
|
134
|
+
heading = stripped.lstrip("#").strip()
|
|
135
|
+
if heading:
|
|
136
|
+
return heading
|
|
137
|
+
return path.stem.replace("-", " ").replace("_", " ").strip() or path.name
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _kind_of(frontmatter: dict[str, Any]) -> ItemKind:
|
|
141
|
+
declared = frontmatter.get("kind")
|
|
142
|
+
if isinstance(declared, str) and declared in _ITEM_KINDS:
|
|
143
|
+
# `ItemKind` is a Literal alias, so the membership test is the narrowing.
|
|
144
|
+
return declared # type: ignore[return-value]
|
|
145
|
+
return DEFAULT_KIND
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _audiences_of(frontmatter: dict[str, Any], relative_path: str) -> tuple[str, ...]:
|
|
149
|
+
declared = frontmatter.get("audience", frontmatter.get("audiences"))
|
|
150
|
+
if isinstance(declared, str) and declared.strip():
|
|
151
|
+
return (declared.strip(),)
|
|
152
|
+
if isinstance(declared, list):
|
|
153
|
+
tags = tuple(item.strip() for item in declared if isinstance(item, str) and item.strip())
|
|
154
|
+
if tags:
|
|
155
|
+
return tags
|
|
156
|
+
lowered = relative_path.lower()
|
|
157
|
+
for fragment, audience in _AUDIENCE_HINTS:
|
|
158
|
+
if fragment in lowered:
|
|
159
|
+
return (audience,)
|
|
160
|
+
return (DEFAULT_AUDIENCE,)
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
_ITEM_KINDS: Final[frozenset[str]] = frozenset(
|
|
164
|
+
{"answer", "procedure", "rationale", "surface", "recipe"}
|
|
165
|
+
)
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def read_document(path: Path, *, root: Path, audience: str | None = None) -> Document:
|
|
169
|
+
"""Read one file into a `Document`.
|
|
170
|
+
|
|
171
|
+
Args:
|
|
172
|
+
path: The file to read.
|
|
173
|
+
root: What `path` is reported relative to.
|
|
174
|
+
audience: An operator override that beats both frontmatter and the path
|
|
175
|
+
heuristic, because the operator is looking at the document and the
|
|
176
|
+
heuristic is looking at its name.
|
|
177
|
+
|
|
178
|
+
Raises:
|
|
179
|
+
AdoptError: ``KNOWLEDGE_SOURCE_UNREADABLE`` when the file cannot be read
|
|
180
|
+
or decoded, and when it exceeds `MAP_MAX_FILE_BYTES`. Refused rather
|
|
181
|
+
than skipped: a document named on the command line and silently
|
|
182
|
+
dropped is a corpus that is quietly smaller than the operator
|
|
183
|
+
believes, which is the shape of every defect Build 1 found by
|
|
184
|
+
running on a real repository.
|
|
185
|
+
"""
|
|
186
|
+
try:
|
|
187
|
+
raw = path.read_bytes()
|
|
188
|
+
except OSError as error:
|
|
189
|
+
raise AdoptError(
|
|
190
|
+
ErrorCode.KNOWLEDGE_SOURCE_UNREADABLE,
|
|
191
|
+
message=f"{str(path)!r} could not be read",
|
|
192
|
+
hint="Name a readable Markdown or text file. An unreadable source is refused "
|
|
193
|
+
"rather than skipped, because a smaller corpus that reports success is "
|
|
194
|
+
"indistinguishable from a complete one.",
|
|
195
|
+
) from error
|
|
196
|
+
if len(raw) > MAP_MAX_FILE_BYTES:
|
|
197
|
+
raise AdoptError(
|
|
198
|
+
ErrorCode.KNOWLEDGE_SOURCE_UNREADABLE,
|
|
199
|
+
message=f"{str(path)!r} is larger than the {MAP_MAX_FILE_BYTES}-byte ingest bound",
|
|
200
|
+
hint="Split the document, or point ingest at the sections worth binding. The "
|
|
201
|
+
"bound is the walk's, shared so one file cannot make a run unbounded.",
|
|
202
|
+
)
|
|
203
|
+
try:
|
|
204
|
+
text = raw.decode("utf-8")
|
|
205
|
+
except UnicodeDecodeError as error:
|
|
206
|
+
raise AdoptError(
|
|
207
|
+
ErrorCode.KNOWLEDGE_SOURCE_UNREADABLE,
|
|
208
|
+
message=f"{str(path)!r} is not valid UTF-8",
|
|
209
|
+
hint="Ingest reads UTF-8 text. A file that is not text is not a document, and "
|
|
210
|
+
"guessing an encoding would put mojibake into the knowledge store.",
|
|
211
|
+
) from error
|
|
212
|
+
|
|
213
|
+
frontmatter, body = split_frontmatter(text)
|
|
214
|
+
relative = _relative_to(path, root)
|
|
215
|
+
return Document(
|
|
216
|
+
path=relative,
|
|
217
|
+
title=_title_of(frontmatter, body, path),
|
|
218
|
+
kind=_kind_of(frontmatter),
|
|
219
|
+
audiences=(audience,) if audience else _audiences_of(frontmatter, relative),
|
|
220
|
+
body_md=body,
|
|
221
|
+
digest=body_digest(body),
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _relative_to(path: Path, root: Path) -> str:
|
|
226
|
+
try:
|
|
227
|
+
return path.resolve().relative_to(root.resolve()).as_posix()
|
|
228
|
+
except ValueError:
|
|
229
|
+
# Outside the root: the absolute path is the honest citation, and a
|
|
230
|
+
# `../..` chain would resolve differently from wherever it is later read.
|
|
231
|
+
return path.resolve().as_posix()
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _candidates(paths: Sequence[Path]) -> Iterator[Path]:
|
|
235
|
+
"""Every file to ingest, through **the one walk**.
|
|
236
|
+
|
|
237
|
+
A directory is enumerated by `adopt_detect.walk_files` -- the walker whose
|
|
238
|
+
own docstring says *"this is the one walk"* -- with the suffix filter
|
|
239
|
+
applied on top. It was `Path.rglob("*")` until T1.6, which has no
|
|
240
|
+
containment rule, no symlink rule, no depth bound, no `.gitignore` scope and
|
|
241
|
+
no count bound: a repository containing `docs/elsewhere -> /etc` copied
|
|
242
|
+
out-of-tree material into the knowledge store, and `_relative_to` then
|
|
243
|
+
recorded the **absolute** path of it as the document's provenance, because a
|
|
244
|
+
path outside the root has no relative form (B2-05).
|
|
245
|
+
|
|
246
|
+
An explicitly named **file** keeps today's policy and is ingested whatever
|
|
247
|
+
its suffix and wherever it lives: an operator naming a path outside the tree
|
|
248
|
+
is making a choice, and the walk's rules exist to stop a *sweep* reaching
|
|
249
|
+
where nobody pointed it.
|
|
250
|
+
|
|
251
|
+
Raises:
|
|
252
|
+
AdoptError: ``KNOWLEDGE_SOURCE_UNREADABLE`` when one directory yields
|
|
253
|
+
more than `DETECT_MAX_FILES` candidates -- contracts §13 already
|
|
254
|
+
names "exceeds the walk's file bound" as this code's subject.
|
|
255
|
+
Refused rather than truncated, on `MAP_TREE_TOO_LARGE`'s argument: a
|
|
256
|
+
corpus quietly smaller than the operator believes reports success
|
|
257
|
+
over knowledge that was never ingested.
|
|
258
|
+
"""
|
|
259
|
+
for path in paths:
|
|
260
|
+
if path.is_dir():
|
|
261
|
+
found: list[Path] = []
|
|
262
|
+
for walked, (_relative, absolute) in enumerate(walk_files(path), 1):
|
|
263
|
+
if walked > DETECT_MAX_FILES:
|
|
264
|
+
raise AdoptError(
|
|
265
|
+
ErrorCode.KNOWLEDGE_SOURCE_UNREADABLE,
|
|
266
|
+
message=f"{str(path)!r} holds more than {DETECT_MAX_FILES} walkable files",
|
|
267
|
+
hint="Ingest a subdirectory, or exclude generated and vendored "
|
|
268
|
+
"trees with .gitignore. The bound is the walk's, shared so one "
|
|
269
|
+
"directory cannot make a run unbounded.",
|
|
270
|
+
)
|
|
271
|
+
if absolute.suffix.lower() in _SUFFIXES:
|
|
272
|
+
found.append(absolute)
|
|
273
|
+
yield from sorted(found)
|
|
274
|
+
elif path.is_file():
|
|
275
|
+
# A file named outright is ingested whatever its suffix: the
|
|
276
|
+
# operator pointed at it, and the suffix filter exists to keep a
|
|
277
|
+
# directory walk from sweeping up images, not to overrule a person.
|
|
278
|
+
yield path
|
|
279
|
+
else:
|
|
280
|
+
raise AdoptError(
|
|
281
|
+
ErrorCode.KNOWLEDGE_SOURCE_UNREADABLE,
|
|
282
|
+
message=f"{str(path)!r} is neither a file nor a directory",
|
|
283
|
+
hint="Name documents or directories that exist. A path that is not there "
|
|
284
|
+
"is refused rather than contributing nothing, because a typo and an "
|
|
285
|
+
"empty directory would otherwise look identical.",
|
|
286
|
+
)
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def discover(
|
|
290
|
+
paths: Sequence[Path], *, root: Path, audience: str | None = None
|
|
291
|
+
) -> tuple[Document, ...]:
|
|
292
|
+
"""Every document under `paths`, deduplicated and ordered by path.
|
|
293
|
+
|
|
294
|
+
Ordering is the writer's, exactly as it is for the export bundle: two runs
|
|
295
|
+
over one tree must produce one sequence of items, or the ids differ and the
|
|
296
|
+
review queue is a different queue.
|
|
297
|
+
"""
|
|
298
|
+
seen: dict[str, Document] = {}
|
|
299
|
+
for candidate in _candidates(paths):
|
|
300
|
+
document = read_document(candidate, root=root, audience=audience)
|
|
301
|
+
seen.setdefault(document.path, document)
|
|
302
|
+
return tuple(seen[path] for path in sorted(seen))
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def audience_is_known(audience: str) -> bool:
|
|
306
|
+
"""Whether an audience is one the pack builder recognises (v6.1 §6 B4)."""
|
|
307
|
+
return audience in AUDIENCES
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
def unknown_audiences(documents: Iterable[Document]) -> tuple[str, ...]:
|
|
311
|
+
"""Audiences outside the vocabulary, sorted -- reported, never rejected."""
|
|
312
|
+
return tuple(
|
|
313
|
+
sorted(
|
|
314
|
+
{
|
|
315
|
+
audience
|
|
316
|
+
for document in documents
|
|
317
|
+
for audience in document.audiences
|
|
318
|
+
if not audience_is_known(audience)
|
|
319
|
+
}
|
|
320
|
+
)
|
|
321
|
+
)
|