adopt-knowledge 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,321 @@
1
+ """Files in, `Document` values out -- the ingest reader.
2
+
3
+ Markdown and plain text only, which is v6.1 §6 Build 2's v1 scope stated as a
4
+ capability rather than as an intention: nothing here can open a PDF or a DOCX,
5
+ so the richer-ingestion trigger has to fire before richer ingestion exists.
6
+
7
+ **Three heuristics, each overridable by frontmatter, each with a stated
8
+ default.** Title, kind and audience are guesses about a document a human wrote
9
+ for their own reasons, and a guess that cannot be corrected is a guess that
10
+ becomes wrong permanently. Frontmatter wins over the heuristic, always.
11
+
12
+ **The digest is over the body, not the file.** Frontmatter that changed while
13
+ the prose did not is not a new revision of the knowledge -- it is a new answer
14
+ to "who is this for", which lands on `audience_tag` instead. This is the same
15
+ reasoning H5 applies to identities one level up: what changed has to be the
16
+ thing the row is about.
17
+ """
18
+
19
+ import hashlib
20
+ from collections.abc import Iterable, Iterator, Sequence
21
+ from dataclasses import dataclass
22
+ from pathlib import Path
23
+ from typing import Any, Final
24
+
25
+ import yaml
26
+
27
+ from adopt_const import DETECT_MAX_FILES, MAP_MAX_FILE_BYTES
28
+ from adopt_detect import walk_files
29
+ from adopt_model._enums import ItemKind
30
+ from adopt_obs import AdoptError, ErrorCode
31
+
32
+ __all__ = [
33
+ "AUDIENCES",
34
+ "DEFAULT_AUDIENCE",
35
+ "DEFAULT_KIND",
36
+ "Document",
37
+ "body_digest",
38
+ "discover",
39
+ "read_document",
40
+ ]
41
+
42
+ #: The audience vocabulary v6.1 §6 Build 4 names for pack assembly. `audience_tag`
43
+ #: holds free text -- the manifest declares no enum -- so the vocabulary lives
44
+ #: here, where the writer is, and `--audience` accepts anything: a firm with a
45
+ #: fifth audience should not have to patch the tool, and a typo is visible in
46
+ #: `adopt gaps` rather than silently unmatched.
47
+ AUDIENCES: Final[tuple[str, ...]] = ("technical", "client_ops", "end_user", "admin")
48
+ DEFAULT_AUDIENCE: Final[str] = "technical"
49
+ #: An ingested document describes how something is done. `rationale` is what
50
+ #: harvest mines (why it is done that way) and is deliberately not the default
51
+ #: here: labelling every README a decision record would fill the decision
52
+ #: appendix of every future pack with installation instructions.
53
+ DEFAULT_KIND: Final[ItemKind] = "procedure"
54
+
55
+ _SUFFIXES: Final[frozenset[str]] = frozenset({".md", ".markdown", ".mdown", ".txt"})
56
+ _FENCE: Final[str] = "---"
57
+ _DIGEST_PREFIX: Final[str] = "sha256"
58
+
59
+ #: Path fragments that name an audience more reliably than the prose does. Order
60
+ #: matters: the first hit wins, so the more specific fragments come first.
61
+ _AUDIENCE_HINTS: Final[tuple[tuple[str, str], ...]] = (
62
+ ("runbook", "client_ops"),
63
+ ("operations", "client_ops"),
64
+ ("ops/", "client_ops"),
65
+ ("admin", "admin"),
66
+ ("install", "admin"),
67
+ ("deploy", "admin"),
68
+ ("user-guide", "end_user"),
69
+ ("user_guide", "end_user"),
70
+ ("guide", "end_user"),
71
+ ("tutorial", "end_user"),
72
+ ("getting-started", "end_user"),
73
+ )
74
+
75
+
76
+ @dataclass(frozen=True, slots=True)
77
+ class Document:
78
+ """One ingestable document, resolved.
79
+
80
+ `path` is POSIX-relative to the scanned root and is what
81
+ `provenance.source_ref` records, so a bundle written on Windows and read on
82
+ Linux cites the same file.
83
+ """
84
+
85
+ path: str
86
+ title: str
87
+ kind: ItemKind
88
+ audiences: tuple[str, ...]
89
+ body_md: str
90
+ digest: str
91
+
92
+
93
+ def body_digest(body: str) -> str:
94
+ """`sha256:<hex>` over the document body.
95
+
96
+ Newlines are normalised first. A checkout with `core.autocrlf=true` must
97
+ not present every document as changed -- the CRLF lesson the 0.3.1 release
98
+ paid for, applied where the next writer would meet it.
99
+ """
100
+ normalised = body.replace("\r\n", "\n").replace("\r", "\n")
101
+ digest = hashlib.sha256(normalised.encode("utf-8")).hexdigest()
102
+ return f"{_DIGEST_PREFIX}:{digest}"
103
+
104
+
105
+ def split_frontmatter(text: str) -> tuple[dict[str, Any], str]:
106
+ """`(frontmatter, body)`. Malformed or absent frontmatter yields `({}, text)`.
107
+
108
+ A document whose frontmatter does not parse is ingested as prose rather than
109
+ refused: the YAML is an optional convenience, and refusing the file would
110
+ lose the knowledge in it over a typo in a field nobody required.
111
+ """
112
+ if not text.startswith(_FENCE):
113
+ return {}, text
114
+ lines = text.splitlines(keepends=True)
115
+ for index in range(1, len(lines)):
116
+ if lines[index].strip() == _FENCE:
117
+ block = "".join(lines[1:index])
118
+ body = "".join(lines[index + 1 :])
119
+ try:
120
+ loaded = yaml.safe_load(block)
121
+ except yaml.YAMLError:
122
+ return {}, text
123
+ return (loaded if isinstance(loaded, dict) else {}), body
124
+ return {}, text
125
+
126
+
127
+ def _title_of(frontmatter: dict[str, Any], body: str, path: Path) -> str:
128
+ declared = frontmatter.get("title")
129
+ if isinstance(declared, str) and declared.strip():
130
+ return declared.strip()
131
+ for line in body.splitlines():
132
+ stripped = line.strip()
133
+ if stripped.startswith("#"):
134
+ heading = stripped.lstrip("#").strip()
135
+ if heading:
136
+ return heading
137
+ return path.stem.replace("-", " ").replace("_", " ").strip() or path.name
138
+
139
+
140
+ def _kind_of(frontmatter: dict[str, Any]) -> ItemKind:
141
+ declared = frontmatter.get("kind")
142
+ if isinstance(declared, str) and declared in _ITEM_KINDS:
143
+ # `ItemKind` is a Literal alias, so the membership test is the narrowing.
144
+ return declared # type: ignore[return-value]
145
+ return DEFAULT_KIND
146
+
147
+
148
+ def _audiences_of(frontmatter: dict[str, Any], relative_path: str) -> tuple[str, ...]:
149
+ declared = frontmatter.get("audience", frontmatter.get("audiences"))
150
+ if isinstance(declared, str) and declared.strip():
151
+ return (declared.strip(),)
152
+ if isinstance(declared, list):
153
+ tags = tuple(item.strip() for item in declared if isinstance(item, str) and item.strip())
154
+ if tags:
155
+ return tags
156
+ lowered = relative_path.lower()
157
+ for fragment, audience in _AUDIENCE_HINTS:
158
+ if fragment in lowered:
159
+ return (audience,)
160
+ return (DEFAULT_AUDIENCE,)
161
+
162
+
163
+ _ITEM_KINDS: Final[frozenset[str]] = frozenset(
164
+ {"answer", "procedure", "rationale", "surface", "recipe"}
165
+ )
166
+
167
+
168
+ def read_document(path: Path, *, root: Path, audience: str | None = None) -> Document:
169
+ """Read one file into a `Document`.
170
+
171
+ Args:
172
+ path: The file to read.
173
+ root: What `path` is reported relative to.
174
+ audience: An operator override that beats both frontmatter and the path
175
+ heuristic, because the operator is looking at the document and the
176
+ heuristic is looking at its name.
177
+
178
+ Raises:
179
+ AdoptError: ``KNOWLEDGE_SOURCE_UNREADABLE`` when the file cannot be read
180
+ or decoded, and when it exceeds `MAP_MAX_FILE_BYTES`. Refused rather
181
+ than skipped: a document named on the command line and silently
182
+ dropped is a corpus that is quietly smaller than the operator
183
+ believes, which is the shape of every defect Build 1 found by
184
+ running on a real repository.
185
+ """
186
+ try:
187
+ raw = path.read_bytes()
188
+ except OSError as error:
189
+ raise AdoptError(
190
+ ErrorCode.KNOWLEDGE_SOURCE_UNREADABLE,
191
+ message=f"{str(path)!r} could not be read",
192
+ hint="Name a readable Markdown or text file. An unreadable source is refused "
193
+ "rather than skipped, because a smaller corpus that reports success is "
194
+ "indistinguishable from a complete one.",
195
+ ) from error
196
+ if len(raw) > MAP_MAX_FILE_BYTES:
197
+ raise AdoptError(
198
+ ErrorCode.KNOWLEDGE_SOURCE_UNREADABLE,
199
+ message=f"{str(path)!r} is larger than the {MAP_MAX_FILE_BYTES}-byte ingest bound",
200
+ hint="Split the document, or point ingest at the sections worth binding. The "
201
+ "bound is the walk's, shared so one file cannot make a run unbounded.",
202
+ )
203
+ try:
204
+ text = raw.decode("utf-8")
205
+ except UnicodeDecodeError as error:
206
+ raise AdoptError(
207
+ ErrorCode.KNOWLEDGE_SOURCE_UNREADABLE,
208
+ message=f"{str(path)!r} is not valid UTF-8",
209
+ hint="Ingest reads UTF-8 text. A file that is not text is not a document, and "
210
+ "guessing an encoding would put mojibake into the knowledge store.",
211
+ ) from error
212
+
213
+ frontmatter, body = split_frontmatter(text)
214
+ relative = _relative_to(path, root)
215
+ return Document(
216
+ path=relative,
217
+ title=_title_of(frontmatter, body, path),
218
+ kind=_kind_of(frontmatter),
219
+ audiences=(audience,) if audience else _audiences_of(frontmatter, relative),
220
+ body_md=body,
221
+ digest=body_digest(body),
222
+ )
223
+
224
+
225
+ def _relative_to(path: Path, root: Path) -> str:
226
+ try:
227
+ return path.resolve().relative_to(root.resolve()).as_posix()
228
+ except ValueError:
229
+ # Outside the root: the absolute path is the honest citation, and a
230
+ # `../..` chain would resolve differently from wherever it is later read.
231
+ return path.resolve().as_posix()
232
+
233
+
234
+ def _candidates(paths: Sequence[Path]) -> Iterator[Path]:
235
+ """Every file to ingest, through **the one walk**.
236
+
237
+ A directory is enumerated by `adopt_detect.walk_files` -- the walker whose
238
+ own docstring says *"this is the one walk"* -- with the suffix filter
239
+ applied on top. It was `Path.rglob("*")` until T1.6, which has no
240
+ containment rule, no symlink rule, no depth bound, no `.gitignore` scope and
241
+ no count bound: a repository containing `docs/elsewhere -> /etc` copied
242
+ out-of-tree material into the knowledge store, and `_relative_to` then
243
+ recorded the **absolute** path of it as the document's provenance, because a
244
+ path outside the root has no relative form (B2-05).
245
+
246
+ An explicitly named **file** keeps today's policy and is ingested whatever
247
+ its suffix and wherever it lives: an operator naming a path outside the tree
248
+ is making a choice, and the walk's rules exist to stop a *sweep* reaching
249
+ where nobody pointed it.
250
+
251
+ Raises:
252
+ AdoptError: ``KNOWLEDGE_SOURCE_UNREADABLE`` when one directory yields
253
+ more than `DETECT_MAX_FILES` candidates -- contracts §13 already
254
+ names "exceeds the walk's file bound" as this code's subject.
255
+ Refused rather than truncated, on `MAP_TREE_TOO_LARGE`'s argument: a
256
+ corpus quietly smaller than the operator believes reports success
257
+ over knowledge that was never ingested.
258
+ """
259
+ for path in paths:
260
+ if path.is_dir():
261
+ found: list[Path] = []
262
+ for walked, (_relative, absolute) in enumerate(walk_files(path), 1):
263
+ if walked > DETECT_MAX_FILES:
264
+ raise AdoptError(
265
+ ErrorCode.KNOWLEDGE_SOURCE_UNREADABLE,
266
+ message=f"{str(path)!r} holds more than {DETECT_MAX_FILES} walkable files",
267
+ hint="Ingest a subdirectory, or exclude generated and vendored "
268
+ "trees with .gitignore. The bound is the walk's, shared so one "
269
+ "directory cannot make a run unbounded.",
270
+ )
271
+ if absolute.suffix.lower() in _SUFFIXES:
272
+ found.append(absolute)
273
+ yield from sorted(found)
274
+ elif path.is_file():
275
+ # A file named outright is ingested whatever its suffix: the
276
+ # operator pointed at it, and the suffix filter exists to keep a
277
+ # directory walk from sweeping up images, not to overrule a person.
278
+ yield path
279
+ else:
280
+ raise AdoptError(
281
+ ErrorCode.KNOWLEDGE_SOURCE_UNREADABLE,
282
+ message=f"{str(path)!r} is neither a file nor a directory",
283
+ hint="Name documents or directories that exist. A path that is not there "
284
+ "is refused rather than contributing nothing, because a typo and an "
285
+ "empty directory would otherwise look identical.",
286
+ )
287
+
288
+
289
+ def discover(
290
+ paths: Sequence[Path], *, root: Path, audience: str | None = None
291
+ ) -> tuple[Document, ...]:
292
+ """Every document under `paths`, deduplicated and ordered by path.
293
+
294
+ Ordering is the writer's, exactly as it is for the export bundle: two runs
295
+ over one tree must produce one sequence of items, or the ids differ and the
296
+ review queue is a different queue.
297
+ """
298
+ seen: dict[str, Document] = {}
299
+ for candidate in _candidates(paths):
300
+ document = read_document(candidate, root=root, audience=audience)
301
+ seen.setdefault(document.path, document)
302
+ return tuple(seen[path] for path in sorted(seen))
303
+
304
+
305
+ def audience_is_known(audience: str) -> bool:
306
+ """Whether an audience is one the pack builder recognises (v6.1 §6 B4)."""
307
+ return audience in AUDIENCES
308
+
309
+
310
+ def unknown_audiences(documents: Iterable[Document]) -> tuple[str, ...]:
311
+ """Audiences outside the vocabulary, sorted -- reported, never rejected."""
312
+ return tuple(
313
+ sorted(
314
+ {
315
+ audience
316
+ for document in documents
317
+ for audience in document.audiences
318
+ if not audience_is_known(audience)
319
+ }
320
+ )
321
+ )