surfaceplate 0.16.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- surfaceplate/MANIFEST.sha256 +230 -0
- surfaceplate/VERSION +1 -0
- surfaceplate/__init__.py +18 -0
- surfaceplate/about.py +48 -0
- surfaceplate/adapters/python.md +5 -0
- surfaceplate/adapters/r.md +3 -0
- surfaceplate/adapters/typescript.md +5 -0
- surfaceplate/adopt/__init__.py +15 -0
- surfaceplate/adopt/catalogue.py +78 -0
- surfaceplate/adopt/defaults.py +278 -0
- surfaceplate/adopt/detect.py +141 -0
- surfaceplate/adopt/discover.py +614 -0
- surfaceplate/adopt/example_answers.py +128 -0
- surfaceplate/adopt/explanations.py +551 -0
- surfaceplate/adopt/flow.py +559 -0
- surfaceplate/adopt/interview.py +237 -0
- surfaceplate/adopt/plan.py +1330 -0
- surfaceplate/adopt/provenance.py +300 -0
- surfaceplate/adopt/render.py +282 -0
- surfaceplate/adopt/scaffold.py +343 -0
- surfaceplate/adopt/sections.py +287 -0
- surfaceplate/adopt/tui/__init__.py +7 -0
- surfaceplate/adopt/tui/app.py +211 -0
- surfaceplate/adopt/tui/app.tcss +294 -0
- surfaceplate/adopt/tui/mark.py +77 -0
- surfaceplate/adopt/tui/screens.py +1772 -0
- surfaceplate/adopt/validators.py +224 -0
- surfaceplate/adopt/wizard.py +932 -0
- surfaceplate/check_conformance.py +3664 -0
- surfaceplate/cli.py +268 -0
- surfaceplate/core/AI_OPERATING_MODEL.md +45 -0
- surfaceplate/core/CONFORMANCE_LEVELS.md +299 -0
- surfaceplate/core/CONTROL_PRINCIPLES.md +14 -0
- surfaceplate/core/PREREQUISITE_GATES.md +322 -0
- surfaceplate/core/REVIEW_AND_EVIDENCE.md +55 -0
- surfaceplate/core/SECURITY_BASELINE.md +29 -0
- surfaceplate/doctor.py +249 -0
- surfaceplate/examples/application-profile.essential.example.yaml +126 -0
- surfaceplate/examples/application-profile.full.example.yaml +315 -0
- surfaceplate/examples/method-registry-entry.example.yaml +78 -0
- surfaceplate/examples/method-run-lineage.example.yaml +44 -0
- surfaceplate/examples/override-record.approved.example.yaml +39 -0
- surfaceplate/install_standard.py +880 -0
- surfaceplate/rules.py +148 -0
- surfaceplate/schemas/README.md +24 -0
- surfaceplate/schemas/application-profile.schema.yaml +320 -0
- surfaceplate/schemas/assurance-evidence.schema.yaml +33 -0
- surfaceplate/schemas/gate-exception.schema.yaml +47 -0
- surfaceplate/schemas/method-registry-entry.schema.yaml +132 -0
- surfaceplate/schemas/method-run-lineage.schema.yaml +72 -0
- surfaceplate/schemas/override-record.schema.yaml +76 -0
- surfaceplate/seeds/CHANGELOG.md +18 -0
- surfaceplate/seeds/activity-register.md +47 -0
- surfaceplate/seeds/adoption-decision-record.md +30 -0
- surfaceplate/seeds/data-source-register.md +18 -0
- surfaceplate/seeds/decision-log.md +28 -0
- surfaceplate/seeds/dependency-review-log.md +18 -0
- surfaceplate/seeds/findings-register.md +26 -0
- surfaceplate/seeds/method-registry-readme.md +7 -0
- surfaceplate/seeds/options-log.md +18 -0
- surfaceplate/seeds/output-validation-log.md +18 -0
- surfaceplate/seeds/overrides-readme.md +7 -0
- surfaceplate/seeds/release-checklist.md +19 -0
- surfaceplate/seeds/risk-classification.md +24 -0
- surfaceplate/seeds/run-lineage-readme.md +7 -0
- surfaceplate/seeds/source-of-truth-matrix.yaml +25 -0
- surfaceplate/seeds/test-conventions.md +20 -0
- surfaceplate/standard/.githooks/.gitattributes +1 -0
- surfaceplate/standard/.githooks/pre-commit +35 -0
- surfaceplate/standard/.github/skills/bug-fix/SKILL.md +57 -0
- surfaceplate/standard/.github/skills/change/SKILL.md +62 -0
- surfaceplate/standard/.github/skills/dependency-update/SKILL.md +61 -0
- surfaceplate/standard/.github/skills/fix-ci/SKILL.md +68 -0
- surfaceplate/standard/.github/skills/release/SKILL.md +71 -0
- surfaceplate/standard/.github/skills/review/SKILL.md +56 -0
- surfaceplate/standard/.github/skills/security-review/SKILL.md +61 -0
- surfaceplate/standard/.github/workflows/standards-conformance.yml +38 -0
- surfaceplate/standard/agent-instructions/activity.md +78 -0
- surfaceplate/standard/agent-instructions/ai-workflow.md +109 -0
- surfaceplate/standard/agent-instructions/authority.md +72 -0
- surfaceplate/standard/agent-instructions/provenance.md +93 -0
- surfaceplate/standard/agent-instructions/security.md +78 -0
- surfaceplate/standard/agent-instructions/tests.md +71 -0
- surfaceplate/standard/conformance-block.md +31 -0
- surfaceplate/templates/application-profile.yaml +113 -0
- surfaceplate/templates/decision-record.md +42 -0
- surfaceplate/templates/gate-exception.yaml +23 -0
- surfaceplate/templates/override-record.yaml +25 -0
- surfaceplate/templates/work-packet.md +33 -0
- surfaceplate-0.16.0.dist-info/METADATA +19 -0
- surfaceplate-0.16.0.dist-info/RECORD +96 -0
- surfaceplate-0.16.0.dist-info/WHEEL +4 -0
- surfaceplate-0.16.0.dist-info/entry_points.txt +2 -0
- surfaceplate-0.16.0.dist-info/licenses/LICENSE +201 -0
- surfaceplate-0.16.0.dist-info/licenses/LICENSE-DOCS +121 -0
- surfaceplate-0.16.0.dist-info/licenses/NOTICE +24 -0
|
@@ -0,0 +1,614 @@
|
|
|
1
|
+
"""What this repository already contains, offered as answers instead of a blank box.
|
|
2
|
+
|
|
3
|
+
`DR-38` records why this exists. A maintainer ran the wizard against a real repository and could
|
|
4
|
+
not finish it, and his verdict was about the interaction model rather than any single defect:
|
|
5
|
+
*"free text is confusing and really prone to errors... the wizard should do a discovery of the repo
|
|
6
|
+
first to identify what is the potential candidate for each question."*
|
|
7
|
+
|
|
8
|
+
So every structural question - which file is the precondition, which paths does the gate cover,
|
|
9
|
+
which CI step implements this control - is answered from a list built by reading the repository,
|
|
10
|
+
and free text remains only where the answer is genuinely prose.
|
|
11
|
+
|
|
12
|
+
**This reconciles `example_answers.py`'s scoping rather than reversing it.** That module refused to
|
|
13
|
+
offer example artefact paths, because "a plausible-looking example risks being copied unedited" -
|
|
14
|
+
and it was right: an invented path that looks real is worse than a blank field. A file that
|
|
15
|
+
actually exists in the adopter's own repository is a different kind of thing entirely. The rule
|
|
16
|
+
this module keeps is the one underneath that decision: **never offer something that isn't there.**
|
|
17
|
+
|
|
18
|
+
**Git-aware throughout.** Candidates come from `git ls-files`, so an ignored build artefact, a
|
|
19
|
+
`.venv`, or an untracked scratch file is never offered as though it were part of the repository.
|
|
20
|
+
That is also what stops the lists being unusably long. Where git is unavailable the functions return
|
|
21
|
+
nothing rather than falling back to a filesystem walk: an empty list makes the field behave as it
|
|
22
|
+
did before, while a walk would quietly start offering junk.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import json
|
|
28
|
+
import subprocess
|
|
29
|
+
from dataclasses import dataclass
|
|
30
|
+
from pathlib import Path
|
|
31
|
+
|
|
32
|
+
# Directories whose contents are plausible governance artefacts. Not exhaustive, and not a
|
|
33
|
+
# judgement about the adopter's layout - just the places this framework's own documents, and both
|
|
34
|
+
# worked examples, actually put things.
|
|
35
|
+
_ARTEFACT_DIRS = ("docs", "governance", "activity", "adr", "decisions", ".github")
|
|
36
|
+
_ARTEFACT_SUFFIXES = (".md", ".yaml", ".yml")
|
|
37
|
+
|
|
38
|
+
# Files that are a lock file by name. `dependency_lock` names one of these.
|
|
39
|
+
_LOCK_FILES = (
|
|
40
|
+
"requirements.txt",
|
|
41
|
+
"requirements.lock",
|
|
42
|
+
"poetry.lock",
|
|
43
|
+
"Pipfile.lock",
|
|
44
|
+
"package-lock.json",
|
|
45
|
+
"yarn.lock",
|
|
46
|
+
"pnpm-lock.yaml",
|
|
47
|
+
"Cargo.lock",
|
|
48
|
+
"go.sum",
|
|
49
|
+
"gemfile.lock",
|
|
50
|
+
"pyproject.toml",
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
# Top-level directories that usually hold the code a gate would guard.
|
|
54
|
+
_SOURCE_DIRS = ("src", "lib", "app", "pkg", "internal", "cmd", "services", "packages")
|
|
55
|
+
|
|
56
|
+
_CI_DIRS = (".github/workflows", ".gitlab-ci.d")
|
|
57
|
+
|
|
58
|
+
# Short enough to read. Forty was "too many options to know which one is the right one", and a list
|
|
59
|
+
# nobody can scan is a list nobody uses. **The cap is applied per field, after ranking, and never
|
|
60
|
+
# to the scan** (`F75`): `scan` used to keep 200 and a repository with 300 files under `docs/` lost
|
|
61
|
+
# its real `activity/register.md` before any gate had ranked it. `plan._from_candidates` and
|
|
62
|
+
# `rank_for_gate` cut to `SHOWN` once the field at hand has put its best candidate first.
|
|
63
|
+
SHOWN = 12 # what an adopter is actually offered, once ranked for the field at hand
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _tracked_files(repo: Path) -> list[str]:
|
|
67
|
+
"""Every path git considers part of this repository, or `[]` if git cannot answer.
|
|
68
|
+
|
|
69
|
+
Deliberately not a filesystem walk. A walk offers `.venv/`, `node_modules/` and build output as
|
|
70
|
+
candidate governance artefacts, which is worse than offering nothing.
|
|
71
|
+
|
|
72
|
+
`-z`, so paths come back verbatim: without it git C-quotes a non-ASCII path
|
|
73
|
+
(`"docs/caf\\303\\251.md"`), and the quoted string was offered while the real file was dropped
|
|
74
|
+
(the review's code item 8).
|
|
75
|
+
"""
|
|
76
|
+
try:
|
|
77
|
+
result = subprocess.run(
|
|
78
|
+
["git", "-C", str(repo), "ls-files", "-z", "--cached", "--exclude-standard"],
|
|
79
|
+
capture_output=True,
|
|
80
|
+
timeout=10,
|
|
81
|
+
)
|
|
82
|
+
except (OSError, subprocess.TimeoutExpired):
|
|
83
|
+
return []
|
|
84
|
+
if result.returncode != 0:
|
|
85
|
+
return []
|
|
86
|
+
return [
|
|
87
|
+
chunk.decode("utf-8", errors="surrogateescape")
|
|
88
|
+
for chunk in result.stdout.split(b"\0")
|
|
89
|
+
if chunk
|
|
90
|
+
]
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _dedupe(values: list[str]) -> list[str]:
|
|
94
|
+
"""De-duplicated, order preserved - each caller ranks its own candidates, and sorting here
|
|
95
|
+
would bury the most likely answer somewhere alphabetical. No cap: see `SHOWN`."""
|
|
96
|
+
return list(dict.fromkeys(values))
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
# The one directory only this framework writes. Everything else it installs is read from the
|
|
100
|
+
# install record at scan time, so the exclusion cannot go stale when the payload changes - except
|
|
101
|
+
# the two files the installer CREATES when absent and then manages a block in, which the record
|
|
102
|
+
# does not list because they are the adopter's to keep. On a repository holding nothing else they
|
|
103
|
+
# are framework output all the same, and offering them proposed `**` as a discovered pathspec.
|
|
104
|
+
_FRAMEWORK_DIR = ".standards/"
|
|
105
|
+
_BLOCK_HOSTS = (".github/copilot-instructions.md", "AGENTS.md")
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def framework_paths(repo: Path) -> tuple[set[str], set[str]]:
|
|
109
|
+
"""`(files, ci_steps)` this framework put into the repository, read from the install record.
|
|
110
|
+
|
|
111
|
+
`F61`: discovery proposed `.github/instructions/authority.instructions.md` as the adopter's
|
|
112
|
+
authority map, the installed workflow's own "Check conformance to Surfaceplate" as their
|
|
113
|
+
contract test, and told a bare repository it appeared to have a CI workflow sixty seconds after
|
|
114
|
+
the installer wrote one. A gate whose precondition is a file the framework installed is
|
|
115
|
+
satisfied the moment the framework is installed, and the checker passes it. So every path in
|
|
116
|
+
the install record's file list, the profile the wizard is about to write, and every step of the
|
|
117
|
+
installed workflow are excluded from every candidate list - not ranked last, excluded.
|
|
118
|
+
"""
|
|
119
|
+
record_path = repo / ".standards" / "INSTALL.json"
|
|
120
|
+
try:
|
|
121
|
+
record = json.loads(record_path.read_text(encoding="utf-8"))
|
|
122
|
+
except (OSError, ValueError):
|
|
123
|
+
return set(), set()
|
|
124
|
+
files = set((record.get("files") or {}).keys()) if isinstance(record.get("files"), dict) else set()
|
|
125
|
+
profile = record.get("profile_path")
|
|
126
|
+
if isinstance(profile, str):
|
|
127
|
+
files.add(profile)
|
|
128
|
+
steps: set[str] = set()
|
|
129
|
+
for rel in files:
|
|
130
|
+
if rel.startswith(".github/workflows/") and rel.endswith((".yml", ".yaml")):
|
|
131
|
+
steps |= set(_steps_in(repo / rel))
|
|
132
|
+
return files, steps
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _steps_in(path: Path) -> list[str]:
|
|
136
|
+
from surfaceplate.check_conformance import load_yaml
|
|
137
|
+
|
|
138
|
+
if not path.is_file():
|
|
139
|
+
return []
|
|
140
|
+
document, _ = load_yaml(path)
|
|
141
|
+
if not isinstance(document, dict) or not isinstance(document.get("jobs"), dict):
|
|
142
|
+
return []
|
|
143
|
+
names: list[str] = []
|
|
144
|
+
for job in document["jobs"].values():
|
|
145
|
+
if not isinstance(job, dict):
|
|
146
|
+
continue
|
|
147
|
+
for step in job.get("steps") or []:
|
|
148
|
+
if isinstance(step, dict) and isinstance(step.get("name"), str):
|
|
149
|
+
names.append(step["name"])
|
|
150
|
+
return names
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _adopters_own(repo: Path) -> list[str]:
|
|
154
|
+
"""The tracked files that are the adopter's, not this framework's."""
|
|
155
|
+
excluded, _ = framework_paths(repo)
|
|
156
|
+
if excluded:
|
|
157
|
+
excluded |= set(_BLOCK_HOSTS)
|
|
158
|
+
return [
|
|
159
|
+
p for p in _tracked_files(repo)
|
|
160
|
+
if p not in excluded and not p.startswith(_FRAMEWORK_DIR)
|
|
161
|
+
]
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
# Where a governance artefact is most likely to be, most likely first. Ranking matters more than
|
|
165
|
+
# it looks: people pick the first plausible line they see, so an alphabetical list that opens with
|
|
166
|
+
# `.github/` because a dot sorts before a letter is actively misleading.
|
|
167
|
+
_ARTEFACT_RANK = ("docs/", "governance/", "activity/", "decisions/", "adr/")
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _adopter_first(paths: list[str]) -> list[str]:
|
|
171
|
+
"""Ranked by how likely the adopter means it: their own governance directories first, then
|
|
172
|
+
their root-level documents, then anything else. What this framework installed is not ranked
|
|
173
|
+
last any more - it is not offered at all (`framework_paths`).
|
|
174
|
+
|
|
175
|
+
Matches a bare directory name as well as a prefix, because the register-directory candidates are
|
|
176
|
+
directories (`governance`) where the artefact candidates are files (`governance/x.yaml`).
|
|
177
|
+
"""
|
|
178
|
+
|
|
179
|
+
def rank(path: str) -> tuple[int, str]:
|
|
180
|
+
for index, prefix in enumerate(_ARTEFACT_RANK):
|
|
181
|
+
if path == prefix.rstrip("/") or path.startswith(prefix):
|
|
182
|
+
return (index, path)
|
|
183
|
+
if "/" not in path:
|
|
184
|
+
return (len(_ARTEFACT_RANK), path)
|
|
185
|
+
return (len(_ARTEFACT_RANK) + 1, path)
|
|
186
|
+
|
|
187
|
+
return sorted(paths, key=rank)
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
# Words that suggest a file is the artefact a particular gate is about. Ranking only; nothing here
|
|
191
|
+
# decides anything, and every candidate stays in the list either way.
|
|
192
|
+
GATE_KEYWORDS: dict[str, tuple[str, ...]] = {
|
|
193
|
+
"work_registration": ("register", "activity", "backlog"),
|
|
194
|
+
"register_currency": ("register", "activity"),
|
|
195
|
+
"work_contract": ("packet", "contract", "brief"),
|
|
196
|
+
"risk_classification": ("risk", "classification"),
|
|
197
|
+
"decision_before_implementation": ("decision", "adr"),
|
|
198
|
+
"records_before_release": ("release", "checklist"),
|
|
199
|
+
"change_record_before_completion": ("changelog", "change"),
|
|
200
|
+
# `F84`: "inventory" matched a work inventory and proposed it as the authority map. The seed's
|
|
201
|
+
# own path still matches on `source_of_truth`.
|
|
202
|
+
"authority_map": ("authority", "source_of_truth"),
|
|
203
|
+
"authority_same_change": ("authority",),
|
|
204
|
+
"test_convention": ("test", "convention"),
|
|
205
|
+
"regression_before_merge": ("regression", "test"),
|
|
206
|
+
"equivalence_evidence": ("equivalence", "protocol", "test"),
|
|
207
|
+
"data_source_lifecycle": ("data", "source"),
|
|
208
|
+
"output_validation_before_external_use": ("output", "validation"),
|
|
209
|
+
"dependency_output_delta": ("dependency", "review"),
|
|
210
|
+
"component_library": ("component", "design"),
|
|
211
|
+
"design_authority": ("design", "policy"),
|
|
212
|
+
"options_before_build": ("option", "design", "decision"),
|
|
213
|
+
"prerequisite_state_ui": ("design", "state", "screen"),
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
# `F94`: a path with one of these components is not a live artefact whatever its name says. Still
|
|
218
|
+
# offered - the adopter chooses from everything found - but never proposed, and ranked last.
|
|
219
|
+
_ARCHIVE_COMPONENTS = ("archive", "archived", "attic", "deprecated")
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def is_archived(path: str) -> bool:
|
|
223
|
+
return any(part.lower() in _ARCHIVE_COMPONENTS for part in path.split("/")[:-1])
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
# `F93`: the words a record directory's name must carry to be proposed for a pattern-C control.
|
|
227
|
+
CONTROL_DIR_WORDS: dict[str, tuple[str, ...]] = {
|
|
228
|
+
"method_registry": ("registry", "method"),
|
|
229
|
+
"overrides": ("override",),
|
|
230
|
+
"run_lineage": ("lineage", "run"),
|
|
231
|
+
"provenance": ("provenance", "lineage", "run"),
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def register_dirs_that_fit(repo: Path, directories, control_id: str) -> list[str]:
|
|
236
|
+
"""The register directories a pattern-C control may be PROPOSED: named for the control, and
|
|
237
|
+
holding no YAML record the control's schema rejects (`SP056`). `F93`: four controls were
|
|
238
|
+
proposed a directory of account configuration because it held YAML."""
|
|
239
|
+
from surfaceplate.check_conformance import PATTERN_C_CONTROLS, load_yaml
|
|
240
|
+
|
|
241
|
+
words = CONTROL_DIR_WORDS.get(control_id, ())
|
|
242
|
+
schema_path = repo / ".standards" / "schemas" / PATTERN_C_CONTROLS.get(control_id, "")
|
|
243
|
+
if not words or not schema_path.is_file():
|
|
244
|
+
return []
|
|
245
|
+
try:
|
|
246
|
+
import jsonschema
|
|
247
|
+
import yaml
|
|
248
|
+
|
|
249
|
+
schema = yaml.safe_load(schema_path.read_text(encoding="utf-8"))
|
|
250
|
+
validator = jsonschema.Draft202012Validator(schema, format_checker=jsonschema.FormatChecker())
|
|
251
|
+
except Exception: # noqa: BLE001 - no schema, no proposal
|
|
252
|
+
return []
|
|
253
|
+
out: list[str] = []
|
|
254
|
+
for directory in directories:
|
|
255
|
+
name = directory.rsplit("/", 1)[-1].lower()
|
|
256
|
+
if not any(w in name for w in words) or is_archived(directory + "/x"):
|
|
257
|
+
continue
|
|
258
|
+
ok = True
|
|
259
|
+
for record in sorted((repo / directory).glob("*.y*ml")):
|
|
260
|
+
document, _ = load_yaml(record)
|
|
261
|
+
if document is None or next(validator.iter_errors(document), None) is not None:
|
|
262
|
+
ok = False
|
|
263
|
+
break
|
|
264
|
+
if ok:
|
|
265
|
+
out.append(directory)
|
|
266
|
+
return out
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def rank_for_gate(
|
|
270
|
+
candidates: tuple[str, ...] | list[str], gate_id: str, limit: int = SHOWN
|
|
271
|
+
) -> list[str]:
|
|
272
|
+
"""The same candidates, most plausible for THIS gate first.
|
|
273
|
+
|
|
274
|
+
A precondition list is only useful if the right answer is near the top; forty alphabetical
|
|
275
|
+
paths is a haystack. Keyword matching is a hint, not a decision - nothing is removed, and the
|
|
276
|
+
adopter still chooses.
|
|
277
|
+
"""
|
|
278
|
+
hit = matched_for_gate(candidates, gate_id, limit=None)
|
|
279
|
+
if not GATE_KEYWORDS.get(gate_id, ()):
|
|
280
|
+
return list(candidates)[:limit]
|
|
281
|
+
rest = [c for c in candidates if c not in hit]
|
|
282
|
+
# Cut AFTER ranking, never before. `F75`: this comment was true here while `scan` capped the
|
|
283
|
+
# list to 200 one level up, so the register was thrown away before this line ever ran. The
|
|
284
|
+
# scan now keeps everything and this is the only cut.
|
|
285
|
+
return (hit + rest)[:limit]
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
def matched_for_gate(
|
|
289
|
+
candidates: tuple[str, ...] | list[str], gate_id: str, limit: int | None = SHOWN
|
|
290
|
+
) -> list[str]:
|
|
291
|
+
"""Only the candidates that actually matched a keyword for THIS gate, best first.
|
|
292
|
+
|
|
293
|
+
`rank_for_gate` orders; this one **discriminates**, and the difference is `F40`. Ranking returns
|
|
294
|
+
every candidate so the dropdown can offer everything - that is `DR-38`'s rule, and the adopter
|
|
295
|
+
still chooses. But `defaults.py` took the top-ranked candidate as a *proposal*, and in a
|
|
296
|
+
repository holding one unrelated file the top of the ranking is simply that file. The wizard
|
|
297
|
+
proposed `README.md` as the precondition for `work_registration`, producing a gate that satisfies
|
|
298
|
+
`SP032` - exists, non-empty, no placeholder - while guarding nothing at all.
|
|
299
|
+
|
|
300
|
+
So a proposal now comes only from this list, and an empty list means no proposal and a question
|
|
301
|
+
asked. That is `DR-40`'s own standard applied to a case it missed: *a field with no honest source
|
|
302
|
+
is left unanswered and still asked.* An unmatched file is not an honest source; it is the only
|
|
303
|
+
file.
|
|
304
|
+
"""
|
|
305
|
+
words = GATE_KEYWORDS.get(gate_id, ())
|
|
306
|
+
if not words:
|
|
307
|
+
return []
|
|
308
|
+
|
|
309
|
+
def score(path: str) -> tuple[int, int, int, str]:
|
|
310
|
+
low = path.lower()
|
|
311
|
+
matches = sum(1 for w in words if w in low)
|
|
312
|
+
# Archived paths last (`F94`); then more keywords first, then shallower paths:
|
|
313
|
+
# `activity/register.md` matches both "activity" and "register" and sits above
|
|
314
|
+
# `activity/ACT-001.md`, which matches one.
|
|
315
|
+
return (1 if is_archived(path) else 0, -matches, path.count("/"), path)
|
|
316
|
+
|
|
317
|
+
hit = sorted((c for c in candidates if any(w in c.lower() for w in words)), key=score)
|
|
318
|
+
return hit if limit is None else hit[:limit]
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def candidate_artefacts(repo: Path) -> list[str]:
|
|
322
|
+
"""Files that could serve as an artefact - every tracked Markdown or YAML file of the adopter's
|
|
323
|
+
own, ranked; the field at hand cuts the list after ranking it for its gate or control.
|
|
324
|
+
|
|
325
|
+
`F88` / `DR-54` (2): this offered only files under a fixed list of directories, so a findings
|
|
326
|
+
register at `org/FINDINGS.md` was never offered. Ranking, not the list, now does the work:
|
|
327
|
+
`_adopter_first` puts the governance directories first, root documents next, the rest after.
|
|
328
|
+
"""
|
|
329
|
+
out = [
|
|
330
|
+
p for p in _adopters_own(repo)
|
|
331
|
+
if p.endswith(_ARTEFACT_SUFFIXES) and not any(p.startswith(d + "/") for d in _CI_DIRS)
|
|
332
|
+
] # a workflow is CI, offered as such elsewhere, never as a governance artefact
|
|
333
|
+
return _dedupe(_adopter_first(out))
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
def free_seeds(repo: Path) -> tuple[dict[str, str], dict[str, str]]:
|
|
337
|
+
"""`(by gate, by control)`: the seeds whose path is free in this repository, so a field can open
|
|
338
|
+
with "create it" (`DR-54` (1)). Read here, once, so `plan.py` needs no repository."""
|
|
339
|
+
from surfaceplate.adopt import scaffold
|
|
340
|
+
|
|
341
|
+
gates = {g: path for g, (path, _s, _w) in scaffold.SEEDABLE.items() if not scaffold._occupied(repo / path)}
|
|
342
|
+
controls = {c: path for c, (path, _s, _w) in scaffold.SEEDABLE_CONTROLS.items() if not scaffold._occupied(repo / path)}
|
|
343
|
+
return gates, controls
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def candidate_register_dirs(repo: Path) -> list[str]:
|
|
347
|
+
"""Directories holding YAML records - what a pattern-C control names (`DR-26`).
|
|
348
|
+
|
|
349
|
+
An empty register is a legitimate answer to those controls, so a directory qualifies by holding
|
|
350
|
+
records OR by sitting where records live; the checker's own rule is that the directory exists
|
|
351
|
+
and contains nothing invalid, never that it contains anything at all.
|
|
352
|
+
"""
|
|
353
|
+
dirs: set[str] = set()
|
|
354
|
+
control_words = tuple(w for words in CONTROL_DIR_WORDS.values() for w in words)
|
|
355
|
+
for path in _adopters_own(repo):
|
|
356
|
+
parent = "/".join(path.split("/")[:-1])
|
|
357
|
+
if not parent or parent.startswith(".github"):
|
|
358
|
+
continue
|
|
359
|
+
name = parent.rsplit("/", 1)[-1].lower()
|
|
360
|
+
# A directory holding records, or one named for a record control that holds none yet -
|
|
361
|
+
# which is what a seeded directory is (`F93`); it would otherwise vanish from the offer
|
|
362
|
+
# the moment it was created.
|
|
363
|
+
if path.endswith((".yaml", ".yml")) or any(w in name for w in control_words):
|
|
364
|
+
dirs.add(parent)
|
|
365
|
+
return _dedupe(_adopter_first(sorted(dirs)))
|
|
366
|
+
|
|
367
|
+
|
|
368
|
+
def candidate_lock_files(repo: Path) -> list[str]:
|
|
369
|
+
"""Dependency lock files, for `dependency_lock`'s implementation reference."""
|
|
370
|
+
tracked = _adopters_own(repo)
|
|
371
|
+
names = {name.lower() for name in _LOCK_FILES}
|
|
372
|
+
return _dedupe(_adopter_first([p for p in tracked if p.split("/")[-1].lower() in names]))
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
def candidate_paths(repo: Path) -> list[str]:
|
|
376
|
+
"""Git pathspecs a gate could cover, as `src/**`-style globs.
|
|
377
|
+
|
|
378
|
+
Offers the top-level directories that actually contain tracked code, plus `**` for a gate that
|
|
379
|
+
covers the whole repository - which is a real answer, not a cop-out, for a small one.
|
|
380
|
+
"""
|
|
381
|
+
own = _adopters_own(repo)
|
|
382
|
+
if not own:
|
|
383
|
+
# Nothing git could read, or nothing but this framework's own files: no pathspec is an
|
|
384
|
+
# honest offer, and `Discovered.is_empty()` can then be true (the review's code item 9).
|
|
385
|
+
return []
|
|
386
|
+
tops: set[str] = set()
|
|
387
|
+
for path in own:
|
|
388
|
+
if "/" not in path:
|
|
389
|
+
continue
|
|
390
|
+
head = path.split("/")[0]
|
|
391
|
+
if head.startswith("."):
|
|
392
|
+
continue
|
|
393
|
+
tops.add(head)
|
|
394
|
+
# Recognised source directories first, then everything else, then the whole repository - which
|
|
395
|
+
# is a real answer for a small one, but rarely the one someone means, so it goes last.
|
|
396
|
+
known = [f"{d}/**" for d in sorted(tops) if d in _SOURCE_DIRS]
|
|
397
|
+
others = [f"{d}/**" for d in sorted(tops) if d not in _SOURCE_DIRS]
|
|
398
|
+
return _dedupe(known + others + ["**"])
|
|
399
|
+
|
|
400
|
+
|
|
401
|
+
def candidate_ci_steps(repo: Path) -> list[str]:
|
|
402
|
+
"""Every named step in every workflow, for a pattern-B control's `implementation_reference`.
|
|
403
|
+
|
|
404
|
+
`check_conformance.find_workflow_step` looks a step up BY name and cannot enumerate, so this
|
|
405
|
+
walks the same `jobs -> steps -> name` nesting it does. That duplication is deliberate and
|
|
406
|
+
narrow: the checker's function answers "does this named step exist and can it fail", which is a
|
|
407
|
+
different question from "what could the adopter pick", and merging them would make the checker
|
|
408
|
+
depend on the wizard.
|
|
409
|
+
"""
|
|
410
|
+
installed_files, installed_steps = framework_paths(repo)
|
|
411
|
+
names: list[str] = []
|
|
412
|
+
for directory in _CI_DIRS:
|
|
413
|
+
d = repo / directory
|
|
414
|
+
if not d.is_dir():
|
|
415
|
+
continue
|
|
416
|
+
for path in sorted(d.glob("*.y*ml")):
|
|
417
|
+
if path.relative_to(repo).as_posix() in installed_files:
|
|
418
|
+
continue # the framework's own workflow is not the adopter's CI (`F61`)
|
|
419
|
+
names.extend(step for step in _steps_in(path) if step not in installed_steps)
|
|
420
|
+
return _dedupe(sorted(dict.fromkeys(names)))
|
|
421
|
+
|
|
422
|
+
|
|
423
|
+
# ---------------------------------------------------------------------------------------------
|
|
424
|
+
# The checker's own rules, applied before proposing (`DR-51` (5)), and what a choice is
|
|
425
|
+
# ---------------------------------------------------------------------------------------------
|
|
426
|
+
|
|
427
|
+
|
|
428
|
+
def content_problem(target: Path) -> str:
|
|
429
|
+
"""Why the checker would reject this file as an artefact, in a few words, or `""`. The same
|
|
430
|
+
two rules `SP032` and `SP051` apply after the path checks: non-empty, no placeholder token.
|
|
431
|
+
A directory has no content rule (a register may be empty)."""
|
|
432
|
+
from surfaceplate import rules
|
|
433
|
+
|
|
434
|
+
if not target.is_file():
|
|
435
|
+
return ""
|
|
436
|
+
try:
|
|
437
|
+
text = target.read_text(encoding="utf-8", errors="replace")
|
|
438
|
+
except OSError:
|
|
439
|
+
return "cannot be read"
|
|
440
|
+
if not text.strip():
|
|
441
|
+
return "is empty"
|
|
442
|
+
if rules.PLACEHOLDER_PATTERN.search(text):
|
|
443
|
+
return "still contains a template placeholder (TBD, TODO or replace-me)"
|
|
444
|
+
return ""
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
def scanner_step(target: Path, scanner: str) -> tuple[str, str]:
|
|
448
|
+
"""`("runs", step name)` where a workflow step runs the scanner; `("mentions", "")` for a
|
|
449
|
+
non-workflow file that mentions it (the checker inspects those no further); `("comment", "")`
|
|
450
|
+
for a workflow that mentions it outside any step; `("absent", "")` otherwise. `SP046`'s rule."""
|
|
451
|
+
from surfaceplate.check_conformance import load_yaml, step_mentions
|
|
452
|
+
|
|
453
|
+
try:
|
|
454
|
+
text = target.read_text(encoding="utf-8", errors="replace")
|
|
455
|
+
except OSError:
|
|
456
|
+
return "absent", ""
|
|
457
|
+
if scanner.lower() not in text.lower():
|
|
458
|
+
return "absent", ""
|
|
459
|
+
document, _ = load_yaml(target)
|
|
460
|
+
if not isinstance(document, dict) or not isinstance(document.get("jobs"), dict):
|
|
461
|
+
return "mentions", ""
|
|
462
|
+
for job in document["jobs"].values():
|
|
463
|
+
if not isinstance(job, dict):
|
|
464
|
+
continue
|
|
465
|
+
for step in job.get("steps") or []:
|
|
466
|
+
if isinstance(step, dict) and step_mentions(step, scanner):
|
|
467
|
+
return "runs", str(step.get("name") or step.get("uses") or step.get("id") or "the scan step")
|
|
468
|
+
return "comment", ""
|
|
469
|
+
|
|
470
|
+
|
|
471
|
+
def scanner_workflows(repo: Path, scanner: str) -> list[str]:
|
|
472
|
+
"""The adopter's workflows where a step runs the named scanner - what `scanner.wired_in`
|
|
473
|
+
offers and proposes. `F83`: the first workflow found was proposed, and the checker then
|
|
474
|
+
reported it never mentions the scanner."""
|
|
475
|
+
installed_files, _ = framework_paths(repo)
|
|
476
|
+
own = set(_adopters_own(repo))
|
|
477
|
+
out: list[str] = []
|
|
478
|
+
for directory in _CI_DIRS:
|
|
479
|
+
d = repo / directory
|
|
480
|
+
if not d.is_dir():
|
|
481
|
+
continue
|
|
482
|
+
for path in sorted(d.glob("*.y*ml")):
|
|
483
|
+
rel = path.relative_to(repo).as_posix()
|
|
484
|
+
if rel in installed_files or rel not in own:
|
|
485
|
+
continue
|
|
486
|
+
if scanner_step(path, scanner)[0] == "runs":
|
|
487
|
+
out.append(rel)
|
|
488
|
+
return _dedupe(out)
|
|
489
|
+
|
|
490
|
+
|
|
491
|
+
def _title_of(target: Path) -> str:
|
|
492
|
+
"""The first heading of a Markdown file, the first comment or key of a YAML file: what a
|
|
493
|
+
reader would take the file to be about, in its own words."""
|
|
494
|
+
try:
|
|
495
|
+
lines = target.read_text(encoding="utf-8", errors="replace").splitlines()
|
|
496
|
+
except OSError:
|
|
497
|
+
return ""
|
|
498
|
+
for line in lines:
|
|
499
|
+
stripped = line.strip()
|
|
500
|
+
if not stripped:
|
|
501
|
+
continue
|
|
502
|
+
if stripped.startswith("#"):
|
|
503
|
+
return stripped.lstrip("#").strip()[:60]
|
|
504
|
+
return stripped[:60]
|
|
505
|
+
return ""
|
|
506
|
+
|
|
507
|
+
|
|
508
|
+
def describe(
|
|
509
|
+
repo: Path,
|
|
510
|
+
path: str,
|
|
511
|
+
*,
|
|
512
|
+
gate_id: str = "",
|
|
513
|
+
scanner: str = "",
|
|
514
|
+
found: "Discovered | None" = None,
|
|
515
|
+
) -> str:
|
|
516
|
+
"""One line about a chosen path, for the help beside the field (`F80`, `DR-51` (4)): what
|
|
517
|
+
discovery saw in it, whether it matched the gate's words, and whether the checker's rules
|
|
518
|
+
would reject it. States what was read; decides nothing."""
|
|
519
|
+
target = repo / path
|
|
520
|
+
if not target.exists():
|
|
521
|
+
return f"{path}: nothing exists at that path in this repository."
|
|
522
|
+
if target.is_dir():
|
|
523
|
+
records = len([p for p in target.glob("*.y*ml")])
|
|
524
|
+
return f"{path}: a directory holding {records} YAML record(s); an empty register is a valid start."
|
|
525
|
+
parts: list[str] = []
|
|
526
|
+
title = _title_of(target)
|
|
527
|
+
parts.append(f'"{title}"' if title else "no heading or first line")
|
|
528
|
+
try:
|
|
529
|
+
count = len(target.read_text(encoding="utf-8", errors="replace").splitlines())
|
|
530
|
+
parts[-1] += f", {count} line(s)"
|
|
531
|
+
except OSError:
|
|
532
|
+
pass
|
|
533
|
+
if gate_id:
|
|
534
|
+
words = [w for w in GATE_KEYWORDS.get(gate_id, ()) if w in path.lower()]
|
|
535
|
+
parts.append(
|
|
536
|
+
f"matched this gate's words: {', '.join(words)}" if words
|
|
537
|
+
else f"did not match this gate's words ({', '.join(GATE_KEYWORDS.get(gate_id, ())) or 'none'}); it is simply a file found here"
|
|
538
|
+
)
|
|
539
|
+
if scanner:
|
|
540
|
+
state, step = scanner_step(target, scanner)
|
|
541
|
+
parts.append(
|
|
542
|
+
{"runs": f"step '{step}' runs {scanner}",
|
|
543
|
+
"mentions": f"mentions {scanner}; not a workflow, so the checker inspects it no further",
|
|
544
|
+
"comment": f"mentions {scanner} but no step runs it; the checker would reject it (SP046)",
|
|
545
|
+
"absent": f"never mentions {scanner}; the checker would reject it (SP046)"}[state]
|
|
546
|
+
)
|
|
547
|
+
problem = (found.rejected.get(path) if found is not None else None) or content_problem(target)
|
|
548
|
+
if problem:
|
|
549
|
+
parts.append(f"the checker would reject it: it {problem} (SP032)")
|
|
550
|
+
return f"{path}: " + "; ".join(parts) + "."
|
|
551
|
+
|
|
552
|
+
|
|
553
|
+
@dataclass(frozen=True)
|
|
554
|
+
class Discovered:
|
|
555
|
+
"""One scan of the repository, passed to `plan.py` so every section can offer real answers."""
|
|
556
|
+
|
|
557
|
+
artefacts: tuple[str, ...] = ()
|
|
558
|
+
register_dirs: tuple[str, ...] = ()
|
|
559
|
+
lock_files: tuple[str, ...] = ()
|
|
560
|
+
paths: tuple[str, ...] = ()
|
|
561
|
+
ci_steps: tuple[str, ...] = ()
|
|
562
|
+
# `DR-51` (5): workflows where a step runs the scanner the examples name, and artefacts the
|
|
563
|
+
# checker's content rules would reject, with the reason. Rejected files stay in `artefacts`
|
|
564
|
+
# (the adopter chooses from everything found) and are never proposed.
|
|
565
|
+
scanner_workflows: tuple[str, ...] = ()
|
|
566
|
+
rejected: dict[str, str] = None # type: ignore[assignment]
|
|
567
|
+
# `DR-54` (1): seeds whose path is free here, by gate id and by control id.
|
|
568
|
+
free_seeds: dict[str, str] = None # type: ignore[assignment]
|
|
569
|
+
free_control_seeds: dict[str, str] = None # type: ignore[assignment]
|
|
570
|
+
# `F93`: per pattern-C control, the register directories it may be proposed.
|
|
571
|
+
register_fit: dict[str, tuple[str, ...]] = None # type: ignore[assignment]
|
|
572
|
+
|
|
573
|
+
def __post_init__(self) -> None:
|
|
574
|
+
for name in ("rejected", "free_seeds", "free_control_seeds", "register_fit"):
|
|
575
|
+
if getattr(self, name) is None:
|
|
576
|
+
object.__setattr__(self, name, {})
|
|
577
|
+
|
|
578
|
+
def is_empty(self) -> bool:
|
|
579
|
+
"""True when nothing of the adopter's could be read - a tree git cannot answer for, or one
|
|
580
|
+
holding nothing but this framework's own files. Every `_from_candidates` field then
|
|
581
|
+
degrades to the plain text box it always was; `plan.py` is the caller."""
|
|
582
|
+
return not (
|
|
583
|
+
self.artefacts or self.register_dirs or self.lock_files or self.paths or self.ci_steps
|
|
584
|
+
)
|
|
585
|
+
|
|
586
|
+
|
|
587
|
+
# The scanner the examples name; `scan` looks for its workflow, and the field's validator
|
|
588
|
+
# re-checks against whatever name the profile ends up carrying.
|
|
589
|
+
DEFAULT_SCANNER = "gitleaks"
|
|
590
|
+
|
|
591
|
+
|
|
592
|
+
def scan(repo: Path, scanner: str = DEFAULT_SCANNER) -> Discovered:
|
|
593
|
+
"""Read the repository once. Cheap enough to do at startup; nothing here writes."""
|
|
594
|
+
artefacts = tuple(candidate_artefacts(repo))
|
|
595
|
+
seeds_by_gate, seeds_by_control = free_seeds(repo)
|
|
596
|
+
register_dirs = tuple(candidate_register_dirs(repo))
|
|
597
|
+
register_fit = {c: tuple(register_dirs_that_fit(repo, register_dirs, c)) for c in CONTROL_DIR_WORDS}
|
|
598
|
+
rejected = {}
|
|
599
|
+
for rel in artefacts:
|
|
600
|
+
problem = content_problem(repo / rel)
|
|
601
|
+
if problem:
|
|
602
|
+
rejected[rel] = problem
|
|
603
|
+
return Discovered(
|
|
604
|
+
artefacts=artefacts,
|
|
605
|
+
register_dirs=register_dirs,
|
|
606
|
+
lock_files=tuple(candidate_lock_files(repo)),
|
|
607
|
+
paths=tuple(candidate_paths(repo)),
|
|
608
|
+
ci_steps=tuple(candidate_ci_steps(repo)),
|
|
609
|
+
scanner_workflows=tuple(scanner_workflows(repo, scanner)),
|
|
610
|
+
rejected=rejected,
|
|
611
|
+
free_seeds=seeds_by_gate,
|
|
612
|
+
free_control_seeds=seeds_by_control,
|
|
613
|
+
register_fit=register_fit,
|
|
614
|
+
)
|