ww-agentic-workflows 1.0.0.dev3__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ww/__init__.py +18 -0
- ww/_bundled_extensions/ww/git/extension.py +1728 -0
- ww/action_execution.py +887 -0
- ww/actions/__init__.py +94 -0
- ww/actions/command.py +444 -0
- ww/actions/contracts.py +699 -0
- ww/actions/extension.py +197 -0
- ww/actions/mcp.py +84 -0
- ww/actions/prompt.py +74 -0
- ww/actions/skill.py +62 -0
- ww/actions/slash_command.py +63 -0
- ww/agents.py +151 -0
- ww/amendments.py +54 -0
- ww/artifacts.py +93 -0
- ww/assessments.py +181 -0
- ww/assets/__init__.py +2 -0
- ww/assets/agent_instructions.md +49 -0
- ww/assets/docs/examples.md +879 -0
- ww/assets/docs/features.md +4639 -0
- ww/assets/docs/specification.md +1876 -0
- ww/assets/noww_skill.md +11 -0
- ww/assets/workflows/catchall.yaml +26 -0
- ww/assets/workflows/onboarding.yaml +586 -0
- ww/assets/workflows/scriptize.yaml +130 -0
- ww/assets/ww-automate_skill.md +23 -0
- ww/assets/ww-deduce-feedback_skill.md +38 -0
- ww/assets/ww-feedback-rules_skill.md +48 -0
- ww/assets/ww-learn-project_skill.md +22 -0
- ww/assets/ww-refresh_skill.md +26 -0
- ww/assets/ww-rule_skill.md +83 -0
- ww/assets/ww-rules-from-artifacts_skill.md +22 -0
- ww/assets/ww-scriptize_skill.md +33 -0
- ww/assets/ww-setup_skill.md +94 -0
- ww/assets/ww-solve_skill.md +23 -0
- ww/assets/ww-suggest_skill.md +32 -0
- ww/assets/ww-wizard_skill.md +105 -0
- ww/assets/ww_skill.md +59 -0
- ww/assignments.py +283 -0
- ww/bootstrap.py +405 -0
- ww/builtin_workflows.py +215 -0
- ww/changes.py +225 -0
- ww/child_coordination.py +482 -0
- ww/children.py +106 -0
- ww/claude_permissions.py +115 -0
- ww/cli/__init__.py +7 -0
- ww/cli/__main__.py +6 -0
- ww/cli/audit.py +129 -0
- ww/cli/catalogs.py +131 -0
- ww/cli/discover.py +607 -0
- ww/cli/initialization.py +898 -0
- ww/cli/lookup.py +287 -0
- ww/cli/main.py +1768 -0
- ww/cli/parser.py +1200 -0
- ww/cli/prompts.py +217 -0
- ww/cli/updates.py +117 -0
- ww/completion_artifacts.py +156 -0
- ww/completion_inputs.py +39 -0
- ww/config/__init__.py +582 -0
- ww/config/actions.py +591 -0
- ww/config/composition.py +571 -0
- ww/config/rules.py +511 -0
- ww/config/steps.py +1220 -0
- ww/config/values.py +223 -0
- ww/config_files.py +191 -0
- ww/config_writes.py +264 -0
- ww/contracts.py +155 -0
- ww/control.py +41 -0
- ww/defaults.py +130 -0
- ww/design_docs.py +32 -0
- ww/discovery.py +104 -0
- ww/documents.py +217 -0
- ww/errors.py +18 -0
- ww/executable.py +43 -0
- ww/execution_models/__init__.py +64 -0
- ww/execution_models/construction.py +148 -0
- ww/execution_models/decoding.py +38 -0
- ww/execution_models/plan_codec.py +565 -0
- ww/execution_models/records.py +1206 -0
- ww/execution_models/runs.py +266 -0
- ww/extensions/__init__.py +40 -0
- ww/extensions/api.py +559 -0
- ww/extensions/registry.py +864 -0
- ww/extensions/store.py +78 -0
- ww/feedback.py +342 -0
- ww/handler_repairs.py +57 -0
- ww/hooks/__init__.py +40 -0
- ww/hooks/agents.py +380 -0
- ww/hooks/install.py +168 -0
- ww/hooks/notices.py +206 -0
- ww/hooks/records.py +209 -0
- ww/hooks/runtime.py +266 -0
- ww/hooks/transcripts.py +183 -0
- ww/inspect.py +896 -0
- ww/instructions/__init__.py +17 -0
- ww/instructions/builder.py +1682 -0
- ww/instructions/commands.py +335 -0
- ww/instructions/handoff.py +149 -0
- ww/instructions/models.py +686 -0
- ww/instructions/policy.py +219 -0
- ww/instructions/text.py +168 -0
- ww/interactions.py +187 -0
- ww/interpolation.py +37 -0
- ww/item_passes.py +167 -0
- ww/items.py +99 -0
- ww/locking.py +207 -0
- ww/metadata_publication.py +230 -0
- ww/onboarding.py +229 -0
- ww/open_work.py +236 -0
- ww/operations.py +193 -0
- ww/operator_ui/__init__.py +16 -0
- ww/operator_ui/page.html +351 -0
- ww/operator_ui/server.py +215 -0
- ww/operator_ui/session.py +389 -0
- ww/operator_ui/sheet.py +104 -0
- ww/operator_ui/view.py +109 -0
- ww/output.py +339 -0
- ww/output_adapters/__init__.py +12 -0
- ww/output_adapters/base.py +25 -0
- ww/output_adapters/json_adapter.py +37 -0
- ww/output_adapters/markdown.py +2293 -0
- ww/output_adapters/rule_pages.py +337 -0
- ww/output_adapters/terminal.py +21 -0
- ww/package_updates.py +167 -0
- ww/plan/__init__.py +38 -0
- ww/plan/actions.py +207 -0
- ww/plan/compiler.py +1492 -0
- ww/plan/constructs.py +456 -0
- ww/plan/models.py +665 -0
- ww/project_config.py +752 -0
- ww/recovery.py +401 -0
- ww/replanning.py +367 -0
- ww/results.py +77 -0
- ww/rule_checks.py +230 -0
- ww/rule_conversion.py +331 -0
- ww/rule_disputes.py +148 -0
- ww/rule_store.py +456 -0
- ww/rule_verification.py +714 -0
- ww/rule_views.py +447 -0
- ww/rule_writes.py +920 -0
- ww/run_coordination.py +158 -0
- ww/runtimes.py +105 -0
- ww/service.py +4405 -0
- ww/setup_apply.py +428 -0
- ww/step_values.py +20 -0
- ww/storage.py +447 -0
- ww/storage_adapters/__init__.py +36 -0
- ww/storage_adapters/base.py +540 -0
- ww/storage_adapters/filesystem.py +370 -0
- ww/storage_adapters/memory.py +195 -0
- ww/storage_adapters/project_metadata.py +69 -0
- ww/storage_adapters/task_document.py +484 -0
- ww/task_ids.py +114 -0
- ww/task_references.py +124 -0
- ww/transitions.py +1619 -0
- ww/updates.py +399 -0
- ww/upgrade.py +95 -0
- ww/validation.py +168 -0
- ww/variables.py +275 -0
- ww/workflow_config.py +854 -0
- ww/workflow_update.py +239 -0
- ww/workflow_validation.py +1260 -0
- ww/workspace.py +50 -0
- ww_agentic_workflows-1.0.0.dev3.dist-info/METADATA +690 -0
- ww_agentic_workflows-1.0.0.dev3.dist-info/RECORD +167 -0
- ww_agentic_workflows-1.0.0.dev3.dist-info/WHEEL +4 -0
- ww_agentic_workflows-1.0.0.dev3.dist-info/entry_points.txt +2 -0
- ww_agentic_workflows-1.0.0.dev3.dist-info/licenses/LICENSE +674 -0
ww/inspect.py
ADDED
|
@@ -0,0 +1,896 @@
|
|
|
1
|
+
# SPDX-License-Identifier: GPL-3.0-or-later
|
|
2
|
+
"""A read-only profile of the checkout, built from its files and ``git``.
|
|
3
|
+
|
|
4
|
+
``ww inspect`` reads the working tree and the local Git history and reports
|
|
5
|
+
facts the setup workflows build a proposal from. Every fact names where it
|
|
6
|
+
came from, or says it was not found. Nothing is written and nothing leaves the
|
|
7
|
+
machine: Git runs as an argument list, without a shell, with a timeout and a
|
|
8
|
+
bounded ``-n``, and never fetches. A failing or slow Git call leaves only that
|
|
9
|
+
fact not found.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
import math
|
|
16
|
+
import os
|
|
17
|
+
import re
|
|
18
|
+
import statistics
|
|
19
|
+
import subprocess
|
|
20
|
+
import sys
|
|
21
|
+
import time
|
|
22
|
+
from collections import Counter
|
|
23
|
+
from collections.abc import Callable, Iterable, Sequence
|
|
24
|
+
from dataclasses import dataclass
|
|
25
|
+
from itertools import pairwise
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
from typing import Any, NamedTuple
|
|
28
|
+
|
|
29
|
+
import yaml
|
|
30
|
+
|
|
31
|
+
if sys.version_info >= (3, 11):
|
|
32
|
+
import tomllib
|
|
33
|
+
else: # pyproject.toml is read from Python 3.11, which ships tomllib.
|
|
34
|
+
tomllib = None
|
|
35
|
+
|
|
36
|
+
DEFAULT_COMMITS = 300
|
|
37
|
+
GIT_TIMEOUT_SECONDS = 15
|
|
38
|
+
DAY_SECONDS = 86_400
|
|
39
|
+
WEEK_SECONDS = 7 * DAY_SECONDS
|
|
40
|
+
ACTIVE_DAYS = 90
|
|
41
|
+
CADENCE_WEEKS = 12
|
|
42
|
+
DAILY_COMMITS_PER_WEEK = 5
|
|
43
|
+
SMALL_TEAM_MAX = 5
|
|
44
|
+
TOP_FIX_PATHS = 5
|
|
45
|
+
RECENT_FIXES = 5
|
|
46
|
+
TOP_HOT_PATHS = 10
|
|
47
|
+
TOP_TICKET_PREFIXES = 3
|
|
48
|
+
TAG_GAPS_FROM = 10
|
|
49
|
+
LIFETIME_MERGES = 10
|
|
50
|
+
LIFETIME_COMMITS = 200
|
|
51
|
+
MAJORITY = 0.5
|
|
52
|
+
# Below this share of merge commits the history reads as rebased or squashed.
|
|
53
|
+
MERGE_STYLE_SHARE = 0.05
|
|
54
|
+
MANIFEST_DEPTH = 2
|
|
55
|
+
MONOREPO_MIN_DIRECTORIES = 2
|
|
56
|
+
# Hash, parents, author e-mail, author time and subject, as LOG_FORMAT asks.
|
|
57
|
+
LOG_FIELDS = 5
|
|
58
|
+
MAX_LISTED = 200_000
|
|
59
|
+
LOG_FORMAT = "--format=%x1e%H%x1f%P%x1f%ae%x1f%at%x1f%s"
|
|
60
|
+
|
|
61
|
+
VERIFY_NAMES = ("test", "lint", "typecheck", "check", "format", "build")
|
|
62
|
+
# Every manifest is listed; package.json, composer.json, Makefile and
|
|
63
|
+
# pyproject.toml also name the verify commands.
|
|
64
|
+
MANIFEST_NAMES = frozenset(
|
|
65
|
+
{"package.json", "pyproject.toml", "Makefile", "composer.json", "go.mod"}
|
|
66
|
+
| {"Cargo.toml", "Gemfile", "pom.xml", "build.gradle", "build.gradle.kts"}
|
|
67
|
+
)
|
|
68
|
+
PYTHON_TOOLS = (
|
|
69
|
+
("pytest", "test", ("python", "-m", "pytest")),
|
|
70
|
+
("ruff", "lint", ("ruff", "check", ".")),
|
|
71
|
+
("mypy", "typecheck", ("mypy",)),
|
|
72
|
+
)
|
|
73
|
+
NODE_LOCKS = (
|
|
74
|
+
("pnpm-lock.yaml", "pnpm"),
|
|
75
|
+
("yarn.lock", "yarn"),
|
|
76
|
+
("bun.lockb", "bun"),
|
|
77
|
+
("bun.lock", "bun"),
|
|
78
|
+
)
|
|
79
|
+
MONOREPO_DIRECTORIES = ("apps", "packages", "services")
|
|
80
|
+
SKIPPED_DIRECTORIES = frozenset(
|
|
81
|
+
{"node_modules", "vendor", "dist", "build", "target", "venv", "__pycache__"}
|
|
82
|
+
)
|
|
83
|
+
AGENT_FILES: tuple[str, ...] = (
|
|
84
|
+
"AGENTS.md",
|
|
85
|
+
"CLAUDE.md",
|
|
86
|
+
"GEMINI.md",
|
|
87
|
+
".codex",
|
|
88
|
+
".claude",
|
|
89
|
+
)
|
|
90
|
+
AGENT_FILES += (".cursor/rules*", ".cursorrules")
|
|
91
|
+
PR_FILES = [
|
|
92
|
+
f"{directory}{name}"
|
|
93
|
+
for directory in (".github/", "", "docs/")
|
|
94
|
+
for name in ("PULL_REQUEST_TEMPLATE*", "pull_request_template*", "CODEOWNERS")
|
|
95
|
+
]
|
|
96
|
+
CI_FILES: tuple[str, ...] = (
|
|
97
|
+
".github/workflows/*.yml",
|
|
98
|
+
".github/workflows/*.yaml",
|
|
99
|
+
".gitlab-ci.yml",
|
|
100
|
+
)
|
|
101
|
+
CI_FILES += ("Jenkinsfile", ".circleci/config.yml")
|
|
102
|
+
|
|
103
|
+
# A subject that reports a fix: it starts, after an optional tracker key, with
|
|
104
|
+
# fix, fixes, fixed, hotfix or revert (so "fix:" and "fix(cli):" too), or it
|
|
105
|
+
# names a regression, e.g. "Fixed the totals", "PROJ-3: fix the crash" or
|
|
106
|
+
# "Guard the cache regression"; "Add the fix loop" does not match.
|
|
107
|
+
FIX_SUBJECT = re.compile(
|
|
108
|
+
r"^(?:\[?[A-Z][A-Z0-9]+-\d+\]?:?\s+)?(?:fix|fixes|fixed|hotfix|revert)\b"
|
|
109
|
+
r"|\bregression\b",
|
|
110
|
+
re.I,
|
|
111
|
+
)
|
|
112
|
+
FIX_RULE = (
|
|
113
|
+
"subject starts with fix, fixes, fixed, hotfix or revert, or names a regression"
|
|
114
|
+
)
|
|
115
|
+
# A tracker key such as an issue ID, upper case in a subject, e.g. "PROJ-12".
|
|
116
|
+
TICKET_KEY = re.compile(r"\b([A-Z][A-Z0-9]+)-\d+\b")
|
|
117
|
+
# The same key in a branch name, in any case, e.g. "feature/proj-12" gives "proj".
|
|
118
|
+
BRANCH_TICKET_KEY = re.compile(r"\b([A-Za-z][A-Za-z0-9]+)-\d+\b")
|
|
119
|
+
# A subject that starts with a tracker key, bracketed or not,
|
|
120
|
+
# e.g. "PROJ-12: Add the cache" or "[PROJ-12] Add the cache".
|
|
121
|
+
TICKET_SUBJECT = re.compile(r"^\[?[A-Z][A-Z0-9]+-\d+\]?[:\s]")
|
|
122
|
+
# A conventional-commits subject: type, optional scope, optional "!", colon,
|
|
123
|
+
# e.g. "fix(cli): keep the cursor".
|
|
124
|
+
CONVENTIONAL_SUBJECT = re.compile(r"^[a-z]+(\([\w./-]+\))?!?: ")
|
|
125
|
+
# The branch a merge commit's subject names, e.g. "Merge branch 'hotfix/x'"
|
|
126
|
+
# gives "hotfix/x" and "Merge pull request #7 from owner/feature/y" gives
|
|
127
|
+
# "owner/feature/y".
|
|
128
|
+
MERGED_BRANCH = re.compile(
|
|
129
|
+
r"^Merge (?:remote-tracking )?branch '([^']+)'|^Merge pull request #\d+ from (\S+)"
|
|
130
|
+
)
|
|
131
|
+
# A Makefile rule's target at the start of a line, not a variable assignment,
|
|
132
|
+
# e.g. "test: deps" gives "test".
|
|
133
|
+
MAKE_TARGET = re.compile(r"^([A-Za-z0-9][\w.-]*)\s*:(?![:=])", re.M)
|
|
134
|
+
# A relative link out of this checkout into a sibling directory,
|
|
135
|
+
# e.g. "../billing-api." gives "billing-api" and "../web.app" gives "web.app".
|
|
136
|
+
SIBLING_LINK = re.compile(r"(?<![\w.])\.\./([A-Za-z0-9][\w-]*(?:\.[\w-]+)*)")
|
|
137
|
+
# A Git remote URL, SSH or HTTPS, ending in ".git",
|
|
138
|
+
# e.g. "git@example.com:team/billing-api.git" gives "billing-api".
|
|
139
|
+
GIT_URL = re.compile(
|
|
140
|
+
r"(?:git@[\w.-]+:|https?://[\w.-]+/)[\w.-]+/([\w-]+(?:\.[\w-]+)*)\.git\b"
|
|
141
|
+
)
|
|
142
|
+
# A Compose file at the root, e.g. "docker-compose.dev.yml" or "compose.yaml".
|
|
143
|
+
COMPOSE_FILE = re.compile(r"(docker-)?compose[\w.-]*\.ya?ml$")
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
class Fact(NamedTuple):
|
|
147
|
+
"""One observation and where it came from; a ``value`` of ``None`` is not found."""
|
|
148
|
+
|
|
149
|
+
value: Any
|
|
150
|
+
evidence: str
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
Section = dict[str, Fact]
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
class Count(NamedTuple):
|
|
157
|
+
name: str
|
|
158
|
+
times: int
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
class Manifest(NamedTuple):
|
|
162
|
+
path: str
|
|
163
|
+
scripts: tuple[str, ...] = ()
|
|
164
|
+
workspaces: tuple[str, ...] = ()
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
class VerifyCommand(NamedTuple):
|
|
168
|
+
"""A command a manifest names for checking the code, as exact argv."""
|
|
169
|
+
|
|
170
|
+
name: str
|
|
171
|
+
argv: tuple[str, ...]
|
|
172
|
+
directory: str
|
|
173
|
+
source: str
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
class Commit(NamedTuple):
|
|
177
|
+
sha: str
|
|
178
|
+
parents: tuple[str, ...]
|
|
179
|
+
author: str
|
|
180
|
+
timestamp: int
|
|
181
|
+
subject: str
|
|
182
|
+
# (path, added + deleted lines); binary files count zero lines.
|
|
183
|
+
changes: tuple[tuple[str, int], ...]
|
|
184
|
+
|
|
185
|
+
@property
|
|
186
|
+
def is_merge(self) -> bool:
|
|
187
|
+
return len(self.parents) > 1
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
@dataclass(frozen=True)
|
|
191
|
+
class Profile:
|
|
192
|
+
"""Everything ``ww inspect`` found, by section; Git sections need Git."""
|
|
193
|
+
|
|
194
|
+
root: str
|
|
195
|
+
commits_window: int
|
|
196
|
+
git_unavailable: str | None
|
|
197
|
+
sections: dict[str, Section]
|
|
198
|
+
|
|
199
|
+
def value(self, section: str, key: str) -> Any:
|
|
200
|
+
return self.sections[section][key].value
|
|
201
|
+
|
|
202
|
+
def to_dict(self) -> dict[str, Any]:
|
|
203
|
+
"""The profile as JSON-ready data; each fact is its value and evidence."""
|
|
204
|
+
plain: dict[str, Any] = _plain(vars(self))
|
|
205
|
+
return plain
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _plain(value: Any) -> Any:
|
|
209
|
+
if hasattr(value, "_asdict"):
|
|
210
|
+
value = value._asdict()
|
|
211
|
+
if isinstance(value, dict):
|
|
212
|
+
return {key: _plain(item) for key, item in value.items()}
|
|
213
|
+
if isinstance(value, (list, tuple)):
|
|
214
|
+
return [_plain(item) for item in value]
|
|
215
|
+
return value
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def _counts(counts: tuple[Count, ...]) -> str:
|
|
219
|
+
return ", ".join(f"{count.name} {count.times}" for count in counts)
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _manifests(manifests: tuple[Manifest, ...]) -> str:
|
|
223
|
+
shown = []
|
|
224
|
+
for manifest in manifests:
|
|
225
|
+
details = [f"{len(manifest.scripts)} scripts"] if manifest.scripts else []
|
|
226
|
+
details += ["workspaces"] if manifest.workspaces else []
|
|
227
|
+
shown.append(manifest.path + (f" [{', '.join(details)}]" if details else ""))
|
|
228
|
+
return ", ".join(shown)
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _verify(commands: tuple[VerifyCommand, ...]) -> str:
|
|
232
|
+
return "; ".join(
|
|
233
|
+
f"{command.name}: `{' '.join(command.argv)}`"
|
|
234
|
+
+ ("" if command.directory == "." else f" in {command.directory}")
|
|
235
|
+
+ f" from {command.source}"
|
|
236
|
+
for command in commands
|
|
237
|
+
)
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
_percent = "{:.0%}".format
|
|
241
|
+
_number = "{:g}".format
|
|
242
|
+
_code = "`{}`".format
|
|
243
|
+
_comma = ", ".join
|
|
244
|
+
_semicolon = "; ".join
|
|
245
|
+
|
|
246
|
+
# Every fact of each section, in page order: its label and how its value reads.
|
|
247
|
+
# The section's title is its key with spaces, e.g. "Hot paths".
|
|
248
|
+
LABELS: dict[str, dict[str, tuple[str, Callable[[Any], str]]]] = {
|
|
249
|
+
"repository": {
|
|
250
|
+
"default_branch": ("Default branch", str),
|
|
251
|
+
"integration_branch": ("Integration branch", str),
|
|
252
|
+
"branch_patterns": ("Branch patterns", _counts),
|
|
253
|
+
"remotes": ("Remotes", _comma),
|
|
254
|
+
"shallow": ("Shallow clone", lambda shallow: "yes" if shallow else "no"),
|
|
255
|
+
"merge_share": ("Merge commits", _percent),
|
|
256
|
+
"merge_style": ("Merge style", str),
|
|
257
|
+
"pr_signals": ("PR signals", _semicolon),
|
|
258
|
+
"tags": ("Tags", str),
|
|
259
|
+
"tag_interval_days": ("Days between tags", _number),
|
|
260
|
+
},
|
|
261
|
+
"activity": {
|
|
262
|
+
"commits": ("Commits read", str),
|
|
263
|
+
"contributors": ("Contributors", str),
|
|
264
|
+
"active_contributors": ("Active in 90 days", str),
|
|
265
|
+
"team_shape": ("Team shape", str),
|
|
266
|
+
"weekly_commits": (
|
|
267
|
+
"Commits per week",
|
|
268
|
+
lambda weekly: " ".join(map(str, weekly)) + " (oldest first)",
|
|
269
|
+
),
|
|
270
|
+
"cadence": ("Cadence", str),
|
|
271
|
+
"files_per_commit": ("Files per commit (median)", _number),
|
|
272
|
+
"lines_per_commit": ("Lines per commit (median)", _number),
|
|
273
|
+
"branch_lifetime_days": ("Branch lifetime in days (median)", _number),
|
|
274
|
+
},
|
|
275
|
+
"fixes": {
|
|
276
|
+
"share": ("Fix share", _percent),
|
|
277
|
+
"paths": ("Paths fixes touch", _counts),
|
|
278
|
+
"recent": (
|
|
279
|
+
"Recent fixes",
|
|
280
|
+
lambda subjects: _semicolon(f'"{s}"' for s in subjects),
|
|
281
|
+
),
|
|
282
|
+
},
|
|
283
|
+
"hot_paths": {"most_changed": ("Most changed", _counts)},
|
|
284
|
+
"layout": {
|
|
285
|
+
"manifests": ("Manifests", _manifests),
|
|
286
|
+
"ci": ("CI", _comma),
|
|
287
|
+
"verify": ("Verify commands", _verify),
|
|
288
|
+
"monorepo_signals": ("Monorepo signals", _semicolon),
|
|
289
|
+
"projects": ("Candidate projects", _comma),
|
|
290
|
+
},
|
|
291
|
+
"conventions": {
|
|
292
|
+
"ticket_prefixes": ("Ticket prefixes", _counts),
|
|
293
|
+
"task_format": ("task_format candidate", _code),
|
|
294
|
+
"ticket_share": ("Subjects led by a ticket key", _percent),
|
|
295
|
+
"conventional_share": ("Conventional-commit subjects", _percent),
|
|
296
|
+
"subject_convention": ("Subject convention", str),
|
|
297
|
+
"commit_format": ("commit_format candidate", _code),
|
|
298
|
+
"agent_files": ("Agent instruction files", _comma),
|
|
299
|
+
},
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def _section(name: str, unknown: str = "", **found: tuple[Any, str]) -> Section:
|
|
304
|
+
"""A section from ``key=(value, evidence)``; a key left out is not found."""
|
|
305
|
+
return {key: Fact(*found.get(key, (None, unknown))) for key in LABELS[name]}
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
def _git(root: Path, *arguments: str) -> str | None:
|
|
309
|
+
"""Run one read-only Git command; ``None`` when it fails or times out."""
|
|
310
|
+
environment = {**os.environ, "GIT_TERMINAL_PROMPT": "0", "GIT_OPTIONAL_LOCKS": "0"}
|
|
311
|
+
try:
|
|
312
|
+
result = subprocess.run(
|
|
313
|
+
["git", "-c", "core.quotePath=false", *arguments],
|
|
314
|
+
cwd=root,
|
|
315
|
+
capture_output=True,
|
|
316
|
+
text=True,
|
|
317
|
+
encoding="utf-8",
|
|
318
|
+
errors="replace",
|
|
319
|
+
check=False,
|
|
320
|
+
timeout=GIT_TIMEOUT_SECONDS,
|
|
321
|
+
env=environment,
|
|
322
|
+
)
|
|
323
|
+
except (OSError, subprocess.SubprocessError):
|
|
324
|
+
return None
|
|
325
|
+
return result.stdout if result.returncode == 0 else None
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
def _lines(output: str | None) -> list[str]:
|
|
329
|
+
return [line for line in (output or "").splitlines() if line.strip()][:MAX_LISTED]
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
def _ref_exists(root: Path, ref: str) -> bool:
|
|
333
|
+
return _git(root, "rev-parse", "--verify", "--quiet", ref) is not None
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
def _read(path: Path) -> str:
|
|
337
|
+
"""The file's text; empty when it cannot be read."""
|
|
338
|
+
try:
|
|
339
|
+
return path.read_text(encoding="utf-8")
|
|
340
|
+
except (OSError, UnicodeDecodeError):
|
|
341
|
+
return ""
|
|
342
|
+
|
|
343
|
+
|
|
344
|
+
def _present(root: Path, patterns: Iterable[str]) -> tuple[str, ...]:
|
|
345
|
+
"""The paths under ``root`` the glob patterns name, each once, in order."""
|
|
346
|
+
paths = (path for pattern in patterns for path in sorted(root.glob(pattern)))
|
|
347
|
+
return tuple(dict.fromkeys(path.relative_to(root).as_posix() for path in paths))
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def parse_log(output: str) -> list[Commit]:
|
|
351
|
+
"""Read ``git log --numstat`` output in :data:`LOG_FORMAT`."""
|
|
352
|
+
commits = []
|
|
353
|
+
for record in output.split("\x1e"):
|
|
354
|
+
header, _, body = record.strip("\n").partition("\n")
|
|
355
|
+
fields = header.split("\x1f")
|
|
356
|
+
if len(fields) != LOG_FIELDS:
|
|
357
|
+
continue
|
|
358
|
+
sha, parents, author, timestamp, subject = fields
|
|
359
|
+
changes = []
|
|
360
|
+
for line in body.splitlines():
|
|
361
|
+
added, _, rest = line.partition("\t")
|
|
362
|
+
deleted, _, path = rest.partition("\t")
|
|
363
|
+
if path:
|
|
364
|
+
numeric = added.isdigit() and deleted.isdigit()
|
|
365
|
+
changes.append((path, int(added) + int(deleted) if numeric else 0))
|
|
366
|
+
stamp = int(timestamp) if timestamp.isdigit() else 0
|
|
367
|
+
heads = tuple(parents.split())
|
|
368
|
+
commits.append(
|
|
369
|
+
Commit(sha, heads, author.lower(), stamp, subject, tuple(changes))
|
|
370
|
+
)
|
|
371
|
+
return commits
|
|
372
|
+
|
|
373
|
+
|
|
374
|
+
def team_shape(active: int) -> str:
|
|
375
|
+
"""Solo for at most one contributor active in 90 days, small to five, else team."""
|
|
376
|
+
if active <= 1:
|
|
377
|
+
return "solo"
|
|
378
|
+
return "small" if active <= SMALL_TEAM_MAX else "team"
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
def cadence(weekly: Sequence[int]) -> str:
|
|
382
|
+
"""Daily at a median of five commits a week or more, weekly at one, else sparse."""
|
|
383
|
+
median = statistics.median(weekly) if weekly else 0
|
|
384
|
+
if median >= DAILY_COMMITS_PER_WEEK:
|
|
385
|
+
return "daily"
|
|
386
|
+
return "weekly" if median >= 1 else "sparse"
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
def history_weeks(timestamps: Iterable[int], now: float) -> int:
|
|
390
|
+
"""The weeks from the oldest commit read to now: at least one, at most 12.
|
|
391
|
+
|
|
392
|
+
A young history is measured over the weeks it has, so a week of daily
|
|
393
|
+
commits reads as daily rather than as one busy week among eleven empty ones.
|
|
394
|
+
"""
|
|
395
|
+
oldest = min(timestamps, default=now)
|
|
396
|
+
return min(CADENCE_WEEKS, max(1, math.ceil((now - oldest) / WEEK_SECONDS)))
|
|
397
|
+
|
|
398
|
+
|
|
399
|
+
def weekly_commits(
|
|
400
|
+
timestamps: Iterable[int], now: float, weeks: int = CADENCE_WEEKS
|
|
401
|
+
) -> tuple[int, ...]:
|
|
402
|
+
"""Commits in each of the last ``weeks`` weeks, oldest week first."""
|
|
403
|
+
counts = [0] * weeks
|
|
404
|
+
for timestamp in timestamps:
|
|
405
|
+
age = int((now - timestamp) // WEEK_SECONDS)
|
|
406
|
+
if 0 <= age < weeks:
|
|
407
|
+
counts[weeks - 1 - age] += 1
|
|
408
|
+
return tuple(counts)
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
def is_fix(subject: str) -> bool:
|
|
412
|
+
return FIX_SUBJECT.search(subject) is not None
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
def top_counts(names: Iterable[str], limit: int) -> tuple[Count, ...]:
|
|
416
|
+
"""The ``limit`` most frequent names, ties in first-seen order."""
|
|
417
|
+
return tuple(Count(*pair) for pair in Counter(names).most_common(limit))
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
def ticket_prefixes(
|
|
421
|
+
subjects: Iterable[str], branches: Iterable[str] = ()
|
|
422
|
+
) -> tuple[Count, ...]:
|
|
423
|
+
"""Upper-case tracker key prefixes, counted once per subject or branch name.
|
|
424
|
+
|
|
425
|
+
Subjects count only upper-case keys; branch names, often lower case, count
|
|
426
|
+
keys in any case.
|
|
427
|
+
"""
|
|
428
|
+
seen: list[str] = []
|
|
429
|
+
for subject in subjects:
|
|
430
|
+
seen.extend(dict.fromkeys(TICKET_KEY.findall(subject)))
|
|
431
|
+
for branch in branches:
|
|
432
|
+
seen.extend(dict.fromkeys(k.upper() for k in BRANCH_TICKET_KEY.findall(branch)))
|
|
433
|
+
return top_counts(seen, TOP_TICKET_PREFIXES)
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
def commit_format_for(ticket_share: float) -> str:
|
|
437
|
+
"""The task ID leads the subject when most subjects start with a key."""
|
|
438
|
+
if ticket_share >= MAJORITY:
|
|
439
|
+
return "{{ww.task.id}}: {{commit_message}}"
|
|
440
|
+
return "{{commit_message}}"
|
|
441
|
+
|
|
442
|
+
|
|
443
|
+
def subject_convention(ticket_share: float, conventional_share: float) -> str:
|
|
444
|
+
"""Whichever convention most subjects follow, else free."""
|
|
445
|
+
if ticket_share >= MAJORITY:
|
|
446
|
+
return "ticket prefix"
|
|
447
|
+
if conventional_share >= MAJORITY:
|
|
448
|
+
return "conventional commits"
|
|
449
|
+
return "free"
|
|
450
|
+
|
|
451
|
+
|
|
452
|
+
def _share(part: int, whole: int) -> float | None:
|
|
453
|
+
return round(part / whole, 3) if whole else None
|
|
454
|
+
|
|
455
|
+
|
|
456
|
+
def _median(values: Sequence[float]) -> float | None:
|
|
457
|
+
return round(float(statistics.median(values)), 1) if values else None
|
|
458
|
+
|
|
459
|
+
|
|
460
|
+
def inspect_checkout(root: Path, commits: int = DEFAULT_COMMITS) -> Profile:
|
|
461
|
+
"""Profile the checkout at ``root`` from at most ``commits`` commits."""
|
|
462
|
+
root = root.resolve()
|
|
463
|
+
files, files_evidence = _tracked_files(root)
|
|
464
|
+
layout = _layout(root, files, files_evidence)
|
|
465
|
+
if _git(root, "rev-parse", "--is-inside-work-tree") is None:
|
|
466
|
+
conventions = _conventions(root, [], [], "no Git history")
|
|
467
|
+
unavailable = "not a Git repository (git rev-parse)"
|
|
468
|
+
sections = {"layout": layout, "conventions": conventions}
|
|
469
|
+
return Profile(str(root), commits, unavailable, sections)
|
|
470
|
+
history: list[Commit] = []
|
|
471
|
+
if _ref_exists(root, "HEAD"):
|
|
472
|
+
log = _git(root, "log", f"-n{commits}", "--no-renames", "--numstat", LOG_FORMAT)
|
|
473
|
+
history = parse_log(log or "")
|
|
474
|
+
window = f"git log -n {commits}"
|
|
475
|
+
branches, branches_evidence = _branch_names(root, history)
|
|
476
|
+
hot = top_counts((path for c in history for path, _ in c.changes), TOP_HOT_PATHS)
|
|
477
|
+
sections = {
|
|
478
|
+
"repository": _repository(root, history, window, branches, branches_evidence),
|
|
479
|
+
"activity": _activity(root, history, window),
|
|
480
|
+
"fixes": _fixes(history, window),
|
|
481
|
+
"hot_paths": _section(
|
|
482
|
+
"hot_paths", most_changed=(hot if history else None, f"{window} --numstat")
|
|
483
|
+
),
|
|
484
|
+
"layout": layout,
|
|
485
|
+
"conventions": _conventions(
|
|
486
|
+
root, history, branches, f"{window}; {branches_evidence}"
|
|
487
|
+
),
|
|
488
|
+
}
|
|
489
|
+
return Profile(str(root), commits, None, sections)
|
|
490
|
+
|
|
491
|
+
|
|
492
|
+
def _branch_names(root: Path, history: list[Commit]) -> tuple[list[str], str]:
|
|
493
|
+
"""Branch names without their remote, plus those merge subjects name."""
|
|
494
|
+
names = [name.partition("/")[2] for name in _refs(root, "refs/remotes")]
|
|
495
|
+
evidence = "git branch -r"
|
|
496
|
+
if not names:
|
|
497
|
+
names = _refs(root, "refs/heads")
|
|
498
|
+
evidence = "git branch"
|
|
499
|
+
merged = []
|
|
500
|
+
for commit in history:
|
|
501
|
+
match = MERGED_BRANCH.match(commit.subject)
|
|
502
|
+
if match and match.group(1):
|
|
503
|
+
merged.append(match.group(1))
|
|
504
|
+
elif match:
|
|
505
|
+
# A pull request names "owner/branch"; the owner is not a lane.
|
|
506
|
+
merged.append(match.group(2).partition("/")[2])
|
|
507
|
+
if merged:
|
|
508
|
+
evidence += ", merge subjects"
|
|
509
|
+
return [name for name in names + merged if name], evidence
|
|
510
|
+
|
|
511
|
+
|
|
512
|
+
def _refs(root: Path, namespace: str) -> list[str]:
|
|
513
|
+
"""Short names of the refs under ``namespace``, without symbolic refs."""
|
|
514
|
+
output = _git(
|
|
515
|
+
root, "for-each-ref", "--format=%(refname:short)%09%(symref)", namespace
|
|
516
|
+
)
|
|
517
|
+
lines = (line.partition("\t") for line in _lines(output))
|
|
518
|
+
return [name for name, _, symref in lines if not symref]
|
|
519
|
+
|
|
520
|
+
|
|
521
|
+
def _repository(
|
|
522
|
+
root: Path, history: list[Commit], window: str, branches: list[str], evidence: str
|
|
523
|
+
) -> Section:
|
|
524
|
+
default = _default_branch(root)
|
|
525
|
+
merges = sum(commit.is_merge for commit in history)
|
|
526
|
+
merge_share = _share(merges, len(history))
|
|
527
|
+
style = None
|
|
528
|
+
if merge_share is not None:
|
|
529
|
+
linear = merge_share < MERGE_STYLE_SHARE
|
|
530
|
+
style = "linear (rebase or squash)" if linear else "merge commits"
|
|
531
|
+
# A lane is a branch's first path segment with its slash, e.g. "feature/".
|
|
532
|
+
lanes = [f"{name.partition('/')[0]}/" for name in branches if "/" in name[1:]]
|
|
533
|
+
shallow = _git(root, "rev-parse", "--is-shallow-repository")
|
|
534
|
+
pulls = sum(commit.subject.startswith("Merge pull request") for commit in history)
|
|
535
|
+
pr_signals = _present(root, PR_FILES)
|
|
536
|
+
if pulls:
|
|
537
|
+
pr_signals += (f'{pulls} "Merge pull request" subjects',)
|
|
538
|
+
tags = "--format=%(creatordate:unix)"
|
|
539
|
+
stamps = _lines(
|
|
540
|
+
_git(root, "for-each-ref", "--sort=-creatordate", tags, "refs/tags")
|
|
541
|
+
)[:TAG_GAPS_FROM]
|
|
542
|
+
gaps = [(int(new) - int(old)) / DAY_SECONDS for new, old in pairwise(stamps)]
|
|
543
|
+
return _section(
|
|
544
|
+
"repository",
|
|
545
|
+
default_branch=default,
|
|
546
|
+
integration_branch=_integration_branch(root, default[0]),
|
|
547
|
+
branch_patterns=(top_counts(lanes, len(lanes) or 1), evidence),
|
|
548
|
+
remotes=(tuple(_lines(_git(root, "remote"))), "git remote"),
|
|
549
|
+
shallow=(
|
|
550
|
+
None if shallow is None else shallow.strip() == "true",
|
|
551
|
+
"git rev-parse --is-shallow-repository",
|
|
552
|
+
),
|
|
553
|
+
merge_share=(merge_share, f"{merges} of {len(history)} commits, {window}"),
|
|
554
|
+
merge_style=(style, f"merge share against {MERGE_STYLE_SHARE:.0%}, {window}"),
|
|
555
|
+
pr_signals=(pr_signals, f"files and subjects, {window}"),
|
|
556
|
+
tags=(len(_lines(_git(root, "tag", "--list"))), "git tag --list"),
|
|
557
|
+
tag_interval_days=(
|
|
558
|
+
_median(gaps),
|
|
559
|
+
f"last {TAG_GAPS_FROM} tags, git for-each-ref",
|
|
560
|
+
),
|
|
561
|
+
)
|
|
562
|
+
|
|
563
|
+
|
|
564
|
+
def _default_branch(root: Path) -> tuple[str | None, str]:
|
|
565
|
+
head = _git(root, "symbolic-ref", "--short", "refs/remotes/origin/HEAD")
|
|
566
|
+
if head:
|
|
567
|
+
return head.strip().partition("/")[2], "git symbolic-ref origin/HEAD"
|
|
568
|
+
for name in ("main", "master"):
|
|
569
|
+
for ref in (f"refs/heads/{name}", f"refs/remotes/origin/{name}"):
|
|
570
|
+
if _ref_exists(root, ref):
|
|
571
|
+
return name, f"{ref} exists"
|
|
572
|
+
current = _git(root, "symbolic-ref", "--short", "HEAD")
|
|
573
|
+
if current:
|
|
574
|
+
return current.strip(), "current branch, git symbolic-ref HEAD"
|
|
575
|
+
return None, "origin/HEAD, main, master"
|
|
576
|
+
|
|
577
|
+
|
|
578
|
+
def _integration_branch(root: Path, default: str | None) -> tuple[str | None, str]:
|
|
579
|
+
"""``dev`` or ``develop`` when merges land on it, else the default branch."""
|
|
580
|
+
for name in ("dev", "develop"):
|
|
581
|
+
for ref in (f"refs/heads/{name}", f"refs/remotes/origin/{name}"):
|
|
582
|
+
merged = _git(
|
|
583
|
+
root, "log", "-n", "1", "--first-parent", "--merges", "--format=%H", ref
|
|
584
|
+
)
|
|
585
|
+
if merged and merged.strip():
|
|
586
|
+
return name, f"merges on {ref}, git log --first-parent --merges"
|
|
587
|
+
return default, "no dev or develop receiving merges; the default branch"
|
|
588
|
+
|
|
589
|
+
|
|
590
|
+
def _activity(root: Path, history: list[Commit], window: str) -> Section:
|
|
591
|
+
if not history:
|
|
592
|
+
return _section("activity", window, commits=(0, window))
|
|
593
|
+
now = time.time()
|
|
594
|
+
stamps = [commit.timestamp for commit in history]
|
|
595
|
+
recent = now - ACTIVE_DAYS * DAY_SECONDS
|
|
596
|
+
active = len({commit.author for commit in history if commit.timestamp >= recent})
|
|
597
|
+
weeks = history_weeks(stamps, now)
|
|
598
|
+
weekly = weekly_commits(stamps, now, weeks)
|
|
599
|
+
span = f"the {weeks} week{'s' * (weeks != 1)} the history spans, at most 12"
|
|
600
|
+
changes = [commit for commit in history if not commit.is_merge]
|
|
601
|
+
numstat = f"non-merge commits, {window} --numstat"
|
|
602
|
+
return _section(
|
|
603
|
+
"activity",
|
|
604
|
+
commits=(len(history), window),
|
|
605
|
+
contributors=(
|
|
606
|
+
len({commit.author for commit in history}),
|
|
607
|
+
f"author e-mails, {window}",
|
|
608
|
+
),
|
|
609
|
+
active_contributors=(active, f"authors in {ACTIVE_DAYS} days, {window}"),
|
|
610
|
+
team_shape=(
|
|
611
|
+
team_shape(active),
|
|
612
|
+
f"{active} active in {ACTIVE_DAYS} days: solo 1, small 2-5, team 6+",
|
|
613
|
+
),
|
|
614
|
+
weekly_commits=(weekly, f"{span}, {window}"),
|
|
615
|
+
cadence=(
|
|
616
|
+
cadence(weekly),
|
|
617
|
+
f"median {statistics.median(weekly):g}/week over {span}: "
|
|
618
|
+
"daily 5+, weekly 1+",
|
|
619
|
+
),
|
|
620
|
+
files_per_commit=(_median([len(c.changes) for c in changes]), numstat),
|
|
621
|
+
lines_per_commit=(
|
|
622
|
+
_median([sum(lines for _, lines in c.changes) for c in changes]),
|
|
623
|
+
numstat,
|
|
624
|
+
),
|
|
625
|
+
branch_lifetime_days=(
|
|
626
|
+
_branch_lifetime(root, history),
|
|
627
|
+
f"first commit to merge, last {LIFETIME_MERGES} merges, git log P1..P2",
|
|
628
|
+
),
|
|
629
|
+
)
|
|
630
|
+
|
|
631
|
+
|
|
632
|
+
def _branch_lifetime(root: Path, history: list[Commit]) -> float | None:
|
|
633
|
+
"""Median days from a merged branch's first own commit to its merge."""
|
|
634
|
+
lifetimes = []
|
|
635
|
+
for merge in [commit for commit in history if commit.is_merge][:LIFETIME_MERGES]:
|
|
636
|
+
branch = f"{merge.parents[0]}..{merge.parents[1]}"
|
|
637
|
+
output = _git(root, "log", f"-n{LIFETIME_COMMITS}", "--format=%at", branch)
|
|
638
|
+
stamps = [int(line) for line in _lines(output) if line.isdigit()]
|
|
639
|
+
if stamps:
|
|
640
|
+
lifetimes.append(max(0, merge.timestamp - min(stamps)) / DAY_SECONDS)
|
|
641
|
+
return _median(lifetimes)
|
|
642
|
+
|
|
643
|
+
|
|
644
|
+
def _fixes(history: list[Commit], window: str) -> Section:
|
|
645
|
+
# Merge subjects name branches ("Merge branch 'hotfix/x'"), not fixes, so
|
|
646
|
+
# the share is over the commits that change something themselves.
|
|
647
|
+
changes = [commit for commit in history if not commit.is_merge]
|
|
648
|
+
fixes = [commit for commit in changes if is_fix(commit.subject)]
|
|
649
|
+
rule = f"{len(fixes)} of {len(changes)} non-merge commits, {FIX_RULE}, {window}"
|
|
650
|
+
if not changes:
|
|
651
|
+
return _section("fixes", window, share=(None, rule))
|
|
652
|
+
paths = top_counts((path for c in fixes for path, _ in c.changes), TOP_FIX_PATHS)
|
|
653
|
+
return _section(
|
|
654
|
+
"fixes",
|
|
655
|
+
share=(_share(len(fixes), len(changes)), rule),
|
|
656
|
+
paths=(paths, f"paths the fix commits touch, {window} --numstat"),
|
|
657
|
+
recent=(tuple(commit.subject for commit in fixes[:RECENT_FIXES]), window),
|
|
658
|
+
)
|
|
659
|
+
|
|
660
|
+
|
|
661
|
+
def _conventions(
|
|
662
|
+
root: Path, history: list[Commit], branches: list[str], evidence: str
|
|
663
|
+
) -> Section:
|
|
664
|
+
subjects = [commit.subject for commit in history if not commit.is_merge]
|
|
665
|
+
prefixes = ticket_prefixes(subjects, branches)
|
|
666
|
+
tickets = sum(TICKET_SUBJECT.match(subject) is not None for subject in subjects)
|
|
667
|
+
conventional = sum(bool(CONVENTIONAL_SUBJECT.match(s)) for s in subjects)
|
|
668
|
+
ticket_share = _share(tickets, len(subjects))
|
|
669
|
+
conventional_share = _share(conventional, len(subjects))
|
|
670
|
+
shares = f"{tickets} ticket-led and {conventional} conventional of {len(subjects)}"
|
|
671
|
+
shares += " non-merge subjects"
|
|
672
|
+
convention = commit_format = None
|
|
673
|
+
if ticket_share is not None:
|
|
674
|
+
convention = subject_convention(ticket_share, conventional_share or 0.0)
|
|
675
|
+
commit_format = commit_format_for(ticket_share)
|
|
676
|
+
task_format = f"{prefixes[0].name}-{{{{digit}}}}" if prefixes else None
|
|
677
|
+
return _section(
|
|
678
|
+
"conventions",
|
|
679
|
+
ticket_prefixes=(prefixes if history or branches else None, evidence),
|
|
680
|
+
task_format=(task_format, "the most frequent tracker key prefix"),
|
|
681
|
+
ticket_share=(ticket_share, shares),
|
|
682
|
+
conventional_share=(conventional_share, shares),
|
|
683
|
+
subject_convention=(
|
|
684
|
+
convention,
|
|
685
|
+
"the convention half the subjects follow, else free",
|
|
686
|
+
),
|
|
687
|
+
commit_format=(
|
|
688
|
+
commit_format,
|
|
689
|
+
"task ID first when half the subjects start with a key",
|
|
690
|
+
),
|
|
691
|
+
agent_files=(_present(root, AGENT_FILES), "files at the project root"),
|
|
692
|
+
)
|
|
693
|
+
|
|
694
|
+
|
|
695
|
+
def _tracked_files(root: Path) -> tuple[list[str], str]:
|
|
696
|
+
"""Paths at most two directories deep: Git's tracked files, else a walk."""
|
|
697
|
+
output = _git(root, "ls-files")
|
|
698
|
+
if output is not None and output.strip():
|
|
699
|
+
paths = [line for line in _lines(output) if line.count("/") <= MANIFEST_DEPTH]
|
|
700
|
+
return paths, "git ls-files"
|
|
701
|
+
paths = []
|
|
702
|
+
for directory, subdirectories, names in os.walk(root):
|
|
703
|
+
relative = Path(directory).relative_to(root)
|
|
704
|
+
subdirectories[:] = sorted(
|
|
705
|
+
name
|
|
706
|
+
for name in subdirectories
|
|
707
|
+
if len(relative.parts) < MANIFEST_DEPTH
|
|
708
|
+
and not name.startswith(".")
|
|
709
|
+
and name not in SKIPPED_DIRECTORIES
|
|
710
|
+
)
|
|
711
|
+
paths.extend((relative / name).as_posix() for name in sorted(names))
|
|
712
|
+
return paths, "files on disk"
|
|
713
|
+
|
|
714
|
+
|
|
715
|
+
def _layout(root: Path, files: list[str], files_evidence: str) -> Section:
|
|
716
|
+
manifests: list[Manifest] = []
|
|
717
|
+
verify: list[VerifyCommand] = []
|
|
718
|
+
for path in files:
|
|
719
|
+
name = path.rpartition("/")[2]
|
|
720
|
+
if name in MANIFEST_NAMES and path.split("/")[0] not in SKIPPED_DIRECTORIES:
|
|
721
|
+
manifest, commands = _read_manifest(root, path)
|
|
722
|
+
manifests.append(manifest)
|
|
723
|
+
verify.extend(commands)
|
|
724
|
+
signals, projects = _monorepo(root, files, manifests)
|
|
725
|
+
return _section(
|
|
726
|
+
"layout",
|
|
727
|
+
manifests=(tuple(manifests), f"root and two levels down, {files_evidence}"),
|
|
728
|
+
ci=(
|
|
729
|
+
tuple(sorted(_present(root, CI_FILES))),
|
|
730
|
+
".github/workflows, .gitlab-ci.yml, Jenkinsfile, .circleci",
|
|
731
|
+
),
|
|
732
|
+
verify=(
|
|
733
|
+
tuple(verify),
|
|
734
|
+
f"scripts and targets named {'/'.join(VERIFY_NAMES)} in the manifests",
|
|
735
|
+
),
|
|
736
|
+
monorepo_signals=(
|
|
737
|
+
signals,
|
|
738
|
+
f"manifests, compose files, README, {files_evidence}",
|
|
739
|
+
),
|
|
740
|
+
projects=(projects, "directories and sibling repositories the signals name"),
|
|
741
|
+
)
|
|
742
|
+
|
|
743
|
+
|
|
744
|
+
def _read_manifest(root: Path, path: str) -> tuple[Manifest, list[VerifyCommand]]:
|
|
745
|
+
"""One manifest and the verify commands it names; unreadable means none."""
|
|
746
|
+
directory, _, name = path.rpartition("/")
|
|
747
|
+
directory = directory or "."
|
|
748
|
+
text = _read(root / path)
|
|
749
|
+
if name in ("package.json", "composer.json"):
|
|
750
|
+
try:
|
|
751
|
+
data = json.loads(text)
|
|
752
|
+
except json.JSONDecodeError:
|
|
753
|
+
data = None
|
|
754
|
+
data = data if isinstance(data, dict) else {}
|
|
755
|
+
scripts = (
|
|
756
|
+
tuple(data["scripts"]) if isinstance(data.get("scripts"), dict) else ()
|
|
757
|
+
)
|
|
758
|
+
listed = data.get("workspaces")
|
|
759
|
+
if isinstance(listed, dict):
|
|
760
|
+
listed = listed.get("packages")
|
|
761
|
+
workspaces = tuple(map(str, listed)) if isinstance(listed, list) else ()
|
|
762
|
+
runner = (
|
|
763
|
+
"composer" if name == "composer.json" else _node_runner(root, directory)
|
|
764
|
+
)
|
|
765
|
+
commands = [
|
|
766
|
+
VerifyCommand(verb, (runner, "run", script), directory, path)
|
|
767
|
+
for script in scripts
|
|
768
|
+
if (verb := script.partition(":")[0]) in VERIFY_NAMES
|
|
769
|
+
]
|
|
770
|
+
return Manifest(path, scripts, workspaces), commands
|
|
771
|
+
if name == "Makefile":
|
|
772
|
+
targets = tuple(dict.fromkeys(MAKE_TARGET.findall(text)))
|
|
773
|
+
return Manifest(path, targets), [
|
|
774
|
+
VerifyCommand(target, ("make", target), directory, path)
|
|
775
|
+
for target in targets
|
|
776
|
+
if target in VERIFY_NAMES
|
|
777
|
+
]
|
|
778
|
+
tools: Any = {}
|
|
779
|
+
if name == "pyproject.toml" and tomllib is not None:
|
|
780
|
+
try:
|
|
781
|
+
tools = tomllib.loads(text).get("tool", {})
|
|
782
|
+
except tomllib.TOMLDecodeError:
|
|
783
|
+
tools = {}
|
|
784
|
+
return Manifest(path), [
|
|
785
|
+
VerifyCommand(verb, argv, directory, f"{path} [tool.{tool}]")
|
|
786
|
+
for tool, verb, argv in PYTHON_TOOLS
|
|
787
|
+
if isinstance(tools, dict) and tool in tools
|
|
788
|
+
]
|
|
789
|
+
|
|
790
|
+
|
|
791
|
+
def _node_runner(root: Path, directory: str) -> str:
|
|
792
|
+
"""The package manager a lock file names, in the package or at the root."""
|
|
793
|
+
for place in dict.fromkeys((root / directory, root)):
|
|
794
|
+
for lock, runner in NODE_LOCKS:
|
|
795
|
+
if (place / lock).exists():
|
|
796
|
+
return runner
|
|
797
|
+
return "npm"
|
|
798
|
+
|
|
799
|
+
|
|
800
|
+
def _monorepo(
|
|
801
|
+
root: Path, files: list[str], manifests: list[Manifest]
|
|
802
|
+
) -> tuple[tuple[str, ...], tuple[str, ...]]:
|
|
803
|
+
"""Signals of several projects in or beside this checkout, and their names.
|
|
804
|
+
|
|
805
|
+
Candidates are the directories holding manifests, when there are at least
|
|
806
|
+
two of them, a workspace list, or an apps/, packages/ or services/
|
|
807
|
+
directory; plus the build directories compose files name and the sibling
|
|
808
|
+
repositories the README links to, which are named and never read.
|
|
809
|
+
"""
|
|
810
|
+
directories = sorted(
|
|
811
|
+
{m.path.rpartition("/")[0] for m in manifests if "/" in m.path}
|
|
812
|
+
)
|
|
813
|
+
signals: list[str] = []
|
|
814
|
+
if len(directories) >= MONOREPO_MIN_DIRECTORIES:
|
|
815
|
+
signals.append(f"{len(directories)} directories with manifests")
|
|
816
|
+
signals.extend(
|
|
817
|
+
f"workspaces in {m.path}: {', '.join(m.workspaces)}"
|
|
818
|
+
for m in manifests
|
|
819
|
+
if m.workspaces
|
|
820
|
+
)
|
|
821
|
+
signals.extend(
|
|
822
|
+
f"{name}/ holds manifests"
|
|
823
|
+
for name in sorted({d.split("/")[0] for d in directories})
|
|
824
|
+
if name in MONOREPO_DIRECTORIES
|
|
825
|
+
)
|
|
826
|
+
candidates = list(directories) if signals else []
|
|
827
|
+
for path in files:
|
|
828
|
+
if "/" not in path and COMPOSE_FILE.match(path):
|
|
829
|
+
services, builds = _compose_services(root / path)
|
|
830
|
+
if services:
|
|
831
|
+
signals.append(f"{path} services: {', '.join(services)}")
|
|
832
|
+
candidates.extend(builds)
|
|
833
|
+
siblings = _readme_siblings(root, files)
|
|
834
|
+
if siblings:
|
|
835
|
+
signals.append(f"README links to sibling repositories: {', '.join(siblings)}")
|
|
836
|
+
candidates.extend(f"../{name}" for name in siblings)
|
|
837
|
+
return tuple(signals), tuple(dict.fromkeys(candidates))
|
|
838
|
+
|
|
839
|
+
|
|
840
|
+
def _compose_services(path: Path) -> tuple[list[str], list[str]]:
|
|
841
|
+
"""Service names, and the build directories inside the checkout that exist."""
|
|
842
|
+
try:
|
|
843
|
+
data = yaml.safe_load(_read(path))
|
|
844
|
+
except yaml.YAMLError:
|
|
845
|
+
data = None
|
|
846
|
+
services = data.get("services") if isinstance(data, dict) else None
|
|
847
|
+
if not isinstance(services, dict):
|
|
848
|
+
return [], []
|
|
849
|
+
base = path.parent.resolve()
|
|
850
|
+
builds = []
|
|
851
|
+
for service in services.values():
|
|
852
|
+
build = service.get("build") if isinstance(service, dict) else None
|
|
853
|
+
if isinstance(build, dict):
|
|
854
|
+
build = build.get("context")
|
|
855
|
+
directory = (base / build).resolve() if isinstance(build, str) else base
|
|
856
|
+
if directory != base and directory.is_dir() and directory.is_relative_to(base):
|
|
857
|
+
builds.append(directory.relative_to(base).as_posix())
|
|
858
|
+
return [str(name) for name in services], builds
|
|
859
|
+
|
|
860
|
+
|
|
861
|
+
def _readme_siblings(root: Path, files: list[str]) -> tuple[str, ...]:
|
|
862
|
+
names: list[str] = []
|
|
863
|
+
for path in files:
|
|
864
|
+
if "/" not in path and path.lower().startswith("readme"):
|
|
865
|
+
text = _read(root / path)
|
|
866
|
+
names += SIBLING_LINK.findall(text) + GIT_URL.findall(text)
|
|
867
|
+
# A README usually shows how to clone this very repository; that is not a
|
|
868
|
+
# sibling.
|
|
869
|
+
own = {root.name}
|
|
870
|
+
for line in _lines(_git(root, "remote", "-v")):
|
|
871
|
+
url = line.split()[1] if len(line.split()) > 1 else ""
|
|
872
|
+
own.add(url.rstrip("/").rpartition("/")[2].removesuffix(".git"))
|
|
873
|
+
return tuple(name for name in dict.fromkeys(names) if name not in own)
|
|
874
|
+
|
|
875
|
+
|
|
876
|
+
def render_markdown(profile: Profile) -> str:
|
|
877
|
+
"""The profile as short sections, one fact per line, evidence in parentheses."""
|
|
878
|
+
lines = [
|
|
879
|
+
f"# Profile of {Path(profile.root).name}",
|
|
880
|
+
"",
|
|
881
|
+
"Read-only facts about this checkout; each names where it came from.",
|
|
882
|
+
]
|
|
883
|
+
if profile.git_unavailable is not None:
|
|
884
|
+
lines += ["", f"Git facts are unavailable: {profile.git_unavailable}."]
|
|
885
|
+
for name, section in profile.sections.items():
|
|
886
|
+
lines += ["", f"## {name.replace('_', ' ').capitalize()}", ""]
|
|
887
|
+
for key, (value, evidence) in section.items():
|
|
888
|
+
label, show = LABELS[name][key]
|
|
889
|
+
if value is None:
|
|
890
|
+
text = "not found"
|
|
891
|
+
elif isinstance(value, tuple) and not value:
|
|
892
|
+
text = "none"
|
|
893
|
+
else:
|
|
894
|
+
text = show(value)
|
|
895
|
+
lines.append(f"- {label}: {text} ({evidence})")
|
|
896
|
+
return "\n".join(lines) + "\n"
|