catalogify 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- catalogify/__init__.py +11 -0
- catalogify/_scripts/__init__.py +0 -0
- catalogify/_scripts/okf-history.sh +157 -0
- catalogify/_scripts/okf-inventory.sh +300 -0
- catalogify/_scripts/validate_okf.py +356 -0
- catalogify/_skill_assets/SKILL.md +135 -0
- catalogify/_skill_assets/__init__.py +0 -0
- catalogify/_skill_assets/okf-config.template.yml +67 -0
- catalogify/_skill_assets/references/clarify.md +102 -0
- catalogify/_skill_assets/references/generate.md +384 -0
- catalogify/_skill_assets/references/update.md +133 -0
- catalogify/_skill_assets/references/validate.md +44 -0
- catalogify/cli.py +71 -0
- catalogify/installer.py +175 -0
- catalogify/runner.py +78 -0
- catalogify-0.5.0.dist-info/METADATA +322 -0
- catalogify-0.5.0.dist-info/RECORD +21 -0
- catalogify-0.5.0.dist-info/WHEEL +5 -0
- catalogify-0.5.0.dist-info/entry_points.txt +5 -0
- catalogify-0.5.0.dist-info/licenses/LICENSE +21 -0
- catalogify-0.5.0.dist-info/top_level.txt +1 -0
catalogify/__init__.py
ADDED
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
"""catalogify — turn a repository into a knowledge catalog an agent can afford to read.
|
|
2
|
+
|
|
3
|
+
Generates an Open Knowledge Format (OKF v0.1) bundle: cross-linked markdown
|
|
4
|
+
concepts with YAML frontmatter describing a codebase's services, modules, APIs,
|
|
5
|
+
data models and operations. Mines git history for the reasoning behind the code,
|
|
6
|
+
and parks what it cannot establish as an open question rather than inventing it.
|
|
7
|
+
|
|
8
|
+
OKF spec: https://github.com/GoogleCloudPlatform/knowledge-catalog/blob/main/okf/SPEC.md
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
__version__ = "0.5.0"
|
|
File without changes
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# okf-history.sh — per-path git history for the OKF enrichment agent.
|
|
3
|
+
#
|
|
4
|
+
# The inventory script gives repo-wide churn signals; this gives the *why*
|
|
5
|
+
# for ONE concept: how a file/dir came to be and how it has changed. The
|
|
6
|
+
# agent calls it while writing or refreshing a concept to ground the
|
|
7
|
+
# "purpose / invariants / gotchas / why it is shaped this way" narrative
|
|
8
|
+
# and to cite commits (OKF §8).
|
|
9
|
+
#
|
|
10
|
+
# Deliberately bounded and diff-free by default so it stays cheap and never
|
|
11
|
+
# leaks secrets from historical diffs. Use --patch only when you explicitly
|
|
12
|
+
# need the change content, and still scrub secrets before writing them into
|
|
13
|
+
# the bundle.
|
|
14
|
+
#
|
|
15
|
+
# Usage:
|
|
16
|
+
# okf-history.sh <path> [<path> ...] # summary for each path
|
|
17
|
+
# okf-history.sh --limit 30 <path> # cap recent commits (default 20)
|
|
18
|
+
# okf-history.sh --patch <path> # include truncated diffs (careful)
|
|
19
|
+
# okf-history.sh --json <path> # machine-readable output
|
|
20
|
+
#
|
|
21
|
+
# Output (text mode) per path:
|
|
22
|
+
# - creation commit (first time the path appears, follows renames)
|
|
23
|
+
# - total non-merge commit count touching the path
|
|
24
|
+
# - most recent non-merge commits (sha, ISO date, subject)
|
|
25
|
+
# - reverts/hotfixes touching the path (subjects matching revert/hotfix/fix)
|
|
26
|
+
|
|
27
|
+
set -euo pipefail
|
|
28
|
+
|
|
29
|
+
LIMIT="${OKF_HISTORY_LIMIT:-20}"
|
|
30
|
+
PATCH=0
|
|
31
|
+
JSON=0
|
|
32
|
+
PATHS=()
|
|
33
|
+
|
|
34
|
+
while [[ $# -gt 0 ]]; do
|
|
35
|
+
case "$1" in
|
|
36
|
+
--limit) LIMIT="${2:-20}"; shift 2 ;;
|
|
37
|
+
--patch) PATCH=1; shift ;;
|
|
38
|
+
--json) JSON=1; shift ;;
|
|
39
|
+
-h|--help)
|
|
40
|
+
sed -n '2,27p' "$0" | sed 's/^# \{0,1\}//'
|
|
41
|
+
exit 0 ;;
|
|
42
|
+
--) shift; while [[ $# -gt 0 ]]; do PATHS+=("$1"); shift; done ;;
|
|
43
|
+
-*) echo "unknown option: $1" >&2; exit 2 ;;
|
|
44
|
+
*) PATHS+=("$1"); shift ;;
|
|
45
|
+
esac
|
|
46
|
+
done
|
|
47
|
+
|
|
48
|
+
if [[ ${#PATHS[@]} -eq 0 ]]; then
|
|
49
|
+
echo "usage: okf-history.sh [--limit N] [--patch] [--json] <path> [<path> ...]" >&2
|
|
50
|
+
exit 2
|
|
51
|
+
fi
|
|
52
|
+
|
|
53
|
+
if ! git rev-parse --git-dir >/dev/null 2>&1; then
|
|
54
|
+
echo "okf-history: not a git repository — no history available." >&2
|
|
55
|
+
exit 0
|
|
56
|
+
fi
|
|
57
|
+
|
|
58
|
+
# --- per-path collectors ---------------------------------------------------
|
|
59
|
+
|
|
60
|
+
path_count() { git log --no-merges --follow --format='%h' -- "$1" 2>/dev/null | wc -l | tr -d ' '; }
|
|
61
|
+
path_created() { git log --no-merges --follow --diff-filter=A --format='%h%x09%cI%x09%an%x09%s' -- "$1" 2>/dev/null | tail -1; }
|
|
62
|
+
path_recent() { git log --no-merges --follow -n "$LIMIT" --format='%h%x09%cI%x09%s' -- "$1" 2>/dev/null; }
|
|
63
|
+
path_fixes() { git log --no-merges --follow -n 200 --format='%h%x09%cI%x09%s' -- "$1" 2>/dev/null \
|
|
64
|
+
| grep -iE $'\t''.*(revert|hotfix|regress|CVE-|security|race|deadlock|leak|corrupt|rollback)' || true; }
|
|
65
|
+
|
|
66
|
+
emit_text() {
|
|
67
|
+
local p="$1"
|
|
68
|
+
echo "=============================================================="
|
|
69
|
+
echo "PATH: $p"
|
|
70
|
+
echo "--------------------------------------------------------------"
|
|
71
|
+
local count created
|
|
72
|
+
count="$(path_count "$p")"
|
|
73
|
+
created="$(path_created "$p")"
|
|
74
|
+
echo "commits (non-merge, --follow): ${count:-0}"
|
|
75
|
+
if [[ -n "$created" ]]; then
|
|
76
|
+
IFS=$'\t' read -r c_sha c_date c_author c_subj <<<"$created"
|
|
77
|
+
echo "created: $c_sha $c_date by $c_author"
|
|
78
|
+
echo " $c_subj"
|
|
79
|
+
fi
|
|
80
|
+
echo
|
|
81
|
+
echo "recent commits (up to $LIMIT):"
|
|
82
|
+
local recent; recent="$(path_recent "$p")"
|
|
83
|
+
if [[ -n "$recent" ]]; then
|
|
84
|
+
printf '%s\n' "$recent" | while IFS=$'\t' read -r sha date subj; do
|
|
85
|
+
printf ' %s %s %s\n' "$sha" "$date" "$subj"
|
|
86
|
+
done
|
|
87
|
+
else
|
|
88
|
+
echo " (none)"
|
|
89
|
+
fi
|
|
90
|
+
local fixes; fixes="$(path_fixes "$p")"
|
|
91
|
+
if [[ -n "$fixes" ]]; then
|
|
92
|
+
echo
|
|
93
|
+
echo "reverts / hotfixes / risk-flagged commits (gotcha signals):"
|
|
94
|
+
printf '%s\n' "$fixes" | while IFS=$'\t' read -r sha date subj; do
|
|
95
|
+
printf ' %s %s %s\n' "$sha" "$date" "$subj"
|
|
96
|
+
done
|
|
97
|
+
fi
|
|
98
|
+
if [[ "$PATCH" -eq 1 ]]; then
|
|
99
|
+
echo
|
|
100
|
+
echo "recent diffs (truncated to 200 lines — SCRUB SECRETS before use):"
|
|
101
|
+
git log --no-merges --follow -n 3 -p --format='--- %h %cI %s' -- "$p" 2>/dev/null | head -200 || true
|
|
102
|
+
fi
|
|
103
|
+
echo
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
emit_json() {
|
|
107
|
+
# Build one JSON object per path via python for correct escaping.
|
|
108
|
+
python3 - "$LIMIT" "$@" <<'PYEOF'
|
|
109
|
+
import json, subprocess, sys
|
|
110
|
+
|
|
111
|
+
limit = sys.argv[1]
|
|
112
|
+
paths = sys.argv[2:]
|
|
113
|
+
|
|
114
|
+
def run(args):
|
|
115
|
+
try:
|
|
116
|
+
return subprocess.run(args, capture_output=True, text=True, check=False).stdout
|
|
117
|
+
except Exception:
|
|
118
|
+
return ""
|
|
119
|
+
|
|
120
|
+
def rows(out):
|
|
121
|
+
r = []
|
|
122
|
+
for ln in out.splitlines():
|
|
123
|
+
if not ln.strip():
|
|
124
|
+
continue
|
|
125
|
+
parts = ln.split("\t")
|
|
126
|
+
r.append(parts)
|
|
127
|
+
return r
|
|
128
|
+
|
|
129
|
+
result = []
|
|
130
|
+
for p in paths:
|
|
131
|
+
count = run(["git", "log", "--no-merges", "--follow", "--format=%h", "--", p])
|
|
132
|
+
created = rows(run(["git", "log", "--no-merges", "--follow", "--diff-filter=A",
|
|
133
|
+
"--format=%h\t%cI\t%an\t%s", "--", p]))
|
|
134
|
+
recent = rows(run(["git", "log", "--no-merges", "--follow", "-n", str(limit),
|
|
135
|
+
"--format=%h\t%cI\t%s", "--", p]))
|
|
136
|
+
obj = {
|
|
137
|
+
"path": p,
|
|
138
|
+
"commit_count": len([l for l in count.splitlines() if l.strip()]),
|
|
139
|
+
"created": (lambda c: {"sha": c[0], "date": c[1], "author": c[2],
|
|
140
|
+
"subject": "\t".join(c[3:])} if c and len(c) >= 4 else None)(
|
|
141
|
+
created[-1] if created else None),
|
|
142
|
+
"recent": [{"sha": r[0], "date": r[1], "subject": "\t".join(r[2:])}
|
|
143
|
+
for r in recent if len(r) >= 3],
|
|
144
|
+
}
|
|
145
|
+
result.append(obj)
|
|
146
|
+
|
|
147
|
+
print(json.dumps(result, indent=2))
|
|
148
|
+
PYEOF
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
if [[ "$JSON" -eq 1 ]]; then
|
|
152
|
+
emit_json "${PATHS[@]}"
|
|
153
|
+
else
|
|
154
|
+
for p in "${PATHS[@]}"; do
|
|
155
|
+
emit_text "$p"
|
|
156
|
+
done
|
|
157
|
+
fi
|
|
@@ -0,0 +1,300 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# okf-inventory.sh — deterministic repository inventory for the OKF
|
|
3
|
+
# enrichment agent. Emits JSON (default: a per-repo temp path, printed to
|
|
4
|
+
# stdout) and a short human-readable summary to stdout.
|
|
5
|
+
#
|
|
6
|
+
# The agent uses this as the factual substrate for concept planning so it
|
|
7
|
+
# doesn't have to guess the repo layout.
|
|
8
|
+
#
|
|
9
|
+
# Usage: okf-inventory.sh [output_path] [--config <okf-config.yml>]
|
|
10
|
+
|
|
11
|
+
set -euo pipefail
|
|
12
|
+
|
|
13
|
+
OUT=""
|
|
14
|
+
CONFIG=""
|
|
15
|
+
while [[ $# -gt 0 ]]; do
|
|
16
|
+
case "$1" in
|
|
17
|
+
--config)
|
|
18
|
+
CONFIG="${2:-}"
|
|
19
|
+
shift 2
|
|
20
|
+
;;
|
|
21
|
+
*)
|
|
22
|
+
OUT="$1"
|
|
23
|
+
shift
|
|
24
|
+
;;
|
|
25
|
+
esac
|
|
26
|
+
done
|
|
27
|
+
|
|
28
|
+
ROOT="$(git rev-parse --show-toplevel 2>/dev/null || pwd)"
|
|
29
|
+
cd "$ROOT"
|
|
30
|
+
|
|
31
|
+
OUT="${OUT:-${TMPDIR:-/tmp}/okf-inventory-$(basename "$ROOT")-$$.json}"
|
|
32
|
+
CAP="${OKF_INVENTORY_CAP:-150}"
|
|
33
|
+
|
|
34
|
+
# Default excludes, aligned with okf-config.template.yml's `exclude:` list.
|
|
35
|
+
DEFAULT_EXCLUDE='^(node_modules|\.git|dist|build|vendor|\.specify|specs)/|(^|/)[^/]*\.min\.js$'
|
|
36
|
+
|
|
37
|
+
# --- helpers ---------------------------------------------------------------
|
|
38
|
+
|
|
39
|
+
# Reads `okf.exclude` glob list from a config YAML (best-effort, no PyYAML
|
|
40
|
+
# dependency) and prints one alternation-ready regex fragment per line.
|
|
41
|
+
# Silent no-op (empty output) if $CONFIG is unset or unreadable.
|
|
42
|
+
config_exclude_regex() {
|
|
43
|
+
if [[ -z "$CONFIG" || ! -f "$CONFIG" ]]; then
|
|
44
|
+
return 0
|
|
45
|
+
fi
|
|
46
|
+
python3 - "$CONFIG" <<'PYEOF'
|
|
47
|
+
import re, sys
|
|
48
|
+
|
|
49
|
+
path = sys.argv[1]
|
|
50
|
+
try:
|
|
51
|
+
with open(path, encoding="utf-8") as f:
|
|
52
|
+
lines = f.readlines()
|
|
53
|
+
except OSError:
|
|
54
|
+
sys.exit(0)
|
|
55
|
+
|
|
56
|
+
in_exclude = False
|
|
57
|
+
globs = []
|
|
58
|
+
for line in lines:
|
|
59
|
+
stripped = line.strip()
|
|
60
|
+
if re.match(r"^exclude\s*:\s*$", stripped):
|
|
61
|
+
in_exclude = True
|
|
62
|
+
continue
|
|
63
|
+
if in_exclude:
|
|
64
|
+
m = re.match(r"^-\s*[\"']?([^\"'#]+)[\"']?\s*(#.*)?$", stripped)
|
|
65
|
+
if m:
|
|
66
|
+
globs.append(m.group(1).strip())
|
|
67
|
+
continue
|
|
68
|
+
if stripped and not stripped.startswith("#"):
|
|
69
|
+
# A non-list, non-comment line ends the exclude block.
|
|
70
|
+
in_exclude = False
|
|
71
|
+
|
|
72
|
+
def glob_to_regex(g):
|
|
73
|
+
# Minimal glob->regex: ** -> .*, * -> [^/]*, escape the rest.
|
|
74
|
+
out = []
|
|
75
|
+
i = 0
|
|
76
|
+
while i < len(g):
|
|
77
|
+
if g[i:i+2] == "**":
|
|
78
|
+
out.append(".*")
|
|
79
|
+
i += 2
|
|
80
|
+
elif g[i] == "*":
|
|
81
|
+
out.append("[^/]*")
|
|
82
|
+
i += 1
|
|
83
|
+
else:
|
|
84
|
+
out.append(re.escape(g[i]))
|
|
85
|
+
i += 1
|
|
86
|
+
return "^" + "".join(out) + "$"
|
|
87
|
+
|
|
88
|
+
for g in globs:
|
|
89
|
+
print(glob_to_regex(g))
|
|
90
|
+
PYEOF
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
EXCLUDE_FRAGMENTS="$(config_exclude_regex || true)"
|
|
94
|
+
if [[ -n "$EXCLUDE_FRAGMENTS" ]]; then
|
|
95
|
+
CONFIG_EXCLUDE_RE="$(printf '%s\n' "$EXCLUDE_FRAGMENTS" | paste -sd '|' -)"
|
|
96
|
+
else
|
|
97
|
+
CONFIG_EXCLUDE_RE=""
|
|
98
|
+
fi
|
|
99
|
+
|
|
100
|
+
list_matching() {
|
|
101
|
+
# $1 = regex on path; respects .gitignore via git ls-files when available.
|
|
102
|
+
# Always applies the default exclude list, plus config excludes if present.
|
|
103
|
+
local raw
|
|
104
|
+
if git rev-parse --git-dir >/dev/null 2>&1; then
|
|
105
|
+
raw="$(git ls-files || true)"
|
|
106
|
+
else
|
|
107
|
+
raw="$(find . -type f | sed 's|^\./||' || true)"
|
|
108
|
+
fi
|
|
109
|
+
printf '%s\n' "$raw" \
|
|
110
|
+
| grep -vE "$DEFAULT_EXCLUDE" \
|
|
111
|
+
| { if [[ -n "$CONFIG_EXCLUDE_RE" ]]; then grep -vE "$CONFIG_EXCLUDE_RE"; else cat; fi; } \
|
|
112
|
+
| grep -E "$1" || true
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
to_json_array() { python3 -c '
|
|
116
|
+
import json, sys
|
|
117
|
+
print(json.dumps([l for l in sys.stdin.read().splitlines() if l.strip()]))'; }
|
|
118
|
+
|
|
119
|
+
# --- collect ---------------------------------------------------------------
|
|
120
|
+
# Whether this is a git repo at all. Everything git-derived below is empty
|
|
121
|
+
# when it is not, and consumers should branch on this rather than on an
|
|
122
|
+
# empty string that could equally mean "no remote configured".
|
|
123
|
+
# Python literals: this value is interpolated into the emitter heredoc below.
|
|
124
|
+
if git rev-parse --git-dir >/dev/null 2>&1; then IS_GIT_REPO=True; else IS_GIT_REPO=False; fi
|
|
125
|
+
REMOTE="$(git remote get-url origin 2>/dev/null || echo "")"
|
|
126
|
+
BRANCH="$(git symbolic-ref --short HEAD 2>/dev/null || echo "")"
|
|
127
|
+
HEAD_SHA="$(git rev-parse --short HEAD 2>/dev/null || echo "")"
|
|
128
|
+
|
|
129
|
+
# --- git history signals ---------------------------------------------------
|
|
130
|
+
# Bounded, deterministic history to inform *significance* (churn) and the
|
|
131
|
+
# *why* (recent commit subjects). All heavy scans are capped so this stays
|
|
132
|
+
# cheap on large repos. Skipped cleanly when not a git repo.
|
|
133
|
+
HIST_COMMITS="${OKF_HISTORY_COMMITS:-2000}" # how many recent commits to scan for churn
|
|
134
|
+
HIST_RECENT="${OKF_HISTORY_RECENT:-20}" # how many recent subjects to surface
|
|
135
|
+
CHURN_TOP="${OKF_CHURN_TOP:-30}" # top-N hottest files to report
|
|
136
|
+
|
|
137
|
+
# Take the first N lines WITHOUT closing the pipe early.
|
|
138
|
+
#
|
|
139
|
+
# `head -N` exits as soon as it has N lines, which sends SIGPIPE to whatever is
|
|
140
|
+
# still writing upstream. Under `set -o pipefail` that surfaces as exit 141 and
|
|
141
|
+
# `set -e` then aborts the whole script. It only bites on repositories large
|
|
142
|
+
# enough that the upstream producer is still writing when head leaves — which is
|
|
143
|
+
# exactly the case this tool is for (first reproduced on kubernetes/kubernetes:
|
|
144
|
+
# 500k LOC, 140k commits). awk drains its input to EOF, so nothing upstream ever
|
|
145
|
+
# sees a closed pipe.
|
|
146
|
+
take() {
|
|
147
|
+
awk -v n="$1" 'NR<=n'
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
apply_excludes() {
|
|
151
|
+
grep -vE "$DEFAULT_EXCLUDE" \
|
|
152
|
+
| { if [[ -n "$CONFIG_EXCLUDE_RE" ]]; then grep -vE "$CONFIG_EXCLUDE_RE"; else cat; fi; }
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
CHURN=""
|
|
156
|
+
RECENT_COMMITS=""
|
|
157
|
+
COMMITS_SCANNED="0"
|
|
158
|
+
if git rev-parse --git-dir >/dev/null 2>&1; then
|
|
159
|
+
# Churn: commit-count per file across the last $HIST_COMMITS non-merge
|
|
160
|
+
# commits. A proxy for "which files carry the most history/gotchas".
|
|
161
|
+
CHURN="$(git log --no-merges -n "$HIST_COMMITS" --pretty=format: --name-only 2>/dev/null \
|
|
162
|
+
| sed '/^$/d' \
|
|
163
|
+
| apply_excludes \
|
|
164
|
+
| sort | uniq -c | sort -rn | take "$CHURN_TOP" \
|
|
165
|
+
| awk '{c=$1; $1=""; sub(/^ /,""); printf "%s\t%s\n", c, $0}')"
|
|
166
|
+
# Recent commit subjects (no merges) — cheap signal for the "why".
|
|
167
|
+
RECENT_COMMITS="$(git log --no-merges -n "$HIST_RECENT" --pretty=format:'%h%x09%cI%x09%s' 2>/dev/null || true)"
|
|
168
|
+
COMMITS_SCANNED="$(git rev-list --no-merges --count -n "$HIST_COMMITS" HEAD 2>/dev/null || echo 0)"
|
|
169
|
+
fi
|
|
170
|
+
|
|
171
|
+
ALL_FILES="$(list_matching '.')"
|
|
172
|
+
FILE_COUNT="$(printf '%s\n' "$ALL_FILES" | sed '/^$/d' | wc -l | tr -d ' ')"
|
|
173
|
+
|
|
174
|
+
# Language histogram by extension (top 15)
|
|
175
|
+
LANG_HIST="$(printf '%s\n' "$ALL_FILES" | sed '/^$/d' \
|
|
176
|
+
| awk -F. 'NF>1 {print $NF}' | sort | uniq -c | sort -rn | take 15 \
|
|
177
|
+
| awk '{printf "%s:%s\n", $2, $1}')"
|
|
178
|
+
|
|
179
|
+
# Each category: raw match count vs capped count, so the caller can detect
|
|
180
|
+
# truncation. take "$CAP" applied uniformly across every category.
|
|
181
|
+
raw_count() { list_matching "$1" | sed '/^$/d' | wc -l | tr -d ' '; }
|
|
182
|
+
|
|
183
|
+
MANIFESTS_RE='(^|/)(package\.json|pyproject\.toml|setup\.py|requirements[^/]*\.txt|go\.mod|Cargo\.toml|pom\.xml|build\.gradle(\.kts)?|Gemfile|composer\.json|\.csproj)$'
|
|
184
|
+
ENTRYPOINTS_RE='(^|/)(main|app|index|server|cli|manage|__main__)\.(py|js|ts|go|rs|rb|java|cs)$'
|
|
185
|
+
API_DEFS_RE='\.(proto|avsc|graphql|gql)$|(^|/)(openapi|swagger)[^/]*\.(ya?ml|json)$'
|
|
186
|
+
ROUTES_RE='(rout|controller|endpoint|handler|view|resource)s?[^/]*\.(py|js|ts|go|rb|java|cs|php)$'
|
|
187
|
+
DB_FILES_RE='(^|/)(migrations?|alembic|db/migrate)/|(model|schema|entit)[^/]*\.(py|js|ts|go|rb|java|cs|sql|prisma)$'
|
|
188
|
+
OPS_FILES_RE='(^|/)(Dockerfile[^/]*|docker-compose[^/]*\.ya?ml|Makefile|Procfile|[^/]*\.tf|helm/.*|k8s/.*\.ya?ml|\.github/workflows/.*\.ya?ml|\.gitlab-ci\.yml|Jenkinsfile|cloudbuild\.ya?ml)$'
|
|
189
|
+
DOCS_RE='(^|/)(README[^/]*|CONTRIBUTING[^/]*|CHANGELOG[^/]*|ARCHITECTURE[^/]*|docs?/.*)\.(md|rst|txt)$'
|
|
190
|
+
CONFIGS_RE='(^|/)(config|settings|conf)[^/]*\.(py|js|ts|ya?ml|json|toml|ini|env\.example)$'
|
|
191
|
+
# ADR / design-decision docs (rationale goldmine) — matched separately from
|
|
192
|
+
# generic docs so the planner can seed `references/` and Design Decision concepts.
|
|
193
|
+
ADR_RE='(^|/)(adr|adrs|decisions?|rfcs?)/|(^|/)(ADR|RFC)[-_0-9]'
|
|
194
|
+
|
|
195
|
+
MANIFESTS="$(list_matching "$MANIFESTS_RE" | take "$CAP")"
|
|
196
|
+
ENTRYPOINTS="$(list_matching "$ENTRYPOINTS_RE" | take "$CAP")"
|
|
197
|
+
API_DEFS="$(list_matching "$API_DEFS_RE" | take "$CAP")"
|
|
198
|
+
ROUTES="$(list_matching "$ROUTES_RE" | take "$CAP")"
|
|
199
|
+
DB_FILES="$(list_matching "$DB_FILES_RE" | take "$CAP")"
|
|
200
|
+
OPS_FILES="$(list_matching "$OPS_FILES_RE" | take "$CAP")"
|
|
201
|
+
DOCS="$(list_matching "$DOCS_RE" | take "$CAP")"
|
|
202
|
+
CONFIGS="$(list_matching "$CONFIGS_RE" | take "$CAP")"
|
|
203
|
+
ADR_DOCS="$(list_matching "$ADR_RE" | take "$CAP")"
|
|
204
|
+
|
|
205
|
+
MANIFESTS_RAW="$(raw_count "$MANIFESTS_RE")"
|
|
206
|
+
ENTRYPOINTS_RAW="$(raw_count "$ENTRYPOINTS_RE")"
|
|
207
|
+
API_DEFS_RAW="$(raw_count "$API_DEFS_RE")"
|
|
208
|
+
ROUTES_RAW="$(raw_count "$ROUTES_RE")"
|
|
209
|
+
DB_FILES_RAW="$(raw_count "$DB_FILES_RE")"
|
|
210
|
+
OPS_FILES_RAW="$(raw_count "$OPS_FILES_RE")"
|
|
211
|
+
DOCS_RAW="$(raw_count "$DOCS_RE")"
|
|
212
|
+
CONFIGS_RAW="$(raw_count "$CONFIGS_RE")"
|
|
213
|
+
ADR_DOCS_RAW="$(raw_count "$ADR_RE")"
|
|
214
|
+
|
|
215
|
+
# Top-level directory sizes (proxy for module significance)
|
|
216
|
+
TOPDIRS="$(printf '%s\n' "$ALL_FILES" | sed '/^$/d' | awk -F/ 'NF>1 {print $1}' | sort | uniq -c | sort -rn | take 20 | awk '{printf "%s:%s\n", $2, $1}')"
|
|
217
|
+
|
|
218
|
+
# --- emit ------------------------------------------------------------------
|
|
219
|
+
python3 - "$OUT" "$CAP" <<PYEOF
|
|
220
|
+
import json, sys, os
|
|
221
|
+
|
|
222
|
+
def lines(s): return [l for l in s.splitlines() if l.strip()]
|
|
223
|
+
|
|
224
|
+
cap = int(sys.argv[2])
|
|
225
|
+
|
|
226
|
+
def category(name_lines, raw_count):
|
|
227
|
+
items = lines(name_lines)
|
|
228
|
+
return {"items": items, "truncated": raw_count > len(items)}
|
|
229
|
+
|
|
230
|
+
categories = {
|
|
231
|
+
"dependency_manifests": category("""$MANIFESTS""", int("$MANIFESTS_RAW" or 0)),
|
|
232
|
+
"entrypoints": category("""$ENTRYPOINTS""", int("$ENTRYPOINTS_RAW" or 0)),
|
|
233
|
+
"api_definitions": category("""$API_DEFS""", int("$API_DEFS_RAW" or 0)),
|
|
234
|
+
"route_like_files": category("""$ROUTES""", int("$ROUTES_RAW" or 0)),
|
|
235
|
+
"data_layer_files": category("""$DB_FILES""", int("$DB_FILES_RAW" or 0)),
|
|
236
|
+
"ops_files": category("""$OPS_FILES""", int("$OPS_FILES_RAW" or 0)),
|
|
237
|
+
"docs": category("""$DOCS""", int("$DOCS_RAW" or 0)),
|
|
238
|
+
"config_files": category("""$CONFIGS""", int("$CONFIGS_RAW" or 0)),
|
|
239
|
+
"adr_docs": category("""$ADR_DOCS""", int("$ADR_DOCS_RAW" or 0)),
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
def churn_list(raw):
|
|
243
|
+
out = []
|
|
244
|
+
for ln in lines(raw):
|
|
245
|
+
parts = ln.split("\t", 1)
|
|
246
|
+
if len(parts) == 2:
|
|
247
|
+
count, path = parts
|
|
248
|
+
try:
|
|
249
|
+
out.append({"file": path.strip(), "commits": int(count.strip())})
|
|
250
|
+
except ValueError:
|
|
251
|
+
pass
|
|
252
|
+
return out
|
|
253
|
+
|
|
254
|
+
def recent_list(raw):
|
|
255
|
+
out = []
|
|
256
|
+
for ln in lines(raw):
|
|
257
|
+
parts = ln.split("\t")
|
|
258
|
+
if len(parts) >= 3:
|
|
259
|
+
out.append({"sha": parts[0], "date": parts[1], "subject": "\t".join(parts[2:])})
|
|
260
|
+
return out
|
|
261
|
+
|
|
262
|
+
churn = churn_list("""$CHURN""")
|
|
263
|
+
|
|
264
|
+
inv = {
|
|
265
|
+
"root": os.getcwd(),
|
|
266
|
+
"git": {
|
|
267
|
+
"is_git_repo": $IS_GIT_REPO,
|
|
268
|
+
"remote": """$REMOTE""",
|
|
269
|
+
"branch": """$BRANCH""",
|
|
270
|
+
"head": """$HEAD_SHA""",
|
|
271
|
+
"history": {
|
|
272
|
+
"commits_scanned": int("$COMMITS_SCANNED" or 0),
|
|
273
|
+
"churn": churn, # hottest files (commit count desc)
|
|
274
|
+
"recent_commits": recent_list("""$RECENT_COMMITS"""),
|
|
275
|
+
"churn_truncated": len(churn) >= int("$CHURN_TOP" or 0) > 0,
|
|
276
|
+
},
|
|
277
|
+
},
|
|
278
|
+
"file_count": int("$FILE_COUNT" or 0),
|
|
279
|
+
"cap": cap,
|
|
280
|
+
"language_histogram": dict(x.split(":") for x in lines("""$LANG_HIST""")),
|
|
281
|
+
"top_level_dirs": dict(x.split(":") for x in lines("""$TOPDIRS""")),
|
|
282
|
+
**{k: v["items"] for k, v in categories.items()},
|
|
283
|
+
"truncated": {k: v["truncated"] for k, v in categories.items() if v["truncated"]},
|
|
284
|
+
}
|
|
285
|
+
with open(sys.argv[1], "w") as f:
|
|
286
|
+
json.dump(inv, f, indent=2)
|
|
287
|
+
|
|
288
|
+
print(f"Inventory written to {sys.argv[1]}")
|
|
289
|
+
if inv["git"]["is_git_repo"]:
|
|
290
|
+
print(f" files: {inv['file_count']} head: {inv['git']['head']} branch: {inv['git']['branch']}")
|
|
291
|
+
else:
|
|
292
|
+
print(f" files: {inv['file_count']} (not a git repository - no history signals)")
|
|
293
|
+
hist = inv["git"]["history"]
|
|
294
|
+
print(f" git history: {hist['commits_scanned']} commits scanned, "
|
|
295
|
+
f"{len(hist['churn'])} hot files, {len(hist['recent_commits'])} recent subjects")
|
|
296
|
+
for k in ("dependency_manifests","entrypoints","api_definitions","route_like_files",
|
|
297
|
+
"data_layer_files","ops_files","docs","config_files","adr_docs"):
|
|
298
|
+
flag = " (truncated)" if k in inv["truncated"] else ""
|
|
299
|
+
print(f" {k}: {len(inv[k])}{flag}")
|
|
300
|
+
PYEOF
|