knoblog 0.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- knoblog/__init__.py +2 -0
- knoblog/__main__.py +3 -0
- knoblog/analyze.py +403 -0
- knoblog/cli.py +240 -0
- knoblog/config.py +133 -0
- knoblog/extractors.py +377 -0
- knoblog/gitio.py +161 -0
- knoblog/posttext.py +458 -0
- knoblog/presets/_template.yml +42 -0
- knoblog/presets/ardupilot.yml +12 -0
- knoblog/presets/generic.yml +6 -0
- knoblog/presets/px4.yml +15 -0
- knoblog/presets/ros2.yml +13 -0
- knoblog/presets/x-algorithm.yml +43 -0
- knoblog/render.py +367 -0
- knoblog/static/style.css +38 -0
- knoblog/summarize.py +270 -0
- knoblog/yamlmini.py +248 -0
- knoblog-0.1.1.dist-info/METADATA +171 -0
- knoblog-0.1.1.dist-info/RECORD +24 -0
- knoblog-0.1.1.dist-info/WHEEL +5 -0
- knoblog-0.1.1.dist-info/entry_points.txt +2 -0
- knoblog-0.1.1.dist-info/licenses/LICENSE +21 -0
- knoblog-0.1.1.dist-info/top_level.txt +1 -0
knoblog/__init__.py
ADDED
knoblog/__main__.py
ADDED
knoblog/analyze.py
ADDED
|
@@ -0,0 +1,403 @@
|
|
|
1
|
+
"""Deterministic analysis of one commit: which watched parameters changed, old -> new.
|
|
2
|
+
|
|
3
|
+
Everything is derived mechanically from git; nothing is guessed.
|
|
4
|
+
"""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import os
|
|
8
|
+
import re
|
|
9
|
+
from collections import Counter, defaultdict
|
|
10
|
+
|
|
11
|
+
from . import config as C
|
|
12
|
+
from .yamlmini import unquote
|
|
13
|
+
from .extractors import KEYED, LITERALISH, auto_extractors, infer_type, parse_num, run_extractor
|
|
14
|
+
from .gitio import EMPTY_TREE, BlobReader, CommitMeta, changed_files, run, unified_diff
|
|
15
|
+
|
|
16
|
+
FAMILY = {"px4_param_define": "px4", "px4_module_yaml": "px4"} # same parameter namespace, different file formats
|
|
17
|
+
CONST_LIKE = {"rust_const", "python_const", "c_const", "jvm_const"} # only diffed in modified files; added/removed only for UPPER names
|
|
18
|
+
DOC_EXTS = {".md", ".txt", ".rst"}
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
# ---------------------------------------------------------------------------
|
|
22
|
+
# File-level noise classification
|
|
23
|
+
# ---------------------------------------------------------------------------
|
|
24
|
+
def _strip_comments(path: str, text: str) -> str:
|
|
25
|
+
ext = os.path.splitext(path)[1]
|
|
26
|
+
if ext in {".py", ".toml", ".yml", ".yaml", ".bazel", ".bzl", ".cfg", ".ini"} or path.endswith("BUILD"):
|
|
27
|
+
text = re.sub(r"(?m)^\s*#.*$", "", text)
|
|
28
|
+
text = re.sub(r"(?m)\s+#[^\n'\"]*$", "", text)
|
|
29
|
+
if ext == ".py":
|
|
30
|
+
text = re.sub(r'(?s)"""(.*?)"""', '""""""', text)
|
|
31
|
+
else:
|
|
32
|
+
text = re.sub(r"(?s)/\*.*?\*/", "", text)
|
|
33
|
+
text = re.sub(r"(?m)(^|\s)//.*$", r"\1", text)
|
|
34
|
+
return text
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def classify_file_change(path: str, old: str | None, new: str | None, noise: list[re.Pattern]) -> str:
|
|
38
|
+
"""'noise' (only configured noise lines changed), 'whitespace', 'comments' or 'code'."""
|
|
39
|
+
if old is None or new is None:
|
|
40
|
+
return "code"
|
|
41
|
+
strip = lambda t: "\n".join(l for l in t.splitlines() if not any(rx.search(l) for rx in noise))
|
|
42
|
+
no_ws = lambda t: re.sub(r"\s+", "", t)
|
|
43
|
+
o, n = strip(old), strip(new)
|
|
44
|
+
if o == n:
|
|
45
|
+
return "noise"
|
|
46
|
+
if no_ws(o) == no_ws(n):
|
|
47
|
+
return "whitespace"
|
|
48
|
+
if no_ws(_strip_comments(path, o)) == no_ws(_strip_comments(path, n)):
|
|
49
|
+
return "comments"
|
|
50
|
+
return "code"
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
# ---------------------------------------------------------------------------
|
|
54
|
+
# Optional: generic numeric tweaks found by pairing -/+ lines in the same hunk
|
|
55
|
+
# ---------------------------------------------------------------------------
|
|
56
|
+
NUM_TOKEN = re.compile(r"(?<![\w.])[-+]?(?:\d[\d_]*\.?\d*(?:[eE][-+]?\d+)?|\.\d+)(?![\w])")
|
|
57
|
+
BOOL_TOKEN = re.compile(r"\b(true|false|True|False)\b")
|
|
58
|
+
TEST_PATH = re.compile(r"(^|/)(tests?|testing|testdata)/|_test\.\w+$|(^|/)test_[^/]+$|Test\.\w+$")
|
|
59
|
+
TEST_LINE = re.compile(r"\bassert|\bexpect\(|\.apply\(&|should|#\[test\]")
|
|
60
|
+
COMMENT_LINE = re.compile(r"^\s*(//|#|\*|/\*|--)")
|
|
61
|
+
LOCK_FILE = re.compile(r"(^|/)(Cargo\.lock|package-lock\.json|yarn\.lock|poetry\.lock|uv\.lock|[^/]*\.lock)$")
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _skeleton(line: str) -> str:
|
|
65
|
+
return re.sub(r"\s+", "", BOOL_TOKEN.sub("?", NUM_TOKEN.sub("#", line)))
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _label_for(line: str) -> str:
|
|
69
|
+
for rx in (r"\s*(?:pub\s+)?(?:const\s+|static\s+|val\s+|var\s+|let\s+)?([A-Za-z_][\w.]*)\s*(?::[^=]*)?=",
|
|
70
|
+
r"\s*([A-Za-z_][\w.]*)\s*:", r"\s*\"([^\"]+)\"\s*[:=]"):
|
|
71
|
+
m = re.match(rx, line)
|
|
72
|
+
if m:
|
|
73
|
+
return m.group(1)
|
|
74
|
+
return ""
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def numeric_tweaks(diff_text: str, noise: list[re.Pattern]) -> list[dict]:
|
|
78
|
+
tweaks: list[dict] = []
|
|
79
|
+
cur_file, minus, plus = None, [], []
|
|
80
|
+
|
|
81
|
+
def flush():
|
|
82
|
+
if not cur_file or not minus or not plus:
|
|
83
|
+
return
|
|
84
|
+
if TEST_PATH.search(cur_file) or os.path.splitext(cur_file)[1] in DOC_EXTS or LOCK_FILE.search(cur_file):
|
|
85
|
+
return
|
|
86
|
+
used = set()
|
|
87
|
+
for m in minus:
|
|
88
|
+
if TEST_LINE.search(m) or COMMENT_LINE.match(m) or any(rx.search(m) for rx in noise):
|
|
89
|
+
continue
|
|
90
|
+
sk = _skeleton(m)
|
|
91
|
+
if len(re.sub(r"[^A-Za-z]", "", sk)) < 2:
|
|
92
|
+
continue
|
|
93
|
+
for idx, p in enumerate(plus):
|
|
94
|
+
if idx in used or _skeleton(p) != sk or p.strip() == m.strip():
|
|
95
|
+
continue
|
|
96
|
+
ov = NUM_TOKEN.findall(m) + BOOL_TOKEN.findall(m)
|
|
97
|
+
nv = NUM_TOKEN.findall(p) + BOOL_TOKEN.findall(p)
|
|
98
|
+
used.add(idx)
|
|
99
|
+
tweaks.append({"file": cur_file, "label": _label_for(p) or _label_for(m), "old_line": m.strip()[:200],
|
|
100
|
+
"new_line": p.strip()[:200],
|
|
101
|
+
"changes": [[o, n] for o, n in zip(ov, nv) if o != n] if len(ov) == len(nv) else []})
|
|
102
|
+
break
|
|
103
|
+
|
|
104
|
+
for line in diff_text.splitlines():
|
|
105
|
+
if line.startswith("diff --git"):
|
|
106
|
+
flush(); minus, plus, cur_file = [], [], None
|
|
107
|
+
elif line.startswith("+++ "):
|
|
108
|
+
cur_file = line[6:] if line.startswith("+++ b/") else None
|
|
109
|
+
elif line.startswith("--- "):
|
|
110
|
+
continue
|
|
111
|
+
elif line.startswith("@@"):
|
|
112
|
+
flush(); minus, plus = [], []
|
|
113
|
+
elif line.startswith("-"):
|
|
114
|
+
minus.append(line[1:])
|
|
115
|
+
elif line.startswith("+"):
|
|
116
|
+
plus.append(line[1:])
|
|
117
|
+
flush()
|
|
118
|
+
return tweaks
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
# ---------------------------------------------------------------------------
|
|
122
|
+
# Helpers
|
|
123
|
+
# ---------------------------------------------------------------------------
|
|
124
|
+
def extractors_for(path: str, text: str, cfg: dict) -> list[tuple[str, dict]]:
|
|
125
|
+
out: list[tuple[str, dict]] = []
|
|
126
|
+
for spec in cfg["extractors"]:
|
|
127
|
+
if spec.get("paths") and not C.match_any(path, spec["paths"]):
|
|
128
|
+
continue
|
|
129
|
+
if spec.get("exclude") and C.match_any(path, spec["exclude"]):
|
|
130
|
+
continue
|
|
131
|
+
names = auto_extractors(path, text) if spec["use"] == "auto" else [spec["use"]]
|
|
132
|
+
for n in names:
|
|
133
|
+
if n not in [x[0] for x in out]:
|
|
134
|
+
out.append((n, spec.get("options") or {}))
|
|
135
|
+
return out
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
_C_NUMEXPR = re.compile(r"(?:[-+*/()\s<>|&~]|0[xX][0-9a-fA-F]+[uUlL]*|\d[\d_.]*(?:[eE][-+]?\d+)?[fFuUlL]*)+")
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _const_literal(v: str) -> bool:
|
|
142
|
+
v = str(v).strip()
|
|
143
|
+
return bool(LITERALISH.fullmatch(v) or (_C_NUMEXPR.fullmatch(v) and re.search(r"\d", v)))
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def change_metrics(old, new) -> dict:
|
|
147
|
+
o, n = parse_num(old), parse_num(new)
|
|
148
|
+
if o is None or n is None:
|
|
149
|
+
return {"pct": None, "ratio": None}
|
|
150
|
+
pct = None if o == 0 else round((n - o) / abs(o) * 100, 4)
|
|
151
|
+
ratio = None if o == 0 or (o < 0) != (n < 0) else round(n / o, 6)
|
|
152
|
+
return {"pct": pct, "ratio": ratio}
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def list_items(v) -> list[str] | None:
|
|
156
|
+
"""Items of a flat list value "[a, b]" (raw strings), or None."""
|
|
157
|
+
v = str(v).strip() if v is not None else ""
|
|
158
|
+
if not (v.startswith("[") and v.endswith("]")):
|
|
159
|
+
return None
|
|
160
|
+
from .yamlmini import _split_flow
|
|
161
|
+
return _split_flow(v[1:-1])
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def same_value(a: str, b: str) -> bool:
|
|
165
|
+
if a == b:
|
|
166
|
+
return True
|
|
167
|
+
if str(a).lower() == str(b).lower() and str(a).lower() in ("true", "false"):
|
|
168
|
+
return True # YAML True == true
|
|
169
|
+
la, lb = list_items(a), list_items(b)
|
|
170
|
+
if la is not None and lb is not None:
|
|
171
|
+
return len(la) == len(lb) and all(same_value(unquote(x), unquote(y)) for x, y in zip(la, lb))
|
|
172
|
+
x, y = parse_num(a), parse_num(b)
|
|
173
|
+
return x is not None and y is not None and x == y and infer_type(a) == infer_type(b) and not ("." in a) ^ ("." in b)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def classify_component(path: str, cfg: dict) -> str | None:
|
|
177
|
+
comp = cfg["components"]
|
|
178
|
+
if comp.get("exclude") and re.search(comp["exclude"], path):
|
|
179
|
+
return None
|
|
180
|
+
for r in comp["rules"]:
|
|
181
|
+
if re.search(r["pattern"], path):
|
|
182
|
+
return r["kind"]
|
|
183
|
+
return None
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
GENERIC_NAMES = {"filter", "task_filter", "filters", "scorer", "scorers", "source", "sources", "rules", "rule"}
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def component_name(path: str) -> str:
|
|
190
|
+
base = os.path.splitext(os.path.basename(path))[0]
|
|
191
|
+
if base.lower() in GENERIC_NAMES:
|
|
192
|
+
return f"{os.path.basename(os.path.dirname(path))}/{base}"
|
|
193
|
+
return base
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def top_module(path: str) -> str:
|
|
197
|
+
return path.split("/", 1)[0] if "/" in path else "(repo root)"
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
# ---------------------------------------------------------------------------
|
|
201
|
+
# Main entry point
|
|
202
|
+
# ---------------------------------------------------------------------------
|
|
203
|
+
def analyze_commit(repo: str, meta: CommitMeta, cfg: dict, repo_url: str) -> dict:
|
|
204
|
+
parent = meta.parents[0] if meta.parents else EMPTY_TREE
|
|
205
|
+
noise = [re.compile(p) for p in cfg["noise"]["line_patterns"]]
|
|
206
|
+
all_files = changed_files(repo, parent, meta.sha)
|
|
207
|
+
files = [f for f in all_files if C.watched(f.path, cfg) and not C.match_any(f.path, cfg["noise"]["ignore_paths"])]
|
|
208
|
+
blobs = BlobReader(repo)
|
|
209
|
+
file_kinds: dict[str, str] = {}
|
|
210
|
+
old_vals: dict[tuple[str, str, str], dict] = {} # (extractor, file, name) -> record
|
|
211
|
+
new_vals: dict[tuple[str, str, str], dict] = {}
|
|
212
|
+
lfs: list[dict] = []
|
|
213
|
+
try:
|
|
214
|
+
for fc in files:
|
|
215
|
+
old_path = fc.old_path or fc.path
|
|
216
|
+
old = blobs.text(parent, old_path) if fc.status != "A" and parent != EMPTY_TREE else None
|
|
217
|
+
new = blobs.text(meta.sha, fc.path) if fc.status != "D" else None
|
|
218
|
+
file_kinds[fc.path] = classify_file_change(fc.path, old, new, noise) if fc.status in ("M", "R") else "code"
|
|
219
|
+
if cfg.get("lfs"):
|
|
220
|
+
lt = [t for t in (old, new) if t and t.startswith("version https://git-lfs")]
|
|
221
|
+
if lt:
|
|
222
|
+
size = lambda t: int(m.group(1)) if t and (m := re.search(r"^size (\d+)", t, re.M)) else None
|
|
223
|
+
lfs.append({"file": fc.path, "status": fc.status, "old_size": size(old), "new_size": size(new)})
|
|
224
|
+
continue
|
|
225
|
+
if file_kinds[fc.path] != "code":
|
|
226
|
+
continue
|
|
227
|
+
for text, bucket, path in ((old, old_vals, old_path), (new, new_vals, fc.path)):
|
|
228
|
+
if not text:
|
|
229
|
+
continue
|
|
230
|
+
for ext, opts in extractors_for(path, text, cfg):
|
|
231
|
+
if ext in CONST_LIKE and fc.status not in ("M", "R"):
|
|
232
|
+
continue
|
|
233
|
+
for name, rec in run_extractor(ext, path, text, opts).items():
|
|
234
|
+
bucket[(ext, fc.path, name)] = rec
|
|
235
|
+
finally:
|
|
236
|
+
blobs.close()
|
|
237
|
+
|
|
238
|
+
changes: list[dict] = []
|
|
239
|
+
gone = [k for k in old_vals if k not in new_vals]
|
|
240
|
+
fresh = [k for k in new_vals if k not in old_vals]
|
|
241
|
+
fam = lambda e: FAMILY.get(e, e)
|
|
242
|
+
gone_by = {(fam(e), n): (e, f) for e, f, n in gone}
|
|
243
|
+
fresh_names = {(fam(e), n) for e, f, n in fresh}
|
|
244
|
+
for key in sorted(set(old_vals) | set(new_vals)):
|
|
245
|
+
ext, f, name = key
|
|
246
|
+
o, n = old_vals.get(key), new_vals.get(key)
|
|
247
|
+
base = {"extractor": ext, "file": f, "name": name, "keyed": ext in KEYED}
|
|
248
|
+
if o and n:
|
|
249
|
+
if same_value(o["value"], n["value"]) and o.get("decl_type") == n.get("decl_type"):
|
|
250
|
+
continue
|
|
251
|
+
if ext in CONST_LIKE and not (_const_literal(o["value"]) and _const_literal(n["value"])):
|
|
252
|
+
continue # expressions/calls on either side: not a value change we can state exactly
|
|
253
|
+
changes.append({**base, "kind": "changed", "old": o["value"], "new": n["value"], "type": n["type"],
|
|
254
|
+
**{k: n[k] for k in ("decl_type", "key") if k in n}, **change_metrics(o["value"], n["value"])})
|
|
255
|
+
elif n:
|
|
256
|
+
src = gone_by.get((fam(ext), name))
|
|
257
|
+
if src is not None:
|
|
258
|
+
ov = old_vals[(src[0], src[1], name)]
|
|
259
|
+
same = same_value(ov["value"], n["value"])
|
|
260
|
+
changes.append({**base, "kind": "moved", "from": src[1], "old": n["value"] if same else ov["value"], "new": n["value"], "type": n["type"],
|
|
261
|
+
**({"old_literal": ov["value"]} if same and ov["value"] != n["value"] else {}),
|
|
262
|
+
**{k: n[k] for k in ("decl_type", "key") if k in n}, **change_metrics(ov["value"], n["value"])})
|
|
263
|
+
elif ext not in CONST_LIKE or name.isupper():
|
|
264
|
+
changes.append({**base, "kind": "added", "old": None, "new": n["value"], "type": n["type"],
|
|
265
|
+
**{k: n[k] for k in ("decl_type", "key") if k in n}, "pct": None, "ratio": None})
|
|
266
|
+
elif o and (fam(ext), name) not in fresh_names and (ext not in CONST_LIKE or name.isupper()):
|
|
267
|
+
changes.append({**base, "kind": "removed", "old": o["value"], "new": None, "type": o["type"],
|
|
268
|
+
**{k: o[k] for k in ("decl_type", "key") if k in o}, "pct": None, "ratio": None})
|
|
269
|
+
|
|
270
|
+
changes = _refine_keyed(changes, {fc.path for fc in files if fc.status == "D"})
|
|
271
|
+
|
|
272
|
+
if cfg.get("track_moves"):
|
|
273
|
+
_mark_still_defined(repo, meta.sha, cfg, [c for c in changes if c["kind"] == "removed" and c["extractor"] not in CONST_LIKE])
|
|
274
|
+
|
|
275
|
+
comps: dict[str, list] = {"added": [], "removed": [], "renamed": [], "changed": []}
|
|
276
|
+
for fc in files:
|
|
277
|
+
kind = classify_component(fc.path, cfg)
|
|
278
|
+
old_kind = classify_component(fc.old_path, cfg) if fc.old_path else kind
|
|
279
|
+
if fc.status == "A" and kind:
|
|
280
|
+
comps["added"].append({"kind": kind, "name": component_name(fc.path), "path": fc.path})
|
|
281
|
+
elif fc.status == "D" and kind:
|
|
282
|
+
comps["removed"].append({"kind": kind, "name": component_name(fc.path), "path": fc.path})
|
|
283
|
+
elif fc.status == "R" and (kind or old_kind):
|
|
284
|
+
if component_name(fc.old_path) != component_name(fc.path):
|
|
285
|
+
comps["renamed"].append({"kind": kind or old_kind, "old": component_name(fc.old_path), "new": component_name(fc.path), "path": fc.path})
|
|
286
|
+
elif file_kinds.get(fc.path) == "code" and kind:
|
|
287
|
+
comps["changed"].append({"kind": kind, "name": component_name(fc.path), "path": fc.path})
|
|
288
|
+
elif fc.status == "M" and kind and file_kinds.get(fc.path) == "code":
|
|
289
|
+
comps["changed"].append({"kind": kind, "name": component_name(fc.path), "path": fc.path})
|
|
290
|
+
|
|
291
|
+
tweaks: list[dict] = []
|
|
292
|
+
if cfg.get("tweaks") and parent != EMPTY_TREE:
|
|
293
|
+
covered = {c["file"] for c in changes} | {l["file"] for l in lfs}
|
|
294
|
+
known = {(c["file"], c["name"]) for c in changes}
|
|
295
|
+
tweaks = [t for t in numeric_tweaks(unified_diff(repo, parent, meta.sha, context=0), noise)
|
|
296
|
+
if C.watched(t["file"], cfg) and file_kinds.get(t["file"]) == "code" and t["changes"]
|
|
297
|
+
and t["file"] not in covered and (t["file"], t["label"]) not in known]
|
|
298
|
+
|
|
299
|
+
modules: dict[str, dict] = defaultdict(lambda: {"added": 0, "removed": 0, "modified": 0, "lines_added": 0, "lines_deleted": 0})
|
|
300
|
+
for fc in files:
|
|
301
|
+
m = modules[top_module(fc.path)]
|
|
302
|
+
m[{"A": "added", "D": "removed"}.get(fc.status, "modified")] += 1
|
|
303
|
+
m["lines_added"] += fc.added
|
|
304
|
+
m["lines_deleted"] += fc.deleted
|
|
305
|
+
kinds = Counter(file_kinds.values())
|
|
306
|
+
meaningful = [fc for fc in files if file_kinds.get(fc.path) == "code"]
|
|
307
|
+
if not meaningful:
|
|
308
|
+
cls = "noop"
|
|
309
|
+
elif all(os.path.splitext(fc.path)[1] in DOC_EXTS for fc in meaningful):
|
|
310
|
+
cls = "docs"
|
|
311
|
+
else:
|
|
312
|
+
cls = "code"
|
|
313
|
+
return {
|
|
314
|
+
"sha": meta.sha, "short": meta.sha[:7], "url": f"{repo_url}/commit/{meta.sha}" if repo_url else "",
|
|
315
|
+
"date": meta.date, "author": meta.author, "subject": meta.subject, "body": meta.body,
|
|
316
|
+
"is_merge": len(meta.parents) > 1, "is_root": not meta.parents, "parent": meta.parents[0] if meta.parents else None,
|
|
317
|
+
"class": cls,
|
|
318
|
+
"stats": {"files": len(files), "files_total": len(all_files), "added_files": sum(f.status == "A" for f in files),
|
|
319
|
+
"removed_files": sum(f.status == "D" for f in files), "lines_added": sum(f.added for f in files),
|
|
320
|
+
"lines_deleted": sum(f.deleted for f in files), "noise_files": kinds.get("noise", 0),
|
|
321
|
+
"whitespace_files": kinds.get("whitespace", 0), "comment_only_files": kinds.get("comments", 0)},
|
|
322
|
+
"modules": dict(sorted(modules.items(), key=lambda kv: -(kv[1]["lines_added"] + kv[1]["lines_deleted"]))),
|
|
323
|
+
"files": [{"status": f.status, "path": f.path, "old_path": f.old_path, "added": f.added, "deleted": f.deleted,
|
|
324
|
+
"kind": file_kinds.get(f.path, "code")} for f in files],
|
|
325
|
+
"changes": changes, "components": comps, "tweaks": tweaks, "lfs": lfs,
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
def _mark_still_defined(repo: str, sha: str, cfg: dict, removed: list[dict]) -> None:
|
|
330
|
+
"""A name that vanished from a changed file may still be defined in another file (it moved). Check with git grep."""
|
|
331
|
+
if not removed or len(removed) > 200:
|
|
332
|
+
return
|
|
333
|
+
names = sorted({c["name"].split(".")[-1] for c in removed})
|
|
334
|
+
pattern = "|".join(re.escape(n) for n in names)
|
|
335
|
+
out = run(["grep", "-l", "-E", rf"\b({pattern})\b", sha, "--", *[f":(glob){p}" for p in cfg["watch"]["paths"]]], cwd=repo, check=False)
|
|
336
|
+
candidates = [l.split(":", 1)[1] for l in out.splitlines() if ":" in l]
|
|
337
|
+
candidates = [p for p in candidates if C.watched(p, cfg)][:50]
|
|
338
|
+
if not candidates:
|
|
339
|
+
return
|
|
340
|
+
blobs = BlobReader(repo)
|
|
341
|
+
try:
|
|
342
|
+
found: dict[tuple[str, str], str] = {}
|
|
343
|
+
for path in candidates:
|
|
344
|
+
text = blobs.text(sha, path)
|
|
345
|
+
if not text:
|
|
346
|
+
continue
|
|
347
|
+
for ext, opts in extractors_for(path, text, cfg):
|
|
348
|
+
for name in run_extractor(ext, path, text, opts):
|
|
349
|
+
found.setdefault((ext, name), path)
|
|
350
|
+
for c in removed:
|
|
351
|
+
p = found.get((c["extractor"], c["name"]))
|
|
352
|
+
if p and p != c["file"]:
|
|
353
|
+
c["still_defined_in"] = p
|
|
354
|
+
finally:
|
|
355
|
+
blobs.close()
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
def _refine_keyed(changes: list[dict], deleted_files: set[str]) -> list[dict]:
|
|
359
|
+
"""Structured-config post-processing.
|
|
360
|
+
|
|
361
|
+
* numeric lists of equal length that changed are split per element (max_velocity[0] 0.26 -> 0.5);
|
|
362
|
+
* a top-level section renamed with identical values (dwb_controller.* -> nav2_controller.*) becomes one
|
|
363
|
+
"renamed" record per key instead of N removed + N added;
|
|
364
|
+
* removals caused by deleting a whole file are flagged (file_deleted) so summaries can collapse them.
|
|
365
|
+
"""
|
|
366
|
+
out: list[dict] = []
|
|
367
|
+
for c in changes:
|
|
368
|
+
if c["kind"] == "changed" and c.get("keyed") and c["type"] == "list":
|
|
369
|
+
lo, ln = list_items(c["old"]), list_items(c["new"])
|
|
370
|
+
if lo and ln and len(lo) == len(ln) and all(parse_num(x) is not None for x in lo + ln):
|
|
371
|
+
for i, (a, b) in enumerate(zip(lo, ln)):
|
|
372
|
+
if not same_value(a, b):
|
|
373
|
+
out.append({**c, "name": f"{c['name']}[{i}]", "old": a, "new": b, "type": "number",
|
|
374
|
+
"list_index": i, **change_metrics(a, b)})
|
|
375
|
+
continue
|
|
376
|
+
out.append(c)
|
|
377
|
+
removed = [c for c in out if c["kind"] == "removed" and c.get("keyed")]
|
|
378
|
+
added = [c for c in out if c["kind"] == "added" and c.get("keyed")]
|
|
379
|
+
split = lambda n: n.split(".", 1) if "." in n else (None, n)
|
|
380
|
+
add_by = {}
|
|
381
|
+
for c in added:
|
|
382
|
+
head, rest = split(c["name"])
|
|
383
|
+
if head is not None:
|
|
384
|
+
add_by.setdefault((c["extractor"], c["file"], rest), []).append(c)
|
|
385
|
+
pairs: dict[tuple[str, str, str], list[tuple[dict, dict]]] = {}
|
|
386
|
+
for c in removed:
|
|
387
|
+
head, rest = split(c["name"])
|
|
388
|
+
for a in add_by.get((c["extractor"], c["file"], rest), []):
|
|
389
|
+
if same_value(c["old"], a["new"]) and split(a["name"])[0] != head:
|
|
390
|
+
pairs.setdefault((c["file"], head, split(a["name"])[0]), []).append((c, a))
|
|
391
|
+
break
|
|
392
|
+
drop: set[int] = set()
|
|
393
|
+
for (f, old_head, new_head), ps in pairs.items():
|
|
394
|
+
total = sum(1 for c in removed if c["file"] == f and split(c["name"])[0] == old_head)
|
|
395
|
+
if len(ps) >= 3 and len(ps) >= 0.8 * total:
|
|
396
|
+
for r, a in ps:
|
|
397
|
+
drop.add(id(r))
|
|
398
|
+
a.update(kind="renamed", old_name=r["name"], old=r["old"], pct=None, ratio=None)
|
|
399
|
+
out = [c for c in out if id(c) not in drop]
|
|
400
|
+
for c in out:
|
|
401
|
+
if c["kind"] == "removed" and c["file"] in deleted_files:
|
|
402
|
+
c["file_deleted"] = True
|
|
403
|
+
return out
|
knoblog/cli.py
ADDED
|
@@ -0,0 +1,240 @@
|
|
|
1
|
+
"""knoblog command line.
|
|
2
|
+
|
|
3
|
+
knoblog run --repo <url or path> --config knoblog.yml [--out site]
|
|
4
|
+
knoblog init [--preset ros2] [--out knoblog.yml]
|
|
5
|
+
knoblog extract FILE [--extractor NAME]
|
|
6
|
+
knoblog presets
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import argparse
|
|
11
|
+
import json
|
|
12
|
+
import os
|
|
13
|
+
import re
|
|
14
|
+
import shutil
|
|
15
|
+
import sys
|
|
16
|
+
import time
|
|
17
|
+
from datetime import datetime, timezone
|
|
18
|
+
|
|
19
|
+
from . import __version__
|
|
20
|
+
from . import config as C
|
|
21
|
+
from . import posttext, render
|
|
22
|
+
from .analyze import CONST_LIKE, analyze_commit
|
|
23
|
+
from .extractors import REGISTRY, auto_extractors, run_extractor
|
|
24
|
+
from .gitio import EMPTY_TREE, ensure_clone, head_rev, list_commits, run, unified_diff
|
|
25
|
+
from .summarize import deterministic, llm_config, llm_summarize, load_note
|
|
26
|
+
|
|
27
|
+
PKG = os.path.dirname(os.path.abspath(__file__))
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def log(msg: str) -> None:
|
|
31
|
+
print(msg, file=sys.stderr, flush=True)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def browse_url(repo: str) -> str:
|
|
35
|
+
"""https URL for commit links from a clone URL (GitHub/GitLab style), or ''."""
|
|
36
|
+
repo = repo.strip()
|
|
37
|
+
m = re.match(r"^git@([^:]+):(.+?)(\.git)?/?$", repo)
|
|
38
|
+
if m:
|
|
39
|
+
return f"https://{m.group(1)}/{m.group(2)}"
|
|
40
|
+
m = re.match(r"^(https?://[^/]+/.+?)(\.git)?/?$", repo)
|
|
41
|
+
return m.group(1) if m else ""
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def resolve_repo(repo: str, cfg: dict, args) -> tuple[str, str]:
|
|
45
|
+
"""Return (local git dir, browsable url)."""
|
|
46
|
+
if os.path.isdir(repo):
|
|
47
|
+
url = cfg.get("repo_url") or browse_url(run(["remote", "get-url", "origin"], cwd=repo, check=False).strip())
|
|
48
|
+
if not args.no_fetch and run(["remote"], cwd=repo, check=False).strip():
|
|
49
|
+
log(f"[fetch] {repo}")
|
|
50
|
+
run(["fetch", "--quiet", "origin"], cwd=repo, check=False)
|
|
51
|
+
return repo, url
|
|
52
|
+
cache = args.cache_dir or os.path.join(".knoblog-cache")
|
|
53
|
+
dest = os.path.join(cache, re.sub(r"[^A-Za-z0-9._-]+", "_", browse_url(repo).split("//")[-1] or repo))
|
|
54
|
+
if not (args.no_fetch and os.path.isdir(dest)):
|
|
55
|
+
log(f"[clone] {repo} -> {dest} ({'blob-less partial' if cfg['clone']['partial'] else 'full'})")
|
|
56
|
+
ensure_clone(repo, dest, cfg.get("branch"), partial=cfg["clone"]["partial"])
|
|
57
|
+
return dest, cfg.get("repo_url") or browse_url(repo)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def summarise(e: dict, cfg: dict, repo: str, meta, llm, state: dict) -> None:
|
|
61
|
+
auto = deterministic(e, cfg)
|
|
62
|
+
e["auto"] = auto
|
|
63
|
+
sc = cfg["summaries"]
|
|
64
|
+
base = cfg.get("_dir") or "."
|
|
65
|
+
rel = lambda p: p if not p or os.path.isabs(p) else os.path.join(base, p)
|
|
66
|
+
note = load_note(rel(sc.get("notes_dir")), e["short"])
|
|
67
|
+
cached = load_note(rel(sc.get("llm_cache_dir")), e["short"])
|
|
68
|
+
mk = lambda src, n: {"source": src, "title": n["title"] or auto["headline"], "markdown": n["markdown"], "bullets": auto["bullets"], "tags": auto["tags"]}
|
|
69
|
+
if note:
|
|
70
|
+
e["summary"] = mk("note", note)
|
|
71
|
+
elif cached:
|
|
72
|
+
e["summary"] = mk("llm", cached)
|
|
73
|
+
elif llm and e["class"] != "noop" and state["llm_calls"] < int(sc.get("llm_limit") or 0) and sc.get("llm_cache_dir"):
|
|
74
|
+
state["llm_calls"] += 1
|
|
75
|
+
try:
|
|
76
|
+
parent = meta.parents[0] if meta.parents else EMPTY_TREE
|
|
77
|
+
text = llm_summarize(llm, e, unified_diff(repo, parent, meta.sha, context=2))
|
|
78
|
+
d = rel(sc["llm_cache_dir"])
|
|
79
|
+
os.makedirs(d, exist_ok=True)
|
|
80
|
+
with open(os.path.join(d, f"{e['short']}.md"), "w", encoding="utf-8") as f:
|
|
81
|
+
f.write(text + "\n")
|
|
82
|
+
e["summary"] = mk("llm", load_note(d, e["short"]))
|
|
83
|
+
except Exception as exc: # deterministic fallback
|
|
84
|
+
log(f"[llm] {e['short']} failed ({exc}); using deterministic summary")
|
|
85
|
+
if "summary" not in e:
|
|
86
|
+
e["summary"] = {"source": "auto", "title": auto["headline"], "markdown": "", "bullets": auto["bullets"], "tags": auto["tags"]}
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def history_records(entries: list[dict], cfg: dict) -> list[dict]:
|
|
90
|
+
"""Flat change records for params.html / changes.json (bulk additions are left to entry pages)."""
|
|
91
|
+
out = []
|
|
92
|
+
limit = cfg["post"]["bulk_added_limit"]
|
|
93
|
+
for e in entries:
|
|
94
|
+
bulk = e["is_root"] or sum(c["kind"] == "added" for c in e["changes"]) > limit
|
|
95
|
+
for c in e["changes"]:
|
|
96
|
+
if bulk and c["kind"] in ("added", "removed"):
|
|
97
|
+
continue
|
|
98
|
+
if (c["kind"] in ("changed", "moved") and c["old"] == c["new"]) or c["kind"] == "renamed" or c.get("file_deleted"):
|
|
99
|
+
continue
|
|
100
|
+
out.append({**render.change_record(e, c), "pct": c.get("pct")})
|
|
101
|
+
return out
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def cmd_run(args) -> int:
|
|
105
|
+
t0 = time.time()
|
|
106
|
+
overrides: dict = {}
|
|
107
|
+
if args.site_url is not None:
|
|
108
|
+
overrides.setdefault("site", {})["url"] = args.site_url
|
|
109
|
+
if args.max_commits:
|
|
110
|
+
overrides.setdefault("history", {})["max_commits"] = args.max_commits
|
|
111
|
+
if args.since:
|
|
112
|
+
overrides.setdefault("history", {})["since"] = args.since
|
|
113
|
+
if args.no_llm:
|
|
114
|
+
overrides.setdefault("summaries", {})["llm"] = False
|
|
115
|
+
if args.branch:
|
|
116
|
+
overrides["branch"] = args.branch
|
|
117
|
+
cfg = C.load(args.config, args.preset, overrides)
|
|
118
|
+
repo_arg = args.repo or cfg.get("repo")
|
|
119
|
+
if not repo_arg:
|
|
120
|
+
raise SystemExit("no repository: pass --repo <url or path> or set `repo:` in the config")
|
|
121
|
+
repo, url = resolve_repo(repo_arg, cfg, args)
|
|
122
|
+
cfg["repo_url"] = url
|
|
123
|
+
cfg["name"] = cfg.get("name") or (url or os.path.abspath(repo)).rstrip("/").split("/")[-1]
|
|
124
|
+
h = cfg["history"]
|
|
125
|
+
rev = head_rev(repo, cfg.get("branch"))
|
|
126
|
+
paths = cfg["watch"]["paths"] if h.get("limit_log_to_watched_paths") and cfg["watch"]["paths"] != ["**"] else None
|
|
127
|
+
commits = list_commits(repo, rev, paths, h.get("max_commits"), h.get("since"), h.get("first_parent", True))
|
|
128
|
+
log(f"[walk] {len(commits)} commits (rev {rev[:7]}, newest {h.get('max_commits')}{', path-limited' if paths else ''})")
|
|
129
|
+
llm = llm_config() if cfg["summaries"].get("llm") else None
|
|
130
|
+
entries, state = [], {"llm_calls": 0}
|
|
131
|
+
for meta in commits:
|
|
132
|
+
e = analyze_commit(repo, meta, cfg, url)
|
|
133
|
+
summarise(e, cfg, repo, meta, llm, state)
|
|
134
|
+
entries.append(e)
|
|
135
|
+
if args.verbose:
|
|
136
|
+
log(f"[entry] {e['date'][:10]} {e['short']} {e['class']:5} {len(e['changes']):4} changes {e['summary']['title'][:80]}")
|
|
137
|
+
posts = posttext.post_text_for(entries, cfg) if cfg["post"].get("enabled") else []
|
|
138
|
+
posts_by_sha = {p["sha"]: p for p in posts}
|
|
139
|
+
now = datetime.now(timezone.utc)
|
|
140
|
+
meta = {"generated_at": now.strftime("%Y-%m-%d %H:%M UTC"), "generated_iso": now.strftime("%Y-%m-%dT%H:%M:%SZ"), "head_sha": rev}
|
|
141
|
+
site = render.Site(cfg, meta)
|
|
142
|
+
out = args.out
|
|
143
|
+
if os.path.isdir(out):
|
|
144
|
+
shutil.rmtree(out)
|
|
145
|
+
os.makedirs(os.path.join(out, "entries"))
|
|
146
|
+
shutil.copy(os.path.join(PKG, "static", "style.css"), os.path.join(out, "style.css"))
|
|
147
|
+
real = [e for e in entries if e["class"] != "noop"]
|
|
148
|
+
noops = [e for e in entries if e["class"] == "noop"]
|
|
149
|
+
newest = list(reversed(real))
|
|
150
|
+
records = history_records(real, cfg)
|
|
151
|
+
newest_records = list(reversed(records))
|
|
152
|
+
seen, current = set(), []
|
|
153
|
+
for r in newest_records:
|
|
154
|
+
if r["kind"] in ("changed", "moved") and r["extractor"] not in CONST_LIKE and (r["name"], r["file"]) not in seen:
|
|
155
|
+
seen.add((r["name"], r["file"]))
|
|
156
|
+
current.append(r)
|
|
157
|
+
render.write(os.path.join(out, "index.html"), site.page(site.title, site.index(newest, list(reversed(noops)), current[:40])))
|
|
158
|
+
render.write(os.path.join(out, "params.html"), site.page("Parameter history · " + site.title, site.params(newest_records)))
|
|
159
|
+
render.write(os.path.join(out, "posts.html"), site.page("Post text · " + site.title, site.posts_page(list(reversed(posts)))))
|
|
160
|
+
for i, e in enumerate(real):
|
|
161
|
+
prev = real[i - 1] if i else None
|
|
162
|
+
nxt = real[i + 1] if i + 1 < len(real) else None
|
|
163
|
+
title = f"{render.fmt_date(e['date'])}: {e['summary']['title'].replace('`', '')}"
|
|
164
|
+
render.write(os.path.join(out, "entries", f"{e['short']}.html"),
|
|
165
|
+
site.page(title + " · " + site.title, site.detail(e, prev, nxt, posts_by_sha.get(e["sha"])), depth=1, description=title))
|
|
166
|
+
render.write(os.path.join(out, "feed.xml"), site.atom(newest))
|
|
167
|
+
render.write(os.path.join(out, "feed.json"), json.dumps(site.json_feed(newest, posts_by_sha), indent=1, ensure_ascii=False))
|
|
168
|
+
for r in newest_records:
|
|
169
|
+
r.pop("pct", None)
|
|
170
|
+
render.write(os.path.join(out, "changes.json"), json.dumps({"generator": f"knoblog {__version__}", "repo": url, "head": rev,
|
|
171
|
+
"generated": meta["generated_iso"], "changes": newest_records}, indent=1, ensure_ascii=False))
|
|
172
|
+
render.write(os.path.join(out, "posts.json"), json.dumps({"generator": f"knoblog {__version__}", "note": "Text only; knoblog never posts.",
|
|
173
|
+
"posts": list(reversed(posts))}, indent=1, ensure_ascii=False))
|
|
174
|
+
render.write(os.path.join(out, "posts.txt"), render.posts_txt(list(reversed(posts))))
|
|
175
|
+
slim = [{k: e[k] for k in ("sha", "short", "url", "date", "class", "is_root", "is_merge", "stats", "summary", "changes", "components", "lfs")}
|
|
176
|
+
for e in newest]
|
|
177
|
+
render.write(os.path.join(out, "entries.json"), json.dumps({"generated": meta["generated_iso"], "head": rev, "entries": slim}, indent=1, ensure_ascii=False))
|
|
178
|
+
render.write(os.path.join(out, ".nojekyll"), "")
|
|
179
|
+
ready = sum(p["status"] == "ready" for p in posts)
|
|
180
|
+
log(f"[done] {len(real)} entries ({len(noops)} no-op), {len(records)} value changes, {ready} post-ready -> {out} in {time.time() - t0:.1f}s")
|
|
181
|
+
if args.github_output and os.environ.get("GITHUB_OUTPUT"):
|
|
182
|
+
with open(os.environ["GITHUB_OUTPUT"], "a", encoding="utf-8") as f:
|
|
183
|
+
f.write(f"entries={len(real)}\nchanges={len(records)}\nposts_ready={ready}\nsite_dir={out}\n")
|
|
184
|
+
return 0
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def cmd_init(args) -> int:
|
|
188
|
+
if os.path.exists(args.out) and not args.force:
|
|
189
|
+
raise SystemExit(f"{args.out} exists (use --force)")
|
|
190
|
+
src = os.path.join(PKG, "presets", "_template.yml")
|
|
191
|
+
text = open(src, encoding="utf-8").read().replace("extends: generic", f"extends: {args.preset}")
|
|
192
|
+
render.write(args.out, text)
|
|
193
|
+
log(f"wrote {args.out} (extends preset {args.preset!r}); edit watch.paths, then: knoblog run --repo <url or path> --config {args.out}")
|
|
194
|
+
return 0
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def cmd_extract(args) -> int:
|
|
198
|
+
text = open(args.file, encoding="utf-8", errors="replace").read()
|
|
199
|
+
names = [args.extractor] if args.extractor else auto_extractors(args.file, text)
|
|
200
|
+
out = {n: run_extractor(n, args.file, text, {}) for n in names}
|
|
201
|
+
print(json.dumps(out, indent=1, ensure_ascii=False))
|
|
202
|
+
return 0
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def main(argv: list[str] | None = None) -> int:
|
|
206
|
+
p = argparse.ArgumentParser(prog="knoblog", description="Plain-English changelogs + X-ready post text for parameter/config changes in a git repo.")
|
|
207
|
+
p.add_argument("--version", action="version", version=f"knoblog {__version__}")
|
|
208
|
+
sub = p.add_subparsers(dest="cmd", required=True)
|
|
209
|
+
r = sub.add_parser("run", help="analyse history and write the site + feeds")
|
|
210
|
+
r.add_argument("--repo", help="git URL or local path (overrides `repo:` in the config)")
|
|
211
|
+
r.add_argument("--config", "-c", help="knoblog.yml (or a preset name)")
|
|
212
|
+
r.add_argument("--preset", help="preset to start from (ros2, px4, ardupilot, x-algorithm, generic)")
|
|
213
|
+
r.add_argument("--out", "-o", default="site")
|
|
214
|
+
r.add_argument("--branch")
|
|
215
|
+
r.add_argument("--max-commits", type=int)
|
|
216
|
+
r.add_argument("--since", help="only commits after this date (YYYY-MM-DD)")
|
|
217
|
+
r.add_argument("--site-url", help="public URL of the site (used in feeds and post links)")
|
|
218
|
+
r.add_argument("--cache-dir", help="where remote repos are cloned (default .knoblog-cache)")
|
|
219
|
+
r.add_argument("--no-fetch", action="store_true", help="use the existing clone as-is")
|
|
220
|
+
r.add_argument("--no-llm", action="store_true", help="never call an LLM")
|
|
221
|
+
r.add_argument("--github-output", action="store_true", help="write counts to $GITHUB_OUTPUT")
|
|
222
|
+
r.add_argument("-v", "--verbose", action="store_true")
|
|
223
|
+
r.set_defaults(func=cmd_run)
|
|
224
|
+
i = sub.add_parser("init", help="write a starter knoblog.yml")
|
|
225
|
+
i.add_argument("--preset", default="generic")
|
|
226
|
+
i.add_argument("--out", default="knoblog.yml")
|
|
227
|
+
i.add_argument("--force", action="store_true")
|
|
228
|
+
i.set_defaults(func=cmd_init)
|
|
229
|
+
x = sub.add_parser("extract", help="print the values knoblog extracts from one file")
|
|
230
|
+
x.add_argument("file")
|
|
231
|
+
x.add_argument("--extractor", choices=sorted(REGISTRY))
|
|
232
|
+
x.set_defaults(func=cmd_extract)
|
|
233
|
+
ps = sub.add_parser("presets", help="list built-in presets")
|
|
234
|
+
ps.set_defaults(func=lambda a: print("\n".join(n for n in C.presets() if not n.startswith("_"))) or 0)
|
|
235
|
+
args = p.parse_args(argv)
|
|
236
|
+
return args.func(args)
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
if __name__ == "__main__":
|
|
240
|
+
raise SystemExit(main())
|