knoblog 0.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
knoblog/__init__.py ADDED
@@ -0,0 +1,2 @@
1
+ """knoblog: plain-English changelogs and X-ready post text for parameter, gain, threshold, flag and config changes."""
2
+ __version__ = "0.1.1"
knoblog/__main__.py ADDED
@@ -0,0 +1,3 @@
1
+ from .cli import main
2
+
3
+ raise SystemExit(main())
knoblog/analyze.py ADDED
@@ -0,0 +1,403 @@
1
+ """Deterministic analysis of one commit: which watched parameters changed, old -> new.
2
+
3
+ Everything is derived mechanically from git; nothing is guessed.
4
+ """
5
+ from __future__ import annotations
6
+
7
+ import os
8
+ import re
9
+ from collections import Counter, defaultdict
10
+
11
+ from . import config as C
12
+ from .yamlmini import unquote
13
+ from .extractors import KEYED, LITERALISH, auto_extractors, infer_type, parse_num, run_extractor
14
+ from .gitio import EMPTY_TREE, BlobReader, CommitMeta, changed_files, run, unified_diff
15
+
16
+ FAMILY = {"px4_param_define": "px4", "px4_module_yaml": "px4"} # same parameter namespace, different file formats
17
+ CONST_LIKE = {"rust_const", "python_const", "c_const", "jvm_const"} # only diffed in modified files; added/removed only for UPPER names
18
+ DOC_EXTS = {".md", ".txt", ".rst"}
19
+
20
+
21
+ # ---------------------------------------------------------------------------
22
+ # File-level noise classification
23
+ # ---------------------------------------------------------------------------
24
+ def _strip_comments(path: str, text: str) -> str:
25
+ ext = os.path.splitext(path)[1]
26
+ if ext in {".py", ".toml", ".yml", ".yaml", ".bazel", ".bzl", ".cfg", ".ini"} or path.endswith("BUILD"):
27
+ text = re.sub(r"(?m)^\s*#.*$", "", text)
28
+ text = re.sub(r"(?m)\s+#[^\n'\"]*$", "", text)
29
+ if ext == ".py":
30
+ text = re.sub(r'(?s)"""(.*?)"""', '""""""', text)
31
+ else:
32
+ text = re.sub(r"(?s)/\*.*?\*/", "", text)
33
+ text = re.sub(r"(?m)(^|\s)//.*$", r"\1", text)
34
+ return text
35
+
36
+
37
+ def classify_file_change(path: str, old: str | None, new: str | None, noise: list[re.Pattern]) -> str:
38
+ """'noise' (only configured noise lines changed), 'whitespace', 'comments' or 'code'."""
39
+ if old is None or new is None:
40
+ return "code"
41
+ strip = lambda t: "\n".join(l for l in t.splitlines() if not any(rx.search(l) for rx in noise))
42
+ no_ws = lambda t: re.sub(r"\s+", "", t)
43
+ o, n = strip(old), strip(new)
44
+ if o == n:
45
+ return "noise"
46
+ if no_ws(o) == no_ws(n):
47
+ return "whitespace"
48
+ if no_ws(_strip_comments(path, o)) == no_ws(_strip_comments(path, n)):
49
+ return "comments"
50
+ return "code"
51
+
52
+
53
+ # ---------------------------------------------------------------------------
54
+ # Optional: generic numeric tweaks found by pairing -/+ lines in the same hunk
55
+ # ---------------------------------------------------------------------------
56
+ NUM_TOKEN = re.compile(r"(?<![\w.])[-+]?(?:\d[\d_]*\.?\d*(?:[eE][-+]?\d+)?|\.\d+)(?![\w])")
57
+ BOOL_TOKEN = re.compile(r"\b(true|false|True|False)\b")
58
+ TEST_PATH = re.compile(r"(^|/)(tests?|testing|testdata)/|_test\.\w+$|(^|/)test_[^/]+$|Test\.\w+$")
59
+ TEST_LINE = re.compile(r"\bassert|\bexpect\(|\.apply\(&|should|#\[test\]")
60
+ COMMENT_LINE = re.compile(r"^\s*(//|#|\*|/\*|--)")
61
+ LOCK_FILE = re.compile(r"(^|/)(Cargo\.lock|package-lock\.json|yarn\.lock|poetry\.lock|uv\.lock|[^/]*\.lock)$")
62
+
63
+
64
+ def _skeleton(line: str) -> str:
65
+ return re.sub(r"\s+", "", BOOL_TOKEN.sub("?", NUM_TOKEN.sub("#", line)))
66
+
67
+
68
+ def _label_for(line: str) -> str:
69
+ for rx in (r"\s*(?:pub\s+)?(?:const\s+|static\s+|val\s+|var\s+|let\s+)?([A-Za-z_][\w.]*)\s*(?::[^=]*)?=",
70
+ r"\s*([A-Za-z_][\w.]*)\s*:", r"\s*\"([^\"]+)\"\s*[:=]"):
71
+ m = re.match(rx, line)
72
+ if m:
73
+ return m.group(1)
74
+ return ""
75
+
76
+
77
+ def numeric_tweaks(diff_text: str, noise: list[re.Pattern]) -> list[dict]:
78
+ tweaks: list[dict] = []
79
+ cur_file, minus, plus = None, [], []
80
+
81
+ def flush():
82
+ if not cur_file or not minus or not plus:
83
+ return
84
+ if TEST_PATH.search(cur_file) or os.path.splitext(cur_file)[1] in DOC_EXTS or LOCK_FILE.search(cur_file):
85
+ return
86
+ used = set()
87
+ for m in minus:
88
+ if TEST_LINE.search(m) or COMMENT_LINE.match(m) or any(rx.search(m) for rx in noise):
89
+ continue
90
+ sk = _skeleton(m)
91
+ if len(re.sub(r"[^A-Za-z]", "", sk)) < 2:
92
+ continue
93
+ for idx, p in enumerate(plus):
94
+ if idx in used or _skeleton(p) != sk or p.strip() == m.strip():
95
+ continue
96
+ ov = NUM_TOKEN.findall(m) + BOOL_TOKEN.findall(m)
97
+ nv = NUM_TOKEN.findall(p) + BOOL_TOKEN.findall(p)
98
+ used.add(idx)
99
+ tweaks.append({"file": cur_file, "label": _label_for(p) or _label_for(m), "old_line": m.strip()[:200],
100
+ "new_line": p.strip()[:200],
101
+ "changes": [[o, n] for o, n in zip(ov, nv) if o != n] if len(ov) == len(nv) else []})
102
+ break
103
+
104
+ for line in diff_text.splitlines():
105
+ if line.startswith("diff --git"):
106
+ flush(); minus, plus, cur_file = [], [], None
107
+ elif line.startswith("+++ "):
108
+ cur_file = line[6:] if line.startswith("+++ b/") else None
109
+ elif line.startswith("--- "):
110
+ continue
111
+ elif line.startswith("@@"):
112
+ flush(); minus, plus = [], []
113
+ elif line.startswith("-"):
114
+ minus.append(line[1:])
115
+ elif line.startswith("+"):
116
+ plus.append(line[1:])
117
+ flush()
118
+ return tweaks
119
+
120
+
121
+ # ---------------------------------------------------------------------------
122
+ # Helpers
123
+ # ---------------------------------------------------------------------------
124
+ def extractors_for(path: str, text: str, cfg: dict) -> list[tuple[str, dict]]:
125
+ out: list[tuple[str, dict]] = []
126
+ for spec in cfg["extractors"]:
127
+ if spec.get("paths") and not C.match_any(path, spec["paths"]):
128
+ continue
129
+ if spec.get("exclude") and C.match_any(path, spec["exclude"]):
130
+ continue
131
+ names = auto_extractors(path, text) if spec["use"] == "auto" else [spec["use"]]
132
+ for n in names:
133
+ if n not in [x[0] for x in out]:
134
+ out.append((n, spec.get("options") or {}))
135
+ return out
136
+
137
+
138
+ _C_NUMEXPR = re.compile(r"(?:[-+*/()\s<>|&~]|0[xX][0-9a-fA-F]+[uUlL]*|\d[\d_.]*(?:[eE][-+]?\d+)?[fFuUlL]*)+")
139
+
140
+
141
+ def _const_literal(v: str) -> bool:
142
+ v = str(v).strip()
143
+ return bool(LITERALISH.fullmatch(v) or (_C_NUMEXPR.fullmatch(v) and re.search(r"\d", v)))
144
+
145
+
146
+ def change_metrics(old, new) -> dict:
147
+ o, n = parse_num(old), parse_num(new)
148
+ if o is None or n is None:
149
+ return {"pct": None, "ratio": None}
150
+ pct = None if o == 0 else round((n - o) / abs(o) * 100, 4)
151
+ ratio = None if o == 0 or (o < 0) != (n < 0) else round(n / o, 6)
152
+ return {"pct": pct, "ratio": ratio}
153
+
154
+
155
+ def list_items(v) -> list[str] | None:
156
+ """Items of a flat list value "[a, b]" (raw strings), or None."""
157
+ v = str(v).strip() if v is not None else ""
158
+ if not (v.startswith("[") and v.endswith("]")):
159
+ return None
160
+ from .yamlmini import _split_flow
161
+ return _split_flow(v[1:-1])
162
+
163
+
164
+ def same_value(a: str, b: str) -> bool:
165
+ if a == b:
166
+ return True
167
+ if str(a).lower() == str(b).lower() and str(a).lower() in ("true", "false"):
168
+ return True # YAML True == true
169
+ la, lb = list_items(a), list_items(b)
170
+ if la is not None and lb is not None:
171
+ return len(la) == len(lb) and all(same_value(unquote(x), unquote(y)) for x, y in zip(la, lb))
172
+ x, y = parse_num(a), parse_num(b)
173
+ return x is not None and y is not None and x == y and infer_type(a) == infer_type(b) and not ("." in a) ^ ("." in b)
174
+
175
+
176
+ def classify_component(path: str, cfg: dict) -> str | None:
177
+ comp = cfg["components"]
178
+ if comp.get("exclude") and re.search(comp["exclude"], path):
179
+ return None
180
+ for r in comp["rules"]:
181
+ if re.search(r["pattern"], path):
182
+ return r["kind"]
183
+ return None
184
+
185
+
186
+ GENERIC_NAMES = {"filter", "task_filter", "filters", "scorer", "scorers", "source", "sources", "rules", "rule"}
187
+
188
+
189
+ def component_name(path: str) -> str:
190
+ base = os.path.splitext(os.path.basename(path))[0]
191
+ if base.lower() in GENERIC_NAMES:
192
+ return f"{os.path.basename(os.path.dirname(path))}/{base}"
193
+ return base
194
+
195
+
196
+ def top_module(path: str) -> str:
197
+ return path.split("/", 1)[0] if "/" in path else "(repo root)"
198
+
199
+
200
+ # ---------------------------------------------------------------------------
201
+ # Main entry point
202
+ # ---------------------------------------------------------------------------
203
+ def analyze_commit(repo: str, meta: CommitMeta, cfg: dict, repo_url: str) -> dict:
204
+ parent = meta.parents[0] if meta.parents else EMPTY_TREE
205
+ noise = [re.compile(p) for p in cfg["noise"]["line_patterns"]]
206
+ all_files = changed_files(repo, parent, meta.sha)
207
+ files = [f for f in all_files if C.watched(f.path, cfg) and not C.match_any(f.path, cfg["noise"]["ignore_paths"])]
208
+ blobs = BlobReader(repo)
209
+ file_kinds: dict[str, str] = {}
210
+ old_vals: dict[tuple[str, str, str], dict] = {} # (extractor, file, name) -> record
211
+ new_vals: dict[tuple[str, str, str], dict] = {}
212
+ lfs: list[dict] = []
213
+ try:
214
+ for fc in files:
215
+ old_path = fc.old_path or fc.path
216
+ old = blobs.text(parent, old_path) if fc.status != "A" and parent != EMPTY_TREE else None
217
+ new = blobs.text(meta.sha, fc.path) if fc.status != "D" else None
218
+ file_kinds[fc.path] = classify_file_change(fc.path, old, new, noise) if fc.status in ("M", "R") else "code"
219
+ if cfg.get("lfs"):
220
+ lt = [t for t in (old, new) if t and t.startswith("version https://git-lfs")]
221
+ if lt:
222
+ size = lambda t: int(m.group(1)) if t and (m := re.search(r"^size (\d+)", t, re.M)) else None
223
+ lfs.append({"file": fc.path, "status": fc.status, "old_size": size(old), "new_size": size(new)})
224
+ continue
225
+ if file_kinds[fc.path] != "code":
226
+ continue
227
+ for text, bucket, path in ((old, old_vals, old_path), (new, new_vals, fc.path)):
228
+ if not text:
229
+ continue
230
+ for ext, opts in extractors_for(path, text, cfg):
231
+ if ext in CONST_LIKE and fc.status not in ("M", "R"):
232
+ continue
233
+ for name, rec in run_extractor(ext, path, text, opts).items():
234
+ bucket[(ext, fc.path, name)] = rec
235
+ finally:
236
+ blobs.close()
237
+
238
+ changes: list[dict] = []
239
+ gone = [k for k in old_vals if k not in new_vals]
240
+ fresh = [k for k in new_vals if k not in old_vals]
241
+ fam = lambda e: FAMILY.get(e, e)
242
+ gone_by = {(fam(e), n): (e, f) for e, f, n in gone}
243
+ fresh_names = {(fam(e), n) for e, f, n in fresh}
244
+ for key in sorted(set(old_vals) | set(new_vals)):
245
+ ext, f, name = key
246
+ o, n = old_vals.get(key), new_vals.get(key)
247
+ base = {"extractor": ext, "file": f, "name": name, "keyed": ext in KEYED}
248
+ if o and n:
249
+ if same_value(o["value"], n["value"]) and o.get("decl_type") == n.get("decl_type"):
250
+ continue
251
+ if ext in CONST_LIKE and not (_const_literal(o["value"]) and _const_literal(n["value"])):
252
+ continue # expressions/calls on either side: not a value change we can state exactly
253
+ changes.append({**base, "kind": "changed", "old": o["value"], "new": n["value"], "type": n["type"],
254
+ **{k: n[k] for k in ("decl_type", "key") if k in n}, **change_metrics(o["value"], n["value"])})
255
+ elif n:
256
+ src = gone_by.get((fam(ext), name))
257
+ if src is not None:
258
+ ov = old_vals[(src[0], src[1], name)]
259
+ same = same_value(ov["value"], n["value"])
260
+ changes.append({**base, "kind": "moved", "from": src[1], "old": n["value"] if same else ov["value"], "new": n["value"], "type": n["type"],
261
+ **({"old_literal": ov["value"]} if same and ov["value"] != n["value"] else {}),
262
+ **{k: n[k] for k in ("decl_type", "key") if k in n}, **change_metrics(ov["value"], n["value"])})
263
+ elif ext not in CONST_LIKE or name.isupper():
264
+ changes.append({**base, "kind": "added", "old": None, "new": n["value"], "type": n["type"],
265
+ **{k: n[k] for k in ("decl_type", "key") if k in n}, "pct": None, "ratio": None})
266
+ elif o and (fam(ext), name) not in fresh_names and (ext not in CONST_LIKE or name.isupper()):
267
+ changes.append({**base, "kind": "removed", "old": o["value"], "new": None, "type": o["type"],
268
+ **{k: o[k] for k in ("decl_type", "key") if k in o}, "pct": None, "ratio": None})
269
+
270
+ changes = _refine_keyed(changes, {fc.path for fc in files if fc.status == "D"})
271
+
272
+ if cfg.get("track_moves"):
273
+ _mark_still_defined(repo, meta.sha, cfg, [c for c in changes if c["kind"] == "removed" and c["extractor"] not in CONST_LIKE])
274
+
275
+ comps: dict[str, list] = {"added": [], "removed": [], "renamed": [], "changed": []}
276
+ for fc in files:
277
+ kind = classify_component(fc.path, cfg)
278
+ old_kind = classify_component(fc.old_path, cfg) if fc.old_path else kind
279
+ if fc.status == "A" and kind:
280
+ comps["added"].append({"kind": kind, "name": component_name(fc.path), "path": fc.path})
281
+ elif fc.status == "D" and kind:
282
+ comps["removed"].append({"kind": kind, "name": component_name(fc.path), "path": fc.path})
283
+ elif fc.status == "R" and (kind or old_kind):
284
+ if component_name(fc.old_path) != component_name(fc.path):
285
+ comps["renamed"].append({"kind": kind or old_kind, "old": component_name(fc.old_path), "new": component_name(fc.path), "path": fc.path})
286
+ elif file_kinds.get(fc.path) == "code" and kind:
287
+ comps["changed"].append({"kind": kind, "name": component_name(fc.path), "path": fc.path})
288
+ elif fc.status == "M" and kind and file_kinds.get(fc.path) == "code":
289
+ comps["changed"].append({"kind": kind, "name": component_name(fc.path), "path": fc.path})
290
+
291
+ tweaks: list[dict] = []
292
+ if cfg.get("tweaks") and parent != EMPTY_TREE:
293
+ covered = {c["file"] for c in changes} | {l["file"] for l in lfs}
294
+ known = {(c["file"], c["name"]) for c in changes}
295
+ tweaks = [t for t in numeric_tweaks(unified_diff(repo, parent, meta.sha, context=0), noise)
296
+ if C.watched(t["file"], cfg) and file_kinds.get(t["file"]) == "code" and t["changes"]
297
+ and t["file"] not in covered and (t["file"], t["label"]) not in known]
298
+
299
+ modules: dict[str, dict] = defaultdict(lambda: {"added": 0, "removed": 0, "modified": 0, "lines_added": 0, "lines_deleted": 0})
300
+ for fc in files:
301
+ m = modules[top_module(fc.path)]
302
+ m[{"A": "added", "D": "removed"}.get(fc.status, "modified")] += 1
303
+ m["lines_added"] += fc.added
304
+ m["lines_deleted"] += fc.deleted
305
+ kinds = Counter(file_kinds.values())
306
+ meaningful = [fc for fc in files if file_kinds.get(fc.path) == "code"]
307
+ if not meaningful:
308
+ cls = "noop"
309
+ elif all(os.path.splitext(fc.path)[1] in DOC_EXTS for fc in meaningful):
310
+ cls = "docs"
311
+ else:
312
+ cls = "code"
313
+ return {
314
+ "sha": meta.sha, "short": meta.sha[:7], "url": f"{repo_url}/commit/{meta.sha}" if repo_url else "",
315
+ "date": meta.date, "author": meta.author, "subject": meta.subject, "body": meta.body,
316
+ "is_merge": len(meta.parents) > 1, "is_root": not meta.parents, "parent": meta.parents[0] if meta.parents else None,
317
+ "class": cls,
318
+ "stats": {"files": len(files), "files_total": len(all_files), "added_files": sum(f.status == "A" for f in files),
319
+ "removed_files": sum(f.status == "D" for f in files), "lines_added": sum(f.added for f in files),
320
+ "lines_deleted": sum(f.deleted for f in files), "noise_files": kinds.get("noise", 0),
321
+ "whitespace_files": kinds.get("whitespace", 0), "comment_only_files": kinds.get("comments", 0)},
322
+ "modules": dict(sorted(modules.items(), key=lambda kv: -(kv[1]["lines_added"] + kv[1]["lines_deleted"]))),
323
+ "files": [{"status": f.status, "path": f.path, "old_path": f.old_path, "added": f.added, "deleted": f.deleted,
324
+ "kind": file_kinds.get(f.path, "code")} for f in files],
325
+ "changes": changes, "components": comps, "tweaks": tweaks, "lfs": lfs,
326
+ }
327
+
328
+
329
+ def _mark_still_defined(repo: str, sha: str, cfg: dict, removed: list[dict]) -> None:
330
+ """A name that vanished from a changed file may still be defined in another file (it moved). Check with git grep."""
331
+ if not removed or len(removed) > 200:
332
+ return
333
+ names = sorted({c["name"].split(".")[-1] for c in removed})
334
+ pattern = "|".join(re.escape(n) for n in names)
335
+ out = run(["grep", "-l", "-E", rf"\b({pattern})\b", sha, "--", *[f":(glob){p}" for p in cfg["watch"]["paths"]]], cwd=repo, check=False)
336
+ candidates = [l.split(":", 1)[1] for l in out.splitlines() if ":" in l]
337
+ candidates = [p for p in candidates if C.watched(p, cfg)][:50]
338
+ if not candidates:
339
+ return
340
+ blobs = BlobReader(repo)
341
+ try:
342
+ found: dict[tuple[str, str], str] = {}
343
+ for path in candidates:
344
+ text = blobs.text(sha, path)
345
+ if not text:
346
+ continue
347
+ for ext, opts in extractors_for(path, text, cfg):
348
+ for name in run_extractor(ext, path, text, opts):
349
+ found.setdefault((ext, name), path)
350
+ for c in removed:
351
+ p = found.get((c["extractor"], c["name"]))
352
+ if p and p != c["file"]:
353
+ c["still_defined_in"] = p
354
+ finally:
355
+ blobs.close()
356
+
357
+
358
+ def _refine_keyed(changes: list[dict], deleted_files: set[str]) -> list[dict]:
359
+ """Structured-config post-processing.
360
+
361
+ * numeric lists of equal length that changed are split per element (max_velocity[0] 0.26 -> 0.5);
362
+ * a top-level section renamed with identical values (dwb_controller.* -> nav2_controller.*) becomes one
363
+ "renamed" record per key instead of N removed + N added;
364
+ * removals caused by deleting a whole file are flagged (file_deleted) so summaries can collapse them.
365
+ """
366
+ out: list[dict] = []
367
+ for c in changes:
368
+ if c["kind"] == "changed" and c.get("keyed") and c["type"] == "list":
369
+ lo, ln = list_items(c["old"]), list_items(c["new"])
370
+ if lo and ln and len(lo) == len(ln) and all(parse_num(x) is not None for x in lo + ln):
371
+ for i, (a, b) in enumerate(zip(lo, ln)):
372
+ if not same_value(a, b):
373
+ out.append({**c, "name": f"{c['name']}[{i}]", "old": a, "new": b, "type": "number",
374
+ "list_index": i, **change_metrics(a, b)})
375
+ continue
376
+ out.append(c)
377
+ removed = [c for c in out if c["kind"] == "removed" and c.get("keyed")]
378
+ added = [c for c in out if c["kind"] == "added" and c.get("keyed")]
379
+ split = lambda n: n.split(".", 1) if "." in n else (None, n)
380
+ add_by = {}
381
+ for c in added:
382
+ head, rest = split(c["name"])
383
+ if head is not None:
384
+ add_by.setdefault((c["extractor"], c["file"], rest), []).append(c)
385
+ pairs: dict[tuple[str, str, str], list[tuple[dict, dict]]] = {}
386
+ for c in removed:
387
+ head, rest = split(c["name"])
388
+ for a in add_by.get((c["extractor"], c["file"], rest), []):
389
+ if same_value(c["old"], a["new"]) and split(a["name"])[0] != head:
390
+ pairs.setdefault((c["file"], head, split(a["name"])[0]), []).append((c, a))
391
+ break
392
+ drop: set[int] = set()
393
+ for (f, old_head, new_head), ps in pairs.items():
394
+ total = sum(1 for c in removed if c["file"] == f and split(c["name"])[0] == old_head)
395
+ if len(ps) >= 3 and len(ps) >= 0.8 * total:
396
+ for r, a in ps:
397
+ drop.add(id(r))
398
+ a.update(kind="renamed", old_name=r["name"], old=r["old"], pct=None, ratio=None)
399
+ out = [c for c in out if id(c) not in drop]
400
+ for c in out:
401
+ if c["kind"] == "removed" and c["file"] in deleted_files:
402
+ c["file_deleted"] = True
403
+ return out
knoblog/cli.py ADDED
@@ -0,0 +1,240 @@
1
+ """knoblog command line.
2
+
3
+ knoblog run --repo <url or path> --config knoblog.yml [--out site]
4
+ knoblog init [--preset ros2] [--out knoblog.yml]
5
+ knoblog extract FILE [--extractor NAME]
6
+ knoblog presets
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import argparse
11
+ import json
12
+ import os
13
+ import re
14
+ import shutil
15
+ import sys
16
+ import time
17
+ from datetime import datetime, timezone
18
+
19
+ from . import __version__
20
+ from . import config as C
21
+ from . import posttext, render
22
+ from .analyze import CONST_LIKE, analyze_commit
23
+ from .extractors import REGISTRY, auto_extractors, run_extractor
24
+ from .gitio import EMPTY_TREE, ensure_clone, head_rev, list_commits, run, unified_diff
25
+ from .summarize import deterministic, llm_config, llm_summarize, load_note
26
+
27
+ PKG = os.path.dirname(os.path.abspath(__file__))
28
+
29
+
30
+ def log(msg: str) -> None:
31
+ print(msg, file=sys.stderr, flush=True)
32
+
33
+
34
+ def browse_url(repo: str) -> str:
35
+ """https URL for commit links from a clone URL (GitHub/GitLab style), or ''."""
36
+ repo = repo.strip()
37
+ m = re.match(r"^git@([^:]+):(.+?)(\.git)?/?$", repo)
38
+ if m:
39
+ return f"https://{m.group(1)}/{m.group(2)}"
40
+ m = re.match(r"^(https?://[^/]+/.+?)(\.git)?/?$", repo)
41
+ return m.group(1) if m else ""
42
+
43
+
44
+ def resolve_repo(repo: str, cfg: dict, args) -> tuple[str, str]:
45
+ """Return (local git dir, browsable url)."""
46
+ if os.path.isdir(repo):
47
+ url = cfg.get("repo_url") or browse_url(run(["remote", "get-url", "origin"], cwd=repo, check=False).strip())
48
+ if not args.no_fetch and run(["remote"], cwd=repo, check=False).strip():
49
+ log(f"[fetch] {repo}")
50
+ run(["fetch", "--quiet", "origin"], cwd=repo, check=False)
51
+ return repo, url
52
+ cache = args.cache_dir or os.path.join(".knoblog-cache")
53
+ dest = os.path.join(cache, re.sub(r"[^A-Za-z0-9._-]+", "_", browse_url(repo).split("//")[-1] or repo))
54
+ if not (args.no_fetch and os.path.isdir(dest)):
55
+ log(f"[clone] {repo} -> {dest} ({'blob-less partial' if cfg['clone']['partial'] else 'full'})")
56
+ ensure_clone(repo, dest, cfg.get("branch"), partial=cfg["clone"]["partial"])
57
+ return dest, cfg.get("repo_url") or browse_url(repo)
58
+
59
+
60
+ def summarise(e: dict, cfg: dict, repo: str, meta, llm, state: dict) -> None:
61
+ auto = deterministic(e, cfg)
62
+ e["auto"] = auto
63
+ sc = cfg["summaries"]
64
+ base = cfg.get("_dir") or "."
65
+ rel = lambda p: p if not p or os.path.isabs(p) else os.path.join(base, p)
66
+ note = load_note(rel(sc.get("notes_dir")), e["short"])
67
+ cached = load_note(rel(sc.get("llm_cache_dir")), e["short"])
68
+ mk = lambda src, n: {"source": src, "title": n["title"] or auto["headline"], "markdown": n["markdown"], "bullets": auto["bullets"], "tags": auto["tags"]}
69
+ if note:
70
+ e["summary"] = mk("note", note)
71
+ elif cached:
72
+ e["summary"] = mk("llm", cached)
73
+ elif llm and e["class"] != "noop" and state["llm_calls"] < int(sc.get("llm_limit") or 0) and sc.get("llm_cache_dir"):
74
+ state["llm_calls"] += 1
75
+ try:
76
+ parent = meta.parents[0] if meta.parents else EMPTY_TREE
77
+ text = llm_summarize(llm, e, unified_diff(repo, parent, meta.sha, context=2))
78
+ d = rel(sc["llm_cache_dir"])
79
+ os.makedirs(d, exist_ok=True)
80
+ with open(os.path.join(d, f"{e['short']}.md"), "w", encoding="utf-8") as f:
81
+ f.write(text + "\n")
82
+ e["summary"] = mk("llm", load_note(d, e["short"]))
83
+ except Exception as exc: # deterministic fallback
84
+ log(f"[llm] {e['short']} failed ({exc}); using deterministic summary")
85
+ if "summary" not in e:
86
+ e["summary"] = {"source": "auto", "title": auto["headline"], "markdown": "", "bullets": auto["bullets"], "tags": auto["tags"]}
87
+
88
+
89
+ def history_records(entries: list[dict], cfg: dict) -> list[dict]:
90
+ """Flat change records for params.html / changes.json (bulk additions are left to entry pages)."""
91
+ out = []
92
+ limit = cfg["post"]["bulk_added_limit"]
93
+ for e in entries:
94
+ bulk = e["is_root"] or sum(c["kind"] == "added" for c in e["changes"]) > limit
95
+ for c in e["changes"]:
96
+ if bulk and c["kind"] in ("added", "removed"):
97
+ continue
98
+ if (c["kind"] in ("changed", "moved") and c["old"] == c["new"]) or c["kind"] == "renamed" or c.get("file_deleted"):
99
+ continue
100
+ out.append({**render.change_record(e, c), "pct": c.get("pct")})
101
+ return out
102
+
103
+
104
+ def cmd_run(args) -> int:
105
+ t0 = time.time()
106
+ overrides: dict = {}
107
+ if args.site_url is not None:
108
+ overrides.setdefault("site", {})["url"] = args.site_url
109
+ if args.max_commits:
110
+ overrides.setdefault("history", {})["max_commits"] = args.max_commits
111
+ if args.since:
112
+ overrides.setdefault("history", {})["since"] = args.since
113
+ if args.no_llm:
114
+ overrides.setdefault("summaries", {})["llm"] = False
115
+ if args.branch:
116
+ overrides["branch"] = args.branch
117
+ cfg = C.load(args.config, args.preset, overrides)
118
+ repo_arg = args.repo or cfg.get("repo")
119
+ if not repo_arg:
120
+ raise SystemExit("no repository: pass --repo <url or path> or set `repo:` in the config")
121
+ repo, url = resolve_repo(repo_arg, cfg, args)
122
+ cfg["repo_url"] = url
123
+ cfg["name"] = cfg.get("name") or (url or os.path.abspath(repo)).rstrip("/").split("/")[-1]
124
+ h = cfg["history"]
125
+ rev = head_rev(repo, cfg.get("branch"))
126
+ paths = cfg["watch"]["paths"] if h.get("limit_log_to_watched_paths") and cfg["watch"]["paths"] != ["**"] else None
127
+ commits = list_commits(repo, rev, paths, h.get("max_commits"), h.get("since"), h.get("first_parent", True))
128
+ log(f"[walk] {len(commits)} commits (rev {rev[:7]}, newest {h.get('max_commits')}{', path-limited' if paths else ''})")
129
+ llm = llm_config() if cfg["summaries"].get("llm") else None
130
+ entries, state = [], {"llm_calls": 0}
131
+ for meta in commits:
132
+ e = analyze_commit(repo, meta, cfg, url)
133
+ summarise(e, cfg, repo, meta, llm, state)
134
+ entries.append(e)
135
+ if args.verbose:
136
+ log(f"[entry] {e['date'][:10]} {e['short']} {e['class']:5} {len(e['changes']):4} changes {e['summary']['title'][:80]}")
137
+ posts = posttext.post_text_for(entries, cfg) if cfg["post"].get("enabled") else []
138
+ posts_by_sha = {p["sha"]: p for p in posts}
139
+ now = datetime.now(timezone.utc)
140
+ meta = {"generated_at": now.strftime("%Y-%m-%d %H:%M UTC"), "generated_iso": now.strftime("%Y-%m-%dT%H:%M:%SZ"), "head_sha": rev}
141
+ site = render.Site(cfg, meta)
142
+ out = args.out
143
+ if os.path.isdir(out):
144
+ shutil.rmtree(out)
145
+ os.makedirs(os.path.join(out, "entries"))
146
+ shutil.copy(os.path.join(PKG, "static", "style.css"), os.path.join(out, "style.css"))
147
+ real = [e for e in entries if e["class"] != "noop"]
148
+ noops = [e for e in entries if e["class"] == "noop"]
149
+ newest = list(reversed(real))
150
+ records = history_records(real, cfg)
151
+ newest_records = list(reversed(records))
152
+ seen, current = set(), []
153
+ for r in newest_records:
154
+ if r["kind"] in ("changed", "moved") and r["extractor"] not in CONST_LIKE and (r["name"], r["file"]) not in seen:
155
+ seen.add((r["name"], r["file"]))
156
+ current.append(r)
157
+ render.write(os.path.join(out, "index.html"), site.page(site.title, site.index(newest, list(reversed(noops)), current[:40])))
158
+ render.write(os.path.join(out, "params.html"), site.page("Parameter history · " + site.title, site.params(newest_records)))
159
+ render.write(os.path.join(out, "posts.html"), site.page("Post text · " + site.title, site.posts_page(list(reversed(posts)))))
160
+ for i, e in enumerate(real):
161
+ prev = real[i - 1] if i else None
162
+ nxt = real[i + 1] if i + 1 < len(real) else None
163
+ title = f"{render.fmt_date(e['date'])}: {e['summary']['title'].replace('`', '')}"
164
+ render.write(os.path.join(out, "entries", f"{e['short']}.html"),
165
+ site.page(title + " · " + site.title, site.detail(e, prev, nxt, posts_by_sha.get(e["sha"])), depth=1, description=title))
166
+ render.write(os.path.join(out, "feed.xml"), site.atom(newest))
167
+ render.write(os.path.join(out, "feed.json"), json.dumps(site.json_feed(newest, posts_by_sha), indent=1, ensure_ascii=False))
168
+ for r in newest_records:
169
+ r.pop("pct", None)
170
+ render.write(os.path.join(out, "changes.json"), json.dumps({"generator": f"knoblog {__version__}", "repo": url, "head": rev,
171
+ "generated": meta["generated_iso"], "changes": newest_records}, indent=1, ensure_ascii=False))
172
+ render.write(os.path.join(out, "posts.json"), json.dumps({"generator": f"knoblog {__version__}", "note": "Text only; knoblog never posts.",
173
+ "posts": list(reversed(posts))}, indent=1, ensure_ascii=False))
174
+ render.write(os.path.join(out, "posts.txt"), render.posts_txt(list(reversed(posts))))
175
+ slim = [{k: e[k] for k in ("sha", "short", "url", "date", "class", "is_root", "is_merge", "stats", "summary", "changes", "components", "lfs")}
176
+ for e in newest]
177
+ render.write(os.path.join(out, "entries.json"), json.dumps({"generated": meta["generated_iso"], "head": rev, "entries": slim}, indent=1, ensure_ascii=False))
178
+ render.write(os.path.join(out, ".nojekyll"), "")
179
+ ready = sum(p["status"] == "ready" for p in posts)
180
+ log(f"[done] {len(real)} entries ({len(noops)} no-op), {len(records)} value changes, {ready} post-ready -> {out} in {time.time() - t0:.1f}s")
181
+ if args.github_output and os.environ.get("GITHUB_OUTPUT"):
182
+ with open(os.environ["GITHUB_OUTPUT"], "a", encoding="utf-8") as f:
183
+ f.write(f"entries={len(real)}\nchanges={len(records)}\nposts_ready={ready}\nsite_dir={out}\n")
184
+ return 0
185
+
186
+
187
+ def cmd_init(args) -> int:
188
+ if os.path.exists(args.out) and not args.force:
189
+ raise SystemExit(f"{args.out} exists (use --force)")
190
+ src = os.path.join(PKG, "presets", "_template.yml")
191
+ text = open(src, encoding="utf-8").read().replace("extends: generic", f"extends: {args.preset}")
192
+ render.write(args.out, text)
193
+ log(f"wrote {args.out} (extends preset {args.preset!r}); edit watch.paths, then: knoblog run --repo <url or path> --config {args.out}")
194
+ return 0
195
+
196
+
197
+ def cmd_extract(args) -> int:
198
+ text = open(args.file, encoding="utf-8", errors="replace").read()
199
+ names = [args.extractor] if args.extractor else auto_extractors(args.file, text)
200
+ out = {n: run_extractor(n, args.file, text, {}) for n in names}
201
+ print(json.dumps(out, indent=1, ensure_ascii=False))
202
+ return 0
203
+
204
+
205
+ def main(argv: list[str] | None = None) -> int:
206
+ p = argparse.ArgumentParser(prog="knoblog", description="Plain-English changelogs + X-ready post text for parameter/config changes in a git repo.")
207
+ p.add_argument("--version", action="version", version=f"knoblog {__version__}")
208
+ sub = p.add_subparsers(dest="cmd", required=True)
209
+ r = sub.add_parser("run", help="analyse history and write the site + feeds")
210
+ r.add_argument("--repo", help="git URL or local path (overrides `repo:` in the config)")
211
+ r.add_argument("--config", "-c", help="knoblog.yml (or a preset name)")
212
+ r.add_argument("--preset", help="preset to start from (ros2, px4, ardupilot, x-algorithm, generic)")
213
+ r.add_argument("--out", "-o", default="site")
214
+ r.add_argument("--branch")
215
+ r.add_argument("--max-commits", type=int)
216
+ r.add_argument("--since", help="only commits after this date (YYYY-MM-DD)")
217
+ r.add_argument("--site-url", help="public URL of the site (used in feeds and post links)")
218
+ r.add_argument("--cache-dir", help="where remote repos are cloned (default .knoblog-cache)")
219
+ r.add_argument("--no-fetch", action="store_true", help="use the existing clone as-is")
220
+ r.add_argument("--no-llm", action="store_true", help="never call an LLM")
221
+ r.add_argument("--github-output", action="store_true", help="write counts to $GITHUB_OUTPUT")
222
+ r.add_argument("-v", "--verbose", action="store_true")
223
+ r.set_defaults(func=cmd_run)
224
+ i = sub.add_parser("init", help="write a starter knoblog.yml")
225
+ i.add_argument("--preset", default="generic")
226
+ i.add_argument("--out", default="knoblog.yml")
227
+ i.add_argument("--force", action="store_true")
228
+ i.set_defaults(func=cmd_init)
229
+ x = sub.add_parser("extract", help="print the values knoblog extracts from one file")
230
+ x.add_argument("file")
231
+ x.add_argument("--extractor", choices=sorted(REGISTRY))
232
+ x.set_defaults(func=cmd_extract)
233
+ ps = sub.add_parser("presets", help="list built-in presets")
234
+ ps.set_defaults(func=lambda a: print("\n".join(n for n in C.presets() if not n.startswith("_"))) or 0)
235
+ args = p.parse_args(argv)
236
+ return args.func(args)
237
+
238
+
239
+ if __name__ == "__main__":
240
+ raise SystemExit(main())