create-caspian-app 1.5.8 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,719 +1,719 @@
1
- """App formatter: Python via ruff, authored markup via djLint (`npm run format`).
2
-
3
- Two surfaces, one command:
4
-
5
- 1. **Python** -- `ruff format` over `main.py`, `src/**`, `settings/*.py`,
6
- `tests/**`. A Python formatter never rewrites string *contents*, so the
7
- markup inside `html(r\"\"\"...\"\"\")` is byte-preserved; this was verified
8
- across every block in `src/` before adopting it.
9
-
10
- 2. **Markup** -- the template inside every `html(r\"\"\"...\"\"\")` call, formatted
11
- with djLint and then *proved* unchanged in rendering before being written.
12
-
13
- Why djLint, and why the proof
14
- -----------------------------
15
- This project layers four dialects in one string: HTML, Jinja `{{ }}`/`{% %}`,
16
- PulsePoint `{ }`, and JavaScript inside `<script>` -- and they nest, e.g.
17
- `class="... {currentUrl === '{{ item['href'] }}' ? 'a' : 'b'}"`. Prettier has no
18
- Jinja awareness: it de-indents `{% for %}` blocks to column 0 and joins
19
- `{% endfor %} {% endfor %}` onto one line. djLint is Jinja-aware and, critically,
20
- does not reflow text -- so PulsePoint expressions survive intact.
21
-
22
- djLint is still a general HTML formatter, and it will make changes that are
23
- correct for HTML but wrong here. The one that matters: it inserts a newline
24
- between a block tag and an adjacent inline or `<x-*>` tag, which renders as a
25
- visible space, because a custom element's `display` is set by CSS the formatter
26
- cannot see. So no block is trusted -- each is proved equivalent by
27
- `_markup_equivalence` before it is written, and skipped with a reason if not.
28
-
29
- Two regions are additionally masked away before djLint runs, so they are
30
- preserved byte-for-byte by construction rather than by proof:
31
-
32
- * `<script>` / `<style>` -- djLint reads `/>` inside a JS regex as a tag
33
- delimiter and rewrites `.replace(/>/g, …)` into `.replace( />/g, …)`.
34
- * `<pre>` / `<textarea>` -- whitespace there renders literally.
35
-
36
- Usage (from the project root):
37
-
38
- python settings/format.py # format Python + markup
39
- python settings/format.py --check # report only; exit 1 if work remains
40
- python settings/format.py --markup # markup only, skip ruff format
41
- python settings/format.py --python # ruff format only, skip markup
42
-
43
- `npm run check:fix` runs this first, so formatting settles before the fixer and
44
- the gate look at the code.
45
- """
46
-
47
- from __future__ import annotations
48
-
49
- import argparse
50
- import ast
51
- import re
52
- import subprocess
53
- import sys
54
- import tempfile
55
- from dataclasses import dataclass, field
56
- from functools import lru_cache
57
- from pathlib import Path
58
-
59
- import _markup_equivalence as eq
60
-
61
- PROJECT_ROOT = Path(__file__).resolve().parents[1]
62
- SCAN_ROOT = PROJECT_ROOT / "src"
63
-
64
- # Generated or vendored trees that are not hand-authored.
65
- EXCLUDED_PARTS = {"__pycache__", "node_modules", ".venv", "prisma"}
66
-
67
- # Paths handed to `ruff format`. Mirrors the gate's Python surface.
68
- PYTHON_TARGETS = ("main.py", "src", "settings", "tests")
69
-
70
- # Excluded from formatting. The test is NOT "was this generated" -- it is
71
- # "is this regenerated wholesale and never hand-edited". The Prisma client is
72
- # rewritten in full by `prisma generate` and is never touched by hand, so
73
- # formatting it is pure churn that the next generate undoes; it also holds no
74
- # markup, so nothing is lost by leaving it alone.
75
- #
76
- # Component libraries under `src/lib/**` are deliberately NOT listed here even
77
- # though a CLI first installed them: they are hand-maintained in this workspace,
78
- # and they hold the majority of the repo's markup blocks. Excluding them would
79
- # remove most of the formatter's reach.
80
- PYTHON_FORMAT_EXCLUDE = ("src/lib/prisma",)
81
-
82
- # djLint settings. Passed explicitly rather than read from pyproject, because
83
- # blocks are formatted in a temp directory outside the project.
84
- DJLINT_INDENT = "2"
85
- DJLINT_MAX_LINE = "120"
86
-
87
- # Register `<x-*>` component tags with djLint. Without this it does not know
88
- # them, treats them as inline, and leaves a component tree flat:
89
- #
90
- # <x-shell>
91
- # <x-brand />
92
- # </x-shell>
93
- #
94
- # Registered, they nest like any other container. This does not weaken the
95
- # safety check -- `_markup_equivalence` still treats `<x-*>` as inline, so any
96
- # block where the new indentation would actually add rendered whitespace is
97
- # still skipped.
98
- DJLINT_CUSTOM_HTML = r"x-[\w-]+"
99
-
100
- OPAQUE = re.compile(r"(<(script|style|pre|textarea)\b[^>]*>)(.*?)(</\2\s*>)", re.I | re.S)
101
-
102
- try:
103
- sys.stdout.reconfigure(encoding="utf-8", errors="replace") # type: ignore[union-attr]
104
- except AttributeError, ValueError:
105
- pass
106
-
107
- _TTY = sys.stdout.isatty()
108
-
109
-
110
- def _c(code: str, text: str) -> str:
111
- return f"\033[{code}m{text}\033[0m" if _TTY else text
112
-
113
-
114
- def red(t: str) -> str:
115
- return _c("31", t)
116
-
117
-
118
- def green(t: str) -> str:
119
- return _c("32", t)
120
-
121
-
122
- def yellow(t: str) -> str:
123
- return _c("33", t)
124
-
125
-
126
- def bold(t: str) -> str:
127
- return _c("1", t)
128
-
129
-
130
- # ---------------------------------------------------------------------------
131
- # masking
132
- # ---------------------------------------------------------------------------
133
-
134
-
135
- def mask_opaque(markup: str) -> tuple[str, dict[str, str]]:
136
- """Replace `<script>`/`<style>`/`<pre>`/`<textarea>` bodies with tokens."""
137
- store: dict[str, str] = {}
138
-
139
- def sub(m: re.Match[str]) -> str:
140
- key = f"PPMASK{len(store):04d}Z"
141
- store[key] = m.group(3)
142
- return f"{m.group(1)}{key}{m.group(4)}"
143
-
144
- return OPAQUE.sub(sub, markup), store
145
-
146
-
147
- def unmask_opaque(markup: str, store: dict[str, str]) -> str:
148
- """Restore masked bodies, re-indenting code to its new nesting depth.
149
-
150
- `<pre>`/`<textarea>` are restored verbatim -- their whitespace renders.
151
- `<script>`/`<style>` are re-indented as a block, which cannot change what
152
- the code does but keeps the output readable.
153
- """
154
- for key, body in store.items():
155
- pattern = re.compile(
156
- r"([ \t]*)(<(script|style|pre|textarea)\b[^>]*>)" + key + r"(</\3\s*>)",
157
- re.I,
158
- )
159
-
160
- def repl(m: re.Match[str], body: str = body) -> str:
161
- indent, open_tag, tag, close_tag = (
162
- m.group(1),
163
- m.group(2),
164
- m.group(3).lower(),
165
- m.group(4),
166
- )
167
- if tag in ("pre", "textarea"):
168
- return f"{indent}{open_tag}{body}{close_tag}"
169
- lines = body.splitlines()
170
- while lines and not lines[0].strip():
171
- lines.pop(0)
172
- while lines and not lines[-1].strip():
173
- lines.pop()
174
- if not lines:
175
- return f"{indent}{open_tag}{close_tag}"
176
- base = min((len(ln) - len(ln.lstrip()) for ln in lines if ln.strip()), default=0)
177
- inner = indent + " "
178
- rendered = "\n".join((inner + ln[base:].rstrip()) if ln.strip() else "" for ln in lines)
179
- return f"{indent}{open_tag}\n{rendered}\n{indent}{close_tag}"
180
-
181
- markup = pattern.sub(repl, markup, count=1)
182
- return markup
183
-
184
-
185
- # ---------------------------------------------------------------------------
186
- # block discovery
187
- # ---------------------------------------------------------------------------
188
-
189
-
190
- @dataclass
191
- class Block:
192
- """One `html(r\"\"\"...\"\"\")` template argument in a Python file."""
193
-
194
- path: Path
195
- lineno: int
196
- start: int # byte offset of the string literal's opening quote
197
- end: int # byte offset just past its closing quote
198
- prefix: str # the literal's opening delimiter, e.g. `r\"\"\"`
199
- quote: str # the closing delimiter
200
- source: str # the markup itself
201
-
202
-
203
- def _rel(path: Path) -> str:
204
- """Project-relative path for reports, tolerating a path outside the root."""
205
- try:
206
- return path.relative_to(PROJECT_ROOT).as_posix()
207
- except ValueError:
208
- return path.as_posix()
209
-
210
-
211
- def _iter_python_files() -> list[Path]:
212
- if not SCAN_ROOT.exists():
213
- return []
214
- return sorted(p for p in SCAN_ROOT.rglob("*.py") if not EXCLUDED_PARTS.intersection(p.parts))
215
-
216
-
217
- def find_blocks(path: Path) -> list[Block]:
218
- """Every raw triple-quoted `html(...)` template argument in one file.
219
-
220
- Only the `html(r\"\"\"...\"\"\")` form is touched. That is the single markup
221
- entrypoint the `templates` gate already enforces (`html-form`), so any other
222
- shape is a gate failure to fix rather than something to reformat.
223
- """
224
- try:
225
- text = path.read_text(encoding="utf-8")
226
- tree = ast.parse(text)
227
- except OSError, UnicodeDecodeError, SyntaxError:
228
- return []
229
-
230
- lines = text.splitlines(keepends=True)
231
- offsets = [0]
232
- for line in lines:
233
- offsets.append(offsets[-1] + len(line))
234
-
235
- blocks: list[Block] = []
236
- for node in ast.walk(tree):
237
- if not isinstance(node, ast.Call):
238
- continue
239
- name = getattr(node.func, "id", None) or getattr(node.func, "attr", None)
240
- if name != "html" or not node.args:
241
- continue
242
- arg = node.args[0]
243
- if not (isinstance(arg, ast.Constant) and isinstance(arg.value, str)):
244
- continue
245
- if arg.end_lineno is None or arg.end_col_offset is None:
246
- continue
247
- start = offsets[arg.lineno - 1] + arg.col_offset
248
- end = offsets[arg.end_lineno - 1] + arg.end_col_offset
249
- literal = text[start:end]
250
- match = re.match(r'^([rR]?)("""|\'\'\')', literal)
251
- if not match or not match.group(1):
252
- continue # not the raw triple-quoted form; `templates` reports it
253
- prefix, quote = match.group(0), match.group(2)
254
- if not literal.endswith(quote):
255
- continue
256
- blocks.append(
257
- Block(
258
- path=path,
259
- lineno=arg.lineno,
260
- start=start,
261
- end=end,
262
- prefix=prefix,
263
- quote=quote,
264
- source=literal[len(prefix) : -len(quote)],
265
- )
266
- )
267
- return blocks
268
-
269
-
270
- # ---------------------------------------------------------------------------
271
- # djLint
272
- # ---------------------------------------------------------------------------
273
-
274
-
275
- def djlint_batch(sources: list[str]) -> list[str | None]:
276
- """Format many markup fragments in a single djLint run.
277
-
278
- djLint costs roughly a second of interpreter start-up per invocation, and
279
- this repo has well over 500 blocks. Writing them all into one temp directory
280
- and reformatting it once turns minutes into seconds.
281
- """
282
- if not sources:
283
- return []
284
- results: list[str | None] = [None] * len(sources)
285
- with tempfile.TemporaryDirectory(prefix="pp-format-") as tmp:
286
- root = Path(tmp)
287
- for index, text in enumerate(sources):
288
- (root / f"{index:05d}.html").write_text(text, encoding="utf-8")
289
- proc = subprocess.run(
290
- [
291
- sys.executable,
292
- "-m",
293
- "djlint",
294
- str(root),
295
- "--reformat",
296
- "--profile",
297
- "jinja",
298
- "--indent",
299
- DJLINT_INDENT,
300
- "--max-line-length",
301
- DJLINT_MAX_LINE,
302
- "--preserve-blank-lines",
303
- "--custom-html",
304
- DJLINT_CUSTOM_HTML,
305
- ],
306
- capture_output=True,
307
- text=True,
308
- encoding="utf-8",
309
- errors="replace",
310
- )
311
- # djLint exits 1 when it reformatted something; only >1 is a real error.
312
- if proc.returncode > 1:
313
- return results
314
- for index in range(len(sources)):
315
- try:
316
- results[index] = (root / f"{index:05d}.html").read_text(encoding="utf-8")
317
- except OSError:
318
- results[index] = None
319
- return results
320
-
321
-
322
- def format_markup(source: str, formatted_skeleton: str | None, store: dict[str, str]) -> str | None:
323
- if formatted_skeleton is None:
324
- return None
325
- return unmask_opaque(formatted_skeleton, store)
326
-
327
-
328
- # ---------------------------------------------------------------------------
329
- # splicing
330
- # ---------------------------------------------------------------------------
331
-
332
-
333
- def render_literal(block: Block, markup: str) -> str | None:
334
- """Rebuild the Python string literal around freshly formatted markup.
335
-
336
- Returns None when the result could not be spliced back safely.
337
- """
338
- body = markup.strip("\n")
339
- if block.quote in body:
340
- return None # would terminate the literal early
341
- if body.endswith("\\"):
342
- return None # a trailing backslash escapes the closing quote
343
- original_multiline = "\n" in block.source.strip("\n") or block.source.startswith("\n")
344
- if "\n" in body or original_multiline:
345
- body = f"\n{body}\n"
346
- return f"{block.prefix}{body}{block.quote}"
347
-
348
-
349
- @lru_cache(maxsize=None)
350
- def jinja_classes(path: Path) -> dict:
351
- """Class lists this file hides behind an expression.
352
-
353
- Recovered from the file's own source so the oracle can see the display of a
354
- root written as `<div {{ attributes }}>` or `class="{helper()}"`. Cached
355
- because every block in a file shares one answer.
356
- """
357
- try:
358
- return eq.class_hints(path.read_text(encoding="utf-8"))
359
- except OSError:
360
- return {}
361
-
362
-
363
- @dataclass
364
- class Skip:
365
- path: str
366
- lineno: int
367
- reason: str
368
-
369
-
370
- @dataclass
371
- class MarkupReport:
372
- scanned: int = 0
373
- formatted: int = 0
374
- already: int = 0
375
- files_changed: int = 0
376
- skips: list[Skip] = field(default_factory=list)
377
- error: str = ""
378
-
379
-
380
- def format_markup_blocks(*, write: bool) -> MarkupReport:
381
- report = MarkupReport()
382
-
383
- files = _iter_python_files()
384
- per_file: dict[Path, list[Block]] = {}
385
- flat: list[tuple[Path, Block]] = []
386
- for path in files:
387
- blocks = find_blocks(path)
388
- if blocks:
389
- per_file[path] = blocks
390
- flat.extend((path, b) for b in blocks)
391
-
392
- report.scanned = len(flat)
393
- if not flat:
394
- return report
395
-
396
- masked: list[str] = []
397
- stores: list[dict[str, str]] = []
398
- for _, block in flat:
399
- skeleton, store = mask_opaque(block.source)
400
- masked.append(skeleton)
401
- stores.append(store)
402
-
403
- outputs = djlint_batch(masked)
404
- if all(o is None for o in outputs):
405
- report.error = "djLint did not run. Install it with `uv sync --group dev`."
406
- return report
407
-
408
- # Decide each block, then apply per file from the bottom up so earlier
409
- # offsets stay valid.
410
- decisions: dict[Path, list[tuple[Block, str]]] = {}
411
- for (path, block), skeleton, store in zip(flat, outputs, stores):
412
- rel = _rel(path)
413
- formatted = format_markup(block.source, skeleton, store)
414
- if formatted is None:
415
- report.skips.append(Skip(rel, block.lineno, "djLint could not format this block"))
416
- continue
417
- if formatted.strip("\n") == block.source.strip("\n"):
418
- report.already += 1
419
- continue
420
- same, why = eq.equivalent(block.source, formatted, jinja_classes(path))
421
- if not same:
422
- report.skips.append(Skip(rel, block.lineno, why))
423
- continue
424
- literal = render_literal(block, formatted)
425
- if literal is None:
426
- report.skips.append(Skip(rel, block.lineno, "could not be spliced back safely"))
427
- continue
428
- decisions.setdefault(path, []).append((block, literal))
429
- report.formatted += 1
430
-
431
- report.files_changed = len(decisions)
432
- if not write:
433
- return report
434
-
435
- for path, items in decisions.items():
436
- text = path.read_text(encoding="utf-8")
437
- for block, literal in sorted(items, key=lambda i: i[0].start, reverse=True):
438
- text = text[: block.start] + literal + text[block.end :]
439
- # Never leave a file that no longer parses, or whose markup did not land
440
- # exactly as intended.
441
- try:
442
- ast.parse(text)
443
- except SyntaxError:
444
- report.skips.append(Skip(_rel(path), 0, "rewrite would not parse; file left unchanged"))
445
- report.formatted -= len(items)
446
- report.files_changed -= 1
447
- continue
448
- path.write_text(text, encoding="utf-8")
449
-
450
- return report
451
-
452
-
453
- # ---------------------------------------------------------------------------
454
- # ruff format
455
- # ---------------------------------------------------------------------------
456
-
457
-
458
- def _ruff_format_cmd(targets: list[str]) -> list[str]:
459
- cmd = [sys.executable, "-m", "ruff", "format", *targets]
460
- for excluded in PYTHON_FORMAT_EXCLUDE:
461
- cmd += ["--exclude", excluded]
462
- return cmd
463
-
464
-
465
- def hug_html_call_openings(paths: list[Path], *, write: bool) -> list[Path]:
466
- """Put `html(r\"\"\"` back on one line, so the markup starts at line 1.
467
-
468
- `ruff format` always explodes a call whose first argument is a multiline
469
- string when the call has other arguments, producing:
470
-
471
- return html(
472
- r\"\"\"
473
- <div>…
474
-
475
- That buries the template one level deeper and separates `html(` from the
476
- markup it opens. The house style is `html(r\"\"\"` with the markup starting
477
- on the next line, so this runs *after* ruff and closes the gap.
478
-
479
- Ruff will re-split these on its next run, in default and preview style
480
- alike, so the two steps must always run as a pair -- which is why
481
- `--check` re-runs the whole pipeline rather than calling
482
- `ruff format --check` directly.
483
-
484
- Returns the paths that needed the fix.
485
- """
486
- touched: list[Path] = []
487
- for path in paths:
488
- try:
489
- text = path.read_text(encoding="utf-8")
490
- except OSError, UnicodeDecodeError:
491
- continue
492
- if "html(" not in text:
493
- continue
494
- try:
495
- tree = ast.parse(text)
496
- except SyntaxError:
497
- continue
498
-
499
- lines = text.splitlines(keepends=True)
500
- offsets = [0]
501
- for line in lines:
502
- offsets.append(offsets[-1] + len(line))
503
-
504
- cuts: list[tuple[int, int]] = []
505
- for node in ast.walk(tree):
506
- if not isinstance(node, ast.Call):
507
- continue
508
- name = getattr(node.func, "id", None) or getattr(node.func, "attr", None)
509
- if name != "html" or not node.args:
510
- continue
511
- arg = node.args[0]
512
- if not (isinstance(arg, ast.Constant) and isinstance(arg.value, str)):
513
- continue
514
- func_line, func_col = node.func.end_lineno, node.func.end_col_offset
515
- if func_line is None or func_col is None:
516
- continue
517
- paren = text.find("(", offsets[func_line - 1] + func_col)
518
- if paren == -1:
519
- continue
520
- literal_start = offsets[arg.lineno - 1] + arg.col_offset
521
- gap = text[paren + 1 : literal_start]
522
- # Only close a gap that is pure whitespace spanning a line break;
523
- # anything else means this is not the shape ruff produced.
524
- if gap and gap.strip() == "" and "\n" in gap:
525
- cuts.append((paren + 1, literal_start))
526
-
527
- if not cuts:
528
- continue
529
- touched.append(path)
530
- if not write:
531
- continue
532
- for start, end in sorted(cuts, reverse=True):
533
- text = text[:start] + text[end:]
534
- try:
535
- ast.parse(text)
536
- except SyntaxError:
537
- continue # leave the file as ruff wrote it rather than risk it
538
- path.write_text(text, encoding="utf-8")
539
- return touched
540
-
541
-
542
- def _python_files_for_hug() -> list[Path]:
543
- """Every formatted Python file that could contain an `html(...)` call."""
544
- found: list[Path] = []
545
- for target in PYTHON_TARGETS:
546
- base = PROJECT_ROOT / target
547
- if not base.exists():
548
- continue
549
- candidates = [base] if base.is_file() else sorted(base.rglob("*.py"))
550
- for path in candidates:
551
- rel = path.relative_to(PROJECT_ROOT).as_posix()
552
- if EXCLUDED_PARTS.intersection(path.parts):
553
- continue
554
- if any(rel.startswith(x) for x in PYTHON_FORMAT_EXCLUDE):
555
- continue
556
- found.append(path)
557
- return found
558
-
559
-
560
- _PYTHON_PIPELINE_ROUNDS = 4
561
-
562
-
563
- def _snapshot(files: list[Path]) -> dict[Path, bytes]:
564
- out: dict[Path, bytes] = {}
565
- for path in files:
566
- try:
567
- out[path] = path.read_bytes()
568
- except OSError:
569
- continue
570
- return out
571
-
572
-
573
- def _python_pipeline(root: Path, targets: list[str], files: list[Path]) -> int:
574
- """Run `ruff format` + the `html(` hug until the tree stops changing.
575
-
576
- One pass is not enough, because the two steps feed each other. Ruff splits
577
- `html(` off its template; the hug rejoins it; and for a call whose template
578
- is the *only* argument, that rejoin then lets ruff pull the closing `)` up
579
- on its next run. Iterating to a fixed point is what makes the result stable
580
- -- and it is what lets `--check` replay this exact function against a mirror
581
- of the tree, instead of trusting `ruff format --check`, which would flag
582
- every hugged call as unformatted because ruff is what splits them.
583
-
584
- It settles in a couple of rounds: a multi-argument call lands on
585
- ruff-splits-then-hug-rejoins, which is a fixed point of the *pair* even
586
- though neither step is idempotent alone.
587
-
588
- Returns the number of files whose bytes changed.
589
- """
590
- initial = _snapshot(files)
591
- for _ in range(_PYTHON_PIPELINE_ROUNDS):
592
- before = _snapshot(files)
593
- subprocess.run(
594
- _ruff_format_cmd(targets),
595
- cwd=root,
596
- capture_output=True,
597
- text=True,
598
- encoding="utf-8",
599
- errors="replace",
600
- )
601
- hug_html_call_openings(files, write=True)
602
- if _snapshot(files) == before:
603
- break
604
- final = _snapshot(files)
605
- return sum(1 for path, data in initial.items() if final.get(path) != data)
606
-
607
-
608
- def run_ruff_format(*, write: bool) -> tuple[bool, str]:
609
- targets = [t for t in PYTHON_TARGETS if (PROJECT_ROOT / t).exists()]
610
- files = _python_files_for_hug()
611
-
612
- if write:
613
- changed = _python_pipeline(PROJECT_ROOT, targets, files)
614
- total = len(files)
615
- if changed:
616
- return True, f"{changed} file(s) reformatted, {total - changed} left unchanged"
617
- return True, f"{total} files already formatted"
618
-
619
- with tempfile.TemporaryDirectory(prefix="pp-pyfmt-") as tmp:
620
- root = Path(tmp)
621
- # Ruff resolves `include`/`exclude` relative to the config's directory,
622
- # so the mirror needs the same relative layout and its own copy of the
623
- # config, or every file is filtered out as "not part of the project".
624
- (root / "pyproject.toml").write_bytes((PROJECT_ROOT / "pyproject.toml").read_bytes())
625
- mirrored: list[tuple[Path, Path]] = []
626
- for path in files:
627
- dest = root / path.relative_to(PROJECT_ROOT)
628
- dest.parent.mkdir(parents=True, exist_ok=True)
629
- try:
630
- dest.write_bytes(path.read_bytes())
631
- except OSError:
632
- continue
633
- mirrored.append((path, dest))
634
-
635
- _python_pipeline(root, targets, [d for _, d in mirrored])
636
- differing = sum(
637
- 1 for original, dest in mirrored if original.read_bytes() != dest.read_bytes()
638
- )
639
-
640
- total = len(mirrored)
641
- if differing:
642
- return (
643
- False,
644
- f"{differing} file(s) would be reformatted, {total - differing} already formatted",
645
- )
646
- return True, f"{total} files already formatted"
647
-
648
-
649
- # ---------------------------------------------------------------------------
650
- # reporting
651
- # ---------------------------------------------------------------------------
652
-
653
-
654
- def print_report(report: MarkupReport, ruff_line: str, *, write: bool) -> None:
655
- verb = "formatted" if write else "would format"
656
- if ruff_line:
657
- print(f"{bold('python')} {ruff_line}")
658
- if report.error:
659
- print(f"{bold('markup')} {red(report.error)}")
660
- return
661
- print(
662
- f"{bold('markup')} {verb} {report.formatted} of {report.scanned} block(s); "
663
- f"{report.already} already formatted; {len(report.skips)} skipped"
664
- )
665
- if not report.skips:
666
- return
667
- print(
668
- f"\n{yellow('skipped')} — djLint's output could not be proved to render "
669
- f"identically, so these were left alone:"
670
- )
671
- by_file: dict[str, list[Skip]] = {}
672
- for skip in report.skips:
673
- by_file.setdefault(skip.path, []).append(skip)
674
- for path in sorted(by_file):
675
- print(f" {path}")
676
- for skip in sorted(by_file[path], key=lambda s: s.lineno):
677
- print(f" {skip.lineno}: {skip.reason}")
678
-
679
-
680
- def main() -> int:
681
- parser = argparse.ArgumentParser(
682
- description="Format app Python (ruff) and authored markup (djLint)."
683
- )
684
- parser.add_argument(
685
- "--check",
686
- action="store_true",
687
- help="report what would change without writing; exit 1 if work remains",
688
- )
689
- parser.add_argument("--python", action="store_true", help="format Python only")
690
- parser.add_argument("--markup", action="store_true", help="format markup only")
691
- args = parser.parse_args()
692
-
693
- do_python = args.python or not args.markup
694
- do_markup = args.markup or not args.python
695
- write = not args.check
696
-
697
- # Markup first, Python second. Reformatting a template changes how many
698
- # lines its string literal spans, which can change how ruff wraps the
699
- # enclosing `html(...)` call. Running ruff last means one pass converges;
700
- # the other order leaves files that `--check` would still flag.
701
- report = MarkupReport()
702
- if do_markup:
703
- report = format_markup_blocks(write=write)
704
-
705
- ruff_ok, ruff_line = (True, "")
706
- if do_python:
707
- ruff_ok, ruff_line = run_ruff_format(write=write)
708
-
709
- print_report(report, ruff_line, write=write)
710
-
711
- if report.error:
712
- return 1
713
- if args.check:
714
- return 0 if (ruff_ok and report.formatted == 0) else 1
715
- return 0
716
-
717
-
718
- if __name__ == "__main__":
719
- raise SystemExit(main())
1
+ """App formatter: Python via ruff, authored markup via djLint (`npm run format`).
2
+
3
+ Two surfaces, one command:
4
+
5
+ 1. **Python** -- `ruff format` over `main.py`, `src/**`, `settings/*.py`,
6
+ `tests/**`. A Python formatter never rewrites string *contents*, so the
7
+ markup inside `html(r\"\"\"...\"\"\")` is byte-preserved; this was verified
8
+ across every block in `src/` before adopting it.
9
+
10
+ 2. **Markup** -- the template inside every `html(r\"\"\"...\"\"\")` call, formatted
11
+ with djLint and then *proved* unchanged in rendering before being written.
12
+
13
+ Why djLint, and why the proof
14
+ -----------------------------
15
+ This project layers four dialects in one string: HTML, Jinja `{{ }}`/`{% %}`,
16
+ PulsePoint `{ }`, and JavaScript inside `<script>` -- and they nest, e.g.
17
+ `class="... {currentUrl === '{{ item['href'] }}' ? 'a' : 'b'}"`. Prettier has no
18
+ Jinja awareness: it de-indents `{% for %}` blocks to column 0 and joins
19
+ `{% endfor %} {% endfor %}` onto one line. djLint is Jinja-aware and, critically,
20
+ does not reflow text -- so PulsePoint expressions survive intact.
21
+
22
+ djLint is still a general HTML formatter, and it will make changes that are
23
+ correct for HTML but wrong here. The one that matters: it inserts a newline
24
+ between a block tag and an adjacent inline or `<x-*>` tag, which renders as a
25
+ visible space, because a custom element's `display` is set by CSS the formatter
26
+ cannot see. So no block is trusted -- each is proved equivalent by
27
+ `_markup_equivalence` before it is written, and skipped with a reason if not.
28
+
29
+ Two regions are additionally masked away before djLint runs, so they are
30
+ preserved byte-for-byte by construction rather than by proof:
31
+
32
+ * `<script>` / `<style>` -- djLint reads `/>` inside a JS regex as a tag
33
+ delimiter and rewrites `.replace(/>/g, …)` into `.replace( />/g, …)`.
34
+ * `<pre>` / `<textarea>` -- whitespace there renders literally.
35
+
36
+ Usage (from the project root):
37
+
38
+ python settings/format.py # format Python + markup
39
+ python settings/format.py --check # report only; exit 1 if work remains
40
+ python settings/format.py --markup # markup only, skip ruff format
41
+ python settings/format.py --python # ruff format only, skip markup
42
+
43
+ `npm run test:fix` runs this first, so formatting settles before the fixer and
44
+ the gate look at the code.
45
+ """
46
+
47
+ from __future__ import annotations
48
+
49
+ import argparse
50
+ import ast
51
+ import re
52
+ import subprocess
53
+ import sys
54
+ import tempfile
55
+ from dataclasses import dataclass, field
56
+ from functools import lru_cache
57
+ from pathlib import Path
58
+
59
+ import _markup_equivalence as eq
60
+
61
+ PROJECT_ROOT = Path(__file__).resolve().parents[1]
62
+ SCAN_ROOT = PROJECT_ROOT / "src"
63
+
64
+ # Generated or vendored trees that are not hand-authored.
65
+ EXCLUDED_PARTS = {"__pycache__", "node_modules", ".venv", "prisma"}
66
+
67
+ # Paths handed to `ruff format`. Mirrors the gate's Python surface.
68
+ PYTHON_TARGETS = ("main.py", "src", "settings", "tests")
69
+
70
+ # Excluded from formatting. The test is NOT "was this generated" -- it is
71
+ # "is this regenerated wholesale and never hand-edited". The Prisma client is
72
+ # rewritten in full by `prisma generate` and is never touched by hand, so
73
+ # formatting it is pure churn that the next generate undoes; it also holds no
74
+ # markup, so nothing is lost by leaving it alone.
75
+ #
76
+ # Component libraries under `src/lib/**` are deliberately NOT listed here even
77
+ # though a CLI first installed them: they are hand-maintained in this workspace,
78
+ # and they hold the majority of the repo's markup blocks. Excluding them would
79
+ # remove most of the formatter's reach.
80
+ PYTHON_FORMAT_EXCLUDE = ("src/lib/prisma",)
81
+
82
+ # djLint settings. Passed explicitly rather than read from pyproject, because
83
+ # blocks are formatted in a temp directory outside the project.
84
+ DJLINT_INDENT = "2"
85
+ DJLINT_MAX_LINE = "120"
86
+
87
+ # Register `<x-*>` component tags with djLint. Without this it does not know
88
+ # them, treats them as inline, and leaves a component tree flat:
89
+ #
90
+ # <x-shell>
91
+ # <x-brand />
92
+ # </x-shell>
93
+ #
94
+ # Registered, they nest like any other container. This does not weaken the
95
+ # safety check -- `_markup_equivalence` still treats `<x-*>` as inline, so any
96
+ # block where the new indentation would actually add rendered whitespace is
97
+ # still skipped.
98
+ DJLINT_CUSTOM_HTML = r"x-[\w-]+"
99
+
100
+ OPAQUE = re.compile(r"(<(script|style|pre|textarea)\b[^>]*>)(.*?)(</\2\s*>)", re.I | re.S)
101
+
102
+ try:
103
+ sys.stdout.reconfigure(encoding="utf-8", errors="replace") # type: ignore[union-attr]
104
+ except AttributeError, ValueError:
105
+ pass
106
+
107
+ _TTY = sys.stdout.isatty()
108
+
109
+
110
+ def _c(code: str, text: str) -> str:
111
+ return f"\033[{code}m{text}\033[0m" if _TTY else text
112
+
113
+
114
+ def red(t: str) -> str:
115
+ return _c("31", t)
116
+
117
+
118
+ def green(t: str) -> str:
119
+ return _c("32", t)
120
+
121
+
122
+ def yellow(t: str) -> str:
123
+ return _c("33", t)
124
+
125
+
126
+ def bold(t: str) -> str:
127
+ return _c("1", t)
128
+
129
+
130
+ # ---------------------------------------------------------------------------
131
+ # masking
132
+ # ---------------------------------------------------------------------------
133
+
134
+
135
+ def mask_opaque(markup: str) -> tuple[str, dict[str, str]]:
136
+ """Replace `<script>`/`<style>`/`<pre>`/`<textarea>` bodies with tokens."""
137
+ store: dict[str, str] = {}
138
+
139
+ def sub(m: re.Match[str]) -> str:
140
+ key = f"PPMASK{len(store):04d}Z"
141
+ store[key] = m.group(3)
142
+ return f"{m.group(1)}{key}{m.group(4)}"
143
+
144
+ return OPAQUE.sub(sub, markup), store
145
+
146
+
147
+ def unmask_opaque(markup: str, store: dict[str, str]) -> str:
148
+ """Restore masked bodies, re-indenting code to its new nesting depth.
149
+
150
+ `<pre>`/`<textarea>` are restored verbatim -- their whitespace renders.
151
+ `<script>`/`<style>` are re-indented as a block, which cannot change what
152
+ the code does but keeps the output readable.
153
+ """
154
+ for key, body in store.items():
155
+ pattern = re.compile(
156
+ r"([ \t]*)(<(script|style|pre|textarea)\b[^>]*>)" + key + r"(</\3\s*>)",
157
+ re.I,
158
+ )
159
+
160
+ def repl(m: re.Match[str], body: str = body) -> str:
161
+ indent, open_tag, tag, close_tag = (
162
+ m.group(1),
163
+ m.group(2),
164
+ m.group(3).lower(),
165
+ m.group(4),
166
+ )
167
+ if tag in ("pre", "textarea"):
168
+ return f"{indent}{open_tag}{body}{close_tag}"
169
+ lines = body.splitlines()
170
+ while lines and not lines[0].strip():
171
+ lines.pop(0)
172
+ while lines and not lines[-1].strip():
173
+ lines.pop()
174
+ if not lines:
175
+ return f"{indent}{open_tag}{close_tag}"
176
+ base = min((len(ln) - len(ln.lstrip()) for ln in lines if ln.strip()), default=0)
177
+ inner = indent + " "
178
+ rendered = "\n".join((inner + ln[base:].rstrip()) if ln.strip() else "" for ln in lines)
179
+ return f"{indent}{open_tag}\n{rendered}\n{indent}{close_tag}"
180
+
181
+ markup = pattern.sub(repl, markup, count=1)
182
+ return markup
183
+
184
+
185
+ # ---------------------------------------------------------------------------
186
+ # block discovery
187
+ # ---------------------------------------------------------------------------
188
+
189
+
190
+ @dataclass
191
+ class Block:
192
+ """One `html(r\"\"\"...\"\"\")` template argument in a Python file."""
193
+
194
+ path: Path
195
+ lineno: int
196
+ start: int # byte offset of the string literal's opening quote
197
+ end: int # byte offset just past its closing quote
198
+ prefix: str # the literal's opening delimiter, e.g. `r\"\"\"`
199
+ quote: str # the closing delimiter
200
+ source: str # the markup itself
201
+
202
+
203
+ def _rel(path: Path) -> str:
204
+ """Project-relative path for reports, tolerating a path outside the root."""
205
+ try:
206
+ return path.relative_to(PROJECT_ROOT).as_posix()
207
+ except ValueError:
208
+ return path.as_posix()
209
+
210
+
211
+ def _iter_python_files() -> list[Path]:
212
+ if not SCAN_ROOT.exists():
213
+ return []
214
+ return sorted(p for p in SCAN_ROOT.rglob("*.py") if not EXCLUDED_PARTS.intersection(p.parts))
215
+
216
+
217
+ def find_blocks(path: Path) -> list[Block]:
218
+ """Every raw triple-quoted `html(...)` template argument in one file.
219
+
220
+ Only the `html(r\"\"\"...\"\"\")` form is touched. That is the single markup
221
+ entrypoint the `templates` gate already enforces (`html-form`), so any other
222
+ shape is a gate failure to fix rather than something to reformat.
223
+ """
224
+ try:
225
+ text = path.read_text(encoding="utf-8")
226
+ tree = ast.parse(text)
227
+ except OSError, UnicodeDecodeError, SyntaxError:
228
+ return []
229
+
230
+ lines = text.splitlines(keepends=True)
231
+ offsets = [0]
232
+ for line in lines:
233
+ offsets.append(offsets[-1] + len(line))
234
+
235
+ blocks: list[Block] = []
236
+ for node in ast.walk(tree):
237
+ if not isinstance(node, ast.Call):
238
+ continue
239
+ name = getattr(node.func, "id", None) or getattr(node.func, "attr", None)
240
+ if name != "html" or not node.args:
241
+ continue
242
+ arg = node.args[0]
243
+ if not (isinstance(arg, ast.Constant) and isinstance(arg.value, str)):
244
+ continue
245
+ if arg.end_lineno is None or arg.end_col_offset is None:
246
+ continue
247
+ start = offsets[arg.lineno - 1] + arg.col_offset
248
+ end = offsets[arg.end_lineno - 1] + arg.end_col_offset
249
+ literal = text[start:end]
250
+ match = re.match(r'^([rR]?)("""|\'\'\')', literal)
251
+ if not match or not match.group(1):
252
+ continue # not the raw triple-quoted form; `templates` reports it
253
+ prefix, quote = match.group(0), match.group(2)
254
+ if not literal.endswith(quote):
255
+ continue
256
+ blocks.append(
257
+ Block(
258
+ path=path,
259
+ lineno=arg.lineno,
260
+ start=start,
261
+ end=end,
262
+ prefix=prefix,
263
+ quote=quote,
264
+ source=literal[len(prefix) : -len(quote)],
265
+ )
266
+ )
267
+ return blocks
268
+
269
+
270
+ # ---------------------------------------------------------------------------
271
+ # djLint
272
+ # ---------------------------------------------------------------------------
273
+
274
+
275
+ def djlint_batch(sources: list[str]) -> list[str | None]:
276
+ """Format many markup fragments in a single djLint run.
277
+
278
+ djLint costs roughly a second of interpreter start-up per invocation, and
279
+ this repo has well over 500 blocks. Writing them all into one temp directory
280
+ and reformatting it once turns minutes into seconds.
281
+ """
282
+ if not sources:
283
+ return []
284
+ results: list[str | None] = [None] * len(sources)
285
+ with tempfile.TemporaryDirectory(prefix="pp-format-") as tmp:
286
+ root = Path(tmp)
287
+ for index, text in enumerate(sources):
288
+ (root / f"{index:05d}.html").write_text(text, encoding="utf-8")
289
+ proc = subprocess.run(
290
+ [
291
+ sys.executable,
292
+ "-m",
293
+ "djlint",
294
+ str(root),
295
+ "--reformat",
296
+ "--profile",
297
+ "jinja",
298
+ "--indent",
299
+ DJLINT_INDENT,
300
+ "--max-line-length",
301
+ DJLINT_MAX_LINE,
302
+ "--preserve-blank-lines",
303
+ "--custom-html",
304
+ DJLINT_CUSTOM_HTML,
305
+ ],
306
+ capture_output=True,
307
+ text=True,
308
+ encoding="utf-8",
309
+ errors="replace",
310
+ )
311
+ # djLint exits 1 when it reformatted something; only >1 is a real error.
312
+ if proc.returncode > 1:
313
+ return results
314
+ for index in range(len(sources)):
315
+ try:
316
+ results[index] = (root / f"{index:05d}.html").read_text(encoding="utf-8")
317
+ except OSError:
318
+ results[index] = None
319
+ return results
320
+
321
+
322
+ def format_markup(source: str, formatted_skeleton: str | None, store: dict[str, str]) -> str | None:
323
+ if formatted_skeleton is None:
324
+ return None
325
+ return unmask_opaque(formatted_skeleton, store)
326
+
327
+
328
+ # ---------------------------------------------------------------------------
329
+ # splicing
330
+ # ---------------------------------------------------------------------------
331
+
332
+
333
+ def render_literal(block: Block, markup: str) -> str | None:
334
+ """Rebuild the Python string literal around freshly formatted markup.
335
+
336
+ Returns None when the result could not be spliced back safely.
337
+ """
338
+ body = markup.strip("\n")
339
+ if block.quote in body:
340
+ return None # would terminate the literal early
341
+ if body.endswith("\\"):
342
+ return None # a trailing backslash escapes the closing quote
343
+ original_multiline = "\n" in block.source.strip("\n") or block.source.startswith("\n")
344
+ if "\n" in body or original_multiline:
345
+ body = f"\n{body}\n"
346
+ return f"{block.prefix}{body}{block.quote}"
347
+
348
+
349
+ @lru_cache(maxsize=None)
350
+ def jinja_classes(path: Path) -> dict:
351
+ """Class lists this file hides behind an expression.
352
+
353
+ Recovered from the file's own source so the oracle can see the display of a
354
+ root written as `<div {{ attributes }}>` or `class="{helper()}"`. Cached
355
+ because every block in a file shares one answer.
356
+ """
357
+ try:
358
+ return eq.class_hints(path.read_text(encoding="utf-8"))
359
+ except OSError:
360
+ return {}
361
+
362
+
363
+ @dataclass
364
+ class Skip:
365
+ path: str
366
+ lineno: int
367
+ reason: str
368
+
369
+
370
+ @dataclass
371
+ class MarkupReport:
372
+ scanned: int = 0
373
+ formatted: int = 0
374
+ already: int = 0
375
+ files_changed: int = 0
376
+ skips: list[Skip] = field(default_factory=list)
377
+ error: str = ""
378
+
379
+
380
+ def format_markup_blocks(*, write: bool) -> MarkupReport:
381
+ report = MarkupReport()
382
+
383
+ files = _iter_python_files()
384
+ per_file: dict[Path, list[Block]] = {}
385
+ flat: list[tuple[Path, Block]] = []
386
+ for path in files:
387
+ blocks = find_blocks(path)
388
+ if blocks:
389
+ per_file[path] = blocks
390
+ flat.extend((path, b) for b in blocks)
391
+
392
+ report.scanned = len(flat)
393
+ if not flat:
394
+ return report
395
+
396
+ masked: list[str] = []
397
+ stores: list[dict[str, str]] = []
398
+ for _, block in flat:
399
+ skeleton, store = mask_opaque(block.source)
400
+ masked.append(skeleton)
401
+ stores.append(store)
402
+
403
+ outputs = djlint_batch(masked)
404
+ if all(o is None for o in outputs):
405
+ report.error = "djLint did not run. Install it with `uv sync --group dev`."
406
+ return report
407
+
408
+ # Decide each block, then apply per file from the bottom up so earlier
409
+ # offsets stay valid.
410
+ decisions: dict[Path, list[tuple[Block, str]]] = {}
411
+ for (path, block), skeleton, store in zip(flat, outputs, stores):
412
+ rel = _rel(path)
413
+ formatted = format_markup(block.source, skeleton, store)
414
+ if formatted is None:
415
+ report.skips.append(Skip(rel, block.lineno, "djLint could not format this block"))
416
+ continue
417
+ if formatted.strip("\n") == block.source.strip("\n"):
418
+ report.already += 1
419
+ continue
420
+ same, why = eq.equivalent(block.source, formatted, jinja_classes(path))
421
+ if not same:
422
+ report.skips.append(Skip(rel, block.lineno, why))
423
+ continue
424
+ literal = render_literal(block, formatted)
425
+ if literal is None:
426
+ report.skips.append(Skip(rel, block.lineno, "could not be spliced back safely"))
427
+ continue
428
+ decisions.setdefault(path, []).append((block, literal))
429
+ report.formatted += 1
430
+
431
+ report.files_changed = len(decisions)
432
+ if not write:
433
+ return report
434
+
435
+ for path, items in decisions.items():
436
+ text = path.read_text(encoding="utf-8")
437
+ for block, literal in sorted(items, key=lambda i: i[0].start, reverse=True):
438
+ text = text[: block.start] + literal + text[block.end :]
439
+ # Never leave a file that no longer parses, or whose markup did not land
440
+ # exactly as intended.
441
+ try:
442
+ ast.parse(text)
443
+ except SyntaxError:
444
+ report.skips.append(Skip(_rel(path), 0, "rewrite would not parse; file left unchanged"))
445
+ report.formatted -= len(items)
446
+ report.files_changed -= 1
447
+ continue
448
+ path.write_text(text, encoding="utf-8")
449
+
450
+ return report
451
+
452
+
453
+ # ---------------------------------------------------------------------------
454
+ # ruff format
455
+ # ---------------------------------------------------------------------------
456
+
457
+
458
+ def _ruff_format_cmd(targets: list[str]) -> list[str]:
459
+ cmd = [sys.executable, "-m", "ruff", "format", *targets]
460
+ for excluded in PYTHON_FORMAT_EXCLUDE:
461
+ cmd += ["--exclude", excluded]
462
+ return cmd
463
+
464
+
465
+ def hug_html_call_openings(paths: list[Path], *, write: bool) -> list[Path]:
466
+ """Put `html(r\"\"\"` back on one line, so the markup starts at line 1.
467
+
468
+ `ruff format` always explodes a call whose first argument is a multiline
469
+ string when the call has other arguments, producing:
470
+
471
+ return html(
472
+ r\"\"\"
473
+ <div>…
474
+
475
+ That buries the template one level deeper and separates `html(` from the
476
+ markup it opens. The house style is `html(r\"\"\"` with the markup starting
477
+ on the next line, so this runs *after* ruff and closes the gap.
478
+
479
+ Ruff will re-split these on its next run, in default and preview style
480
+ alike, so the two steps must always run as a pair -- which is why
481
+ `--check` re-runs the whole pipeline rather than calling
482
+ `ruff format --check` directly.
483
+
484
+ Returns the paths that needed the fix.
485
+ """
486
+ touched: list[Path] = []
487
+ for path in paths:
488
+ try:
489
+ text = path.read_text(encoding="utf-8")
490
+ except OSError, UnicodeDecodeError:
491
+ continue
492
+ if "html(" not in text:
493
+ continue
494
+ try:
495
+ tree = ast.parse(text)
496
+ except SyntaxError:
497
+ continue
498
+
499
+ lines = text.splitlines(keepends=True)
500
+ offsets = [0]
501
+ for line in lines:
502
+ offsets.append(offsets[-1] + len(line))
503
+
504
+ cuts: list[tuple[int, int]] = []
505
+ for node in ast.walk(tree):
506
+ if not isinstance(node, ast.Call):
507
+ continue
508
+ name = getattr(node.func, "id", None) or getattr(node.func, "attr", None)
509
+ if name != "html" or not node.args:
510
+ continue
511
+ arg = node.args[0]
512
+ if not (isinstance(arg, ast.Constant) and isinstance(arg.value, str)):
513
+ continue
514
+ func_line, func_col = node.func.end_lineno, node.func.end_col_offset
515
+ if func_line is None or func_col is None:
516
+ continue
517
+ paren = text.find("(", offsets[func_line - 1] + func_col)
518
+ if paren == -1:
519
+ continue
520
+ literal_start = offsets[arg.lineno - 1] + arg.col_offset
521
+ gap = text[paren + 1 : literal_start]
522
+ # Only close a gap that is pure whitespace spanning a line break;
523
+ # anything else means this is not the shape ruff produced.
524
+ if gap and gap.strip() == "" and "\n" in gap:
525
+ cuts.append((paren + 1, literal_start))
526
+
527
+ if not cuts:
528
+ continue
529
+ touched.append(path)
530
+ if not write:
531
+ continue
532
+ for start, end in sorted(cuts, reverse=True):
533
+ text = text[:start] + text[end:]
534
+ try:
535
+ ast.parse(text)
536
+ except SyntaxError:
537
+ continue # leave the file as ruff wrote it rather than risk it
538
+ path.write_text(text, encoding="utf-8")
539
+ return touched
540
+
541
+
542
+ def _python_files_for_hug() -> list[Path]:
543
+ """Every formatted Python file that could contain an `html(...)` call."""
544
+ found: list[Path] = []
545
+ for target in PYTHON_TARGETS:
546
+ base = PROJECT_ROOT / target
547
+ if not base.exists():
548
+ continue
549
+ candidates = [base] if base.is_file() else sorted(base.rglob("*.py"))
550
+ for path in candidates:
551
+ rel = path.relative_to(PROJECT_ROOT).as_posix()
552
+ if EXCLUDED_PARTS.intersection(path.parts):
553
+ continue
554
+ if any(rel.startswith(x) for x in PYTHON_FORMAT_EXCLUDE):
555
+ continue
556
+ found.append(path)
557
+ return found
558
+
559
+
560
+ _PYTHON_PIPELINE_ROUNDS = 4
561
+
562
+
563
+ def _snapshot(files: list[Path]) -> dict[Path, bytes]:
564
+ out: dict[Path, bytes] = {}
565
+ for path in files:
566
+ try:
567
+ out[path] = path.read_bytes()
568
+ except OSError:
569
+ continue
570
+ return out
571
+
572
+
573
+ def _python_pipeline(root: Path, targets: list[str], files: list[Path]) -> int:
574
+ """Run `ruff format` + the `html(` hug until the tree stops changing.
575
+
576
+ One pass is not enough, because the two steps feed each other. Ruff splits
577
+ `html(` off its template; the hug rejoins it; and for a call whose template
578
+ is the *only* argument, that rejoin then lets ruff pull the closing `)` up
579
+ on its next run. Iterating to a fixed point is what makes the result stable
580
+ -- and it is what lets `--check` replay this exact function against a mirror
581
+ of the tree, instead of trusting `ruff format --check`, which would flag
582
+ every hugged call as unformatted because ruff is what splits them.
583
+
584
+ It settles in a couple of rounds: a multi-argument call lands on
585
+ ruff-splits-then-hug-rejoins, which is a fixed point of the *pair* even
586
+ though neither step is idempotent alone.
587
+
588
+ Returns the number of files whose bytes changed.
589
+ """
590
+ initial = _snapshot(files)
591
+ for _ in range(_PYTHON_PIPELINE_ROUNDS):
592
+ before = _snapshot(files)
593
+ subprocess.run(
594
+ _ruff_format_cmd(targets),
595
+ cwd=root,
596
+ capture_output=True,
597
+ text=True,
598
+ encoding="utf-8",
599
+ errors="replace",
600
+ )
601
+ hug_html_call_openings(files, write=True)
602
+ if _snapshot(files) == before:
603
+ break
604
+ final = _snapshot(files)
605
+ return sum(1 for path, data in initial.items() if final.get(path) != data)
606
+
607
+
608
+ def run_ruff_format(*, write: bool) -> tuple[bool, str]:
609
+ targets = [t for t in PYTHON_TARGETS if (PROJECT_ROOT / t).exists()]
610
+ files = _python_files_for_hug()
611
+
612
+ if write:
613
+ changed = _python_pipeline(PROJECT_ROOT, targets, files)
614
+ total = len(files)
615
+ if changed:
616
+ return True, f"{changed} file(s) reformatted, {total - changed} left unchanged"
617
+ return True, f"{total} files already formatted"
618
+
619
+ with tempfile.TemporaryDirectory(prefix="pp-pyfmt-") as tmp:
620
+ root = Path(tmp)
621
+ # Ruff resolves `include`/`exclude` relative to the config's directory,
622
+ # so the mirror needs the same relative layout and its own copy of the
623
+ # config, or every file is filtered out as "not part of the project".
624
+ (root / "pyproject.toml").write_bytes((PROJECT_ROOT / "pyproject.toml").read_bytes())
625
+ mirrored: list[tuple[Path, Path]] = []
626
+ for path in files:
627
+ dest = root / path.relative_to(PROJECT_ROOT)
628
+ dest.parent.mkdir(parents=True, exist_ok=True)
629
+ try:
630
+ dest.write_bytes(path.read_bytes())
631
+ except OSError:
632
+ continue
633
+ mirrored.append((path, dest))
634
+
635
+ _python_pipeline(root, targets, [d for _, d in mirrored])
636
+ differing = sum(
637
+ 1 for original, dest in mirrored if original.read_bytes() != dest.read_bytes()
638
+ )
639
+
640
+ total = len(mirrored)
641
+ if differing:
642
+ return (
643
+ False,
644
+ f"{differing} file(s) would be reformatted, {total - differing} already formatted",
645
+ )
646
+ return True, f"{total} files already formatted"
647
+
648
+
649
+ # ---------------------------------------------------------------------------
650
+ # reporting
651
+ # ---------------------------------------------------------------------------
652
+
653
+
654
+ def print_report(report: MarkupReport, ruff_line: str, *, write: bool) -> None:
655
+ verb = "formatted" if write else "would format"
656
+ if ruff_line:
657
+ print(f"{bold('python')} {ruff_line}")
658
+ if report.error:
659
+ print(f"{bold('markup')} {red(report.error)}")
660
+ return
661
+ print(
662
+ f"{bold('markup')} {verb} {report.formatted} of {report.scanned} block(s); "
663
+ f"{report.already} already formatted; {len(report.skips)} skipped"
664
+ )
665
+ if not report.skips:
666
+ return
667
+ print(
668
+ f"\n{yellow('skipped')} — djLint's output could not be proved to render "
669
+ f"identically, so these were left alone:"
670
+ )
671
+ by_file: dict[str, list[Skip]] = {}
672
+ for skip in report.skips:
673
+ by_file.setdefault(skip.path, []).append(skip)
674
+ for path in sorted(by_file):
675
+ print(f" {path}")
676
+ for skip in sorted(by_file[path], key=lambda s: s.lineno):
677
+ print(f" {skip.lineno}: {skip.reason}")
678
+
679
+
680
+ def main() -> int:
681
+ parser = argparse.ArgumentParser(
682
+ description="Format app Python (ruff) and authored markup (djLint)."
683
+ )
684
+ parser.add_argument(
685
+ "--check",
686
+ action="store_true",
687
+ help="report what would change without writing; exit 1 if work remains",
688
+ )
689
+ parser.add_argument("--python", action="store_true", help="format Python only")
690
+ parser.add_argument("--markup", action="store_true", help="format markup only")
691
+ args = parser.parse_args()
692
+
693
+ do_python = args.python or not args.markup
694
+ do_markup = args.markup or not args.python
695
+ write = not args.check
696
+
697
+ # Markup first, Python second. Reformatting a template changes how many
698
+ # lines its string literal spans, which can change how ruff wraps the
699
+ # enclosing `html(...)` call. Running ruff last means one pass converges;
700
+ # the other order leaves files that `--check` would still flag.
701
+ report = MarkupReport()
702
+ if do_markup:
703
+ report = format_markup_blocks(write=write)
704
+
705
+ ruff_ok, ruff_line = (True, "")
706
+ if do_python:
707
+ ruff_ok, ruff_line = run_ruff_format(write=write)
708
+
709
+ print_report(report, ruff_line, write=write)
710
+
711
+ if report.error:
712
+ return 1
713
+ if args.check:
714
+ return 0 if (ruff_ok and report.formatted == 0) else 1
715
+ return 0
716
+
717
+
718
+ if __name__ == "__main__":
719
+ raise SystemExit(main())