casecrash 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- casecrash/__init__.py +5 -0
- casecrash/cli.py +147 -0
- casecrash/core.py +285 -0
- casecrash-0.1.0.dist-info/METADATA +132 -0
- casecrash-0.1.0.dist-info/RECORD +9 -0
- casecrash-0.1.0.dist-info/WHEEL +5 -0
- casecrash-0.1.0.dist-info/entry_points.txt +2 -0
- casecrash-0.1.0.dist-info/licenses/LICENSE +21 -0
- casecrash-0.1.0.dist-info/top_level.txt +1 -0
casecrash/__init__.py
ADDED
casecrash/cli.py
ADDED
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
"""Command-line interface for casecrash."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import json
|
|
7
|
+
import os
|
|
8
|
+
import sys
|
|
9
|
+
|
|
10
|
+
from . import __version__
|
|
11
|
+
from .core import (
|
|
12
|
+
GitNotFound,
|
|
13
|
+
NotAGitRepo,
|
|
14
|
+
escaped_display,
|
|
15
|
+
normalization_form,
|
|
16
|
+
safe_display,
|
|
17
|
+
scan,
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
EXIT_OK = 0
|
|
21
|
+
EXIT_COLLISIONS = 1
|
|
22
|
+
EXIT_ERROR = 2
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
26
|
+
parser = argparse.ArgumentParser(
|
|
27
|
+
prog="casecrash",
|
|
28
|
+
description=(
|
|
29
|
+
"Find filenames in your git index that would collide on "
|
|
30
|
+
"case-insensitive or Unicode-normalizing filesystems "
|
|
31
|
+
"(macOS, Windows) — before they break someone's checkout."
|
|
32
|
+
),
|
|
33
|
+
)
|
|
34
|
+
parser.add_argument(
|
|
35
|
+
"path",
|
|
36
|
+
nargs="?",
|
|
37
|
+
default=".",
|
|
38
|
+
help="directory inside the git repository to scan (default: .)",
|
|
39
|
+
)
|
|
40
|
+
parser.add_argument(
|
|
41
|
+
"--format",
|
|
42
|
+
choices=["text", "json"],
|
|
43
|
+
default="text",
|
|
44
|
+
help="output format (default: text)",
|
|
45
|
+
)
|
|
46
|
+
parser.add_argument(
|
|
47
|
+
"--version",
|
|
48
|
+
action="version",
|
|
49
|
+
version="casecrash " + __version__,
|
|
50
|
+
)
|
|
51
|
+
return parser
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _platform_note(kind: str) -> str:
|
|
55
|
+
if kind == "case":
|
|
56
|
+
return "collide on case-insensitive filesystems (macOS, Windows)"
|
|
57
|
+
if kind == "normalization":
|
|
58
|
+
return "collide after Unicode normalization (macOS)"
|
|
59
|
+
return "identical index entries (index corruption)"
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _format_name(raw: bytes) -> str:
|
|
63
|
+
form = normalization_form(raw)
|
|
64
|
+
shown = escaped_display(raw)
|
|
65
|
+
if form:
|
|
66
|
+
return "'%s' [%s]" % (shown, form)
|
|
67
|
+
return "'%s'" % shown
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def render_text(result) -> str:
|
|
71
|
+
lines = []
|
|
72
|
+
if not result.collisions:
|
|
73
|
+
lines.append(
|
|
74
|
+
"casecrash: no filename collisions in the git index "
|
|
75
|
+
"(%d files scanned%s)." % (
|
|
76
|
+
result.files_scanned,
|
|
77
|
+
", %d submodule(s) skipped" % result.gitlinks_skipped
|
|
78
|
+
if result.gitlinks_skipped else "",
|
|
79
|
+
)
|
|
80
|
+
)
|
|
81
|
+
return "\n".join(lines) + "\n"
|
|
82
|
+
|
|
83
|
+
lines.append(
|
|
84
|
+
"casecrash: found %d filename collision(s) in %s "
|
|
85
|
+
"(%d files scanned):"
|
|
86
|
+
% (len(result.collisions), result.repo_root, result.files_scanned)
|
|
87
|
+
)
|
|
88
|
+
for collision in result.collisions:
|
|
89
|
+
lines.append("")
|
|
90
|
+
lines.append("[%s] %s:" % (collision.kind, _platform_note(collision.kind)))
|
|
91
|
+
for raw in collision.names:
|
|
92
|
+
lines.append(" %s" % _format_name(raw))
|
|
93
|
+
lines.append(
|
|
94
|
+
" -> breaks on: %s" % ", ".join(collision.platforms)
|
|
95
|
+
)
|
|
96
|
+
if result.gitlinks_skipped:
|
|
97
|
+
lines.append("")
|
|
98
|
+
lines.append(
|
|
99
|
+
"note: %d submodule(s) skipped (not checked)"
|
|
100
|
+
% result.gitlinks_skipped
|
|
101
|
+
)
|
|
102
|
+
lines.append("")
|
|
103
|
+
lines.append(
|
|
104
|
+
"Fix: rename one of each colliding pair before a macOS/Windows "
|
|
105
|
+
"user clones this repo. "
|
|
106
|
+
"'git mv' the odd one out and commit."
|
|
107
|
+
)
|
|
108
|
+
return "\n".join(lines) + "\n"
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def render_json(result) -> str:
|
|
112
|
+
payload = {
|
|
113
|
+
"tool": "casecrash",
|
|
114
|
+
"version": __version__,
|
|
115
|
+
"repo_root": result.repo_root,
|
|
116
|
+
"files_scanned": result.files_scanned,
|
|
117
|
+
"gitlinks_skipped": result.gitlinks_skipped,
|
|
118
|
+
"collisions": [
|
|
119
|
+
{
|
|
120
|
+
"kind": collision.kind,
|
|
121
|
+
"platforms": collision.platforms,
|
|
122
|
+
"files": [safe_display(n) for n in collision.names],
|
|
123
|
+
}
|
|
124
|
+
for collision in result.collisions
|
|
125
|
+
],
|
|
126
|
+
}
|
|
127
|
+
return json.dumps(payload, indent=2, ensure_ascii=True) + "\n"
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def main(argv: list[str] | None = None) -> int:
|
|
131
|
+
args = build_parser().parse_args(argv)
|
|
132
|
+
try:
|
|
133
|
+
result = scan(args.path)
|
|
134
|
+
except (NotAGitRepo, GitNotFound) as exc:
|
|
135
|
+
print("casecrash: error: %s" % exc, file=sys.stderr)
|
|
136
|
+
return EXIT_ERROR
|
|
137
|
+
|
|
138
|
+
if args.format == "json":
|
|
139
|
+
sys.stdout.write(render_json(result))
|
|
140
|
+
else:
|
|
141
|
+
sys.stdout.write(render_text(result))
|
|
142
|
+
|
|
143
|
+
return EXIT_COLLISIONS if result.has_collisions else EXIT_OK
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
if __name__ == "__main__":
|
|
147
|
+
sys.exit(main())
|
casecrash/core.py
ADDED
|
@@ -0,0 +1,285 @@
|
|
|
1
|
+
"""Core collision detection for casecrash.
|
|
2
|
+
|
|
3
|
+
Reads the git index (``git ls-files -s -z``) and groups tracked paths by
|
|
4
|
+
``NFC(casefold(name))``. Inside each group the cause is classified:
|
|
5
|
+
|
|
6
|
+
* ``case`` — names share the same NFC form but differ (pure case
|
|
7
|
+
difference). Breaks on macOS and Windows.
|
|
8
|
+
* ``normalization`` — NFC forms differ by more than case (e.g. NFC ``café``
|
|
9
|
+
vs NFD ``café``). Breaks on macOS, whose filesystems compare names
|
|
10
|
+
without regard to normalization.
|
|
11
|
+
* ``exact-duplicate`` — the identical (name, mode, sha) entry appears twice,
|
|
12
|
+
which indicates index corruption.
|
|
13
|
+
|
|
14
|
+
Filenames are handled as bytes end-to-end and only decoded for display with
|
|
15
|
+
``surrogateescape``, so undecodable names can never crash the tool.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import os
|
|
21
|
+
import subprocess
|
|
22
|
+
import unicodedata
|
|
23
|
+
from dataclasses import dataclass, field
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class NotAGitRepo(Exception):
|
|
27
|
+
"""Raised when the target directory is not inside a git repository."""
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class GitNotFound(Exception):
|
|
31
|
+
"""Raised when the git executable cannot be found."""
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
GITLINK_MODE = b"160000"
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@dataclass
|
|
38
|
+
class IndexEntry:
|
|
39
|
+
name: bytes
|
|
40
|
+
mode: bytes
|
|
41
|
+
sha: bytes
|
|
42
|
+
stage: bytes
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@dataclass
|
|
46
|
+
class Collision:
|
|
47
|
+
kind: str # "case" | "normalization" | "exact-duplicate"
|
|
48
|
+
names: list[bytes] = field(default_factory=list)
|
|
49
|
+
|
|
50
|
+
@property
|
|
51
|
+
def platforms(self) -> list[str]:
|
|
52
|
+
if self.kind == "case":
|
|
53
|
+
return ["macOS", "Windows"]
|
|
54
|
+
if self.kind == "normalization":
|
|
55
|
+
return ["macOS"]
|
|
56
|
+
return ["macOS", "Windows", "Linux"]
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@dataclass
|
|
60
|
+
class ScanResult:
|
|
61
|
+
repo_root: str
|
|
62
|
+
files_scanned: int
|
|
63
|
+
gitlinks_skipped: int
|
|
64
|
+
collisions: list[Collision] = field(default_factory=list)
|
|
65
|
+
|
|
66
|
+
@property
|
|
67
|
+
def has_collisions(self) -> bool:
|
|
68
|
+
return bool(self.collisions)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
# ---------------------------------------------------------------------------
|
|
72
|
+
# display helpers
|
|
73
|
+
# ---------------------------------------------------------------------------
|
|
74
|
+
|
|
75
|
+
def _safe_char(ch: str) -> str:
|
|
76
|
+
code = ord(ch)
|
|
77
|
+
if 0xDC80 <= code <= 0xDCFF: # surrogateescape byte
|
|
78
|
+
return "\\x%02x" % (code - 0xDC00)
|
|
79
|
+
return ch
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def safe_display(raw: bytes) -> str:
|
|
83
|
+
"""Lossless-ish printable form of a raw filename.
|
|
84
|
+
|
|
85
|
+
Undecodable bytes become ``\\xNN`` escapes; everything else passes
|
|
86
|
+
through unchanged.
|
|
87
|
+
"""
|
|
88
|
+
return "".join(_safe_char(c) for c in raw.decode("utf-8", "surrogateescape"))
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def escaped_display(raw: bytes) -> str:
|
|
92
|
+
"""Like :func:`safe_display` but renders every non-ASCII char as \\uXXXX.
|
|
93
|
+
|
|
94
|
+
Makes NFC-vs-NFD differences visible in reports.
|
|
95
|
+
"""
|
|
96
|
+
out = []
|
|
97
|
+
for ch in raw.decode("utf-8", "surrogateescape"):
|
|
98
|
+
code = ord(ch)
|
|
99
|
+
if 0xDC80 <= code <= 0xDCFF:
|
|
100
|
+
out.append("\\x%02x" % (code - 0xDC00))
|
|
101
|
+
elif code > 0x7E or code < 0x20:
|
|
102
|
+
out.append("\\u%04x" % code)
|
|
103
|
+
else:
|
|
104
|
+
out.append(ch)
|
|
105
|
+
return "".join(out)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def normalization_form(raw: bytes) -> str:
|
|
109
|
+
"""Best-effort NFC/NFD label for a filename, '' when not applicable."""
|
|
110
|
+
try:
|
|
111
|
+
text = raw.decode("utf-8")
|
|
112
|
+
except UnicodeDecodeError:
|
|
113
|
+
return ""
|
|
114
|
+
nfc = unicodedata.normalize("NFC", text)
|
|
115
|
+
nfd = unicodedata.normalize("NFD", text)
|
|
116
|
+
if text == nfc and text != nfd:
|
|
117
|
+
return "NFC"
|
|
118
|
+
if text == nfd and text != nfc:
|
|
119
|
+
return "NFD"
|
|
120
|
+
if text != nfc and text != nfd:
|
|
121
|
+
return "mixed"
|
|
122
|
+
return "" # pure ASCII, form is irrelevant
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
# ---------------------------------------------------------------------------
|
|
126
|
+
# index reading
|
|
127
|
+
# ---------------------------------------------------------------------------
|
|
128
|
+
|
|
129
|
+
def _run_git(args: list[str], cwd: str) -> subprocess.CompletedProcess:
|
|
130
|
+
try:
|
|
131
|
+
return subprocess.run(
|
|
132
|
+
["git"] + args,
|
|
133
|
+
cwd=cwd,
|
|
134
|
+
stdout=subprocess.PIPE,
|
|
135
|
+
stderr=subprocess.PIPE,
|
|
136
|
+
)
|
|
137
|
+
except FileNotFoundError as exc:
|
|
138
|
+
raise GitNotFound(
|
|
139
|
+
"the 'git' executable was not found on PATH; "
|
|
140
|
+
"casecrash needs git to read the index"
|
|
141
|
+
) from exc
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def repo_root(path: str) -> str:
|
|
145
|
+
"""Return the repository root for *path*, or raise NotAGitRepo."""
|
|
146
|
+
if os.path.isfile(path):
|
|
147
|
+
path = os.path.dirname(os.path.abspath(path)) or "."
|
|
148
|
+
proc = _run_git(["rev-parse", "--show-toplevel"], cwd=path)
|
|
149
|
+
if proc.returncode != 0:
|
|
150
|
+
raise NotAGitRepo(
|
|
151
|
+
"%r is not inside a git repository" % os.path.abspath(path)
|
|
152
|
+
)
|
|
153
|
+
return proc.stdout.decode("utf-8", "replace").strip()
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def read_index(path: str) -> tuple[list[IndexEntry], int]:
|
|
157
|
+
"""Read the git index. Returns (entries, gitlink_count).
|
|
158
|
+
|
|
159
|
+
Gitlinks (submodules) are counted and excluded from the entries so they
|
|
160
|
+
never participate in collision detection.
|
|
161
|
+
"""
|
|
162
|
+
root = repo_root(path)
|
|
163
|
+
proc = _run_git(["ls-files", "-s", "-z", "--full-name"], cwd=root)
|
|
164
|
+
if proc.returncode != 0:
|
|
165
|
+
raise NotAGitRepo(
|
|
166
|
+
"could not read the git index of %r: %s"
|
|
167
|
+
% (root, proc.stderr.decode("utf-8", "replace").strip())
|
|
168
|
+
)
|
|
169
|
+
entries: list[IndexEntry] = []
|
|
170
|
+
gitlinks = 0
|
|
171
|
+
for record in proc.stdout.split(b"\x00"):
|
|
172
|
+
if not record:
|
|
173
|
+
continue
|
|
174
|
+
head, _, name = record.partition(b"\t")
|
|
175
|
+
parts = head.split(b" ")
|
|
176
|
+
if len(parts) < 3 or not name:
|
|
177
|
+
continue
|
|
178
|
+
mode, sha, stage = parts[0], parts[1], parts[2]
|
|
179
|
+
if mode == GITLINK_MODE:
|
|
180
|
+
gitlinks += 1
|
|
181
|
+
continue
|
|
182
|
+
entries.append(IndexEntry(name=name, mode=mode, sha=sha, stage=stage))
|
|
183
|
+
return entries, gitlinks
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
# ---------------------------------------------------------------------------
|
|
187
|
+
# collision detection
|
|
188
|
+
# ---------------------------------------------------------------------------
|
|
189
|
+
|
|
190
|
+
def _collision_key(raw: bytes) -> tuple[str, object]:
|
|
191
|
+
"""Grouping key: NFC(casefold(name)).
|
|
192
|
+
|
|
193
|
+
Names that are not valid UTF-8 fall back to byte-level lowercasing so
|
|
194
|
+
ASCII case collisions in weird filenames are still caught.
|
|
195
|
+
"""
|
|
196
|
+
try:
|
|
197
|
+
text = raw.decode("utf-8")
|
|
198
|
+
except UnicodeDecodeError:
|
|
199
|
+
return ("raw", raw.lower())
|
|
200
|
+
return ("text", unicodedata.normalize("NFC", text.casefold()))
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def _pair_kind(a: bytes, b: bytes) -> str:
|
|
204
|
+
"""Classify why two colliding names collide.
|
|
205
|
+
|
|
206
|
+
* ``case`` — the names are equal under Unicode casefolding, so they
|
|
207
|
+
differ only by case (``README.md``/``readme.md``, ``É``/``é``).
|
|
208
|
+
Breaks on macOS and Windows.
|
|
209
|
+
* ``normalization`` — the casefolded forms still differ, so the names
|
|
210
|
+
are distinct spellings that only become equal after Unicode
|
|
211
|
+
normalization (NFC ``café`` vs NFD ``café``). Breaks on macOS,
|
|
212
|
+
whose filesystems compare names without regard to normalization.
|
|
213
|
+
"""
|
|
214
|
+
try:
|
|
215
|
+
fa = a.decode("utf-8", "surrogateescape").casefold()
|
|
216
|
+
fb = b.decode("utf-8", "surrogateescape").casefold()
|
|
217
|
+
except UnicodeDecodeError: # pragma: no cover - defensive
|
|
218
|
+
return "case"
|
|
219
|
+
return "case" if fa == fb else "normalization"
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _classify(names: list[bytes]) -> str:
|
|
223
|
+
"""Classify a group of distinct names sharing a collision key.
|
|
224
|
+
|
|
225
|
+
A group is labeled ``normalization`` if any pair inside it collides
|
|
226
|
+
for normalization reasons; otherwise it is a pure ``case`` collision.
|
|
227
|
+
"""
|
|
228
|
+
for i, first in enumerate(names):
|
|
229
|
+
for other in names[i + 1:]:
|
|
230
|
+
if _pair_kind(first, other) == "normalization":
|
|
231
|
+
return "normalization"
|
|
232
|
+
return "case"
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def find_collisions(entries: list[IndexEntry]) -> list[Collision]:
|
|
236
|
+
"""Detect collisions among index entries."""
|
|
237
|
+
collisions: list[Collision] = []
|
|
238
|
+
|
|
239
|
+
# 1. Exact duplicates: identical (name, mode, sha) more than once.
|
|
240
|
+
# (Same name at different unmerged stages is normal merge state,
|
|
241
|
+
# not corruption, so stage is deliberately ignored here... in fact
|
|
242
|
+
# we require the full triple to repeat.)
|
|
243
|
+
seen: dict[tuple[bytes, bytes, bytes], int] = {}
|
|
244
|
+
for entry in entries:
|
|
245
|
+
key = (entry.name, entry.mode, entry.sha)
|
|
246
|
+
seen[key] = seen.get(key, 0) + 1
|
|
247
|
+
for (name, _mode, _sha), count in seen.items():
|
|
248
|
+
if count > 1:
|
|
249
|
+
collisions.append(
|
|
250
|
+
Collision(kind="exact-duplicate", names=[name, name])
|
|
251
|
+
)
|
|
252
|
+
|
|
253
|
+
# 2. Case / normalization collisions over distinct names.
|
|
254
|
+
distinct: dict[bytes, None] = {}
|
|
255
|
+
for entry in entries:
|
|
256
|
+
distinct.setdefault(entry.name)
|
|
257
|
+
groups: dict[tuple[str, object], list[bytes]] = {}
|
|
258
|
+
for name in distinct:
|
|
259
|
+
groups.setdefault(_collision_key(name), []).append(name)
|
|
260
|
+
|
|
261
|
+
for names in groups.values():
|
|
262
|
+
if len(names) < 2:
|
|
263
|
+
continue
|
|
264
|
+
collisions.append(
|
|
265
|
+
Collision(kind=_classify(sorted(names)), names=sorted(names))
|
|
266
|
+
)
|
|
267
|
+
|
|
268
|
+
# Deterministic order: exact duplicates first, then by first filename.
|
|
269
|
+
order = {"exact-duplicate": 0, "normalization": 1, "case": 2}
|
|
270
|
+
collisions.sort(
|
|
271
|
+
key=lambda c: (order.get(c.kind, 3), safe_display(c.names[0]))
|
|
272
|
+
)
|
|
273
|
+
return collisions
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def scan(path: str = ".") -> ScanResult:
|
|
277
|
+
"""Scan the git index containing *path* for filename collisions."""
|
|
278
|
+
root = repo_root(path)
|
|
279
|
+
entries, gitlinks = read_index(path)
|
|
280
|
+
return ScanResult(
|
|
281
|
+
repo_root=root,
|
|
282
|
+
files_scanned=len({e.name for e in entries}),
|
|
283
|
+
gitlinks_skipped=gitlinks,
|
|
284
|
+
collisions=find_collisions(entries),
|
|
285
|
+
)
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: casecrash
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Find filenames in your git index that would collide on macOS or Windows checkouts
|
|
5
|
+
Author: hao li
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/hahahahahahahahah6/casecrash
|
|
8
|
+
Keywords: git,macos,windows,filename,case-sensitivity,unicode,ci
|
|
9
|
+
Classifier: Development Status :: 4 - Beta
|
|
10
|
+
Classifier: Environment :: Console
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Topic :: Software Development :: Version Control :: Git
|
|
15
|
+
Classifier: Topic :: Utilities
|
|
16
|
+
Requires-Python: >=3.9
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
License-File: LICENSE
|
|
19
|
+
Dynamic: license-file
|
|
20
|
+
|
|
21
|
+
# casecrash
|
|
22
|
+
|
|
23
|
+
**Git can track two files your Mac cannot check out.**
|
|
24
|
+
|
|
25
|
+
`README.md` and `readme.md` can sit side by side in a git repo — until
|
|
26
|
+
someone on macOS or Windows clones it and the checkout explodes. The same
|
|
27
|
+
goes for Unicode lookalikes: NFC `café` vs NFD `café` look identical in
|
|
28
|
+
your terminal but are different byte sequences, and macOS treats them as
|
|
29
|
+
the same file.
|
|
30
|
+
|
|
31
|
+
casecrash reads your git index and flags every pair of tracked files that
|
|
32
|
+
would collide on a case-insensitive or Unicode-normalizing filesystem —
|
|
33
|
+
*before* it breaks a teammate's checkout.
|
|
34
|
+
|
|
35
|
+
## Quickstart
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
pip install casecrash
|
|
39
|
+
cd your-repo
|
|
40
|
+
casecrash
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
```
|
|
44
|
+
casecrash: found 2 filename collision(s) in /tmp/demorepo (5 files scanned):
|
|
45
|
+
|
|
46
|
+
[normalization] collide after Unicode normalization (macOS):
|
|
47
|
+
'cafe\u0301.md' [NFD]
|
|
48
|
+
'caf\u00e9.md' [NFC]
|
|
49
|
+
-> breaks on: macOS
|
|
50
|
+
|
|
51
|
+
[case] collide on case-insensitive filesystems (macOS, Windows):
|
|
52
|
+
'README.md'
|
|
53
|
+
'readme.md'
|
|
54
|
+
-> breaks on: macOS, Windows
|
|
55
|
+
|
|
56
|
+
Fix: rename one of each colliding pair before a macOS/Windows user clones this repo. 'git mv' the odd one out and commit.
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
No collisions? Exit 0 and a clean bill of health:
|
|
60
|
+
|
|
61
|
+
```
|
|
62
|
+
casecrash: no filename collisions in the git index (128 files scanned).
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
## What it detects
|
|
66
|
+
|
|
67
|
+
| Kind | Example | Breaks on |
|
|
68
|
+
|---|---|---|
|
|
69
|
+
| `case` | `README.md` vs `readme.md`, `É` vs `é` | macOS, Windows |
|
|
70
|
+
| `normalization` | NFC `café` vs NFD `café` | macOS (APFS compares names without regard to normalization) |
|
|
71
|
+
| `exact-duplicate` | identical index entry twice | everywhere (index corruption) |
|
|
72
|
+
|
|
73
|
+
Detection is done on the git index (`git ls-files`), not the working
|
|
74
|
+
tree, so it works on any OS — a Linux CI runner can catch a collision
|
|
75
|
+
that would only bite macOS users. Filenames are handled as raw bytes, so
|
|
76
|
+
undecodable names can never crash the tool. Submodules are skipped (and
|
|
77
|
+
counted).
|
|
78
|
+
|
|
79
|
+
## CI usage
|
|
80
|
+
|
|
81
|
+
casecrash exits **1** when collisions are found, **0** when clean, **2** on
|
|
82
|
+
errors — so it drops straight into any pipeline:
|
|
83
|
+
|
|
84
|
+
```yaml
|
|
85
|
+
# .github/workflows/casecrash.yml
|
|
86
|
+
name: casecheck
|
|
87
|
+
on: [push, pull_request]
|
|
88
|
+
jobs:
|
|
89
|
+
casecrash:
|
|
90
|
+
runs-on: ubuntu-latest
|
|
91
|
+
steps:
|
|
92
|
+
- uses: actions/checkout@v4
|
|
93
|
+
- uses: actions/setup-python@v5
|
|
94
|
+
with:
|
|
95
|
+
python-version: "3.12"
|
|
96
|
+
- run: pip install casecrash
|
|
97
|
+
- run: casecrash
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
JSON output for scripting:
|
|
101
|
+
|
|
102
|
+
```bash
|
|
103
|
+
casecrash --format json
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
## Why this happens
|
|
107
|
+
|
|
108
|
+
Git was born on case-sensitive Linux filesystems and its index happily
|
|
109
|
+
stores `File.txt` and `file.txt` as two separate entries. macOS and
|
|
110
|
+
Windows disagree: checking out the second name silently overwrites the
|
|
111
|
+
first (or the clone fails outright). Normalization collisions are
|
|
112
|
+
sneakier — macOS's HFS+/APFS normalizes filenames, so two
|
|
113
|
+
byte-different, canonically-equivalent names are the same file there.
|
|
114
|
+
|
|
115
|
+
This bites real projects: agents and scripts that generate files with
|
|
116
|
+
slightly different casings, Unicode filenames pasted from different
|
|
117
|
+
platforms, and the occasional CVE (see
|
|
118
|
+
[CVE-2021-21300](https://github.blog/security/vulnerability-research/github-security-lab-cve-2021-21300/),
|
|
119
|
+
where git's own checkout could be tricked on case-insensitive
|
|
120
|
+
filesystems).
|
|
121
|
+
|
|
122
|
+
## Install
|
|
123
|
+
|
|
124
|
+
```bash
|
|
125
|
+
pip install casecrash
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
Requires Python 3.9+ and git. Zero dependencies.
|
|
129
|
+
|
|
130
|
+
## License
|
|
131
|
+
|
|
132
|
+
MIT
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
casecrash/__init__.py,sha256=R2rcdGlaRrNotjlCC4vWLPDab9lu_bSfzo7DmV29HWo,191
|
|
2
|
+
casecrash/cli.py,sha256=-pWt2vIBf5xGFRx-X_wqQGs0WxhrTGbF6k3lmZgZy_M,4111
|
|
3
|
+
casecrash/core.py,sha256=3wqhjcJJohtmMLg2-72rhvv1PLMEZ366geLycMqLaVE,9428
|
|
4
|
+
casecrash-0.1.0.dist-info/licenses/LICENSE,sha256=OYhQDg7nxWoFtnF6J8NkFUA8NMw5TKv9a1WTXHY-fu4,1063
|
|
5
|
+
casecrash-0.1.0.dist-info/METADATA,sha256=xGorsyE_xQhKF1i1k4fMcvTZy6omLlDRMt67jGv4BTs,4038
|
|
6
|
+
casecrash-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
7
|
+
casecrash-0.1.0.dist-info/entry_points.txt,sha256=HYfGDKu-1RxOjfEpqjq8XKqvQXX3HWQHCCaWVWwsGTw,49
|
|
8
|
+
casecrash-0.1.0.dist-info/top_level.txt,sha256=q1xuSiKusSGVmX-P6XXE9lBP-cFfn3UaUikjNu578zU,10
|
|
9
|
+
casecrash-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 hao li
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
casecrash
|