findfmt 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
findfmt/__init__.py ADDED
@@ -0,0 +1,22 @@
1
+ """findfmt: A .gitignore-aware file discovery and classification suite."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from findfmt.classifier import classify_file, get_known_tags
6
+ from findfmt.cli import get_version, main
7
+ from findfmt.models import FileInfo, TraversalConfig
8
+ from findfmt.traversal import find_files, traverse_directory
9
+
10
+ __version__ = get_version()
11
+
12
+ __all__ = [
13
+ "FileInfo",
14
+ "TraversalConfig",
15
+ "__version__",
16
+ "classify_file",
17
+ "find_files",
18
+ "get_known_tags",
19
+ "get_version",
20
+ "main",
21
+ "traverse_directory",
22
+ ]
findfmt/__main__.py ADDED
@@ -0,0 +1,10 @@
1
+ """Entry point for python -m findfmt."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import sys
6
+
7
+ from findfmt.cli import main
8
+
9
+ if __name__ == "__main__": # pragma: no cover
10
+ sys.exit(main())
findfmt/classifier.py ADDED
@@ -0,0 +1,258 @@
1
+ """File classification and content analysis engine."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import mimetypes
6
+ import os
7
+ from typing import TYPE_CHECKING
8
+
9
+ import identify.identify as identify_engine
10
+
11
+ from findfmt.models import FileInfo
12
+
13
+ if TYPE_CHECKING:
14
+ from pathlib import Path
15
+
16
+
17
+ def get_known_tags() -> frozenset[str]:
18
+ """Retrieve all known format and type tags recognized by the engine.
19
+
20
+ Returns:
21
+ frozenset of all recognized string tags.
22
+ """
23
+ return frozenset(identify_engine.ALL_TAGS)
24
+
25
+
26
+ def extract_shebang(path: Path) -> str | None:
27
+ """Extract the shebang interpreter from the top of an executable file.
28
+
29
+ Args:
30
+ path: Path to the target file.
31
+
32
+ Returns:
33
+ The raw shebang line string if present and readable, or None.
34
+ """
35
+ try:
36
+ if not path.is_file():
37
+ return None
38
+
39
+ with path.open("rb") as f:
40
+ first_line = f.readline(512)
41
+ if first_line.startswith(b"#!"):
42
+ return first_line.decode("utf-8", errors="replace").strip()
43
+ except (OSError, UnicodeDecodeError):
44
+ return None
45
+
46
+ return None
47
+
48
+
49
+ def _resolve_relative(path: Path, root_path: Path | None) -> Path:
50
+ """Compute relative path from root_path if provided.
51
+
52
+ Args:
53
+ path: Target file path.
54
+ root_path: Optional root directory path.
55
+
56
+ Returns:
57
+ Relative path if root_path is given and matches, otherwise original path.
58
+ """
59
+ if root_path is None:
60
+ return path
61
+
62
+ try:
63
+ return path.resolve().relative_to(root_path.resolve())
64
+ except (ValueError, OSError):
65
+ return path
66
+
67
+
68
+ def _identify_tags(
69
+ path: Path,
70
+ str_path: str,
71
+ *,
72
+ exists: bool,
73
+ is_regular_or_symlink: bool = True,
74
+ ) -> set[str]:
75
+ """Extract identify tags with fallback handling.
76
+
77
+ Args:
78
+ path: Target file path.
79
+ str_path: String representation of path.
80
+ exists: Flag indicating whether path exists on disk.
81
+ is_regular_or_symlink: Flag indicating whether path is regular file or symlink.
82
+
83
+ Returns:
84
+ Set of tag strings identified for the file.
85
+ """
86
+ try:
87
+ if exists and is_regular_or_symlink:
88
+ return set(identify_engine.tags_from_path(str_path))
89
+
90
+ return set(identify_engine.tags_from_filename(path.name))
91
+ except (ValueError, OSError):
92
+ return set(identify_engine.tags_from_filename(path.name))
93
+
94
+
95
+ def _shebang_info(
96
+ path: Path,
97
+ str_path: str,
98
+ *,
99
+ is_file: bool,
100
+ is_symlink: bool,
101
+ ) -> tuple[str | None, set[str]]:
102
+ """Extract shebang and interpreter tags if applicable.
103
+
104
+ Args:
105
+ path: Target file path.
106
+ str_path: String representation of path.
107
+ is_file: Flag indicating whether path is a regular file.
108
+ is_symlink: Flag indicating whether path is a symbolic link.
109
+
110
+ Returns:
111
+ Tuple containing shebang string (or None) and set of tags.
112
+ """
113
+ if not (is_file and not is_symlink):
114
+ return None, set()
115
+
116
+ shebang = extract_shebang(path)
117
+ tags: set[str] = set()
118
+ if shebang:
119
+ parts = identify_engine.parse_shebang_from_file(str_path)
120
+ for part in parts:
121
+ tags.update(identify_engine.tags_from_interpreter(part))
122
+
123
+ return shebang, tags
124
+
125
+
126
+ def _mime_info(str_path: str) -> tuple[str | None, set[str]]:
127
+ """Extract MIME type and MIME-derived tags.
128
+
129
+ Args:
130
+ str_path: String representation of the file path.
131
+
132
+ Returns:
133
+ Tuple containing MIME type string (or None) and set of MIME tags.
134
+ """
135
+ mime_type, _ = mimetypes.guess_type(str_path)
136
+ tags: set[str] = set()
137
+ if mime_type:
138
+ tags.add(f"mime:{mime_type}")
139
+ category = mime_type.split("/", 1)[0]
140
+ tags.add(category)
141
+
142
+ return mime_type, tags
143
+
144
+
145
+ def _check_executable(path: Path, *, exists: bool, shebang: str | None = None) -> bool:
146
+ """Determine whether a file is executable across POSIX and Windows platforms.
147
+
148
+ Args:
149
+ path: File path to check.
150
+ exists: Flag indicating whether path exists on disk.
151
+ shebang: Optional shebang line string if detected.
152
+
153
+ Returns:
154
+ True if the file is considered executable on the host platform.
155
+ """
156
+ if not exists:
157
+ return False
158
+
159
+ if os.name == "nt":
160
+ pathext = os.environ.get("PATHEXT", ".COM;.EXE;.BAT;.CMD;.VBS;.JS;.WS;.MSC").split(";")
161
+ has_pathext = any(path.name.upper().endswith(ext.upper()) for ext in pathext if ext)
162
+ return has_pathext or bool(shebang)
163
+
164
+ return os.access(path, os.X_OK)
165
+
166
+
167
+ def _inspect_metadata(path: Path) -> tuple[bool, bool, bool, int]:
168
+ """Inspect filesystem metadata safely.
169
+
170
+ Args:
171
+ path: File path to inspect.
172
+
173
+ Returns:
174
+ Tuple of (is_symlink, exists, is_file, size_bytes).
175
+ """
176
+ try:
177
+ is_symlink = path.is_symlink()
178
+ except OSError:
179
+ is_symlink = False
180
+
181
+ try:
182
+ stat_result = path.lstat() if is_symlink else path.stat()
183
+ size_bytes = stat_result.st_size
184
+ except OSError:
185
+ size_bytes = 0
186
+
187
+ try:
188
+ exists = path.exists()
189
+ except OSError:
190
+ exists = False
191
+
192
+ try:
193
+ is_file = path.is_file() if exists else False
194
+ except OSError:
195
+ is_file = False
196
+
197
+ return is_symlink, exists, is_file, size_bytes
198
+
199
+
200
+ def _adjust_nt_tags(tags: set[str], *, is_executable: bool) -> None:
201
+ """Adjust executable tags on Windows platform.
202
+
203
+ Args:
204
+ tags: Set of tags to modify in place.
205
+ is_executable: Flag indicating whether path is executable.
206
+ """
207
+ if os.name != "nt":
208
+ return
209
+
210
+ if is_executable:
211
+ tags.discard("non-executable")
212
+ tags.add("executable")
213
+ else:
214
+ tags.discard("executable")
215
+ tags.add("non-executable")
216
+
217
+
218
+ def classify_file(path: Path, root_path: Path | None = None) -> FileInfo:
219
+ """Classify a given file path by inspection of name, content, and metadata.
220
+
221
+ Args:
222
+ path: The path of the file to classify.
223
+ root_path: Optional root directory used to compute relative_path.
224
+
225
+ Returns:
226
+ FileInfo containing metadata, identified tags, shebang, and MIME info.
227
+ """
228
+ str_path = str(path)
229
+ is_symlink, exists, is_file, size_bytes = _inspect_metadata(path)
230
+
231
+ tags = _identify_tags(
232
+ path,
233
+ str_path,
234
+ exists=exists,
235
+ is_regular_or_symlink=is_file or is_symlink,
236
+ )
237
+ shebang, shebang_tags = _shebang_info(path, str_path, is_file=is_file, is_symlink=is_symlink)
238
+ tags.update(shebang_tags)
239
+
240
+ is_executable = _check_executable(path, exists=exists, shebang=shebang)
241
+ _adjust_nt_tags(tags, is_executable=is_executable)
242
+
243
+ mime_type, mime_tags = _mime_info(str_path)
244
+ tags.update(mime_tags)
245
+
246
+ relative_path = _resolve_relative(path, root_path)
247
+ resolved_path = path.resolve() if exists else path
248
+
249
+ return FileInfo(
250
+ path=resolved_path,
251
+ relative_path=relative_path,
252
+ tags=frozenset(tags),
253
+ shebang=shebang,
254
+ mime_type=mime_type,
255
+ is_executable=is_executable,
256
+ is_symlink=is_symlink,
257
+ size_bytes=size_bytes,
258
+ )
findfmt/cli.py ADDED
@@ -0,0 +1,262 @@
1
+ """Command-line interface for findfmt."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import sys
7
+ from collections import Counter
8
+ from importlib.metadata import PackageNotFoundError, version
9
+ from pathlib import Path
10
+ from typing import TYPE_CHECKING
11
+
12
+ from findfmt.classifier import get_known_tags
13
+ from findfmt.models import TraversalConfig
14
+ from findfmt.traversal import find_files
15
+
16
+ if TYPE_CHECKING:
17
+ from collections.abc import Sequence
18
+
19
+ from findfmt.models import FileInfo
20
+
21
+
22
+ def get_version() -> str:
23
+ """Retrieve package version or fallback string.
24
+
25
+ Returns:
26
+ Version string.
27
+ """
28
+ try:
29
+ return version("findfmt")
30
+ except PackageNotFoundError:
31
+ return "0.1.0.dev0"
32
+
33
+
34
+ def build_parser() -> argparse.ArgumentParser:
35
+ """Construct command-line argument parser.
36
+
37
+ Returns:
38
+ Configured ArgumentParser instance.
39
+ """
40
+ parser = argparse.ArgumentParser(
41
+ prog="findfmt",
42
+ description=(
43
+ "A .gitignore-aware file discovery and classification suite that locates "
44
+ "files by content format, shebang, and MIME tag."
45
+ ),
46
+ formatter_class=argparse.RawDescriptionHelpFormatter,
47
+ )
48
+
49
+ parser.add_argument(
50
+ "paths",
51
+ nargs="*",
52
+ default=["."],
53
+ help="One or more directory or file paths to inspect (default: current directory).",
54
+ )
55
+
56
+ parser.add_argument(
57
+ "-t",
58
+ "--type",
59
+ "--tag",
60
+ dest="tags",
61
+ action="append",
62
+ help="Tag or comma-separated tags to match (e.g. 'python', 'yaml,json', 'executable').",
63
+ )
64
+
65
+ parser.add_argument(
66
+ "-e",
67
+ "--exclude",
68
+ "--exclude-tag",
69
+ dest="exclude_tags",
70
+ action="append",
71
+ help="Tag or comma-separated tags to exclude.",
72
+ )
73
+
74
+ parser.add_argument(
75
+ "--all-tags",
76
+ action="store_true",
77
+ help="Require matching files to have ALL specified tags rather than ANY tag.",
78
+ )
79
+
80
+ parser.add_argument(
81
+ "--shebang",
82
+ dest="shebang",
83
+ help="Filter files whose shebang contains this interpreter or pattern.",
84
+ )
85
+
86
+ parser.add_argument(
87
+ "--no-ignore",
88
+ action="store_true",
89
+ help="Do not respect .gitignore rules during traversal.",
90
+ )
91
+
92
+ parser.add_argument(
93
+ "--hidden",
94
+ action="store_true",
95
+ help="Include hidden files and directories.",
96
+ )
97
+
98
+ parser.add_argument(
99
+ "-L",
100
+ "--follow-symlinks",
101
+ action="store_true",
102
+ help="Follow symbolic links during traversal.",
103
+ )
104
+
105
+ parser.add_argument(
106
+ "--absolute",
107
+ action="store_true",
108
+ help="Output absolute paths rather than paths relative to the traversal root.",
109
+ )
110
+
111
+ parser.add_argument(
112
+ "-0",
113
+ "--print0",
114
+ action="store_true",
115
+ help=r"Delimit path outputs with a NUL (\0) character instead of a newline.",
116
+ )
117
+
118
+ parser.add_argument(
119
+ "-l",
120
+ "--list-tags",
121
+ action="store_true",
122
+ help="Display identified tags alongside each matched path.",
123
+ )
124
+
125
+ parser.add_argument(
126
+ "-s",
127
+ "--summary",
128
+ action="store_true",
129
+ help="Print summary match statistics to stderr.",
130
+ )
131
+
132
+ parser.add_argument(
133
+ "--known-tags",
134
+ action="store_true",
135
+ help="List all known classification tags supported by the engine and exit.",
136
+ )
137
+
138
+ parser.add_argument(
139
+ "-v",
140
+ "--version",
141
+ action="version",
142
+ version=f"%(prog)s {get_version()}",
143
+ )
144
+
145
+ return parser
146
+
147
+
148
+ def parse_tag_arguments(tag_args: Sequence[str] | None) -> frozenset[str]:
149
+ """Parse repeatable or comma-delimited tag arguments into a normalized frozenset.
150
+
151
+ Args:
152
+ tag_args: Raw arguments provided to tag filter flags.
153
+
154
+ Returns:
155
+ frozenset of lowercase tag strings.
156
+ """
157
+ if not tag_args:
158
+ return frozenset[str]()
159
+
160
+ result: set[str] = set()
161
+ for arg in tag_args:
162
+ for tag in arg.split(","):
163
+ cleaned = tag.strip().lower()
164
+ if cleaned:
165
+ result.add(cleaned)
166
+
167
+ return frozenset(result)
168
+
169
+
170
+ def _format_match(file_info: FileInfo, *, absolute: bool, show_tags: bool, delimiter: str) -> str:
171
+ """Format matching file information for stdout output.
172
+
173
+ Args:
174
+ file_info: Classified file information.
175
+ absolute: Whether to format using absolute path.
176
+ show_tags: Whether to append comma-separated tags.
177
+ delimiter: End of line delimiter string.
178
+
179
+ Returns:
180
+ Formatted string for output.
181
+ """
182
+ path_str = str(file_info.path if absolute else file_info.relative_path)
183
+ if show_tags:
184
+ tags_repr = ", ".join(sorted(file_info.tags))
185
+ return f"{path_str} [{tags_repr}]{delimiter}"
186
+
187
+ return f"{path_str}{delimiter}"
188
+
189
+
190
+ def _write_summary(match_count: int, tag_counter: Counter[str]) -> None:
191
+ """Write execution summary to stderr.
192
+
193
+ Args:
194
+ match_count: Total number of files matched.
195
+ tag_counter: Frequency counter of tags matched.
196
+ """
197
+ sys.stderr.write(f"\n--- findfmt summary ---\nMatched files: {match_count}\n")
198
+ if tag_counter:
199
+ sys.stderr.write("Top tags:\n")
200
+ for tag, count in tag_counter.most_common(10):
201
+ sys.stderr.write(f" {tag}: {count}\n")
202
+
203
+
204
+ def main(argv: Sequence[str] | None = None) -> int:
205
+ """Main CLI entrypoint.
206
+
207
+ Args:
208
+ argv: Optional command-line arguments (defaults to sys.argv[1:]).
209
+
210
+ Returns:
211
+ Integer exit code (0 for success, 1 on error).
212
+ """
213
+ parser = build_parser()
214
+ args = parser.parse_args(argv)
215
+
216
+ if args.known_tags:
217
+ for tag in sorted(get_known_tags()):
218
+ sys.stdout.write(f"{tag}\n")
219
+
220
+ return 0
221
+
222
+ root_paths = tuple(Path(p) for p in args.paths)
223
+ config = TraversalConfig(
224
+ root_paths=root_paths,
225
+ include_tags=parse_tag_arguments(args.tags),
226
+ exclude_tags=parse_tag_arguments(args.exclude_tags),
227
+ all_tags=args.all_tags,
228
+ shebang_filter=args.shebang,
229
+ respect_gitignore=not args.no_ignore,
230
+ include_hidden=args.hidden,
231
+ follow_symlinks=args.follow_symlinks,
232
+ relative_paths=not args.absolute,
233
+ null_delimited=args.print0,
234
+ show_tags=args.list_tags,
235
+ show_summary=args.summary,
236
+ )
237
+
238
+ tag_counter: Counter[str] = Counter()
239
+ match_count = 0
240
+ delimiter = "\0" if config.null_delimited else "\n"
241
+
242
+ for file_info in find_files(config):
243
+ match_count += 1
244
+ if config.show_summary:
245
+ tag_counter.update(file_info.tags)
246
+
247
+ formatted = _format_match(
248
+ file_info,
249
+ absolute=args.absolute,
250
+ show_tags=config.show_tags,
251
+ delimiter=delimiter,
252
+ )
253
+ sys.stdout.write(formatted)
254
+
255
+ if config.show_summary:
256
+ _write_summary(match_count, tag_counter)
257
+
258
+ return 0
259
+
260
+
261
+ if __name__ == "__main__":
262
+ sys.exit(main())
findfmt/models.py ADDED
@@ -0,0 +1,64 @@
1
+ """Core data models and configurations for findfmt."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass, field
6
+ from pathlib import Path
7
+
8
+
9
+ @dataclass(frozen=True, slots=True)
10
+ class FileInfo:
11
+ """Represents file metadata, classification tags, and content attributes.
12
+
13
+ Attributes:
14
+ path: Absolute path to the file.
15
+ relative_path: Path relative to the traversal root.
16
+ tags: Set of tags assigned by the classifier (e.g., 'python', 'text').
17
+ shebang: Extracted shebang line if present, otherwise None.
18
+ mime_type: Detected MIME type if available, otherwise None.
19
+ is_executable: Whether the file has executable permissions.
20
+ is_symlink: Whether the path is a symbolic link.
21
+ size_bytes: File size in bytes.
22
+ """
23
+
24
+ path: Path
25
+ relative_path: Path
26
+ tags: frozenset[str] = field(default_factory=frozenset)
27
+ shebang: str | None = None
28
+ mime_type: str | None = None
29
+ is_executable: bool = False
30
+ is_symlink: bool = False
31
+ size_bytes: int = 0
32
+
33
+
34
+ @dataclass(frozen=True, slots=True)
35
+ class TraversalConfig:
36
+ r"""Configuration options for repository traversal and format filtering.
37
+
38
+ Attributes:
39
+ root_paths: Roots from which to begin file discovery.
40
+ include_tags: Tags that files must have to be included.
41
+ exclude_tags: Tags that disqualify a file from inclusion.
42
+ all_tags: If True, file must match all include_tags; if False, any tag.
43
+ shebang_filter: Substring or interpreter name required in shebang.
44
+ respect_gitignore: Whether to ignore paths matched by gitignore rules.
45
+ include_hidden: Whether to inspect hidden files and directories.
46
+ follow_symlinks: Whether to resolve and traverse symbolic links.
47
+ relative_paths: Whether to output relative paths rather than absolute.
48
+ null_delimited: Whether to delimit output paths with NUL bytes (\0).
49
+ show_tags: Whether to print identified tags alongside file paths.
50
+ show_summary: Whether to output summary statistics of matches.
51
+ """
52
+
53
+ root_paths: tuple[Path, ...] = field(default_factory=lambda: (Path(),))
54
+ include_tags: frozenset[str] = field(default_factory=frozenset)
55
+ exclude_tags: frozenset[str] = field(default_factory=frozenset)
56
+ all_tags: bool = False
57
+ shebang_filter: str | None = None
58
+ respect_gitignore: bool = True
59
+ include_hidden: bool = False
60
+ follow_symlinks: bool = False
61
+ relative_paths: bool = True
62
+ null_delimited: bool = False
63
+ show_tags: bool = False
64
+ show_summary: bool = False
findfmt/py.typed ADDED
File without changes
findfmt/traversal.py ADDED
@@ -0,0 +1,297 @@
1
+ """Git-conscious directory traversal and filtering engine."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import contextlib
6
+ import os
7
+ from pathlib import Path
8
+ from typing import TYPE_CHECKING, Any
9
+
10
+ import pathspec
11
+
12
+ from findfmt.classifier import classify_file
13
+
14
+ if TYPE_CHECKING:
15
+ from collections.abc import Iterator
16
+
17
+ from findfmt.models import FileInfo, TraversalConfig
18
+
19
+ PathSpecType = pathspec.PathSpec[Any]
20
+
21
+
22
+ def load_gitignore_spec(directory: Path) -> PathSpecType | None:
23
+ """Load gitignore patterns from a directory if a .gitignore file exists.
24
+
25
+ Args:
26
+ directory: Directory to check for .gitignore.
27
+
28
+ Returns:
29
+ PathSpec instance if rules exist, or None.
30
+ """
31
+ gitignore_file = directory / ".gitignore"
32
+ if gitignore_file.is_file():
33
+ try:
34
+ with gitignore_file.open("r", encoding="utf-8", errors="replace") as f:
35
+ lines = [line.strip() for line in f if line.strip() and not line.startswith("#")]
36
+ if lines:
37
+ return pathspec.PathSpec.from_lines("gitignore", lines)
38
+ except OSError:
39
+ return None
40
+
41
+ return None
42
+
43
+
44
+ def load_git_exclude_spec(root: Path) -> PathSpecType | None:
45
+ """Load exclude patterns from .git/info/exclude if present.
46
+
47
+ Args:
48
+ root: Git repository root directory.
49
+
50
+ Returns:
51
+ PathSpec instance if exclude patterns exist, or None.
52
+ """
53
+ exclude_file = root / ".git" / "info" / "exclude"
54
+ if exclude_file.is_file():
55
+ try:
56
+ with exclude_file.open("r", encoding="utf-8", errors="replace") as f:
57
+ lines = [line.strip() for line in f if line.strip() and not line.startswith("#")]
58
+ if lines:
59
+ return pathspec.PathSpec.from_lines("gitignore", lines)
60
+ except OSError:
61
+ return None
62
+
63
+ return None
64
+
65
+
66
+ def should_skip_dir(dir_name: str, *, include_hidden: bool) -> bool:
67
+ """Determine if a directory should be skipped during descent.
68
+
69
+ Args:
70
+ dir_name: Basename of directory.
71
+ include_hidden: Whether hidden directories are included.
72
+
73
+ Returns:
74
+ True if the directory should be skipped, False otherwise.
75
+ """
76
+ if dir_name == ".git":
77
+ return True
78
+
79
+ return bool(not include_hidden and dir_name.startswith("."))
80
+
81
+
82
+ def matches_filter(file_info: FileInfo, config: TraversalConfig) -> bool:
83
+ """Determine whether a classified file satisfies traversal filter criteria.
84
+
85
+ Args:
86
+ file_info: The classified FileInfo object.
87
+ config: The active TraversalConfig filter settings.
88
+
89
+ Returns:
90
+ True if the file satisfies all filters, False otherwise.
91
+ """
92
+ tags = file_info.tags
93
+
94
+ # Check exclude tags first
95
+ if config.exclude_tags and (tags & config.exclude_tags):
96
+ return False
97
+
98
+ # Check include tags
99
+ if config.include_tags:
100
+ if config.all_tags and not config.include_tags.issubset(tags):
101
+ return False
102
+
103
+ if not config.all_tags and not (tags & config.include_tags):
104
+ return False
105
+
106
+ # Check shebang filter
107
+ if config.shebang_filter:
108
+ return bool(file_info.shebang and config.shebang_filter in file_info.shebang)
109
+
110
+ return True
111
+
112
+
113
+ def _load_active_specs(
114
+ root: Path,
115
+ active_specs: tuple[tuple[Path, PathSpecType], ...],
116
+ *,
117
+ respect_gitignore: bool,
118
+ ) -> tuple[tuple[Path, PathSpecType], ...]:
119
+ """Assemble active gitignore specifications for a given directory."""
120
+ if not respect_gitignore:
121
+ return ()
122
+
123
+ current_specs = list(active_specs)
124
+ if not active_specs:
125
+ git_exclude = load_git_exclude_spec(root)
126
+ if git_exclude:
127
+ current_specs.append((root, git_exclude))
128
+
129
+ local_spec = load_gitignore_spec(root)
130
+ if local_spec:
131
+ current_specs.append((root, local_spec))
132
+
133
+ return tuple(current_specs)
134
+
135
+
136
+ def _is_path_ignored(
137
+ entry_path: Path,
138
+ *,
139
+ is_dir: bool,
140
+ specs: tuple[tuple[Path, PathSpecType], ...],
141
+ ) -> bool:
142
+ """Check if an entry matches any active gitignore rule."""
143
+ for base_dir, spec in specs:
144
+ if not entry_path.is_relative_to(base_dir):
145
+ continue
146
+
147
+ rel = entry_path.relative_to(base_dir)
148
+ rel_str = str(rel) + ("/" if is_dir else "")
149
+ if spec.match_file(rel_str):
150
+ return True
151
+
152
+ return False
153
+
154
+
155
+ def _scan_directory_entries(
156
+ root: Path,
157
+ config: TraversalConfig,
158
+ specs: tuple[tuple[Path, PathSpecType], ...],
159
+ ) -> tuple[list[Path], list[Path]]:
160
+ """Scan and partition directory entries into subdirectories and files."""
161
+ try:
162
+ entries = sorted(os.scandir(root), key=lambda e: e.name)
163
+ except OSError:
164
+ return [], []
165
+
166
+ subdirs: list[Path] = []
167
+ files: list[Path] = []
168
+
169
+ for entry in entries:
170
+ if not config.include_hidden and entry.name.startswith("."):
171
+ continue
172
+
173
+ entry_path = Path(entry.path)
174
+ is_dir = entry.is_dir(follow_symlinks=config.follow_symlinks)
175
+
176
+ if config.respect_gitignore and _is_path_ignored(entry_path, is_dir=is_dir, specs=specs):
177
+ continue
178
+
179
+ if is_dir:
180
+ if not should_skip_dir(entry.name, include_hidden=config.include_hidden):
181
+ subdirs.append(entry_path)
182
+ elif entry.is_file(follow_symlinks=config.follow_symlinks):
183
+ files.append(entry_path)
184
+
185
+ return subdirs, files
186
+
187
+
188
+ def _should_skip_symlink_dir(
189
+ subdir: Path,
190
+ *,
191
+ follow_symlinks: bool,
192
+ visited: set[Path] | None,
193
+ ) -> bool:
194
+ """Determine whether a candidate directory should be skipped to break symlink loops.
195
+
196
+ Args:
197
+ subdir: Directory path candidate.
198
+ follow_symlinks: Traversal configuration flag for following symlinks.
199
+ visited: Set of canonical directory paths already visited.
200
+
201
+ Returns:
202
+ True if the directory should be skipped, False otherwise.
203
+ """
204
+ if not (follow_symlinks and visited is not None):
205
+ return False
206
+
207
+ try:
208
+ resolved = subdir.resolve()
209
+ except OSError:
210
+ return True
211
+
212
+ if resolved in visited:
213
+ return True
214
+
215
+ visited.add(resolved)
216
+ return False
217
+
218
+
219
+ def traverse_directory(
220
+ root: Path,
221
+ config: TraversalConfig,
222
+ active_specs: tuple[tuple[Path, PathSpecType], ...] = (),
223
+ base_root: Path | None = None,
224
+ visited_dirs: set[Path] | None = None,
225
+ ) -> Iterator[FileInfo]:
226
+ """Recursively traverse a directory hierarchy honoring .gitignore and filters.
227
+
228
+ Args:
229
+ root: The current directory to traverse.
230
+ config: Traversal configuration options.
231
+ active_specs: Inherited parent gitignore specs with their base paths.
232
+ base_root: The top-level root directory used for relative paths.
233
+ visited_dirs: Tracked set of canonical directory paths visited to break cycles.
234
+
235
+ Yields:
236
+ FileInfo objects for matching files in deterministic sorted order.
237
+ """
238
+ effective_base = base_root if base_root is not None else root
239
+ active_visited = visited_dirs
240
+ if config.follow_symlinks and active_visited is None:
241
+ active_visited = set()
242
+ with contextlib.suppress(OSError):
243
+ active_visited.add(root.resolve())
244
+
245
+ specs = _load_active_specs(
246
+ root,
247
+ active_specs,
248
+ respect_gitignore=config.respect_gitignore,
249
+ )
250
+ subdirs, files = _scan_directory_entries(root, config, specs)
251
+
252
+ for file_path in files:
253
+ info = classify_file(file_path, root_path=effective_base)
254
+ if matches_filter(info, config):
255
+ yield info
256
+
257
+ for subdir in subdirs:
258
+ if _should_skip_symlink_dir(
259
+ subdir,
260
+ follow_symlinks=config.follow_symlinks,
261
+ visited=active_visited,
262
+ ):
263
+ continue
264
+
265
+ yield from traverse_directory(
266
+ subdir,
267
+ config,
268
+ specs,
269
+ base_root=effective_base,
270
+ visited_dirs=active_visited,
271
+ )
272
+
273
+
274
+ def find_files(config: TraversalConfig) -> Iterator[FileInfo]:
275
+ """Discover and classify files across all configured root paths.
276
+
277
+ Args:
278
+ config: Traversal and filtering configuration.
279
+
280
+ Yields:
281
+ FileInfo objects matching the criteria.
282
+ """
283
+ for root in sorted(config.root_paths):
284
+ resolved_root = root.resolve()
285
+ if not resolved_root.exists():
286
+ continue
287
+
288
+ if resolved_root.is_file():
289
+ info = classify_file(resolved_root, root_path=resolved_root.parent)
290
+ if matches_filter(info, config):
291
+ yield info
292
+ else:
293
+ yield from traverse_directory(
294
+ resolved_root,
295
+ config,
296
+ base_root=resolved_root,
297
+ )
@@ -0,0 +1,220 @@
1
+ Metadata-Version: 2.5
2
+ Name: findfmt
3
+ Version: 0.1.0
4
+ Summary: A.gitignore-aware file discovery and classification suite that locates files by content format, shebang, and MIME tag for automated linting, formatting, and CI pipelines.
5
+ Project-URL: Changelog, https://github.com/bdperkin/findfmt/blob/main/CHANGELOG.md
6
+ Project-URL: Documentation, https://bdperkin.github.io/findfmt
7
+ Project-URL: Homepage, https://github.com/bdperkin/findfmt
8
+ Project-URL: Issues, https://github.com/bdperkin/findfmt/issues
9
+ Project-URL: Repository, https://github.com/bdperkin/findfmt
10
+ Author-email: Brandon Perkins <bdperkin@gmail.com>
11
+ License: MIT
12
+ License-File: LICENSE
13
+ Keywords: classification,discovery,find,format,gitignore,identify,linting,mime,shebang
14
+ Classifier: Development Status :: 4 - Beta
15
+ Classifier: Environment :: Console
16
+ Classifier: Intended Audience :: Developers
17
+ Classifier: License :: OSI Approved :: MIT License
18
+ Classifier: Operating System :: OS Independent
19
+ Classifier: Programming Language :: Python :: 3 :: Only
20
+ Classifier: Programming Language :: Python :: 3.10
21
+ Classifier: Programming Language :: Python :: 3.11
22
+ Classifier: Programming Language :: Python :: 3.12
23
+ Classifier: Programming Language :: Python :: 3.13
24
+ Classifier: Topic :: Software Development :: Quality Assurance
25
+ Classifier: Topic :: Utilities
26
+ Requires-Python: >=3.10
27
+ Requires-Dist: identify>=2.6
28
+ Requires-Dist: pathspec>=0.12
29
+ Requires-Dist: tomli>=2.0.1; python_version < '3.11'
30
+ Description-Content-Type: text/markdown
31
+
32
+ <p align="center">
33
+ <img src="assets/logo.svg" alt="findfmt Logo" width="600" />
34
+ </p>
35
+
36
+ <p align="center">
37
+ <a href="https://github.com/bdperkin/findfmt/actions/workflows/ci.yml"><img src="https://github.com/bdperkin/findfmt/actions/workflows/ci.yml/badge.svg" alt="CI Status" /></a>
38
+ <a href="https://github.com/bdperkin/findfmt/actions/workflows/codeql.yml"><img src="https://github.com/bdperkin/findfmt/actions/workflows/codeql.yml/badge.svg" alt="CodeQL Analysis" /></a>
39
+ <a href="https://github.com/bdperkin/findfmt/actions/workflows/docs.yml"><img src="https://github.com/bdperkin/findfmt/actions/workflows/docs.yml/badge.svg" alt="Documentation Status" /></a>
40
+ <a href="https://results.pre-commit.ci/latest/github/bdperkin/findfmt/main"><img src="https://results.pre-commit.ci/badge/github/bdperkin/findfmt/main.svg" alt="pre-commit.ci status" /></a>
41
+ <a href="https://codecov.io/gh/bdperkin/findfmt"><img src="https://codecov.io/gh/bdperkin/findfmt/branch/main/graph/badge.svg" alt="Coverage" /></a>
42
+ </p>
43
+
44
+ <p align="center">
45
+ <a href="https://pypi.org/project/findfmt/"><img src="https://img.shields.io/pypi/v/findfmt.svg?logo=pypi&logoColor=white" alt="PyPI Version" /></a>
46
+ <a href="https://pypi.org/project/findfmt/"><img src="https://img.shields.io/pypi/pyversions/findfmt.svg?logo=python&logoColor=white" alt="Python Versions" /></a>
47
+ <a href="https://pypi.org/project/findfmt/"><img src="https://img.shields.io/pypi/wheel/findfmt.svg" alt="PyPI Wheel" /></a>
48
+ <a href="https://github.com/bdperkin/findfmt/pkgs/container/findfmt"><img src="https://img.shields.io/badge/GHCR-container-blue?logo=docker&logoColor=white" alt="GHCR Container" /></a>
49
+ <a href="https://github.com/bdperkin/findfmt/blob/main/LICENSE"><img src="https://img.shields.io/badge/License-MIT-blue.svg" alt="License: MIT" /></a>
50
+ </p>
51
+
52
+ <p align="center">
53
+ <a href="https://github.com/astral-sh/uv"><img src="https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/uv/main/assets/badge/v0.json" alt="uv" /></a>
54
+ <a href="https://github.com/astral-sh/ruff"><img src="https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/ruff/main/assets/badge/v2.json" alt="Ruff" /></a>
55
+ <a href="https://github.com/astral-sh/ty"><img src="https://img.shields.io/badge/type--checked-ty-blueviolet" alt="Type-checked: ty" /></a>
56
+ <a href="https://interrogate.readthedocs.io/"><img src="https://img.shields.io/badge/interrogate-100%25-brightgreen" alt="Docstring Coverage: 100%" /></a>
57
+ <a href="https://conventionalcommits.org"><img src="https://img.shields.io/badge/Conventional%20Commits-1.0.0-yellow.svg" alt="Conventional Commits" /></a>
58
+ <a href="ACCESSIBILITY.md"><img src="https://img.shields.io/badge/accessibility-WCAG%202.1%20AA-blue" alt="Accessibility WCAG 2.1 AA" /></a>
59
+ </p>
60
+
61
+ ______________________________________________________________________
62
+
63
+ > **A `.gitignore`-aware file discovery and classification suite that locates files by content
64
+ > format, shebang, and MIME tag for automated linting, formatting, and CI pipelines.**
65
+
66
+ ## 1. Why findfmt?
67
+
68
+ Unlike traditional `find` or globbing tools that rely strictly on file extensions, `findfmt`:
69
+
70
+ - **Understands file content**: Identifies format by content, shebang (`#!/usr/bin/env python3`),
71
+ and MIME types using the `identify` engine.
72
+ - **Respects Git**: Traverses trees hierarchically while pruning `.gitignore` and
73
+ `.git/info/exclude` paths early before descending into large directories (e.g. `node_modules/`,
74
+ `.venv/`).
75
+ - **Deterministic**: Always returns clean, relative, deterministically sorted paths optimized for
76
+ subshells, xargs, and automation.
77
+
78
+ ______________________________________________________________________
79
+
80
+ ## 2. Installation
81
+
82
+ ### 2.1. With `uv` (Recommended)
83
+
84
+ Run directly with `uvx`:
85
+
86
+ ```bash
87
+ uvx findfmt --help
88
+ ```
89
+
90
+ Install globally:
91
+
92
+ ```bash
93
+ uv tool install findfmt
94
+ ```
95
+
96
+ Or add to your project:
97
+
98
+ ```bash
99
+ uv add findfmt
100
+ ```
101
+
102
+ ### 2.2. With `pip`
103
+
104
+ ```bash
105
+ pip install findfmt
106
+ ```
107
+
108
+ ### 2.3. With Docker / Container
109
+
110
+ ```bash
111
+ docker pull ghcr.io/bdperkin/findfmt:latest
112
+ docker run --rm -v "$(pwd)":/workspace -w /workspace ghcr.io/bdperkin/findfmt:latest -t python
113
+ ```
114
+
115
+ ______________________________________________________________________
116
+
117
+ ## 3. Usage
118
+
119
+ ### 3.1. Locate Files by Tag / Format
120
+
121
+ ```bash
122
+ # Locate all Python files
123
+ findfmt -t python
124
+
125
+ # Locate all YAML and JSON files
126
+ findfmt -t yaml,json
127
+
128
+ # Locate shell scripts
129
+ findfmt -t shell
130
+ ```
131
+
132
+ ### 3.2. Shebang Filtering
133
+
134
+ ```bash
135
+ # Locate files with bash shebang
136
+ findfmt --shebang bash
137
+
138
+ # Locate scripts executing with python
139
+ findfmt --shebang python
140
+ ```
141
+
142
+ ### 3.3. Pipe Safely to Linters and Tools
143
+
144
+ Use `-0` for NUL-delimited output with `xargs -0`:
145
+
146
+ ```bash
147
+ # Format discovered Python files
148
+ findfmt -t python -0 | xargs -0 ruff format
149
+
150
+ # Lint shell scripts with shellcheck
151
+ findfmt -t shell -0 | xargs -r -0 shellcheck
152
+ ```
153
+
154
+ ### 3.4. Inspect Tags & Summaries
155
+
156
+ ```bash
157
+ # Print matched files and their classification tags
158
+ findfmt -l -t python
159
+
160
+ # Print summary statistics to stderr
161
+ findfmt -s
162
+ ```
163
+
164
+ ______________________________________________________________________
165
+
166
+ ## 4. CLI Options
167
+
168
+ | Flag | Description |
169
+ | ------------------------------ | -------------------------------------------------- |
170
+ | `-t, --tag, --type` | Match files containing specified tag(s) |
171
+ | `-e, --exclude, --exclude-tag` | Exclude files containing specified tag(s) |
172
+ | `--all-tags` | Require match against all include tags (AND logic) |
173
+ | `--shebang` | Match shebang interpreter name or pattern |
174
+ | `--no-ignore` | Do not prune paths matching `.gitignore` |
175
+ | `--hidden` | Inspect hidden files and directories |
176
+ | `-0, --print0` | NUL-delimited output for `xargs -0` |
177
+ | `-l, --list-tags` | Print tags alongside file paths |
178
+ | `-s, --summary` | Print match frequencies to stderr |
179
+ | `--absolute` | Output absolute rather than relative paths |
180
+ | `--known-tags` | List all supported classification tags |
181
+ | `-v, --version` | Display version and exit |
182
+
183
+ ______________________________________________________________________
184
+
185
+ ## 5. Development & Testing
186
+
187
+ This project enforces 100% test coverage and strict type checking:
188
+
189
+ ```bash
190
+ # Clone the repository
191
+ git clone https://github.com/bdperkin/findfmt.git
192
+ cd findfmt
193
+
194
+ # Install dependencies with uv
195
+ uv sync --all-groups
196
+
197
+ # Run tests and verify 100% coverage
198
+ uv run pytest
199
+
200
+ # Run linting and typing
201
+ uv run ruff check
202
+ uv run ty check
203
+
204
+ # Run full verification suite
205
+ uv run python tools/verify_quality.py
206
+ ```
207
+
208
+ ______________________________________________________________________
209
+
210
+ ## 6. Governance & Community
211
+
212
+ - [Contributing Guidelines](CONTRIBUTING.md)
213
+ - [Code of Conduct](CODE_OF_CONDUCT.md)
214
+ - [Security Policy](SECURITY.md)
215
+ - [Support Information](SUPPORT.md)
216
+ - [Accessibility Statement](ACCESSIBILITY.md)
217
+
218
+ ## 7. License
219
+
220
+ [MIT License](LICENSE) © 2026 Brandon Perkins.
@@ -0,0 +1,12 @@
1
+ findfmt/__init__.py,sha256=XEJBW3qrJKLsExfWyvx3M5tIRKfyBECf8UA4Uw24Vbk,549
2
+ findfmt/__main__.py,sha256=Y38nYVtAEYxkPM-8pJSmIiIL1maIJEPnsl8-ZXsdCek,188
3
+ findfmt/classifier.py,sha256=KCkYwjcHOG9-IhWTIApZ_zjPLX_lLrOSOaMdKpY8pG0,7089
4
+ findfmt/cli.py,sha256=A3T_JkiDzUrAq0mpXw6Yp6aUB-LKiulnqrYAOpbvQ0c,6964
5
+ findfmt/models.py,sha256=3903OEMGOY9w39TSYZWbsJItcnO7WTy_20wwoQwjHvE,2585
6
+ findfmt/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
7
+ findfmt/traversal.py,sha256=EjYA7-pEJIclK24MBik4AyBc4F2PdaBj1Q3K76TQHoE,8681
8
+ findfmt-0.1.0.dist-info/METADATA,sha256=icoW7gl_O3t19evOKUXmJxynfPrTInmUYI8o1sqiVHM,8745
9
+ findfmt-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
10
+ findfmt-0.1.0.dist-info/entry_points.txt,sha256=1mZdpaiLuEes0P-UzwIy7yvg0YChbQiFUpBQodfKZmE,45
11
+ findfmt-0.1.0.dist-info/licenses/LICENSE,sha256=3U-6gkW73x_v0Qc79rs2mQfYVYQAawO7tHREHDkO3UA,1072
12
+ findfmt-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ findfmt = findfmt.cli:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Brandon Perkins
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.