fpathlib 0.1.5__tar.gz → 0.1.7__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. {fpathlib-0.1.5/src/fpathlib.egg-info → fpathlib-0.1.7}/PKG-INFO +3 -3
  2. {fpathlib-0.1.5 → fpathlib-0.1.7}/README.md +2 -2
  3. {fpathlib-0.1.5 → fpathlib-0.1.7}/conda-recipe/meta.yaml +2 -2
  4. {fpathlib-0.1.5 → fpathlib-0.1.7}/docs/source/api.rst +4 -2
  5. {fpathlib-0.1.5 → fpathlib-0.1.7}/docs/source/index.rst +2 -2
  6. {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib/__init__.py +6 -6
  7. {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib/_version.py +3 -3
  8. {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib/expand.py +27 -34
  9. {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib/ext/polars.py +170 -72
  10. {fpathlib-0.1.5 → fpathlib-0.1.7/src/fpathlib.egg-info}/PKG-INFO +3 -3
  11. fpathlib-0.1.7/src/fpathlib.egg-info/scm_version.json +8 -0
  12. fpathlib-0.1.5/src/fpathlib.egg-info/scm_version.json +0 -8
  13. {fpathlib-0.1.5 → fpathlib-0.1.7}/.gitignore +0 -0
  14. {fpathlib-0.1.5 → fpathlib-0.1.7}/LICENSE +0 -0
  15. {fpathlib-0.1.5 → fpathlib-0.1.7}/MANIFEST.in +0 -0
  16. {fpathlib-0.1.5 → fpathlib-0.1.7}/TODO.txt +0 -0
  17. {fpathlib-0.1.5 → fpathlib-0.1.7}/docs/Makefile +0 -0
  18. {fpathlib-0.1.5 → fpathlib-0.1.7}/docs/make.bat +0 -0
  19. {fpathlib-0.1.5 → fpathlib-0.1.7}/docs/source/conf.py +0 -0
  20. {fpathlib-0.1.5 → fpathlib-0.1.7}/pyproject.toml +0 -0
  21. {fpathlib-0.1.5 → fpathlib-0.1.7}/scripts/conda.sh +0 -0
  22. {fpathlib-0.1.5 → fpathlib-0.1.7}/scripts/deploy.sh +0 -0
  23. {fpathlib-0.1.5 → fpathlib-0.1.7}/scripts/docs.sh +0 -0
  24. {fpathlib-0.1.5 → fpathlib-0.1.7}/scripts/pypi.sh +0 -0
  25. {fpathlib-0.1.5 → fpathlib-0.1.7}/scripts/tag.sh +0 -0
  26. {fpathlib-0.1.5 → fpathlib-0.1.7}/setup.cfg +0 -0
  27. {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib/ext/__init__.py +0 -0
  28. {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib/fpath.py +0 -0
  29. {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib/path.py +0 -0
  30. {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib/utils.py +0 -0
  31. {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib.egg-info/SOURCES.txt +0 -0
  32. {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib.egg-info/dependency_links.txt +0 -0
  33. {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib.egg-info/requires.txt +0 -0
  34. {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib.egg-info/scm_file_list.json +0 -0
  35. {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: fpathlib
3
- Version: 0.1.5
3
+ Version: 0.1.7
4
4
  Summary: A package to combine paths with metadata
5
5
  Author-email: "C. Lockhart" <clockha2@gmu.edu>
6
6
  Requires-Python: >=3.12
@@ -25,10 +25,10 @@ does this with an f-string-like path pattern -- `FPath` -- whose
25
25
  `{variable}` fields are captured out of every matching path on disk.
26
26
 
27
27
  ```python
28
- from fpathlib import expand_fpath
28
+ from fpathlib import expand
29
29
 
30
30
  # Given files like data/tr1/output/0/job2.log, data/tr2/output/1/job0.log, ...
31
- expanded = expand_fpath("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
31
+ expanded = expand("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
32
32
 
33
33
  for path in expanded:
34
34
  print(path, path.metadata)
@@ -8,10 +8,10 @@ does this with an f-string-like path pattern -- `FPath` -- whose
8
8
  `{variable}` fields are captured out of every matching path on disk.
9
9
 
10
10
  ```python
11
- from fpathlib import expand_fpath
11
+ from fpathlib import expand
12
12
 
13
13
  # Given files like data/tr1/output/0/job2.log, data/tr2/output/1/job0.log, ...
14
- expanded = expand_fpath("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
14
+ expanded = expand("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
15
15
 
16
16
  for path in expanded:
17
17
  print(path, path.metadata)
@@ -1,5 +1,5 @@
1
1
  {% set name = "fpathlib" %}
2
- {% set version = "0.1.4" %}
2
+ {% set version = "0.1.6" %}
3
3
 
4
4
  package:
5
5
  name: {{ name|lower }}
@@ -7,7 +7,7 @@ package:
7
7
 
8
8
  source:
9
9
  url: https://pypi.io/packages/source/{{ name[0] }}/{{ name }}/fpathlib-{{ version }}.tar.gz
10
- sha256: 8ff44332a465c2adb5c45f94d6ee79c958fb50c39e896b16734d0d8208026c77
10
+ sha256: a7a3b1d99483a33a81db47de608a6e791a5afa155d062a5914fd9eee98878da1
11
11
 
12
12
  build:
13
13
  noarch: python
@@ -23,9 +23,11 @@ FPath and ExpandedFPath
23
23
  Expanding paths
24
24
  ----------------
25
25
 
26
- .. autofunction:: fpathlib.expand_fpath
26
+ .. autofunction:: fpathlib.expand
27
27
 
28
- .. autofunction:: fpathlib.expand_fpath_decorator
28
+ .. autofunction:: fpathlib.iexpand
29
+
30
+ .. autofunction:: fpathlib.expand_arg
29
31
 
30
32
  .. autofunction:: fpathlib.is_expandable
31
33
 
@@ -11,10 +11,10 @@ own filename.
11
11
 
12
12
  .. code-block:: python
13
13
 
14
- from fpathlib import expand_fpath
14
+ from fpathlib import expand
15
15
 
16
16
  # Given files like data/tr1/output/0/job2.log, data/tr2/output/1/job0.log, ...
17
- expanded = expand_fpath("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
17
+ expanded = expand("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
18
18
 
19
19
  for path in expanded:
20
20
  print(path, path.metadata)
@@ -1,9 +1,9 @@
1
1
  from fpathlib.path import Path
2
2
  from fpathlib.fpath import FPath, ExpandedFPath
3
3
  from fpathlib.expand import (
4
- expand_fpath,
5
- iexpand_fpath,
6
- expand_fpath_decorator,
4
+ expand,
5
+ iexpand,
6
+ expand_arg,
7
7
  is_expandable,
8
8
  )
9
9
 
@@ -11,8 +11,8 @@ __all__ = [
11
11
  "Path",
12
12
  "FPath",
13
13
  "ExpandedFPath",
14
- "expand_fpath",
15
- "iexpand_fpath",
16
- "expand_fpath_decorator",
14
+ "expand",
15
+ "iexpand",
16
+ "expand_arg",
17
17
  "is_expandable",
18
18
  ]
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '0.1.5'
22
- __version_tuple__ = version_tuple = (0, 1, 5)
21
+ __version__ = version = '0.1.7'
22
+ __version_tuple__ = version_tuple = (0, 1, 7)
23
23
 
24
- __commit_id__ = commit_id = 'gae98b27f0'
24
+ __commit_id__ = commit_id = 'g6fb0fd96d'
@@ -4,7 +4,7 @@ import parse
4
4
  from .fpath import ExpandedFPath, FPath
5
5
 
6
6
 
7
- def expand_fpath(fpath, *, exclude_path_patterns=None, require_metadata=True):
7
+ def expand(fpath, *, exclude_path_patterns=None, require_metadata=True):
8
8
  """
9
9
  Use an f-string to extract out a collection of paths, where the f-string variables
10
10
  are captured and stored along the path name. This is a convenience function that
@@ -31,9 +31,9 @@ def expand_fpath(fpath, *, exclude_path_patterns=None, require_metadata=True):
31
31
  )
32
32
 
33
33
 
34
- def iexpand_fpath(fpath, *, exclude_path_patterns=None, require_metadata=True, errors="raise"):
34
+ def iexpand(fpath, *, exclude_path_patterns=None, require_metadata=True, errors="raise"):
35
35
  """
36
- Generator equivalent of :func:`.expand_fpath` -- lazily yields each
36
+ Generator equivalent of :func:`.expand` -- lazily yields each
37
37
  matching :obj:`.Path` instead of building the whole
38
38
  :obj:`.ExpandedFPath` up front. This is a convenience function that
39
39
  simply creates an :obj:`FPath` and calls its :meth:`.FPath.iexpand`
@@ -63,58 +63,51 @@ def iexpand_fpath(fpath, *, exclude_path_patterns=None, require_metadata=True, e
63
63
  )
64
64
 
65
65
 
66
- def expand_fpath_decorator(f=None, require_expandable=False, post_process=None):
66
+ def expand_arg(f=None, require_expandable=False):
67
67
  """
68
- Decorator for :func:`.expand_fpath`.
68
+ Decorator for :func:`.expand`. Expands `fpath` (if it has {}
69
+ named captures) before calling the wrapped function with the result;
70
+ otherwise calls it with `fpath` unchanged. Callers that need to react
71
+ differently depending on whether expansion actually happened (e.g. to
72
+ join in captured metadata) should check `isinstance(expanded_fpath,
73
+ ExpandedFPath)` themselves, inside the wrapped function -- this
74
+ decorator doesn't hook into that, it only handles expansion.
69
75
 
70
76
  Parameters
71
77
  ----------
72
78
  f : :obj:`callable`
73
- A function that takes an :obj:`.ExpandedFPath` as its first argument.
79
+ A function that takes an :obj:`.ExpandedFPath` (or, if not
80
+ expandable and `require_expandable` is False, the original `fpath`)
81
+ as its first argument.
74
82
  require_expandable : :obj:`bool`
75
- Whether to require `fpath` to have {} named captures. If False
76
- (the default), a plain literal path or glob is passed straight
77
- through to `f` unexpanded, with `post_process` skipped -- there's
78
- no ExpandedFPath to hand it in that case. (Default: False)
79
- post_process : :obj:`callable`
80
- A function that takes the output of `f` and the :obj:`.ExpandedFPath`.
81
- (Default: None).
83
+ Whether to require `fpath` to have {} named captures, raising
84
+ ValueError otherwise. If False (the default), a plain literal path
85
+ or glob is passed straight through to `f` unexpanded. (Default: False)
82
86
  """
83
87
 
84
88
  def decorator(f):
85
89
  @wraps(f)
86
90
  def wrapper(fpath, *args, **kwargs):
87
- # If fpath is an :obj:`ExpandedFPath`, just call f with it.
91
+ # If fpath is already an :obj:`ExpandedFPath`, just call f with it.
88
92
  if isinstance(fpath, ExpandedFPath):
89
- expanded_fpath = fpath
90
- result = f(fpath, *args, **kwargs)
93
+ return f(fpath, *args, **kwargs)
91
94
 
92
95
  # Expand fpath if it is expandable.
93
- elif is_expandable(fpath):
96
+ if is_expandable(fpath):
94
97
  exclude_path_patterns = kwargs.pop("exclude_path_patterns", None)
95
98
  require_metadata = kwargs.pop("require_metadata", True)
96
- expanded_fpath = expand_fpath(
99
+ expanded_fpath = expand(
97
100
  fpath,
98
101
  exclude_path_patterns=exclude_path_patterns,
99
102
  require_metadata=require_metadata,
100
103
  )
101
- result = f(expanded_fpath, *args, **kwargs)
102
-
103
- # fpath has no {} captures, so there's no ExpandedFPath to
104
- # build -- nothing for post_process (e.g. join_metadata) to
105
- # join metadata from. Call f directly and return immediately,
106
- # skipping post_process entirely, rather than falling through
107
- # to it with no expanded_fpath to give it.
108
- else:
109
- if require_expandable:
110
- msg = f"fpath not expandable: '{fpath}'"
111
- raise ValueError(msg)
112
- return f(fpath, *args, **kwargs)
113
-
114
- if post_process is not None:
115
- result = post_process(result, expanded_fpath)
104
+ return f(expanded_fpath, *args, **kwargs)
116
105
 
117
- return result
106
+ # fpath has no {} captures.
107
+ if require_expandable:
108
+ msg = f"fpath not expandable: '{fpath}'"
109
+ raise ValueError(msg)
110
+ return f(fpath, *args, **kwargs)
118
111
 
119
112
  return wrapper
120
113
 
@@ -1,6 +1,6 @@
1
1
  from functools import wraps
2
2
  import polars as _polars
3
- from fpathlib import expand_fpath_decorator, ExpandedFPath
3
+ from fpathlib import expand_arg, is_expandable, ExpandedFPath
4
4
 
5
5
 
6
6
  def __getattr__(name):
@@ -13,15 +13,59 @@ def __getattr__(name):
13
13
  return getattr(_polars, name)
14
14
 
15
15
 
16
- def join_metadata(df, expanded_fpath):
17
- return df.join(
18
- expanded_fpath.to_polars(lazy=isinstance(df, _polars.LazyFrame)),
19
- on="fname",
20
- )
21
-
16
+ def join_metadata(f):
17
+ """
18
+ Wraps a scan_csv-shaped function `f(source, *args, **kwargs) ->
19
+ LazyFrame` with metadata-joining and "_fname" cleanup, applied
20
+ automatically to whatever `f` returns. Must be applied *inside*
21
+ @expand_arg (i.e. listed closer to `def`), so `source`
22
+ here is always the already-expanded value, not the raw caller-supplied
23
+ pattern:
24
+
25
+ @expand_arg
26
+ @join_metadata
27
+ def scan_csv(source, ...): ...
28
+
29
+ Guards against getting that order backwards: if `source` still looks
30
+ like an unexpanded {} pattern at this point, expansion can only have
31
+ failed to run (wrong decorator order, or this decorator used without
32
+ @expand_arg at all) -- raise immediately rather than
33
+ silently joining no metadata.
34
+ """
22
35
 
23
- @expand_fpath_decorator(post_process=join_metadata)
24
- def read_csv(expanded_fpath, *args, **kwargs):
36
+ @wraps(f)
37
+ def wrapper(source, *args, **kwargs):
38
+ lf = f(source, *args, **kwargs)
39
+ if not isinstance(source, ExpandedFPath) and is_expandable(source):
40
+ msg = (
41
+ f"join_metadata received an unexpanded pattern "
42
+ f"{source!r} -- @expand_arg must be the outer "
43
+ "decorator, applied above (not below) @join_metadata"
44
+ )
45
+ raise RuntimeError(msg)
46
+
47
+ # Join captured {} metadata into `lf`, keyed on the reserved
48
+ # "_fname" column scan_csv/scan_parquet always create, then always
49
+ # drop "_fname" itself. Callers that want to keep the file path
50
+ # under a different name must alias it to that name *before* this
51
+ # runs (inside `f`'s own body) -- by the time this returns,
52
+ # "_fname" is gone either way.
53
+ if isinstance(source, ExpandedFPath):
54
+ # "fname" is ExpandedFPath.to_polars()'s normal, public column
55
+ # name; rename it to the same reserved "_fname" scan_csv/
56
+ # scan_parquet use, so the join key can never collide with an
57
+ # `include_file_paths` name a caller chose (including "fname"
58
+ # itself).
59
+ metadata = source.to_polars(lazy=isinstance(lf, _polars.LazyFrame))
60
+ metadata = metadata.rename({"fname": "_fname"})
61
+ lf = lf.join(metadata, on="_fname")
62
+
63
+ return lf.drop("_fname")
64
+
65
+ return wrapper
66
+
67
+
68
+ def read_csv(source, *args, **kwargs):
25
69
  """
26
70
  Read the paths in the collection as CSV files, and return a
27
71
  :obj:`polars.DataFrame` along with the metadata captured from the path
@@ -29,8 +73,8 @@ def read_csv(expanded_fpath, *args, **kwargs):
29
73
 
30
74
  Parameters
31
75
  ----------
32
- expanded_fpath : :obj:`fpathlib.ExpandedFPath`
33
- An expanded f-string path.
76
+ source : :obj:`fpathlib.ExpandedFPath`, or any type accepted by :obj:`polars.read_csv`
77
+ An expanded f-string path, or a plain path/glob.
34
78
  *args
35
79
  Positional arguments to pass to :meth:`polars.read_csv`.
36
80
  **kwargs
@@ -41,14 +85,14 @@ def read_csv(expanded_fpath, *args, **kwargs):
41
85
  :obj:`polars.DataFrame`
42
86
  """
43
87
 
44
- return scan_csv.__wrapped__(expanded_fpath, *args, **kwargs).collect()
88
+ return scan_csv(source, *args, **kwargs).collect()
45
89
 
46
90
 
47
- @expand_fpath_decorator(post_process=join_metadata)
48
91
  def read_txt(
49
- expanded_fpath,
50
- filter_expr=None,
51
- separator=None,
92
+ source,
93
+ line_filter=None,
94
+ separator=r"\s+",
95
+ strip_initial_spaces=True,
52
96
  new_columns=None,
53
97
  has_header=False,
54
98
  *args,
@@ -63,10 +107,13 @@ def read_txt(
63
107
 
64
108
  Parameters
65
109
  ----------
66
- expanded_fpath : :obj:`fpathlib.ExpandedFPath`
67
- An expanded f-string path.
68
- filter_expr : :obj:`polars.Expr`, optional
69
- Filter the lines before splitting by the separator (if provided).
110
+ source : :obj:`fpathlib.ExpandedFPath`, or any type accepted by :obj:`polars.scan_csv`
111
+ An expanded f-string path, or a plain path/glob.
112
+ line_filter : :obj:`callable`, optional
113
+ A function that takes a :obj:`polars.Expr` for the line's text and
114
+ returns a boolean :obj:`polars.Expr`, used to filter lines before
115
+ splitting by the separator. E.g.
116
+ `lambda line: line.str.starts_with("#").not_()`.
70
117
  separator : :obj:`str`, optional
71
118
  Deliminatorg to split each line into fields.
72
119
  new_columns : :obj:`list`[:obj:`str`], optional
@@ -86,10 +133,11 @@ def read_txt(
86
133
  :obj:`polars.DataFrame`
87
134
  """
88
135
 
89
- return scan_txt.__wrapped__(
90
- expanded_fpath,
91
- filter_expr=filter_expr,
136
+ return scan_txt(
137
+ source,
138
+ line_filter=line_filter,
92
139
  separator=separator,
140
+ strip_initial_spaces=strip_initial_spaces,
93
141
  new_columns=new_columns,
94
142
  has_header=has_header,
95
143
  *args,
@@ -97,8 +145,9 @@ def read_txt(
97
145
  ).collect()
98
146
 
99
147
 
100
- @expand_fpath_decorator(post_process=join_metadata)
101
- def scan_csv(expanded_fpath, *args, **kwargs):
148
+ @expand_arg
149
+ @join_metadata
150
+ def scan_csv(source, include_file_paths=None, *args, **kwargs):
102
151
  """
103
152
  Scan the paths in the collection as CSV files, and return a
104
153
  :obj:`polars.LazyFrame` along with the metadata captured from the path
@@ -106,8 +155,12 @@ def scan_csv(expanded_fpath, *args, **kwargs):
106
155
 
107
156
  Parameters
108
157
  ----------
109
- expanded_fpath : :obj:`fpathlib.ExpandedFPath`
110
- An expanded f-string path.
158
+ source : :obj:`fpathlib.ExpandedFPath`, or any type accepted by :obj:`polars.scan_csv`
159
+ An expanded f-string path, or a plain path/glob.
160
+ include_file_paths : :obj:`str`, optional
161
+ Name to give a column of each row's source file path in the
162
+ output. The file path is always used internally to join captured
163
+ metadata; without this, it isn't kept in the result. (Default: None)
111
164
  *args
112
165
  Positional arguments to pass to :meth:`polars.scan_csv`.
113
166
  **kwargs
@@ -119,17 +172,19 @@ def scan_csv(expanded_fpath, *args, **kwargs):
119
172
  """
120
173
 
121
174
  lf = _polars.scan_csv(
122
- expanded_fpath,
123
- include_file_paths="fname",
175
+ source,
176
+ include_file_paths="_fname",
124
177
  *args,
125
178
  **kwargs,
126
179
  )
127
-
180
+ if include_file_paths is not None:
181
+ lf = lf.with_columns(_polars.col("_fname").alias(include_file_paths))
128
182
  return lf
129
183
 
130
184
 
131
- @expand_fpath_decorator(post_process=join_metadata)
132
- def scan_parquet(expanded_fpath, *args, **kwargs):
185
+ @expand_arg
186
+ @join_metadata
187
+ def scan_parquet(source, include_file_paths=None, *args, **kwargs):
133
188
  """
134
189
  Scan the paths in the collection as a parquet file, and return a
135
190
  :obj:`polars.LazyFrame` along with the metadata captured from the path
@@ -137,8 +192,12 @@ def scan_parquet(expanded_fpath, *args, **kwargs):
137
192
 
138
193
  Parameters
139
194
  ----------
140
- expanded_fpath : :obj:`fpathlib.ExpandedFPath`
141
- An expanded f-string path.
195
+ source : :obj:`fpathlib.ExpandedFPath`, or any type accepted by :obj:`polars.scan_parquet`
196
+ An expanded f-string path, or a plain path/glob.
197
+ include_file_paths : :obj:`str`, optional
198
+ Name to give a column of each row's source file path in the
199
+ output. The file path is always used internally to join captured
200
+ metadata; without this, it isn't kept in the result. (Default: None)
142
201
  *args
143
202
  Positional arguments to pass to :meth:`polars.scan_parquet`.
144
203
  **kwargs
@@ -150,24 +209,26 @@ def scan_parquet(expanded_fpath, *args, **kwargs):
150
209
  """
151
210
 
152
211
  lf = _polars.scan_parquet(
153
- expanded_fpath,
154
- include_file_paths="fname",
212
+ source,
213
+ include_file_paths="_fname",
155
214
  *args,
156
215
  **kwargs,
157
216
  )
158
-
217
+ if include_file_paths is not None:
218
+ lf = lf.with_columns(_polars.col("_fname").alias(include_file_paths))
159
219
  return lf
160
220
 
161
221
 
162
- # TODO rename expanded_fpath as source
163
- @expand_fpath_decorator(post_process=join_metadata)
222
+ @expand_arg
164
223
  def scan_txt(
165
- expanded_fpath,
166
- filter_expr=None,
167
- separator=None,
224
+ source,
225
+ line_filter=None,
226
+ separator=r"\s+",
227
+ strip_initial_spaces=True,
168
228
  new_columns=None,
169
229
  has_header=False,
170
- keep_line=False,
230
+ include_line=None,
231
+ include_file_paths=None,
171
232
  usecols=None,
172
233
  validate_schema=True,
173
234
  *args,
@@ -182,10 +243,13 @@ def scan_txt(
182
243
 
183
244
  Parameters
184
245
  ----------
185
- expanded_fpath : :obj:`fpathlib.ExpandedFPath`
186
- An expanded f-string path.
187
- filter_expr : :obj:`polars.Expr`, optional
188
- Filter the lines before splitting by the separator (if provided).
246
+ source : :obj:`fpathlib.ExpandedFPath`, or any type accepted by :obj:`polars.scan_csv`
247
+ An expanded f-string path, or a plain path/glob.
248
+ line_filter : :obj:`callable`, optional
249
+ A function that takes a :obj:`polars.Expr` for the line's text and
250
+ returns a boolean :obj:`polars.Expr`, used to filter lines before
251
+ splitting by the separator. E.g.
252
+ `lambda line: line.str.starts_with("#").not_()`.
189
253
  separator : :obj:`str`, optional
190
254
  Deliminatorg to split each line into fields.
191
255
  new_columns : :obj:`list`[:obj:`str`], optional
@@ -195,8 +259,13 @@ def scan_txt(
195
259
  Whether the text files have a header line that should be skipped. The header
196
260
  must have the same delimiter as the separator provided in `separator`.
197
261
  (Default: False)
198
- keep_line : :obj:`bool`
199
- Whether to keep the original line as a column in the output.
262
+ include_line : :obj:`str`, optional
263
+ Name to give a column of each row's original, unsplit line text in
264
+ the output. Without this, it isn't kept in the result. (Default: None)
265
+ include_file_paths : :obj:`str`, optional
266
+ Name to give a column of each row's source file path in the
267
+ output. The file path is always used internally to join captured
268
+ metadata; without this, it isn't kept in the result. (Default: None)
200
269
  usecols : :obj:`list`[:obj:`int`], optional
201
270
  Indexes of columns to keep in the output. If not provided, all columns are kept. Only applicable if `separator` is provided.
202
271
  validate_schema : :obj:`bool`
@@ -220,53 +289,68 @@ def scan_txt(
220
289
  :obj:`polars.LazyFrame`
221
290
  """
222
291
 
223
- # TODO there are forbidden variables that should not be in expanded_fpath
224
- # such as 'line' and 'fields' and 'fname'
292
+ # TODO there are forbidden variables that should not be in source
293
+ # such as '_line' and 'fields' and '_fname'
225
294
 
226
295
  # TODO schema and schema_overrides is probably broken
227
296
 
228
- lf = _polars.scan_csv(
229
- expanded_fpath,
230
- include_file_paths="fname",
297
+ # Delegates to scan_csv (rather than calling _polars.scan_csv and
298
+ # doing the metadata-join/include_file_paths handling directly) so
299
+ # that handling -- including the recursive single-file sample call
300
+ # below, since source[0] is a plain Path, not an ExpandedFPath --
301
+ # isn't duplicated here. scan_csv's own @expand_arg is a
302
+ # no-op on `source` at this point: it's already been resolved by
303
+ # scan_txt's decorator, so scan_csv just sees an ExpandedFPath or an
304
+ # already-unexpandable literal/glob, either way with nothing left to
305
+ # expand.
306
+ lf = scan_csv(
307
+ source,
308
+ include_file_paths=include_file_paths,
231
309
  separator="\n",
232
- new_columns=["line"],
310
+ new_columns=["_line"],
233
311
  has_header=False,
234
312
  **kwargs,
235
313
  )
236
314
 
315
+ # Strip leading whitespace from each line if requested.
316
+ if strip_initial_spaces:
317
+ lf = lf.with_columns(_polars.col("_line").str.strip_chars_start())
318
+
237
319
  # Can filter lines before doing any further processing
238
320
  # This could be to remove lines with comments, etc.
239
- if filter_expr is not None:
240
- lf = lf.filter(filter_expr)
321
+ if line_filter is not None:
322
+ lf = lf.filter(line_filter(_polars.col("_line")))
323
+
324
+ if include_line is not None:
325
+ lf = lf.with_columns(_polars.col("_line").alias(include_line))
241
326
 
242
327
  # Separate lines into fields using `separator`
243
328
  if separator is not None:
244
329
  # Separate line into fields by separator
245
330
  lf = lf.with_columns(
246
- _polars.col("line").str.split(separator, literal=False).alias("fields")
331
+ _polars.col("_line").str.split(separator, literal=False).alias("fields")
247
332
  )
248
-
249
- if not keep_line:
250
- lf = lf.drop("line")
333
+ lf = lf.drop("_line")
251
334
 
252
335
  # With many matched files, inferring the field count/dtypes directly
253
336
  # against the full glob is extremely slow (every file has to be opened
254
337
  # before a `.head()` takes effect). Instead, recurse on a single
255
- # representative file -- expanded_fpath[0] -- and reuse its already-cheap
338
+ # representative file -- source[0] -- and reuse its already-cheap
256
339
  # (single-file) inference below instead of duplicating it here. Computed
257
340
  # once here since it's needed by both the field-count branch below and
258
341
  # the dtype-inference branch further down.
259
342
  infer_schema = kwargs.get("infer_schema", True)
260
343
  sample_schema = None
261
344
  if (
262
- isinstance(expanded_fpath, ExpandedFPath)
263
- and len(expanded_fpath) > 1
345
+ isinstance(source, ExpandedFPath)
346
+ and len(source) > 1
264
347
  and (usecols is None or infer_schema)
265
348
  ):
266
349
  sample_schema = scan_txt(
267
- expanded_fpath[0],
268
- filter_expr=filter_expr,
350
+ source[0],
351
+ line_filter=line_filter,
269
352
  separator=separator,
353
+ strip_initial_spaces=strip_initial_spaces,
270
354
  new_columns=new_columns,
271
355
  has_header=has_header,
272
356
  usecols=usecols,
@@ -282,7 +366,10 @@ def scan_txt(
282
366
 
283
367
  else:
284
368
  if sample_schema is not None:
285
- n_fields = len(sample_schema) - 1 # minus 'fname'
369
+ # The recursive sample call above is always non-expandable
370
+ # (source[0] is a plain Path), so it already drops "_fname"
371
+ # itself -- no adjustment needed here.
372
+ n_fields = len(sample_schema)
286
373
  else:
287
374
  n_fields = (
288
375
  lf.head(1)
@@ -339,11 +426,9 @@ def scan_txt(
339
426
  # Infer dtypes?
340
427
  if infer_schema:
341
428
  if sample_schema is not None:
342
- inferred_schema = {
343
- name: dtype
344
- for name, dtype in sample_schema.items()
345
- if name != "fname"
346
- }
429
+ # The recursive sample call already drops "_fname" (see
430
+ # above), so no need to filter it back out here.
431
+ inferred_schema = dict(sample_schema.items())
347
432
  else:
348
433
  sample = (
349
434
  lf.head(kwargs.get("infer_schema_length", 100))
@@ -354,4 +439,17 @@ def scan_txt(
354
439
  inferred_schema = _polars.read_csv(sample).schema
355
440
  lf = lf.cast(inferred_schema)
356
441
 
442
+ elif include_line is None:
443
+ # No separator means each row is just the raw line -- that's the
444
+ # whole point of this mode, so it has to stay visible somehow.
445
+ # Fall back to the traditional "line" name rather than leaking the
446
+ # internal "_line" name, since the caller didn't ask for a specific
447
+ # one via include_line.
448
+ lf = lf.rename({"_line": "line"})
449
+ else:
450
+ # include_line was given, so the alias was already created above
451
+ # (before this if/elif/else) -- drop the internal "_line" itself so
452
+ # it doesn't also leak into the output alongside it.
453
+ lf = lf.drop("_line")
454
+
357
455
  return lf
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: fpathlib
3
- Version: 0.1.5
3
+ Version: 0.1.7
4
4
  Summary: A package to combine paths with metadata
5
5
  Author-email: "C. Lockhart" <clockha2@gmu.edu>
6
6
  Requires-Python: >=3.12
@@ -25,10 +25,10 @@ does this with an f-string-like path pattern -- `FPath` -- whose
25
25
  `{variable}` fields are captured out of every matching path on disk.
26
26
 
27
27
  ```python
28
- from fpathlib import expand_fpath
28
+ from fpathlib import expand
29
29
 
30
30
  # Given files like data/tr1/output/0/job2.log, data/tr2/output/1/job0.log, ...
31
- expanded = expand_fpath("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
31
+ expanded = expand("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
32
32
 
33
33
  for path in expanded:
34
34
  print(path, path.metadata)
@@ -0,0 +1,8 @@
1
+ {
2
+ "tag": "0.1.7",
3
+ "distance": 0,
4
+ "node": "g6fb0fd96d4a8743d6b3c15b956553e52d9bd391f",
5
+ "dirty": false,
6
+ "branch": "main",
7
+ "node_date": "2026-09-24"
8
+ }
@@ -1,8 +0,0 @@
1
- {
2
- "tag": "0.1.5",
3
- "distance": 0,
4
- "node": "gae98b27f09095479f5ae8272b52f73655b3c8a4b",
5
- "dirty": false,
6
- "branch": "main",
7
- "node_date": "2026-09-17"
8
- }
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes