fpathlib 0.1.4.dev16__tar.gz → 0.1.6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. {fpathlib-0.1.4.dev16/src/fpathlib.egg-info → fpathlib-0.1.6}/PKG-INFO +3 -3
  2. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/README.md +2 -2
  3. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/conda-recipe/meta.yaml +2 -2
  4. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/docs/source/api.rst +4 -2
  5. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/docs/source/index.rst +2 -2
  6. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/scripts/pypi.sh +8 -1
  7. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib/__init__.py +6 -6
  8. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib/_version.py +3 -3
  9. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib/expand.py +30 -32
  10. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib/ext/polars.py +160 -70
  11. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6/src/fpathlib.egg-info}/PKG-INFO +3 -3
  12. fpathlib-0.1.6/src/fpathlib.egg-info/scm_version.json +8 -0
  13. fpathlib-0.1.4.dev16/src/fpathlib.egg-info/scm_version.json +0 -8
  14. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/.gitignore +0 -0
  15. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/LICENSE +0 -0
  16. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/MANIFEST.in +0 -0
  17. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/TODO.txt +0 -0
  18. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/docs/Makefile +0 -0
  19. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/docs/make.bat +0 -0
  20. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/docs/source/conf.py +0 -0
  21. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/pyproject.toml +0 -0
  22. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/scripts/conda.sh +0 -0
  23. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/scripts/deploy.sh +0 -0
  24. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/scripts/docs.sh +0 -0
  25. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/scripts/tag.sh +0 -0
  26. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/setup.cfg +0 -0
  27. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib/ext/__init__.py +0 -0
  28. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib/fpath.py +0 -0
  29. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib/path.py +0 -0
  30. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib/utils.py +0 -0
  31. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib.egg-info/SOURCES.txt +0 -0
  32. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib.egg-info/dependency_links.txt +0 -0
  33. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib.egg-info/requires.txt +0 -0
  34. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib.egg-info/scm_file_list.json +0 -0
  35. {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: fpathlib
3
- Version: 0.1.4.dev16
3
+ Version: 0.1.6
4
4
  Summary: A package to combine paths with metadata
5
5
  Author-email: "C. Lockhart" <clockha2@gmu.edu>
6
6
  Requires-Python: >=3.12
@@ -25,10 +25,10 @@ does this with an f-string-like path pattern -- `FPath` -- whose
25
25
  `{variable}` fields are captured out of every matching path on disk.
26
26
 
27
27
  ```python
28
- from fpathlib import expand_fpath
28
+ from fpathlib import expand
29
29
 
30
30
  # Given files like data/tr1/output/0/job2.log, data/tr2/output/1/job0.log, ...
31
- expanded = expand_fpath("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
31
+ expanded = expand("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
32
32
 
33
33
  for path in expanded:
34
34
  print(path, path.metadata)
@@ -8,10 +8,10 @@ does this with an f-string-like path pattern -- `FPath` -- whose
8
8
  `{variable}` fields are captured out of every matching path on disk.
9
9
 
10
10
  ```python
11
- from fpathlib import expand_fpath
11
+ from fpathlib import expand
12
12
 
13
13
  # Given files like data/tr1/output/0/job2.log, data/tr2/output/1/job0.log, ...
14
- expanded = expand_fpath("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
14
+ expanded = expand("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
15
15
 
16
16
  for path in expanded:
17
17
  print(path, path.metadata)
@@ -1,5 +1,5 @@
1
1
  {% set name = "fpathlib" %}
2
- {% set version = "0.1.3" %}
2
+ {% set version = "0.1.5" %}
3
3
 
4
4
  package:
5
5
  name: {{ name|lower }}
@@ -7,7 +7,7 @@ package:
7
7
 
8
8
  source:
9
9
  url: https://pypi.io/packages/source/{{ name[0] }}/{{ name }}/fpathlib-{{ version }}.tar.gz
10
- sha256: 059dbf937c6e965ef33f15aa214b2f354550d5dfa26b9030357954cce8c51874
10
+ sha256: 083299d5a41577484af5a8447d185a6ab1c190b2dfe31690375349c163ea61ab
11
11
 
12
12
  build:
13
13
  noarch: python
@@ -23,9 +23,11 @@ FPath and ExpandedFPath
23
23
  Expanding paths
24
24
  ----------------
25
25
 
26
- .. autofunction:: fpathlib.expand_fpath
26
+ .. autofunction:: fpathlib.expand
27
27
 
28
- .. autofunction:: fpathlib.expand_fpath_decorator
28
+ .. autofunction:: fpathlib.iexpand
29
+
30
+ .. autofunction:: fpathlib.expand_arg
29
31
 
30
32
  .. autofunction:: fpathlib.is_expandable
31
33
 
@@ -11,10 +11,10 @@ own filename.
11
11
 
12
12
  .. code-block:: python
13
13
 
14
- from fpathlib import expand_fpath
14
+ from fpathlib import expand
15
15
 
16
16
  # Given files like data/tr1/output/0/job2.log, data/tr2/output/1/job0.log, ...
17
- expanded = expand_fpath("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
17
+ expanded = expand("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
18
18
 
19
19
  for path in expanded:
20
20
  print(path, path.metadata)
@@ -21,7 +21,14 @@ python3 -m build
21
21
  # 0.1.3/0.1.2 tag-collision incident slipped past a git-describe check even
22
22
  # though the build itself came out as a .devN), inspect what actually got
23
23
  # built. This is the real signal of whether the release is clean.
24
- dev_artifacts=$(ls dist/ | grep -c '\.dev[0-9]')
24
+ #
25
+ # `grep -c` exits 1 (not just prints "0") when it finds zero matches --
26
+ # with `set -e` active (deploy.sh sets it, and this script is sourced into
27
+ # that same shell), that silently killed the whole deploy right here for
28
+ # every *clean* release build (the exact case with 0 .dev files), before
29
+ # ever reaching twine upload. `|| true` keeps the count without letting
30
+ # grep's "no matches" exit status abort the script.
31
+ dev_artifacts=$(ls dist/ | grep -c '\.dev[0-9]' || true)
25
32
  if [ $allow_dev -eq 0 ] && [ "$dev_artifacts" != "0" ]
26
33
  then
27
34
  echo "built version is a dev version (tag doesn't point at a clean, distinct commit -- check 'git describe --tags --long' and 'git status'), not uploading to pypi"
@@ -1,9 +1,9 @@
1
1
  from fpathlib.path import Path
2
2
  from fpathlib.fpath import FPath, ExpandedFPath
3
3
  from fpathlib.expand import (
4
- expand_fpath,
5
- iexpand_fpath,
6
- expand_fpath_decorator,
4
+ expand,
5
+ iexpand,
6
+ expand_arg,
7
7
  is_expandable,
8
8
  )
9
9
 
@@ -11,8 +11,8 @@ __all__ = [
11
11
  "Path",
12
12
  "FPath",
13
13
  "ExpandedFPath",
14
- "expand_fpath",
15
- "iexpand_fpath",
16
- "expand_fpath_decorator",
14
+ "expand",
15
+ "iexpand",
16
+ "expand_arg",
17
17
  "is_expandable",
18
18
  ]
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '0.1.4.dev16'
22
- __version_tuple__ = version_tuple = (0, 1, 4, 'dev16')
21
+ __version__ = version = '0.1.6'
22
+ __version_tuple__ = version_tuple = (0, 1, 6)
23
23
 
24
- __commit_id__ = commit_id = 'g851c5dad8'
24
+ __commit_id__ = commit_id = 'g5162f8cd0'
@@ -4,7 +4,7 @@ import parse
4
4
  from .fpath import ExpandedFPath, FPath
5
5
 
6
6
 
7
- def expand_fpath(fpath, *, exclude_path_patterns=None, require_metadata=True):
7
+ def expand(fpath, *, exclude_path_patterns=None, require_metadata=True):
8
8
  """
9
9
  Use an f-string to extract out a collection of paths, where the f-string variables
10
10
  are captured and stored along the path name. This is a convenience function that
@@ -31,9 +31,9 @@ def expand_fpath(fpath, *, exclude_path_patterns=None, require_metadata=True):
31
31
  )
32
32
 
33
33
 
34
- def iexpand_fpath(fpath, *, exclude_path_patterns=None, require_metadata=True, errors="raise"):
34
+ def iexpand(fpath, *, exclude_path_patterns=None, require_metadata=True, errors="raise"):
35
35
  """
36
- Generator equivalent of :func:`.expand_fpath` -- lazily yields each
36
+ Generator equivalent of :func:`.expand` -- lazily yields each
37
37
  matching :obj:`.Path` instead of building the whole
38
38
  :obj:`.ExpandedFPath` up front. This is a convenience function that
39
39
  simply creates an :obj:`FPath` and calls its :meth:`.FPath.iexpand`
@@ -63,53 +63,51 @@ def iexpand_fpath(fpath, *, exclude_path_patterns=None, require_metadata=True, e
63
63
  )
64
64
 
65
65
 
66
- def expand_fpath_decorator(f=None, require_expandable=True, post_process=None):
66
+ def expand_arg(f=None, require_expandable=False):
67
67
  """
68
- Decorator for :func:`.expand_fpath`.
68
+ Decorator for :func:`.expand`. Expands `fpath` (if it has {}
69
+ named captures) before calling the wrapped function with the result;
70
+ otherwise calls it with `fpath` unchanged. Callers that need to react
71
+ differently depending on whether expansion actually happened (e.g. to
72
+ join in captured metadata) should check `isinstance(expanded_fpath,
73
+ ExpandedFPath)` themselves, inside the wrapped function -- this
74
+ decorator doesn't hook into that, it only handles expansion.
69
75
 
70
76
  Parameters
71
77
  ----------
72
78
  f : :obj:`callable`
73
- A function that takes an :obj:`.ExpandedFPath` as its first argument.
74
- post_process : :obj:`callable`
75
- A function that takes the output of `f` and the :obj:`.ExpandedFPath`.
76
- (Default: None).
79
+ A function that takes an :obj:`.ExpandedFPath` (or, if not
80
+ expandable and `require_expandable` is False, the original `fpath`)
81
+ as its first argument.
82
+ require_expandable : :obj:`bool`
83
+ Whether to require `fpath` to have {} named captures, raising
84
+ ValueError otherwise. If False (the default), a plain literal path
85
+ or glob is passed straight through to `f` unexpanded. (Default: False)
77
86
  """
78
87
 
79
88
  def decorator(f):
80
89
  @wraps(f)
81
90
  def wrapper(fpath, *args, **kwargs):
82
- # If fpath is an :obj:`ExpandedFPath`, just call f with it.
91
+ # If fpath is already an :obj:`ExpandedFPath`, just call f with it.
83
92
  if isinstance(fpath, ExpandedFPath):
84
- expanded_fpath = fpath
85
- result = f(fpath, *args, **kwargs)
86
-
87
- # Otherwise, expand fpath and call f with the result.
88
- else:
89
- # What happens if fpath is not expandable?
90
- # If require_expandable is True, raise an error.
91
- # Otherwise, just call f with the original fpath.
92
- if not is_expandable(fpath):
93
- if require_expandable:
94
- msg = f"fpath not expandable: '{fpath}'"
95
- raise ValueError(msg)
96
- return f(fpath, *args, **kwargs)
97
-
98
- # We know fpath is expandable, so we can expand it and call f
93
+ return f(fpath, *args, **kwargs)
94
+
95
+ # Expand fpath if it is expandable.
96
+ if is_expandable(fpath):
99
97
  exclude_path_patterns = kwargs.pop("exclude_path_patterns", None)
100
98
  require_metadata = kwargs.pop("require_metadata", True)
101
- expanded_fpath = expand_fpath(
99
+ expanded_fpath = expand(
102
100
  fpath,
103
101
  exclude_path_patterns=exclude_path_patterns,
104
102
  require_metadata=require_metadata,
105
103
  )
106
- result = f(expanded_fpath, *args, **kwargs)
107
-
108
- # If post_process is provided, call it with the result and the expanded_fpath.
109
- if post_process is not None:
110
- result = post_process(result, expanded_fpath)
104
+ return f(expanded_fpath, *args, **kwargs)
111
105
 
112
- return result
106
+ # fpath has no {} captures.
107
+ if require_expandable:
108
+ msg = f"fpath not expandable: '{fpath}'"
109
+ raise ValueError(msg)
110
+ return f(fpath, *args, **kwargs)
113
111
 
114
112
  return wrapper
115
113
 
@@ -1,6 +1,6 @@
1
1
  from functools import wraps
2
2
  import polars as _polars
3
- from fpathlib import expand_fpath_decorator, ExpandedFPath
3
+ from fpathlib import expand_arg, is_expandable, ExpandedFPath
4
4
 
5
5
 
6
6
  def __getattr__(name):
@@ -13,15 +13,59 @@ def __getattr__(name):
13
13
  return getattr(_polars, name)
14
14
 
15
15
 
16
- def join_metadata(df, expanded_fpath):
17
- return df.join(
18
- expanded_fpath.to_polars(lazy=isinstance(df, _polars.LazyFrame)),
19
- on="fname",
20
- )
21
-
16
+ def join_metadata(f):
17
+ """
18
+ Wraps a scan_csv-shaped function `f(source, *args, **kwargs) ->
19
+ LazyFrame` with metadata-joining and "_fname" cleanup, applied
20
+ automatically to whatever `f` returns. Must be applied *inside*
21
+ @expand_arg (i.e. listed closer to `def`), so `source`
22
+ here is always the already-expanded value, not the raw caller-supplied
23
+ pattern:
24
+
25
+ @expand_arg
26
+ @join_metadata
27
+ def scan_csv(source, ...): ...
28
+
29
+ Guards against getting that order backwards: if `source` still looks
30
+ like an unexpanded {} pattern at this point, expansion can only have
31
+ failed to run (wrong decorator order, or this decorator used without
32
+ @expand_arg at all) -- raise immediately rather than
33
+ silently joining no metadata.
34
+ """
22
35
 
23
- @expand_fpath_decorator(post_process=join_metadata)
24
- def read_csv(expanded_fpath, *args, **kwargs):
36
+ @wraps(f)
37
+ def wrapper(source, *args, **kwargs):
38
+ lf = f(source, *args, **kwargs)
39
+ if not isinstance(source, ExpandedFPath) and is_expandable(source):
40
+ msg = (
41
+ f"join_metadata received an unexpanded pattern "
42
+ f"{source!r} -- @expand_arg must be the outer "
43
+ "decorator, applied above (not below) @join_metadata"
44
+ )
45
+ raise RuntimeError(msg)
46
+
47
+ # Join captured {} metadata into `lf`, keyed on the reserved
48
+ # "_fname" column scan_csv/scan_parquet always create, then always
49
+ # drop "_fname" itself. Callers that want to keep the file path
50
+ # under a different name must alias it to that name *before* this
51
+ # runs (inside `f`'s own body) -- by the time this returns,
52
+ # "_fname" is gone either way.
53
+ if isinstance(source, ExpandedFPath):
54
+ # "fname" is ExpandedFPath.to_polars()'s normal, public column
55
+ # name; rename it to the same reserved "_fname" scan_csv/
56
+ # scan_parquet use, so the join key can never collide with an
57
+ # `include_file_paths` name a caller chose (including "fname"
58
+ # itself).
59
+ metadata = source.to_polars(lazy=isinstance(lf, _polars.LazyFrame))
60
+ metadata = metadata.rename({"fname": "_fname"})
61
+ lf = lf.join(metadata, on="_fname")
62
+
63
+ return lf.drop("_fname")
64
+
65
+ return wrapper
66
+
67
+
68
+ def read_csv(source, *args, **kwargs):
25
69
  """
26
70
  Read the paths in the collection as CSV files, and return a
27
71
  :obj:`polars.DataFrame` along with the metadata captured from the path
@@ -29,8 +73,8 @@ def read_csv(expanded_fpath, *args, **kwargs):
29
73
 
30
74
  Parameters
31
75
  ----------
32
- expanded_fpath : :obj:`fpathlib.ExpandedFPath`
33
- An expanded f-string path.
76
+ source : :obj:`fpathlib.ExpandedFPath`, or any type accepted by :obj:`polars.read_csv`
77
+ An expanded f-string path, or a plain path/glob.
34
78
  *args
35
79
  Positional arguments to pass to :meth:`polars.read_csv`.
36
80
  **kwargs
@@ -41,13 +85,12 @@ def read_csv(expanded_fpath, *args, **kwargs):
41
85
  :obj:`polars.DataFrame`
42
86
  """
43
87
 
44
- return scan_csv.__wrapped__(expanded_fpath, *args, **kwargs).collect()
88
+ return scan_csv(source, *args, **kwargs).collect()
45
89
 
46
90
 
47
- @expand_fpath_decorator(post_process=join_metadata)
48
91
  def read_txt(
49
- expanded_fpath,
50
- filter_expr=None,
92
+ source,
93
+ line_filter=None,
51
94
  separator=None,
52
95
  new_columns=None,
53
96
  has_header=False,
@@ -63,10 +106,13 @@ def read_txt(
63
106
 
64
107
  Parameters
65
108
  ----------
66
- expanded_fpath : :obj:`fpathlib.ExpandedFPath`
67
- An expanded f-string path.
68
- filter_expr : :obj:`polars.Expr`, optional
69
- Filter the lines before splitting by the separator (if provided).
109
+ source : :obj:`fpathlib.ExpandedFPath`, or any type accepted by :obj:`polars.scan_csv`
110
+ An expanded f-string path, or a plain path/glob.
111
+ line_filter : :obj:`callable`, optional
112
+ A function that takes a :obj:`polars.Expr` for the line's text and
113
+ returns a boolean :obj:`polars.Expr`, used to filter lines before
114
+ splitting by the separator. E.g.
115
+ `lambda line: line.str.starts_with("#").not_()`.
70
116
  separator : :obj:`str`, optional
71
117
  Deliminatorg to split each line into fields.
72
118
  new_columns : :obj:`list`[:obj:`str`], optional
@@ -86,9 +132,9 @@ def read_txt(
86
132
  :obj:`polars.DataFrame`
87
133
  """
88
134
 
89
- return scan_txt.__wrapped__(
90
- expanded_fpath,
91
- filter_expr=filter_expr,
135
+ return scan_txt(
136
+ source,
137
+ line_filter=line_filter,
92
138
  separator=separator,
93
139
  new_columns=new_columns,
94
140
  has_header=has_header,
@@ -97,8 +143,9 @@ def read_txt(
97
143
  ).collect()
98
144
 
99
145
 
100
- @expand_fpath_decorator(post_process=join_metadata)
101
- def scan_csv(expanded_fpath, *args, **kwargs):
146
+ @expand_arg
147
+ @join_metadata
148
+ def scan_csv(source, include_file_paths=None, *args, **kwargs):
102
149
  """
103
150
  Scan the paths in the collection as CSV files, and return a
104
151
  :obj:`polars.LazyFrame` along with the metadata captured from the path
@@ -106,8 +153,12 @@ def scan_csv(expanded_fpath, *args, **kwargs):
106
153
 
107
154
  Parameters
108
155
  ----------
109
- expanded_fpath : :obj:`fpathlib.ExpandedFPath`
110
- An expanded f-string path.
156
+ source : :obj:`fpathlib.ExpandedFPath`, or any type accepted by :obj:`polars.scan_csv`
157
+ An expanded f-string path, or a plain path/glob.
158
+ include_file_paths : :obj:`str`, optional
159
+ Name to give a column of each row's source file path in the
160
+ output. The file path is always used internally to join captured
161
+ metadata; without this, it isn't kept in the result. (Default: None)
111
162
  *args
112
163
  Positional arguments to pass to :meth:`polars.scan_csv`.
113
164
  **kwargs
@@ -119,17 +170,19 @@ def scan_csv(expanded_fpath, *args, **kwargs):
119
170
  """
120
171
 
121
172
  lf = _polars.scan_csv(
122
- expanded_fpath,
123
- include_file_paths="fname",
173
+ source,
174
+ include_file_paths="_fname",
124
175
  *args,
125
176
  **kwargs,
126
177
  )
127
-
178
+ if include_file_paths is not None:
179
+ lf = lf.with_columns(_polars.col("_fname").alias(include_file_paths))
128
180
  return lf
129
181
 
130
182
 
131
- @expand_fpath_decorator(post_process=join_metadata)
132
- def scan_parquet(expanded_fpath, *args, **kwargs):
183
+ @expand_arg
184
+ @join_metadata
185
+ def scan_parquet(source, include_file_paths=None, *args, **kwargs):
133
186
  """
134
187
  Scan the paths in the collection as a parquet file, and return a
135
188
  :obj:`polars.LazyFrame` along with the metadata captured from the path
@@ -137,8 +190,12 @@ def scan_parquet(expanded_fpath, *args, **kwargs):
137
190
 
138
191
  Parameters
139
192
  ----------
140
- expanded_fpath : :obj:`fpathlib.ExpandedFPath`
141
- An expanded f-string path.
193
+ source : :obj:`fpathlib.ExpandedFPath`, or any type accepted by :obj:`polars.scan_parquet`
194
+ An expanded f-string path, or a plain path/glob.
195
+ include_file_paths : :obj:`str`, optional
196
+ Name to give a column of each row's source file path in the
197
+ output. The file path is always used internally to join captured
198
+ metadata; without this, it isn't kept in the result. (Default: None)
142
199
  *args
143
200
  Positional arguments to pass to :meth:`polars.scan_parquet`.
144
201
  **kwargs
@@ -150,24 +207,25 @@ def scan_parquet(expanded_fpath, *args, **kwargs):
150
207
  """
151
208
 
152
209
  lf = _polars.scan_parquet(
153
- expanded_fpath,
154
- include_file_paths="fname",
210
+ source,
211
+ include_file_paths="_fname",
155
212
  *args,
156
213
  **kwargs,
157
214
  )
158
-
215
+ if include_file_paths is not None:
216
+ lf = lf.with_columns(_polars.col("_fname").alias(include_file_paths))
159
217
  return lf
160
218
 
161
219
 
162
- # TODO rename expanded_fpath as source
163
- @expand_fpath_decorator(require_expandable=False, post_process=join_metadata)
220
+ @expand_arg
164
221
  def scan_txt(
165
- expanded_fpath,
166
- filter_expr=None,
222
+ source,
223
+ line_filter=None,
167
224
  separator=None,
168
225
  new_columns=None,
169
226
  has_header=False,
170
- keep_line=False,
227
+ include_line=None,
228
+ include_file_paths=None,
171
229
  usecols=None,
172
230
  validate_schema=True,
173
231
  *args,
@@ -182,10 +240,13 @@ def scan_txt(
182
240
 
183
241
  Parameters
184
242
  ----------
185
- expanded_fpath : :obj:`fpathlib.ExpandedFPath`
186
- An expanded f-string path.
187
- filter_expr : :obj:`polars.Expr`, optional
188
- Filter the lines before splitting by the separator (if provided).
243
+ source : :obj:`fpathlib.ExpandedFPath`, or any type accepted by :obj:`polars.scan_csv`
244
+ An expanded f-string path, or a plain path/glob.
245
+ line_filter : :obj:`callable`, optional
246
+ A function that takes a :obj:`polars.Expr` for the line's text and
247
+ returns a boolean :obj:`polars.Expr`, used to filter lines before
248
+ splitting by the separator. E.g.
249
+ `lambda line: line.str.starts_with("#").not_()`.
189
250
  separator : :obj:`str`, optional
190
251
  Deliminatorg to split each line into fields.
191
252
  new_columns : :obj:`list`[:obj:`str`], optional
@@ -195,8 +256,13 @@ def scan_txt(
195
256
  Whether the text files have a header line that should be skipped. The header
196
257
  must have the same delimiter as the separator provided in `separator`.
197
258
  (Default: False)
198
- keep_line : :obj:`bool`
199
- Whether to keep the original line as a column in the output.
259
+ include_line : :obj:`str`, optional
260
+ Name to give a column of each row's original, unsplit line text in
261
+ the output. Without this, it isn't kept in the result. (Default: None)
262
+ include_file_paths : :obj:`str`, optional
263
+ Name to give a column of each row's source file path in the
264
+ output. The file path is always used internally to join captured
265
+ metadata; without this, it isn't kept in the result. (Default: None)
200
266
  usecols : :obj:`list`[:obj:`int`], optional
201
267
  Indexes of columns to keep in the output. If not provided, all columns are kept. Only applicable if `separator` is provided.
202
268
  validate_schema : :obj:`bool`
@@ -220,52 +286,62 @@ def scan_txt(
220
286
  :obj:`polars.LazyFrame`
221
287
  """
222
288
 
223
- # TODO there are forbidden variables that should not be in expanded_fpath
224
- # such as 'line' and 'fields' and 'fname'
289
+ # TODO there are forbidden variables that should not be in source
290
+ # such as '_line' and 'fields' and '_fname'
225
291
 
226
292
  # TODO schema and schema_overrides is probably broken
227
293
 
228
- lf = _polars.scan_csv(
229
- expanded_fpath,
230
- include_file_paths="fname",
294
+ # Delegates to scan_csv (rather than calling _polars.scan_csv and
295
+ # doing the metadata-join/include_file_paths handling directly) so
296
+ # that handling -- including the recursive single-file sample call
297
+ # below, since source[0] is a plain Path, not an ExpandedFPath --
298
+ # isn't duplicated here. scan_csv's own @expand_arg is a
299
+ # no-op on `source` at this point: it's already been resolved by
300
+ # scan_txt's decorator, so scan_csv just sees an ExpandedFPath or an
301
+ # already-unexpandable literal/glob, either way with nothing left to
302
+ # expand.
303
+ lf = scan_csv(
304
+ source,
305
+ include_file_paths=include_file_paths,
231
306
  separator="\n",
232
- new_columns=["line"],
307
+ new_columns=["_line"],
233
308
  has_header=False,
234
309
  **kwargs,
235
310
  )
236
311
 
237
312
  # Can filter lines before doing any further processing
238
313
  # This could be to remove lines with comments, etc.
239
- if filter_expr is not None:
240
- lf = lf.filter(filter_expr)
314
+ if line_filter is not None:
315
+ lf = lf.filter(line_filter(_polars.col("_line")))
316
+
317
+ if include_line is not None:
318
+ lf = lf.with_columns(_polars.col("_line").alias(include_line))
241
319
 
242
320
  # Separate lines into fields using `separator`
243
321
  if separator is not None:
244
322
  # Separate line into fields by separator
245
323
  lf = lf.with_columns(
246
- _polars.col("line").str.split(separator, literal=False).alias("fields")
324
+ _polars.col("_line").str.split(separator, literal=False).alias("fields")
247
325
  )
248
-
249
- if not keep_line:
250
- lf = lf.drop("line")
326
+ lf = lf.drop("_line")
251
327
 
252
328
  # With many matched files, inferring the field count/dtypes directly
253
329
  # against the full glob is extremely slow (every file has to be opened
254
330
  # before a `.head()` takes effect). Instead, recurse on a single
255
- # representative file -- expanded_fpath[0] -- and reuse its already-cheap
331
+ # representative file -- source[0] -- and reuse its already-cheap
256
332
  # (single-file) inference below instead of duplicating it here. Computed
257
333
  # once here since it's needed by both the field-count branch below and
258
334
  # the dtype-inference branch further down.
259
335
  infer_schema = kwargs.get("infer_schema", True)
260
336
  sample_schema = None
261
337
  if (
262
- isinstance(expanded_fpath, ExpandedFPath)
263
- and len(expanded_fpath) > 1
338
+ isinstance(source, ExpandedFPath)
339
+ and len(source) > 1
264
340
  and (usecols is None or infer_schema)
265
341
  ):
266
342
  sample_schema = scan_txt(
267
- expanded_fpath[0],
268
- filter_expr=filter_expr,
343
+ source[0],
344
+ line_filter=line_filter,
269
345
  separator=separator,
270
346
  new_columns=new_columns,
271
347
  has_header=has_header,
@@ -282,7 +358,10 @@ def scan_txt(
282
358
 
283
359
  else:
284
360
  if sample_schema is not None:
285
- n_fields = len(sample_schema) - 1 # minus 'fname'
361
+ # The recursive sample call above is always non-expandable
362
+ # (source[0] is a plain Path), so it already drops "_fname"
363
+ # itself -- no adjustment needed here.
364
+ n_fields = len(sample_schema)
286
365
  else:
287
366
  n_fields = (
288
367
  lf.head(1)
@@ -339,11 +418,9 @@ def scan_txt(
339
418
  # Infer dtypes?
340
419
  if infer_schema:
341
420
  if sample_schema is not None:
342
- inferred_schema = {
343
- name: dtype
344
- for name, dtype in sample_schema.items()
345
- if name != "fname"
346
- }
421
+ # The recursive sample call already drops "_fname" (see
422
+ # above), so no need to filter it back out here.
423
+ inferred_schema = dict(sample_schema.items())
347
424
  else:
348
425
  sample = (
349
426
  lf.head(kwargs.get("infer_schema_length", 100))
@@ -354,4 +431,17 @@ def scan_txt(
354
431
  inferred_schema = _polars.read_csv(sample).schema
355
432
  lf = lf.cast(inferred_schema)
356
433
 
434
+ elif include_line is None:
435
+ # No separator means each row is just the raw line -- that's the
436
+ # whole point of this mode, so it has to stay visible somehow.
437
+ # Fall back to the traditional "line" name rather than leaking the
438
+ # internal "_line" name, since the caller didn't ask for a specific
439
+ # one via include_line.
440
+ lf = lf.rename({"_line": "line"})
441
+ else:
442
+ # include_line was given, so the alias was already created above
443
+ # (before this if/elif/else) -- drop the internal "_line" itself so
444
+ # it doesn't also leak into the output alongside it.
445
+ lf = lf.drop("_line")
446
+
357
447
  return lf
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: fpathlib
3
- Version: 0.1.4.dev16
3
+ Version: 0.1.6
4
4
  Summary: A package to combine paths with metadata
5
5
  Author-email: "C. Lockhart" <clockha2@gmu.edu>
6
6
  Requires-Python: >=3.12
@@ -25,10 +25,10 @@ does this with an f-string-like path pattern -- `FPath` -- whose
25
25
  `{variable}` fields are captured out of every matching path on disk.
26
26
 
27
27
  ```python
28
- from fpathlib import expand_fpath
28
+ from fpathlib import expand
29
29
 
30
30
  # Given files like data/tr1/output/0/job2.log, data/tr2/output/1/job0.log, ...
31
- expanded = expand_fpath("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
31
+ expanded = expand("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
32
32
 
33
33
  for path in expanded:
34
34
  print(path, path.metadata)
@@ -0,0 +1,8 @@
1
+ {
2
+ "tag": "0.1.6",
3
+ "distance": 0,
4
+ "node": "g5162f8cd0811db2b8291cbf05ff5ad278451dc81",
5
+ "dirty": false,
6
+ "branch": "main",
7
+ "node_date": "2026-09-21"
8
+ }
@@ -1,8 +0,0 @@
1
- {
2
- "tag": "0.1.3",
3
- "distance": 16,
4
- "node": "g851c5dad81b573754c9c1d7508faca0f7b1ab986",
5
- "dirty": false,
6
- "branch": "main",
7
- "node_date": "2026-09-17"
8
- }
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes