fpathlib 0.1.5__tar.gz → 0.1.7__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fpathlib-0.1.5/src/fpathlib.egg-info → fpathlib-0.1.7}/PKG-INFO +3 -3
- {fpathlib-0.1.5 → fpathlib-0.1.7}/README.md +2 -2
- {fpathlib-0.1.5 → fpathlib-0.1.7}/conda-recipe/meta.yaml +2 -2
- {fpathlib-0.1.5 → fpathlib-0.1.7}/docs/source/api.rst +4 -2
- {fpathlib-0.1.5 → fpathlib-0.1.7}/docs/source/index.rst +2 -2
- {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib/__init__.py +6 -6
- {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib/_version.py +3 -3
- {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib/expand.py +27 -34
- {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib/ext/polars.py +170 -72
- {fpathlib-0.1.5 → fpathlib-0.1.7/src/fpathlib.egg-info}/PKG-INFO +3 -3
- fpathlib-0.1.7/src/fpathlib.egg-info/scm_version.json +8 -0
- fpathlib-0.1.5/src/fpathlib.egg-info/scm_version.json +0 -8
- {fpathlib-0.1.5 → fpathlib-0.1.7}/.gitignore +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/LICENSE +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/MANIFEST.in +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/TODO.txt +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/docs/Makefile +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/docs/make.bat +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/docs/source/conf.py +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/pyproject.toml +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/scripts/conda.sh +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/scripts/deploy.sh +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/scripts/docs.sh +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/scripts/pypi.sh +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/scripts/tag.sh +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/setup.cfg +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib/ext/__init__.py +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib/fpath.py +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib/path.py +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib/utils.py +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib.egg-info/SOURCES.txt +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib.egg-info/dependency_links.txt +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib.egg-info/requires.txt +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib.egg-info/scm_file_list.json +0 -0
- {fpathlib-0.1.5 → fpathlib-0.1.7}/src/fpathlib.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: fpathlib
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.7
|
|
4
4
|
Summary: A package to combine paths with metadata
|
|
5
5
|
Author-email: "C. Lockhart" <clockha2@gmu.edu>
|
|
6
6
|
Requires-Python: >=3.12
|
|
@@ -25,10 +25,10 @@ does this with an f-string-like path pattern -- `FPath` -- whose
|
|
|
25
25
|
`{variable}` fields are captured out of every matching path on disk.
|
|
26
26
|
|
|
27
27
|
```python
|
|
28
|
-
from fpathlib import
|
|
28
|
+
from fpathlib import expand
|
|
29
29
|
|
|
30
30
|
# Given files like data/tr1/output/0/job2.log, data/tr2/output/1/job0.log, ...
|
|
31
|
-
expanded =
|
|
31
|
+
expanded = expand("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
|
|
32
32
|
|
|
33
33
|
for path in expanded:
|
|
34
34
|
print(path, path.metadata)
|
|
@@ -8,10 +8,10 @@ does this with an f-string-like path pattern -- `FPath` -- whose
|
|
|
8
8
|
`{variable}` fields are captured out of every matching path on disk.
|
|
9
9
|
|
|
10
10
|
```python
|
|
11
|
-
from fpathlib import
|
|
11
|
+
from fpathlib import expand
|
|
12
12
|
|
|
13
13
|
# Given files like data/tr1/output/0/job2.log, data/tr2/output/1/job0.log, ...
|
|
14
|
-
expanded =
|
|
14
|
+
expanded = expand("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
|
|
15
15
|
|
|
16
16
|
for path in expanded:
|
|
17
17
|
print(path, path.metadata)
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{% set name = "fpathlib" %}
|
|
2
|
-
{% set version = "0.1.
|
|
2
|
+
{% set version = "0.1.6" %}
|
|
3
3
|
|
|
4
4
|
package:
|
|
5
5
|
name: {{ name|lower }}
|
|
@@ -7,7 +7,7 @@ package:
|
|
|
7
7
|
|
|
8
8
|
source:
|
|
9
9
|
url: https://pypi.io/packages/source/{{ name[0] }}/{{ name }}/fpathlib-{{ version }}.tar.gz
|
|
10
|
-
sha256:
|
|
10
|
+
sha256: a7a3b1d99483a33a81db47de608a6e791a5afa155d062a5914fd9eee98878da1
|
|
11
11
|
|
|
12
12
|
build:
|
|
13
13
|
noarch: python
|
|
@@ -23,9 +23,11 @@ FPath and ExpandedFPath
|
|
|
23
23
|
Expanding paths
|
|
24
24
|
----------------
|
|
25
25
|
|
|
26
|
-
.. autofunction:: fpathlib.
|
|
26
|
+
.. autofunction:: fpathlib.expand
|
|
27
27
|
|
|
28
|
-
.. autofunction:: fpathlib.
|
|
28
|
+
.. autofunction:: fpathlib.iexpand
|
|
29
|
+
|
|
30
|
+
.. autofunction:: fpathlib.expand_arg
|
|
29
31
|
|
|
30
32
|
.. autofunction:: fpathlib.is_expandable
|
|
31
33
|
|
|
@@ -11,10 +11,10 @@ own filename.
|
|
|
11
11
|
|
|
12
12
|
.. code-block:: python
|
|
13
13
|
|
|
14
|
-
from fpathlib import
|
|
14
|
+
from fpathlib import expand
|
|
15
15
|
|
|
16
16
|
# Given files like data/tr1/output/0/job2.log, data/tr2/output/1/job0.log, ...
|
|
17
|
-
expanded =
|
|
17
|
+
expanded = expand("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
|
|
18
18
|
|
|
19
19
|
for path in expanded:
|
|
20
20
|
print(path, path.metadata)
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
from fpathlib.path import Path
|
|
2
2
|
from fpathlib.fpath import FPath, ExpandedFPath
|
|
3
3
|
from fpathlib.expand import (
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
4
|
+
expand,
|
|
5
|
+
iexpand,
|
|
6
|
+
expand_arg,
|
|
7
7
|
is_expandable,
|
|
8
8
|
)
|
|
9
9
|
|
|
@@ -11,8 +11,8 @@ __all__ = [
|
|
|
11
11
|
"Path",
|
|
12
12
|
"FPath",
|
|
13
13
|
"ExpandedFPath",
|
|
14
|
-
"
|
|
15
|
-
"
|
|
16
|
-
"
|
|
14
|
+
"expand",
|
|
15
|
+
"iexpand",
|
|
16
|
+
"expand_arg",
|
|
17
17
|
"is_expandable",
|
|
18
18
|
]
|
|
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
|
|
|
18
18
|
commit_id: str | None
|
|
19
19
|
__commit_id__: str | None
|
|
20
20
|
|
|
21
|
-
__version__ = version = '0.1.
|
|
22
|
-
__version_tuple__ = version_tuple = (0, 1,
|
|
21
|
+
__version__ = version = '0.1.7'
|
|
22
|
+
__version_tuple__ = version_tuple = (0, 1, 7)
|
|
23
23
|
|
|
24
|
-
__commit_id__ = commit_id = '
|
|
24
|
+
__commit_id__ = commit_id = 'g6fb0fd96d'
|
|
@@ -4,7 +4,7 @@ import parse
|
|
|
4
4
|
from .fpath import ExpandedFPath, FPath
|
|
5
5
|
|
|
6
6
|
|
|
7
|
-
def
|
|
7
|
+
def expand(fpath, *, exclude_path_patterns=None, require_metadata=True):
|
|
8
8
|
"""
|
|
9
9
|
Use an f-string to extract out a collection of paths, where the f-string variables
|
|
10
10
|
are captured and stored along the path name. This is a convenience function that
|
|
@@ -31,9 +31,9 @@ def expand_fpath(fpath, *, exclude_path_patterns=None, require_metadata=True):
|
|
|
31
31
|
)
|
|
32
32
|
|
|
33
33
|
|
|
34
|
-
def
|
|
34
|
+
def iexpand(fpath, *, exclude_path_patterns=None, require_metadata=True, errors="raise"):
|
|
35
35
|
"""
|
|
36
|
-
Generator equivalent of :func:`.
|
|
36
|
+
Generator equivalent of :func:`.expand` -- lazily yields each
|
|
37
37
|
matching :obj:`.Path` instead of building the whole
|
|
38
38
|
:obj:`.ExpandedFPath` up front. This is a convenience function that
|
|
39
39
|
simply creates an :obj:`FPath` and calls its :meth:`.FPath.iexpand`
|
|
@@ -63,58 +63,51 @@ def iexpand_fpath(fpath, *, exclude_path_patterns=None, require_metadata=True, e
|
|
|
63
63
|
)
|
|
64
64
|
|
|
65
65
|
|
|
66
|
-
def
|
|
66
|
+
def expand_arg(f=None, require_expandable=False):
|
|
67
67
|
"""
|
|
68
|
-
Decorator for :func:`.
|
|
68
|
+
Decorator for :func:`.expand`. Expands `fpath` (if it has {}
|
|
69
|
+
named captures) before calling the wrapped function with the result;
|
|
70
|
+
otherwise calls it with `fpath` unchanged. Callers that need to react
|
|
71
|
+
differently depending on whether expansion actually happened (e.g. to
|
|
72
|
+
join in captured metadata) should check `isinstance(expanded_fpath,
|
|
73
|
+
ExpandedFPath)` themselves, inside the wrapped function -- this
|
|
74
|
+
decorator doesn't hook into that, it only handles expansion.
|
|
69
75
|
|
|
70
76
|
Parameters
|
|
71
77
|
----------
|
|
72
78
|
f : :obj:`callable`
|
|
73
|
-
A function that takes an :obj:`.ExpandedFPath`
|
|
79
|
+
A function that takes an :obj:`.ExpandedFPath` (or, if not
|
|
80
|
+
expandable and `require_expandable` is False, the original `fpath`)
|
|
81
|
+
as its first argument.
|
|
74
82
|
require_expandable : :obj:`bool`
|
|
75
|
-
Whether to require `fpath` to have {} named captures
|
|
76
|
-
(the default), a plain literal path
|
|
77
|
-
through to `f` unexpanded
|
|
78
|
-
no ExpandedFPath to hand it in that case. (Default: False)
|
|
79
|
-
post_process : :obj:`callable`
|
|
80
|
-
A function that takes the output of `f` and the :obj:`.ExpandedFPath`.
|
|
81
|
-
(Default: None).
|
|
83
|
+
Whether to require `fpath` to have {} named captures, raising
|
|
84
|
+
ValueError otherwise. If False (the default), a plain literal path
|
|
85
|
+
or glob is passed straight through to `f` unexpanded. (Default: False)
|
|
82
86
|
"""
|
|
83
87
|
|
|
84
88
|
def decorator(f):
|
|
85
89
|
@wraps(f)
|
|
86
90
|
def wrapper(fpath, *args, **kwargs):
|
|
87
|
-
# If fpath is an :obj:`ExpandedFPath`, just call f with it.
|
|
91
|
+
# If fpath is already an :obj:`ExpandedFPath`, just call f with it.
|
|
88
92
|
if isinstance(fpath, ExpandedFPath):
|
|
89
|
-
|
|
90
|
-
result = f(fpath, *args, **kwargs)
|
|
93
|
+
return f(fpath, *args, **kwargs)
|
|
91
94
|
|
|
92
95
|
# Expand fpath if it is expandable.
|
|
93
|
-
|
|
96
|
+
if is_expandable(fpath):
|
|
94
97
|
exclude_path_patterns = kwargs.pop("exclude_path_patterns", None)
|
|
95
98
|
require_metadata = kwargs.pop("require_metadata", True)
|
|
96
|
-
expanded_fpath =
|
|
99
|
+
expanded_fpath = expand(
|
|
97
100
|
fpath,
|
|
98
101
|
exclude_path_patterns=exclude_path_patterns,
|
|
99
102
|
require_metadata=require_metadata,
|
|
100
103
|
)
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
# fpath has no {} captures, so there's no ExpandedFPath to
|
|
104
|
-
# build -- nothing for post_process (e.g. join_metadata) to
|
|
105
|
-
# join metadata from. Call f directly and return immediately,
|
|
106
|
-
# skipping post_process entirely, rather than falling through
|
|
107
|
-
# to it with no expanded_fpath to give it.
|
|
108
|
-
else:
|
|
109
|
-
if require_expandable:
|
|
110
|
-
msg = f"fpath not expandable: '{fpath}'"
|
|
111
|
-
raise ValueError(msg)
|
|
112
|
-
return f(fpath, *args, **kwargs)
|
|
113
|
-
|
|
114
|
-
if post_process is not None:
|
|
115
|
-
result = post_process(result, expanded_fpath)
|
|
104
|
+
return f(expanded_fpath, *args, **kwargs)
|
|
116
105
|
|
|
117
|
-
|
|
106
|
+
# fpath has no {} captures.
|
|
107
|
+
if require_expandable:
|
|
108
|
+
msg = f"fpath not expandable: '{fpath}'"
|
|
109
|
+
raise ValueError(msg)
|
|
110
|
+
return f(fpath, *args, **kwargs)
|
|
118
111
|
|
|
119
112
|
return wrapper
|
|
120
113
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
from functools import wraps
|
|
2
2
|
import polars as _polars
|
|
3
|
-
from fpathlib import
|
|
3
|
+
from fpathlib import expand_arg, is_expandable, ExpandedFPath
|
|
4
4
|
|
|
5
5
|
|
|
6
6
|
def __getattr__(name):
|
|
@@ -13,15 +13,59 @@ def __getattr__(name):
|
|
|
13
13
|
return getattr(_polars, name)
|
|
14
14
|
|
|
15
15
|
|
|
16
|
-
def join_metadata(
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
16
|
+
def join_metadata(f):
|
|
17
|
+
"""
|
|
18
|
+
Wraps a scan_csv-shaped function `f(source, *args, **kwargs) ->
|
|
19
|
+
LazyFrame` with metadata-joining and "_fname" cleanup, applied
|
|
20
|
+
automatically to whatever `f` returns. Must be applied *inside*
|
|
21
|
+
@expand_arg (i.e. listed closer to `def`), so `source`
|
|
22
|
+
here is always the already-expanded value, not the raw caller-supplied
|
|
23
|
+
pattern:
|
|
24
|
+
|
|
25
|
+
@expand_arg
|
|
26
|
+
@join_metadata
|
|
27
|
+
def scan_csv(source, ...): ...
|
|
28
|
+
|
|
29
|
+
Guards against getting that order backwards: if `source` still looks
|
|
30
|
+
like an unexpanded {} pattern at this point, expansion can only have
|
|
31
|
+
failed to run (wrong decorator order, or this decorator used without
|
|
32
|
+
@expand_arg at all) -- raise immediately rather than
|
|
33
|
+
silently joining no metadata.
|
|
34
|
+
"""
|
|
22
35
|
|
|
23
|
-
@
|
|
24
|
-
def
|
|
36
|
+
@wraps(f)
|
|
37
|
+
def wrapper(source, *args, **kwargs):
|
|
38
|
+
lf = f(source, *args, **kwargs)
|
|
39
|
+
if not isinstance(source, ExpandedFPath) and is_expandable(source):
|
|
40
|
+
msg = (
|
|
41
|
+
f"join_metadata received an unexpanded pattern "
|
|
42
|
+
f"{source!r} -- @expand_arg must be the outer "
|
|
43
|
+
"decorator, applied above (not below) @join_metadata"
|
|
44
|
+
)
|
|
45
|
+
raise RuntimeError(msg)
|
|
46
|
+
|
|
47
|
+
# Join captured {} metadata into `lf`, keyed on the reserved
|
|
48
|
+
# "_fname" column scan_csv/scan_parquet always create, then always
|
|
49
|
+
# drop "_fname" itself. Callers that want to keep the file path
|
|
50
|
+
# under a different name must alias it to that name *before* this
|
|
51
|
+
# runs (inside `f`'s own body) -- by the time this returns,
|
|
52
|
+
# "_fname" is gone either way.
|
|
53
|
+
if isinstance(source, ExpandedFPath):
|
|
54
|
+
# "fname" is ExpandedFPath.to_polars()'s normal, public column
|
|
55
|
+
# name; rename it to the same reserved "_fname" scan_csv/
|
|
56
|
+
# scan_parquet use, so the join key can never collide with an
|
|
57
|
+
# `include_file_paths` name a caller chose (including "fname"
|
|
58
|
+
# itself).
|
|
59
|
+
metadata = source.to_polars(lazy=isinstance(lf, _polars.LazyFrame))
|
|
60
|
+
metadata = metadata.rename({"fname": "_fname"})
|
|
61
|
+
lf = lf.join(metadata, on="_fname")
|
|
62
|
+
|
|
63
|
+
return lf.drop("_fname")
|
|
64
|
+
|
|
65
|
+
return wrapper
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def read_csv(source, *args, **kwargs):
|
|
25
69
|
"""
|
|
26
70
|
Read the paths in the collection as CSV files, and return a
|
|
27
71
|
:obj:`polars.DataFrame` along with the metadata captured from the path
|
|
@@ -29,8 +73,8 @@ def read_csv(expanded_fpath, *args, **kwargs):
|
|
|
29
73
|
|
|
30
74
|
Parameters
|
|
31
75
|
----------
|
|
32
|
-
|
|
33
|
-
An expanded f-string path.
|
|
76
|
+
source : :obj:`fpathlib.ExpandedFPath`, or any type accepted by :obj:`polars.read_csv`
|
|
77
|
+
An expanded f-string path, or a plain path/glob.
|
|
34
78
|
*args
|
|
35
79
|
Positional arguments to pass to :meth:`polars.read_csv`.
|
|
36
80
|
**kwargs
|
|
@@ -41,14 +85,14 @@ def read_csv(expanded_fpath, *args, **kwargs):
|
|
|
41
85
|
:obj:`polars.DataFrame`
|
|
42
86
|
"""
|
|
43
87
|
|
|
44
|
-
return scan_csv
|
|
88
|
+
return scan_csv(source, *args, **kwargs).collect()
|
|
45
89
|
|
|
46
90
|
|
|
47
|
-
@expand_fpath_decorator(post_process=join_metadata)
|
|
48
91
|
def read_txt(
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
separator=
|
|
92
|
+
source,
|
|
93
|
+
line_filter=None,
|
|
94
|
+
separator=r"\s+",
|
|
95
|
+
strip_initial_spaces=True,
|
|
52
96
|
new_columns=None,
|
|
53
97
|
has_header=False,
|
|
54
98
|
*args,
|
|
@@ -63,10 +107,13 @@ def read_txt(
|
|
|
63
107
|
|
|
64
108
|
Parameters
|
|
65
109
|
----------
|
|
66
|
-
|
|
67
|
-
An expanded f-string path.
|
|
68
|
-
|
|
69
|
-
|
|
110
|
+
source : :obj:`fpathlib.ExpandedFPath`, or any type accepted by :obj:`polars.scan_csv`
|
|
111
|
+
An expanded f-string path, or a plain path/glob.
|
|
112
|
+
line_filter : :obj:`callable`, optional
|
|
113
|
+
A function that takes a :obj:`polars.Expr` for the line's text and
|
|
114
|
+
returns a boolean :obj:`polars.Expr`, used to filter lines before
|
|
115
|
+
splitting by the separator. E.g.
|
|
116
|
+
`lambda line: line.str.starts_with("#").not_()`.
|
|
70
117
|
separator : :obj:`str`, optional
|
|
71
118
|
Deliminatorg to split each line into fields.
|
|
72
119
|
new_columns : :obj:`list`[:obj:`str`], optional
|
|
@@ -86,10 +133,11 @@ def read_txt(
|
|
|
86
133
|
:obj:`polars.DataFrame`
|
|
87
134
|
"""
|
|
88
135
|
|
|
89
|
-
return scan_txt
|
|
90
|
-
|
|
91
|
-
|
|
136
|
+
return scan_txt(
|
|
137
|
+
source,
|
|
138
|
+
line_filter=line_filter,
|
|
92
139
|
separator=separator,
|
|
140
|
+
strip_initial_spaces=strip_initial_spaces,
|
|
93
141
|
new_columns=new_columns,
|
|
94
142
|
has_header=has_header,
|
|
95
143
|
*args,
|
|
@@ -97,8 +145,9 @@ def read_txt(
|
|
|
97
145
|
).collect()
|
|
98
146
|
|
|
99
147
|
|
|
100
|
-
@
|
|
101
|
-
|
|
148
|
+
@expand_arg
|
|
149
|
+
@join_metadata
|
|
150
|
+
def scan_csv(source, include_file_paths=None, *args, **kwargs):
|
|
102
151
|
"""
|
|
103
152
|
Scan the paths in the collection as CSV files, and return a
|
|
104
153
|
:obj:`polars.LazyFrame` along with the metadata captured from the path
|
|
@@ -106,8 +155,12 @@ def scan_csv(expanded_fpath, *args, **kwargs):
|
|
|
106
155
|
|
|
107
156
|
Parameters
|
|
108
157
|
----------
|
|
109
|
-
|
|
110
|
-
An expanded f-string path.
|
|
158
|
+
source : :obj:`fpathlib.ExpandedFPath`, or any type accepted by :obj:`polars.scan_csv`
|
|
159
|
+
An expanded f-string path, or a plain path/glob.
|
|
160
|
+
include_file_paths : :obj:`str`, optional
|
|
161
|
+
Name to give a column of each row's source file path in the
|
|
162
|
+
output. The file path is always used internally to join captured
|
|
163
|
+
metadata; without this, it isn't kept in the result. (Default: None)
|
|
111
164
|
*args
|
|
112
165
|
Positional arguments to pass to :meth:`polars.scan_csv`.
|
|
113
166
|
**kwargs
|
|
@@ -119,17 +172,19 @@ def scan_csv(expanded_fpath, *args, **kwargs):
|
|
|
119
172
|
"""
|
|
120
173
|
|
|
121
174
|
lf = _polars.scan_csv(
|
|
122
|
-
|
|
123
|
-
include_file_paths="
|
|
175
|
+
source,
|
|
176
|
+
include_file_paths="_fname",
|
|
124
177
|
*args,
|
|
125
178
|
**kwargs,
|
|
126
179
|
)
|
|
127
|
-
|
|
180
|
+
if include_file_paths is not None:
|
|
181
|
+
lf = lf.with_columns(_polars.col("_fname").alias(include_file_paths))
|
|
128
182
|
return lf
|
|
129
183
|
|
|
130
184
|
|
|
131
|
-
@
|
|
132
|
-
|
|
185
|
+
@expand_arg
|
|
186
|
+
@join_metadata
|
|
187
|
+
def scan_parquet(source, include_file_paths=None, *args, **kwargs):
|
|
133
188
|
"""
|
|
134
189
|
Scan the paths in the collection as a parquet file, and return a
|
|
135
190
|
:obj:`polars.LazyFrame` along with the metadata captured from the path
|
|
@@ -137,8 +192,12 @@ def scan_parquet(expanded_fpath, *args, **kwargs):
|
|
|
137
192
|
|
|
138
193
|
Parameters
|
|
139
194
|
----------
|
|
140
|
-
|
|
141
|
-
An expanded f-string path.
|
|
195
|
+
source : :obj:`fpathlib.ExpandedFPath`, or any type accepted by :obj:`polars.scan_parquet`
|
|
196
|
+
An expanded f-string path, or a plain path/glob.
|
|
197
|
+
include_file_paths : :obj:`str`, optional
|
|
198
|
+
Name to give a column of each row's source file path in the
|
|
199
|
+
output. The file path is always used internally to join captured
|
|
200
|
+
metadata; without this, it isn't kept in the result. (Default: None)
|
|
142
201
|
*args
|
|
143
202
|
Positional arguments to pass to :meth:`polars.scan_parquet`.
|
|
144
203
|
**kwargs
|
|
@@ -150,24 +209,26 @@ def scan_parquet(expanded_fpath, *args, **kwargs):
|
|
|
150
209
|
"""
|
|
151
210
|
|
|
152
211
|
lf = _polars.scan_parquet(
|
|
153
|
-
|
|
154
|
-
include_file_paths="
|
|
212
|
+
source,
|
|
213
|
+
include_file_paths="_fname",
|
|
155
214
|
*args,
|
|
156
215
|
**kwargs,
|
|
157
216
|
)
|
|
158
|
-
|
|
217
|
+
if include_file_paths is not None:
|
|
218
|
+
lf = lf.with_columns(_polars.col("_fname").alias(include_file_paths))
|
|
159
219
|
return lf
|
|
160
220
|
|
|
161
221
|
|
|
162
|
-
|
|
163
|
-
@expand_fpath_decorator(post_process=join_metadata)
|
|
222
|
+
@expand_arg
|
|
164
223
|
def scan_txt(
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
separator=
|
|
224
|
+
source,
|
|
225
|
+
line_filter=None,
|
|
226
|
+
separator=r"\s+",
|
|
227
|
+
strip_initial_spaces=True,
|
|
168
228
|
new_columns=None,
|
|
169
229
|
has_header=False,
|
|
170
|
-
|
|
230
|
+
include_line=None,
|
|
231
|
+
include_file_paths=None,
|
|
171
232
|
usecols=None,
|
|
172
233
|
validate_schema=True,
|
|
173
234
|
*args,
|
|
@@ -182,10 +243,13 @@ def scan_txt(
|
|
|
182
243
|
|
|
183
244
|
Parameters
|
|
184
245
|
----------
|
|
185
|
-
|
|
186
|
-
An expanded f-string path.
|
|
187
|
-
|
|
188
|
-
|
|
246
|
+
source : :obj:`fpathlib.ExpandedFPath`, or any type accepted by :obj:`polars.scan_csv`
|
|
247
|
+
An expanded f-string path, or a plain path/glob.
|
|
248
|
+
line_filter : :obj:`callable`, optional
|
|
249
|
+
A function that takes a :obj:`polars.Expr` for the line's text and
|
|
250
|
+
returns a boolean :obj:`polars.Expr`, used to filter lines before
|
|
251
|
+
splitting by the separator. E.g.
|
|
252
|
+
`lambda line: line.str.starts_with("#").not_()`.
|
|
189
253
|
separator : :obj:`str`, optional
|
|
190
254
|
Deliminatorg to split each line into fields.
|
|
191
255
|
new_columns : :obj:`list`[:obj:`str`], optional
|
|
@@ -195,8 +259,13 @@ def scan_txt(
|
|
|
195
259
|
Whether the text files have a header line that should be skipped. The header
|
|
196
260
|
must have the same delimiter as the separator provided in `separator`.
|
|
197
261
|
(Default: False)
|
|
198
|
-
|
|
199
|
-
|
|
262
|
+
include_line : :obj:`str`, optional
|
|
263
|
+
Name to give a column of each row's original, unsplit line text in
|
|
264
|
+
the output. Without this, it isn't kept in the result. (Default: None)
|
|
265
|
+
include_file_paths : :obj:`str`, optional
|
|
266
|
+
Name to give a column of each row's source file path in the
|
|
267
|
+
output. The file path is always used internally to join captured
|
|
268
|
+
metadata; without this, it isn't kept in the result. (Default: None)
|
|
200
269
|
usecols : :obj:`list`[:obj:`int`], optional
|
|
201
270
|
Indexes of columns to keep in the output. If not provided, all columns are kept. Only applicable if `separator` is provided.
|
|
202
271
|
validate_schema : :obj:`bool`
|
|
@@ -220,53 +289,68 @@ def scan_txt(
|
|
|
220
289
|
:obj:`polars.LazyFrame`
|
|
221
290
|
"""
|
|
222
291
|
|
|
223
|
-
# TODO there are forbidden variables that should not be in
|
|
224
|
-
# such as '
|
|
292
|
+
# TODO there are forbidden variables that should not be in source
|
|
293
|
+
# such as '_line' and 'fields' and '_fname'
|
|
225
294
|
|
|
226
295
|
# TODO schema and schema_overrides is probably broken
|
|
227
296
|
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
297
|
+
# Delegates to scan_csv (rather than calling _polars.scan_csv and
|
|
298
|
+
# doing the metadata-join/include_file_paths handling directly) so
|
|
299
|
+
# that handling -- including the recursive single-file sample call
|
|
300
|
+
# below, since source[0] is a plain Path, not an ExpandedFPath --
|
|
301
|
+
# isn't duplicated here. scan_csv's own @expand_arg is a
|
|
302
|
+
# no-op on `source` at this point: it's already been resolved by
|
|
303
|
+
# scan_txt's decorator, so scan_csv just sees an ExpandedFPath or an
|
|
304
|
+
# already-unexpandable literal/glob, either way with nothing left to
|
|
305
|
+
# expand.
|
|
306
|
+
lf = scan_csv(
|
|
307
|
+
source,
|
|
308
|
+
include_file_paths=include_file_paths,
|
|
231
309
|
separator="\n",
|
|
232
|
-
new_columns=["
|
|
310
|
+
new_columns=["_line"],
|
|
233
311
|
has_header=False,
|
|
234
312
|
**kwargs,
|
|
235
313
|
)
|
|
236
314
|
|
|
315
|
+
# Strip leading whitespace from each line if requested.
|
|
316
|
+
if strip_initial_spaces:
|
|
317
|
+
lf = lf.with_columns(_polars.col("_line").str.strip_chars_start())
|
|
318
|
+
|
|
237
319
|
# Can filter lines before doing any further processing
|
|
238
320
|
# This could be to remove lines with comments, etc.
|
|
239
|
-
if
|
|
240
|
-
lf = lf.filter(
|
|
321
|
+
if line_filter is not None:
|
|
322
|
+
lf = lf.filter(line_filter(_polars.col("_line")))
|
|
323
|
+
|
|
324
|
+
if include_line is not None:
|
|
325
|
+
lf = lf.with_columns(_polars.col("_line").alias(include_line))
|
|
241
326
|
|
|
242
327
|
# Separate lines into fields using `separator`
|
|
243
328
|
if separator is not None:
|
|
244
329
|
# Separate line into fields by separator
|
|
245
330
|
lf = lf.with_columns(
|
|
246
|
-
_polars.col("
|
|
331
|
+
_polars.col("_line").str.split(separator, literal=False).alias("fields")
|
|
247
332
|
)
|
|
248
|
-
|
|
249
|
-
if not keep_line:
|
|
250
|
-
lf = lf.drop("line")
|
|
333
|
+
lf = lf.drop("_line")
|
|
251
334
|
|
|
252
335
|
# With many matched files, inferring the field count/dtypes directly
|
|
253
336
|
# against the full glob is extremely slow (every file has to be opened
|
|
254
337
|
# before a `.head()` takes effect). Instead, recurse on a single
|
|
255
|
-
# representative file --
|
|
338
|
+
# representative file -- source[0] -- and reuse its already-cheap
|
|
256
339
|
# (single-file) inference below instead of duplicating it here. Computed
|
|
257
340
|
# once here since it's needed by both the field-count branch below and
|
|
258
341
|
# the dtype-inference branch further down.
|
|
259
342
|
infer_schema = kwargs.get("infer_schema", True)
|
|
260
343
|
sample_schema = None
|
|
261
344
|
if (
|
|
262
|
-
isinstance(
|
|
263
|
-
and len(
|
|
345
|
+
isinstance(source, ExpandedFPath)
|
|
346
|
+
and len(source) > 1
|
|
264
347
|
and (usecols is None or infer_schema)
|
|
265
348
|
):
|
|
266
349
|
sample_schema = scan_txt(
|
|
267
|
-
|
|
268
|
-
|
|
350
|
+
source[0],
|
|
351
|
+
line_filter=line_filter,
|
|
269
352
|
separator=separator,
|
|
353
|
+
strip_initial_spaces=strip_initial_spaces,
|
|
270
354
|
new_columns=new_columns,
|
|
271
355
|
has_header=has_header,
|
|
272
356
|
usecols=usecols,
|
|
@@ -282,7 +366,10 @@ def scan_txt(
|
|
|
282
366
|
|
|
283
367
|
else:
|
|
284
368
|
if sample_schema is not None:
|
|
285
|
-
|
|
369
|
+
# The recursive sample call above is always non-expandable
|
|
370
|
+
# (source[0] is a plain Path), so it already drops "_fname"
|
|
371
|
+
# itself -- no adjustment needed here.
|
|
372
|
+
n_fields = len(sample_schema)
|
|
286
373
|
else:
|
|
287
374
|
n_fields = (
|
|
288
375
|
lf.head(1)
|
|
@@ -339,11 +426,9 @@ def scan_txt(
|
|
|
339
426
|
# Infer dtypes?
|
|
340
427
|
if infer_schema:
|
|
341
428
|
if sample_schema is not None:
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
if name != "fname"
|
|
346
|
-
}
|
|
429
|
+
# The recursive sample call already drops "_fname" (see
|
|
430
|
+
# above), so no need to filter it back out here.
|
|
431
|
+
inferred_schema = dict(sample_schema.items())
|
|
347
432
|
else:
|
|
348
433
|
sample = (
|
|
349
434
|
lf.head(kwargs.get("infer_schema_length", 100))
|
|
@@ -354,4 +439,17 @@ def scan_txt(
|
|
|
354
439
|
inferred_schema = _polars.read_csv(sample).schema
|
|
355
440
|
lf = lf.cast(inferred_schema)
|
|
356
441
|
|
|
442
|
+
elif include_line is None:
|
|
443
|
+
# No separator means each row is just the raw line -- that's the
|
|
444
|
+
# whole point of this mode, so it has to stay visible somehow.
|
|
445
|
+
# Fall back to the traditional "line" name rather than leaking the
|
|
446
|
+
# internal "_line" name, since the caller didn't ask for a specific
|
|
447
|
+
# one via include_line.
|
|
448
|
+
lf = lf.rename({"_line": "line"})
|
|
449
|
+
else:
|
|
450
|
+
# include_line was given, so the alias was already created above
|
|
451
|
+
# (before this if/elif/else) -- drop the internal "_line" itself so
|
|
452
|
+
# it doesn't also leak into the output alongside it.
|
|
453
|
+
lf = lf.drop("_line")
|
|
454
|
+
|
|
357
455
|
return lf
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: fpathlib
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.7
|
|
4
4
|
Summary: A package to combine paths with metadata
|
|
5
5
|
Author-email: "C. Lockhart" <clockha2@gmu.edu>
|
|
6
6
|
Requires-Python: >=3.12
|
|
@@ -25,10 +25,10 @@ does this with an f-string-like path pattern -- `FPath` -- whose
|
|
|
25
25
|
`{variable}` fields are captured out of every matching path on disk.
|
|
26
26
|
|
|
27
27
|
```python
|
|
28
|
-
from fpathlib import
|
|
28
|
+
from fpathlib import expand
|
|
29
29
|
|
|
30
30
|
# Given files like data/tr1/output/0/job2.log, data/tr2/output/1/job0.log, ...
|
|
31
|
-
expanded =
|
|
31
|
+
expanded = expand("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
|
|
32
32
|
|
|
33
33
|
for path in expanded:
|
|
34
34
|
print(path, path.metadata)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|