fpathlib 0.1.4.dev16__tar.gz → 0.1.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fpathlib-0.1.4.dev16/src/fpathlib.egg-info → fpathlib-0.1.6}/PKG-INFO +3 -3
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/README.md +2 -2
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/conda-recipe/meta.yaml +2 -2
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/docs/source/api.rst +4 -2
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/docs/source/index.rst +2 -2
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/scripts/pypi.sh +8 -1
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib/__init__.py +6 -6
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib/_version.py +3 -3
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib/expand.py +30 -32
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib/ext/polars.py +160 -70
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6/src/fpathlib.egg-info}/PKG-INFO +3 -3
- fpathlib-0.1.6/src/fpathlib.egg-info/scm_version.json +8 -0
- fpathlib-0.1.4.dev16/src/fpathlib.egg-info/scm_version.json +0 -8
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/.gitignore +0 -0
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/LICENSE +0 -0
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/MANIFEST.in +0 -0
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/TODO.txt +0 -0
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/docs/Makefile +0 -0
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/docs/make.bat +0 -0
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/docs/source/conf.py +0 -0
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/pyproject.toml +0 -0
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/scripts/conda.sh +0 -0
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/scripts/deploy.sh +0 -0
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/scripts/docs.sh +0 -0
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/scripts/tag.sh +0 -0
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/setup.cfg +0 -0
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib/ext/__init__.py +0 -0
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib/fpath.py +0 -0
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib/path.py +0 -0
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib/utils.py +0 -0
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib.egg-info/SOURCES.txt +0 -0
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib.egg-info/dependency_links.txt +0 -0
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib.egg-info/requires.txt +0 -0
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib.egg-info/scm_file_list.json +0 -0
- {fpathlib-0.1.4.dev16 → fpathlib-0.1.6}/src/fpathlib.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: fpathlib
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.6
|
|
4
4
|
Summary: A package to combine paths with metadata
|
|
5
5
|
Author-email: "C. Lockhart" <clockha2@gmu.edu>
|
|
6
6
|
Requires-Python: >=3.12
|
|
@@ -25,10 +25,10 @@ does this with an f-string-like path pattern -- `FPath` -- whose
|
|
|
25
25
|
`{variable}` fields are captured out of every matching path on disk.
|
|
26
26
|
|
|
27
27
|
```python
|
|
28
|
-
from fpathlib import
|
|
28
|
+
from fpathlib import expand
|
|
29
29
|
|
|
30
30
|
# Given files like data/tr1/output/0/job2.log, data/tr2/output/1/job0.log, ...
|
|
31
|
-
expanded =
|
|
31
|
+
expanded = expand("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
|
|
32
32
|
|
|
33
33
|
for path in expanded:
|
|
34
34
|
print(path, path.metadata)
|
|
@@ -8,10 +8,10 @@ does this with an f-string-like path pattern -- `FPath` -- whose
|
|
|
8
8
|
`{variable}` fields are captured out of every matching path on disk.
|
|
9
9
|
|
|
10
10
|
```python
|
|
11
|
-
from fpathlib import
|
|
11
|
+
from fpathlib import expand
|
|
12
12
|
|
|
13
13
|
# Given files like data/tr1/output/0/job2.log, data/tr2/output/1/job0.log, ...
|
|
14
|
-
expanded =
|
|
14
|
+
expanded = expand("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
|
|
15
15
|
|
|
16
16
|
for path in expanded:
|
|
17
17
|
print(path, path.metadata)
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{% set name = "fpathlib" %}
|
|
2
|
-
{% set version = "0.1.
|
|
2
|
+
{% set version = "0.1.5" %}
|
|
3
3
|
|
|
4
4
|
package:
|
|
5
5
|
name: {{ name|lower }}
|
|
@@ -7,7 +7,7 @@ package:
|
|
|
7
7
|
|
|
8
8
|
source:
|
|
9
9
|
url: https://pypi.io/packages/source/{{ name[0] }}/{{ name }}/fpathlib-{{ version }}.tar.gz
|
|
10
|
-
sha256:
|
|
10
|
+
sha256: 083299d5a41577484af5a8447d185a6ab1c190b2dfe31690375349c163ea61ab
|
|
11
11
|
|
|
12
12
|
build:
|
|
13
13
|
noarch: python
|
|
@@ -23,9 +23,11 @@ FPath and ExpandedFPath
|
|
|
23
23
|
Expanding paths
|
|
24
24
|
----------------
|
|
25
25
|
|
|
26
|
-
.. autofunction:: fpathlib.
|
|
26
|
+
.. autofunction:: fpathlib.expand
|
|
27
27
|
|
|
28
|
-
.. autofunction:: fpathlib.
|
|
28
|
+
.. autofunction:: fpathlib.iexpand
|
|
29
|
+
|
|
30
|
+
.. autofunction:: fpathlib.expand_arg
|
|
29
31
|
|
|
30
32
|
.. autofunction:: fpathlib.is_expandable
|
|
31
33
|
|
|
@@ -11,10 +11,10 @@ own filename.
|
|
|
11
11
|
|
|
12
12
|
.. code-block:: python
|
|
13
13
|
|
|
14
|
-
from fpathlib import
|
|
14
|
+
from fpathlib import expand
|
|
15
15
|
|
|
16
16
|
# Given files like data/tr1/output/0/job2.log, data/tr2/output/1/job0.log, ...
|
|
17
|
-
expanded =
|
|
17
|
+
expanded = expand("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
|
|
18
18
|
|
|
19
19
|
for path in expanded:
|
|
20
20
|
print(path, path.metadata)
|
|
@@ -21,7 +21,14 @@ python3 -m build
|
|
|
21
21
|
# 0.1.3/0.1.2 tag-collision incident slipped past a git-describe check even
|
|
22
22
|
# though the build itself came out as a .devN), inspect what actually got
|
|
23
23
|
# built. This is the real signal of whether the release is clean.
|
|
24
|
-
|
|
24
|
+
#
|
|
25
|
+
# `grep -c` exits 1 (not just prints "0") when it finds zero matches --
|
|
26
|
+
# with `set -e` active (deploy.sh sets it, and this script is sourced into
|
|
27
|
+
# that same shell), that silently killed the whole deploy right here for
|
|
28
|
+
# every *clean* release build (the exact case with 0 .dev files), before
|
|
29
|
+
# ever reaching twine upload. `|| true` keeps the count without letting
|
|
30
|
+
# grep's "no matches" exit status abort the script.
|
|
31
|
+
dev_artifacts=$(ls dist/ | grep -c '\.dev[0-9]' || true)
|
|
25
32
|
if [ $allow_dev -eq 0 ] && [ "$dev_artifacts" != "0" ]
|
|
26
33
|
then
|
|
27
34
|
echo "built version is a dev version (tag doesn't point at a clean, distinct commit -- check 'git describe --tags --long' and 'git status'), not uploading to pypi"
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
from fpathlib.path import Path
|
|
2
2
|
from fpathlib.fpath import FPath, ExpandedFPath
|
|
3
3
|
from fpathlib.expand import (
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
4
|
+
expand,
|
|
5
|
+
iexpand,
|
|
6
|
+
expand_arg,
|
|
7
7
|
is_expandable,
|
|
8
8
|
)
|
|
9
9
|
|
|
@@ -11,8 +11,8 @@ __all__ = [
|
|
|
11
11
|
"Path",
|
|
12
12
|
"FPath",
|
|
13
13
|
"ExpandedFPath",
|
|
14
|
-
"
|
|
15
|
-
"
|
|
16
|
-
"
|
|
14
|
+
"expand",
|
|
15
|
+
"iexpand",
|
|
16
|
+
"expand_arg",
|
|
17
17
|
"is_expandable",
|
|
18
18
|
]
|
|
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
|
|
|
18
18
|
commit_id: str | None
|
|
19
19
|
__commit_id__: str | None
|
|
20
20
|
|
|
21
|
-
__version__ = version = '0.1.
|
|
22
|
-
__version_tuple__ = version_tuple = (0, 1,
|
|
21
|
+
__version__ = version = '0.1.6'
|
|
22
|
+
__version_tuple__ = version_tuple = (0, 1, 6)
|
|
23
23
|
|
|
24
|
-
__commit_id__ = commit_id = '
|
|
24
|
+
__commit_id__ = commit_id = 'g5162f8cd0'
|
|
@@ -4,7 +4,7 @@ import parse
|
|
|
4
4
|
from .fpath import ExpandedFPath, FPath
|
|
5
5
|
|
|
6
6
|
|
|
7
|
-
def
|
|
7
|
+
def expand(fpath, *, exclude_path_patterns=None, require_metadata=True):
|
|
8
8
|
"""
|
|
9
9
|
Use an f-string to extract out a collection of paths, where the f-string variables
|
|
10
10
|
are captured and stored along the path name. This is a convenience function that
|
|
@@ -31,9 +31,9 @@ def expand_fpath(fpath, *, exclude_path_patterns=None, require_metadata=True):
|
|
|
31
31
|
)
|
|
32
32
|
|
|
33
33
|
|
|
34
|
-
def
|
|
34
|
+
def iexpand(fpath, *, exclude_path_patterns=None, require_metadata=True, errors="raise"):
|
|
35
35
|
"""
|
|
36
|
-
Generator equivalent of :func:`.
|
|
36
|
+
Generator equivalent of :func:`.expand` -- lazily yields each
|
|
37
37
|
matching :obj:`.Path` instead of building the whole
|
|
38
38
|
:obj:`.ExpandedFPath` up front. This is a convenience function that
|
|
39
39
|
simply creates an :obj:`FPath` and calls its :meth:`.FPath.iexpand`
|
|
@@ -63,53 +63,51 @@ def iexpand_fpath(fpath, *, exclude_path_patterns=None, require_metadata=True, e
|
|
|
63
63
|
)
|
|
64
64
|
|
|
65
65
|
|
|
66
|
-
def
|
|
66
|
+
def expand_arg(f=None, require_expandable=False):
|
|
67
67
|
"""
|
|
68
|
-
Decorator for :func:`.
|
|
68
|
+
Decorator for :func:`.expand`. Expands `fpath` (if it has {}
|
|
69
|
+
named captures) before calling the wrapped function with the result;
|
|
70
|
+
otherwise calls it with `fpath` unchanged. Callers that need to react
|
|
71
|
+
differently depending on whether expansion actually happened (e.g. to
|
|
72
|
+
join in captured metadata) should check `isinstance(expanded_fpath,
|
|
73
|
+
ExpandedFPath)` themselves, inside the wrapped function -- this
|
|
74
|
+
decorator doesn't hook into that, it only handles expansion.
|
|
69
75
|
|
|
70
76
|
Parameters
|
|
71
77
|
----------
|
|
72
78
|
f : :obj:`callable`
|
|
73
|
-
A function that takes an :obj:`.ExpandedFPath`
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
79
|
+
A function that takes an :obj:`.ExpandedFPath` (or, if not
|
|
80
|
+
expandable and `require_expandable` is False, the original `fpath`)
|
|
81
|
+
as its first argument.
|
|
82
|
+
require_expandable : :obj:`bool`
|
|
83
|
+
Whether to require `fpath` to have {} named captures, raising
|
|
84
|
+
ValueError otherwise. If False (the default), a plain literal path
|
|
85
|
+
or glob is passed straight through to `f` unexpanded. (Default: False)
|
|
77
86
|
"""
|
|
78
87
|
|
|
79
88
|
def decorator(f):
|
|
80
89
|
@wraps(f)
|
|
81
90
|
def wrapper(fpath, *args, **kwargs):
|
|
82
|
-
# If fpath is an :obj:`ExpandedFPath`, just call f with it.
|
|
91
|
+
# If fpath is already an :obj:`ExpandedFPath`, just call f with it.
|
|
83
92
|
if isinstance(fpath, ExpandedFPath):
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
else:
|
|
89
|
-
# What happens if fpath is not expandable?
|
|
90
|
-
# If require_expandable is True, raise an error.
|
|
91
|
-
# Otherwise, just call f with the original fpath.
|
|
92
|
-
if not is_expandable(fpath):
|
|
93
|
-
if require_expandable:
|
|
94
|
-
msg = f"fpath not expandable: '{fpath}'"
|
|
95
|
-
raise ValueError(msg)
|
|
96
|
-
return f(fpath, *args, **kwargs)
|
|
97
|
-
|
|
98
|
-
# We know fpath is expandable, so we can expand it and call f
|
|
93
|
+
return f(fpath, *args, **kwargs)
|
|
94
|
+
|
|
95
|
+
# Expand fpath if it is expandable.
|
|
96
|
+
if is_expandable(fpath):
|
|
99
97
|
exclude_path_patterns = kwargs.pop("exclude_path_patterns", None)
|
|
100
98
|
require_metadata = kwargs.pop("require_metadata", True)
|
|
101
|
-
expanded_fpath =
|
|
99
|
+
expanded_fpath = expand(
|
|
102
100
|
fpath,
|
|
103
101
|
exclude_path_patterns=exclude_path_patterns,
|
|
104
102
|
require_metadata=require_metadata,
|
|
105
103
|
)
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
# If post_process is provided, call it with the result and the expanded_fpath.
|
|
109
|
-
if post_process is not None:
|
|
110
|
-
result = post_process(result, expanded_fpath)
|
|
104
|
+
return f(expanded_fpath, *args, **kwargs)
|
|
111
105
|
|
|
112
|
-
|
|
106
|
+
# fpath has no {} captures.
|
|
107
|
+
if require_expandable:
|
|
108
|
+
msg = f"fpath not expandable: '{fpath}'"
|
|
109
|
+
raise ValueError(msg)
|
|
110
|
+
return f(fpath, *args, **kwargs)
|
|
113
111
|
|
|
114
112
|
return wrapper
|
|
115
113
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
from functools import wraps
|
|
2
2
|
import polars as _polars
|
|
3
|
-
from fpathlib import
|
|
3
|
+
from fpathlib import expand_arg, is_expandable, ExpandedFPath
|
|
4
4
|
|
|
5
5
|
|
|
6
6
|
def __getattr__(name):
|
|
@@ -13,15 +13,59 @@ def __getattr__(name):
|
|
|
13
13
|
return getattr(_polars, name)
|
|
14
14
|
|
|
15
15
|
|
|
16
|
-
def join_metadata(
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
16
|
+
def join_metadata(f):
|
|
17
|
+
"""
|
|
18
|
+
Wraps a scan_csv-shaped function `f(source, *args, **kwargs) ->
|
|
19
|
+
LazyFrame` with metadata-joining and "_fname" cleanup, applied
|
|
20
|
+
automatically to whatever `f` returns. Must be applied *inside*
|
|
21
|
+
@expand_arg (i.e. listed closer to `def`), so `source`
|
|
22
|
+
here is always the already-expanded value, not the raw caller-supplied
|
|
23
|
+
pattern:
|
|
24
|
+
|
|
25
|
+
@expand_arg
|
|
26
|
+
@join_metadata
|
|
27
|
+
def scan_csv(source, ...): ...
|
|
28
|
+
|
|
29
|
+
Guards against getting that order backwards: if `source` still looks
|
|
30
|
+
like an unexpanded {} pattern at this point, expansion can only have
|
|
31
|
+
failed to run (wrong decorator order, or this decorator used without
|
|
32
|
+
@expand_arg at all) -- raise immediately rather than
|
|
33
|
+
silently joining no metadata.
|
|
34
|
+
"""
|
|
22
35
|
|
|
23
|
-
@
|
|
24
|
-
def
|
|
36
|
+
@wraps(f)
|
|
37
|
+
def wrapper(source, *args, **kwargs):
|
|
38
|
+
lf = f(source, *args, **kwargs)
|
|
39
|
+
if not isinstance(source, ExpandedFPath) and is_expandable(source):
|
|
40
|
+
msg = (
|
|
41
|
+
f"join_metadata received an unexpanded pattern "
|
|
42
|
+
f"{source!r} -- @expand_arg must be the outer "
|
|
43
|
+
"decorator, applied above (not below) @join_metadata"
|
|
44
|
+
)
|
|
45
|
+
raise RuntimeError(msg)
|
|
46
|
+
|
|
47
|
+
# Join captured {} metadata into `lf`, keyed on the reserved
|
|
48
|
+
# "_fname" column scan_csv/scan_parquet always create, then always
|
|
49
|
+
# drop "_fname" itself. Callers that want to keep the file path
|
|
50
|
+
# under a different name must alias it to that name *before* this
|
|
51
|
+
# runs (inside `f`'s own body) -- by the time this returns,
|
|
52
|
+
# "_fname" is gone either way.
|
|
53
|
+
if isinstance(source, ExpandedFPath):
|
|
54
|
+
# "fname" is ExpandedFPath.to_polars()'s normal, public column
|
|
55
|
+
# name; rename it to the same reserved "_fname" scan_csv/
|
|
56
|
+
# scan_parquet use, so the join key can never collide with an
|
|
57
|
+
# `include_file_paths` name a caller chose (including "fname"
|
|
58
|
+
# itself).
|
|
59
|
+
metadata = source.to_polars(lazy=isinstance(lf, _polars.LazyFrame))
|
|
60
|
+
metadata = metadata.rename({"fname": "_fname"})
|
|
61
|
+
lf = lf.join(metadata, on="_fname")
|
|
62
|
+
|
|
63
|
+
return lf.drop("_fname")
|
|
64
|
+
|
|
65
|
+
return wrapper
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def read_csv(source, *args, **kwargs):
|
|
25
69
|
"""
|
|
26
70
|
Read the paths in the collection as CSV files, and return a
|
|
27
71
|
:obj:`polars.DataFrame` along with the metadata captured from the path
|
|
@@ -29,8 +73,8 @@ def read_csv(expanded_fpath, *args, **kwargs):
|
|
|
29
73
|
|
|
30
74
|
Parameters
|
|
31
75
|
----------
|
|
32
|
-
|
|
33
|
-
An expanded f-string path.
|
|
76
|
+
source : :obj:`fpathlib.ExpandedFPath`, or any type accepted by :obj:`polars.read_csv`
|
|
77
|
+
An expanded f-string path, or a plain path/glob.
|
|
34
78
|
*args
|
|
35
79
|
Positional arguments to pass to :meth:`polars.read_csv`.
|
|
36
80
|
**kwargs
|
|
@@ -41,13 +85,12 @@ def read_csv(expanded_fpath, *args, **kwargs):
|
|
|
41
85
|
:obj:`polars.DataFrame`
|
|
42
86
|
"""
|
|
43
87
|
|
|
44
|
-
return scan_csv
|
|
88
|
+
return scan_csv(source, *args, **kwargs).collect()
|
|
45
89
|
|
|
46
90
|
|
|
47
|
-
@expand_fpath_decorator(post_process=join_metadata)
|
|
48
91
|
def read_txt(
|
|
49
|
-
|
|
50
|
-
|
|
92
|
+
source,
|
|
93
|
+
line_filter=None,
|
|
51
94
|
separator=None,
|
|
52
95
|
new_columns=None,
|
|
53
96
|
has_header=False,
|
|
@@ -63,10 +106,13 @@ def read_txt(
|
|
|
63
106
|
|
|
64
107
|
Parameters
|
|
65
108
|
----------
|
|
66
|
-
|
|
67
|
-
An expanded f-string path.
|
|
68
|
-
|
|
69
|
-
|
|
109
|
+
source : :obj:`fpathlib.ExpandedFPath`, or any type accepted by :obj:`polars.scan_csv`
|
|
110
|
+
An expanded f-string path, or a plain path/glob.
|
|
111
|
+
line_filter : :obj:`callable`, optional
|
|
112
|
+
A function that takes a :obj:`polars.Expr` for the line's text and
|
|
113
|
+
returns a boolean :obj:`polars.Expr`, used to filter lines before
|
|
114
|
+
splitting by the separator. E.g.
|
|
115
|
+
`lambda line: line.str.starts_with("#").not_()`.
|
|
70
116
|
separator : :obj:`str`, optional
|
|
71
117
|
Deliminatorg to split each line into fields.
|
|
72
118
|
new_columns : :obj:`list`[:obj:`str`], optional
|
|
@@ -86,9 +132,9 @@ def read_txt(
|
|
|
86
132
|
:obj:`polars.DataFrame`
|
|
87
133
|
"""
|
|
88
134
|
|
|
89
|
-
return scan_txt
|
|
90
|
-
|
|
91
|
-
|
|
135
|
+
return scan_txt(
|
|
136
|
+
source,
|
|
137
|
+
line_filter=line_filter,
|
|
92
138
|
separator=separator,
|
|
93
139
|
new_columns=new_columns,
|
|
94
140
|
has_header=has_header,
|
|
@@ -97,8 +143,9 @@ def read_txt(
|
|
|
97
143
|
).collect()
|
|
98
144
|
|
|
99
145
|
|
|
100
|
-
@
|
|
101
|
-
|
|
146
|
+
@expand_arg
|
|
147
|
+
@join_metadata
|
|
148
|
+
def scan_csv(source, include_file_paths=None, *args, **kwargs):
|
|
102
149
|
"""
|
|
103
150
|
Scan the paths in the collection as CSV files, and return a
|
|
104
151
|
:obj:`polars.LazyFrame` along with the metadata captured from the path
|
|
@@ -106,8 +153,12 @@ def scan_csv(expanded_fpath, *args, **kwargs):
|
|
|
106
153
|
|
|
107
154
|
Parameters
|
|
108
155
|
----------
|
|
109
|
-
|
|
110
|
-
An expanded f-string path.
|
|
156
|
+
source : :obj:`fpathlib.ExpandedFPath`, or any type accepted by :obj:`polars.scan_csv`
|
|
157
|
+
An expanded f-string path, or a plain path/glob.
|
|
158
|
+
include_file_paths : :obj:`str`, optional
|
|
159
|
+
Name to give a column of each row's source file path in the
|
|
160
|
+
output. The file path is always used internally to join captured
|
|
161
|
+
metadata; without this, it isn't kept in the result. (Default: None)
|
|
111
162
|
*args
|
|
112
163
|
Positional arguments to pass to :meth:`polars.scan_csv`.
|
|
113
164
|
**kwargs
|
|
@@ -119,17 +170,19 @@ def scan_csv(expanded_fpath, *args, **kwargs):
|
|
|
119
170
|
"""
|
|
120
171
|
|
|
121
172
|
lf = _polars.scan_csv(
|
|
122
|
-
|
|
123
|
-
include_file_paths="
|
|
173
|
+
source,
|
|
174
|
+
include_file_paths="_fname",
|
|
124
175
|
*args,
|
|
125
176
|
**kwargs,
|
|
126
177
|
)
|
|
127
|
-
|
|
178
|
+
if include_file_paths is not None:
|
|
179
|
+
lf = lf.with_columns(_polars.col("_fname").alias(include_file_paths))
|
|
128
180
|
return lf
|
|
129
181
|
|
|
130
182
|
|
|
131
|
-
@
|
|
132
|
-
|
|
183
|
+
@expand_arg
|
|
184
|
+
@join_metadata
|
|
185
|
+
def scan_parquet(source, include_file_paths=None, *args, **kwargs):
|
|
133
186
|
"""
|
|
134
187
|
Scan the paths in the collection as a parquet file, and return a
|
|
135
188
|
:obj:`polars.LazyFrame` along with the metadata captured from the path
|
|
@@ -137,8 +190,12 @@ def scan_parquet(expanded_fpath, *args, **kwargs):
|
|
|
137
190
|
|
|
138
191
|
Parameters
|
|
139
192
|
----------
|
|
140
|
-
|
|
141
|
-
An expanded f-string path.
|
|
193
|
+
source : :obj:`fpathlib.ExpandedFPath`, or any type accepted by :obj:`polars.scan_parquet`
|
|
194
|
+
An expanded f-string path, or a plain path/glob.
|
|
195
|
+
include_file_paths : :obj:`str`, optional
|
|
196
|
+
Name to give a column of each row's source file path in the
|
|
197
|
+
output. The file path is always used internally to join captured
|
|
198
|
+
metadata; without this, it isn't kept in the result. (Default: None)
|
|
142
199
|
*args
|
|
143
200
|
Positional arguments to pass to :meth:`polars.scan_parquet`.
|
|
144
201
|
**kwargs
|
|
@@ -150,24 +207,25 @@ def scan_parquet(expanded_fpath, *args, **kwargs):
|
|
|
150
207
|
"""
|
|
151
208
|
|
|
152
209
|
lf = _polars.scan_parquet(
|
|
153
|
-
|
|
154
|
-
include_file_paths="
|
|
210
|
+
source,
|
|
211
|
+
include_file_paths="_fname",
|
|
155
212
|
*args,
|
|
156
213
|
**kwargs,
|
|
157
214
|
)
|
|
158
|
-
|
|
215
|
+
if include_file_paths is not None:
|
|
216
|
+
lf = lf.with_columns(_polars.col("_fname").alias(include_file_paths))
|
|
159
217
|
return lf
|
|
160
218
|
|
|
161
219
|
|
|
162
|
-
|
|
163
|
-
@expand_fpath_decorator(require_expandable=False, post_process=join_metadata)
|
|
220
|
+
@expand_arg
|
|
164
221
|
def scan_txt(
|
|
165
|
-
|
|
166
|
-
|
|
222
|
+
source,
|
|
223
|
+
line_filter=None,
|
|
167
224
|
separator=None,
|
|
168
225
|
new_columns=None,
|
|
169
226
|
has_header=False,
|
|
170
|
-
|
|
227
|
+
include_line=None,
|
|
228
|
+
include_file_paths=None,
|
|
171
229
|
usecols=None,
|
|
172
230
|
validate_schema=True,
|
|
173
231
|
*args,
|
|
@@ -182,10 +240,13 @@ def scan_txt(
|
|
|
182
240
|
|
|
183
241
|
Parameters
|
|
184
242
|
----------
|
|
185
|
-
|
|
186
|
-
An expanded f-string path.
|
|
187
|
-
|
|
188
|
-
|
|
243
|
+
source : :obj:`fpathlib.ExpandedFPath`, or any type accepted by :obj:`polars.scan_csv`
|
|
244
|
+
An expanded f-string path, or a plain path/glob.
|
|
245
|
+
line_filter : :obj:`callable`, optional
|
|
246
|
+
A function that takes a :obj:`polars.Expr` for the line's text and
|
|
247
|
+
returns a boolean :obj:`polars.Expr`, used to filter lines before
|
|
248
|
+
splitting by the separator. E.g.
|
|
249
|
+
`lambda line: line.str.starts_with("#").not_()`.
|
|
189
250
|
separator : :obj:`str`, optional
|
|
190
251
|
Deliminatorg to split each line into fields.
|
|
191
252
|
new_columns : :obj:`list`[:obj:`str`], optional
|
|
@@ -195,8 +256,13 @@ def scan_txt(
|
|
|
195
256
|
Whether the text files have a header line that should be skipped. The header
|
|
196
257
|
must have the same delimiter as the separator provided in `separator`.
|
|
197
258
|
(Default: False)
|
|
198
|
-
|
|
199
|
-
|
|
259
|
+
include_line : :obj:`str`, optional
|
|
260
|
+
Name to give a column of each row's original, unsplit line text in
|
|
261
|
+
the output. Without this, it isn't kept in the result. (Default: None)
|
|
262
|
+
include_file_paths : :obj:`str`, optional
|
|
263
|
+
Name to give a column of each row's source file path in the
|
|
264
|
+
output. The file path is always used internally to join captured
|
|
265
|
+
metadata; without this, it isn't kept in the result. (Default: None)
|
|
200
266
|
usecols : :obj:`list`[:obj:`int`], optional
|
|
201
267
|
Indexes of columns to keep in the output. If not provided, all columns are kept. Only applicable if `separator` is provided.
|
|
202
268
|
validate_schema : :obj:`bool`
|
|
@@ -220,52 +286,62 @@ def scan_txt(
|
|
|
220
286
|
:obj:`polars.LazyFrame`
|
|
221
287
|
"""
|
|
222
288
|
|
|
223
|
-
# TODO there are forbidden variables that should not be in
|
|
224
|
-
# such as '
|
|
289
|
+
# TODO there are forbidden variables that should not be in source
|
|
290
|
+
# such as '_line' and 'fields' and '_fname'
|
|
225
291
|
|
|
226
292
|
# TODO schema and schema_overrides is probably broken
|
|
227
293
|
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
294
|
+
# Delegates to scan_csv (rather than calling _polars.scan_csv and
|
|
295
|
+
# doing the metadata-join/include_file_paths handling directly) so
|
|
296
|
+
# that handling -- including the recursive single-file sample call
|
|
297
|
+
# below, since source[0] is a plain Path, not an ExpandedFPath --
|
|
298
|
+
# isn't duplicated here. scan_csv's own @expand_arg is a
|
|
299
|
+
# no-op on `source` at this point: it's already been resolved by
|
|
300
|
+
# scan_txt's decorator, so scan_csv just sees an ExpandedFPath or an
|
|
301
|
+
# already-unexpandable literal/glob, either way with nothing left to
|
|
302
|
+
# expand.
|
|
303
|
+
lf = scan_csv(
|
|
304
|
+
source,
|
|
305
|
+
include_file_paths=include_file_paths,
|
|
231
306
|
separator="\n",
|
|
232
|
-
new_columns=["
|
|
307
|
+
new_columns=["_line"],
|
|
233
308
|
has_header=False,
|
|
234
309
|
**kwargs,
|
|
235
310
|
)
|
|
236
311
|
|
|
237
312
|
# Can filter lines before doing any further processing
|
|
238
313
|
# This could be to remove lines with comments, etc.
|
|
239
|
-
if
|
|
240
|
-
lf = lf.filter(
|
|
314
|
+
if line_filter is not None:
|
|
315
|
+
lf = lf.filter(line_filter(_polars.col("_line")))
|
|
316
|
+
|
|
317
|
+
if include_line is not None:
|
|
318
|
+
lf = lf.with_columns(_polars.col("_line").alias(include_line))
|
|
241
319
|
|
|
242
320
|
# Separate lines into fields using `separator`
|
|
243
321
|
if separator is not None:
|
|
244
322
|
# Separate line into fields by separator
|
|
245
323
|
lf = lf.with_columns(
|
|
246
|
-
_polars.col("
|
|
324
|
+
_polars.col("_line").str.split(separator, literal=False).alias("fields")
|
|
247
325
|
)
|
|
248
|
-
|
|
249
|
-
if not keep_line:
|
|
250
|
-
lf = lf.drop("line")
|
|
326
|
+
lf = lf.drop("_line")
|
|
251
327
|
|
|
252
328
|
# With many matched files, inferring the field count/dtypes directly
|
|
253
329
|
# against the full glob is extremely slow (every file has to be opened
|
|
254
330
|
# before a `.head()` takes effect). Instead, recurse on a single
|
|
255
|
-
# representative file --
|
|
331
|
+
# representative file -- source[0] -- and reuse its already-cheap
|
|
256
332
|
# (single-file) inference below instead of duplicating it here. Computed
|
|
257
333
|
# once here since it's needed by both the field-count branch below and
|
|
258
334
|
# the dtype-inference branch further down.
|
|
259
335
|
infer_schema = kwargs.get("infer_schema", True)
|
|
260
336
|
sample_schema = None
|
|
261
337
|
if (
|
|
262
|
-
isinstance(
|
|
263
|
-
and len(
|
|
338
|
+
isinstance(source, ExpandedFPath)
|
|
339
|
+
and len(source) > 1
|
|
264
340
|
and (usecols is None or infer_schema)
|
|
265
341
|
):
|
|
266
342
|
sample_schema = scan_txt(
|
|
267
|
-
|
|
268
|
-
|
|
343
|
+
source[0],
|
|
344
|
+
line_filter=line_filter,
|
|
269
345
|
separator=separator,
|
|
270
346
|
new_columns=new_columns,
|
|
271
347
|
has_header=has_header,
|
|
@@ -282,7 +358,10 @@ def scan_txt(
|
|
|
282
358
|
|
|
283
359
|
else:
|
|
284
360
|
if sample_schema is not None:
|
|
285
|
-
|
|
361
|
+
# The recursive sample call above is always non-expandable
|
|
362
|
+
# (source[0] is a plain Path), so it already drops "_fname"
|
|
363
|
+
# itself -- no adjustment needed here.
|
|
364
|
+
n_fields = len(sample_schema)
|
|
286
365
|
else:
|
|
287
366
|
n_fields = (
|
|
288
367
|
lf.head(1)
|
|
@@ -339,11 +418,9 @@ def scan_txt(
|
|
|
339
418
|
# Infer dtypes?
|
|
340
419
|
if infer_schema:
|
|
341
420
|
if sample_schema is not None:
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
if name != "fname"
|
|
346
|
-
}
|
|
421
|
+
# The recursive sample call already drops "_fname" (see
|
|
422
|
+
# above), so no need to filter it back out here.
|
|
423
|
+
inferred_schema = dict(sample_schema.items())
|
|
347
424
|
else:
|
|
348
425
|
sample = (
|
|
349
426
|
lf.head(kwargs.get("infer_schema_length", 100))
|
|
@@ -354,4 +431,17 @@ def scan_txt(
|
|
|
354
431
|
inferred_schema = _polars.read_csv(sample).schema
|
|
355
432
|
lf = lf.cast(inferred_schema)
|
|
356
433
|
|
|
434
|
+
elif include_line is None:
|
|
435
|
+
# No separator means each row is just the raw line -- that's the
|
|
436
|
+
# whole point of this mode, so it has to stay visible somehow.
|
|
437
|
+
# Fall back to the traditional "line" name rather than leaking the
|
|
438
|
+
# internal "_line" name, since the caller didn't ask for a specific
|
|
439
|
+
# one via include_line.
|
|
440
|
+
lf = lf.rename({"_line": "line"})
|
|
441
|
+
else:
|
|
442
|
+
# include_line was given, so the alias was already created above
|
|
443
|
+
# (before this if/elif/else) -- drop the internal "_line" itself so
|
|
444
|
+
# it doesn't also leak into the output alongside it.
|
|
445
|
+
lf = lf.drop("_line")
|
|
446
|
+
|
|
357
447
|
return lf
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: fpathlib
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.6
|
|
4
4
|
Summary: A package to combine paths with metadata
|
|
5
5
|
Author-email: "C. Lockhart" <clockha2@gmu.edu>
|
|
6
6
|
Requires-Python: >=3.12
|
|
@@ -25,10 +25,10 @@ does this with an f-string-like path pattern -- `FPath` -- whose
|
|
|
25
25
|
`{variable}` fields are captured out of every matching path on disk.
|
|
26
26
|
|
|
27
27
|
```python
|
|
28
|
-
from fpathlib import
|
|
28
|
+
from fpathlib import expand
|
|
29
29
|
|
|
30
30
|
# Given files like data/tr1/output/0/job2.log, data/tr2/output/1/job0.log, ...
|
|
31
|
-
expanded =
|
|
31
|
+
expanded = expand("data/tr{trajectory:d}/output/{replica:d}/job{job:d}.log")
|
|
32
32
|
|
|
33
33
|
for path in expanded:
|
|
34
34
|
print(path, path.metadata)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|