mango-metadata-from-tables 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mango_metadata_from_tables-1.0.0/PKG-INFO +143 -0
- mango_metadata_from_tables-1.0.0/README.md +124 -0
- mango_metadata_from_tables-1.0.0/pyproject.toml +44 -0
- mango_metadata_from_tables-1.0.0/src/mango_metadata_from_tables/__init__.py +19 -0
- mango_metadata_from_tables-1.0.0/src/mango_metadata_from_tables/__main__.py +4 -0
- mango_metadata_from_tables-1.0.0/src/mango_metadata_from_tables/cli.py +20 -0
- mango_metadata_from_tables-1.0.0/src/mango_metadata_from_tables/create_config.py +149 -0
- mango_metadata_from_tables-1.0.0/src/mango_metadata_from_tables/dataframe2avus.py +119 -0
- mango_metadata_from_tables-1.0.0/src/mango_metadata_from_tables/preprocessing.py +263 -0
- mango_metadata_from_tables-1.0.0/src/mango_metadata_from_tables/prompts.py +289 -0
- mango_metadata_from_tables-1.0.0/src/mango_metadata_from_tables/read_table.py +65 -0
- mango_metadata_from_tables-1.0.0/src/mango_metadata_from_tables/run.py +113 -0
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: mango-metadata-from-tables
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Use this package to process tabular files in which each row represents an iRODS data object or collection and each column contains either an identifier or metadata to add to this item. It supports plain text files and Excel files, which could be stored locally or in iRODS itself.
|
|
5
|
+
Author: Mariana Montes, Jef Scheepers
|
|
6
|
+
Requires-Dist: click>=8.3.2
|
|
7
|
+
Requires-Dist: jinja2>=3.1.6
|
|
8
|
+
Requires-Dist: mango-mdschema==1.1.0
|
|
9
|
+
Requires-Dist: odfpy>=1.4.1
|
|
10
|
+
Requires-Dist: openpyxl==3.1.5
|
|
11
|
+
Requires-Dist: pandas>=2.2.3
|
|
12
|
+
Requires-Dist: pygments>=2.20.0
|
|
13
|
+
Requires-Dist: python-irodsclient>=3.2.0
|
|
14
|
+
Requires-Dist: pyyaml==6.0.2
|
|
15
|
+
Requires-Dist: rich==13.9.4
|
|
16
|
+
Requires-Python: >=3.10
|
|
17
|
+
Project-URL: repository, https://github.com/kuleuven/mango-metadata-from-tables
|
|
18
|
+
Description-Content-Type: text/markdown
|
|
19
|
+
|
|
20
|
+
# Python package to extract metadata from tables
|
|
21
|
+
|
|
22
|
+
Use this package to process tabular files in which each row represents an iRODS data object or collection
|
|
23
|
+
and each column contains either an identifier or metadata to add to this item.
|
|
24
|
+
It supports plain text files and Excel files, which could be stored locally or in iRODS itself.
|
|
25
|
+
|
|
26
|
+
To get started, create a virtual environment with pip and install the dependencies described in the [requirements file](./requirements.txt):
|
|
27
|
+
|
|
28
|
+
```sh
|
|
29
|
+
python -m venv venv
|
|
30
|
+
source venv/bin/activate
|
|
31
|
+
pip install -e .
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
Now, you can run the script with the command `mango-metadata-from-tables`.
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
## Usage
|
|
38
|
+
|
|
39
|
+
This module can run on the command line with two commands: `setup` and `run`.
|
|
40
|
+
|
|
41
|
+
The `setup` command takes as arguments the path to a tabular file (local or in iRODS) and
|
|
42
|
+
the desired path for the output YAML, asks the user questions about how to
|
|
43
|
+
parse the tabular file, and outputs a configuration file.
|
|
44
|
+
|
|
45
|
+
This configuration file can then be provided as the `--config` option to the `run`
|
|
46
|
+
command in order to standardize tabular files and properly obtain paths to data objects
|
|
47
|
+
and attach metadata to them based on the columns of these files.
|
|
48
|
+
|
|
49
|
+
**Note:** Empty values in the table are ignored, and will not be added as metadata.
|
|
50
|
+
In some cases, users may find it meaningful to include the absence of a value as contextual information.
|
|
51
|
+
In those cases, we advice to use a value like "Unknown", "Not applicable" or "NA" in your table instead.
|
|
52
|
+
|
|
53
|
+
### Examples
|
|
54
|
+
|
|
55
|
+
### A small csv file
|
|
56
|
+
|
|
57
|
+
The following file simulates having [a small semicolon-separated file](./tests/testdata/testdata.csv)
|
|
58
|
+
with absolute paths in a "dataobject" column and a few columns with metadata.
|
|
59
|
+
|
|
60
|
+
First, with the `setup` command, we answer a few questions on how to parse the tabular file
|
|
61
|
+
and create a "test-config.yaml" configuration file that keeps track of the answers.
|
|
62
|
+
|
|
63
|
+
Then, with the `run` command, we use the information on the configuration YAML file to parse
|
|
64
|
+
the tabular file and, because it's just a "dry run", we simulate adding the metadata to each
|
|
65
|
+
data object or collection. Note that this `run` command could then also be used on other tabular files
|
|
66
|
+
with the same properties as the original one.
|
|
67
|
+
|
|
68
|
+
```sh
|
|
69
|
+
mango-metadata-from-tables setup testdata/testdata.csv test-config.yaml --sep ";"
|
|
70
|
+
mango-metadata-from-tables run testdata/testdata.csv --config test-config.yaml --dry-run
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
### A larger Excel file with multiple sheets
|
|
74
|
+
|
|
75
|
+
In this second example the file is an [Excel file with multiple sheets](./tests/testdata/bigger-testdata.xlsx),
|
|
76
|
+
including one that has no relevant metadata. Again, with the `setup` command we indicate
|
|
77
|
+
how the Excel should be parsed and record the answers in a YAML configuration file.
|
|
78
|
+
Then, with the `run` command we parse the Excel and simulate adding the metadata.
|
|
79
|
+
|
|
80
|
+
```sh
|
|
81
|
+
mango-metadata-from-tables setup testdata/bigger-testdata.xlsx bigger-test-config.yaml
|
|
82
|
+
mango-metadata-from-tables run testdata/bigger-testdata.xlsx --config bigger-test-config.yaml --dry-run
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
## `setup`
|
|
86
|
+
|
|
87
|
+
The configuration file can be created as follows:
|
|
88
|
+
|
|
89
|
+
```sh
|
|
90
|
+
mango-metadata-from-tables setup filename output_path
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
In this case `filename` is the path to a tabular file (csv, tsv, Excel...),
|
|
94
|
+
stored either locally or in iRODS. If it lives in iRODS, the `--irods` flag should be used,
|
|
95
|
+
so that an iRODS session is started:
|
|
96
|
+
|
|
97
|
+
```sh
|
|
98
|
+
mango-metadata-from-tables setup /zone/home/project/path/to/tabular output_path --irods
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
If the tabular file is a plain text file, it is possible to specify a column separator
|
|
102
|
+
with the `--sep` option, which has "," as a default. If a wrong separator is provided and
|
|
103
|
+
the parser finds a single column, it will warn you and give you the possibility to correct it.
|
|
104
|
+
|
|
105
|
+
```sh
|
|
106
|
+
mango-metadata-from-tables setup testdata/testdata.csv test-output.yml --sep ";"
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
If the file can be found and opened as a dataframe, the user will be prompted with questions
|
|
110
|
+
that will later guide preprocessing of equivalent tabular files:
|
|
111
|
+
|
|
112
|
+
- If there are multiple sheets in an Excel file, which one(s) should be used?
|
|
113
|
+
- Which of the columns contains a unique identifier of the data objects that metadata has to be attached to?
|
|
114
|
+
- If the unique identifier is not an absolute path, is it a relative path or part of filename?
|
|
115
|
+
And if so, within which collection should the data objects be found?
|
|
116
|
+
- Should any columns be whitelisted or blacklisted?
|
|
117
|
+
|
|
118
|
+
The final YAML will be printed on the console and saved as a file locally.
|
|
119
|
+
|
|
120
|
+
## `run`
|
|
121
|
+
|
|
122
|
+
Given a path to a tabular file with metadata and a YAML with the settings to preprocess it,
|
|
123
|
+
metadata can be added with the `run` command:
|
|
124
|
+
|
|
125
|
+
```sh
|
|
126
|
+
mango-metadata-from-tables run path_to_tabular --config path/to/config.yml
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
For testing purposes, it is possible to use
|
|
130
|
+
the `--dry-run` flag, which simulates the preprocessing and identification of metadata and
|
|
131
|
+
prints a small report at the end.
|
|
132
|
+
An iRODS session will be initiated always, so **make sure you have a valid active iRODS Session**.
|
|
133
|
+
For the testdata, that is the icts zone in quality.
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
```sh
|
|
137
|
+
mango-metadata-from-tables run path_to_tabular --config path/to/config.yml --dry-run
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
It is not necessary to rerun both `setup` and `run` for each tabular file:
|
|
141
|
+
if you have several tabular files with the same properties, and that thus can be described by the same
|
|
142
|
+
YAML configuration file, you just need to run `setup` with one of them, and then
|
|
143
|
+
`run` with each of the tabular files and the same configuration file.
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
# Python package to extract metadata from tables
|
|
2
|
+
|
|
3
|
+
Use this package to process tabular files in which each row represents an iRODS data object or collection
|
|
4
|
+
and each column contains either an identifier or metadata to add to this item.
|
|
5
|
+
It supports plain text files and Excel files, which could be stored locally or in iRODS itself.
|
|
6
|
+
|
|
7
|
+
To get started, create a virtual environment with pip and install the dependencies described in the [requirements file](./requirements.txt):
|
|
8
|
+
|
|
9
|
+
```sh
|
|
10
|
+
python -m venv venv
|
|
11
|
+
source venv/bin/activate
|
|
12
|
+
pip install -e .
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
Now, you can run the script with the command `mango-metadata-from-tables`.
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
## Usage
|
|
19
|
+
|
|
20
|
+
This module can run on the command line with two commands: `setup` and `run`.
|
|
21
|
+
|
|
22
|
+
The `setup` command takes as arguments the path to a tabular file (local or in iRODS) and
|
|
23
|
+
the desired path for the output YAML, asks the user questions about how to
|
|
24
|
+
parse the tabular file, and outputs a configuration file.
|
|
25
|
+
|
|
26
|
+
This configuration file can then be provided as the `--config` option to the `run`
|
|
27
|
+
command in order to standardize tabular files and properly obtain paths to data objects
|
|
28
|
+
and attach metadata to them based on the columns of these files.
|
|
29
|
+
|
|
30
|
+
**Note:** Empty values in the table are ignored, and will not be added as metadata.
|
|
31
|
+
In some cases, users may find it meaningful to include the absence of a value as contextual information.
|
|
32
|
+
In those cases, we advice to use a value like "Unknown", "Not applicable" or "NA" in your table instead.
|
|
33
|
+
|
|
34
|
+
### Examples
|
|
35
|
+
|
|
36
|
+
### A small csv file
|
|
37
|
+
|
|
38
|
+
The following file simulates having [a small semicolon-separated file](./tests/testdata/testdata.csv)
|
|
39
|
+
with absolute paths in a "dataobject" column and a few columns with metadata.
|
|
40
|
+
|
|
41
|
+
First, with the `setup` command, we answer a few questions on how to parse the tabular file
|
|
42
|
+
and create a "test-config.yaml" configuration file that keeps track of the answers.
|
|
43
|
+
|
|
44
|
+
Then, with the `run` command, we use the information on the configuration YAML file to parse
|
|
45
|
+
the tabular file and, because it's just a "dry run", we simulate adding the metadata to each
|
|
46
|
+
data object or collection. Note that this `run` command could then also be used on other tabular files
|
|
47
|
+
with the same properties as the original one.
|
|
48
|
+
|
|
49
|
+
```sh
|
|
50
|
+
mango-metadata-from-tables setup testdata/testdata.csv test-config.yaml --sep ";"
|
|
51
|
+
mango-metadata-from-tables run testdata/testdata.csv --config test-config.yaml --dry-run
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
### A larger Excel file with multiple sheets
|
|
55
|
+
|
|
56
|
+
In this second example the file is an [Excel file with multiple sheets](./tests/testdata/bigger-testdata.xlsx),
|
|
57
|
+
including one that has no relevant metadata. Again, with the `setup` command we indicate
|
|
58
|
+
how the Excel should be parsed and record the answers in a YAML configuration file.
|
|
59
|
+
Then, with the `run` command we parse the Excel and simulate adding the metadata.
|
|
60
|
+
|
|
61
|
+
```sh
|
|
62
|
+
mango-metadata-from-tables setup testdata/bigger-testdata.xlsx bigger-test-config.yaml
|
|
63
|
+
mango-metadata-from-tables run testdata/bigger-testdata.xlsx --config bigger-test-config.yaml --dry-run
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
## `setup`
|
|
67
|
+
|
|
68
|
+
The configuration file can be created as follows:
|
|
69
|
+
|
|
70
|
+
```sh
|
|
71
|
+
mango-metadata-from-tables setup filename output_path
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
In this case `filename` is the path to a tabular file (csv, tsv, Excel...),
|
|
75
|
+
stored either locally or in iRODS. If it lives in iRODS, the `--irods` flag should be used,
|
|
76
|
+
so that an iRODS session is started:
|
|
77
|
+
|
|
78
|
+
```sh
|
|
79
|
+
mango-metadata-from-tables setup /zone/home/project/path/to/tabular output_path --irods
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
If the tabular file is a plain text file, it is possible to specify a column separator
|
|
83
|
+
with the `--sep` option, which has "," as a default. If a wrong separator is provided and
|
|
84
|
+
the parser finds a single column, it will warn you and give you the possibility to correct it.
|
|
85
|
+
|
|
86
|
+
```sh
|
|
87
|
+
mango-metadata-from-tables setup testdata/testdata.csv test-output.yml --sep ";"
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
If the file can be found and opened as a dataframe, the user will be prompted with questions
|
|
91
|
+
that will later guide preprocessing of equivalent tabular files:
|
|
92
|
+
|
|
93
|
+
- If there are multiple sheets in an Excel file, which one(s) should be used?
|
|
94
|
+
- Which of the columns contains a unique identifier of the data objects that metadata has to be attached to?
|
|
95
|
+
- If the unique identifier is not an absolute path, is it a relative path or part of filename?
|
|
96
|
+
And if so, within which collection should the data objects be found?
|
|
97
|
+
- Should any columns be whitelisted or blacklisted?
|
|
98
|
+
|
|
99
|
+
The final YAML will be printed on the console and saved as a file locally.
|
|
100
|
+
|
|
101
|
+
## `run`
|
|
102
|
+
|
|
103
|
+
Given a path to a tabular file with metadata and a YAML with the settings to preprocess it,
|
|
104
|
+
metadata can be added with the `run` command:
|
|
105
|
+
|
|
106
|
+
```sh
|
|
107
|
+
mango-metadata-from-tables run path_to_tabular --config path/to/config.yml
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
For testing purposes, it is possible to use
|
|
111
|
+
the `--dry-run` flag, which simulates the preprocessing and identification of metadata and
|
|
112
|
+
prints a small report at the end.
|
|
113
|
+
An iRODS session will be initiated always, so **make sure you have a valid active iRODS Session**.
|
|
114
|
+
For the testdata, that is the icts zone in quality.
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
```sh
|
|
118
|
+
mango-metadata-from-tables run path_to_tabular --config path/to/config.yml --dry-run
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
It is not necessary to rerun both `setup` and `run` for each tabular file:
|
|
122
|
+
if you have several tabular files with the same properties, and that thus can be described by the same
|
|
123
|
+
YAML configuration file, you just need to run `setup` with one of them, and then
|
|
124
|
+
`run` with each of the tabular files and the same configuration file.
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "mango-metadata-from-tables"
|
|
3
|
+
version = "1.0.0"
|
|
4
|
+
description = "Use this package to process tabular files in which each row represents an iRODS data object or collection and each column contains either an identifier or metadata to add to this item. It supports plain text files and Excel files, which could be stored locally or in iRODS itself."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
authors = [
|
|
7
|
+
{ name = "Mariana Montes" },
|
|
8
|
+
{ name = "Jef Scheepers" }
|
|
9
|
+
]
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
dependencies = [
|
|
12
|
+
"click>=8.3.2",
|
|
13
|
+
"jinja2>=3.1.6",
|
|
14
|
+
"mango-mdschema==1.1.0",
|
|
15
|
+
"odfpy>=1.4.1",
|
|
16
|
+
"openpyxl==3.1.5",
|
|
17
|
+
"pandas>=2.2.3",
|
|
18
|
+
"pygments>=2.20.0",
|
|
19
|
+
"python-irodsclient>=3.2.0",
|
|
20
|
+
"pyyaml==6.0.2",
|
|
21
|
+
"rich==13.9.4",
|
|
22
|
+
]
|
|
23
|
+
|
|
24
|
+
[project.scripts]
|
|
25
|
+
mango-metadata-from-tables = "mango_metadata_from_tables.cli:mdtab"
|
|
26
|
+
|
|
27
|
+
[project.urls]
|
|
28
|
+
repository = "https://github.com/kuleuven/mango-metadata-from-tables"
|
|
29
|
+
|
|
30
|
+
[build-system]
|
|
31
|
+
requires = ["uv_build>=0.9.10,<0.10.0"]
|
|
32
|
+
build-backend = "uv_build"
|
|
33
|
+
|
|
34
|
+
[tool.pytest.ini_options]
|
|
35
|
+
pythonpath = ["src"]
|
|
36
|
+
|
|
37
|
+
[dependency-groups]
|
|
38
|
+
dev = [
|
|
39
|
+
"pylint>=4.0.8",
|
|
40
|
+
"pytest>=9.0.3,<9.1.0",
|
|
41
|
+
"pytest-cases>=3.9.1",
|
|
42
|
+
"pytest-cov>=7.1.0",
|
|
43
|
+
]
|
|
44
|
+
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
from enum import StrEnum
|
|
2
|
+
|
|
3
|
+
from rich.console import Console
|
|
4
|
+
|
|
5
|
+
DATAOBJECT = "dataobject"
|
|
6
|
+
EXCLUDE_NONSCHEMA_MD = "exclude_non_schema_metadata"
|
|
7
|
+
EXCLUDE_INVALID_SCHEMA_MD = "exclude_invalid_schema_metadata"
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class ItemType(StrEnum):
|
|
11
|
+
DATAOBJECT = "data object"
|
|
12
|
+
COLLECTION = "collection"
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
console = Console()
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def main() -> None:
|
|
19
|
+
print("Hello from mango-metadata-from-tables!")
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
import click
|
|
2
|
+
from .run import run
|
|
3
|
+
from .create_config import setup
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
# These are the functions that would be called from the command line :)
|
|
7
|
+
# (and use click)
|
|
8
|
+
@click.group()
|
|
9
|
+
def mdtab():
|
|
10
|
+
"""Process tabular files to add iRODS metadata to data objects."""
|
|
11
|
+
pass
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
mdtab.add_command(setup)
|
|
15
|
+
mdtab.add_command(run)
|
|
16
|
+
|
|
17
|
+
# endregion
|
|
18
|
+
|
|
19
|
+
if __name__ == "__main__":
|
|
20
|
+
mdtab()
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
import click
|
|
2
|
+
from rich.prompt import Prompt, Confirm
|
|
3
|
+
from rich.console import Group
|
|
4
|
+
from rich.syntax import Syntax
|
|
5
|
+
from .prompts import (
|
|
6
|
+
select_sheets,
|
|
7
|
+
classify_target_item_column,
|
|
8
|
+
filter_columns,
|
|
9
|
+
ask_multivalue_columns,
|
|
10
|
+
list_columns_with_character,
|
|
11
|
+
ask_about_schemas,
|
|
12
|
+
)
|
|
13
|
+
from rich.markdown import Markdown
|
|
14
|
+
from .preprocessing import get_sheets
|
|
15
|
+
from . import console
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@click.command()
|
|
19
|
+
@click.option("--sep", default=",", help="Separator for plain text files.")
|
|
20
|
+
@click.option(
|
|
21
|
+
"--irods/--no-irods", default=False, help="Whether an iRODS session is needed."
|
|
22
|
+
)
|
|
23
|
+
@click.argument("example")
|
|
24
|
+
@click.argument("output", type=click.File("w"))
|
|
25
|
+
def setup(example, output, sep=",", irods=False):
|
|
26
|
+
"""
|
|
27
|
+
Generate configuration file.
|
|
28
|
+
|
|
29
|
+
EXAMPLE is the path to the tabular file to parse.
|
|
30
|
+
|
|
31
|
+
OUTPUT is the path where the configuration file will be saved.
|
|
32
|
+
"""
|
|
33
|
+
import yaml
|
|
34
|
+
|
|
35
|
+
while True:
|
|
36
|
+
sheets = get_sheets(example, sep, irods)
|
|
37
|
+
if len(sheets) > 1:
|
|
38
|
+
break
|
|
39
|
+
column_names = list(sheets.values())[0].columns
|
|
40
|
+
if len(column_names) > 1:
|
|
41
|
+
break
|
|
42
|
+
update_separator = Confirm.ask(
|
|
43
|
+
f"Your sheet has only one column: `{column_names[0]}`, \
|
|
44
|
+
would you like to provide another separator?"
|
|
45
|
+
)
|
|
46
|
+
if update_separator:
|
|
47
|
+
sep = Prompt.ask("Which separator would you like to try now?") or " "
|
|
48
|
+
else:
|
|
49
|
+
break
|
|
50
|
+
|
|
51
|
+
# select which sheets to use, if there are more than one
|
|
52
|
+
selection_of_sheets = select_sheets(sheets)
|
|
53
|
+
sheets = {k: v for k, v in sheets.items() if k in selection_of_sheets}
|
|
54
|
+
|
|
55
|
+
# get info on the item column, if there is any
|
|
56
|
+
path_info = classify_target_item_column(sheets)
|
|
57
|
+
item_column = path_info["item_column"]
|
|
58
|
+
|
|
59
|
+
# if there is a dedicated item column,
|
|
60
|
+
# only keep the sheets that contain that column
|
|
61
|
+
if item_column:
|
|
62
|
+
sheets = {k: v for k, v in sheets.items() if item_column in v.columns}
|
|
63
|
+
|
|
64
|
+
# start config yaml with the info we have
|
|
65
|
+
for_yaml = {
|
|
66
|
+
"sheets": list(sheets.keys()),
|
|
67
|
+
"separator": sep,
|
|
68
|
+
"item_type": path_info["item_type"].name,
|
|
69
|
+
"path_column": {
|
|
70
|
+
"column_name": item_column,
|
|
71
|
+
"path_type": path_info["path_type"],
|
|
72
|
+
"pattern": path_info["pattern"],
|
|
73
|
+
"workdir": path_info["workdir"],
|
|
74
|
+
},
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
# ask if any columns need to be blacklisted OR whitelisted
|
|
78
|
+
all_column_names = list(
|
|
79
|
+
set(
|
|
80
|
+
col
|
|
81
|
+
for sheet in sheets.values()
|
|
82
|
+
for col in sheet.columns
|
|
83
|
+
if col != item_column
|
|
84
|
+
)
|
|
85
|
+
)
|
|
86
|
+
column_filter = filter_columns(all_column_names)
|
|
87
|
+
|
|
88
|
+
# calculating columns based on filter
|
|
89
|
+
if column_filter.get("whitelist", False):
|
|
90
|
+
filtered_columns = [
|
|
91
|
+
col for col in all_column_names if col in column_filter["whitelist"]
|
|
92
|
+
]
|
|
93
|
+
elif column_filter.get("blacklist", False):
|
|
94
|
+
filtered_columns = [
|
|
95
|
+
col for col in all_column_names if col not in column_filter["blacklist"]
|
|
96
|
+
]
|
|
97
|
+
else:
|
|
98
|
+
filtered_columns = all_column_names
|
|
99
|
+
|
|
100
|
+
# update yaml with column information
|
|
101
|
+
for_yaml.update(column_filter)
|
|
102
|
+
|
|
103
|
+
if Confirm.ask(
|
|
104
|
+
"Do your sheet(s) have columns which may contain multiple values per row?"
|
|
105
|
+
):
|
|
106
|
+
|
|
107
|
+
valid_separator_chosen = False
|
|
108
|
+
while not valid_separator_chosen:
|
|
109
|
+
multivalue_separator = Prompt.ask(
|
|
110
|
+
"What is the separator of your columns with multiple values?"
|
|
111
|
+
)
|
|
112
|
+
if multivalue_separator == sep:
|
|
113
|
+
print(
|
|
114
|
+
"This separator cannot be used, because it is used to separate your columns."
|
|
115
|
+
)
|
|
116
|
+
continue
|
|
117
|
+
columns_with_separator = list_columns_with_character(
|
|
118
|
+
sheets.values(), filtered_columns, multivalue_separator
|
|
119
|
+
)
|
|
120
|
+
if len(columns_with_separator) == 0:
|
|
121
|
+
print(
|
|
122
|
+
"This separator cannot be used, because it does not appear in your file."
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
else:
|
|
126
|
+
valid_separator_chosen = True
|
|
127
|
+
|
|
128
|
+
# ask for multivalue columns
|
|
129
|
+
multivalue_columns = ask_multivalue_columns(
|
|
130
|
+
list(col for col in columns_with_separator if col != item_column)
|
|
131
|
+
)
|
|
132
|
+
# update yaml with multivalue columns information
|
|
133
|
+
for_yaml["multivalue_separator"] = multivalue_separator
|
|
134
|
+
for_yaml["multivalue_columns"] = multivalue_columns
|
|
135
|
+
|
|
136
|
+
# ask about schema metadata
|
|
137
|
+
if mango_schema_info := ask_about_schemas():
|
|
138
|
+
for_yaml["mango_schema"] = mango_schema_info
|
|
139
|
+
|
|
140
|
+
# create yaml from the dictionary
|
|
141
|
+
yml = yaml.dump(for_yaml, default_flow_style=False, indent=2)
|
|
142
|
+
# Make a group and indicate where it is saved
|
|
143
|
+
panel_group = Group(
|
|
144
|
+
Markdown("# This is your config yaml"),
|
|
145
|
+
Syntax(yml, "yaml"),
|
|
146
|
+
Markdown(f"_It will be saved in `{output.name}`._"),
|
|
147
|
+
)
|
|
148
|
+
console.print(panel_group)
|
|
149
|
+
click.echo(yml, file=output)
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
import pandas as pd
|
|
2
|
+
from irods.session import iRODSSession
|
|
3
|
+
from mango_mdschema import Schema
|
|
4
|
+
from irods.meta import iRODSMeta, AVUOperation
|
|
5
|
+
from collections.abc import Generator
|
|
6
|
+
from . import console, DATAOBJECT, ItemType
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def unlist_value(value: list, field) -> str | int:
|
|
10
|
+
"""Unlist values for non repeatable fields"""
|
|
11
|
+
if len(value) == 1 and not field.repeatable:
|
|
12
|
+
return value[0]
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def dict_to_avus(
|
|
16
|
+
row: dict,
|
|
17
|
+
schema: Schema = None,
|
|
18
|
+
exclude_non_schema_metadata: bool = True,
|
|
19
|
+
exclude_invalid_schema_metadata: bool = False,
|
|
20
|
+
) -> list[iRODSMeta]:
|
|
21
|
+
"""Convert a dictionary of metadata name-value pairs into a list of iRODSMeta"""
|
|
22
|
+
if schema is not None:
|
|
23
|
+
dict_to_validate = {
|
|
24
|
+
k: unlist_value(v, schema.fields[k])
|
|
25
|
+
for k, v in row.items()
|
|
26
|
+
if k in schema.fields
|
|
27
|
+
}
|
|
28
|
+
valid_schema_metadata = schema.validate(
|
|
29
|
+
dict_to_validate
|
|
30
|
+
) # dict with metadata that passed the schema
|
|
31
|
+
for k, v in valid_schema_metadata.items():
|
|
32
|
+
if v is None:
|
|
33
|
+
line1 = f"Found a value '{dict_to_validate[k]}' of column '{k}' that does not match the schema."
|
|
34
|
+
line2 = (
|
|
35
|
+
"It will be excluded."
|
|
36
|
+
if exclude_invalid_schema_metadata
|
|
37
|
+
else "It will be added as non-schema metadata."
|
|
38
|
+
)
|
|
39
|
+
console.print(f"{line1} {line2}")
|
|
40
|
+
schema_avus = schema.to_avus(valid_schema_metadata)
|
|
41
|
+
if len(schema_avus) > 0:
|
|
42
|
+
schema_avus += [
|
|
43
|
+
iRODSMeta(f"{schema.prefix}.{schema.name}.__version__", schema.version)
|
|
44
|
+
]
|
|
45
|
+
|
|
46
|
+
# create empty dict if other metadata is ignored; otherwise dict of metadata that did not pass
|
|
47
|
+
def is_invalid_schema_metadata(k):
|
|
48
|
+
return k in dict_to_validate and valid_schema_metadata.get(k, None) is None
|
|
49
|
+
|
|
50
|
+
def is_nonschema_metadata(k):
|
|
51
|
+
return k not in schema.fields
|
|
52
|
+
|
|
53
|
+
invalid_schema_metadata = (
|
|
54
|
+
{}
|
|
55
|
+
if exclude_invalid_schema_metadata
|
|
56
|
+
else {k: v for k, v in row.items() if is_invalid_schema_metadata(k)}
|
|
57
|
+
)
|
|
58
|
+
nonschema_metadata = (
|
|
59
|
+
{}
|
|
60
|
+
if exclude_non_schema_metadata
|
|
61
|
+
else {k: v for k, v in row.items() if is_nonschema_metadata(k)}
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
other_metadata = {**nonschema_metadata, **invalid_schema_metadata}
|
|
65
|
+
else: # if there is no schema
|
|
66
|
+
schema_avus = [] # no schema metadata
|
|
67
|
+
other_metadata = row # all metadata
|
|
68
|
+
|
|
69
|
+
non_schema_avus = [
|
|
70
|
+
iRODSMeta(str(key), str(value_item))
|
|
71
|
+
for key, value in other_metadata.items() # empty if all metadata is from schema or the other metadata is ignored
|
|
72
|
+
for value_item in value
|
|
73
|
+
if not pd.isna(value_item)
|
|
74
|
+
]
|
|
75
|
+
return schema_avus + non_schema_avus
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def generate_rows(
|
|
79
|
+
dataframe: pd.DataFrame, multivalue_columns: list, multivalue_separator: str
|
|
80
|
+
) -> Generator[tuple]:
|
|
81
|
+
"""Yield a tuple of filename and metadata-dictionary from a dataframe"""
|
|
82
|
+
for _, row in dataframe.iterrows():
|
|
83
|
+
md_dict = {}
|
|
84
|
+
for k, v in row.items():
|
|
85
|
+
if k != DATAOBJECT:
|
|
86
|
+
if k in multivalue_columns and isinstance(v, str):
|
|
87
|
+
md_dict[k] = [
|
|
88
|
+
val.strip()
|
|
89
|
+
for val in v.split(multivalue_separator)
|
|
90
|
+
if val.strip()
|
|
91
|
+
]
|
|
92
|
+
else:
|
|
93
|
+
md_dict[k] = [v]
|
|
94
|
+
yield (row[DATAOBJECT], md_dict)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def apply_metadata_to_data_object(
|
|
98
|
+
path: str,
|
|
99
|
+
avu_dict: dict,
|
|
100
|
+
schema_instructions: dict,
|
|
101
|
+
session: iRODSSession,
|
|
102
|
+
item_type: ItemType = ItemType.DATAOBJECT,
|
|
103
|
+
):
|
|
104
|
+
"""Add metadata from a dictionary to a given data object or collection"""
|
|
105
|
+
manager = (
|
|
106
|
+
session.data_objects
|
|
107
|
+
if item_type == ItemType.DATAOBJECT
|
|
108
|
+
else session.collections
|
|
109
|
+
)
|
|
110
|
+
try:
|
|
111
|
+
obj = manager.get(path)
|
|
112
|
+
avus = dict_to_avus(avu_dict, **schema_instructions)
|
|
113
|
+
obj.metadata.apply_atomic_operations(
|
|
114
|
+
*[AVUOperation(operation="add", avu=item) for item in avus]
|
|
115
|
+
)
|
|
116
|
+
return avus
|
|
117
|
+
except Exception as e:
|
|
118
|
+
print(e)
|
|
119
|
+
return 0
|
|
@@ -0,0 +1,263 @@
|
|
|
1
|
+
import io
|
|
2
|
+
import os
|
|
3
|
+
import pandas as pd
|
|
4
|
+
import click
|
|
5
|
+
import jinja2
|
|
6
|
+
import datetime
|
|
7
|
+
import pathlib
|
|
8
|
+
from irods.session import iRODSSession
|
|
9
|
+
from irods.column import Criterion
|
|
10
|
+
from irods.models import Collection, DataObject
|
|
11
|
+
import yaml
|
|
12
|
+
from mango_mdschema.schema import Schema, get_mango_schema
|
|
13
|
+
from .read_table import parse_tabular_file
|
|
14
|
+
from . import (
|
|
15
|
+
DATAOBJECT,
|
|
16
|
+
EXCLUDE_NONSCHEMA_MD,
|
|
17
|
+
EXCLUDE_INVALID_SCHEMA_MD,
|
|
18
|
+
console,
|
|
19
|
+
ItemType,
|
|
20
|
+
)
|
|
21
|
+
from typing import Callable
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def search_objects_with_identifier(
|
|
25
|
+
session: iRODSSession, workingdirectory: str, identifier: str, exact_match: bool
|
|
26
|
+
):
|
|
27
|
+
"""Searches a given project for data objects starting with a certain identifier
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
Arguments
|
|
31
|
+
---------
|
|
32
|
+
session: obj
|
|
33
|
+
An iRODSSession object
|
|
34
|
+
|
|
35
|
+
workingdirectory: str
|
|
36
|
+
Path to the collection in iRODS
|
|
37
|
+
|
|
38
|
+
identifier: str
|
|
39
|
+
The identifier you want to search for
|
|
40
|
+
|
|
41
|
+
Returns
|
|
42
|
+
-------
|
|
43
|
+
paths: str
|
|
44
|
+
A list of data object paths matching the identifier
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
operator = "=" if exact_match else "like"
|
|
48
|
+
query = (
|
|
49
|
+
session.query(DataObject.name, Collection.name)
|
|
50
|
+
.filter(Criterion("like", Collection.name, workingdirectory + "%"))
|
|
51
|
+
.filter(Criterion(operator, DataObject.name, identifier + "%"))
|
|
52
|
+
)
|
|
53
|
+
paths = [f"{result[Collection.name]}/{result[DataObject.name]}" for result in query]
|
|
54
|
+
return paths
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def query_dataobjects_with_filename(
|
|
58
|
+
session, df, filename_column, workingdirectory, exact_match=True
|
|
59
|
+
):
|
|
60
|
+
"""
|
|
61
|
+
Queries data objects in iRODS based on identifiers in the dataframe,
|
|
62
|
+
and creates a row for each result with the accompanying metadata.
|
|
63
|
+
"""
|
|
64
|
+
|
|
65
|
+
new_rows = []
|
|
66
|
+
for index, identifier in enumerate(df[filename_column]):
|
|
67
|
+
paths = search_objects_with_identifier(
|
|
68
|
+
session, workingdirectory, identifier, exact_match
|
|
69
|
+
)
|
|
70
|
+
for path in paths:
|
|
71
|
+
new_row = df.iloc[index].drop(filename_column)
|
|
72
|
+
new_row[DATAOBJECT] = path
|
|
73
|
+
# create a 1 row dataframe, which needs to be transposed (hence the T)
|
|
74
|
+
new_rows.append(new_row.to_frame().T)
|
|
75
|
+
if len(new_rows) > 0:
|
|
76
|
+
new_df = pd.concat(new_rows, ignore_index=True)
|
|
77
|
+
else:
|
|
78
|
+
columns = [column for column in df.columns if column != filename_column]
|
|
79
|
+
columns.append(DATAOBJECT)
|
|
80
|
+
new_df = pd.DataFrame(columns=columns)
|
|
81
|
+
return new_df
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def render_single_path_from_pattern(
|
|
85
|
+
row: pd.Series, pattern: str, env: jinja2.Environment
|
|
86
|
+
) -> str:
|
|
87
|
+
"""Render a path from a pattern using a single DataFrame row."""
|
|
88
|
+
|
|
89
|
+
path_template = env.from_string(pattern)
|
|
90
|
+
return path_template.render(row.to_dict())
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def create_path_based_on_pattern(
|
|
94
|
+
df: pd.DataFrame, pattern: str, env: jinja2.Environment
|
|
95
|
+
):
|
|
96
|
+
"""Create a column for data object paths based on info from other columns"""
|
|
97
|
+
|
|
98
|
+
path_template = env.from_string(pattern)
|
|
99
|
+
constructed_paths = [
|
|
100
|
+
path_template.render(row.to_dict()) for _, row in df.iterrows()
|
|
101
|
+
]
|
|
102
|
+
df[DATAOBJECT] = constructed_paths
|
|
103
|
+
return df
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def chain_collection_and_filename(
|
|
107
|
+
df: pd.DataFrame, filename_column: str, workingdirectory: str
|
|
108
|
+
):
|
|
109
|
+
"""Renames the column with the relative data object or collection path and completes it with the parent collection path"""
|
|
110
|
+
df = df.rename(columns={filename_column: DATAOBJECT})
|
|
111
|
+
df[DATAOBJECT] = [str(pathlib.PurePosixPath(workingdirectory, x)) for x in df[DATAOBJECT]]
|
|
112
|
+
return df
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
# filters for creating patterns
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def date_format(value, input_format="%d-%m-%Y", output_format="%Y-%m-%d"):
|
|
119
|
+
"""
|
|
120
|
+
Return a date in the specified output format
|
|
121
|
+
|
|
122
|
+
If the value is a datetime object, it will be converted immediately.
|
|
123
|
+
If the value is a string, it will be converted to a datetime object
|
|
124
|
+
according to the input format, and then converted to the output format.
|
|
125
|
+
"""
|
|
126
|
+
|
|
127
|
+
if isinstance(value, str):
|
|
128
|
+
value = datetime.datetime.strptime(value, input_format)
|
|
129
|
+
# If already a datetime object, skip conversion
|
|
130
|
+
return value.strftime(output_format)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def create_jinja_environment_with_filters():
|
|
134
|
+
jinja_environment = jinja2.Environment()
|
|
135
|
+
jinja_environment.filters["date_format"] = date_format
|
|
136
|
+
return jinja_environment
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def validate_schema_columns(sheets: dict[pd.DataFrame], schema: Schema) -> list[str]:
|
|
140
|
+
if schema is None:
|
|
141
|
+
return list(sheets.keys())
|
|
142
|
+
|
|
143
|
+
required_fields = [
|
|
144
|
+
field_name
|
|
145
|
+
for field_name, field in schema.fields.items()
|
|
146
|
+
if field.required and not field.default
|
|
147
|
+
]
|
|
148
|
+
sheets_for_schema = [
|
|
149
|
+
sheetname
|
|
150
|
+
for sheetname, sheet in sheets.items()
|
|
151
|
+
if all(field_name in sheet.columns for field_name in required_fields)
|
|
152
|
+
]
|
|
153
|
+
if len(sheets_for_schema) == 0:
|
|
154
|
+
raise KeyError(
|
|
155
|
+
"None of the sheets contain all the required fields of the schema."
|
|
156
|
+
)
|
|
157
|
+
return sheets_for_schema
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def process_tabular_file(
|
|
161
|
+
filename: str | pathlib.Path,
|
|
162
|
+
config: io.StringIO | click.File,
|
|
163
|
+
session: iRODSSession | None = None,
|
|
164
|
+
) -> dict:
|
|
165
|
+
"""Apply the preprocessing to a file based on a configuration file"""
|
|
166
|
+
yml = yaml.safe_load(config)
|
|
167
|
+
sheets = parse_tabular_file(filename, session, yml.get("separator", None))
|
|
168
|
+
item_type = yml.get("item_type", ItemType.DATAOBJECT.name)
|
|
169
|
+
sheets_to_return = {}
|
|
170
|
+
for sheetname, sheet in sheets.items():
|
|
171
|
+
if sheetname not in yml["sheets"]:
|
|
172
|
+
continue
|
|
173
|
+
path_column_name = yml["path_column"]["column_name"]
|
|
174
|
+
if (
|
|
175
|
+
item_type == ItemType.DATAOBJECT.name
|
|
176
|
+
and yml["path_column"]["path_type"] == "part"
|
|
177
|
+
):
|
|
178
|
+
if session is None:
|
|
179
|
+
raise ValueError("Cannot query paths with no iRODS session")
|
|
180
|
+
sheet = query_dataobjects_with_filename(
|
|
181
|
+
session,
|
|
182
|
+
sheet,
|
|
183
|
+
path_column_name,
|
|
184
|
+
yml["path_column"]["workdir"],
|
|
185
|
+
exact_match=False,
|
|
186
|
+
)
|
|
187
|
+
if sheet.empty:
|
|
188
|
+
continue
|
|
189
|
+
elif yml["path_column"]["path_type"] == "relative":
|
|
190
|
+
sheet = chain_collection_and_filename(
|
|
191
|
+
sheet, path_column_name, yml["path_column"]["workdir"]
|
|
192
|
+
)
|
|
193
|
+
elif yml["path_column"]["path_type"] == "pattern":
|
|
194
|
+
env = create_jinja_environment_with_filters()
|
|
195
|
+
sheet = create_path_based_on_pattern(
|
|
196
|
+
sheet, yml["path_column"]["pattern"], env
|
|
197
|
+
)
|
|
198
|
+
else:
|
|
199
|
+
sheet = sheet.rename(columns={path_column_name: DATAOBJECT})
|
|
200
|
+
|
|
201
|
+
if "whitelist" in yml:
|
|
202
|
+
sheet = sheet[
|
|
203
|
+
[c for c in sheet.columns if c in [DATAOBJECT] + yml["whitelist"]]
|
|
204
|
+
]
|
|
205
|
+
elif "blacklist" in yml:
|
|
206
|
+
sheet = sheet[[c for c in sheet.columns if c not in yml["blacklist"]]]
|
|
207
|
+
sheets_to_return[sheetname] = sheet
|
|
208
|
+
|
|
209
|
+
multivalue_columns = yml.get("multivalue_columns", [])
|
|
210
|
+
multivalue_separator = yml.get("multivalue_separator", "")
|
|
211
|
+
schema_info = yml.get("mango_schema", {})
|
|
212
|
+
if "path" in schema_info:
|
|
213
|
+
|
|
214
|
+
def get_schema(path):
|
|
215
|
+
match path:
|
|
216
|
+
case {"realm": realm, "schema": schema}:
|
|
217
|
+
return Schema(
|
|
218
|
+
get_mango_schema(session, realm=realm, schema_name=schema)
|
|
219
|
+
)
|
|
220
|
+
case str(path):
|
|
221
|
+
return Schema(path) if os.path.exists(path) else None
|
|
222
|
+
|
|
223
|
+
schema = get_schema(schema_info["path"])
|
|
224
|
+
else:
|
|
225
|
+
schema = None
|
|
226
|
+
if schema:
|
|
227
|
+
schema_instructions = {
|
|
228
|
+
"schema": schema,
|
|
229
|
+
EXCLUDE_NONSCHEMA_MD: schema_info.get(EXCLUDE_NONSCHEMA_MD, True),
|
|
230
|
+
EXCLUDE_INVALID_SCHEMA_MD: schema_info.get(
|
|
231
|
+
EXCLUDE_INVALID_SCHEMA_MD, False
|
|
232
|
+
),
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
else:
|
|
236
|
+
console.print("No schema found, metadata will be added as is.")
|
|
237
|
+
schema_instructions = {}
|
|
238
|
+
|
|
239
|
+
processed_config_data = {
|
|
240
|
+
"item_type": ItemType[yml.get("item_type", ItemType.DATAOBJECT)],
|
|
241
|
+
"sheets": sheets_to_return,
|
|
242
|
+
"multivalue_columns": multivalue_columns,
|
|
243
|
+
"multivalue_separator": multivalue_separator,
|
|
244
|
+
"schema_instructions": schema_instructions,
|
|
245
|
+
}
|
|
246
|
+
return processed_config_data
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
# only connect to irods if requested
|
|
250
|
+
def get_sheets(example: str, sep=",", irods=False):
|
|
251
|
+
"""Parse a tabular file in iRODS or locally, with the right separator"""
|
|
252
|
+
if irods:
|
|
253
|
+
try:
|
|
254
|
+
env_file = os.environ["IRODS_ENVIRONMENT_FILE"]
|
|
255
|
+
except KeyError:
|
|
256
|
+
env_file = os.path.expanduser("~/.irods/irods_environment.json")
|
|
257
|
+
|
|
258
|
+
ssl_settings = {}
|
|
259
|
+
with iRODSSession(irods_env_file=env_file, **ssl_settings) as session:
|
|
260
|
+
sheets = parse_tabular_file(example, session, sep)
|
|
261
|
+
else:
|
|
262
|
+
sheets = parse_tabular_file(example, separator=sep)
|
|
263
|
+
return sheets
|
|
@@ -0,0 +1,289 @@
|
|
|
1
|
+
import os.path
|
|
2
|
+
import pandas as pd
|
|
3
|
+
from . import console, ItemType, EXCLUDE_INVALID_SCHEMA_MD, EXCLUDE_NONSCHEMA_MD
|
|
4
|
+
from rich.markdown import Markdown
|
|
5
|
+
from rich.prompt import Prompt, Confirm
|
|
6
|
+
from .preprocessing import (
|
|
7
|
+
create_jinja_environment_with_filters,
|
|
8
|
+
render_single_path_from_pattern,
|
|
9
|
+
)
|
|
10
|
+
from typing import Set
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def explain_multiple_choice():
|
|
14
|
+
console.print(
|
|
15
|
+
"Type one answer at a time, pressing Enter afterwards. \
|
|
16
|
+
Press Enter twice when you are done.",
|
|
17
|
+
style="italic magenta",
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def select_sheets(sheet_collection: dict) -> list:
|
|
22
|
+
"""Ask user to choose which sheets to use from an Excel"""
|
|
23
|
+
selection_of_sheets = list(sheet_collection.keys())
|
|
24
|
+
if len(sheet_collection) == 1:
|
|
25
|
+
if selection_of_sheets[0] == "single_sheet":
|
|
26
|
+
console.print(
|
|
27
|
+
"You have provided a plain text file, no multiple sheets, great work!"
|
|
28
|
+
)
|
|
29
|
+
else:
|
|
30
|
+
console.print(
|
|
31
|
+
Markdown(
|
|
32
|
+
f"The file you provided has only one sheet: `{selection_of_sheets[0]}`."
|
|
33
|
+
)
|
|
34
|
+
)
|
|
35
|
+
return selection_of_sheets[0]
|
|
36
|
+
all_sheets = Confirm.ask("Would you like to use all of the available sheets?")
|
|
37
|
+
if all_sheets:
|
|
38
|
+
return selection_of_sheets
|
|
39
|
+
explain_multiple_choice()
|
|
40
|
+
selected_sheets = []
|
|
41
|
+
while True:
|
|
42
|
+
selected_sheet = Prompt.ask(
|
|
43
|
+
"Which of the available sheets would you like to select?",
|
|
44
|
+
choices=selection_of_sheets + [""],
|
|
45
|
+
)
|
|
46
|
+
if selected_sheet:
|
|
47
|
+
selected_sheets.append(selected_sheet)
|
|
48
|
+
else:
|
|
49
|
+
break
|
|
50
|
+
return selected_sheets
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def identify_item_column(sheet_collection: dict) -> str:
|
|
54
|
+
"""Ask user which column contains the unique data object or collection information"""
|
|
55
|
+
columns = set([col for sheet in sheet_collection.values() for col in sheet.columns])
|
|
56
|
+
dfs = "dataframe has" if len(sheet_collection) == 1 else "dataframes have"
|
|
57
|
+
cols = "1 column" if len(columns) == 1 else f"{len(columns)} columns"
|
|
58
|
+
column_intro = f"Your {dfs} {cols}:\n\n"
|
|
59
|
+
column_list = "\n\n".join(f"- {col}" for col in columns)
|
|
60
|
+
console.print(Markdown(column_intro + column_list))
|
|
61
|
+
return Prompt.ask(
|
|
62
|
+
"Which column contains an unique identifier for the target item?",
|
|
63
|
+
choices=columns,
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def test_pattern_on_first_column(sheet_collection: dict[pd.DataFrame], pattern: str):
|
|
68
|
+
"""
|
|
69
|
+
Apply a pattern on the first row of the first dataframe of a dictionary of dataframes.
|
|
70
|
+
"""
|
|
71
|
+
|
|
72
|
+
df = list(sheet_collection.values())[0]
|
|
73
|
+
row = df.iloc[0]
|
|
74
|
+
env = create_jinja_environment_with_filters()
|
|
75
|
+
try:
|
|
76
|
+
result = render_single_path_from_pattern(row, pattern, env)
|
|
77
|
+
except Exception as e:
|
|
78
|
+
message = f"The following error occured while trying to apply the pattern to a row:\n {type(e).__name__}: {repr(e.args)}"
|
|
79
|
+
print(message)
|
|
80
|
+
result = None
|
|
81
|
+
return result
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def classify_target_item_column(
|
|
85
|
+
sheet_collection: dict,
|
|
86
|
+
) -> dict:
|
|
87
|
+
|
|
88
|
+
import re
|
|
89
|
+
|
|
90
|
+
item_type = Prompt.ask(
|
|
91
|
+
"Will you be annotating data objects or colletions?", choices=ItemType
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
message = """
|
|
95
|
+
In order to add metadata to your TARGET_ITEMs, each row needs
|
|
96
|
+
to have a reference to your TARGET_ITEM.
|
|
97
|
+
|
|
98
|
+
For your table, how can we find the TARGET_ITEM in each row?
|
|
99
|
+
|
|
100
|
+
1) A column contains the absolute path to the TARGET_ITEM
|
|
101
|
+
2) A column contains the relative path to the TARGET_ITEM
|
|
102
|
+
3) The absolute path of the data TARGET_ITEM can be reconstructed by combining
|
|
103
|
+
info of multiple columns and strings.
|
|
104
|
+
"""
|
|
105
|
+
message = message.replace("TARGET_ITEM", item_type)
|
|
106
|
+
choice_mapping = {"1": "absolute", "2": "relative", "3": "pattern"}
|
|
107
|
+
|
|
108
|
+
if item_type == ItemType.DATAOBJECT.value:
|
|
109
|
+
message += "4) A column contains part of the data object name\n"
|
|
110
|
+
choice_mapping["4"] = "part"
|
|
111
|
+
|
|
112
|
+
answer = Prompt.ask(message, choices=list(choice_mapping.keys()))
|
|
113
|
+
path_type = choice_mapping[answer]
|
|
114
|
+
workdir = ""
|
|
115
|
+
pattern = ""
|
|
116
|
+
if path_type == "pattern":
|
|
117
|
+
item_column = ""
|
|
118
|
+
pattern_question = """
|
|
119
|
+
Provide a path pattern using double curly braces ({{ }}) to reference column names.
|
|
120
|
+
Example: '/zone/home/project/{{ lab }}_{{ experiment }}.txt' will use values from the 'lab' and 'experiment' columns in each row.
|
|
121
|
+
You can also use filters to modify the values of the columns before using them in the path.
|
|
122
|
+
For more information, see the documentation in docs/construct_path_from_columns.md.\n"""
|
|
123
|
+
pattern_okay = False
|
|
124
|
+
while not pattern_okay:
|
|
125
|
+
pattern = Prompt.ask(pattern_question)
|
|
126
|
+
preview = test_pattern_on_first_column(sheet_collection, pattern)
|
|
127
|
+
if preview is None:
|
|
128
|
+
print("The pattern you provided is not valid.")
|
|
129
|
+
else:
|
|
130
|
+
pattern_ok_message = (
|
|
131
|
+
f"Based on the pattern you provided, the first row in your file contains the following path: {preview}."
|
|
132
|
+
"\n Does this look okay?"
|
|
133
|
+
)
|
|
134
|
+
pattern_okay = Confirm.ask(pattern_ok_message)
|
|
135
|
+
|
|
136
|
+
print(
|
|
137
|
+
f"Great! Data objects will be found by combining columns and strings in the following pattern: {pattern}"
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
else:
|
|
141
|
+
item_column = identify_item_column(sheet_collection)
|
|
142
|
+
if path_type in ["relative", "part"]:
|
|
143
|
+
while not re.match("/[a-z_]+/home/[^/]+/?", workdir):
|
|
144
|
+
workdir = Prompt.ask(
|
|
145
|
+
"What is the absolute path of the collection where we can find these data objects? \
|
|
146
|
+
(It should start with `/{zone}/home/{project}/...`)"
|
|
147
|
+
)
|
|
148
|
+
if path_type == "relative":
|
|
149
|
+
console.print(
|
|
150
|
+
Markdown(
|
|
151
|
+
f"Great! The relative paths in `{item_column}` will be chained to `{workdir}`!"
|
|
152
|
+
)
|
|
153
|
+
)
|
|
154
|
+
elif path_type == "part":
|
|
155
|
+
console.print(
|
|
156
|
+
Markdown(
|
|
157
|
+
f"Great! Data objects will be found by querying the contents of `{item_column}` within `{workdir}`!"
|
|
158
|
+
)
|
|
159
|
+
)
|
|
160
|
+
enum_mapping = {x.value: x for x in ItemType}
|
|
161
|
+
|
|
162
|
+
return {
|
|
163
|
+
"item_type": enum_mapping[item_type],
|
|
164
|
+
"item_column": item_column,
|
|
165
|
+
"path_type": path_type,
|
|
166
|
+
"pattern": pattern,
|
|
167
|
+
"workdir": workdir,
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def filter_columns(columns: list) -> dict:
|
|
172
|
+
"""Ask user to blacklist or whitelist columns"""
|
|
173
|
+
filter_how = Prompt.ask(
|
|
174
|
+
"Would you like to whitelist or blacklist some columns?",
|
|
175
|
+
choices=["whitelist", "blacklist", "neither"],
|
|
176
|
+
default="neither",
|
|
177
|
+
)
|
|
178
|
+
if filter_how == "neither":
|
|
179
|
+
return {}
|
|
180
|
+
explain_multiple_choice()
|
|
181
|
+
# using a set to get a list without duplicates
|
|
182
|
+
filter_what = set()
|
|
183
|
+
# creating a 'choices'-list so we don't modify the original columns list
|
|
184
|
+
choices = columns.copy() + [""]
|
|
185
|
+
while any(c for c in choices):
|
|
186
|
+
ans = Prompt.ask(
|
|
187
|
+
f"Which column(s) would you like to {filter_how}?", choices=choices
|
|
188
|
+
)
|
|
189
|
+
if ans:
|
|
190
|
+
filter_what.add(ans)
|
|
191
|
+
choices.remove(ans)
|
|
192
|
+
else:
|
|
193
|
+
break
|
|
194
|
+
# convert set back to list because a set cannot
|
|
195
|
+
# be added to a yml
|
|
196
|
+
filter_what = list(filter_what)
|
|
197
|
+
if len(filter_what) == 0:
|
|
198
|
+
return {}
|
|
199
|
+
return {filter_how: filter_what}
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def ask_multivalue_columns(columns: list) -> list:
|
|
203
|
+
"""Ask user whether the sheets contain any colums with multiple values"""
|
|
204
|
+
|
|
205
|
+
explain_multiple_choice()
|
|
206
|
+
# using a set to get a list without duplicates
|
|
207
|
+
multivalue_columns = set()
|
|
208
|
+
choices = columns.copy() + [""]
|
|
209
|
+
# creating a 'choices'-list so we don't modify the original columns list
|
|
210
|
+
while any(c for c in choices):
|
|
211
|
+
ans = Prompt.ask(
|
|
212
|
+
"Which column(s) can contain multiple values?",
|
|
213
|
+
choices=choices,
|
|
214
|
+
)
|
|
215
|
+
if ans:
|
|
216
|
+
multivalue_columns.add(ans)
|
|
217
|
+
choices.remove(ans)
|
|
218
|
+
else:
|
|
219
|
+
break
|
|
220
|
+
|
|
221
|
+
# convert set back to list because a set cannot
|
|
222
|
+
# be added to a yml
|
|
223
|
+
multivalue_columns = list(multivalue_columns)
|
|
224
|
+
return multivalue_columns
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def list_columns_with_character(
|
|
228
|
+
dfs: list[pd.DataFrame], eligible_columns: list, character: str
|
|
229
|
+
) -> Set[str]:
|
|
230
|
+
"""
|
|
231
|
+
Given a list of pandas DataFrames, return a set of column names
|
|
232
|
+
where at least one value contains the specified character.
|
|
233
|
+
"""
|
|
234
|
+
|
|
235
|
+
return set(
|
|
236
|
+
col
|
|
237
|
+
for df in dfs
|
|
238
|
+
for col in df.columns
|
|
239
|
+
if col in eligible_columns
|
|
240
|
+
and df[col].dtype == object
|
|
241
|
+
and df[col].astype(str).str.contains(character, na=False, regex=False).any()
|
|
242
|
+
) # not sure about how this gets split in rows
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def ask_about_schemas() -> dict | None:
|
|
246
|
+
if not Confirm.ask(
|
|
247
|
+
"Do you have a ManGO metadata schema to validate your metadata?"
|
|
248
|
+
):
|
|
249
|
+
return
|
|
250
|
+
if Confirm.ask(
|
|
251
|
+
"Does the schema exist in ManGO? (We cannot verify the correctness at this stage yet)"
|
|
252
|
+
):
|
|
253
|
+
realm = Prompt.ask(
|
|
254
|
+
"Please provide the name of the project/realm the schema belongs to"
|
|
255
|
+
)
|
|
256
|
+
schema = Prompt.ask(
|
|
257
|
+
f"Please provide the name of the published schema in the {realm} realm"
|
|
258
|
+
)
|
|
259
|
+
# @todo validate against iRODS?
|
|
260
|
+
schema_file = {"realm": realm, "schema": schema}
|
|
261
|
+
else:
|
|
262
|
+
schema_file = ""
|
|
263
|
+
while not os.path.exists(schema_file):
|
|
264
|
+
# TODO add mango-mdschema validation OF the schema file
|
|
265
|
+
schema_file = Prompt.ask("Please provide a valid path for your schema: ")
|
|
266
|
+
if not schema_file:
|
|
267
|
+
print("Changed your mind? We won't use a schema then!")
|
|
268
|
+
break
|
|
269
|
+
if not schema_file:
|
|
270
|
+
return
|
|
271
|
+
invalid_schema_metadata_question = (
|
|
272
|
+
"Should we discard invalid schema values? "
|
|
273
|
+
"(Otherwise, they will be added as non-schema metadata, "
|
|
274
|
+
"e.g. 'size=medium' instead of 'mgs.schema.size=medium')"
|
|
275
|
+
)
|
|
276
|
+
exclude_invalid_schema_metadata = Confirm.ask(
|
|
277
|
+
invalid_schema_metadata_question, default=False
|
|
278
|
+
)
|
|
279
|
+
nonschema_metadata_question = (
|
|
280
|
+
"Should we discard the columns not covered by schema? "
|
|
281
|
+
"(If you say no, they will be added as non-schema metadata):"
|
|
282
|
+
)
|
|
283
|
+
|
|
284
|
+
exclude_nonschema_metadata = Confirm.ask(nonschema_metadata_question, default=True)
|
|
285
|
+
return {
|
|
286
|
+
"path": schema_file,
|
|
287
|
+
EXCLUDE_NONSCHEMA_MD: exclude_nonschema_metadata,
|
|
288
|
+
EXCLUDE_INVALID_SCHEMA_MD: exclude_invalid_schema_metadata,
|
|
289
|
+
}
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
import pandas as pd
|
|
2
|
+
import pathlib
|
|
3
|
+
from irods.exception import DataObjectDoesNotExist, CollectionDoesNotExist
|
|
4
|
+
from irods.data_object import iRODSDataObject
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def create_file_object(path: str | pathlib.Path, session=None):
|
|
8
|
+
"""Turn path to file into a file-like object.
|
|
9
|
+
|
|
10
|
+
Args:
|
|
11
|
+
session: (iRODSSession or None): session to connect to iRODS.
|
|
12
|
+
Session can be set to None for testing non-irods functionalities.
|
|
13
|
+
path (str): Path to tabular file
|
|
14
|
+
|
|
15
|
+
Raises:
|
|
16
|
+
FileNotFoundError: If the file cannot be found locally or in ManGO.
|
|
17
|
+
|
|
18
|
+
Returns:
|
|
19
|
+
pathlib.Path or irods.iRODSDataObject: File-like object to read metadata from.
|
|
20
|
+
"""
|
|
21
|
+
ppath = pathlib.Path(path)
|
|
22
|
+
if ppath.suffix not in [".xlsx", ".csv", ".tsv"]:
|
|
23
|
+
raise IOError("Filetype not accepted")
|
|
24
|
+
if ppath.exists():
|
|
25
|
+
return ppath
|
|
26
|
+
if session:
|
|
27
|
+
try:
|
|
28
|
+
return session.data_objects.get(path)
|
|
29
|
+
except DataObjectDoesNotExist or CollectionDoesNotExist as e:
|
|
30
|
+
raise e
|
|
31
|
+
raise FileNotFoundError
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def parse_tabular_file(path: str | pathlib.Path, session=None, separator: str = ","):
|
|
35
|
+
"""Parse tabular file.
|
|
36
|
+
|
|
37
|
+
Args:
|
|
38
|
+
path (str): Path to the tabular file.
|
|
39
|
+
session (iRODSSession or None): session to connect to iRODS.
|
|
40
|
+
If it is none, it is assumed that we are testing
|
|
41
|
+
separator (str, optional): Separator for plain text files.. Defaults to ",".
|
|
42
|
+
|
|
43
|
+
Raises:
|
|
44
|
+
IOError: If the file cannot be parsed (it is not .xlsx, .csv or .tsv) it won't be read.
|
|
45
|
+
|
|
46
|
+
Returns:
|
|
47
|
+
dict: Dictionary of pandas.DataFrames with sheet names as keys.
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
file = create_file_object(path, session)
|
|
51
|
+
if file.name.endswith("xlsx"):
|
|
52
|
+
# Local excel files are binary and should be opened with 'rb'.
|
|
53
|
+
# However, iRODS implemented their 'open' method differently,
|
|
54
|
+
# and there you should use just 'r' instead
|
|
55
|
+
reading_mode = "r" if isinstance(file, iRODSDataObject) else "rb"
|
|
56
|
+
with file.open(reading_mode) as f:
|
|
57
|
+
sheets = pd.read_excel(f, sheet_name=None)
|
|
58
|
+
if any(x.strip() != x for x in sheets.keys()):
|
|
59
|
+
sheets = {k.strip(): v for k, v in sheets.items()}
|
|
60
|
+
else:
|
|
61
|
+
# these types are not binary and should be opened with 'r'
|
|
62
|
+
sheets = {"single_sheet": pd.read_csv(str(path), sep=separator)}
|
|
63
|
+
for sheet in sheets.values():
|
|
64
|
+
sheet.columns = sheet.columns.str.strip()
|
|
65
|
+
return sheets
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
import io
|
|
2
|
+
import os.path
|
|
3
|
+
import pathlib
|
|
4
|
+
from irods.session import iRODSSession
|
|
5
|
+
import click
|
|
6
|
+
from rich.markdown import Markdown
|
|
7
|
+
from rich.progress import track
|
|
8
|
+
from .dataframe2avus import dict_to_avus, generate_rows, apply_metadata_to_data_object
|
|
9
|
+
from .preprocessing import process_tabular_file, validate_schema_columns
|
|
10
|
+
from . import console
|
|
11
|
+
|
|
12
|
+
# region Chains
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@click.command()
|
|
16
|
+
@click.option(
|
|
17
|
+
"--config",
|
|
18
|
+
type=click.File("r"),
|
|
19
|
+
required=True,
|
|
20
|
+
help="Configuration file created by `setup`.",
|
|
21
|
+
)
|
|
22
|
+
@click.option("--dry-run", is_flag=True, help="Simulate applying the metadata.")
|
|
23
|
+
@click.argument("filename")
|
|
24
|
+
def run(filename, config, dry_run=False):
|
|
25
|
+
"""Apply metadata from a tabular file to data objects.
|
|
26
|
+
|
|
27
|
+
FILENAME is the path to the tabular file containing the metadata.
|
|
28
|
+
It should have some column with a unique identifier for the data objects,
|
|
29
|
+
and columns for other metadata fields.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
tabular_file = pathlib.Path(filename)
|
|
33
|
+
if not tabular_file.exists():
|
|
34
|
+
raise FileExistsError
|
|
35
|
+
try:
|
|
36
|
+
env_file = os.environ["IRODS_ENVIRONMENT_FILE"]
|
|
37
|
+
except KeyError:
|
|
38
|
+
env_file = os.path.expanduser("~/.irods/irods_environment.json")
|
|
39
|
+
|
|
40
|
+
ssl_settings = {}
|
|
41
|
+
with iRODSSession(irods_env_file=env_file, **ssl_settings) as session:
|
|
42
|
+
apply_metadata_from_table(tabular_file, config, dry_run, session)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def apply_metadata_from_table(
|
|
46
|
+
filename: str | pathlib.Path,
|
|
47
|
+
config: io.StringIO | click.File,
|
|
48
|
+
dry_run: bool = False,
|
|
49
|
+
session: iRODSSession | None = None,
|
|
50
|
+
):
|
|
51
|
+
if session is None:
|
|
52
|
+
console.print("No session found, we'll try a dry-run!")
|
|
53
|
+
processed_config_data = process_tabular_file(filename, config, session)
|
|
54
|
+
|
|
55
|
+
sheets = processed_config_data["sheets"]
|
|
56
|
+
item_type = processed_config_data["item_type"]
|
|
57
|
+
multivalue_columns = processed_config_data["multivalue_columns"]
|
|
58
|
+
multivalue_separator = processed_config_data["multivalue_separator"]
|
|
59
|
+
schema_instructions = processed_config_data["schema_instructions"]
|
|
60
|
+
sheets_for_schemas = validate_schema_columns(
|
|
61
|
+
sheets, schema_instructions.get("schema", None)
|
|
62
|
+
)
|
|
63
|
+
for sheetname, sheet in sheets.items():
|
|
64
|
+
sheet_schema_instructions = (
|
|
65
|
+
schema_instructions if sheetname in sheets_for_schemas else {}
|
|
66
|
+
)
|
|
67
|
+
progress_message = f"Adding metadata from {sheetname + ' in ' if len(sheets) > 1 else ''}`{filename}`..."
|
|
68
|
+
n = 0
|
|
69
|
+
errors = 0
|
|
70
|
+
min_avus = None
|
|
71
|
+
max_avus = None
|
|
72
|
+
|
|
73
|
+
# loop over each row printing a progress bar
|
|
74
|
+
for item, md_dict in track(
|
|
75
|
+
generate_rows(sheet, multivalue_columns, multivalue_separator),
|
|
76
|
+
description=progress_message,
|
|
77
|
+
):
|
|
78
|
+
if session and not dry_run:
|
|
79
|
+
avus = apply_metadata_to_data_object(
|
|
80
|
+
item, md_dict, sheet_schema_instructions, session, item_type
|
|
81
|
+
)
|
|
82
|
+
yield {"item": item, "avus": avus}
|
|
83
|
+
else:
|
|
84
|
+
console.print(
|
|
85
|
+
f"Creating the following AVUs for {item_type.value} {item}:"
|
|
86
|
+
)
|
|
87
|
+
avus = dict_to_avus(md_dict, **sheet_schema_instructions)
|
|
88
|
+
print(avus)
|
|
89
|
+
yield {"item": item, "avus": avus}
|
|
90
|
+
if avus:
|
|
91
|
+
n += 1
|
|
92
|
+
if max_avus is None or len(avus) > max_avus:
|
|
93
|
+
max_avus = len(avus)
|
|
94
|
+
if min_avus is None or len(avus) < min_avus:
|
|
95
|
+
min_avus = len(avus)
|
|
96
|
+
else:
|
|
97
|
+
errors += 1
|
|
98
|
+
|
|
99
|
+
avu_length_range = (
|
|
100
|
+
max_avus if min_avus == max_avus else f"{min_avus} to {max_avus}"
|
|
101
|
+
)
|
|
102
|
+
console.print(
|
|
103
|
+
Markdown(
|
|
104
|
+
# This calculation may not be correct anymore in case of multiple values,
|
|
105
|
+
# since the md_dict of each object can now have a different length
|
|
106
|
+
f"{'Simulated' if dry_run else 'Applied'} {avu_length_range} AVUs for each of {n} {item_type.value}s"
|
|
107
|
+
)
|
|
108
|
+
)
|
|
109
|
+
if errors > 0:
|
|
110
|
+
console.print(
|
|
111
|
+
f"{errors} data objects were skipped because the paths were not valid!",
|
|
112
|
+
style="red bold",
|
|
113
|
+
)
|