mlcroissant 0.0.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mlcroissant-0.0.2/PKG-INFO +153 -0
- mlcroissant-0.0.2/README.md +121 -0
- mlcroissant-0.0.2/mlcroissant/__init__.py +16 -0
- mlcroissant-0.0.2/mlcroissant/_src/__init__.py +0 -0
- mlcroissant-0.0.2/mlcroissant/_src/core/__init__.py +0 -0
- mlcroissant-0.0.2/mlcroissant/_src/core/constants.py +112 -0
- mlcroissant-0.0.2/mlcroissant/_src/core/constants_test.py +14 -0
- mlcroissant-0.0.2/mlcroissant/_src/core/data_types.py +34 -0
- mlcroissant-0.0.2/mlcroissant/_src/core/git.py +37 -0
- mlcroissant-0.0.2/mlcroissant/_src/core/git_test.py +54 -0
- mlcroissant-0.0.2/mlcroissant/_src/core/graphs/__init__.py +0 -0
- mlcroissant-0.0.2/mlcroissant/_src/core/graphs/utils.py +46 -0
- mlcroissant-0.0.2/mlcroissant/_src/core/issues.py +83 -0
- mlcroissant-0.0.2/mlcroissant/_src/core/issues_test.py +40 -0
- mlcroissant-0.0.2/mlcroissant/_src/core/json_ld.py +221 -0
- mlcroissant-0.0.2/mlcroissant/_src/core/json_ld_test.py +58 -0
- mlcroissant-0.0.2/mlcroissant/_src/core/optional.py +90 -0
- mlcroissant-0.0.2/mlcroissant/_src/core/optional_test.py +27 -0
- mlcroissant-0.0.2/mlcroissant/_src/core/path.py +50 -0
- mlcroissant-0.0.2/mlcroissant/_src/core/types.py +5 -0
- mlcroissant-0.0.2/mlcroissant/_src/datasets.py +116 -0
- mlcroissant-0.0.2/mlcroissant/_src/datasets_test.py +125 -0
- mlcroissant-0.0.2/mlcroissant/_src/nodes.py +20 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/__init__.py +5 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/base_operation.py +33 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/execute.py +111 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/execute_test.py +30 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/graph.py +265 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/__init__.py +25 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/concatenate.py +30 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/concatenate_test.py +9 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/data.py +20 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/data_test.py +9 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/download.py +170 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/download_test.py +70 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/extract.py +60 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/extract_test.py +29 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/field.py +67 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/field_test.py +9 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/filter.py +42 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/filter_test.py +9 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/group.py +16 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/group_test.py +9 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/init.py +16 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/init_test.py +9 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/join.py +67 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/join_test.py +9 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/local_directory.py +25 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/local_directory_test.py +11 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/parse_json.py +20 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/parse_json_test.py +29 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/read.py +88 -0
- mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/read_test.py +45 -0
- mlcroissant-0.0.2/mlcroissant/_src/structure_graph/__init__.py +0 -0
- mlcroissant-0.0.2/mlcroissant/_src/structure_graph/base_node.py +204 -0
- mlcroissant-0.0.2/mlcroissant/_src/structure_graph/base_node_test.py +81 -0
- mlcroissant-0.0.2/mlcroissant/_src/structure_graph/graph.py +142 -0
- mlcroissant-0.0.2/mlcroissant/_src/structure_graph/graph_test.py +47 -0
- mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/__init__.py +0 -0
- mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/field.py +196 -0
- mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/field_test.py +52 -0
- mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/file_object.py +95 -0
- mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/file_object_test.py +58 -0
- mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/file_set.py +76 -0
- mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/file_set_test.py +53 -0
- mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/metadata.py +197 -0
- mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/metadata_test.py +63 -0
- mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/rdf.py +41 -0
- mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/rdf_test.py +21 -0
- mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/record_set.py +125 -0
- mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/record_set_test.py +107 -0
- mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/source.py +321 -0
- mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/source_test.py +268 -0
- mlcroissant-0.0.2/mlcroissant/_src/tests/__init__.py +0 -0
- mlcroissant-0.0.2/mlcroissant/_src/tests/nodes.py +85 -0
- mlcroissant-0.0.2/mlcroissant/_src/tests/records.py +38 -0
- mlcroissant-0.0.2/mlcroissant/_src/tests/records_test.py +33 -0
- mlcroissant-0.0.2/mlcroissant.egg-info/PKG-INFO +153 -0
- mlcroissant-0.0.2/mlcroissant.egg-info/SOURCES.txt +94 -0
- mlcroissant-0.0.2/mlcroissant.egg-info/dependency_links.txt +1 -0
- mlcroissant-0.0.2/mlcroissant.egg-info/requires.txt +29 -0
- mlcroissant-0.0.2/mlcroissant.egg-info/top_level.txt +3 -0
- mlcroissant-0.0.2/pyproject.toml +83 -0
- mlcroissant-0.0.2/scripts/__init__.py +1 -0
- mlcroissant-0.0.2/scripts/from_huggingface_to_croissant.py +192 -0
- mlcroissant-0.0.2/scripts/from_huggingface_to_croissant_test.py +34 -0
- mlcroissant-0.0.2/scripts/load.py +106 -0
- mlcroissant-0.0.2/scripts/load_test.py +20 -0
- mlcroissant-0.0.2/scripts/migrations/__init__.py +1 -0
- mlcroissant-0.0.2/scripts/migrations/migrate.py +136 -0
- mlcroissant-0.0.2/scripts/migrations/previous/202307171508.py +104 -0
- mlcroissant-0.0.2/scripts/migrations/previous/202307201041.py +25 -0
- mlcroissant-0.0.2/scripts/migrations/previous/202308312000.py +44 -0
- mlcroissant-0.0.2/scripts/migrations/previous/202309061700.py +38 -0
- mlcroissant-0.0.2/scripts/validate.py +50 -0
- mlcroissant-0.0.2/setup.cfg +4 -0
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: mlcroissant
|
|
3
|
+
Version: 0.0.2
|
|
4
|
+
Summary: MLCommons datasets format.
|
|
5
|
+
Author: Joaquin Vanschoren, Jos van der Velde, Omar Benjelloun, Peter Mattson, Pieter Gijsbers, Pierre Marcenac, Pierre Ruyssen, Prabhant Singh
|
|
6
|
+
Description-Content-Type: text/markdown
|
|
7
|
+
Requires-Dist: absl-py
|
|
8
|
+
Requires-Dist: etils[epath]
|
|
9
|
+
Requires-Dist: jsonpath-rw
|
|
10
|
+
Requires-Dist: networkx
|
|
11
|
+
Requires-Dist: pandas
|
|
12
|
+
Requires-Dist: rdflib
|
|
13
|
+
Requires-Dist: requests
|
|
14
|
+
Requires-Dist: tqdm
|
|
15
|
+
Provides-Extra: dev
|
|
16
|
+
Requires-Dist: black; extra == "dev"
|
|
17
|
+
Requires-Dist: datasets; extra == "dev"
|
|
18
|
+
Requires-Dist: flake8-docstrings; extra == "dev"
|
|
19
|
+
Requires-Dist: mlcroissant[git]; extra == "dev"
|
|
20
|
+
Requires-Dist: mlcroissant[image]; extra == "dev"
|
|
21
|
+
Requires-Dist: mlcroissant[parquet]; extra == "dev"
|
|
22
|
+
Requires-Dist: pyflakes; extra == "dev"
|
|
23
|
+
Requires-Dist: pylint; extra == "dev"
|
|
24
|
+
Requires-Dist: pytest; extra == "dev"
|
|
25
|
+
Requires-Dist: pytype; extra == "dev"
|
|
26
|
+
Provides-Extra: git
|
|
27
|
+
Requires-Dist: GitPython; extra == "git"
|
|
28
|
+
Provides-Extra: image
|
|
29
|
+
Requires-Dist: Pillow; extra == "image"
|
|
30
|
+
Provides-Extra: parquet
|
|
31
|
+
Requires-Dist: pyarrow; extra == "parquet"
|
|
32
|
+
|
|
33
|
+
# mlcroissant 🥐
|
|
34
|
+
|
|
35
|
+
Discover `mlcroissant 🥐` with this
|
|
36
|
+
[introduction tutorial in Google Colab](https://colab.sandbox.google.com/github/mlcommons/croissant/blob/main/python/mlcroissant/recipes/introduction.ipynb).
|
|
37
|
+
|
|
38
|
+
## Python requirements
|
|
39
|
+
|
|
40
|
+
Python version >= 3.10.
|
|
41
|
+
|
|
42
|
+
If you do not have a Python environment:
|
|
43
|
+
|
|
44
|
+
```bash
|
|
45
|
+
python3 -m venv ~/py3
|
|
46
|
+
source ~/py3/bin/activate
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
## Install
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
python -m pip install ".[dev]"
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
## Verify/load a Croissant dataset
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
python scripts/validate.py --file ../../datasets/titanic/metadata.json
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
The command:
|
|
62
|
+
|
|
63
|
+
- Exits with 0, prints `Done` and displays encountered warnings, when no error was found in the file.
|
|
64
|
+
- Exits with 1 and displays all encountered errors/warnings, otherwise.
|
|
65
|
+
|
|
66
|
+
Similarly, you can generate a dataset by launching:
|
|
67
|
+
|
|
68
|
+
```bash
|
|
69
|
+
python scripts/load.py \
|
|
70
|
+
--file ../../datasets/titanic/metadata.json \
|
|
71
|
+
--record_set passengers \
|
|
72
|
+
--num_records 10
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
## Programmatically build JSON-LD files
|
|
76
|
+
|
|
77
|
+
You can programmatically build Croissant JSON-LD files using the Python API.
|
|
78
|
+
|
|
79
|
+
```python
|
|
80
|
+
import mlcroissant as mlc
|
|
81
|
+
metadata=mlc.nodes.Metadata(
|
|
82
|
+
name="...",
|
|
83
|
+
)
|
|
84
|
+
metadata.to_json() # this returns the JSON-LD file.
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
For a full working example, refer to
|
|
88
|
+
[the script to convert Hugging Face datasets to Croissant files](./scripts/from_huggingface_to_croissant.py).
|
|
89
|
+
This script uses the Python API to programmatically build JSON-LD files.
|
|
90
|
+
|
|
91
|
+
## Run tests
|
|
92
|
+
|
|
93
|
+
All tests can be run from the Makefile:
|
|
94
|
+
|
|
95
|
+
```bash
|
|
96
|
+
make tests
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
## Design
|
|
100
|
+
|
|
101
|
+
The most important modules in the library are:
|
|
102
|
+
|
|
103
|
+
- [`mlcroissant/_src/structure_graph`](./mlcroissant/_src/structure_graph/graph.py) is responsible for the **static analysis** of the Croissant files. We convert Croissant files to a Python representation called "**structure graph**" (using [NetworkX](https://networkx.org/)). In the process, we catch any static analysis issues (e.g., a missing mandatory field or a logic problem in the file).
|
|
104
|
+
- [`mlcroissant/_src/operation_graph`](./mlcroissant/_src/operation_graph/graph.py) is responsible for the **dynamic analysis** of the Croissant files (i.e., actually loading the dataset by yielding examples). We convert the structure graph into an "**operation graph**". Operations are the unit transformations that allow to build the dataset (like [`Download`](./mlcroissant/_src/operation_graph/operations/download.py), [`Extract`](./mlcroissant/_src/operation_graph/operations/extract.py), etc).
|
|
105
|
+
|
|
106
|
+
Other important modules are:
|
|
107
|
+
|
|
108
|
+
- [`mlcroissant/_src/core`](./mlcroissant/_src/core) defines all needed core internals. For instance, [`Issues`](./mlcroissant/_src/core/issues.py) are a way to track errors and warning during the analysis of Croissant files.
|
|
109
|
+
- [`mlcroissant/__init__.py`](./mlcroissant/__init__.py) declares the public API with [`mlcroissant.Dataset`](./mlcroissant/_src/datasets.py).
|
|
110
|
+
|
|
111
|
+
For the full design, refer to the [design doc](https://docs.google.com/document/d/1zYQIUX9ae1sZOOBq9OCsJ8JW8-Ejy3NLSeqaI5LtOEM/edit?resourcekey=0-CK78DfFvF7fnufyZqF3h3Q) for an overview of the implementation.
|
|
112
|
+
|
|
113
|
+
## Contribute
|
|
114
|
+
|
|
115
|
+
All contributions are welcome! We even have [good first issues](https://github.com/mlcommons/croissant/issues?q=is%3Aissue+is%3Aopen+label%3A%22good+first+issue%22) to start in the project. Refer to the [GitHub project](https://github.com/orgs/mlcommons/projects/26) for more detailed user stories and read above how the repo is [designed](#design).
|
|
116
|
+
|
|
117
|
+
An easy way to contribute to `mlcroissant` is using Croissant's configured [codespaces](https://docs.github.com/en/codespaces/overview).
|
|
118
|
+
To start a codespace:
|
|
119
|
+
|
|
120
|
+
- On Croissant's main [repo page](https://github.com/mlcommons/croissant), click on the `<Code>` button and select the `Codespaces` tab. You can start a new codespace by clicking on the `+` sign on the left side of the tab. By default, the codespace will start on Croissant's `main` branch, unless you select otherwise from the branches drop-down menu on the left side.
|
|
121
|
+
- While building the environment, your codespaces will install all `mlcroissant`'s required dependencies - so that you can start coding right away! Of course, you can [further personalize](https://docs.github.com/en/codespaces/customizing-your-codespace/personalizing-github-codespaces-for-your-account) your codespace.
|
|
122
|
+
- To start contributing to Croissant:
|
|
123
|
+
- Create a new branch from the `Terminal` tab in the bottom panel of your codespace with `git checkout -b feature/my-awesome-new-feature`
|
|
124
|
+
- You can create new commits, and run most git commands from the `Source Control` tab in the left panel of your codespace. Alternatively, use the `Terminal` in the bottom panel of your codespace.
|
|
125
|
+
- Iterate on your code until all tests are green (you can run tests with `make pytest` or form the `Tests` tab in the left panel of your codespace).
|
|
126
|
+
- Open a pull request (PR) with the main branch of https://github.com/mlcommons/croissant, and ask for feedback!
|
|
127
|
+
|
|
128
|
+
Alternatively, you can contribute to `mlcroissant` using the "classic" GitHub workflow:
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
## Debug
|
|
132
|
+
|
|
133
|
+
You can debug the validation of the file using the `--debug` flag:
|
|
134
|
+
|
|
135
|
+
```bash
|
|
136
|
+
python scripts/validate.py --file ../../datasets/titanic/metadata.json --debug
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
This will:
|
|
140
|
+
1. print extra information, like the generated nodes;
|
|
141
|
+
2. save the generated structure graph to a folder indicated in the logs.
|
|
142
|
+
|
|
143
|
+
## Publishing wheels
|
|
144
|
+
|
|
145
|
+
Publishing is done manually.
|
|
146
|
+
We are in the process of setting up an automatic deployment with GitHub Actions.
|
|
147
|
+
|
|
148
|
+
1. Bump the version in `croissant/python/mlcroissant/pyproject.toml`.
|
|
149
|
+
1. Build locally:
|
|
150
|
+
```bash
|
|
151
|
+
python -m build
|
|
152
|
+
```
|
|
153
|
+
1. Upload to pypi.org.
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
# mlcroissant 🥐
|
|
2
|
+
|
|
3
|
+
Discover `mlcroissant 🥐` with this
|
|
4
|
+
[introduction tutorial in Google Colab](https://colab.sandbox.google.com/github/mlcommons/croissant/blob/main/python/mlcroissant/recipes/introduction.ipynb).
|
|
5
|
+
|
|
6
|
+
## Python requirements
|
|
7
|
+
|
|
8
|
+
Python version >= 3.10.
|
|
9
|
+
|
|
10
|
+
If you do not have a Python environment:
|
|
11
|
+
|
|
12
|
+
```bash
|
|
13
|
+
python3 -m venv ~/py3
|
|
14
|
+
source ~/py3/bin/activate
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
## Install
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
python -m pip install ".[dev]"
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
## Verify/load a Croissant dataset
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
python scripts/validate.py --file ../../datasets/titanic/metadata.json
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
The command:
|
|
30
|
+
|
|
31
|
+
- Exits with 0, prints `Done` and displays encountered warnings, when no error was found in the file.
|
|
32
|
+
- Exits with 1 and displays all encountered errors/warnings, otherwise.
|
|
33
|
+
|
|
34
|
+
Similarly, you can generate a dataset by launching:
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
python scripts/load.py \
|
|
38
|
+
--file ../../datasets/titanic/metadata.json \
|
|
39
|
+
--record_set passengers \
|
|
40
|
+
--num_records 10
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
## Programmatically build JSON-LD files
|
|
44
|
+
|
|
45
|
+
You can programmatically build Croissant JSON-LD files using the Python API.
|
|
46
|
+
|
|
47
|
+
```python
|
|
48
|
+
import mlcroissant as mlc
|
|
49
|
+
metadata=mlc.nodes.Metadata(
|
|
50
|
+
name="...",
|
|
51
|
+
)
|
|
52
|
+
metadata.to_json() # this returns the JSON-LD file.
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
For a full working example, refer to
|
|
56
|
+
[the script to convert Hugging Face datasets to Croissant files](./scripts/from_huggingface_to_croissant.py).
|
|
57
|
+
This script uses the Python API to programmatically build JSON-LD files.
|
|
58
|
+
|
|
59
|
+
## Run tests
|
|
60
|
+
|
|
61
|
+
All tests can be run from the Makefile:
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
make tests
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
## Design
|
|
68
|
+
|
|
69
|
+
The most important modules in the library are:
|
|
70
|
+
|
|
71
|
+
- [`mlcroissant/_src/structure_graph`](./mlcroissant/_src/structure_graph/graph.py) is responsible for the **static analysis** of the Croissant files. We convert Croissant files to a Python representation called "**structure graph**" (using [NetworkX](https://networkx.org/)). In the process, we catch any static analysis issues (e.g., a missing mandatory field or a logic problem in the file).
|
|
72
|
+
- [`mlcroissant/_src/operation_graph`](./mlcroissant/_src/operation_graph/graph.py) is responsible for the **dynamic analysis** of the Croissant files (i.e., actually loading the dataset by yielding examples). We convert the structure graph into an "**operation graph**". Operations are the unit transformations that allow to build the dataset (like [`Download`](./mlcroissant/_src/operation_graph/operations/download.py), [`Extract`](./mlcroissant/_src/operation_graph/operations/extract.py), etc).
|
|
73
|
+
|
|
74
|
+
Other important modules are:
|
|
75
|
+
|
|
76
|
+
- [`mlcroissant/_src/core`](./mlcroissant/_src/core) defines all needed core internals. For instance, [`Issues`](./mlcroissant/_src/core/issues.py) are a way to track errors and warning during the analysis of Croissant files.
|
|
77
|
+
- [`mlcroissant/__init__.py`](./mlcroissant/__init__.py) declares the public API with [`mlcroissant.Dataset`](./mlcroissant/_src/datasets.py).
|
|
78
|
+
|
|
79
|
+
For the full design, refer to the [design doc](https://docs.google.com/document/d/1zYQIUX9ae1sZOOBq9OCsJ8JW8-Ejy3NLSeqaI5LtOEM/edit?resourcekey=0-CK78DfFvF7fnufyZqF3h3Q) for an overview of the implementation.
|
|
80
|
+
|
|
81
|
+
## Contribute
|
|
82
|
+
|
|
83
|
+
All contributions are welcome! We even have [good first issues](https://github.com/mlcommons/croissant/issues?q=is%3Aissue+is%3Aopen+label%3A%22good+first+issue%22) to start in the project. Refer to the [GitHub project](https://github.com/orgs/mlcommons/projects/26) for more detailed user stories and read above how the repo is [designed](#design).
|
|
84
|
+
|
|
85
|
+
An easy way to contribute to `mlcroissant` is using Croissant's configured [codespaces](https://docs.github.com/en/codespaces/overview).
|
|
86
|
+
To start a codespace:
|
|
87
|
+
|
|
88
|
+
- On Croissant's main [repo page](https://github.com/mlcommons/croissant), click on the `<Code>` button and select the `Codespaces` tab. You can start a new codespace by clicking on the `+` sign on the left side of the tab. By default, the codespace will start on Croissant's `main` branch, unless you select otherwise from the branches drop-down menu on the left side.
|
|
89
|
+
- While building the environment, your codespaces will install all `mlcroissant`'s required dependencies - so that you can start coding right away! Of course, you can [further personalize](https://docs.github.com/en/codespaces/customizing-your-codespace/personalizing-github-codespaces-for-your-account) your codespace.
|
|
90
|
+
- To start contributing to Croissant:
|
|
91
|
+
- Create a new branch from the `Terminal` tab in the bottom panel of your codespace with `git checkout -b feature/my-awesome-new-feature`
|
|
92
|
+
- You can create new commits, and run most git commands from the `Source Control` tab in the left panel of your codespace. Alternatively, use the `Terminal` in the bottom panel of your codespace.
|
|
93
|
+
- Iterate on your code until all tests are green (you can run tests with `make pytest` or form the `Tests` tab in the left panel of your codespace).
|
|
94
|
+
- Open a pull request (PR) with the main branch of https://github.com/mlcommons/croissant, and ask for feedback!
|
|
95
|
+
|
|
96
|
+
Alternatively, you can contribute to `mlcroissant` using the "classic" GitHub workflow:
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
## Debug
|
|
100
|
+
|
|
101
|
+
You can debug the validation of the file using the `--debug` flag:
|
|
102
|
+
|
|
103
|
+
```bash
|
|
104
|
+
python scripts/validate.py --file ../../datasets/titanic/metadata.json --debug
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
This will:
|
|
108
|
+
1. print extra information, like the generated nodes;
|
|
109
|
+
2. save the generated structure graph to a folder indicated in the logs.
|
|
110
|
+
|
|
111
|
+
## Publishing wheels
|
|
112
|
+
|
|
113
|
+
Publishing is done manually.
|
|
114
|
+
We are in the process of setting up an automatic deployment with GitHub Actions.
|
|
115
|
+
|
|
116
|
+
1. Bump the version in `croissant/python/mlcroissant/pyproject.toml`.
|
|
117
|
+
1. Build locally:
|
|
118
|
+
```bash
|
|
119
|
+
python -m build
|
|
120
|
+
```
|
|
121
|
+
1. Upload to pypi.org.
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
"""Defines the public interface to the `mlcroissant` package."""
|
|
2
|
+
from mlcroissant._src import nodes
|
|
3
|
+
from mlcroissant._src.core import constants
|
|
4
|
+
from mlcroissant._src.core.issues import ValidationError
|
|
5
|
+
from mlcroissant._src.datasets import Dataset
|
|
6
|
+
from mlcroissant._src.datasets import Records
|
|
7
|
+
from mlcroissant._src.structure_graph.nodes.field import Field
|
|
8
|
+
|
|
9
|
+
__all__ = [
|
|
10
|
+
"constants",
|
|
11
|
+
"Dataset",
|
|
12
|
+
"Field",
|
|
13
|
+
"nodes",
|
|
14
|
+
"Records",
|
|
15
|
+
"ValidationError",
|
|
16
|
+
]
|
|
File without changes
|
|
File without changes
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""constants module."""
|
|
2
|
+
|
|
3
|
+
from etils import epath
|
|
4
|
+
import rdflib
|
|
5
|
+
from rdflib import namespace
|
|
6
|
+
from rdflib import term
|
|
7
|
+
|
|
8
|
+
# MLCommons-defined URIs (still draft).
|
|
9
|
+
ML_COMMONS = rdflib.Namespace("http://mlcommons.org/schema/")
|
|
10
|
+
ML_COMMONS_COLUMN = ML_COMMONS.column
|
|
11
|
+
ML_COMMONS_DATA = ML_COMMONS.data
|
|
12
|
+
ML_COMMONS_DATA_TYPE = ML_COMMONS.dataType
|
|
13
|
+
ML_COMMONS_DATA_TYPE_BOUNDING_BOX = ML_COMMONS.BoundingBox
|
|
14
|
+
ML_COMMONS_EXTRACT = ML_COMMONS.extract
|
|
15
|
+
ML_COMMONS_FILE_PROPERTY = ML_COMMONS.fileProperty
|
|
16
|
+
ML_COMMONS_FIELD = ML_COMMONS.field
|
|
17
|
+
ML_COMMONS_FIELD_TYPE = ML_COMMONS.Field
|
|
18
|
+
# ML_COMMONS.format is understood as the `format` method on the class Namespace.
|
|
19
|
+
ML_COMMONS_FORMAT = term.URIRef("http://mlcommons.org/schema/format")
|
|
20
|
+
ML_COMMONS_INCLUDES = ML_COMMONS.includes
|
|
21
|
+
ML_COMMONS_IS_ENUMERATION = ML_COMMONS.isEnumeration
|
|
22
|
+
ML_COMMONS_JSON_PATH = ML_COMMONS.jsonPath
|
|
23
|
+
ML_COMMONS_PARENT_FIELD = ML_COMMONS.parentField
|
|
24
|
+
ML_COMMONS_PATH = ML_COMMONS.path
|
|
25
|
+
ML_COMMONS_RECORD_SET = ML_COMMONS.recordSet
|
|
26
|
+
ML_COMMONS_RECORD_SET_TYPE = ML_COMMONS.RecordSet
|
|
27
|
+
ML_COMMONS_REFERENCES = ML_COMMONS.references
|
|
28
|
+
ML_COMMONS_REGEX = ML_COMMONS.regex
|
|
29
|
+
ML_COMMONS_REPEATED = ML_COMMONS.repeated
|
|
30
|
+
# ML_COMMONS.replace is understood as the `replace` method on the class Namespace.
|
|
31
|
+
ML_COMMONS_REPLACE = term.URIRef("http://mlcommons.org/schema/replace")
|
|
32
|
+
ML_COMMONS_SEPARATOR = ML_COMMONS.separator
|
|
33
|
+
ML_COMMONS_SOURCE = ML_COMMONS.source
|
|
34
|
+
ML_COMMONS_SUB_FIELD = ML_COMMONS.subField
|
|
35
|
+
ML_COMMONS_SUB_FIELD_TYPE = ML_COMMONS.SubField
|
|
36
|
+
ML_COMMONS_TRANSFORM = ML_COMMONS.transform
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
# RDF standard URIs.
|
|
40
|
+
# For "@type" key:
|
|
41
|
+
RDF_TYPE = namespace.RDF.type
|
|
42
|
+
|
|
43
|
+
# Schema.org standard URIs.
|
|
44
|
+
SCHEMA_ORG_CITATION = namespace.SDO.citation
|
|
45
|
+
SCHEMA_ORG_CONTAINED_IN = namespace.SDO.containedIn
|
|
46
|
+
SCHEMA_ORG_CONTENT_SIZE = namespace.SDO.contentSize
|
|
47
|
+
SCHEMA_ORG_CONTENT_URL = namespace.SDO.contentUrl
|
|
48
|
+
SCHEMA_ORG_DATASET = namespace.SDO.Dataset
|
|
49
|
+
SCHEMA_ORG_DATA_TYPE_BOOL = namespace.SDO.Boolean
|
|
50
|
+
SCHEMA_ORG_DATA_TYPE_DATE = namespace.SDO.Date
|
|
51
|
+
SCHEMA_ORG_DATA_TYPE_FLOAT = namespace.SDO.Float
|
|
52
|
+
SCHEMA_ORG_DATA_TYPE_IMAGE_OBJECT = namespace.SDO.ImageObject
|
|
53
|
+
SCHEMA_ORG_DATA_TYPE_INTEGER = namespace.SDO.Integer
|
|
54
|
+
SCHEMA_ORG_DATA_TYPE_TEXT = namespace.SDO.Text
|
|
55
|
+
SCHEMA_ORG_DATA_TYPE_URL = namespace.SDO.URL
|
|
56
|
+
SCHEMA_ORG_DESCRIPTION = namespace.SDO.description
|
|
57
|
+
SCHEMA_ORG_DISTRIBUTION = namespace.SDO.distribution
|
|
58
|
+
SCHEMA_ORG_EMAIL = namespace.SDO.email
|
|
59
|
+
SCHEMA_ORG_ENCODING_FORMAT = namespace.SDO.encodingFormat
|
|
60
|
+
SCHEMA_ORG_LICENSE = namespace.SDO.license
|
|
61
|
+
SCHEMA_ORG_NAME = namespace.SDO.name
|
|
62
|
+
SCHEMA_ORG_SHA256 = namespace.SDO.sha256
|
|
63
|
+
SCHEMA_ORG_URL = namespace.SDO.url
|
|
64
|
+
|
|
65
|
+
# Schema.org URIs that do not exist yet in the standard.
|
|
66
|
+
SCHEMA_ORG = rdflib.Namespace("https://schema.org/")
|
|
67
|
+
SCHEMA_ORG_KEY = SCHEMA_ORG.key
|
|
68
|
+
SCHEMA_ORG_FILE_OBJECT = SCHEMA_ORG.FileObject
|
|
69
|
+
SCHEMA_ORG_FILE_SET = SCHEMA_ORG.FileSet
|
|
70
|
+
SCHEMA_ORG_MD5 = SCHEMA_ORG.md5
|
|
71
|
+
|
|
72
|
+
TO_CROISSANT = {
|
|
73
|
+
ML_COMMONS_TRANSFORM: "transforms",
|
|
74
|
+
ML_COMMONS_COLUMN: "csv_column",
|
|
75
|
+
ML_COMMONS_DATA_TYPE: "data_type",
|
|
76
|
+
ML_COMMONS_DATA: "data",
|
|
77
|
+
ML_COMMONS_EXTRACT: "extract",
|
|
78
|
+
ML_COMMONS_FIELD: "field",
|
|
79
|
+
ML_COMMONS_FILE_PROPERTY: "file_property",
|
|
80
|
+
ML_COMMONS_FORMAT: "format",
|
|
81
|
+
ML_COMMONS_INCLUDES: "includes",
|
|
82
|
+
ML_COMMONS_JSON_PATH: "json_path",
|
|
83
|
+
ML_COMMONS_REFERENCES: "references",
|
|
84
|
+
ML_COMMONS_REGEX: "regex",
|
|
85
|
+
ML_COMMONS_REPLACE: "replace",
|
|
86
|
+
ML_COMMONS_SEPARATOR: "separator",
|
|
87
|
+
ML_COMMONS_SOURCE: "source",
|
|
88
|
+
SCHEMA_ORG_CITATION: "citation",
|
|
89
|
+
SCHEMA_ORG_CONTAINED_IN: "contained_in",
|
|
90
|
+
SCHEMA_ORG_CONTENT_SIZE: "content_size",
|
|
91
|
+
SCHEMA_ORG_CONTENT_URL: "content_url",
|
|
92
|
+
SCHEMA_ORG_DESCRIPTION: "description",
|
|
93
|
+
SCHEMA_ORG_DISTRIBUTION: "distribution",
|
|
94
|
+
SCHEMA_ORG_ENCODING_FORMAT: "encoding_format",
|
|
95
|
+
SCHEMA_ORG_LICENSE: "license",
|
|
96
|
+
SCHEMA_ORG_MD5: "md5",
|
|
97
|
+
SCHEMA_ORG_NAME: "name",
|
|
98
|
+
SCHEMA_ORG_SHA256: "sha256",
|
|
99
|
+
SCHEMA_ORG_URL: "url",
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
FROM_CROISSANT = {v: k for k, v in TO_CROISSANT.items()}
|
|
103
|
+
|
|
104
|
+
# Environment variables
|
|
105
|
+
CROISSANT_CACHE = epath.Path("~/.cache/croissant").expanduser()
|
|
106
|
+
DOWNLOAD_PATH = CROISSANT_CACHE / "download"
|
|
107
|
+
EXTRACT_PATH = CROISSANT_CACHE / "extract"
|
|
108
|
+
CROISSANT_GIT_USERNAME = "CROISSANT_GIT_USERNAME"
|
|
109
|
+
CROISSANT_GIT_PASSWORD = "CROISSANT_GIT_PASSWORD"
|
|
110
|
+
|
|
111
|
+
# Encoding formats
|
|
112
|
+
GIT_HTTPS_ENCODING_FORMAT = "git+https"
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""constants_test module."""
|
|
2
|
+
|
|
3
|
+
from mlcroissant._src.core.constants import TO_CROISSANT
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def test_to_croissant_values_are_unique():
|
|
7
|
+
deja_vu = {}
|
|
8
|
+
for key, value in TO_CROISSANT.items():
|
|
9
|
+
if value in deja_vu:
|
|
10
|
+
raise ValueError(
|
|
11
|
+
f"Keys {key} and {deja_vu[value]} define the same Croissant value:"
|
|
12
|
+
f" {value}."
|
|
13
|
+
)
|
|
14
|
+
deja_vu[value] = key
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""data_types module."""
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
|
|
5
|
+
from mlcroissant._src.core import constants
|
|
6
|
+
from mlcroissant._src.core.issues import Issues
|
|
7
|
+
from mlcroissant._src.core.types import Json
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def check_expected_type(issues: Issues, jsonld: Json, expected_type: str):
|
|
11
|
+
"""Checks that JSON-LD `jsonld` has "@type" == expected_type."""
|
|
12
|
+
node_name = jsonld.get(constants.SCHEMA_ORG_NAME, "<unknown node>")
|
|
13
|
+
node_type = jsonld.get("@type")
|
|
14
|
+
if node_type != expected_type:
|
|
15
|
+
issues.add_error(
|
|
16
|
+
f'"{node_name}" should have an attribute "@type": "{expected_type}". Got'
|
|
17
|
+
f" {node_type} instead."
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
EXPECTED_DATA_TYPES: dict[str, type] = {
|
|
22
|
+
constants.ML_COMMONS_DATA_TYPE_BOUNDING_BOX: (
|
|
23
|
+
constants.ML_COMMONS_DATA_TYPE_BOUNDING_BOX
|
|
24
|
+
),
|
|
25
|
+
constants.SCHEMA_ORG_DATA_TYPE_BOOL: bool,
|
|
26
|
+
constants.SCHEMA_ORG_DATA_TYPE_DATE: pd.Timestamp,
|
|
27
|
+
constants.SCHEMA_ORG_DATA_TYPE_FLOAT: float,
|
|
28
|
+
constants.SCHEMA_ORG_DATA_TYPE_IMAGE_OBJECT: (
|
|
29
|
+
constants.SCHEMA_ORG_DATA_TYPE_IMAGE_OBJECT
|
|
30
|
+
),
|
|
31
|
+
constants.SCHEMA_ORG_DATA_TYPE_INTEGER: int,
|
|
32
|
+
constants.SCHEMA_ORG_DATA_TYPE_TEXT: str,
|
|
33
|
+
constants.SCHEMA_ORG_DATA_TYPE_URL: str,
|
|
34
|
+
}
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""git module."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
|
|
5
|
+
from absl import logging
|
|
6
|
+
from etils import epath
|
|
7
|
+
|
|
8
|
+
from mlcroissant._src.core.optional import deps
|
|
9
|
+
from mlcroissant._src.core.path import Path
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def is_git_lfs_file(filepath: epath.Path) -> bool:
|
|
13
|
+
"""Returns whether a file a non-downloaded git-lfs file by checking its header.
|
|
14
|
+
|
|
15
|
+
An optimization of this function would be to only read the file if there is a git
|
|
16
|
+
repository in the distribution.
|
|
17
|
+
"""
|
|
18
|
+
with open(filepath, "rb") as file:
|
|
19
|
+
# Only read the first line of the file. In the future, this could be a problem,
|
|
20
|
+
# e.g. if we accept *.txt files and the file starts with the same header.
|
|
21
|
+
first_line = file.readline()
|
|
22
|
+
if first_line.startswith(b"version https://git-lfs.github.com/spec"):
|
|
23
|
+
return True
|
|
24
|
+
return False
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def download_git_lfs_file(file: Path):
|
|
28
|
+
"""Downloads a specific git-lfs file within its repo."""
|
|
29
|
+
# Path(filepath="/tmp/full/path.json", fullpath="path.json")
|
|
30
|
+
# => working_dir = "/tmp/full"
|
|
31
|
+
fullpath = os.fspath(file.fullpath)
|
|
32
|
+
working_dir = os.fspath(file.filepath).rsplit(fullpath)[0]
|
|
33
|
+
repo = deps.git.Git(working_dir)
|
|
34
|
+
logging.info(
|
|
35
|
+
"Downloading git-lfs file: %s in working dir: %s", fullpath, working_dir
|
|
36
|
+
)
|
|
37
|
+
repo.execute(["git", "lfs", "pull", "--include", fullpath])
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""Tests for git."""
|
|
2
|
+
|
|
3
|
+
import pathlib
|
|
4
|
+
import tempfile
|
|
5
|
+
from unittest import mock
|
|
6
|
+
|
|
7
|
+
from etils import epath
|
|
8
|
+
import git
|
|
9
|
+
|
|
10
|
+
from mlcroissant._src.core.git import download_git_lfs_file
|
|
11
|
+
from mlcroissant._src.core.git import is_git_lfs_file
|
|
12
|
+
from mlcroissant._src.core.path import Path
|
|
13
|
+
|
|
14
|
+
_GIT_LFS_CONTENT = lambda: """version https://git-lfs.github.com/spec/v1
|
|
15
|
+
oid sha256:5e2785fcd9098567a49d6e62e328923d955b307b6dcd0492f6234e96e670772a
|
|
16
|
+
size 309207547
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
_NON_GIT_LFS_CONTENT = lambda: """name,age
|
|
20
|
+
a,1
|
|
21
|
+
b,2
|
|
22
|
+
c,3"""
|
|
23
|
+
|
|
24
|
+
_NON_ASCII_CONTENT = lambda: (255).to_bytes(1, byteorder="big")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def test_is_git_lfs_file():
|
|
28
|
+
with tempfile.TemporaryDirectory() as tempdir:
|
|
29
|
+
tempdir = epath.Path(tempdir)
|
|
30
|
+
|
|
31
|
+
gitlfs_file = tempdir / "gitlfs"
|
|
32
|
+
gitlfs_file.write_text(_GIT_LFS_CONTENT())
|
|
33
|
+
assert is_git_lfs_file(gitlfs_file)
|
|
34
|
+
|
|
35
|
+
no_gitlfs_file = tempdir / "no-gitlfs"
|
|
36
|
+
no_gitlfs_file.write_text(_NON_GIT_LFS_CONTENT())
|
|
37
|
+
assert not is_git_lfs_file(no_gitlfs_file)
|
|
38
|
+
|
|
39
|
+
no_ascii_file = tempdir / "no-ascii"
|
|
40
|
+
no_ascii_file.write_bytes(_NON_ASCII_CONTENT())
|
|
41
|
+
assert not is_git_lfs_file(no_ascii_file)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def test_download_git_lfs_file():
|
|
45
|
+
file = Path(
|
|
46
|
+
filepath=epath.Path("/tmp/full/path.json"),
|
|
47
|
+
fullpath=pathlib.PurePath("path.json"),
|
|
48
|
+
)
|
|
49
|
+
with mock.patch.object(git, "Git", autospec=True) as git_mock:
|
|
50
|
+
download_git_lfs_file(file)
|
|
51
|
+
git_mock.assert_called_once_with("/tmp/full/")
|
|
52
|
+
git_mock.return_value.execute.assert_called_once_with(
|
|
53
|
+
["git", "lfs", "pull", "--include", "path.json"]
|
|
54
|
+
)
|
|
File without changes
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
"""Utils for graphs."""
|
|
2
|
+
|
|
3
|
+
import time
|
|
4
|
+
|
|
5
|
+
import networkx as nx
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def pretty_print_graph(graph: nx.Graph, simplify=False):
|
|
9
|
+
"""Pretty prints a NetworkX graph.
|
|
10
|
+
|
|
11
|
+
Warning: this function is for debugging purposes only.
|
|
12
|
+
|
|
13
|
+
Args:
|
|
14
|
+
graph: Any NetworkX graph.
|
|
15
|
+
simplify: Whether to print a simplified version of nodes.
|
|
16
|
+
"""
|
|
17
|
+
if simplify:
|
|
18
|
+
simple_graph = nx.Graph()
|
|
19
|
+
for x, y in graph.edges():
|
|
20
|
+
x = getattr(x, "uid", x)
|
|
21
|
+
y = getattr(y, "uid", y)
|
|
22
|
+
simple_graph.add_edge(x, y)
|
|
23
|
+
graph = simple_graph
|
|
24
|
+
agraph = nx.nx_agraph.to_agraph(graph)
|
|
25
|
+
agraph.layout(prog="dot")
|
|
26
|
+
temporary_file = f"/tmp/graph_{time.time()}.png"
|
|
27
|
+
agraph.draw(temporary_file, args="-Gnodesep=0.01 -Gfont_size=1", prog="dot")
|
|
28
|
+
print(f"Generated a graph and saved it in: {temporary_file}")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def print_graph_traversal(graph: nx.Graph):
|
|
32
|
+
"""Pretty prints a NetworkX graph.
|
|
33
|
+
|
|
34
|
+
Warning: this function is for debugging purposes only.
|
|
35
|
+
|
|
36
|
+
Args:
|
|
37
|
+
graph: Any NetworkX graph.
|
|
38
|
+
"""
|
|
39
|
+
visited = {}
|
|
40
|
+
print("--- Graph traversal ---")
|
|
41
|
+
for start, end, _ in nx.edge_bfs(graph):
|
|
42
|
+
for node in [start, end]:
|
|
43
|
+
if node.name not in visited:
|
|
44
|
+
print(f"Visited: {node.name}")
|
|
45
|
+
visited[node.name] = True
|
|
46
|
+
print("Done traversing the graph.")
|