mlcroissant 0.0.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. mlcroissant-0.0.2/PKG-INFO +153 -0
  2. mlcroissant-0.0.2/README.md +121 -0
  3. mlcroissant-0.0.2/mlcroissant/__init__.py +16 -0
  4. mlcroissant-0.0.2/mlcroissant/_src/__init__.py +0 -0
  5. mlcroissant-0.0.2/mlcroissant/_src/core/__init__.py +0 -0
  6. mlcroissant-0.0.2/mlcroissant/_src/core/constants.py +112 -0
  7. mlcroissant-0.0.2/mlcroissant/_src/core/constants_test.py +14 -0
  8. mlcroissant-0.0.2/mlcroissant/_src/core/data_types.py +34 -0
  9. mlcroissant-0.0.2/mlcroissant/_src/core/git.py +37 -0
  10. mlcroissant-0.0.2/mlcroissant/_src/core/git_test.py +54 -0
  11. mlcroissant-0.0.2/mlcroissant/_src/core/graphs/__init__.py +0 -0
  12. mlcroissant-0.0.2/mlcroissant/_src/core/graphs/utils.py +46 -0
  13. mlcroissant-0.0.2/mlcroissant/_src/core/issues.py +83 -0
  14. mlcroissant-0.0.2/mlcroissant/_src/core/issues_test.py +40 -0
  15. mlcroissant-0.0.2/mlcroissant/_src/core/json_ld.py +221 -0
  16. mlcroissant-0.0.2/mlcroissant/_src/core/json_ld_test.py +58 -0
  17. mlcroissant-0.0.2/mlcroissant/_src/core/optional.py +90 -0
  18. mlcroissant-0.0.2/mlcroissant/_src/core/optional_test.py +27 -0
  19. mlcroissant-0.0.2/mlcroissant/_src/core/path.py +50 -0
  20. mlcroissant-0.0.2/mlcroissant/_src/core/types.py +5 -0
  21. mlcroissant-0.0.2/mlcroissant/_src/datasets.py +116 -0
  22. mlcroissant-0.0.2/mlcroissant/_src/datasets_test.py +125 -0
  23. mlcroissant-0.0.2/mlcroissant/_src/nodes.py +20 -0
  24. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/__init__.py +5 -0
  25. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/base_operation.py +33 -0
  26. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/execute.py +111 -0
  27. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/execute_test.py +30 -0
  28. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/graph.py +265 -0
  29. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/__init__.py +25 -0
  30. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/concatenate.py +30 -0
  31. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/concatenate_test.py +9 -0
  32. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/data.py +20 -0
  33. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/data_test.py +9 -0
  34. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/download.py +170 -0
  35. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/download_test.py +70 -0
  36. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/extract.py +60 -0
  37. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/extract_test.py +29 -0
  38. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/field.py +67 -0
  39. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/field_test.py +9 -0
  40. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/filter.py +42 -0
  41. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/filter_test.py +9 -0
  42. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/group.py +16 -0
  43. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/group_test.py +9 -0
  44. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/init.py +16 -0
  45. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/init_test.py +9 -0
  46. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/join.py +67 -0
  47. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/join_test.py +9 -0
  48. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/local_directory.py +25 -0
  49. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/local_directory_test.py +11 -0
  50. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/parse_json.py +20 -0
  51. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/parse_json_test.py +29 -0
  52. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/read.py +88 -0
  53. mlcroissant-0.0.2/mlcroissant/_src/operation_graph/operations/read_test.py +45 -0
  54. mlcroissant-0.0.2/mlcroissant/_src/structure_graph/__init__.py +0 -0
  55. mlcroissant-0.0.2/mlcroissant/_src/structure_graph/base_node.py +204 -0
  56. mlcroissant-0.0.2/mlcroissant/_src/structure_graph/base_node_test.py +81 -0
  57. mlcroissant-0.0.2/mlcroissant/_src/structure_graph/graph.py +142 -0
  58. mlcroissant-0.0.2/mlcroissant/_src/structure_graph/graph_test.py +47 -0
  59. mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/__init__.py +0 -0
  60. mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/field.py +196 -0
  61. mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/field_test.py +52 -0
  62. mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/file_object.py +95 -0
  63. mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/file_object_test.py +58 -0
  64. mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/file_set.py +76 -0
  65. mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/file_set_test.py +53 -0
  66. mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/metadata.py +197 -0
  67. mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/metadata_test.py +63 -0
  68. mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/rdf.py +41 -0
  69. mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/rdf_test.py +21 -0
  70. mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/record_set.py +125 -0
  71. mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/record_set_test.py +107 -0
  72. mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/source.py +321 -0
  73. mlcroissant-0.0.2/mlcroissant/_src/structure_graph/nodes/source_test.py +268 -0
  74. mlcroissant-0.0.2/mlcroissant/_src/tests/__init__.py +0 -0
  75. mlcroissant-0.0.2/mlcroissant/_src/tests/nodes.py +85 -0
  76. mlcroissant-0.0.2/mlcroissant/_src/tests/records.py +38 -0
  77. mlcroissant-0.0.2/mlcroissant/_src/tests/records_test.py +33 -0
  78. mlcroissant-0.0.2/mlcroissant.egg-info/PKG-INFO +153 -0
  79. mlcroissant-0.0.2/mlcroissant.egg-info/SOURCES.txt +94 -0
  80. mlcroissant-0.0.2/mlcroissant.egg-info/dependency_links.txt +1 -0
  81. mlcroissant-0.0.2/mlcroissant.egg-info/requires.txt +29 -0
  82. mlcroissant-0.0.2/mlcroissant.egg-info/top_level.txt +3 -0
  83. mlcroissant-0.0.2/pyproject.toml +83 -0
  84. mlcroissant-0.0.2/scripts/__init__.py +1 -0
  85. mlcroissant-0.0.2/scripts/from_huggingface_to_croissant.py +192 -0
  86. mlcroissant-0.0.2/scripts/from_huggingface_to_croissant_test.py +34 -0
  87. mlcroissant-0.0.2/scripts/load.py +106 -0
  88. mlcroissant-0.0.2/scripts/load_test.py +20 -0
  89. mlcroissant-0.0.2/scripts/migrations/__init__.py +1 -0
  90. mlcroissant-0.0.2/scripts/migrations/migrate.py +136 -0
  91. mlcroissant-0.0.2/scripts/migrations/previous/202307171508.py +104 -0
  92. mlcroissant-0.0.2/scripts/migrations/previous/202307201041.py +25 -0
  93. mlcroissant-0.0.2/scripts/migrations/previous/202308312000.py +44 -0
  94. mlcroissant-0.0.2/scripts/migrations/previous/202309061700.py +38 -0
  95. mlcroissant-0.0.2/scripts/validate.py +50 -0
  96. mlcroissant-0.0.2/setup.cfg +4 -0
@@ -0,0 +1,153 @@
1
+ Metadata-Version: 2.1
2
+ Name: mlcroissant
3
+ Version: 0.0.2
4
+ Summary: MLCommons datasets format.
5
+ Author: Joaquin Vanschoren, Jos van der Velde, Omar Benjelloun, Peter Mattson, Pieter Gijsbers, Pierre Marcenac, Pierre Ruyssen, Prabhant Singh
6
+ Description-Content-Type: text/markdown
7
+ Requires-Dist: absl-py
8
+ Requires-Dist: etils[epath]
9
+ Requires-Dist: jsonpath-rw
10
+ Requires-Dist: networkx
11
+ Requires-Dist: pandas
12
+ Requires-Dist: rdflib
13
+ Requires-Dist: requests
14
+ Requires-Dist: tqdm
15
+ Provides-Extra: dev
16
+ Requires-Dist: black; extra == "dev"
17
+ Requires-Dist: datasets; extra == "dev"
18
+ Requires-Dist: flake8-docstrings; extra == "dev"
19
+ Requires-Dist: mlcroissant[git]; extra == "dev"
20
+ Requires-Dist: mlcroissant[image]; extra == "dev"
21
+ Requires-Dist: mlcroissant[parquet]; extra == "dev"
22
+ Requires-Dist: pyflakes; extra == "dev"
23
+ Requires-Dist: pylint; extra == "dev"
24
+ Requires-Dist: pytest; extra == "dev"
25
+ Requires-Dist: pytype; extra == "dev"
26
+ Provides-Extra: git
27
+ Requires-Dist: GitPython; extra == "git"
28
+ Provides-Extra: image
29
+ Requires-Dist: Pillow; extra == "image"
30
+ Provides-Extra: parquet
31
+ Requires-Dist: pyarrow; extra == "parquet"
32
+
33
+ # mlcroissant 🥐
34
+
35
+ Discover `mlcroissant 🥐` with this
36
+ [introduction tutorial in Google Colab](https://colab.sandbox.google.com/github/mlcommons/croissant/blob/main/python/mlcroissant/recipes/introduction.ipynb).
37
+
38
+ ## Python requirements
39
+
40
+ Python version >= 3.10.
41
+
42
+ If you do not have a Python environment:
43
+
44
+ ```bash
45
+ python3 -m venv ~/py3
46
+ source ~/py3/bin/activate
47
+ ```
48
+
49
+ ## Install
50
+
51
+ ```bash
52
+ python -m pip install ".[dev]"
53
+ ```
54
+
55
+ ## Verify/load a Croissant dataset
56
+
57
+ ```bash
58
+ python scripts/validate.py --file ../../datasets/titanic/metadata.json
59
+ ```
60
+
61
+ The command:
62
+
63
+ - Exits with 0, prints `Done` and displays encountered warnings, when no error was found in the file.
64
+ - Exits with 1 and displays all encountered errors/warnings, otherwise.
65
+
66
+ Similarly, you can generate a dataset by launching:
67
+
68
+ ```bash
69
+ python scripts/load.py \
70
+ --file ../../datasets/titanic/metadata.json \
71
+ --record_set passengers \
72
+ --num_records 10
73
+ ```
74
+
75
+ ## Programmatically build JSON-LD files
76
+
77
+ You can programmatically build Croissant JSON-LD files using the Python API.
78
+
79
+ ```python
80
+ import mlcroissant as mlc
81
+ metadata=mlc.nodes.Metadata(
82
+ name="...",
83
+ )
84
+ metadata.to_json() # this returns the JSON-LD file.
85
+ ```
86
+
87
+ For a full working example, refer to
88
+ [the script to convert Hugging Face datasets to Croissant files](./scripts/from_huggingface_to_croissant.py).
89
+ This script uses the Python API to programmatically build JSON-LD files.
90
+
91
+ ## Run tests
92
+
93
+ All tests can be run from the Makefile:
94
+
95
+ ```bash
96
+ make tests
97
+ ```
98
+
99
+ ## Design
100
+
101
+ The most important modules in the library are:
102
+
103
+ - [`mlcroissant/_src/structure_graph`](./mlcroissant/_src/structure_graph/graph.py) is responsible for the **static analysis** of the Croissant files. We convert Croissant files to a Python representation called "**structure graph**" (using [NetworkX](https://networkx.org/)). In the process, we catch any static analysis issues (e.g., a missing mandatory field or a logic problem in the file).
104
+ - [`mlcroissant/_src/operation_graph`](./mlcroissant/_src/operation_graph/graph.py) is responsible for the **dynamic analysis** of the Croissant files (i.e., actually loading the dataset by yielding examples). We convert the structure graph into an "**operation graph**". Operations are the unit transformations that allow to build the dataset (like [`Download`](./mlcroissant/_src/operation_graph/operations/download.py), [`Extract`](./mlcroissant/_src/operation_graph/operations/extract.py), etc).
105
+
106
+ Other important modules are:
107
+
108
+ - [`mlcroissant/_src/core`](./mlcroissant/_src/core) defines all needed core internals. For instance, [`Issues`](./mlcroissant/_src/core/issues.py) are a way to track errors and warning during the analysis of Croissant files.
109
+ - [`mlcroissant/__init__.py`](./mlcroissant/__init__.py) declares the public API with [`mlcroissant.Dataset`](./mlcroissant/_src/datasets.py).
110
+
111
+ For the full design, refer to the [design doc](https://docs.google.com/document/d/1zYQIUX9ae1sZOOBq9OCsJ8JW8-Ejy3NLSeqaI5LtOEM/edit?resourcekey=0-CK78DfFvF7fnufyZqF3h3Q) for an overview of the implementation.
112
+
113
+ ## Contribute
114
+
115
+ All contributions are welcome! We even have [good first issues](https://github.com/mlcommons/croissant/issues?q=is%3Aissue+is%3Aopen+label%3A%22good+first+issue%22) to start in the project. Refer to the [GitHub project](https://github.com/orgs/mlcommons/projects/26) for more detailed user stories and read above how the repo is [designed](#design).
116
+
117
+ An easy way to contribute to `mlcroissant` is using Croissant's configured [codespaces](https://docs.github.com/en/codespaces/overview).
118
+ To start a codespace:
119
+
120
+ - On Croissant's main [repo page](https://github.com/mlcommons/croissant), click on the `<Code>` button and select the `Codespaces` tab. You can start a new codespace by clicking on the `+` sign on the left side of the tab. By default, the codespace will start on Croissant's `main` branch, unless you select otherwise from the branches drop-down menu on the left side.
121
+ - While building the environment, your codespaces will install all `mlcroissant`'s required dependencies - so that you can start coding right away! Of course, you can [further personalize](https://docs.github.com/en/codespaces/customizing-your-codespace/personalizing-github-codespaces-for-your-account) your codespace.
122
+ - To start contributing to Croissant:
123
+ - Create a new branch from the `Terminal` tab in the bottom panel of your codespace with `git checkout -b feature/my-awesome-new-feature`
124
+ - You can create new commits, and run most git commands from the `Source Control` tab in the left panel of your codespace. Alternatively, use the `Terminal` in the bottom panel of your codespace.
125
+ - Iterate on your code until all tests are green (you can run tests with `make pytest` or form the `Tests` tab in the left panel of your codespace).
126
+ - Open a pull request (PR) with the main branch of https://github.com/mlcommons/croissant, and ask for feedback!
127
+
128
+ Alternatively, you can contribute to `mlcroissant` using the "classic" GitHub workflow:
129
+
130
+
131
+ ## Debug
132
+
133
+ You can debug the validation of the file using the `--debug` flag:
134
+
135
+ ```bash
136
+ python scripts/validate.py --file ../../datasets/titanic/metadata.json --debug
137
+ ```
138
+
139
+ This will:
140
+ 1. print extra information, like the generated nodes;
141
+ 2. save the generated structure graph to a folder indicated in the logs.
142
+
143
+ ## Publishing wheels
144
+
145
+ Publishing is done manually.
146
+ We are in the process of setting up an automatic deployment with GitHub Actions.
147
+
148
+ 1. Bump the version in `croissant/python/mlcroissant/pyproject.toml`.
149
+ 1. Build locally:
150
+ ```bash
151
+ python -m build
152
+ ```
153
+ 1. Upload to pypi.org.
@@ -0,0 +1,121 @@
1
+ # mlcroissant 🥐
2
+
3
+ Discover `mlcroissant 🥐` with this
4
+ [introduction tutorial in Google Colab](https://colab.sandbox.google.com/github/mlcommons/croissant/blob/main/python/mlcroissant/recipes/introduction.ipynb).
5
+
6
+ ## Python requirements
7
+
8
+ Python version >= 3.10.
9
+
10
+ If you do not have a Python environment:
11
+
12
+ ```bash
13
+ python3 -m venv ~/py3
14
+ source ~/py3/bin/activate
15
+ ```
16
+
17
+ ## Install
18
+
19
+ ```bash
20
+ python -m pip install ".[dev]"
21
+ ```
22
+
23
+ ## Verify/load a Croissant dataset
24
+
25
+ ```bash
26
+ python scripts/validate.py --file ../../datasets/titanic/metadata.json
27
+ ```
28
+
29
+ The command:
30
+
31
+ - Exits with 0, prints `Done` and displays encountered warnings, when no error was found in the file.
32
+ - Exits with 1 and displays all encountered errors/warnings, otherwise.
33
+
34
+ Similarly, you can generate a dataset by launching:
35
+
36
+ ```bash
37
+ python scripts/load.py \
38
+ --file ../../datasets/titanic/metadata.json \
39
+ --record_set passengers \
40
+ --num_records 10
41
+ ```
42
+
43
+ ## Programmatically build JSON-LD files
44
+
45
+ You can programmatically build Croissant JSON-LD files using the Python API.
46
+
47
+ ```python
48
+ import mlcroissant as mlc
49
+ metadata=mlc.nodes.Metadata(
50
+ name="...",
51
+ )
52
+ metadata.to_json() # this returns the JSON-LD file.
53
+ ```
54
+
55
+ For a full working example, refer to
56
+ [the script to convert Hugging Face datasets to Croissant files](./scripts/from_huggingface_to_croissant.py).
57
+ This script uses the Python API to programmatically build JSON-LD files.
58
+
59
+ ## Run tests
60
+
61
+ All tests can be run from the Makefile:
62
+
63
+ ```bash
64
+ make tests
65
+ ```
66
+
67
+ ## Design
68
+
69
+ The most important modules in the library are:
70
+
71
+ - [`mlcroissant/_src/structure_graph`](./mlcroissant/_src/structure_graph/graph.py) is responsible for the **static analysis** of the Croissant files. We convert Croissant files to a Python representation called "**structure graph**" (using [NetworkX](https://networkx.org/)). In the process, we catch any static analysis issues (e.g., a missing mandatory field or a logic problem in the file).
72
+ - [`mlcroissant/_src/operation_graph`](./mlcroissant/_src/operation_graph/graph.py) is responsible for the **dynamic analysis** of the Croissant files (i.e., actually loading the dataset by yielding examples). We convert the structure graph into an "**operation graph**". Operations are the unit transformations that allow to build the dataset (like [`Download`](./mlcroissant/_src/operation_graph/operations/download.py), [`Extract`](./mlcroissant/_src/operation_graph/operations/extract.py), etc).
73
+
74
+ Other important modules are:
75
+
76
+ - [`mlcroissant/_src/core`](./mlcroissant/_src/core) defines all needed core internals. For instance, [`Issues`](./mlcroissant/_src/core/issues.py) are a way to track errors and warning during the analysis of Croissant files.
77
+ - [`mlcroissant/__init__.py`](./mlcroissant/__init__.py) declares the public API with [`mlcroissant.Dataset`](./mlcroissant/_src/datasets.py).
78
+
79
+ For the full design, refer to the [design doc](https://docs.google.com/document/d/1zYQIUX9ae1sZOOBq9OCsJ8JW8-Ejy3NLSeqaI5LtOEM/edit?resourcekey=0-CK78DfFvF7fnufyZqF3h3Q) for an overview of the implementation.
80
+
81
+ ## Contribute
82
+
83
+ All contributions are welcome! We even have [good first issues](https://github.com/mlcommons/croissant/issues?q=is%3Aissue+is%3Aopen+label%3A%22good+first+issue%22) to start in the project. Refer to the [GitHub project](https://github.com/orgs/mlcommons/projects/26) for more detailed user stories and read above how the repo is [designed](#design).
84
+
85
+ An easy way to contribute to `mlcroissant` is using Croissant's configured [codespaces](https://docs.github.com/en/codespaces/overview).
86
+ To start a codespace:
87
+
88
+ - On Croissant's main [repo page](https://github.com/mlcommons/croissant), click on the `<Code>` button and select the `Codespaces` tab. You can start a new codespace by clicking on the `+` sign on the left side of the tab. By default, the codespace will start on Croissant's `main` branch, unless you select otherwise from the branches drop-down menu on the left side.
89
+ - While building the environment, your codespaces will install all `mlcroissant`'s required dependencies - so that you can start coding right away! Of course, you can [further personalize](https://docs.github.com/en/codespaces/customizing-your-codespace/personalizing-github-codespaces-for-your-account) your codespace.
90
+ - To start contributing to Croissant:
91
+ - Create a new branch from the `Terminal` tab in the bottom panel of your codespace with `git checkout -b feature/my-awesome-new-feature`
92
+ - You can create new commits, and run most git commands from the `Source Control` tab in the left panel of your codespace. Alternatively, use the `Terminal` in the bottom panel of your codespace.
93
+ - Iterate on your code until all tests are green (you can run tests with `make pytest` or form the `Tests` tab in the left panel of your codespace).
94
+ - Open a pull request (PR) with the main branch of https://github.com/mlcommons/croissant, and ask for feedback!
95
+
96
+ Alternatively, you can contribute to `mlcroissant` using the "classic" GitHub workflow:
97
+
98
+
99
+ ## Debug
100
+
101
+ You can debug the validation of the file using the `--debug` flag:
102
+
103
+ ```bash
104
+ python scripts/validate.py --file ../../datasets/titanic/metadata.json --debug
105
+ ```
106
+
107
+ This will:
108
+ 1. print extra information, like the generated nodes;
109
+ 2. save the generated structure graph to a folder indicated in the logs.
110
+
111
+ ## Publishing wheels
112
+
113
+ Publishing is done manually.
114
+ We are in the process of setting up an automatic deployment with GitHub Actions.
115
+
116
+ 1. Bump the version in `croissant/python/mlcroissant/pyproject.toml`.
117
+ 1. Build locally:
118
+ ```bash
119
+ python -m build
120
+ ```
121
+ 1. Upload to pypi.org.
@@ -0,0 +1,16 @@
1
+ """Defines the public interface to the `mlcroissant` package."""
2
+ from mlcroissant._src import nodes
3
+ from mlcroissant._src.core import constants
4
+ from mlcroissant._src.core.issues import ValidationError
5
+ from mlcroissant._src.datasets import Dataset
6
+ from mlcroissant._src.datasets import Records
7
+ from mlcroissant._src.structure_graph.nodes.field import Field
8
+
9
+ __all__ = [
10
+ "constants",
11
+ "Dataset",
12
+ "Field",
13
+ "nodes",
14
+ "Records",
15
+ "ValidationError",
16
+ ]
File without changes
File without changes
@@ -0,0 +1,112 @@
1
+ """constants module."""
2
+
3
+ from etils import epath
4
+ import rdflib
5
+ from rdflib import namespace
6
+ from rdflib import term
7
+
8
+ # MLCommons-defined URIs (still draft).
9
+ ML_COMMONS = rdflib.Namespace("http://mlcommons.org/schema/")
10
+ ML_COMMONS_COLUMN = ML_COMMONS.column
11
+ ML_COMMONS_DATA = ML_COMMONS.data
12
+ ML_COMMONS_DATA_TYPE = ML_COMMONS.dataType
13
+ ML_COMMONS_DATA_TYPE_BOUNDING_BOX = ML_COMMONS.BoundingBox
14
+ ML_COMMONS_EXTRACT = ML_COMMONS.extract
15
+ ML_COMMONS_FILE_PROPERTY = ML_COMMONS.fileProperty
16
+ ML_COMMONS_FIELD = ML_COMMONS.field
17
+ ML_COMMONS_FIELD_TYPE = ML_COMMONS.Field
18
+ # ML_COMMONS.format is understood as the `format` method on the class Namespace.
19
+ ML_COMMONS_FORMAT = term.URIRef("http://mlcommons.org/schema/format")
20
+ ML_COMMONS_INCLUDES = ML_COMMONS.includes
21
+ ML_COMMONS_IS_ENUMERATION = ML_COMMONS.isEnumeration
22
+ ML_COMMONS_JSON_PATH = ML_COMMONS.jsonPath
23
+ ML_COMMONS_PARENT_FIELD = ML_COMMONS.parentField
24
+ ML_COMMONS_PATH = ML_COMMONS.path
25
+ ML_COMMONS_RECORD_SET = ML_COMMONS.recordSet
26
+ ML_COMMONS_RECORD_SET_TYPE = ML_COMMONS.RecordSet
27
+ ML_COMMONS_REFERENCES = ML_COMMONS.references
28
+ ML_COMMONS_REGEX = ML_COMMONS.regex
29
+ ML_COMMONS_REPEATED = ML_COMMONS.repeated
30
+ # ML_COMMONS.replace is understood as the `replace` method on the class Namespace.
31
+ ML_COMMONS_REPLACE = term.URIRef("http://mlcommons.org/schema/replace")
32
+ ML_COMMONS_SEPARATOR = ML_COMMONS.separator
33
+ ML_COMMONS_SOURCE = ML_COMMONS.source
34
+ ML_COMMONS_SUB_FIELD = ML_COMMONS.subField
35
+ ML_COMMONS_SUB_FIELD_TYPE = ML_COMMONS.SubField
36
+ ML_COMMONS_TRANSFORM = ML_COMMONS.transform
37
+
38
+
39
+ # RDF standard URIs.
40
+ # For "@type" key:
41
+ RDF_TYPE = namespace.RDF.type
42
+
43
+ # Schema.org standard URIs.
44
+ SCHEMA_ORG_CITATION = namespace.SDO.citation
45
+ SCHEMA_ORG_CONTAINED_IN = namespace.SDO.containedIn
46
+ SCHEMA_ORG_CONTENT_SIZE = namespace.SDO.contentSize
47
+ SCHEMA_ORG_CONTENT_URL = namespace.SDO.contentUrl
48
+ SCHEMA_ORG_DATASET = namespace.SDO.Dataset
49
+ SCHEMA_ORG_DATA_TYPE_BOOL = namespace.SDO.Boolean
50
+ SCHEMA_ORG_DATA_TYPE_DATE = namespace.SDO.Date
51
+ SCHEMA_ORG_DATA_TYPE_FLOAT = namespace.SDO.Float
52
+ SCHEMA_ORG_DATA_TYPE_IMAGE_OBJECT = namespace.SDO.ImageObject
53
+ SCHEMA_ORG_DATA_TYPE_INTEGER = namespace.SDO.Integer
54
+ SCHEMA_ORG_DATA_TYPE_TEXT = namespace.SDO.Text
55
+ SCHEMA_ORG_DATA_TYPE_URL = namespace.SDO.URL
56
+ SCHEMA_ORG_DESCRIPTION = namespace.SDO.description
57
+ SCHEMA_ORG_DISTRIBUTION = namespace.SDO.distribution
58
+ SCHEMA_ORG_EMAIL = namespace.SDO.email
59
+ SCHEMA_ORG_ENCODING_FORMAT = namespace.SDO.encodingFormat
60
+ SCHEMA_ORG_LICENSE = namespace.SDO.license
61
+ SCHEMA_ORG_NAME = namespace.SDO.name
62
+ SCHEMA_ORG_SHA256 = namespace.SDO.sha256
63
+ SCHEMA_ORG_URL = namespace.SDO.url
64
+
65
+ # Schema.org URIs that do not exist yet in the standard.
66
+ SCHEMA_ORG = rdflib.Namespace("https://schema.org/")
67
+ SCHEMA_ORG_KEY = SCHEMA_ORG.key
68
+ SCHEMA_ORG_FILE_OBJECT = SCHEMA_ORG.FileObject
69
+ SCHEMA_ORG_FILE_SET = SCHEMA_ORG.FileSet
70
+ SCHEMA_ORG_MD5 = SCHEMA_ORG.md5
71
+
72
+ TO_CROISSANT = {
73
+ ML_COMMONS_TRANSFORM: "transforms",
74
+ ML_COMMONS_COLUMN: "csv_column",
75
+ ML_COMMONS_DATA_TYPE: "data_type",
76
+ ML_COMMONS_DATA: "data",
77
+ ML_COMMONS_EXTRACT: "extract",
78
+ ML_COMMONS_FIELD: "field",
79
+ ML_COMMONS_FILE_PROPERTY: "file_property",
80
+ ML_COMMONS_FORMAT: "format",
81
+ ML_COMMONS_INCLUDES: "includes",
82
+ ML_COMMONS_JSON_PATH: "json_path",
83
+ ML_COMMONS_REFERENCES: "references",
84
+ ML_COMMONS_REGEX: "regex",
85
+ ML_COMMONS_REPLACE: "replace",
86
+ ML_COMMONS_SEPARATOR: "separator",
87
+ ML_COMMONS_SOURCE: "source",
88
+ SCHEMA_ORG_CITATION: "citation",
89
+ SCHEMA_ORG_CONTAINED_IN: "contained_in",
90
+ SCHEMA_ORG_CONTENT_SIZE: "content_size",
91
+ SCHEMA_ORG_CONTENT_URL: "content_url",
92
+ SCHEMA_ORG_DESCRIPTION: "description",
93
+ SCHEMA_ORG_DISTRIBUTION: "distribution",
94
+ SCHEMA_ORG_ENCODING_FORMAT: "encoding_format",
95
+ SCHEMA_ORG_LICENSE: "license",
96
+ SCHEMA_ORG_MD5: "md5",
97
+ SCHEMA_ORG_NAME: "name",
98
+ SCHEMA_ORG_SHA256: "sha256",
99
+ SCHEMA_ORG_URL: "url",
100
+ }
101
+
102
+ FROM_CROISSANT = {v: k for k, v in TO_CROISSANT.items()}
103
+
104
+ # Environment variables
105
+ CROISSANT_CACHE = epath.Path("~/.cache/croissant").expanduser()
106
+ DOWNLOAD_PATH = CROISSANT_CACHE / "download"
107
+ EXTRACT_PATH = CROISSANT_CACHE / "extract"
108
+ CROISSANT_GIT_USERNAME = "CROISSANT_GIT_USERNAME"
109
+ CROISSANT_GIT_PASSWORD = "CROISSANT_GIT_PASSWORD"
110
+
111
+ # Encoding formats
112
+ GIT_HTTPS_ENCODING_FORMAT = "git+https"
@@ -0,0 +1,14 @@
1
+ """constants_test module."""
2
+
3
+ from mlcroissant._src.core.constants import TO_CROISSANT
4
+
5
+
6
+ def test_to_croissant_values_are_unique():
7
+ deja_vu = {}
8
+ for key, value in TO_CROISSANT.items():
9
+ if value in deja_vu:
10
+ raise ValueError(
11
+ f"Keys {key} and {deja_vu[value]} define the same Croissant value:"
12
+ f" {value}."
13
+ )
14
+ deja_vu[value] = key
@@ -0,0 +1,34 @@
1
+ """data_types module."""
2
+
3
+ import pandas as pd
4
+
5
+ from mlcroissant._src.core import constants
6
+ from mlcroissant._src.core.issues import Issues
7
+ from mlcroissant._src.core.types import Json
8
+
9
+
10
+ def check_expected_type(issues: Issues, jsonld: Json, expected_type: str):
11
+ """Checks that JSON-LD `jsonld` has "@type" == expected_type."""
12
+ node_name = jsonld.get(constants.SCHEMA_ORG_NAME, "<unknown node>")
13
+ node_type = jsonld.get("@type")
14
+ if node_type != expected_type:
15
+ issues.add_error(
16
+ f'"{node_name}" should have an attribute "@type": "{expected_type}". Got'
17
+ f" {node_type} instead."
18
+ )
19
+
20
+
21
+ EXPECTED_DATA_TYPES: dict[str, type] = {
22
+ constants.ML_COMMONS_DATA_TYPE_BOUNDING_BOX: (
23
+ constants.ML_COMMONS_DATA_TYPE_BOUNDING_BOX
24
+ ),
25
+ constants.SCHEMA_ORG_DATA_TYPE_BOOL: bool,
26
+ constants.SCHEMA_ORG_DATA_TYPE_DATE: pd.Timestamp,
27
+ constants.SCHEMA_ORG_DATA_TYPE_FLOAT: float,
28
+ constants.SCHEMA_ORG_DATA_TYPE_IMAGE_OBJECT: (
29
+ constants.SCHEMA_ORG_DATA_TYPE_IMAGE_OBJECT
30
+ ),
31
+ constants.SCHEMA_ORG_DATA_TYPE_INTEGER: int,
32
+ constants.SCHEMA_ORG_DATA_TYPE_TEXT: str,
33
+ constants.SCHEMA_ORG_DATA_TYPE_URL: str,
34
+ }
@@ -0,0 +1,37 @@
1
+ """git module."""
2
+
3
+ import os
4
+
5
+ from absl import logging
6
+ from etils import epath
7
+
8
+ from mlcroissant._src.core.optional import deps
9
+ from mlcroissant._src.core.path import Path
10
+
11
+
12
+ def is_git_lfs_file(filepath: epath.Path) -> bool:
13
+ """Returns whether a file a non-downloaded git-lfs file by checking its header.
14
+
15
+ An optimization of this function would be to only read the file if there is a git
16
+ repository in the distribution.
17
+ """
18
+ with open(filepath, "rb") as file:
19
+ # Only read the first line of the file. In the future, this could be a problem,
20
+ # e.g. if we accept *.txt files and the file starts with the same header.
21
+ first_line = file.readline()
22
+ if first_line.startswith(b"version https://git-lfs.github.com/spec"):
23
+ return True
24
+ return False
25
+
26
+
27
+ def download_git_lfs_file(file: Path):
28
+ """Downloads a specific git-lfs file within its repo."""
29
+ # Path(filepath="/tmp/full/path.json", fullpath="path.json")
30
+ # => working_dir = "/tmp/full"
31
+ fullpath = os.fspath(file.fullpath)
32
+ working_dir = os.fspath(file.filepath).rsplit(fullpath)[0]
33
+ repo = deps.git.Git(working_dir)
34
+ logging.info(
35
+ "Downloading git-lfs file: %s in working dir: %s", fullpath, working_dir
36
+ )
37
+ repo.execute(["git", "lfs", "pull", "--include", fullpath])
@@ -0,0 +1,54 @@
1
+ """Tests for git."""
2
+
3
+ import pathlib
4
+ import tempfile
5
+ from unittest import mock
6
+
7
+ from etils import epath
8
+ import git
9
+
10
+ from mlcroissant._src.core.git import download_git_lfs_file
11
+ from mlcroissant._src.core.git import is_git_lfs_file
12
+ from mlcroissant._src.core.path import Path
13
+
14
+ _GIT_LFS_CONTENT = lambda: """version https://git-lfs.github.com/spec/v1
15
+ oid sha256:5e2785fcd9098567a49d6e62e328923d955b307b6dcd0492f6234e96e670772a
16
+ size 309207547
17
+ """
18
+
19
+ _NON_GIT_LFS_CONTENT = lambda: """name,age
20
+ a,1
21
+ b,2
22
+ c,3"""
23
+
24
+ _NON_ASCII_CONTENT = lambda: (255).to_bytes(1, byteorder="big")
25
+
26
+
27
+ def test_is_git_lfs_file():
28
+ with tempfile.TemporaryDirectory() as tempdir:
29
+ tempdir = epath.Path(tempdir)
30
+
31
+ gitlfs_file = tempdir / "gitlfs"
32
+ gitlfs_file.write_text(_GIT_LFS_CONTENT())
33
+ assert is_git_lfs_file(gitlfs_file)
34
+
35
+ no_gitlfs_file = tempdir / "no-gitlfs"
36
+ no_gitlfs_file.write_text(_NON_GIT_LFS_CONTENT())
37
+ assert not is_git_lfs_file(no_gitlfs_file)
38
+
39
+ no_ascii_file = tempdir / "no-ascii"
40
+ no_ascii_file.write_bytes(_NON_ASCII_CONTENT())
41
+ assert not is_git_lfs_file(no_ascii_file)
42
+
43
+
44
+ def test_download_git_lfs_file():
45
+ file = Path(
46
+ filepath=epath.Path("/tmp/full/path.json"),
47
+ fullpath=pathlib.PurePath("path.json"),
48
+ )
49
+ with mock.patch.object(git, "Git", autospec=True) as git_mock:
50
+ download_git_lfs_file(file)
51
+ git_mock.assert_called_once_with("/tmp/full/")
52
+ git_mock.return_value.execute.assert_called_once_with(
53
+ ["git", "lfs", "pull", "--include", "path.json"]
54
+ )
@@ -0,0 +1,46 @@
1
+ """Utils for graphs."""
2
+
3
+ import time
4
+
5
+ import networkx as nx
6
+
7
+
8
+ def pretty_print_graph(graph: nx.Graph, simplify=False):
9
+ """Pretty prints a NetworkX graph.
10
+
11
+ Warning: this function is for debugging purposes only.
12
+
13
+ Args:
14
+ graph: Any NetworkX graph.
15
+ simplify: Whether to print a simplified version of nodes.
16
+ """
17
+ if simplify:
18
+ simple_graph = nx.Graph()
19
+ for x, y in graph.edges():
20
+ x = getattr(x, "uid", x)
21
+ y = getattr(y, "uid", y)
22
+ simple_graph.add_edge(x, y)
23
+ graph = simple_graph
24
+ agraph = nx.nx_agraph.to_agraph(graph)
25
+ agraph.layout(prog="dot")
26
+ temporary_file = f"/tmp/graph_{time.time()}.png"
27
+ agraph.draw(temporary_file, args="-Gnodesep=0.01 -Gfont_size=1", prog="dot")
28
+ print(f"Generated a graph and saved it in: {temporary_file}")
29
+
30
+
31
+ def print_graph_traversal(graph: nx.Graph):
32
+ """Pretty prints a NetworkX graph.
33
+
34
+ Warning: this function is for debugging purposes only.
35
+
36
+ Args:
37
+ graph: Any NetworkX graph.
38
+ """
39
+ visited = {}
40
+ print("--- Graph traversal ---")
41
+ for start, end, _ in nx.edge_bfs(graph):
42
+ for node in [start, end]:
43
+ if node.name not in visited:
44
+ print(f"Visited: {node.name}")
45
+ visited[node.name] = True
46
+ print("Done traversing the graph.")