kcai-data-sampling-job 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- kcai_data_sampling_job-0.1.0/PKG-INFO +149 -0
- kcai_data_sampling_job-0.1.0/README.md +126 -0
- kcai_data_sampling_job-0.1.0/pyproject.toml +42 -0
- kcai_data_sampling_job-0.1.0/requirements.txt +6 -0
- kcai_data_sampling_job-0.1.0/setup.cfg +4 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/__init__.py +9 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/_version_.py +24 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/cli.py +380 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/dataloaders/__init__.py +24 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/dataloaders/api/__init__.py +13 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/dataloaders/api/parquet.py +595 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/dataloaders/filters.py +34 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/job.py +326 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/outputwriter/__init__.py +15 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/outputwriter/api/__init__.py +12 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/outputwriter/api/images.py +100 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/outputwriter/api/parquet.py +81 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/py.typed +0 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/utils/__init__.py +5 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/utils/images.py +90 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/utils/shared.py +43 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job.egg-info/PKG-INFO +149 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job.egg-info/SOURCES.txt +26 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job.egg-info/dependency_links.txt +1 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job.egg-info/entry_points.txt +6 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job.egg-info/requires.txt +6 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job.egg-info/scm_file_list.json +22 -0
- kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job.egg-info/top_level.txt +1 -0
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: kcai-data-sampling-job
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Pipeline orchestration, generic table loaders/writers, and the YAML CLI for kcai data-sampling
|
|
5
|
+
Author-email: Safenai <support@safenai.io>
|
|
6
|
+
License-Expression: Apache-2.0
|
|
7
|
+
Keywords: ml,data,augmentation,robustness,pipeline
|
|
8
|
+
Classifier: Development Status :: 3 - Alpha
|
|
9
|
+
Classifier: Intended Audience :: Developers
|
|
10
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
15
|
+
Requires-Python: >=3.11
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
Requires-Dist: numpy
|
|
18
|
+
Requires-Dist: Pillow
|
|
19
|
+
Requires-Dist: pyarrow
|
|
20
|
+
Requires-Dist: PyYAML
|
|
21
|
+
Requires-Dist: tqdm
|
|
22
|
+
Requires-Dist: kcai-data-sampling-core
|
|
23
|
+
|
|
24
|
+
# kcai-data-sampling-job
|
|
25
|
+
|
|
26
|
+
Orchestration for kcai data-sampling: everything that turns a *selection* into
|
|
27
|
+
files on disk. This package holds no algorithms — it is the layer that runs the
|
|
28
|
+
methods other packages provide.
|
|
29
|
+
|
|
30
|
+
A campaign reads a source table, applies each configured transformation
|
|
31
|
+
**independently** to the same source sample, and writes two things per output: a
|
|
32
|
+
**ledger** row describing what was produced, and the **payload** file holding the
|
|
33
|
+
sample itself. The ledger is metadata only; the pixels go to their own
|
|
34
|
+
content-addressed file.
|
|
35
|
+
|
|
36
|
+
## What it ships today
|
|
37
|
+
|
|
38
|
+
- `SamplingJob` — the run loop.
|
|
39
|
+
- The IO plugins: a `parquet` dataloader, and `images` and `parquet` output
|
|
40
|
+
writers.
|
|
41
|
+
- The ledger writer.
|
|
42
|
+
- The `process` command, driven by YAML.
|
|
43
|
+
|
|
44
|
+
**This is where the IO plugins live, and where image files are handled.** It is
|
|
45
|
+
also the only part of the workspace that depends on Pillow, and it needs it
|
|
46
|
+
because it is the only part that encodes a decoded array to PNG. The
|
|
47
|
+
transformation packages themselves stay pure numpy.
|
|
48
|
+
|
|
49
|
+
## Install
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
pip install "kcai-data-sampling[job]"
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
This extra is also what turns on the `process` command:
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
kcai-data-sampling process -p config.yaml
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
Without it the command is not there — the umbrella CLI builds its command list
|
|
62
|
+
from what is importable, so a base install offers `version` and `list` only.
|
|
63
|
+
|
|
64
|
+
## How a run works
|
|
65
|
+
|
|
66
|
+
The job is a streaming loop. For each chunk of `load_batch_size` samples it
|
|
67
|
+
decodes a selection into memory, runs every configured transformation over that
|
|
68
|
+
chunk through the runner, encodes each output as a payload, buffers the metadata
|
|
69
|
+
rows, and drops the chunk.
|
|
70
|
+
|
|
71
|
+
**Batching changes nothing in the output.** Transformations are unary and the
|
|
72
|
+
pairing is chunk-local, so the same selection yields byte-identical results at
|
|
73
|
+
any `load_batch_size`. That is why randomness is seeded per parent sample rather
|
|
74
|
+
than per batch.
|
|
75
|
+
|
|
76
|
+
**There is no chain.** Every transformation is applied to the *same* source
|
|
77
|
+
sample and each output is its own row — no intermediate stage, no
|
|
78
|
+
transformation-of-a-transformation. Two transformations listed together produce
|
|
79
|
+
two rows, not one row reached in two steps.
|
|
80
|
+
|
|
81
|
+
## Output
|
|
82
|
+
|
|
83
|
+
**The ledger** is one row per generated sample, recording the lineage and the
|
|
84
|
+
declaration: which selection and loader it came from, the source `parent_id`, the
|
|
85
|
+
output `id`, the algorithm, its `family`, `arity`, `reversible`, the resolved
|
|
86
|
+
`params`, the `seed`, and the `tool_model` / `target_model` names. Enough to
|
|
87
|
+
reproduce or audit a row without opening the payload.
|
|
88
|
+
|
|
89
|
+
**Source columns are opt-in and never overwrite the generated ones.** `include`
|
|
90
|
+
and `exclude` are pattern lists selecting source columns to carry through;
|
|
91
|
+
`exclude` wins over `include`, and any source column sharing a name with a
|
|
92
|
+
generated column is dropped from the pass-through rather than allowed to
|
|
93
|
+
overwrite it. With neither set, nothing is passed through — the ledger holds
|
|
94
|
+
only the generated columns.
|
|
95
|
+
|
|
96
|
+
**Payload files are content-addressed.** The file name is
|
|
97
|
+
`{selection}__{id}__{c6}.png`, where `c6` is a short hash of the decoded payload
|
|
98
|
+
bytes. Identical pixels plus the same output id therefore name the same file, so
|
|
99
|
+
artifact paths are invariant to buffering and flush order. The output id already
|
|
100
|
+
covers the recipe — parent, algorithm, settings, seed — and the hash only guards
|
|
101
|
+
content collisions.
|
|
102
|
+
|
|
103
|
+
## Example
|
|
104
|
+
|
|
105
|
+
```yaml
|
|
106
|
+
dataloaders:
|
|
107
|
+
loaders:
|
|
108
|
+
- name: images
|
|
109
|
+
type: parquet
|
|
110
|
+
path: /path/to/samples.parquet
|
|
111
|
+
id_column: id
|
|
112
|
+
load_batch_size: 10
|
|
113
|
+
decode: img_bytes
|
|
114
|
+
sample_path:
|
|
115
|
+
- column: path
|
|
116
|
+
prefix: /path/to/images
|
|
117
|
+
|
|
118
|
+
operations:
|
|
119
|
+
outputs:
|
|
120
|
+
path: /path/to/outputs/{selection}/ledger.parquet
|
|
121
|
+
write_samples: true
|
|
122
|
+
samples_dir: /path/to/outputs/{selection}/payloads
|
|
123
|
+
flush_batch_size: 128
|
|
124
|
+
include: [height, width]
|
|
125
|
+
transformations:
|
|
126
|
+
- name: horizontal_flip
|
|
127
|
+
type: horizontal_flip
|
|
128
|
+
- name: crop_resize
|
|
129
|
+
type: crop_resize
|
|
130
|
+
fraction:
|
|
131
|
+
range: [0.3, 0.5]
|
|
132
|
+
step: 0.1
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
`{selection}` in an output path is substituted per selection, so one config can
|
|
136
|
+
write several campaigns into separate directories. The `include` list names
|
|
137
|
+
source columns to carry into the ledger beyond the generated ones.
|
|
138
|
+
|
|
139
|
+
## Extending it
|
|
140
|
+
|
|
141
|
+
The job discovers its parts through entry points, so a package can add a
|
|
142
|
+
dataloader or an output writer without changing anything here:
|
|
143
|
+
|
|
144
|
+
- `kcai_data_sampling.dataloaders`
|
|
145
|
+
- `kcai_data_sampling.outputwriter`
|
|
146
|
+
|
|
147
|
+
Add your own dependencies to *your* package. This one is where IO lives, and
|
|
148
|
+
keeping it free of algorithm-specific requirements is what lets a project compose
|
|
149
|
+
a run out of any mix of the packages above.
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
# kcai-data-sampling-job
|
|
2
|
+
|
|
3
|
+
Orchestration for kcai data-sampling: everything that turns a *selection* into
|
|
4
|
+
files on disk. This package holds no algorithms — it is the layer that runs the
|
|
5
|
+
methods other packages provide.
|
|
6
|
+
|
|
7
|
+
A campaign reads a source table, applies each configured transformation
|
|
8
|
+
**independently** to the same source sample, and writes two things per output: a
|
|
9
|
+
**ledger** row describing what was produced, and the **payload** file holding the
|
|
10
|
+
sample itself. The ledger is metadata only; the pixels go to their own
|
|
11
|
+
content-addressed file.
|
|
12
|
+
|
|
13
|
+
## What it ships today
|
|
14
|
+
|
|
15
|
+
- `SamplingJob` — the run loop.
|
|
16
|
+
- The IO plugins: a `parquet` dataloader, and `images` and `parquet` output
|
|
17
|
+
writers.
|
|
18
|
+
- The ledger writer.
|
|
19
|
+
- The `process` command, driven by YAML.
|
|
20
|
+
|
|
21
|
+
**This is where the IO plugins live, and where image files are handled.** It is
|
|
22
|
+
also the only part of the workspace that depends on Pillow, and it needs it
|
|
23
|
+
because it is the only part that encodes a decoded array to PNG. The
|
|
24
|
+
transformation packages themselves stay pure numpy.
|
|
25
|
+
|
|
26
|
+
## Install
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
pip install "kcai-data-sampling[job]"
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
This extra is also what turns on the `process` command:
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
kcai-data-sampling process -p config.yaml
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
Without it the command is not there — the umbrella CLI builds its command list
|
|
39
|
+
from what is importable, so a base install offers `version` and `list` only.
|
|
40
|
+
|
|
41
|
+
## How a run works
|
|
42
|
+
|
|
43
|
+
The job is a streaming loop. For each chunk of `load_batch_size` samples it
|
|
44
|
+
decodes a selection into memory, runs every configured transformation over that
|
|
45
|
+
chunk through the runner, encodes each output as a payload, buffers the metadata
|
|
46
|
+
rows, and drops the chunk.
|
|
47
|
+
|
|
48
|
+
**Batching changes nothing in the output.** Transformations are unary and the
|
|
49
|
+
pairing is chunk-local, so the same selection yields byte-identical results at
|
|
50
|
+
any `load_batch_size`. That is why randomness is seeded per parent sample rather
|
|
51
|
+
than per batch.
|
|
52
|
+
|
|
53
|
+
**There is no chain.** Every transformation is applied to the *same* source
|
|
54
|
+
sample and each output is its own row — no intermediate stage, no
|
|
55
|
+
transformation-of-a-transformation. Two transformations listed together produce
|
|
56
|
+
two rows, not one row reached in two steps.
|
|
57
|
+
|
|
58
|
+
## Output
|
|
59
|
+
|
|
60
|
+
**The ledger** is one row per generated sample, recording the lineage and the
|
|
61
|
+
declaration: which selection and loader it came from, the source `parent_id`, the
|
|
62
|
+
output `id`, the algorithm, its `family`, `arity`, `reversible`, the resolved
|
|
63
|
+
`params`, the `seed`, and the `tool_model` / `target_model` names. Enough to
|
|
64
|
+
reproduce or audit a row without opening the payload.
|
|
65
|
+
|
|
66
|
+
**Source columns are opt-in and never overwrite the generated ones.** `include`
|
|
67
|
+
and `exclude` are pattern lists selecting source columns to carry through;
|
|
68
|
+
`exclude` wins over `include`, and any source column sharing a name with a
|
|
69
|
+
generated column is dropped from the pass-through rather than allowed to
|
|
70
|
+
overwrite it. With neither set, nothing is passed through — the ledger holds
|
|
71
|
+
only the generated columns.
|
|
72
|
+
|
|
73
|
+
**Payload files are content-addressed.** The file name is
|
|
74
|
+
`{selection}__{id}__{c6}.png`, where `c6` is a short hash of the decoded payload
|
|
75
|
+
bytes. Identical pixels plus the same output id therefore name the same file, so
|
|
76
|
+
artifact paths are invariant to buffering and flush order. The output id already
|
|
77
|
+
covers the recipe — parent, algorithm, settings, seed — and the hash only guards
|
|
78
|
+
content collisions.
|
|
79
|
+
|
|
80
|
+
## Example
|
|
81
|
+
|
|
82
|
+
```yaml
|
|
83
|
+
dataloaders:
|
|
84
|
+
loaders:
|
|
85
|
+
- name: images
|
|
86
|
+
type: parquet
|
|
87
|
+
path: /path/to/samples.parquet
|
|
88
|
+
id_column: id
|
|
89
|
+
load_batch_size: 10
|
|
90
|
+
decode: img_bytes
|
|
91
|
+
sample_path:
|
|
92
|
+
- column: path
|
|
93
|
+
prefix: /path/to/images
|
|
94
|
+
|
|
95
|
+
operations:
|
|
96
|
+
outputs:
|
|
97
|
+
path: /path/to/outputs/{selection}/ledger.parquet
|
|
98
|
+
write_samples: true
|
|
99
|
+
samples_dir: /path/to/outputs/{selection}/payloads
|
|
100
|
+
flush_batch_size: 128
|
|
101
|
+
include: [height, width]
|
|
102
|
+
transformations:
|
|
103
|
+
- name: horizontal_flip
|
|
104
|
+
type: horizontal_flip
|
|
105
|
+
- name: crop_resize
|
|
106
|
+
type: crop_resize
|
|
107
|
+
fraction:
|
|
108
|
+
range: [0.3, 0.5]
|
|
109
|
+
step: 0.1
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
`{selection}` in an output path is substituted per selection, so one config can
|
|
113
|
+
write several campaigns into separate directories. The `include` list names
|
|
114
|
+
source columns to carry into the ledger beyond the generated ones.
|
|
115
|
+
|
|
116
|
+
## Extending it
|
|
117
|
+
|
|
118
|
+
The job discovers its parts through entry points, so a package can add a
|
|
119
|
+
dataloader or an output writer without changing anything here:
|
|
120
|
+
|
|
121
|
+
- `kcai_data_sampling.dataloaders`
|
|
122
|
+
- `kcai_data_sampling.outputwriter`
|
|
123
|
+
|
|
124
|
+
Add your own dependencies to *your* package. This one is where IO lives, and
|
|
125
|
+
keeping it free of algorithm-specific requirements is what lets a project compose
|
|
126
|
+
a run out of any mix of the packages above.
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "kcai-data-sampling-job"
|
|
3
|
+
dynamic = ["dependencies", "version"]
|
|
4
|
+
description = "Pipeline orchestration, generic table loaders/writers, and the YAML CLI for kcai data-sampling"
|
|
5
|
+
authors = [
|
|
6
|
+
{name = "Safenai", email = "support@safenai.io"},
|
|
7
|
+
]
|
|
8
|
+
license = "Apache-2.0"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.11"
|
|
11
|
+
keywords = ["ml", "data", "augmentation", "robustness", "pipeline"]
|
|
12
|
+
classifiers = [
|
|
13
|
+
"Development Status :: 3 - Alpha",
|
|
14
|
+
"Intended Audience :: Developers",
|
|
15
|
+
"Topic :: Software Development :: Libraries :: Python Modules",
|
|
16
|
+
"Programming Language :: Python :: 3.11",
|
|
17
|
+
"Programming Language :: Python :: 3.12",
|
|
18
|
+
"Programming Language :: Python :: 3.13",
|
|
19
|
+
"Programming Language :: Python :: 3.14",
|
|
20
|
+
]
|
|
21
|
+
|
|
22
|
+
[project.entry-points."kcai_data_sampling.dataloaders"]
|
|
23
|
+
parquet = "kcai_data_sampling_job.dataloaders.api.parquet:ParquetDataLoader"
|
|
24
|
+
|
|
25
|
+
[project.entry-points."kcai_data_sampling.outputwriter"]
|
|
26
|
+
images = "kcai_data_sampling_job.outputwriter.api.images:ImagesOutputWriter"
|
|
27
|
+
parquet = "kcai_data_sampling_job.outputwriter.api.parquet:ParquetOutputWriter"
|
|
28
|
+
|
|
29
|
+
[build-system]
|
|
30
|
+
requires = ["setuptools>=80", "setuptools-scm[simple]>=8"]
|
|
31
|
+
build-backend = "setuptools.build_meta"
|
|
32
|
+
|
|
33
|
+
[tool.setuptools_scm]
|
|
34
|
+
version_file = "src/kcai_data_sampling_job/_version_.py"
|
|
35
|
+
root = "../.."
|
|
36
|
+
fallback_version = "0.0.1"
|
|
37
|
+
|
|
38
|
+
[tool.setuptools.dynamic]
|
|
39
|
+
dependencies = {file = ["requirements.txt"]}
|
|
40
|
+
|
|
41
|
+
[tool.setuptools.packages.find]
|
|
42
|
+
where = ["src"]
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
"""Pipeline orchestration, YAML CLI, and generic table IO.
|
|
2
|
+
|
|
3
|
+
The ``parquet`` column loader first decodes `img_bytes` rows to in-memory
|
|
4
|
+
samples, then the transformation interface runs over them; the ``parquet``
|
|
5
|
+
output writer owns the metadata-only ledger. The image sample readers/writers
|
|
6
|
+
are plugins shipped by ``kcai-data-sampling-images``.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from kcai_data_sampling_job._version_ import __version__ as __version__
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# file generated by vcs-versioning
|
|
2
|
+
# don't change, don't track in version control
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
__all__ = [
|
|
6
|
+
"__version__",
|
|
7
|
+
"__version_tuple__",
|
|
8
|
+
"version",
|
|
9
|
+
"version_tuple",
|
|
10
|
+
"__commit_id__",
|
|
11
|
+
"commit_id",
|
|
12
|
+
]
|
|
13
|
+
|
|
14
|
+
version: str
|
|
15
|
+
__version__: str
|
|
16
|
+
__version_tuple__: tuple[int | str, ...]
|
|
17
|
+
version_tuple: tuple[int | str, ...]
|
|
18
|
+
commit_id: str | None
|
|
19
|
+
__commit_id__: str | None
|
|
20
|
+
|
|
21
|
+
__version__ = version = '0.1.0'
|
|
22
|
+
__version_tuple__ = version_tuple = (0, 1, 0)
|
|
23
|
+
|
|
24
|
+
__commit_id__ = commit_id = None
|