kcai-data-sampling-job 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (28) hide show
  1. kcai_data_sampling_job-0.1.0/PKG-INFO +149 -0
  2. kcai_data_sampling_job-0.1.0/README.md +126 -0
  3. kcai_data_sampling_job-0.1.0/pyproject.toml +42 -0
  4. kcai_data_sampling_job-0.1.0/requirements.txt +6 -0
  5. kcai_data_sampling_job-0.1.0/setup.cfg +4 -0
  6. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/__init__.py +9 -0
  7. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/_version_.py +24 -0
  8. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/cli.py +380 -0
  9. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/dataloaders/__init__.py +24 -0
  10. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/dataloaders/api/__init__.py +13 -0
  11. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/dataloaders/api/parquet.py +595 -0
  12. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/dataloaders/filters.py +34 -0
  13. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/job.py +326 -0
  14. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/outputwriter/__init__.py +15 -0
  15. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/outputwriter/api/__init__.py +12 -0
  16. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/outputwriter/api/images.py +100 -0
  17. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/outputwriter/api/parquet.py +81 -0
  18. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/py.typed +0 -0
  19. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/utils/__init__.py +5 -0
  20. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/utils/images.py +90 -0
  21. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job/utils/shared.py +43 -0
  22. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job.egg-info/PKG-INFO +149 -0
  23. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job.egg-info/SOURCES.txt +26 -0
  24. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job.egg-info/dependency_links.txt +1 -0
  25. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job.egg-info/entry_points.txt +6 -0
  26. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job.egg-info/requires.txt +6 -0
  27. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job.egg-info/scm_file_list.json +22 -0
  28. kcai_data_sampling_job-0.1.0/src/kcai_data_sampling_job.egg-info/top_level.txt +1 -0
@@ -0,0 +1,149 @@
1
+ Metadata-Version: 2.4
2
+ Name: kcai-data-sampling-job
3
+ Version: 0.1.0
4
+ Summary: Pipeline orchestration, generic table loaders/writers, and the YAML CLI for kcai data-sampling
5
+ Author-email: Safenai <support@safenai.io>
6
+ License-Expression: Apache-2.0
7
+ Keywords: ml,data,augmentation,robustness,pipeline
8
+ Classifier: Development Status :: 3 - Alpha
9
+ Classifier: Intended Audience :: Developers
10
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
11
+ Classifier: Programming Language :: Python :: 3.11
12
+ Classifier: Programming Language :: Python :: 3.12
13
+ Classifier: Programming Language :: Python :: 3.13
14
+ Classifier: Programming Language :: Python :: 3.14
15
+ Requires-Python: >=3.11
16
+ Description-Content-Type: text/markdown
17
+ Requires-Dist: numpy
18
+ Requires-Dist: Pillow
19
+ Requires-Dist: pyarrow
20
+ Requires-Dist: PyYAML
21
+ Requires-Dist: tqdm
22
+ Requires-Dist: kcai-data-sampling-core
23
+
24
+ # kcai-data-sampling-job
25
+
26
+ Orchestration for kcai data-sampling: everything that turns a *selection* into
27
+ files on disk. This package holds no algorithms — it is the layer that runs the
28
+ methods other packages provide.
29
+
30
+ A campaign reads a source table, applies each configured transformation
31
+ **independently** to the same source sample, and writes two things per output: a
32
+ **ledger** row describing what was produced, and the **payload** file holding the
33
+ sample itself. The ledger is metadata only; the pixels go to their own
34
+ content-addressed file.
35
+
36
+ ## What it ships today
37
+
38
+ - `SamplingJob` — the run loop.
39
+ - The IO plugins: a `parquet` dataloader, and `images` and `parquet` output
40
+ writers.
41
+ - The ledger writer.
42
+ - The `process` command, driven by YAML.
43
+
44
+ **This is where the IO plugins live, and where image files are handled.** It is
45
+ also the only part of the workspace that depends on Pillow, and it needs it
46
+ because it is the only part that encodes a decoded array to PNG. The
47
+ transformation packages themselves stay pure numpy.
48
+
49
+ ## Install
50
+
51
+ ```bash
52
+ pip install "kcai-data-sampling[job]"
53
+ ```
54
+
55
+ This extra is also what turns on the `process` command:
56
+
57
+ ```bash
58
+ kcai-data-sampling process -p config.yaml
59
+ ```
60
+
61
+ Without it the command is not there — the umbrella CLI builds its command list
62
+ from what is importable, so a base install offers `version` and `list` only.
63
+
64
+ ## How a run works
65
+
66
+ The job is a streaming loop. For each chunk of `load_batch_size` samples it
67
+ decodes a selection into memory, runs every configured transformation over that
68
+ chunk through the runner, encodes each output as a payload, buffers the metadata
69
+ rows, and drops the chunk.
70
+
71
+ **Batching changes nothing in the output.** Transformations are unary and the
72
+ pairing is chunk-local, so the same selection yields byte-identical results at
73
+ any `load_batch_size`. That is why randomness is seeded per parent sample rather
74
+ than per batch.
75
+
76
+ **There is no chain.** Every transformation is applied to the *same* source
77
+ sample and each output is its own row — no intermediate stage, no
78
+ transformation-of-a-transformation. Two transformations listed together produce
79
+ two rows, not one row reached in two steps.
80
+
81
+ ## Output
82
+
83
+ **The ledger** is one row per generated sample, recording the lineage and the
84
+ declaration: which selection and loader it came from, the source `parent_id`, the
85
+ output `id`, the algorithm, its `family`, `arity`, `reversible`, the resolved
86
+ `params`, the `seed`, and the `tool_model` / `target_model` names. Enough to
87
+ reproduce or audit a row without opening the payload.
88
+
89
+ **Source columns are opt-in and never overwrite the generated ones.** `include`
90
+ and `exclude` are pattern lists selecting source columns to carry through;
91
+ `exclude` wins over `include`, and any source column sharing a name with a
92
+ generated column is dropped from the pass-through rather than allowed to
93
+ overwrite it. With neither set, nothing is passed through — the ledger holds
94
+ only the generated columns.
95
+
96
+ **Payload files are content-addressed.** The file name is
97
+ `{selection}__{id}__{c6}.png`, where `c6` is a short hash of the decoded payload
98
+ bytes. Identical pixels plus the same output id therefore name the same file, so
99
+ artifact paths are invariant to buffering and flush order. The output id already
100
+ covers the recipe — parent, algorithm, settings, seed — and the hash only guards
101
+ content collisions.
102
+
103
+ ## Example
104
+
105
+ ```yaml
106
+ dataloaders:
107
+ loaders:
108
+ - name: images
109
+ type: parquet
110
+ path: /path/to/samples.parquet
111
+ id_column: id
112
+ load_batch_size: 10
113
+ decode: img_bytes
114
+ sample_path:
115
+ - column: path
116
+ prefix: /path/to/images
117
+
118
+ operations:
119
+ outputs:
120
+ path: /path/to/outputs/{selection}/ledger.parquet
121
+ write_samples: true
122
+ samples_dir: /path/to/outputs/{selection}/payloads
123
+ flush_batch_size: 128
124
+ include: [height, width]
125
+ transformations:
126
+ - name: horizontal_flip
127
+ type: horizontal_flip
128
+ - name: crop_resize
129
+ type: crop_resize
130
+ fraction:
131
+ range: [0.3, 0.5]
132
+ step: 0.1
133
+ ```
134
+
135
+ `{selection}` in an output path is substituted per selection, so one config can
136
+ write several campaigns into separate directories. The `include` list names
137
+ source columns to carry into the ledger beyond the generated ones.
138
+
139
+ ## Extending it
140
+
141
+ The job discovers its parts through entry points, so a package can add a
142
+ dataloader or an output writer without changing anything here:
143
+
144
+ - `kcai_data_sampling.dataloaders`
145
+ - `kcai_data_sampling.outputwriter`
146
+
147
+ Add your own dependencies to *your* package. This one is where IO lives, and
148
+ keeping it free of algorithm-specific requirements is what lets a project compose
149
+ a run out of any mix of the packages above.
@@ -0,0 +1,126 @@
1
+ # kcai-data-sampling-job
2
+
3
+ Orchestration for kcai data-sampling: everything that turns a *selection* into
4
+ files on disk. This package holds no algorithms — it is the layer that runs the
5
+ methods other packages provide.
6
+
7
+ A campaign reads a source table, applies each configured transformation
8
+ **independently** to the same source sample, and writes two things per output: a
9
+ **ledger** row describing what was produced, and the **payload** file holding the
10
+ sample itself. The ledger is metadata only; the pixels go to their own
11
+ content-addressed file.
12
+
13
+ ## What it ships today
14
+
15
+ - `SamplingJob` — the run loop.
16
+ - The IO plugins: a `parquet` dataloader, and `images` and `parquet` output
17
+ writers.
18
+ - The ledger writer.
19
+ - The `process` command, driven by YAML.
20
+
21
+ **This is where the IO plugins live, and where image files are handled.** It is
22
+ also the only part of the workspace that depends on Pillow, and it needs it
23
+ because it is the only part that encodes a decoded array to PNG. The
24
+ transformation packages themselves stay pure numpy.
25
+
26
+ ## Install
27
+
28
+ ```bash
29
+ pip install "kcai-data-sampling[job]"
30
+ ```
31
+
32
+ This extra is also what turns on the `process` command:
33
+
34
+ ```bash
35
+ kcai-data-sampling process -p config.yaml
36
+ ```
37
+
38
+ Without it the command is not there — the umbrella CLI builds its command list
39
+ from what is importable, so a base install offers `version` and `list` only.
40
+
41
+ ## How a run works
42
+
43
+ The job is a streaming loop. For each chunk of `load_batch_size` samples it
44
+ decodes a selection into memory, runs every configured transformation over that
45
+ chunk through the runner, encodes each output as a payload, buffers the metadata
46
+ rows, and drops the chunk.
47
+
48
+ **Batching changes nothing in the output.** Transformations are unary and the
49
+ pairing is chunk-local, so the same selection yields byte-identical results at
50
+ any `load_batch_size`. That is why randomness is seeded per parent sample rather
51
+ than per batch.
52
+
53
+ **There is no chain.** Every transformation is applied to the *same* source
54
+ sample and each output is its own row — no intermediate stage, no
55
+ transformation-of-a-transformation. Two transformations listed together produce
56
+ two rows, not one row reached in two steps.
57
+
58
+ ## Output
59
+
60
+ **The ledger** is one row per generated sample, recording the lineage and the
61
+ declaration: which selection and loader it came from, the source `parent_id`, the
62
+ output `id`, the algorithm, its `family`, `arity`, `reversible`, the resolved
63
+ `params`, the `seed`, and the `tool_model` / `target_model` names. Enough to
64
+ reproduce or audit a row without opening the payload.
65
+
66
+ **Source columns are opt-in and never overwrite the generated ones.** `include`
67
+ and `exclude` are pattern lists selecting source columns to carry through;
68
+ `exclude` wins over `include`, and any source column sharing a name with a
69
+ generated column is dropped from the pass-through rather than allowed to
70
+ overwrite it. With neither set, nothing is passed through — the ledger holds
71
+ only the generated columns.
72
+
73
+ **Payload files are content-addressed.** The file name is
74
+ `{selection}__{id}__{c6}.png`, where `c6` is a short hash of the decoded payload
75
+ bytes. Identical pixels plus the same output id therefore name the same file, so
76
+ artifact paths are invariant to buffering and flush order. The output id already
77
+ covers the recipe — parent, algorithm, settings, seed — and the hash only guards
78
+ content collisions.
79
+
80
+ ## Example
81
+
82
+ ```yaml
83
+ dataloaders:
84
+ loaders:
85
+ - name: images
86
+ type: parquet
87
+ path: /path/to/samples.parquet
88
+ id_column: id
89
+ load_batch_size: 10
90
+ decode: img_bytes
91
+ sample_path:
92
+ - column: path
93
+ prefix: /path/to/images
94
+
95
+ operations:
96
+ outputs:
97
+ path: /path/to/outputs/{selection}/ledger.parquet
98
+ write_samples: true
99
+ samples_dir: /path/to/outputs/{selection}/payloads
100
+ flush_batch_size: 128
101
+ include: [height, width]
102
+ transformations:
103
+ - name: horizontal_flip
104
+ type: horizontal_flip
105
+ - name: crop_resize
106
+ type: crop_resize
107
+ fraction:
108
+ range: [0.3, 0.5]
109
+ step: 0.1
110
+ ```
111
+
112
+ `{selection}` in an output path is substituted per selection, so one config can
113
+ write several campaigns into separate directories. The `include` list names
114
+ source columns to carry into the ledger beyond the generated ones.
115
+
116
+ ## Extending it
117
+
118
+ The job discovers its parts through entry points, so a package can add a
119
+ dataloader or an output writer without changing anything here:
120
+
121
+ - `kcai_data_sampling.dataloaders`
122
+ - `kcai_data_sampling.outputwriter`
123
+
124
+ Add your own dependencies to *your* package. This one is where IO lives, and
125
+ keeping it free of algorithm-specific requirements is what lets a project compose
126
+ a run out of any mix of the packages above.
@@ -0,0 +1,42 @@
1
+ [project]
2
+ name = "kcai-data-sampling-job"
3
+ dynamic = ["dependencies", "version"]
4
+ description = "Pipeline orchestration, generic table loaders/writers, and the YAML CLI for kcai data-sampling"
5
+ authors = [
6
+ {name = "Safenai", email = "support@safenai.io"},
7
+ ]
8
+ license = "Apache-2.0"
9
+ readme = "README.md"
10
+ requires-python = ">=3.11"
11
+ keywords = ["ml", "data", "augmentation", "robustness", "pipeline"]
12
+ classifiers = [
13
+ "Development Status :: 3 - Alpha",
14
+ "Intended Audience :: Developers",
15
+ "Topic :: Software Development :: Libraries :: Python Modules",
16
+ "Programming Language :: Python :: 3.11",
17
+ "Programming Language :: Python :: 3.12",
18
+ "Programming Language :: Python :: 3.13",
19
+ "Programming Language :: Python :: 3.14",
20
+ ]
21
+
22
+ [project.entry-points."kcai_data_sampling.dataloaders"]
23
+ parquet = "kcai_data_sampling_job.dataloaders.api.parquet:ParquetDataLoader"
24
+
25
+ [project.entry-points."kcai_data_sampling.outputwriter"]
26
+ images = "kcai_data_sampling_job.outputwriter.api.images:ImagesOutputWriter"
27
+ parquet = "kcai_data_sampling_job.outputwriter.api.parquet:ParquetOutputWriter"
28
+
29
+ [build-system]
30
+ requires = ["setuptools>=80", "setuptools-scm[simple]>=8"]
31
+ build-backend = "setuptools.build_meta"
32
+
33
+ [tool.setuptools_scm]
34
+ version_file = "src/kcai_data_sampling_job/_version_.py"
35
+ root = "../.."
36
+ fallback_version = "0.0.1"
37
+
38
+ [tool.setuptools.dynamic]
39
+ dependencies = {file = ["requirements.txt"]}
40
+
41
+ [tool.setuptools.packages.find]
42
+ where = ["src"]
@@ -0,0 +1,6 @@
1
+ numpy
2
+ Pillow
3
+ pyarrow
4
+ PyYAML
5
+ tqdm
6
+ kcai-data-sampling-core
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,9 @@
1
+ """Pipeline orchestration, YAML CLI, and generic table IO.
2
+
3
+ The ``parquet`` column loader first decodes `img_bytes` rows to in-memory
4
+ samples, then the transformation interface runs over them; the ``parquet``
5
+ output writer owns the metadata-only ledger. The image sample readers/writers
6
+ are plugins shipped by ``kcai-data-sampling-images``.
7
+ """
8
+
9
+ from kcai_data_sampling_job._version_ import __version__ as __version__
@@ -0,0 +1,24 @@
1
+ # file generated by vcs-versioning
2
+ # don't change, don't track in version control
3
+ from __future__ import annotations
4
+
5
+ __all__ = [
6
+ "__version__",
7
+ "__version_tuple__",
8
+ "version",
9
+ "version_tuple",
10
+ "__commit_id__",
11
+ "commit_id",
12
+ ]
13
+
14
+ version: str
15
+ __version__: str
16
+ __version_tuple__: tuple[int | str, ...]
17
+ version_tuple: tuple[int | str, ...]
18
+ commit_id: str | None
19
+ __commit_id__: str | None
20
+
21
+ __version__ = version = '0.1.0'
22
+ __version_tuple__ = version_tuple = (0, 1, 0)
23
+
24
+ __commit_id__ = commit_id = None