caption-flow 0.5.2__tar.gz → 0.5.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {caption_flow-0.5.2 → caption_flow-0.5.4}/PKG-INFO +126 -42
- caption_flow-0.5.2/src/caption_flow.egg-info/PKG-INFO → caption_flow-0.5.4/README.md +42 -59
- {caption_flow-0.5.2 → caption_flow-0.5.4}/pyproject.toml +11 -51
- caption_flow-0.5.4/setup.py +129 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/__init__.py +1 -1
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/cli.py +13 -3
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/processors/huggingface.py +118 -20
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/processors/webdataset.py +35 -8
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/storage/exporter.py +107 -1
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/storage/manager.py +1 -1
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/utils/vllm_config.py +13 -1
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/workers/caption.py +49 -2
- caption_flow-0.5.2/README.md → caption_flow-0.5.4/src/caption_flow.egg-info/PKG-INFO +143 -4
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow.egg-info/SOURCES.txt +1 -0
- caption_flow-0.5.4/src/caption_flow.egg-info/requires.txt +87 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/tests/test_cli.py +71 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/tests/test_exporter.py +132 -1
- {caption_flow-0.5.2 → caption_flow-0.5.4}/tests/test_processors.py +163 -66
- {caption_flow-0.5.2 → caption_flow-0.5.4}/tests/test_vllm_config.py +16 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/tests/test_webdataset_ranges.py +35 -30
- {caption_flow-0.5.2 → caption_flow-0.5.4}/tests/test_worker_caption.py +199 -0
- caption_flow-0.5.2/src/caption_flow.egg-info/requires.txt +0 -38
- {caption_flow-0.5.2 → caption_flow-0.5.4}/LICENSE +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/setup.cfg +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/models.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/monitor.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/orchestrator.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/processors/__init__.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/processors/base.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/processors/local_filesystem.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/storage/__init__.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/utils/__init__.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/utils/auth.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/utils/caption_utils.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/utils/certificates.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/utils/checkpoint_tracker.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/utils/chunk_tracker.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/utils/image_processor.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/utils/json_utils.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/utils/prompt_template.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/viewer.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/workers/base.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow/workers/data.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow.egg-info/dependency_links.txt +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow.egg-info/entry_points.txt +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/src/caption_flow.egg-info/top_level.txt +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/tests/test_caption_utils.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/tests/test_certificates.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/tests/test_config_reload.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/tests/test_duplicate_job_assignments.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/tests/test_fix_verification.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/tests/test_huggingface_ranges.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/tests/test_json_utils.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/tests/test_main.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/tests/test_monitor.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/tests/test_range_level_distribution.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/tests/test_storage_components.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/tests/test_viewer.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/tests/test_worker_reconnection_complete.py +0 -0
- {caption_flow-0.5.2 → caption_flow-0.5.4}/tests/test_worker_reconnection_sequence.py +0 -0
|
@@ -1,57 +1,103 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: caption-flow
|
|
3
|
-
Version: 0.5.
|
|
3
|
+
Version: 0.5.4
|
|
4
4
|
Summary: Self-contained distributed community captioning system
|
|
5
|
+
Author: bghira
|
|
5
6
|
Author-email: bghira <bghira@users.github.com>
|
|
6
|
-
License:
|
|
7
|
+
License-Expression: AGPL-3.0-or-later
|
|
8
|
+
Project-URL: Homepage, https://github.com/bghira/CaptionFlow
|
|
9
|
+
Project-URL: Repository, https://github.com/bghira/CaptionFlow
|
|
7
10
|
Keywords: captioning,distributed,vllm,dataset,community
|
|
8
11
|
Classifier: Development Status :: 4 - Beta
|
|
9
12
|
Classifier: Intended Audience :: Developers
|
|
10
|
-
Classifier: License :: OSI Approved :: MIT License
|
|
11
13
|
Classifier: Programming Language :: Python :: 3
|
|
12
14
|
Classifier: Programming Language :: Python :: 3.11
|
|
13
15
|
Classifier: Programming Language :: Python :: 3.12
|
|
14
16
|
Classifier: Programming Language :: Python :: 3.13
|
|
15
|
-
Requires-Python:
|
|
17
|
+
Requires-Python: >=3.11,<3.14
|
|
16
18
|
Description-Content-Type: text/markdown
|
|
17
19
|
License-File: LICENSE
|
|
18
|
-
Requires-Dist: websockets
|
|
19
|
-
Requires-Dist: pyarrow
|
|
20
|
-
Requires-Dist: click
|
|
21
|
-
Requires-Dist: pydantic
|
|
22
|
-
Requires-Dist: aiofiles
|
|
23
|
-
Requires-Dist: rich
|
|
24
|
-
Requires-Dist: cryptography
|
|
25
|
-
Requires-Dist:
|
|
26
|
-
Requires-Dist: certbot
|
|
27
|
-
Requires-Dist: numpy
|
|
28
|
-
Requires-Dist:
|
|
29
|
-
Requires-Dist:
|
|
30
|
-
Requires-Dist:
|
|
31
|
-
Requires-Dist:
|
|
32
|
-
Requires-Dist:
|
|
33
|
-
Requires-Dist:
|
|
34
|
-
Requires-Dist:
|
|
35
|
-
Requires-Dist:
|
|
36
|
-
Requires-Dist:
|
|
37
|
-
Requires-Dist:
|
|
38
|
-
Requires-Dist:
|
|
39
|
-
Requires-Dist:
|
|
20
|
+
Requires-Dist: websockets<17.0,>=16.0
|
|
21
|
+
Requires-Dist: pyarrow<26.0.0,>=21.0.0
|
|
22
|
+
Requires-Dist: click<9.0.0,>=8.2.0
|
|
23
|
+
Requires-Dist: pydantic<3.0.0,>=2.12.0
|
|
24
|
+
Requires-Dist: aiofiles<26.0.0,>=24.1.0
|
|
25
|
+
Requires-Dist: rich<16.0.0,>=14.0.0
|
|
26
|
+
Requires-Dist: cryptography<50.0.0,>=45.0.0
|
|
27
|
+
Requires-Dist: PyYAML<7.0.0,>=6.0.2
|
|
28
|
+
Requires-Dist: certbot<6.0.0,>=5.0.0
|
|
29
|
+
Requires-Dist: numpy<3.0.0,>=2.2.0
|
|
30
|
+
Requires-Dist: Pillow<13.0.0,>=11.3.0
|
|
31
|
+
Requires-Dist: pandas<4.0.0,>=3.0.0
|
|
32
|
+
Requires-Dist: datasets<6.0.0,>=5.0.0
|
|
33
|
+
Requires-Dist: boto3<2.0.0,>=1.43.0
|
|
34
|
+
Requires-Dist: webshart<0.6.0,>=0.5.2
|
|
35
|
+
Requires-Dist: pylance<9.0.0,>=8.0.0
|
|
36
|
+
Requires-Dist: duckdb<2.0.0,>=1.5.0
|
|
37
|
+
Requires-Dist: aiohttp<4.0.0,>=3.13.3
|
|
38
|
+
Requires-Dist: fastapi<0.137.0,>=0.133.0
|
|
39
|
+
Requires-Dist: uvicorn<1.0.0,>=0.35.0
|
|
40
|
+
Requires-Dist: huggingface-hub<2.0.0,>=1.5.0
|
|
41
|
+
Requires-Dist: opencv-python-headless<6.0.0,>=4.13.0
|
|
42
|
+
Requires-Dist: psutil<8.0.0,>=7.0.0
|
|
43
|
+
Requires-Dist: requests<3.0.0,>=2.32.0
|
|
44
|
+
Requires-Dist: tqdm<5.0.0,>=4.67.0
|
|
45
|
+
Requires-Dist: urwid<5.0.0,>=3.0.2
|
|
40
46
|
Provides-Extra: vllm
|
|
41
|
-
Requires-Dist:
|
|
42
|
-
Requires-Dist:
|
|
43
|
-
Requires-Dist:
|
|
47
|
+
Requires-Dist: vllm<0.26.0,>=0.25.1; extra == "vllm"
|
|
48
|
+
Requires-Dist: torch==2.11.0; extra == "vllm"
|
|
49
|
+
Requires-Dist: torchvision==0.26.0; extra == "vllm"
|
|
50
|
+
Requires-Dist: torchaudio==2.11.0; extra == "vllm"
|
|
51
|
+
Requires-Dist: transformers<6.0.0,>=5.5.3; extra == "vllm"
|
|
52
|
+
Requires-Dist: qwen-vl-utils<0.1.0,>=0.0.14; extra == "vllm"
|
|
53
|
+
Provides-Extra: captioning
|
|
54
|
+
Requires-Dist: vllm<0.26.0,>=0.25.1; extra == "captioning"
|
|
55
|
+
Requires-Dist: torch==2.11.0; extra == "captioning"
|
|
56
|
+
Requires-Dist: torchvision==0.26.0; extra == "captioning"
|
|
57
|
+
Requires-Dist: torchaudio==2.11.0; extra == "captioning"
|
|
58
|
+
Requires-Dist: transformers<6.0.0,>=5.5.3; extra == "captioning"
|
|
59
|
+
Requires-Dist: qwen-vl-utils<0.1.0,>=0.0.14; extra == "captioning"
|
|
60
|
+
Provides-Extra: cpu
|
|
61
|
+
Requires-Dist: torch>=2.11.0; extra == "cpu"
|
|
62
|
+
Requires-Dist: torchvision>=0.26.0; extra == "cpu"
|
|
63
|
+
Requires-Dist: torchaudio>=2.11.0; extra == "cpu"
|
|
64
|
+
Provides-Extra: cuda
|
|
65
|
+
Requires-Dist: torch>=2.11.0; extra == "cuda"
|
|
66
|
+
Requires-Dist: torchvision>=0.26.0; extra == "cuda"
|
|
67
|
+
Requires-Dist: torchaudio>=2.11.0; extra == "cuda"
|
|
68
|
+
Provides-Extra: cuda13
|
|
69
|
+
Requires-Dist: torch>=2.11.0; extra == "cuda13"
|
|
70
|
+
Requires-Dist: torchvision>=0.26.0; extra == "cuda13"
|
|
71
|
+
Requires-Dist: torchaudio>=2.11.0; extra == "cuda13"
|
|
72
|
+
Provides-Extra: rocm
|
|
73
|
+
Requires-Dist: torch>=2.11.0; extra == "rocm"
|
|
74
|
+
Requires-Dist: torchvision>=0.26.0; extra == "rocm"
|
|
75
|
+
Requires-Dist: torchaudio>=2.11.0; extra == "rocm"
|
|
76
|
+
Provides-Extra: apple
|
|
77
|
+
Requires-Dist: torch>=2.13.0; extra == "apple"
|
|
78
|
+
Requires-Dist: torchvision>=0.28.0; extra == "apple"
|
|
79
|
+
Requires-Dist: torchaudio>=2.11.0; extra == "apple"
|
|
80
|
+
Requires-Dist: vllm-metal==0.1.0; (platform_system == "Darwin" and platform_machine == "arm64" and python_version >= "3.12") and extra == "apple"
|
|
44
81
|
Provides-Extra: dev
|
|
45
|
-
Requires-Dist: pytest
|
|
46
|
-
Requires-Dist: pytest-asyncio
|
|
47
|
-
Requires-Dist: pytest-cov
|
|
48
|
-
Requires-Dist:
|
|
49
|
-
Requires-Dist:
|
|
50
|
-
Requires-Dist:
|
|
51
|
-
Requires-Dist:
|
|
52
|
-
|
|
53
|
-
Requires-Dist:
|
|
82
|
+
Requires-Dist: pytest<9.0.0,>=8.0.0; extra == "dev"
|
|
83
|
+
Requires-Dist: pytest-asyncio<2.0.0,>=1.1.0; extra == "dev"
|
|
84
|
+
Requires-Dist: pytest-cov<7.0.0,>=6.0.0; extra == "dev"
|
|
85
|
+
Requires-Dist: transformers<6.0.0,>=5.5.3; extra == "dev"
|
|
86
|
+
Requires-Dist: black<26.0.0,>=25.0.0; extra == "dev"
|
|
87
|
+
Requires-Dist: ruff<1.0.0,>=0.12.0; extra == "dev"
|
|
88
|
+
Requires-Dist: mypy<2.0.0,>=1.17.0; extra == "dev"
|
|
89
|
+
Provides-Extra: all
|
|
90
|
+
Requires-Dist: vllm<0.26.0,>=0.25.1; extra == "all"
|
|
91
|
+
Requires-Dist: torch==2.11.0; extra == "all"
|
|
92
|
+
Requires-Dist: torchvision==0.26.0; extra == "all"
|
|
93
|
+
Requires-Dist: torchaudio==2.11.0; extra == "all"
|
|
94
|
+
Requires-Dist: transformers<6.0.0,>=5.5.3; extra == "all"
|
|
95
|
+
Requires-Dist: qwen-vl-utils<0.1.0,>=0.0.14; extra == "all"
|
|
96
|
+
Dynamic: author
|
|
54
97
|
Dynamic: license-file
|
|
98
|
+
Dynamic: provides-extra
|
|
99
|
+
Dynamic: requires-dist
|
|
100
|
+
Dynamic: requires-python
|
|
55
101
|
|
|
56
102
|
# CaptionFlow
|
|
57
103
|
|
|
@@ -63,6 +109,8 @@ scalable, fault-tolerant **vLLM-powered image captioning**.
|
|
|
63
109
|
|
|
64
110
|
a fast websocket-based orchestrator paired with lightweight gpu workers achieves exceptional performance for batched requests through vLLM.
|
|
65
111
|
|
|
112
|
+
CaptionFlow is also integrated in [bghira/SimpleTuner](https://github.com/bghira/SimpleTuner), where it powers an end-to-end caption-to-training workflow through the SimpleTuner WebUI. Use CaptionFlow directly when you want a standalone distributed captioning system, or use it through SimpleTuner when you want dataset captioning, caption review/export, and model training managed as one suite.
|
|
113
|
+
|
|
66
114
|
* **orchestrator**: hands out work in chunked shards, collects captions, checkpoints progress, and keeps simple stats.
|
|
67
115
|
* **workers (vLLM)**: connect to the orchestrator, stream in image samples, batch them, and generate 1..N captions per image using prompts supplied by the orchestrator.
|
|
68
116
|
* **config-driven**: all components read YAML config; flags can override.
|
|
@@ -76,11 +124,43 @@ a fast websocket-based orchestrator paired with lightweight gpu workers achieves
|
|
|
76
124
|
```bash
|
|
77
125
|
python -m venv .venv
|
|
78
126
|
source .venv/bin/activate # windows: .venv\Scripts\activate
|
|
79
|
-
pip install
|
|
127
|
+
pip install --upgrade pip
|
|
128
|
+
pip install "caption-flow[vllm]"
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
For an orchestrator or monitor-only install, use `pip install -e .`.
|
|
132
|
+
`.[captioning]` is an alias for `.[vllm]` for integrations such as
|
|
133
|
+
SimpleTuner. Terminal image previews remain optional because the current
|
|
134
|
+
`term-image` release requires an older Pillow major than CaptionFlow uses.
|
|
135
|
+
|
|
136
|
+
On Apple Silicon, install the MPS-compatible PyTorch chain and pinned Metal
|
|
137
|
+
plugin with:
|
|
138
|
+
|
|
139
|
+
```bash
|
|
140
|
+
pip install -e ".[apple]"
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
CaptionFlow can use vLLM on Apple Silicon too, but the normal Linux `vllm`
|
|
144
|
+
wheel is not the Apple install path. The pinned `vllm-metal` dependency is
|
|
145
|
+
selected automatically on native arm64 Python 3.12+. For a ready-to-run
|
|
146
|
+
Metal worker, use the upstream installer, which also builds/installs the
|
|
147
|
+
Apple-specific vLLM core:
|
|
148
|
+
|
|
149
|
+
```bash
|
|
150
|
+
curl -fsSL https://raw.githubusercontent.com/vllm-project/vllm-metal/main/install.sh | bash
|
|
151
|
+
source ~/.venv-vllm-metal/bin/activate
|
|
152
|
+
pip install -e .
|
|
80
153
|
```
|
|
81
154
|
|
|
155
|
+
Do not combine `.[apple]` with the Linux `.[vllm]` extra.
|
|
156
|
+
|
|
157
|
+
For native macOS CPU vLLM instead, follow the
|
|
158
|
+
[official source-build instructions](https://docs.vllm.ai/en/stable/getting_started/installation/cpu/?device=apple).
|
|
159
|
+
|
|
82
160
|
## quickstart (single box)
|
|
83
161
|
|
|
162
|
+
for a full caption-to-training workflow with a web interface, use the SimpleTuner WebUI integration. the standalone flow below is best when you want to run CaptionFlow directly, contribute workers to a cluster, or export captions for your own downstream training pipeline.
|
|
163
|
+
|
|
84
164
|
1. copy + edit the sample configs
|
|
85
165
|
|
|
86
166
|
```bash
|
|
@@ -127,15 +207,18 @@ Usage: caption-flow export [OPTIONS]
|
|
|
127
207
|
Export caption data to various formats.
|
|
128
208
|
|
|
129
209
|
Options:
|
|
130
|
-
--format [jsonl|json|csv|txt|huggingface_hub|all] Export format (default: jsonl)
|
|
210
|
+
--format [jsonl|json|csv|txt|parquet|webshart|lance|huggingface_hub|all] Export format (default: jsonl)
|
|
131
211
|
```
|
|
132
212
|
|
|
133
213
|
* **jsonl**: create JSON line file in the specified `--output` path
|
|
134
214
|
* **csv**: exports CSV-compatible data columns to the `--output` path containing incomplete metadata
|
|
135
215
|
* **json**: creates a `.json` file for each sample inside the `--output` subdirectory containing **complete** metadata; useful for webdatasets
|
|
136
216
|
* **txt**: creates `.txt` file for each sample inside the `--output` subdirectory containing ONLY captions
|
|
217
|
+
* **webshart**: updates an **existing per-shard metadata `.json` file** by writing captions under the plural `captions` key. for this format, pass `--output` as the path to the existing shard metadata JSON file when exporting one shard. if you export multiple shards, pass `--output` as a directory containing one existing `{shard_name}.json` file per shard.
|
|
137
218
|
* **huggingface_hub**: creates a dataset on Hugging Face Hub, possibly `--private` and `--nsfw` where necessary
|
|
138
|
-
* **all**: creates
|
|
219
|
+
* **all**: creates the directory/file-generating export formats in a specified `--output` directory. prefer a directory here; `webshart` is a special case that expects existing per-shard metadata `.json` files rather than creating new metadata files.
|
|
220
|
+
|
|
221
|
+
> note: `--output` paths ending in `.json` are treated specially for `webshart`. use a directory for normal multi-format exports and an existing shard metadata JSON file only when intentionally updating a `webshart` shard.
|
|
139
222
|
|
|
140
223
|
---
|
|
141
224
|
|
|
@@ -162,6 +245,7 @@ Options:
|
|
|
162
245
|
## dataset formats
|
|
163
246
|
|
|
164
247
|
* huggingface hub or local based URL list datasets that are compatible with the datasets library
|
|
248
|
+
* huggingface hub datasets that are simple containers of raw image files
|
|
165
249
|
* webdatasets shards containing full image data; also can be hosted on the hub
|
|
166
250
|
* local folder filled with images; orchestrator will serve the data to workers
|
|
167
251
|
|
|
@@ -240,7 +324,7 @@ PRs welcome. keep it simple and fast.
|
|
|
240
324
|
|
|
241
325
|
To contribute compute to a cluster:
|
|
242
326
|
|
|
243
|
-
1. Install caption-flow: `pip install caption-flow`
|
|
327
|
+
1. Install caption-flow: `pip install "caption-flow[vllm]"`
|
|
244
328
|
2. Get a worker token from the project maintainer
|
|
245
329
|
3. Run: `caption-flow worker --server wss://project.domain.com:8765 --token YOUR_TOKEN`
|
|
246
330
|
|
|
@@ -1,58 +1,3 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: caption-flow
|
|
3
|
-
Version: 0.5.2
|
|
4
|
-
Summary: Self-contained distributed community captioning system
|
|
5
|
-
Author-email: bghira <bghira@users.github.com>
|
|
6
|
-
License: MIT
|
|
7
|
-
Keywords: captioning,distributed,vllm,dataset,community
|
|
8
|
-
Classifier: Development Status :: 4 - Beta
|
|
9
|
-
Classifier: Intended Audience :: Developers
|
|
10
|
-
Classifier: License :: OSI Approved :: MIT License
|
|
11
|
-
Classifier: Programming Language :: Python :: 3
|
|
12
|
-
Classifier: Programming Language :: Python :: 3.11
|
|
13
|
-
Classifier: Programming Language :: Python :: 3.12
|
|
14
|
-
Classifier: Programming Language :: Python :: 3.13
|
|
15
|
-
Requires-Python: <3.14,>=3.11
|
|
16
|
-
Description-Content-Type: text/markdown
|
|
17
|
-
License-File: LICENSE
|
|
18
|
-
Requires-Dist: websockets>=12.0
|
|
19
|
-
Requires-Dist: pyarrow>=14.0.0
|
|
20
|
-
Requires-Dist: click>=8.1.0
|
|
21
|
-
Requires-Dist: pydantic>=2.0.0
|
|
22
|
-
Requires-Dist: aiofiles>=23.0.0
|
|
23
|
-
Requires-Dist: rich>=13.0.0
|
|
24
|
-
Requires-Dist: cryptography>=41.0.0
|
|
25
|
-
Requires-Dist: pyyaml>=6.0
|
|
26
|
-
Requires-Dist: certbot>=2.0.0
|
|
27
|
-
Requires-Dist: numpy>=1.24.0
|
|
28
|
-
Requires-Dist: pillow>=10.0.0
|
|
29
|
-
Requires-Dist: webdataset<2.0.0,>=1.0.2
|
|
30
|
-
Requires-Dist: pandas<3.0.0,>=2.3.1
|
|
31
|
-
Requires-Dist: arrow<2.0.0,>=1.3.0
|
|
32
|
-
Requires-Dist: datasets<5.0.0,>=4.0.0
|
|
33
|
-
Requires-Dist: boto3<2.0.0,>=1.40.11
|
|
34
|
-
Requires-Dist: torchdata<0.12.0,>=0.11.0
|
|
35
|
-
Requires-Dist: textual<6.0.0,>=5.3.0
|
|
36
|
-
Requires-Dist: urwid<4.0.0,>=3.0.2
|
|
37
|
-
Requires-Dist: webshart<0.5.0,>=0.4.3
|
|
38
|
-
Requires-Dist: pylance<0.36.0,>=0.35.0
|
|
39
|
-
Requires-Dist: duckdb<2.0.0,>=1.3.2
|
|
40
|
-
Provides-Extra: vllm
|
|
41
|
-
Requires-Dist: numpy<2.3.0,>=2.2.0; extra == "vllm"
|
|
42
|
-
Requires-Dist: vllm<0.20.0,>=0.19.1; extra == "vllm"
|
|
43
|
-
Requires-Dist: transformers<6.0.0,>=5.5.1; extra == "vllm"
|
|
44
|
-
Provides-Extra: dev
|
|
45
|
-
Requires-Dist: pytest>=7.4.0; extra == "dev"
|
|
46
|
-
Requires-Dist: pytest-asyncio>=0.21.0; extra == "dev"
|
|
47
|
-
Requires-Dist: pytest-cov>=4.1.0; extra == "dev"
|
|
48
|
-
Requires-Dist: black>=23.0.0; extra == "dev"
|
|
49
|
-
Requires-Dist: ruff>=0.1.0; extra == "dev"
|
|
50
|
-
Requires-Dist: mypy>=1.5.0; extra == "dev"
|
|
51
|
-
Requires-Dist: numpy<2.3.0,>=2.2.0; extra == "dev"
|
|
52
|
-
Requires-Dist: vllm<0.20.0,>=0.19.1; extra == "dev"
|
|
53
|
-
Requires-Dist: transformers<6.0.0,>=5.5.1; extra == "dev"
|
|
54
|
-
Dynamic: license-file
|
|
55
|
-
|
|
56
1
|
# CaptionFlow
|
|
57
2
|
|
|
58
3
|
<!-- [](https://github.com/bghira/CaptionFlow/actions/workflows/tests.yml) -->
|
|
@@ -63,6 +8,8 @@ scalable, fault-tolerant **vLLM-powered image captioning**.
|
|
|
63
8
|
|
|
64
9
|
a fast websocket-based orchestrator paired with lightweight gpu workers achieves exceptional performance for batched requests through vLLM.
|
|
65
10
|
|
|
11
|
+
CaptionFlow is also integrated in [bghira/SimpleTuner](https://github.com/bghira/SimpleTuner), where it powers an end-to-end caption-to-training workflow through the SimpleTuner WebUI. Use CaptionFlow directly when you want a standalone distributed captioning system, or use it through SimpleTuner when you want dataset captioning, caption review/export, and model training managed as one suite.
|
|
12
|
+
|
|
66
13
|
* **orchestrator**: hands out work in chunked shards, collects captions, checkpoints progress, and keeps simple stats.
|
|
67
14
|
* **workers (vLLM)**: connect to the orchestrator, stream in image samples, batch them, and generate 1..N captions per image using prompts supplied by the orchestrator.
|
|
68
15
|
* **config-driven**: all components read YAML config; flags can override.
|
|
@@ -76,11 +23,43 @@ a fast websocket-based orchestrator paired with lightweight gpu workers achieves
|
|
|
76
23
|
```bash
|
|
77
24
|
python -m venv .venv
|
|
78
25
|
source .venv/bin/activate # windows: .venv\Scripts\activate
|
|
79
|
-
pip install
|
|
26
|
+
pip install --upgrade pip
|
|
27
|
+
pip install "caption-flow[vllm]"
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
For an orchestrator or monitor-only install, use `pip install -e .`.
|
|
31
|
+
`.[captioning]` is an alias for `.[vllm]` for integrations such as
|
|
32
|
+
SimpleTuner. Terminal image previews remain optional because the current
|
|
33
|
+
`term-image` release requires an older Pillow major than CaptionFlow uses.
|
|
34
|
+
|
|
35
|
+
On Apple Silicon, install the MPS-compatible PyTorch chain and pinned Metal
|
|
36
|
+
plugin with:
|
|
37
|
+
|
|
38
|
+
```bash
|
|
39
|
+
pip install -e ".[apple]"
|
|
80
40
|
```
|
|
81
41
|
|
|
42
|
+
CaptionFlow can use vLLM on Apple Silicon too, but the normal Linux `vllm`
|
|
43
|
+
wheel is not the Apple install path. The pinned `vllm-metal` dependency is
|
|
44
|
+
selected automatically on native arm64 Python 3.12+. For a ready-to-run
|
|
45
|
+
Metal worker, use the upstream installer, which also builds/installs the
|
|
46
|
+
Apple-specific vLLM core:
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
curl -fsSL https://raw.githubusercontent.com/vllm-project/vllm-metal/main/install.sh | bash
|
|
50
|
+
source ~/.venv-vllm-metal/bin/activate
|
|
51
|
+
pip install -e .
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
Do not combine `.[apple]` with the Linux `.[vllm]` extra.
|
|
55
|
+
|
|
56
|
+
For native macOS CPU vLLM instead, follow the
|
|
57
|
+
[official source-build instructions](https://docs.vllm.ai/en/stable/getting_started/installation/cpu/?device=apple).
|
|
58
|
+
|
|
82
59
|
## quickstart (single box)
|
|
83
60
|
|
|
61
|
+
for a full caption-to-training workflow with a web interface, use the SimpleTuner WebUI integration. the standalone flow below is best when you want to run CaptionFlow directly, contribute workers to a cluster, or export captions for your own downstream training pipeline.
|
|
62
|
+
|
|
84
63
|
1. copy + edit the sample configs
|
|
85
64
|
|
|
86
65
|
```bash
|
|
@@ -127,15 +106,18 @@ Usage: caption-flow export [OPTIONS]
|
|
|
127
106
|
Export caption data to various formats.
|
|
128
107
|
|
|
129
108
|
Options:
|
|
130
|
-
--format [jsonl|json|csv|txt|huggingface_hub|all] Export format (default: jsonl)
|
|
109
|
+
--format [jsonl|json|csv|txt|parquet|webshart|lance|huggingface_hub|all] Export format (default: jsonl)
|
|
131
110
|
```
|
|
132
111
|
|
|
133
112
|
* **jsonl**: create JSON line file in the specified `--output` path
|
|
134
113
|
* **csv**: exports CSV-compatible data columns to the `--output` path containing incomplete metadata
|
|
135
114
|
* **json**: creates a `.json` file for each sample inside the `--output` subdirectory containing **complete** metadata; useful for webdatasets
|
|
136
115
|
* **txt**: creates `.txt` file for each sample inside the `--output` subdirectory containing ONLY captions
|
|
116
|
+
* **webshart**: updates an **existing per-shard metadata `.json` file** by writing captions under the plural `captions` key. for this format, pass `--output` as the path to the existing shard metadata JSON file when exporting one shard. if you export multiple shards, pass `--output` as a directory containing one existing `{shard_name}.json` file per shard.
|
|
137
117
|
* **huggingface_hub**: creates a dataset on Hugging Face Hub, possibly `--private` and `--nsfw` where necessary
|
|
138
|
-
* **all**: creates
|
|
118
|
+
* **all**: creates the directory/file-generating export formats in a specified `--output` directory. prefer a directory here; `webshart` is a special case that expects existing per-shard metadata `.json` files rather than creating new metadata files.
|
|
119
|
+
|
|
120
|
+
> note: `--output` paths ending in `.json` are treated specially for `webshart`. use a directory for normal multi-format exports and an existing shard metadata JSON file only when intentionally updating a `webshart` shard.
|
|
139
121
|
|
|
140
122
|
---
|
|
141
123
|
|
|
@@ -162,6 +144,7 @@ Options:
|
|
|
162
144
|
## dataset formats
|
|
163
145
|
|
|
164
146
|
* huggingface hub or local based URL list datasets that are compatible with the datasets library
|
|
147
|
+
* huggingface hub datasets that are simple containers of raw image files
|
|
165
148
|
* webdatasets shards containing full image data; also can be hosted on the hub
|
|
166
149
|
* local folder filled with images; orchestrator will serve the data to workers
|
|
167
150
|
|
|
@@ -240,7 +223,7 @@ PRs welcome. keep it simple and fast.
|
|
|
240
223
|
|
|
241
224
|
To contribute compute to a cluster:
|
|
242
225
|
|
|
243
|
-
1. Install caption-flow: `pip install caption-flow`
|
|
226
|
+
1. Install caption-flow: `pip install "caption-flow[vllm]"`
|
|
244
227
|
2. Get a worker token from the project maintainer
|
|
245
228
|
3. Run: `caption-flow worker --server wss://project.domain.com:8765 --token YOUR_TOKEN`
|
|
246
229
|
|
|
@@ -1,71 +1,31 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68.0.0", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
1
5
|
[project]
|
|
2
6
|
name = "caption-flow"
|
|
3
|
-
|
|
7
|
+
dynamic = ["version", "dependencies", "optional-dependencies"]
|
|
4
8
|
description = "Self-contained distributed community captioning system"
|
|
5
9
|
readme = "README.md"
|
|
10
|
+
license = "AGPL-3.0-or-later"
|
|
6
11
|
requires-python = ">=3.11,<3.14"
|
|
7
|
-
license = { text = "MIT" }
|
|
8
12
|
authors = [{ name = "bghira", email = "bghira@users.github.com" }]
|
|
9
13
|
keywords = ["captioning", "distributed", "vllm", "dataset", "community"]
|
|
10
14
|
classifiers = [
|
|
11
15
|
"Development Status :: 4 - Beta",
|
|
12
16
|
"Intended Audience :: Developers",
|
|
13
|
-
"License :: OSI Approved :: MIT License",
|
|
14
17
|
"Programming Language :: Python :: 3",
|
|
15
18
|
"Programming Language :: Python :: 3.11",
|
|
16
19
|
"Programming Language :: Python :: 3.12",
|
|
17
20
|
"Programming Language :: Python :: 3.13",
|
|
18
21
|
]
|
|
19
22
|
|
|
20
|
-
dependencies = [
|
|
21
|
-
"websockets>=12.0",
|
|
22
|
-
"pyarrow>=14.0.0",
|
|
23
|
-
"click>=8.1.0",
|
|
24
|
-
"pydantic>=2.0.0",
|
|
25
|
-
"aiofiles>=23.0.0",
|
|
26
|
-
"rich>=13.0.0",
|
|
27
|
-
"cryptography>=41.0.0",
|
|
28
|
-
"pyyaml>=6.0",
|
|
29
|
-
"certbot>=2.0.0",
|
|
30
|
-
"numpy>=1.24.0",
|
|
31
|
-
"pillow>=10.0.0",
|
|
32
|
-
"webdataset (>=1.0.2,<2.0.0)",
|
|
33
|
-
"pandas (>=2.3.1,<3.0.0)",
|
|
34
|
-
"arrow (>=1.3.0,<2.0.0)",
|
|
35
|
-
"datasets (>=4.0.0,<5.0.0)",
|
|
36
|
-
"boto3 (>=1.40.11,<2.0.0)",
|
|
37
|
-
"torchdata (>=0.11.0,<0.12.0)",
|
|
38
|
-
"textual (>=5.3.0,<6.0.0)",
|
|
39
|
-
"urwid (>=3.0.2,<4.0.0)",
|
|
40
|
-
"webshart (>=0.4.3,<0.5.0)",
|
|
41
|
-
"pylance (>=0.35.0,<0.36.0)",
|
|
42
|
-
"duckdb (>=1.3.2,<2.0.0)",
|
|
43
|
-
]
|
|
44
|
-
|
|
45
|
-
[project.optional-dependencies]
|
|
46
|
-
vllm = [
|
|
47
|
-
"numpy (>=2.2.0,<2.3.0)",
|
|
48
|
-
"vllm (>=0.19.1,<0.20.0)",
|
|
49
|
-
"transformers (>=5.5.1,<6.0.0)",
|
|
50
|
-
]
|
|
51
|
-
dev = [
|
|
52
|
-
"pytest>=7.4.0",
|
|
53
|
-
"pytest-asyncio>=0.21.0",
|
|
54
|
-
"pytest-cov>=4.1.0",
|
|
55
|
-
"black>=23.0.0",
|
|
56
|
-
"ruff>=0.1.0",
|
|
57
|
-
"mypy>=1.5.0",
|
|
58
|
-
"numpy (>=2.2.0,<2.3.0)",
|
|
59
|
-
"vllm (>=0.19.1,<0.20.0)",
|
|
60
|
-
"transformers (>=5.5.1,<6.0.0)",
|
|
61
|
-
]
|
|
62
|
-
|
|
63
23
|
[project.scripts]
|
|
64
24
|
caption-flow = "caption_flow.cli:main"
|
|
65
25
|
|
|
66
|
-
[
|
|
67
|
-
|
|
68
|
-
|
|
26
|
+
[project.urls]
|
|
27
|
+
Homepage = "https://github.com/bghira/CaptionFlow"
|
|
28
|
+
Repository = "https://github.com/bghira/CaptionFlow"
|
|
69
29
|
|
|
70
30
|
[tool.setuptools.packages.find]
|
|
71
31
|
where = ["src"]
|
|
@@ -98,5 +58,5 @@ warn_return_any = true
|
|
|
98
58
|
warn_unused_configs = true
|
|
99
59
|
disallow_untyped_defs = true
|
|
100
60
|
|
|
101
|
-
[tool.
|
|
102
|
-
|
|
61
|
+
[tool.setuptools.dynamic]
|
|
62
|
+
version = { attr = "caption_flow.__version__" }
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
|
|
3
|
+
"""Setuptools configuration for CaptionFlow.
|
|
4
|
+
|
|
5
|
+
The project keeps dependency selection here so that the GPU worker stack can
|
|
6
|
+
be installed separately from the orchestrator and monitor components.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
from setuptools import find_packages, setup
|
|
12
|
+
|
|
13
|
+
ROOT = Path(__file__).parent
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def get_version() -> str:
|
|
17
|
+
"""Read the package version without importing CaptionFlow's dependencies."""
|
|
18
|
+
version_file = ROOT / "src" / "caption_flow" / "__init__.py"
|
|
19
|
+
for line in version_file.read_text(encoding="utf-8").splitlines():
|
|
20
|
+
if line.startswith("__version__"):
|
|
21
|
+
return line.split("=", 1)[1].strip().strip("\"'")
|
|
22
|
+
raise RuntimeError("Unable to determine CaptionFlow version")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
base_deps = [
|
|
26
|
+
"websockets>=16.0,<17.0",
|
|
27
|
+
"pyarrow>=21.0.0,<26.0.0",
|
|
28
|
+
"click>=8.2.0,<9.0.0",
|
|
29
|
+
"pydantic>=2.12.0,<3.0.0",
|
|
30
|
+
"aiofiles>=24.1.0,<26.0.0",
|
|
31
|
+
"rich>=14.0.0,<16.0.0",
|
|
32
|
+
"cryptography>=45.0.0,<50.0.0",
|
|
33
|
+
"PyYAML>=6.0.2,<7.0.0",
|
|
34
|
+
"certbot>=5.0.0,<6.0.0",
|
|
35
|
+
"numpy>=2.2.0,<3.0.0",
|
|
36
|
+
"Pillow>=11.3.0,<13.0.0",
|
|
37
|
+
"pandas>=3.0.0,<4.0.0",
|
|
38
|
+
"datasets>=5.0.0,<6.0.0",
|
|
39
|
+
"boto3>=1.43.0,<2.0.0",
|
|
40
|
+
"webshart>=0.5.2,<0.6.0",
|
|
41
|
+
"pylance>=8.0.0,<9.0.0",
|
|
42
|
+
"duckdb>=1.5.0,<2.0.0",
|
|
43
|
+
"aiohttp>=3.13.3,<4.0.0",
|
|
44
|
+
"fastapi>=0.133.0,<0.137.0",
|
|
45
|
+
"uvicorn>=0.35.0,<1.0.0",
|
|
46
|
+
"huggingface-hub>=1.5.0,<2.0.0",
|
|
47
|
+
"opencv-python-headless>=4.13.0,<6.0.0",
|
|
48
|
+
"psutil>=7.0.0,<8.0.0",
|
|
49
|
+
"requests>=2.32.0,<3.0.0",
|
|
50
|
+
"tqdm>=4.67.0,<5.0.0",
|
|
51
|
+
"urwid>=3.0.2,<5.0.0",
|
|
52
|
+
]
|
|
53
|
+
|
|
54
|
+
# vLLM 0.25.1 currently selects the PyTorch 2.11.0 family. Keep these
|
|
55
|
+
# explicit so a worker install cannot silently combine incompatible torch
|
|
56
|
+
# packages when another dependency broadens its requirements.
|
|
57
|
+
vllm_deps = [
|
|
58
|
+
"vllm>=0.25.1,<0.26.0",
|
|
59
|
+
"torch==2.11.0",
|
|
60
|
+
"torchvision==0.26.0",
|
|
61
|
+
"torchaudio==2.11.0",
|
|
62
|
+
"transformers>=5.5.3,<6.0.0",
|
|
63
|
+
"qwen-vl-utils>=0.0.14,<0.1.0",
|
|
64
|
+
]
|
|
65
|
+
|
|
66
|
+
PYTORCH_DEPENDENCIES = [
|
|
67
|
+
"torch>=2.11.0",
|
|
68
|
+
"torchvision>=0.26.0",
|
|
69
|
+
"torchaudio>=2.11.0",
|
|
70
|
+
]
|
|
71
|
+
|
|
72
|
+
APPLE_PYTORCH_DEPENDENCIES = [
|
|
73
|
+
"torch>=2.13.0",
|
|
74
|
+
"torchvision>=0.28.0",
|
|
75
|
+
"torchaudio>=2.11.0",
|
|
76
|
+
# vLLM Metal publishes Apple Silicon wheels for Python 3.12+.
|
|
77
|
+
"vllm-metal==0.1.0; platform_system == 'Darwin' and platform_machine == 'arm64' and python_version >= '3.12'",
|
|
78
|
+
]
|
|
79
|
+
|
|
80
|
+
extras_require = {
|
|
81
|
+
"vllm": vllm_deps,
|
|
82
|
+
"captioning": vllm_deps,
|
|
83
|
+
"cpu": PYTORCH_DEPENDENCIES,
|
|
84
|
+
"cuda": PYTORCH_DEPENDENCIES,
|
|
85
|
+
"cuda13": PYTORCH_DEPENDENCIES,
|
|
86
|
+
"rocm": PYTORCH_DEPENDENCIES,
|
|
87
|
+
"apple": APPLE_PYTORCH_DEPENDENCIES,
|
|
88
|
+
"dev": [
|
|
89
|
+
"pytest>=8.0.0,<9.0.0",
|
|
90
|
+
"pytest-asyncio>=1.1.0,<2.0.0",
|
|
91
|
+
"pytest-cov>=6.0.0,<7.0.0",
|
|
92
|
+
"transformers>=5.5.3,<6.0.0",
|
|
93
|
+
"black>=25.0.0,<26.0.0",
|
|
94
|
+
"ruff>=0.12.0,<1.0.0",
|
|
95
|
+
"mypy>=1.17.0,<2.0.0",
|
|
96
|
+
],
|
|
97
|
+
"all": [*vllm_deps],
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
setup(
|
|
102
|
+
name="caption-flow",
|
|
103
|
+
version=get_version(),
|
|
104
|
+
description="Self-contained distributed community captioning system",
|
|
105
|
+
long_description=(ROOT / "README.md").read_text(encoding="utf-8"),
|
|
106
|
+
long_description_content_type="text/markdown",
|
|
107
|
+
author="bghira",
|
|
108
|
+
author_email="bghira@users.github.com",
|
|
109
|
+
packages=find_packages(where="src"),
|
|
110
|
+
package_dir={"": "src"},
|
|
111
|
+
include_package_data=True,
|
|
112
|
+
python_requires=">=3.11,<3.14",
|
|
113
|
+
install_requires=base_deps,
|
|
114
|
+
extras_require=extras_require,
|
|
115
|
+
entry_points={"console_scripts": ["caption-flow=caption_flow.cli:main"]},
|
|
116
|
+
classifiers=[
|
|
117
|
+
"Development Status :: 4 - Beta",
|
|
118
|
+
"Intended Audience :: Developers",
|
|
119
|
+
"Programming Language :: Python :: 3",
|
|
120
|
+
"Programming Language :: Python :: 3.11",
|
|
121
|
+
"Programming Language :: Python :: 3.12",
|
|
122
|
+
"Programming Language :: Python :: 3.13",
|
|
123
|
+
],
|
|
124
|
+
keywords=["captioning", "distributed", "vllm", "dataset", "community"],
|
|
125
|
+
project_urls={
|
|
126
|
+
"Homepage": "https://github.com/bghira/CaptionFlow",
|
|
127
|
+
"Repository": "https://github.com/bghira/CaptionFlow",
|
|
128
|
+
},
|
|
129
|
+
)
|
|
@@ -367,8 +367,8 @@ def orchestrator(ctx, config: Optional[str], **kwargs):
|
|
|
367
367
|
@click.option(
|
|
368
368
|
"--when_finished",
|
|
369
369
|
type=click.Choice(["stay_connected", "shutdown", "post_exec_hook"]),
|
|
370
|
-
default=
|
|
371
|
-
help="Action when all captions are complete (default: stay_connected)",
|
|
370
|
+
default=None,
|
|
371
|
+
help="Action when all captions are complete (default: config value or stay_connected)",
|
|
372
372
|
)
|
|
373
373
|
@click.option("--post_exec_hook", help="Path to executable for post_exec_hook action")
|
|
374
374
|
@click.pass_context
|
|
@@ -1498,7 +1498,17 @@ async def _run_export_process(
|
|
|
1498
1498
|
@click.option(
|
|
1499
1499
|
"--format",
|
|
1500
1500
|
type=click.Choice(
|
|
1501
|
-
[
|
|
1501
|
+
[
|
|
1502
|
+
"jsonl",
|
|
1503
|
+
"json",
|
|
1504
|
+
"csv",
|
|
1505
|
+
"txt",
|
|
1506
|
+
"parquet",
|
|
1507
|
+
"webshart",
|
|
1508
|
+
"lance",
|
|
1509
|
+
"huggingface_hub",
|
|
1510
|
+
"all",
|
|
1511
|
+
],
|
|
1502
1512
|
case_sensitive=False,
|
|
1503
1513
|
),
|
|
1504
1514
|
default="jsonl",
|