convmerge 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- convmerge-0.2.0/.gitignore +14 -0
- convmerge-0.2.0/LICENSE +21 -0
- convmerge-0.2.0/PKG-INFO +154 -0
- convmerge-0.2.0/README.md +120 -0
- convmerge-0.2.0/pyproject.toml +62 -0
- convmerge-0.2.0/src/convmerge/__init__.py +3 -0
- convmerge-0.2.0/src/convmerge/__main__.py +6 -0
- convmerge-0.2.0/src/convmerge/adapters/__init__.py +28 -0
- convmerge-0.2.0/src/convmerge/adapters/alpaca.py +38 -0
- convmerge-0.2.0/src/convmerge/adapters/chat.py +205 -0
- convmerge-0.2.0/src/convmerge/adapters/sharegpt.py +55 -0
- convmerge-0.2.0/src/convmerge/cli.py +367 -0
- convmerge-0.2.0/src/convmerge/convert.py +74 -0
- convmerge-0.2.0/src/convmerge/emitters.py +62 -0
- convmerge-0.2.0/src/convmerge/fetch/__init__.py +38 -0
- convmerge-0.2.0/src/convmerge/fetch/auth.py +62 -0
- convmerge-0.2.0/src/convmerge/fetch/git.py +71 -0
- convmerge-0.2.0/src/convmerge/fetch/github.py +123 -0
- convmerge-0.2.0/src/convmerge/fetch/hf.py +47 -0
- convmerge-0.2.0/src/convmerge/fetch/manifest.py +178 -0
- convmerge-0.2.0/src/convmerge/fetch/runner.py +213 -0
- convmerge-0.2.0/src/convmerge/models.py +21 -0
- convmerge-0.2.0/src/convmerge/normalize/__init__.py +43 -0
- convmerge-0.2.0/src/convmerge/normalize/convert_turns.py +78 -0
- convmerge-0.2.0/src/convmerge/normalize/dedup.py +90 -0
- convmerge-0.2.0/src/convmerge/normalize/jsonl.py +213 -0
- convmerge-0.2.0/src/convmerge/normalize/parquet.py +39 -0
- convmerge-0.2.0/src/convmerge/normalize/schema.py +84 -0
- convmerge-0.2.0/src/convmerge/normalize/turns.py +90 -0
convmerge-0.2.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 convmerge contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
convmerge-0.2.0/PKG-INFO
ADDED
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: convmerge
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Fetch, normalize, and convert heterogeneous chat/instruct datasets into a single LLM training format
|
|
5
|
+
Project-URL: Homepage, https://github.com/snowmuffin/convmerge
|
|
6
|
+
Project-URL: Repository, https://github.com/snowmuffin/convmerge
|
|
7
|
+
Project-URL: Issues, https://github.com/snowmuffin/convmerge/issues
|
|
8
|
+
Author: convmerge contributors
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: chat,dataset,jsonl,llm,sft
|
|
12
|
+
Classifier: Development Status :: 2 - Pre-Alpha
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Requires-Python: >=3.10
|
|
20
|
+
Provides-Extra: dev
|
|
21
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
22
|
+
Requires-Dist: ruff>=0.4; extra == 'dev'
|
|
23
|
+
Provides-Extra: fetch
|
|
24
|
+
Requires-Dist: pyyaml>=6.0; extra == 'fetch'
|
|
25
|
+
Provides-Extra: fetch-all
|
|
26
|
+
Requires-Dist: datasets>=2.16; extra == 'fetch-all'
|
|
27
|
+
Requires-Dist: pyyaml>=6.0; extra == 'fetch-all'
|
|
28
|
+
Provides-Extra: fetch-hf
|
|
29
|
+
Requires-Dist: datasets>=2.16; extra == 'fetch-hf'
|
|
30
|
+
Requires-Dist: pyyaml>=6.0; extra == 'fetch-hf'
|
|
31
|
+
Provides-Extra: parquet
|
|
32
|
+
Requires-Dist: pyarrow>=14; extra == 'parquet'
|
|
33
|
+
Description-Content-Type: text/markdown
|
|
34
|
+
|
|
35
|
+
# convmerge
|
|
36
|
+
|
|
37
|
+
Fetch, normalize, and convert heterogeneous chat / instruct datasets into a
|
|
38
|
+
**single LLM training format** (JSONL).
|
|
39
|
+
|
|
40
|
+
**Repository:** [github.com/snowmuffin/convmerge](https://github.com/snowmuffin/convmerge)
|
|
41
|
+
**Status:** pre-1.0; APIs and CLI may change between minor versions.
|
|
42
|
+
|
|
43
|
+
## Install
|
|
44
|
+
|
|
45
|
+
```bash
|
|
46
|
+
pip install convmerge # core: convert, normalize, dedupe, turns
|
|
47
|
+
pip install "convmerge[fetch]" # + YAML manifest fetcher (GitHub)
|
|
48
|
+
pip install "convmerge[fetch-hf]" # + HuggingFace entries (adds ``datasets``)
|
|
49
|
+
pip install "convmerge[fetch-all]" # all fetch-related extras
|
|
50
|
+
pip install "convmerge[parquet]" # + parquet streaming input
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
Or from a clone:
|
|
54
|
+
|
|
55
|
+
```bash
|
|
56
|
+
git clone https://github.com/snowmuffin/convmerge.git
|
|
57
|
+
cd convmerge
|
|
58
|
+
python -m venv .venv && source .venv/bin/activate
|
|
59
|
+
pip install -e ".[dev,fetch-all,parquet]"
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
## The four use cases
|
|
63
|
+
|
|
64
|
+
### 1. `fetch` — pull raw data from HF + GitHub via a YAML manifest
|
|
65
|
+
|
|
66
|
+
```yaml
|
|
67
|
+
# manifest.yaml
|
|
68
|
+
version: 1
|
|
69
|
+
defaults: { output_root: ./raw, resume: true }
|
|
70
|
+
auth: { hf_token_env: HF_TOKEN, github_token_env: GITHUB_TOKEN }
|
|
71
|
+
datasets:
|
|
72
|
+
- { name: alpaca-ko, hf: MarkrAI/KoCommercial-Dataset, split: train }
|
|
73
|
+
- { name: orca-raw,
|
|
74
|
+
url: https://raw.githubusercontent.com/org/repo/main/data/train.jsonl }
|
|
75
|
+
- { name: repo-tree,
|
|
76
|
+
url: https://github.com/org/example-repo, ext: [".jsonl"] }
|
|
77
|
+
- { name: big-lfs,
|
|
78
|
+
url: https://github.com/org/big-lfs-repo, mode: clone, lfs: true }
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
```bash
|
|
82
|
+
convmerge fetch manifest.yaml -o ./raw
|
|
83
|
+
# or one-shot shortcuts:
|
|
84
|
+
convmerge fetch hf://org/dataset -o ./raw --split train
|
|
85
|
+
convmerge fetch https://github.com/org/repo -o ./raw --ext .jsonl
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Tokens resolve in order CLI flag → file → env var, and are redacted from logs.
|
|
89
|
+
See [docs/fetch.md](docs/fetch.md) for the full schema.
|
|
90
|
+
|
|
91
|
+
### 2. `normalize` — reshape parquet / messy JSON into clean JSONL
|
|
92
|
+
|
|
93
|
+
```bash
|
|
94
|
+
convmerge normalize -i ./raw -o ./jsonl
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
Handles parquet (streamed via `pyarrow`), top-level JSON arrays, concatenated
|
|
98
|
+
single-line JSON (`{...}{...}{...}`), and already-valid JSONL. A directory
|
|
99
|
+
input is walked recursively and mirrored under the output directory.
|
|
100
|
+
|
|
101
|
+
### 3. `convert` — adapter + emitter pipeline
|
|
102
|
+
|
|
103
|
+
```bash
|
|
104
|
+
convmerge convert -i ./jsonl/alpaca.jsonl -o ./train/alpaca.messages.jsonl \
|
|
105
|
+
--from alpaca --format messages
|
|
106
|
+
|
|
107
|
+
convmerge convert -i ./jsonl/mixed.jsonl -o ./train/mixed.messages.jsonl \
|
|
108
|
+
--from auto --format messages # auto-detecting chat adapter
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
Adapters: `alpaca`, `sharegpt`, `chat` (alias `auto`).
|
|
112
|
+
Emitters: `messages`, `alpaca`.
|
|
113
|
+
|
|
114
|
+
### 4. `dedupe` / `turns` — final cleanup + train/eval split hook
|
|
115
|
+
|
|
116
|
+
```bash
|
|
117
|
+
convmerge dedupe -i ./train/mixed.messages.jsonl -o ./train/mixed.dedup.jsonl
|
|
118
|
+
convmerge turns -i ./train/mixed.dedup.jsonl \
|
|
119
|
+
--single-out ./train/single.jsonl \
|
|
120
|
+
--multi-out ./train/multi.jsonl
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
See [docs/format.md](docs/format.md) for adapter / emitter schemas and
|
|
124
|
+
[docs/fetch.md](docs/fetch.md) for manifest details.
|
|
125
|
+
|
|
126
|
+
## Development
|
|
127
|
+
|
|
128
|
+
See [CONTRIBUTING.md](CONTRIBUTING.md). CI runs Ruff + pytest on Python
|
|
129
|
+
3.10 – 3.12.
|
|
130
|
+
|
|
131
|
+
```bash
|
|
132
|
+
ruff check src tests
|
|
133
|
+
pytest -q
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
## PyPI release (maintainers)
|
|
137
|
+
|
|
138
|
+
Releases run from [`.github/workflows/publish.yml`](.github/workflows/publish.yml)
|
|
139
|
+
on pushing a `v*` tag. Publishing authenticates via the **`PYPI_API_TOKEN`**
|
|
140
|
+
GitHub Actions secret (a PyPI API token scoped to the `convmerge` project).
|
|
141
|
+
|
|
142
|
+
1. Create an API token on [pypi.org](https://pypi.org/manage/account/token/)
|
|
143
|
+
scoped to `convmerge`.
|
|
144
|
+
2. In the GitHub repo, *Settings → Secrets and variables → Actions → New
|
|
145
|
+
repository secret*, add `PYPI_API_TOKEN` with the token value.
|
|
146
|
+
3. Tag and push: `git tag v0.2.0 && git push origin v0.2.0`.
|
|
147
|
+
|
|
148
|
+
## Changelog
|
|
149
|
+
|
|
150
|
+
[CHANGELOG.md](CHANGELOG.md)
|
|
151
|
+
|
|
152
|
+
## License
|
|
153
|
+
|
|
154
|
+
MIT
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
# convmerge
|
|
2
|
+
|
|
3
|
+
Fetch, normalize, and convert heterogeneous chat / instruct datasets into a
|
|
4
|
+
**single LLM training format** (JSONL).
|
|
5
|
+
|
|
6
|
+
**Repository:** [github.com/snowmuffin/convmerge](https://github.com/snowmuffin/convmerge)
|
|
7
|
+
**Status:** pre-1.0; APIs and CLI may change between minor versions.
|
|
8
|
+
|
|
9
|
+
## Install
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
pip install convmerge # core: convert, normalize, dedupe, turns
|
|
13
|
+
pip install "convmerge[fetch]" # + YAML manifest fetcher (GitHub)
|
|
14
|
+
pip install "convmerge[fetch-hf]" # + HuggingFace entries (adds ``datasets``)
|
|
15
|
+
pip install "convmerge[fetch-all]" # all fetch-related extras
|
|
16
|
+
pip install "convmerge[parquet]" # + parquet streaming input
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
Or from a clone:
|
|
20
|
+
|
|
21
|
+
```bash
|
|
22
|
+
git clone https://github.com/snowmuffin/convmerge.git
|
|
23
|
+
cd convmerge
|
|
24
|
+
python -m venv .venv && source .venv/bin/activate
|
|
25
|
+
pip install -e ".[dev,fetch-all,parquet]"
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
## The four use cases
|
|
29
|
+
|
|
30
|
+
### 1. `fetch` — pull raw data from HF + GitHub via a YAML manifest
|
|
31
|
+
|
|
32
|
+
```yaml
|
|
33
|
+
# manifest.yaml
|
|
34
|
+
version: 1
|
|
35
|
+
defaults: { output_root: ./raw, resume: true }
|
|
36
|
+
auth: { hf_token_env: HF_TOKEN, github_token_env: GITHUB_TOKEN }
|
|
37
|
+
datasets:
|
|
38
|
+
- { name: alpaca-ko, hf: MarkrAI/KoCommercial-Dataset, split: train }
|
|
39
|
+
- { name: orca-raw,
|
|
40
|
+
url: https://raw.githubusercontent.com/org/repo/main/data/train.jsonl }
|
|
41
|
+
- { name: repo-tree,
|
|
42
|
+
url: https://github.com/org/example-repo, ext: [".jsonl"] }
|
|
43
|
+
- { name: big-lfs,
|
|
44
|
+
url: https://github.com/org/big-lfs-repo, mode: clone, lfs: true }
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
convmerge fetch manifest.yaml -o ./raw
|
|
49
|
+
# or one-shot shortcuts:
|
|
50
|
+
convmerge fetch hf://org/dataset -o ./raw --split train
|
|
51
|
+
convmerge fetch https://github.com/org/repo -o ./raw --ext .jsonl
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
Tokens resolve in order CLI flag → file → env var, and are redacted from logs.
|
|
55
|
+
See [docs/fetch.md](docs/fetch.md) for the full schema.
|
|
56
|
+
|
|
57
|
+
### 2. `normalize` — reshape parquet / messy JSON into clean JSONL
|
|
58
|
+
|
|
59
|
+
```bash
|
|
60
|
+
convmerge normalize -i ./raw -o ./jsonl
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
Handles parquet (streamed via `pyarrow`), top-level JSON arrays, concatenated
|
|
64
|
+
single-line JSON (`{...}{...}{...}`), and already-valid JSONL. A directory
|
|
65
|
+
input is walked recursively and mirrored under the output directory.
|
|
66
|
+
|
|
67
|
+
### 3. `convert` — adapter + emitter pipeline
|
|
68
|
+
|
|
69
|
+
```bash
|
|
70
|
+
convmerge convert -i ./jsonl/alpaca.jsonl -o ./train/alpaca.messages.jsonl \
|
|
71
|
+
--from alpaca --format messages
|
|
72
|
+
|
|
73
|
+
convmerge convert -i ./jsonl/mixed.jsonl -o ./train/mixed.messages.jsonl \
|
|
74
|
+
--from auto --format messages # auto-detecting chat adapter
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
Adapters: `alpaca`, `sharegpt`, `chat` (alias `auto`).
|
|
78
|
+
Emitters: `messages`, `alpaca`.
|
|
79
|
+
|
|
80
|
+
### 4. `dedupe` / `turns` — final cleanup + train/eval split hook
|
|
81
|
+
|
|
82
|
+
```bash
|
|
83
|
+
convmerge dedupe -i ./train/mixed.messages.jsonl -o ./train/mixed.dedup.jsonl
|
|
84
|
+
convmerge turns -i ./train/mixed.dedup.jsonl \
|
|
85
|
+
--single-out ./train/single.jsonl \
|
|
86
|
+
--multi-out ./train/multi.jsonl
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
See [docs/format.md](docs/format.md) for adapter / emitter schemas and
|
|
90
|
+
[docs/fetch.md](docs/fetch.md) for manifest details.
|
|
91
|
+
|
|
92
|
+
## Development
|
|
93
|
+
|
|
94
|
+
See [CONTRIBUTING.md](CONTRIBUTING.md). CI runs Ruff + pytest on Python
|
|
95
|
+
3.10 – 3.12.
|
|
96
|
+
|
|
97
|
+
```bash
|
|
98
|
+
ruff check src tests
|
|
99
|
+
pytest -q
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
## PyPI release (maintainers)
|
|
103
|
+
|
|
104
|
+
Releases run from [`.github/workflows/publish.yml`](.github/workflows/publish.yml)
|
|
105
|
+
on pushing a `v*` tag. Publishing authenticates via the **`PYPI_API_TOKEN`**
|
|
106
|
+
GitHub Actions secret (a PyPI API token scoped to the `convmerge` project).
|
|
107
|
+
|
|
108
|
+
1. Create an API token on [pypi.org](https://pypi.org/manage/account/token/)
|
|
109
|
+
scoped to `convmerge`.
|
|
110
|
+
2. In the GitHub repo, *Settings → Secrets and variables → Actions → New
|
|
111
|
+
repository secret*, add `PYPI_API_TOKEN` with the token value.
|
|
112
|
+
3. Tag and push: `git tag v0.2.0 && git push origin v0.2.0`.
|
|
113
|
+
|
|
114
|
+
## Changelog
|
|
115
|
+
|
|
116
|
+
[CHANGELOG.md](CHANGELOG.md)
|
|
117
|
+
|
|
118
|
+
## License
|
|
119
|
+
|
|
120
|
+
MIT
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "convmerge"
|
|
7
|
+
version = "0.2.0"
|
|
8
|
+
description = "Fetch, normalize, and convert heterogeneous chat/instruct datasets into a single LLM training format"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
requires-python = ">=3.10"
|
|
12
|
+
authors = [{ name = "convmerge contributors" }]
|
|
13
|
+
keywords = ["llm", "sft", "jsonl", "chat", "dataset"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 2 - Pre-Alpha",
|
|
16
|
+
"Intended Audience :: Developers",
|
|
17
|
+
"License :: OSI Approved :: MIT License",
|
|
18
|
+
"Programming Language :: Python :: 3",
|
|
19
|
+
"Programming Language :: Python :: 3.10",
|
|
20
|
+
"Programming Language :: Python :: 3.11",
|
|
21
|
+
"Programming Language :: Python :: 3.12",
|
|
22
|
+
]
|
|
23
|
+
|
|
24
|
+
dependencies = []
|
|
25
|
+
|
|
26
|
+
[project.optional-dependencies]
|
|
27
|
+
# YAML manifest fetcher (GitHub raw / Trees API / git clone). Uses stdlib urllib for HTTP.
|
|
28
|
+
fetch = ["pyyaml>=6.0"]
|
|
29
|
+
# HuggingFace support for fetch (delegates to the ``datasets`` library).
|
|
30
|
+
fetch-hf = ["pyyaml>=6.0", "datasets>=2.16"]
|
|
31
|
+
# Alias: everything fetch-related.
|
|
32
|
+
fetch-all = ["pyyaml>=6.0", "datasets>=2.16"]
|
|
33
|
+
# Parquet streaming input for normalize.
|
|
34
|
+
parquet = ["pyarrow>=14"]
|
|
35
|
+
dev = ["pytest>=8.0", "ruff>=0.4"]
|
|
36
|
+
|
|
37
|
+
[project.scripts]
|
|
38
|
+
convmerge = "convmerge.cli:main"
|
|
39
|
+
|
|
40
|
+
[project.urls]
|
|
41
|
+
Homepage = "https://github.com/snowmuffin/convmerge"
|
|
42
|
+
Repository = "https://github.com/snowmuffin/convmerge"
|
|
43
|
+
Issues = "https://github.com/snowmuffin/convmerge/issues"
|
|
44
|
+
|
|
45
|
+
[tool.hatch.build.targets.wheel]
|
|
46
|
+
packages = ["src/convmerge"]
|
|
47
|
+
|
|
48
|
+
[tool.hatch.build.targets.sdist]
|
|
49
|
+
include = ["/src"]
|
|
50
|
+
|
|
51
|
+
[tool.ruff]
|
|
52
|
+
line-length = 100
|
|
53
|
+
target-version = "py310"
|
|
54
|
+
|
|
55
|
+
[tool.ruff.format]
|
|
56
|
+
quote-style = "double"
|
|
57
|
+
|
|
58
|
+
[tool.ruff.lint]
|
|
59
|
+
select = ["E", "F", "I", "UP"]
|
|
60
|
+
|
|
61
|
+
[tool.pytest.ini_options]
|
|
62
|
+
testpaths = ["tests"]
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
"""Source-format adapters: raw records → TrainingExample."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Callable, Iterator
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from convmerge.adapters.alpaca import iter_from_alpaca_line
|
|
9
|
+
from convmerge.adapters.chat import iter_from_chat_line
|
|
10
|
+
from convmerge.adapters.sharegpt import iter_from_sharegpt_line
|
|
11
|
+
from convmerge.models import TrainingExample
|
|
12
|
+
|
|
13
|
+
AdapterFn = Callable[[dict[str, Any]], Iterator[TrainingExample]]
|
|
14
|
+
|
|
15
|
+
ADAPTERS: dict[str, AdapterFn] = {
|
|
16
|
+
"alpaca": iter_from_alpaca_line,
|
|
17
|
+
"sharegpt": iter_from_sharegpt_line,
|
|
18
|
+
"chat": iter_from_chat_line,
|
|
19
|
+
# ``auto`` is an alias for ``chat`` since the chat adapter is already auto-detecting.
|
|
20
|
+
"auto": iter_from_chat_line,
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def get_adapter(name: str) -> AdapterFn:
|
|
25
|
+
if name not in ADAPTERS:
|
|
26
|
+
known = ", ".join(sorted(ADAPTERS))
|
|
27
|
+
raise ValueError(f"Unknown adapter {name!r}. Choose one of: {known}")
|
|
28
|
+
return ADAPTERS[name]
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""Alpaca-style instruction / input / output → TrainingExample."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Iterator
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from convmerge.models import ChatMessage, TrainingExample
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def iter_from_alpaca_line(record: dict[str, Any]) -> Iterator[TrainingExample]:
|
|
12
|
+
"""
|
|
13
|
+
One JSON object per line: instruction, optional input, output.
|
|
14
|
+
|
|
15
|
+
Maps to a single user message + single assistant message.
|
|
16
|
+
"""
|
|
17
|
+
instruction = (record.get("instruction") or "").strip()
|
|
18
|
+
inp = (record.get("input") or "").strip()
|
|
19
|
+
output = (record.get("output") or "").strip() or (record.get("response") or "").strip()
|
|
20
|
+
|
|
21
|
+
user_parts = [instruction]
|
|
22
|
+
if inp:
|
|
23
|
+
user_parts.append(inp)
|
|
24
|
+
user_content = "\n".join(user_parts).strip()
|
|
25
|
+
|
|
26
|
+
if not user_content and not output:
|
|
27
|
+
return
|
|
28
|
+
|
|
29
|
+
messages: list[ChatMessage] = []
|
|
30
|
+
if user_content:
|
|
31
|
+
messages.append(ChatMessage(role="user", content=user_content))
|
|
32
|
+
if output:
|
|
33
|
+
messages.append(ChatMessage(role="assistant", content=output))
|
|
34
|
+
|
|
35
|
+
if not messages:
|
|
36
|
+
return
|
|
37
|
+
|
|
38
|
+
yield TrainingExample(messages=messages, meta={"source": "alpaca"})
|
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
"""Auto-detecting chat adapter.
|
|
2
|
+
|
|
3
|
+
Routes a raw record to the right internal shape by looking at which keys are
|
|
4
|
+
present. Handles the common messy shapes seen across SFT datasets:
|
|
5
|
+
|
|
6
|
+
- ``messages`` / ``conversation`` / ``conversations`` lists with
|
|
7
|
+
``{role, content}`` or ``{from, value}`` entries.
|
|
8
|
+
- Pairwise preference rows (``conversation_a`` / ``conversation_b``), with an
|
|
9
|
+
optional ``winner`` field; emits only the winner branch by default.
|
|
10
|
+
- Plain ``text`` strings (yielded as a single assistant message).
|
|
11
|
+
- Alpaca-style ``instruction`` / ``input`` / ``output`` (delegates to the
|
|
12
|
+
existing alpaca adapter).
|
|
13
|
+
|
|
14
|
+
Users can override the key lists and role map to teach it about bespoke schemas
|
|
15
|
+
without writing a new adapter from scratch.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from collections.abc import Iterator
|
|
21
|
+
from typing import Any
|
|
22
|
+
|
|
23
|
+
from convmerge.adapters.alpaca import iter_from_alpaca_line
|
|
24
|
+
from convmerge.models import ChatMessage, TrainingExample
|
|
25
|
+
|
|
26
|
+
# Default mapping from common ShareGPT-style ``from`` values onto standard roles.
|
|
27
|
+
DEFAULT_ROLE_MAP: dict[str, str] = {
|
|
28
|
+
"human": "user",
|
|
29
|
+
"user": "user",
|
|
30
|
+
"gpt": "assistant",
|
|
31
|
+
"assistant": "assistant",
|
|
32
|
+
"bing": "assistant",
|
|
33
|
+
"bot": "assistant",
|
|
34
|
+
"system": "system",
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
# Keys searched for the chat-list container, in priority order.
|
|
38
|
+
DEFAULT_CONVERSATION_KEYS: tuple[str, ...] = ("messages", "conversation", "conversations")
|
|
39
|
+
|
|
40
|
+
# Keys treated as role labels inside a chat-list entry.
|
|
41
|
+
DEFAULT_ROLE_KEYS: tuple[str, ...] = ("role", "from")
|
|
42
|
+
|
|
43
|
+
# Keys treated as message content inside a chat-list entry.
|
|
44
|
+
DEFAULT_CONTENT_KEYS: tuple[str, ...] = ("content", "value", "text")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def iter_from_chat_line(
|
|
48
|
+
record: dict[str, Any],
|
|
49
|
+
*,
|
|
50
|
+
conversation_keys: tuple[str, ...] = DEFAULT_CONVERSATION_KEYS,
|
|
51
|
+
role_keys: tuple[str, ...] = DEFAULT_ROLE_KEYS,
|
|
52
|
+
content_keys: tuple[str, ...] = DEFAULT_CONTENT_KEYS,
|
|
53
|
+
role_map: dict[str, str] | None = None,
|
|
54
|
+
pairwise_mode: str = "winner",
|
|
55
|
+
instruction_keys: tuple[str, ...] = ("instruction", "question", "prompt"),
|
|
56
|
+
output_keys: tuple[str, ...] = ("output", "response", "answer"),
|
|
57
|
+
input_keys: tuple[str, ...] = ("input", "context"),
|
|
58
|
+
) -> Iterator[TrainingExample]:
|
|
59
|
+
"""Yield zero or more :class:`TrainingExample` from a single raw record.
|
|
60
|
+
|
|
61
|
+
``pairwise_mode`` controls how ``conversation_a`` / ``conversation_b`` rows
|
|
62
|
+
are handled:
|
|
63
|
+
|
|
64
|
+
- ``"winner"`` (default): emit only the branch named by the ``winner`` field;
|
|
65
|
+
emit nothing when ``winner`` is absent or unrecognised.
|
|
66
|
+
- ``"both"``: emit both branches as independent examples.
|
|
67
|
+
- ``"a"`` / ``"b"``: always emit the chosen branch.
|
|
68
|
+
"""
|
|
69
|
+
role_map = role_map or DEFAULT_ROLE_MAP
|
|
70
|
+
|
|
71
|
+
if "conversation_a" in record and "conversation_b" in record:
|
|
72
|
+
yield from _iter_pairwise(
|
|
73
|
+
record,
|
|
74
|
+
role_keys=role_keys,
|
|
75
|
+
content_keys=content_keys,
|
|
76
|
+
role_map=role_map,
|
|
77
|
+
pairwise_mode=pairwise_mode,
|
|
78
|
+
)
|
|
79
|
+
return
|
|
80
|
+
|
|
81
|
+
for key in conversation_keys:
|
|
82
|
+
convs = record.get(key)
|
|
83
|
+
if isinstance(convs, list) and convs:
|
|
84
|
+
msgs = _coerce_messages(
|
|
85
|
+
convs, role_keys=role_keys, content_keys=content_keys, role_map=role_map
|
|
86
|
+
)
|
|
87
|
+
if msgs:
|
|
88
|
+
yield TrainingExample(messages=msgs, meta={"source": "chat"})
|
|
89
|
+
return
|
|
90
|
+
|
|
91
|
+
txt = record.get("text")
|
|
92
|
+
if isinstance(txt, str) and txt.strip():
|
|
93
|
+
yield TrainingExample(
|
|
94
|
+
messages=[ChatMessage(role="assistant", content=txt.strip())],
|
|
95
|
+
meta={"source": "chat:text"},
|
|
96
|
+
)
|
|
97
|
+
return
|
|
98
|
+
|
|
99
|
+
# Fall back to the alpaca adapter, but let callers override the key priority.
|
|
100
|
+
remapped = _remap_for_alpaca(record, instruction_keys, input_keys, output_keys)
|
|
101
|
+
if remapped is not None:
|
|
102
|
+
yield from iter_from_alpaca_line(remapped)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _iter_pairwise(
|
|
106
|
+
record: dict[str, Any],
|
|
107
|
+
*,
|
|
108
|
+
role_keys: tuple[str, ...],
|
|
109
|
+
content_keys: tuple[str, ...],
|
|
110
|
+
role_map: dict[str, str],
|
|
111
|
+
pairwise_mode: str,
|
|
112
|
+
) -> Iterator[TrainingExample]:
|
|
113
|
+
a = record.get("conversation_a")
|
|
114
|
+
b = record.get("conversation_b")
|
|
115
|
+
winner = str(record.get("winner") or "").lower().strip()
|
|
116
|
+
|
|
117
|
+
branches: list[tuple[str, Any]] = []
|
|
118
|
+
if pairwise_mode == "both":
|
|
119
|
+
branches = [("a", a), ("b", b)]
|
|
120
|
+
elif pairwise_mode == "a":
|
|
121
|
+
branches = [("a", a)]
|
|
122
|
+
elif pairwise_mode == "b":
|
|
123
|
+
branches = [("b", b)]
|
|
124
|
+
elif pairwise_mode == "winner":
|
|
125
|
+
if winner in ("model_a", "a"):
|
|
126
|
+
branches = [("a", a)]
|
|
127
|
+
elif winner in ("model_b", "b"):
|
|
128
|
+
branches = [("b", b)]
|
|
129
|
+
# Tie / unknown: emit nothing.
|
|
130
|
+
else:
|
|
131
|
+
raise ValueError(
|
|
132
|
+
f"Unknown pairwise_mode {pairwise_mode!r}. Use 'winner', 'both', 'a', or 'b'."
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
for label, convs in branches:
|
|
136
|
+
if not isinstance(convs, list) or not convs:
|
|
137
|
+
continue
|
|
138
|
+
msgs = _coerce_messages(
|
|
139
|
+
convs, role_keys=role_keys, content_keys=content_keys, role_map=role_map
|
|
140
|
+
)
|
|
141
|
+
if msgs:
|
|
142
|
+
yield TrainingExample(
|
|
143
|
+
messages=msgs,
|
|
144
|
+
meta={"source": "chat:pairwise", "branch": label},
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _coerce_messages(
|
|
149
|
+
convs: list[Any],
|
|
150
|
+
*,
|
|
151
|
+
role_keys: tuple[str, ...],
|
|
152
|
+
content_keys: tuple[str, ...],
|
|
153
|
+
role_map: dict[str, str],
|
|
154
|
+
) -> list[ChatMessage]:
|
|
155
|
+
out: list[ChatMessage] = []
|
|
156
|
+
for item in convs:
|
|
157
|
+
if not isinstance(item, dict):
|
|
158
|
+
continue
|
|
159
|
+
role_raw: str | None = None
|
|
160
|
+
for rk in role_keys:
|
|
161
|
+
v = item.get(rk)
|
|
162
|
+
if isinstance(v, str) and v.strip():
|
|
163
|
+
role_raw = v.strip().lower()
|
|
164
|
+
break
|
|
165
|
+
if role_raw is None:
|
|
166
|
+
continue
|
|
167
|
+
role = role_map.get(role_raw, role_raw)
|
|
168
|
+
|
|
169
|
+
content: str | None = None
|
|
170
|
+
for ck in content_keys:
|
|
171
|
+
v = item.get(ck)
|
|
172
|
+
if isinstance(v, str):
|
|
173
|
+
content = v
|
|
174
|
+
break
|
|
175
|
+
if content is None:
|
|
176
|
+
continue
|
|
177
|
+
out.append(ChatMessage(role=role, content=content))
|
|
178
|
+
return out
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def _remap_for_alpaca(
|
|
182
|
+
record: dict[str, Any],
|
|
183
|
+
instruction_keys: tuple[str, ...],
|
|
184
|
+
input_keys: tuple[str, ...],
|
|
185
|
+
output_keys: tuple[str, ...],
|
|
186
|
+
) -> dict[str, Any] | None:
|
|
187
|
+
"""Pick the first matching key for each slot and return a standard alpaca row."""
|
|
188
|
+
instr = _first_string(record, instruction_keys)
|
|
189
|
+
out = _first_string(record, output_keys)
|
|
190
|
+
if instr is None and out is None:
|
|
191
|
+
return None
|
|
192
|
+
inp = _first_string(record, input_keys) or ""
|
|
193
|
+
return {
|
|
194
|
+
"instruction": instr or "",
|
|
195
|
+
"input": inp,
|
|
196
|
+
"output": out or "",
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def _first_string(record: dict[str, Any], keys: tuple[str, ...]) -> str | None:
|
|
201
|
+
for k in keys:
|
|
202
|
+
v = record.get(k)
|
|
203
|
+
if isinstance(v, str) and v.strip():
|
|
204
|
+
return v
|
|
205
|
+
return None
|