convmerge 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,14 @@
1
+ .venv/
2
+ venv/
3
+ __pycache__/
4
+ *.py[cod]
5
+ .pytest_cache/
6
+ .ruff_cache/
7
+ .mypy_cache/
8
+ dist/
9
+ build/
10
+ *.egg-info/
11
+ .coverage
12
+ htmlcov/
13
+ *.egg
14
+ .DS_Store
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 convmerge contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,154 @@
1
+ Metadata-Version: 2.4
2
+ Name: convmerge
3
+ Version: 0.2.0
4
+ Summary: Fetch, normalize, and convert heterogeneous chat/instruct datasets into a single LLM training format
5
+ Project-URL: Homepage, https://github.com/snowmuffin/convmerge
6
+ Project-URL: Repository, https://github.com/snowmuffin/convmerge
7
+ Project-URL: Issues, https://github.com/snowmuffin/convmerge/issues
8
+ Author: convmerge contributors
9
+ License-Expression: MIT
10
+ License-File: LICENSE
11
+ Keywords: chat,dataset,jsonl,llm,sft
12
+ Classifier: Development Status :: 2 - Pre-Alpha
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: License :: OSI Approved :: MIT License
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Requires-Python: >=3.10
20
+ Provides-Extra: dev
21
+ Requires-Dist: pytest>=8.0; extra == 'dev'
22
+ Requires-Dist: ruff>=0.4; extra == 'dev'
23
+ Provides-Extra: fetch
24
+ Requires-Dist: pyyaml>=6.0; extra == 'fetch'
25
+ Provides-Extra: fetch-all
26
+ Requires-Dist: datasets>=2.16; extra == 'fetch-all'
27
+ Requires-Dist: pyyaml>=6.0; extra == 'fetch-all'
28
+ Provides-Extra: fetch-hf
29
+ Requires-Dist: datasets>=2.16; extra == 'fetch-hf'
30
+ Requires-Dist: pyyaml>=6.0; extra == 'fetch-hf'
31
+ Provides-Extra: parquet
32
+ Requires-Dist: pyarrow>=14; extra == 'parquet'
33
+ Description-Content-Type: text/markdown
34
+
35
+ # convmerge
36
+
37
+ Fetch, normalize, and convert heterogeneous chat / instruct datasets into a
38
+ **single LLM training format** (JSONL).
39
+
40
+ **Repository:** [github.com/snowmuffin/convmerge](https://github.com/snowmuffin/convmerge)
41
+ **Status:** pre-1.0; APIs and CLI may change between minor versions.
42
+
43
+ ## Install
44
+
45
+ ```bash
46
+ pip install convmerge # core: convert, normalize, dedupe, turns
47
+ pip install "convmerge[fetch]" # + YAML manifest fetcher (GitHub)
48
+ pip install "convmerge[fetch-hf]" # + HuggingFace entries (adds ``datasets``)
49
+ pip install "convmerge[fetch-all]" # all fetch-related extras
50
+ pip install "convmerge[parquet]" # + parquet streaming input
51
+ ```
52
+
53
+ Or from a clone:
54
+
55
+ ```bash
56
+ git clone https://github.com/snowmuffin/convmerge.git
57
+ cd convmerge
58
+ python -m venv .venv && source .venv/bin/activate
59
+ pip install -e ".[dev,fetch-all,parquet]"
60
+ ```
61
+
62
+ ## The four use cases
63
+
64
+ ### 1. `fetch` — pull raw data from HF + GitHub via a YAML manifest
65
+
66
+ ```yaml
67
+ # manifest.yaml
68
+ version: 1
69
+ defaults: { output_root: ./raw, resume: true }
70
+ auth: { hf_token_env: HF_TOKEN, github_token_env: GITHUB_TOKEN }
71
+ datasets:
72
+ - { name: alpaca-ko, hf: MarkrAI/KoCommercial-Dataset, split: train }
73
+ - { name: orca-raw,
74
+ url: https://raw.githubusercontent.com/org/repo/main/data/train.jsonl }
75
+ - { name: repo-tree,
76
+ url: https://github.com/org/example-repo, ext: [".jsonl"] }
77
+ - { name: big-lfs,
78
+ url: https://github.com/org/big-lfs-repo, mode: clone, lfs: true }
79
+ ```
80
+
81
+ ```bash
82
+ convmerge fetch manifest.yaml -o ./raw
83
+ # or one-shot shortcuts:
84
+ convmerge fetch hf://org/dataset -o ./raw --split train
85
+ convmerge fetch https://github.com/org/repo -o ./raw --ext .jsonl
86
+ ```
87
+
88
+ Tokens resolve in order CLI flag → file → env var, and are redacted from logs.
89
+ See [docs/fetch.md](docs/fetch.md) for the full schema.
90
+
91
+ ### 2. `normalize` — reshape parquet / messy JSON into clean JSONL
92
+
93
+ ```bash
94
+ convmerge normalize -i ./raw -o ./jsonl
95
+ ```
96
+
97
+ Handles parquet (streamed via `pyarrow`), top-level JSON arrays, concatenated
98
+ single-line JSON (`{...}{...}{...}`), and already-valid JSONL. A directory
99
+ input is walked recursively and mirrored under the output directory.
100
+
101
+ ### 3. `convert` — adapter + emitter pipeline
102
+
103
+ ```bash
104
+ convmerge convert -i ./jsonl/alpaca.jsonl -o ./train/alpaca.messages.jsonl \
105
+ --from alpaca --format messages
106
+
107
+ convmerge convert -i ./jsonl/mixed.jsonl -o ./train/mixed.messages.jsonl \
108
+ --from auto --format messages # auto-detecting chat adapter
109
+ ```
110
+
111
+ Adapters: `alpaca`, `sharegpt`, `chat` (alias `auto`).
112
+ Emitters: `messages`, `alpaca`.
113
+
114
+ ### 4. `dedupe` / `turns` — final cleanup + train/eval split hook
115
+
116
+ ```bash
117
+ convmerge dedupe -i ./train/mixed.messages.jsonl -o ./train/mixed.dedup.jsonl
118
+ convmerge turns -i ./train/mixed.dedup.jsonl \
119
+ --single-out ./train/single.jsonl \
120
+ --multi-out ./train/multi.jsonl
121
+ ```
122
+
123
+ See [docs/format.md](docs/format.md) for adapter / emitter schemas and
124
+ [docs/fetch.md](docs/fetch.md) for manifest details.
125
+
126
+ ## Development
127
+
128
+ See [CONTRIBUTING.md](CONTRIBUTING.md). CI runs Ruff + pytest on Python
129
+ 3.10 – 3.12.
130
+
131
+ ```bash
132
+ ruff check src tests
133
+ pytest -q
134
+ ```
135
+
136
+ ## PyPI release (maintainers)
137
+
138
+ Releases run from [`.github/workflows/publish.yml`](.github/workflows/publish.yml)
139
+ on pushing a `v*` tag. Publishing authenticates via the **`PYPI_API_TOKEN`**
140
+ GitHub Actions secret (a PyPI API token scoped to the `convmerge` project).
141
+
142
+ 1. Create an API token on [pypi.org](https://pypi.org/manage/account/token/)
143
+ scoped to `convmerge`.
144
+ 2. In the GitHub repo, *Settings → Secrets and variables → Actions → New
145
+ repository secret*, add `PYPI_API_TOKEN` with the token value.
146
+ 3. Tag and push: `git tag v0.2.0 && git push origin v0.2.0`.
147
+
148
+ ## Changelog
149
+
150
+ [CHANGELOG.md](CHANGELOG.md)
151
+
152
+ ## License
153
+
154
+ MIT
@@ -0,0 +1,120 @@
1
+ # convmerge
2
+
3
+ Fetch, normalize, and convert heterogeneous chat / instruct datasets into a
4
+ **single LLM training format** (JSONL).
5
+
6
+ **Repository:** [github.com/snowmuffin/convmerge](https://github.com/snowmuffin/convmerge)
7
+ **Status:** pre-1.0; APIs and CLI may change between minor versions.
8
+
9
+ ## Install
10
+
11
+ ```bash
12
+ pip install convmerge # core: convert, normalize, dedupe, turns
13
+ pip install "convmerge[fetch]" # + YAML manifest fetcher (GitHub)
14
+ pip install "convmerge[fetch-hf]" # + HuggingFace entries (adds ``datasets``)
15
+ pip install "convmerge[fetch-all]" # all fetch-related extras
16
+ pip install "convmerge[parquet]" # + parquet streaming input
17
+ ```
18
+
19
+ Or from a clone:
20
+
21
+ ```bash
22
+ git clone https://github.com/snowmuffin/convmerge.git
23
+ cd convmerge
24
+ python -m venv .venv && source .venv/bin/activate
25
+ pip install -e ".[dev,fetch-all,parquet]"
26
+ ```
27
+
28
+ ## The four use cases
29
+
30
+ ### 1. `fetch` — pull raw data from HF + GitHub via a YAML manifest
31
+
32
+ ```yaml
33
+ # manifest.yaml
34
+ version: 1
35
+ defaults: { output_root: ./raw, resume: true }
36
+ auth: { hf_token_env: HF_TOKEN, github_token_env: GITHUB_TOKEN }
37
+ datasets:
38
+ - { name: alpaca-ko, hf: MarkrAI/KoCommercial-Dataset, split: train }
39
+ - { name: orca-raw,
40
+ url: https://raw.githubusercontent.com/org/repo/main/data/train.jsonl }
41
+ - { name: repo-tree,
42
+ url: https://github.com/org/example-repo, ext: [".jsonl"] }
43
+ - { name: big-lfs,
44
+ url: https://github.com/org/big-lfs-repo, mode: clone, lfs: true }
45
+ ```
46
+
47
+ ```bash
48
+ convmerge fetch manifest.yaml -o ./raw
49
+ # or one-shot shortcuts:
50
+ convmerge fetch hf://org/dataset -o ./raw --split train
51
+ convmerge fetch https://github.com/org/repo -o ./raw --ext .jsonl
52
+ ```
53
+
54
+ Tokens resolve in order CLI flag → file → env var, and are redacted from logs.
55
+ See [docs/fetch.md](docs/fetch.md) for the full schema.
56
+
57
+ ### 2. `normalize` — reshape parquet / messy JSON into clean JSONL
58
+
59
+ ```bash
60
+ convmerge normalize -i ./raw -o ./jsonl
61
+ ```
62
+
63
+ Handles parquet (streamed via `pyarrow`), top-level JSON arrays, concatenated
64
+ single-line JSON (`{...}{...}{...}`), and already-valid JSONL. A directory
65
+ input is walked recursively and mirrored under the output directory.
66
+
67
+ ### 3. `convert` — adapter + emitter pipeline
68
+
69
+ ```bash
70
+ convmerge convert -i ./jsonl/alpaca.jsonl -o ./train/alpaca.messages.jsonl \
71
+ --from alpaca --format messages
72
+
73
+ convmerge convert -i ./jsonl/mixed.jsonl -o ./train/mixed.messages.jsonl \
74
+ --from auto --format messages # auto-detecting chat adapter
75
+ ```
76
+
77
+ Adapters: `alpaca`, `sharegpt`, `chat` (alias `auto`).
78
+ Emitters: `messages`, `alpaca`.
79
+
80
+ ### 4. `dedupe` / `turns` — final cleanup + train/eval split hook
81
+
82
+ ```bash
83
+ convmerge dedupe -i ./train/mixed.messages.jsonl -o ./train/mixed.dedup.jsonl
84
+ convmerge turns -i ./train/mixed.dedup.jsonl \
85
+ --single-out ./train/single.jsonl \
86
+ --multi-out ./train/multi.jsonl
87
+ ```
88
+
89
+ See [docs/format.md](docs/format.md) for adapter / emitter schemas and
90
+ [docs/fetch.md](docs/fetch.md) for manifest details.
91
+
92
+ ## Development
93
+
94
+ See [CONTRIBUTING.md](CONTRIBUTING.md). CI runs Ruff + pytest on Python
95
+ 3.10 – 3.12.
96
+
97
+ ```bash
98
+ ruff check src tests
99
+ pytest -q
100
+ ```
101
+
102
+ ## PyPI release (maintainers)
103
+
104
+ Releases run from [`.github/workflows/publish.yml`](.github/workflows/publish.yml)
105
+ on pushing a `v*` tag. Publishing authenticates via the **`PYPI_API_TOKEN`**
106
+ GitHub Actions secret (a PyPI API token scoped to the `convmerge` project).
107
+
108
+ 1. Create an API token on [pypi.org](https://pypi.org/manage/account/token/)
109
+ scoped to `convmerge`.
110
+ 2. In the GitHub repo, *Settings → Secrets and variables → Actions → New
111
+ repository secret*, add `PYPI_API_TOKEN` with the token value.
112
+ 3. Tag and push: `git tag v0.2.0 && git push origin v0.2.0`.
113
+
114
+ ## Changelog
115
+
116
+ [CHANGELOG.md](CHANGELOG.md)
117
+
118
+ ## License
119
+
120
+ MIT
@@ -0,0 +1,62 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "convmerge"
7
+ version = "0.2.0"
8
+ description = "Fetch, normalize, and convert heterogeneous chat/instruct datasets into a single LLM training format"
9
+ readme = "README.md"
10
+ license = "MIT"
11
+ requires-python = ">=3.10"
12
+ authors = [{ name = "convmerge contributors" }]
13
+ keywords = ["llm", "sft", "jsonl", "chat", "dataset"]
14
+ classifiers = [
15
+ "Development Status :: 2 - Pre-Alpha",
16
+ "Intended Audience :: Developers",
17
+ "License :: OSI Approved :: MIT License",
18
+ "Programming Language :: Python :: 3",
19
+ "Programming Language :: Python :: 3.10",
20
+ "Programming Language :: Python :: 3.11",
21
+ "Programming Language :: Python :: 3.12",
22
+ ]
23
+
24
+ dependencies = []
25
+
26
+ [project.optional-dependencies]
27
+ # YAML manifest fetcher (GitHub raw / Trees API / git clone). Uses stdlib urllib for HTTP.
28
+ fetch = ["pyyaml>=6.0"]
29
+ # HuggingFace support for fetch (delegates to the ``datasets`` library).
30
+ fetch-hf = ["pyyaml>=6.0", "datasets>=2.16"]
31
+ # Alias: everything fetch-related.
32
+ fetch-all = ["pyyaml>=6.0", "datasets>=2.16"]
33
+ # Parquet streaming input for normalize.
34
+ parquet = ["pyarrow>=14"]
35
+ dev = ["pytest>=8.0", "ruff>=0.4"]
36
+
37
+ [project.scripts]
38
+ convmerge = "convmerge.cli:main"
39
+
40
+ [project.urls]
41
+ Homepage = "https://github.com/snowmuffin/convmerge"
42
+ Repository = "https://github.com/snowmuffin/convmerge"
43
+ Issues = "https://github.com/snowmuffin/convmerge/issues"
44
+
45
+ [tool.hatch.build.targets.wheel]
46
+ packages = ["src/convmerge"]
47
+
48
+ [tool.hatch.build.targets.sdist]
49
+ include = ["/src"]
50
+
51
+ [tool.ruff]
52
+ line-length = 100
53
+ target-version = "py310"
54
+
55
+ [tool.ruff.format]
56
+ quote-style = "double"
57
+
58
+ [tool.ruff.lint]
59
+ select = ["E", "F", "I", "UP"]
60
+
61
+ [tool.pytest.ini_options]
62
+ testpaths = ["tests"]
@@ -0,0 +1,3 @@
1
+ """convmerge — merge heterogeneous sources into a single LLM training format."""
2
+
3
+ __version__ = "0.2.0"
@@ -0,0 +1,6 @@
1
+ """Allow ``python -m convmerge``."""
2
+
3
+ from convmerge.cli import main
4
+
5
+ if __name__ == "__main__":
6
+ main()
@@ -0,0 +1,28 @@
1
+ """Source-format adapters: raw records → TrainingExample."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Callable, Iterator
6
+ from typing import Any
7
+
8
+ from convmerge.adapters.alpaca import iter_from_alpaca_line
9
+ from convmerge.adapters.chat import iter_from_chat_line
10
+ from convmerge.adapters.sharegpt import iter_from_sharegpt_line
11
+ from convmerge.models import TrainingExample
12
+
13
+ AdapterFn = Callable[[dict[str, Any]], Iterator[TrainingExample]]
14
+
15
+ ADAPTERS: dict[str, AdapterFn] = {
16
+ "alpaca": iter_from_alpaca_line,
17
+ "sharegpt": iter_from_sharegpt_line,
18
+ "chat": iter_from_chat_line,
19
+ # ``auto`` is an alias for ``chat`` since the chat adapter is already auto-detecting.
20
+ "auto": iter_from_chat_line,
21
+ }
22
+
23
+
24
+ def get_adapter(name: str) -> AdapterFn:
25
+ if name not in ADAPTERS:
26
+ known = ", ".join(sorted(ADAPTERS))
27
+ raise ValueError(f"Unknown adapter {name!r}. Choose one of: {known}")
28
+ return ADAPTERS[name]
@@ -0,0 +1,38 @@
1
+ """Alpaca-style instruction / input / output → TrainingExample."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Iterator
6
+ from typing import Any
7
+
8
+ from convmerge.models import ChatMessage, TrainingExample
9
+
10
+
11
+ def iter_from_alpaca_line(record: dict[str, Any]) -> Iterator[TrainingExample]:
12
+ """
13
+ One JSON object per line: instruction, optional input, output.
14
+
15
+ Maps to a single user message + single assistant message.
16
+ """
17
+ instruction = (record.get("instruction") or "").strip()
18
+ inp = (record.get("input") or "").strip()
19
+ output = (record.get("output") or "").strip() or (record.get("response") or "").strip()
20
+
21
+ user_parts = [instruction]
22
+ if inp:
23
+ user_parts.append(inp)
24
+ user_content = "\n".join(user_parts).strip()
25
+
26
+ if not user_content and not output:
27
+ return
28
+
29
+ messages: list[ChatMessage] = []
30
+ if user_content:
31
+ messages.append(ChatMessage(role="user", content=user_content))
32
+ if output:
33
+ messages.append(ChatMessage(role="assistant", content=output))
34
+
35
+ if not messages:
36
+ return
37
+
38
+ yield TrainingExample(messages=messages, meta={"source": "alpaca"})
@@ -0,0 +1,205 @@
1
+ """Auto-detecting chat adapter.
2
+
3
+ Routes a raw record to the right internal shape by looking at which keys are
4
+ present. Handles the common messy shapes seen across SFT datasets:
5
+
6
+ - ``messages`` / ``conversation`` / ``conversations`` lists with
7
+ ``{role, content}`` or ``{from, value}`` entries.
8
+ - Pairwise preference rows (``conversation_a`` / ``conversation_b``), with an
9
+ optional ``winner`` field; emits only the winner branch by default.
10
+ - Plain ``text`` strings (yielded as a single assistant message).
11
+ - Alpaca-style ``instruction`` / ``input`` / ``output`` (delegates to the
12
+ existing alpaca adapter).
13
+
14
+ Users can override the key lists and role map to teach it about bespoke schemas
15
+ without writing a new adapter from scratch.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ from collections.abc import Iterator
21
+ from typing import Any
22
+
23
+ from convmerge.adapters.alpaca import iter_from_alpaca_line
24
+ from convmerge.models import ChatMessage, TrainingExample
25
+
26
+ # Default mapping from common ShareGPT-style ``from`` values onto standard roles.
27
+ DEFAULT_ROLE_MAP: dict[str, str] = {
28
+ "human": "user",
29
+ "user": "user",
30
+ "gpt": "assistant",
31
+ "assistant": "assistant",
32
+ "bing": "assistant",
33
+ "bot": "assistant",
34
+ "system": "system",
35
+ }
36
+
37
+ # Keys searched for the chat-list container, in priority order.
38
+ DEFAULT_CONVERSATION_KEYS: tuple[str, ...] = ("messages", "conversation", "conversations")
39
+
40
+ # Keys treated as role labels inside a chat-list entry.
41
+ DEFAULT_ROLE_KEYS: tuple[str, ...] = ("role", "from")
42
+
43
+ # Keys treated as message content inside a chat-list entry.
44
+ DEFAULT_CONTENT_KEYS: tuple[str, ...] = ("content", "value", "text")
45
+
46
+
47
+ def iter_from_chat_line(
48
+ record: dict[str, Any],
49
+ *,
50
+ conversation_keys: tuple[str, ...] = DEFAULT_CONVERSATION_KEYS,
51
+ role_keys: tuple[str, ...] = DEFAULT_ROLE_KEYS,
52
+ content_keys: tuple[str, ...] = DEFAULT_CONTENT_KEYS,
53
+ role_map: dict[str, str] | None = None,
54
+ pairwise_mode: str = "winner",
55
+ instruction_keys: tuple[str, ...] = ("instruction", "question", "prompt"),
56
+ output_keys: tuple[str, ...] = ("output", "response", "answer"),
57
+ input_keys: tuple[str, ...] = ("input", "context"),
58
+ ) -> Iterator[TrainingExample]:
59
+ """Yield zero or more :class:`TrainingExample` from a single raw record.
60
+
61
+ ``pairwise_mode`` controls how ``conversation_a`` / ``conversation_b`` rows
62
+ are handled:
63
+
64
+ - ``"winner"`` (default): emit only the branch named by the ``winner`` field;
65
+ emit nothing when ``winner`` is absent or unrecognised.
66
+ - ``"both"``: emit both branches as independent examples.
67
+ - ``"a"`` / ``"b"``: always emit the chosen branch.
68
+ """
69
+ role_map = role_map or DEFAULT_ROLE_MAP
70
+
71
+ if "conversation_a" in record and "conversation_b" in record:
72
+ yield from _iter_pairwise(
73
+ record,
74
+ role_keys=role_keys,
75
+ content_keys=content_keys,
76
+ role_map=role_map,
77
+ pairwise_mode=pairwise_mode,
78
+ )
79
+ return
80
+
81
+ for key in conversation_keys:
82
+ convs = record.get(key)
83
+ if isinstance(convs, list) and convs:
84
+ msgs = _coerce_messages(
85
+ convs, role_keys=role_keys, content_keys=content_keys, role_map=role_map
86
+ )
87
+ if msgs:
88
+ yield TrainingExample(messages=msgs, meta={"source": "chat"})
89
+ return
90
+
91
+ txt = record.get("text")
92
+ if isinstance(txt, str) and txt.strip():
93
+ yield TrainingExample(
94
+ messages=[ChatMessage(role="assistant", content=txt.strip())],
95
+ meta={"source": "chat:text"},
96
+ )
97
+ return
98
+
99
+ # Fall back to the alpaca adapter, but let callers override the key priority.
100
+ remapped = _remap_for_alpaca(record, instruction_keys, input_keys, output_keys)
101
+ if remapped is not None:
102
+ yield from iter_from_alpaca_line(remapped)
103
+
104
+
105
+ def _iter_pairwise(
106
+ record: dict[str, Any],
107
+ *,
108
+ role_keys: tuple[str, ...],
109
+ content_keys: tuple[str, ...],
110
+ role_map: dict[str, str],
111
+ pairwise_mode: str,
112
+ ) -> Iterator[TrainingExample]:
113
+ a = record.get("conversation_a")
114
+ b = record.get("conversation_b")
115
+ winner = str(record.get("winner") or "").lower().strip()
116
+
117
+ branches: list[tuple[str, Any]] = []
118
+ if pairwise_mode == "both":
119
+ branches = [("a", a), ("b", b)]
120
+ elif pairwise_mode == "a":
121
+ branches = [("a", a)]
122
+ elif pairwise_mode == "b":
123
+ branches = [("b", b)]
124
+ elif pairwise_mode == "winner":
125
+ if winner in ("model_a", "a"):
126
+ branches = [("a", a)]
127
+ elif winner in ("model_b", "b"):
128
+ branches = [("b", b)]
129
+ # Tie / unknown: emit nothing.
130
+ else:
131
+ raise ValueError(
132
+ f"Unknown pairwise_mode {pairwise_mode!r}. Use 'winner', 'both', 'a', or 'b'."
133
+ )
134
+
135
+ for label, convs in branches:
136
+ if not isinstance(convs, list) or not convs:
137
+ continue
138
+ msgs = _coerce_messages(
139
+ convs, role_keys=role_keys, content_keys=content_keys, role_map=role_map
140
+ )
141
+ if msgs:
142
+ yield TrainingExample(
143
+ messages=msgs,
144
+ meta={"source": "chat:pairwise", "branch": label},
145
+ )
146
+
147
+
148
+ def _coerce_messages(
149
+ convs: list[Any],
150
+ *,
151
+ role_keys: tuple[str, ...],
152
+ content_keys: tuple[str, ...],
153
+ role_map: dict[str, str],
154
+ ) -> list[ChatMessage]:
155
+ out: list[ChatMessage] = []
156
+ for item in convs:
157
+ if not isinstance(item, dict):
158
+ continue
159
+ role_raw: str | None = None
160
+ for rk in role_keys:
161
+ v = item.get(rk)
162
+ if isinstance(v, str) and v.strip():
163
+ role_raw = v.strip().lower()
164
+ break
165
+ if role_raw is None:
166
+ continue
167
+ role = role_map.get(role_raw, role_raw)
168
+
169
+ content: str | None = None
170
+ for ck in content_keys:
171
+ v = item.get(ck)
172
+ if isinstance(v, str):
173
+ content = v
174
+ break
175
+ if content is None:
176
+ continue
177
+ out.append(ChatMessage(role=role, content=content))
178
+ return out
179
+
180
+
181
+ def _remap_for_alpaca(
182
+ record: dict[str, Any],
183
+ instruction_keys: tuple[str, ...],
184
+ input_keys: tuple[str, ...],
185
+ output_keys: tuple[str, ...],
186
+ ) -> dict[str, Any] | None:
187
+ """Pick the first matching key for each slot and return a standard alpaca row."""
188
+ instr = _first_string(record, instruction_keys)
189
+ out = _first_string(record, output_keys)
190
+ if instr is None and out is None:
191
+ return None
192
+ inp = _first_string(record, input_keys) or ""
193
+ return {
194
+ "instruction": instr or "",
195
+ "input": inp,
196
+ "output": out or "",
197
+ }
198
+
199
+
200
+ def _first_string(record: dict[str, Any], keys: tuple[str, ...]) -> str | None:
201
+ for k in keys:
202
+ v = record.get(k)
203
+ if isinstance(v, str) and v.strip():
204
+ return v
205
+ return None