convmerge 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
convmerge/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """convmerge — merge heterogeneous sources into a single LLM training format."""
2
+
3
+ __version__ = "0.2.0"
convmerge/__main__.py ADDED
@@ -0,0 +1,6 @@
1
+ """Allow ``python -m convmerge``."""
2
+
3
+ from convmerge.cli import main
4
+
5
+ if __name__ == "__main__":
6
+ main()
@@ -0,0 +1,28 @@
1
+ """Source-format adapters: raw records → TrainingExample."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Callable, Iterator
6
+ from typing import Any
7
+
8
+ from convmerge.adapters.alpaca import iter_from_alpaca_line
9
+ from convmerge.adapters.chat import iter_from_chat_line
10
+ from convmerge.adapters.sharegpt import iter_from_sharegpt_line
11
+ from convmerge.models import TrainingExample
12
+
13
+ AdapterFn = Callable[[dict[str, Any]], Iterator[TrainingExample]]
14
+
15
+ ADAPTERS: dict[str, AdapterFn] = {
16
+ "alpaca": iter_from_alpaca_line,
17
+ "sharegpt": iter_from_sharegpt_line,
18
+ "chat": iter_from_chat_line,
19
+ # ``auto`` is an alias for ``chat`` since the chat adapter is already auto-detecting.
20
+ "auto": iter_from_chat_line,
21
+ }
22
+
23
+
24
+ def get_adapter(name: str) -> AdapterFn:
25
+ if name not in ADAPTERS:
26
+ known = ", ".join(sorted(ADAPTERS))
27
+ raise ValueError(f"Unknown adapter {name!r}. Choose one of: {known}")
28
+ return ADAPTERS[name]
@@ -0,0 +1,38 @@
1
+ """Alpaca-style instruction / input / output → TrainingExample."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Iterator
6
+ from typing import Any
7
+
8
+ from convmerge.models import ChatMessage, TrainingExample
9
+
10
+
11
+ def iter_from_alpaca_line(record: dict[str, Any]) -> Iterator[TrainingExample]:
12
+ """
13
+ One JSON object per line: instruction, optional input, output.
14
+
15
+ Maps to a single user message + single assistant message.
16
+ """
17
+ instruction = (record.get("instruction") or "").strip()
18
+ inp = (record.get("input") or "").strip()
19
+ output = (record.get("output") or "").strip() or (record.get("response") or "").strip()
20
+
21
+ user_parts = [instruction]
22
+ if inp:
23
+ user_parts.append(inp)
24
+ user_content = "\n".join(user_parts).strip()
25
+
26
+ if not user_content and not output:
27
+ return
28
+
29
+ messages: list[ChatMessage] = []
30
+ if user_content:
31
+ messages.append(ChatMessage(role="user", content=user_content))
32
+ if output:
33
+ messages.append(ChatMessage(role="assistant", content=output))
34
+
35
+ if not messages:
36
+ return
37
+
38
+ yield TrainingExample(messages=messages, meta={"source": "alpaca"})
@@ -0,0 +1,205 @@
1
+ """Auto-detecting chat adapter.
2
+
3
+ Routes a raw record to the right internal shape by looking at which keys are
4
+ present. Handles the common messy shapes seen across SFT datasets:
5
+
6
+ - ``messages`` / ``conversation`` / ``conversations`` lists with
7
+ ``{role, content}`` or ``{from, value}`` entries.
8
+ - Pairwise preference rows (``conversation_a`` / ``conversation_b``), with an
9
+ optional ``winner`` field; emits only the winner branch by default.
10
+ - Plain ``text`` strings (yielded as a single assistant message).
11
+ - Alpaca-style ``instruction`` / ``input`` / ``output`` (delegates to the
12
+ existing alpaca adapter).
13
+
14
+ Users can override the key lists and role map to teach it about bespoke schemas
15
+ without writing a new adapter from scratch.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ from collections.abc import Iterator
21
+ from typing import Any
22
+
23
+ from convmerge.adapters.alpaca import iter_from_alpaca_line
24
+ from convmerge.models import ChatMessage, TrainingExample
25
+
26
+ # Default mapping from common ShareGPT-style ``from`` values onto standard roles.
27
+ DEFAULT_ROLE_MAP: dict[str, str] = {
28
+ "human": "user",
29
+ "user": "user",
30
+ "gpt": "assistant",
31
+ "assistant": "assistant",
32
+ "bing": "assistant",
33
+ "bot": "assistant",
34
+ "system": "system",
35
+ }
36
+
37
+ # Keys searched for the chat-list container, in priority order.
38
+ DEFAULT_CONVERSATION_KEYS: tuple[str, ...] = ("messages", "conversation", "conversations")
39
+
40
+ # Keys treated as role labels inside a chat-list entry.
41
+ DEFAULT_ROLE_KEYS: tuple[str, ...] = ("role", "from")
42
+
43
+ # Keys treated as message content inside a chat-list entry.
44
+ DEFAULT_CONTENT_KEYS: tuple[str, ...] = ("content", "value", "text")
45
+
46
+
47
+ def iter_from_chat_line(
48
+ record: dict[str, Any],
49
+ *,
50
+ conversation_keys: tuple[str, ...] = DEFAULT_CONVERSATION_KEYS,
51
+ role_keys: tuple[str, ...] = DEFAULT_ROLE_KEYS,
52
+ content_keys: tuple[str, ...] = DEFAULT_CONTENT_KEYS,
53
+ role_map: dict[str, str] | None = None,
54
+ pairwise_mode: str = "winner",
55
+ instruction_keys: tuple[str, ...] = ("instruction", "question", "prompt"),
56
+ output_keys: tuple[str, ...] = ("output", "response", "answer"),
57
+ input_keys: tuple[str, ...] = ("input", "context"),
58
+ ) -> Iterator[TrainingExample]:
59
+ """Yield zero or more :class:`TrainingExample` from a single raw record.
60
+
61
+ ``pairwise_mode`` controls how ``conversation_a`` / ``conversation_b`` rows
62
+ are handled:
63
+
64
+ - ``"winner"`` (default): emit only the branch named by the ``winner`` field;
65
+ emit nothing when ``winner`` is absent or unrecognised.
66
+ - ``"both"``: emit both branches as independent examples.
67
+ - ``"a"`` / ``"b"``: always emit the chosen branch.
68
+ """
69
+ role_map = role_map or DEFAULT_ROLE_MAP
70
+
71
+ if "conversation_a" in record and "conversation_b" in record:
72
+ yield from _iter_pairwise(
73
+ record,
74
+ role_keys=role_keys,
75
+ content_keys=content_keys,
76
+ role_map=role_map,
77
+ pairwise_mode=pairwise_mode,
78
+ )
79
+ return
80
+
81
+ for key in conversation_keys:
82
+ convs = record.get(key)
83
+ if isinstance(convs, list) and convs:
84
+ msgs = _coerce_messages(
85
+ convs, role_keys=role_keys, content_keys=content_keys, role_map=role_map
86
+ )
87
+ if msgs:
88
+ yield TrainingExample(messages=msgs, meta={"source": "chat"})
89
+ return
90
+
91
+ txt = record.get("text")
92
+ if isinstance(txt, str) and txt.strip():
93
+ yield TrainingExample(
94
+ messages=[ChatMessage(role="assistant", content=txt.strip())],
95
+ meta={"source": "chat:text"},
96
+ )
97
+ return
98
+
99
+ # Fall back to the alpaca adapter, but let callers override the key priority.
100
+ remapped = _remap_for_alpaca(record, instruction_keys, input_keys, output_keys)
101
+ if remapped is not None:
102
+ yield from iter_from_alpaca_line(remapped)
103
+
104
+
105
+ def _iter_pairwise(
106
+ record: dict[str, Any],
107
+ *,
108
+ role_keys: tuple[str, ...],
109
+ content_keys: tuple[str, ...],
110
+ role_map: dict[str, str],
111
+ pairwise_mode: str,
112
+ ) -> Iterator[TrainingExample]:
113
+ a = record.get("conversation_a")
114
+ b = record.get("conversation_b")
115
+ winner = str(record.get("winner") or "").lower().strip()
116
+
117
+ branches: list[tuple[str, Any]] = []
118
+ if pairwise_mode == "both":
119
+ branches = [("a", a), ("b", b)]
120
+ elif pairwise_mode == "a":
121
+ branches = [("a", a)]
122
+ elif pairwise_mode == "b":
123
+ branches = [("b", b)]
124
+ elif pairwise_mode == "winner":
125
+ if winner in ("model_a", "a"):
126
+ branches = [("a", a)]
127
+ elif winner in ("model_b", "b"):
128
+ branches = [("b", b)]
129
+ # Tie / unknown: emit nothing.
130
+ else:
131
+ raise ValueError(
132
+ f"Unknown pairwise_mode {pairwise_mode!r}. Use 'winner', 'both', 'a', or 'b'."
133
+ )
134
+
135
+ for label, convs in branches:
136
+ if not isinstance(convs, list) or not convs:
137
+ continue
138
+ msgs = _coerce_messages(
139
+ convs, role_keys=role_keys, content_keys=content_keys, role_map=role_map
140
+ )
141
+ if msgs:
142
+ yield TrainingExample(
143
+ messages=msgs,
144
+ meta={"source": "chat:pairwise", "branch": label},
145
+ )
146
+
147
+
148
+ def _coerce_messages(
149
+ convs: list[Any],
150
+ *,
151
+ role_keys: tuple[str, ...],
152
+ content_keys: tuple[str, ...],
153
+ role_map: dict[str, str],
154
+ ) -> list[ChatMessage]:
155
+ out: list[ChatMessage] = []
156
+ for item in convs:
157
+ if not isinstance(item, dict):
158
+ continue
159
+ role_raw: str | None = None
160
+ for rk in role_keys:
161
+ v = item.get(rk)
162
+ if isinstance(v, str) and v.strip():
163
+ role_raw = v.strip().lower()
164
+ break
165
+ if role_raw is None:
166
+ continue
167
+ role = role_map.get(role_raw, role_raw)
168
+
169
+ content: str | None = None
170
+ for ck in content_keys:
171
+ v = item.get(ck)
172
+ if isinstance(v, str):
173
+ content = v
174
+ break
175
+ if content is None:
176
+ continue
177
+ out.append(ChatMessage(role=role, content=content))
178
+ return out
179
+
180
+
181
+ def _remap_for_alpaca(
182
+ record: dict[str, Any],
183
+ instruction_keys: tuple[str, ...],
184
+ input_keys: tuple[str, ...],
185
+ output_keys: tuple[str, ...],
186
+ ) -> dict[str, Any] | None:
187
+ """Pick the first matching key for each slot and return a standard alpaca row."""
188
+ instr = _first_string(record, instruction_keys)
189
+ out = _first_string(record, output_keys)
190
+ if instr is None and out is None:
191
+ return None
192
+ inp = _first_string(record, input_keys) or ""
193
+ return {
194
+ "instruction": instr or "",
195
+ "input": inp,
196
+ "output": out or "",
197
+ }
198
+
199
+
200
+ def _first_string(record: dict[str, Any], keys: tuple[str, ...]) -> str | None:
201
+ for k in keys:
202
+ v = record.get(k)
203
+ if isinstance(v, str) and v.strip():
204
+ return v
205
+ return None
@@ -0,0 +1,55 @@
1
+ """ShareGPT-style conversations (from/value) → TrainingExample."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Iterator
6
+ from typing import Any
7
+
8
+ from convmerge.models import ChatMessage, TrainingExample
9
+
10
+ # Common ShareGPT role labels
11
+ _FROM_TO_ROLE: dict[str, str] = {
12
+ "human": "user",
13
+ "user": "user",
14
+ "gpt": "assistant",
15
+ "assistant": "assistant",
16
+ "system": "system",
17
+ "bing": "assistant",
18
+ }
19
+
20
+
21
+ def _normalize_role(from_key: str) -> str:
22
+ return _FROM_TO_ROLE.get(from_key.lower().strip(), from_key.lower().strip())
23
+
24
+
25
+ def iter_from_sharegpt_line(record: dict[str, Any]) -> Iterator[TrainingExample]:
26
+ """
27
+ One JSON object with ``conversations``: list of ``{"from": ..., "value": ...}``.
28
+
29
+ Emits one TrainingExample per consecutive human→assistant pair (typical SFT).
30
+ """
31
+ convs = record.get("conversations")
32
+ if not isinstance(convs, list) or len(convs) < 2:
33
+ return
34
+
35
+ i = 0
36
+ while i + 1 < len(convs):
37
+ a, b = convs[i], convs[i + 1]
38
+ if not isinstance(a, dict) or not isinstance(b, dict):
39
+ i += 1
40
+ continue
41
+ ra = _normalize_role(str(a.get("from", "")))
42
+ rb = _normalize_role(str(b.get("from", "")))
43
+ va = (a.get("value") or "").strip()
44
+ vb = (b.get("value") or "").strip()
45
+ if ra == "user" and rb == "assistant":
46
+ yield TrainingExample(
47
+ messages=[
48
+ ChatMessage(role="user", content=va),
49
+ ChatMessage(role="assistant", content=vb),
50
+ ],
51
+ meta={"source": "sharegpt"},
52
+ )
53
+ i += 2
54
+ else:
55
+ i += 1