convmerge 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- convmerge/__init__.py +3 -0
- convmerge/__main__.py +6 -0
- convmerge/adapters/__init__.py +28 -0
- convmerge/adapters/alpaca.py +38 -0
- convmerge/adapters/chat.py +205 -0
- convmerge/adapters/sharegpt.py +55 -0
- convmerge/cli.py +367 -0
- convmerge/convert.py +74 -0
- convmerge/emitters.py +62 -0
- convmerge/fetch/__init__.py +38 -0
- convmerge/fetch/auth.py +62 -0
- convmerge/fetch/git.py +71 -0
- convmerge/fetch/github.py +123 -0
- convmerge/fetch/hf.py +47 -0
- convmerge/fetch/manifest.py +178 -0
- convmerge/fetch/runner.py +213 -0
- convmerge/models.py +21 -0
- convmerge/normalize/__init__.py +43 -0
- convmerge/normalize/convert_turns.py +78 -0
- convmerge/normalize/dedup.py +90 -0
- convmerge/normalize/jsonl.py +213 -0
- convmerge/normalize/parquet.py +39 -0
- convmerge/normalize/schema.py +84 -0
- convmerge/normalize/turns.py +90 -0
- convmerge-0.2.0.dist-info/METADATA +154 -0
- convmerge-0.2.0.dist-info/RECORD +29 -0
- convmerge-0.2.0.dist-info/WHEEL +4 -0
- convmerge-0.2.0.dist-info/entry_points.txt +2 -0
- convmerge-0.2.0.dist-info/licenses/LICENSE +21 -0
convmerge/__init__.py
ADDED
convmerge/__main__.py
ADDED
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
"""Source-format adapters: raw records → TrainingExample."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Callable, Iterator
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from convmerge.adapters.alpaca import iter_from_alpaca_line
|
|
9
|
+
from convmerge.adapters.chat import iter_from_chat_line
|
|
10
|
+
from convmerge.adapters.sharegpt import iter_from_sharegpt_line
|
|
11
|
+
from convmerge.models import TrainingExample
|
|
12
|
+
|
|
13
|
+
AdapterFn = Callable[[dict[str, Any]], Iterator[TrainingExample]]
|
|
14
|
+
|
|
15
|
+
ADAPTERS: dict[str, AdapterFn] = {
|
|
16
|
+
"alpaca": iter_from_alpaca_line,
|
|
17
|
+
"sharegpt": iter_from_sharegpt_line,
|
|
18
|
+
"chat": iter_from_chat_line,
|
|
19
|
+
# ``auto`` is an alias for ``chat`` since the chat adapter is already auto-detecting.
|
|
20
|
+
"auto": iter_from_chat_line,
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def get_adapter(name: str) -> AdapterFn:
|
|
25
|
+
if name not in ADAPTERS:
|
|
26
|
+
known = ", ".join(sorted(ADAPTERS))
|
|
27
|
+
raise ValueError(f"Unknown adapter {name!r}. Choose one of: {known}")
|
|
28
|
+
return ADAPTERS[name]
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""Alpaca-style instruction / input / output → TrainingExample."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Iterator
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from convmerge.models import ChatMessage, TrainingExample
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def iter_from_alpaca_line(record: dict[str, Any]) -> Iterator[TrainingExample]:
|
|
12
|
+
"""
|
|
13
|
+
One JSON object per line: instruction, optional input, output.
|
|
14
|
+
|
|
15
|
+
Maps to a single user message + single assistant message.
|
|
16
|
+
"""
|
|
17
|
+
instruction = (record.get("instruction") or "").strip()
|
|
18
|
+
inp = (record.get("input") or "").strip()
|
|
19
|
+
output = (record.get("output") or "").strip() or (record.get("response") or "").strip()
|
|
20
|
+
|
|
21
|
+
user_parts = [instruction]
|
|
22
|
+
if inp:
|
|
23
|
+
user_parts.append(inp)
|
|
24
|
+
user_content = "\n".join(user_parts).strip()
|
|
25
|
+
|
|
26
|
+
if not user_content and not output:
|
|
27
|
+
return
|
|
28
|
+
|
|
29
|
+
messages: list[ChatMessage] = []
|
|
30
|
+
if user_content:
|
|
31
|
+
messages.append(ChatMessage(role="user", content=user_content))
|
|
32
|
+
if output:
|
|
33
|
+
messages.append(ChatMessage(role="assistant", content=output))
|
|
34
|
+
|
|
35
|
+
if not messages:
|
|
36
|
+
return
|
|
37
|
+
|
|
38
|
+
yield TrainingExample(messages=messages, meta={"source": "alpaca"})
|
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
"""Auto-detecting chat adapter.
|
|
2
|
+
|
|
3
|
+
Routes a raw record to the right internal shape by looking at which keys are
|
|
4
|
+
present. Handles the common messy shapes seen across SFT datasets:
|
|
5
|
+
|
|
6
|
+
- ``messages`` / ``conversation`` / ``conversations`` lists with
|
|
7
|
+
``{role, content}`` or ``{from, value}`` entries.
|
|
8
|
+
- Pairwise preference rows (``conversation_a`` / ``conversation_b``), with an
|
|
9
|
+
optional ``winner`` field; emits only the winner branch by default.
|
|
10
|
+
- Plain ``text`` strings (yielded as a single assistant message).
|
|
11
|
+
- Alpaca-style ``instruction`` / ``input`` / ``output`` (delegates to the
|
|
12
|
+
existing alpaca adapter).
|
|
13
|
+
|
|
14
|
+
Users can override the key lists and role map to teach it about bespoke schemas
|
|
15
|
+
without writing a new adapter from scratch.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from collections.abc import Iterator
|
|
21
|
+
from typing import Any
|
|
22
|
+
|
|
23
|
+
from convmerge.adapters.alpaca import iter_from_alpaca_line
|
|
24
|
+
from convmerge.models import ChatMessage, TrainingExample
|
|
25
|
+
|
|
26
|
+
# Default mapping from common ShareGPT-style ``from`` values onto standard roles.
|
|
27
|
+
DEFAULT_ROLE_MAP: dict[str, str] = {
|
|
28
|
+
"human": "user",
|
|
29
|
+
"user": "user",
|
|
30
|
+
"gpt": "assistant",
|
|
31
|
+
"assistant": "assistant",
|
|
32
|
+
"bing": "assistant",
|
|
33
|
+
"bot": "assistant",
|
|
34
|
+
"system": "system",
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
# Keys searched for the chat-list container, in priority order.
|
|
38
|
+
DEFAULT_CONVERSATION_KEYS: tuple[str, ...] = ("messages", "conversation", "conversations")
|
|
39
|
+
|
|
40
|
+
# Keys treated as role labels inside a chat-list entry.
|
|
41
|
+
DEFAULT_ROLE_KEYS: tuple[str, ...] = ("role", "from")
|
|
42
|
+
|
|
43
|
+
# Keys treated as message content inside a chat-list entry.
|
|
44
|
+
DEFAULT_CONTENT_KEYS: tuple[str, ...] = ("content", "value", "text")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def iter_from_chat_line(
|
|
48
|
+
record: dict[str, Any],
|
|
49
|
+
*,
|
|
50
|
+
conversation_keys: tuple[str, ...] = DEFAULT_CONVERSATION_KEYS,
|
|
51
|
+
role_keys: tuple[str, ...] = DEFAULT_ROLE_KEYS,
|
|
52
|
+
content_keys: tuple[str, ...] = DEFAULT_CONTENT_KEYS,
|
|
53
|
+
role_map: dict[str, str] | None = None,
|
|
54
|
+
pairwise_mode: str = "winner",
|
|
55
|
+
instruction_keys: tuple[str, ...] = ("instruction", "question", "prompt"),
|
|
56
|
+
output_keys: tuple[str, ...] = ("output", "response", "answer"),
|
|
57
|
+
input_keys: tuple[str, ...] = ("input", "context"),
|
|
58
|
+
) -> Iterator[TrainingExample]:
|
|
59
|
+
"""Yield zero or more :class:`TrainingExample` from a single raw record.
|
|
60
|
+
|
|
61
|
+
``pairwise_mode`` controls how ``conversation_a`` / ``conversation_b`` rows
|
|
62
|
+
are handled:
|
|
63
|
+
|
|
64
|
+
- ``"winner"`` (default): emit only the branch named by the ``winner`` field;
|
|
65
|
+
emit nothing when ``winner`` is absent or unrecognised.
|
|
66
|
+
- ``"both"``: emit both branches as independent examples.
|
|
67
|
+
- ``"a"`` / ``"b"``: always emit the chosen branch.
|
|
68
|
+
"""
|
|
69
|
+
role_map = role_map or DEFAULT_ROLE_MAP
|
|
70
|
+
|
|
71
|
+
if "conversation_a" in record and "conversation_b" in record:
|
|
72
|
+
yield from _iter_pairwise(
|
|
73
|
+
record,
|
|
74
|
+
role_keys=role_keys,
|
|
75
|
+
content_keys=content_keys,
|
|
76
|
+
role_map=role_map,
|
|
77
|
+
pairwise_mode=pairwise_mode,
|
|
78
|
+
)
|
|
79
|
+
return
|
|
80
|
+
|
|
81
|
+
for key in conversation_keys:
|
|
82
|
+
convs = record.get(key)
|
|
83
|
+
if isinstance(convs, list) and convs:
|
|
84
|
+
msgs = _coerce_messages(
|
|
85
|
+
convs, role_keys=role_keys, content_keys=content_keys, role_map=role_map
|
|
86
|
+
)
|
|
87
|
+
if msgs:
|
|
88
|
+
yield TrainingExample(messages=msgs, meta={"source": "chat"})
|
|
89
|
+
return
|
|
90
|
+
|
|
91
|
+
txt = record.get("text")
|
|
92
|
+
if isinstance(txt, str) and txt.strip():
|
|
93
|
+
yield TrainingExample(
|
|
94
|
+
messages=[ChatMessage(role="assistant", content=txt.strip())],
|
|
95
|
+
meta={"source": "chat:text"},
|
|
96
|
+
)
|
|
97
|
+
return
|
|
98
|
+
|
|
99
|
+
# Fall back to the alpaca adapter, but let callers override the key priority.
|
|
100
|
+
remapped = _remap_for_alpaca(record, instruction_keys, input_keys, output_keys)
|
|
101
|
+
if remapped is not None:
|
|
102
|
+
yield from iter_from_alpaca_line(remapped)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _iter_pairwise(
|
|
106
|
+
record: dict[str, Any],
|
|
107
|
+
*,
|
|
108
|
+
role_keys: tuple[str, ...],
|
|
109
|
+
content_keys: tuple[str, ...],
|
|
110
|
+
role_map: dict[str, str],
|
|
111
|
+
pairwise_mode: str,
|
|
112
|
+
) -> Iterator[TrainingExample]:
|
|
113
|
+
a = record.get("conversation_a")
|
|
114
|
+
b = record.get("conversation_b")
|
|
115
|
+
winner = str(record.get("winner") or "").lower().strip()
|
|
116
|
+
|
|
117
|
+
branches: list[tuple[str, Any]] = []
|
|
118
|
+
if pairwise_mode == "both":
|
|
119
|
+
branches = [("a", a), ("b", b)]
|
|
120
|
+
elif pairwise_mode == "a":
|
|
121
|
+
branches = [("a", a)]
|
|
122
|
+
elif pairwise_mode == "b":
|
|
123
|
+
branches = [("b", b)]
|
|
124
|
+
elif pairwise_mode == "winner":
|
|
125
|
+
if winner in ("model_a", "a"):
|
|
126
|
+
branches = [("a", a)]
|
|
127
|
+
elif winner in ("model_b", "b"):
|
|
128
|
+
branches = [("b", b)]
|
|
129
|
+
# Tie / unknown: emit nothing.
|
|
130
|
+
else:
|
|
131
|
+
raise ValueError(
|
|
132
|
+
f"Unknown pairwise_mode {pairwise_mode!r}. Use 'winner', 'both', 'a', or 'b'."
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
for label, convs in branches:
|
|
136
|
+
if not isinstance(convs, list) or not convs:
|
|
137
|
+
continue
|
|
138
|
+
msgs = _coerce_messages(
|
|
139
|
+
convs, role_keys=role_keys, content_keys=content_keys, role_map=role_map
|
|
140
|
+
)
|
|
141
|
+
if msgs:
|
|
142
|
+
yield TrainingExample(
|
|
143
|
+
messages=msgs,
|
|
144
|
+
meta={"source": "chat:pairwise", "branch": label},
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _coerce_messages(
|
|
149
|
+
convs: list[Any],
|
|
150
|
+
*,
|
|
151
|
+
role_keys: tuple[str, ...],
|
|
152
|
+
content_keys: tuple[str, ...],
|
|
153
|
+
role_map: dict[str, str],
|
|
154
|
+
) -> list[ChatMessage]:
|
|
155
|
+
out: list[ChatMessage] = []
|
|
156
|
+
for item in convs:
|
|
157
|
+
if not isinstance(item, dict):
|
|
158
|
+
continue
|
|
159
|
+
role_raw: str | None = None
|
|
160
|
+
for rk in role_keys:
|
|
161
|
+
v = item.get(rk)
|
|
162
|
+
if isinstance(v, str) and v.strip():
|
|
163
|
+
role_raw = v.strip().lower()
|
|
164
|
+
break
|
|
165
|
+
if role_raw is None:
|
|
166
|
+
continue
|
|
167
|
+
role = role_map.get(role_raw, role_raw)
|
|
168
|
+
|
|
169
|
+
content: str | None = None
|
|
170
|
+
for ck in content_keys:
|
|
171
|
+
v = item.get(ck)
|
|
172
|
+
if isinstance(v, str):
|
|
173
|
+
content = v
|
|
174
|
+
break
|
|
175
|
+
if content is None:
|
|
176
|
+
continue
|
|
177
|
+
out.append(ChatMessage(role=role, content=content))
|
|
178
|
+
return out
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def _remap_for_alpaca(
|
|
182
|
+
record: dict[str, Any],
|
|
183
|
+
instruction_keys: tuple[str, ...],
|
|
184
|
+
input_keys: tuple[str, ...],
|
|
185
|
+
output_keys: tuple[str, ...],
|
|
186
|
+
) -> dict[str, Any] | None:
|
|
187
|
+
"""Pick the first matching key for each slot and return a standard alpaca row."""
|
|
188
|
+
instr = _first_string(record, instruction_keys)
|
|
189
|
+
out = _first_string(record, output_keys)
|
|
190
|
+
if instr is None and out is None:
|
|
191
|
+
return None
|
|
192
|
+
inp = _first_string(record, input_keys) or ""
|
|
193
|
+
return {
|
|
194
|
+
"instruction": instr or "",
|
|
195
|
+
"input": inp,
|
|
196
|
+
"output": out or "",
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def _first_string(record: dict[str, Any], keys: tuple[str, ...]) -> str | None:
|
|
201
|
+
for k in keys:
|
|
202
|
+
v = record.get(k)
|
|
203
|
+
if isinstance(v, str) and v.strip():
|
|
204
|
+
return v
|
|
205
|
+
return None
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""ShareGPT-style conversations (from/value) → TrainingExample."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Iterator
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from convmerge.models import ChatMessage, TrainingExample
|
|
9
|
+
|
|
10
|
+
# Common ShareGPT role labels
|
|
11
|
+
_FROM_TO_ROLE: dict[str, str] = {
|
|
12
|
+
"human": "user",
|
|
13
|
+
"user": "user",
|
|
14
|
+
"gpt": "assistant",
|
|
15
|
+
"assistant": "assistant",
|
|
16
|
+
"system": "system",
|
|
17
|
+
"bing": "assistant",
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _normalize_role(from_key: str) -> str:
|
|
22
|
+
return _FROM_TO_ROLE.get(from_key.lower().strip(), from_key.lower().strip())
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def iter_from_sharegpt_line(record: dict[str, Any]) -> Iterator[TrainingExample]:
|
|
26
|
+
"""
|
|
27
|
+
One JSON object with ``conversations``: list of ``{"from": ..., "value": ...}``.
|
|
28
|
+
|
|
29
|
+
Emits one TrainingExample per consecutive human→assistant pair (typical SFT).
|
|
30
|
+
"""
|
|
31
|
+
convs = record.get("conversations")
|
|
32
|
+
if not isinstance(convs, list) or len(convs) < 2:
|
|
33
|
+
return
|
|
34
|
+
|
|
35
|
+
i = 0
|
|
36
|
+
while i + 1 < len(convs):
|
|
37
|
+
a, b = convs[i], convs[i + 1]
|
|
38
|
+
if not isinstance(a, dict) or not isinstance(b, dict):
|
|
39
|
+
i += 1
|
|
40
|
+
continue
|
|
41
|
+
ra = _normalize_role(str(a.get("from", "")))
|
|
42
|
+
rb = _normalize_role(str(b.get("from", "")))
|
|
43
|
+
va = (a.get("value") or "").strip()
|
|
44
|
+
vb = (b.get("value") or "").strip()
|
|
45
|
+
if ra == "user" and rb == "assistant":
|
|
46
|
+
yield TrainingExample(
|
|
47
|
+
messages=[
|
|
48
|
+
ChatMessage(role="user", content=va),
|
|
49
|
+
ChatMessage(role="assistant", content=vb),
|
|
50
|
+
],
|
|
51
|
+
meta={"source": "sharegpt"},
|
|
52
|
+
)
|
|
53
|
+
i += 2
|
|
54
|
+
else:
|
|
55
|
+
i += 1
|