smoltrace 0.0.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- smoltrace/__init__.py +18 -0
- smoltrace/cleanup.py +219 -0
- smoltrace/cli.py +98 -0
- smoltrace/core.py +706 -0
- smoltrace/main.py +138 -0
- smoltrace/otel.py +649 -0
- smoltrace/tools.py +84 -0
- smoltrace/utils.py +1173 -0
- smoltrace-0.0.2.dist-info/METADATA +741 -0
- smoltrace-0.0.2.dist-info/RECORD +14 -0
- smoltrace-0.0.2.dist-info/WHEEL +5 -0
- smoltrace-0.0.2.dist-info/entry_points.txt +3 -0
- smoltrace-0.0.2.dist-info/licenses/LICENSE +201 -0
- smoltrace-0.0.2.dist-info/top_level.txt +1 -0
smoltrace/__init__.py
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""
|
|
2
|
+
SMOLTRACE - Comprehensive benchmarking and evaluation framework for smolagents.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
__version__ = "0.0.1"
|
|
6
|
+
|
|
7
|
+
# Export main functions
|
|
8
|
+
from .core import run_evaluation
|
|
9
|
+
from .utils import (cleanup_datasets, discover_smoltrace_datasets, filter_runs,
|
|
10
|
+
group_datasets_by_run)
|
|
11
|
+
|
|
12
|
+
__all__ = [
|
|
13
|
+
"run_evaluation",
|
|
14
|
+
"cleanup_datasets",
|
|
15
|
+
"discover_smoltrace_datasets",
|
|
16
|
+
"group_datasets_by_run",
|
|
17
|
+
"filter_runs",
|
|
18
|
+
]
|
smoltrace/cleanup.py
ADDED
|
@@ -0,0 +1,219 @@
|
|
|
1
|
+
# smoltrace/cleanup.py
|
|
2
|
+
"""
|
|
3
|
+
CLI command for cleaning up SMOLTRACE datasets from HuggingFace Hub.
|
|
4
|
+
|
|
5
|
+
Usage:
|
|
6
|
+
smoltrace-cleanup --dry-run
|
|
7
|
+
smoltrace-cleanup --older-than 7 --no-dry-run
|
|
8
|
+
smoltrace-cleanup --keep-recent 5 --no-dry-run
|
|
9
|
+
smoltrace-cleanup --incomplete-only --no-dry-run
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
import argparse
|
|
13
|
+
import os
|
|
14
|
+
import sys
|
|
15
|
+
|
|
16
|
+
from .utils import cleanup_datasets
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def parse_older_than(value: str) -> int:
|
|
20
|
+
"""
|
|
21
|
+
Parse --older-than argument.
|
|
22
|
+
|
|
23
|
+
Supports formats:
|
|
24
|
+
- "7d" or "7" → 7 days
|
|
25
|
+
- "30d" → 30 days
|
|
26
|
+
- "1w" → 7 days
|
|
27
|
+
- "1m" → 30 days
|
|
28
|
+
|
|
29
|
+
Args:
|
|
30
|
+
value: String value to parse
|
|
31
|
+
|
|
32
|
+
Returns:
|
|
33
|
+
Number of days
|
|
34
|
+
"""
|
|
35
|
+
value = value.strip().lower()
|
|
36
|
+
|
|
37
|
+
# Handle "Nd" format
|
|
38
|
+
if value.endswith("d"):
|
|
39
|
+
return int(value[:-1])
|
|
40
|
+
|
|
41
|
+
# Handle "Nw" format (weeks)
|
|
42
|
+
if value.endswith("w"):
|
|
43
|
+
return int(value[:-1]) * 7
|
|
44
|
+
|
|
45
|
+
# Handle "Nm" format (months - approximate as 30 days)
|
|
46
|
+
if value.endswith("m"):
|
|
47
|
+
return int(value[:-1]) * 30
|
|
48
|
+
|
|
49
|
+
# Handle just a number (assume days)
|
|
50
|
+
try:
|
|
51
|
+
return int(value)
|
|
52
|
+
except ValueError:
|
|
53
|
+
raise ValueError(f"Invalid --older-than format: {value}. Use format like: 7d, 30d, 1w, 1m")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def main():
|
|
57
|
+
"""Main entry point for smoltrace-cleanup CLI command."""
|
|
58
|
+
parser = argparse.ArgumentParser(
|
|
59
|
+
prog="smoltrace-cleanup",
|
|
60
|
+
description="Cleanup SMOLTRACE datasets from HuggingFace Hub",
|
|
61
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
62
|
+
epilog="""
|
|
63
|
+
Examples:
|
|
64
|
+
# Dry-run: Show what would be deleted (safe, default)
|
|
65
|
+
smoltrace-cleanup
|
|
66
|
+
|
|
67
|
+
# Delete datasets older than 7 days
|
|
68
|
+
smoltrace-cleanup --older-than 7d --no-dry-run
|
|
69
|
+
|
|
70
|
+
# Keep only 5 most recent evaluations
|
|
71
|
+
smoltrace-cleanup --keep-recent 5 --no-dry-run
|
|
72
|
+
|
|
73
|
+
# Delete incomplete runs (missing traces or metrics)
|
|
74
|
+
smoltrace-cleanup --incomplete-only --no-dry-run
|
|
75
|
+
|
|
76
|
+
# Delete only results datasets, keep traces and metrics
|
|
77
|
+
smoltrace-cleanup --only results --older-than 30d --no-dry-run
|
|
78
|
+
|
|
79
|
+
# Batch mode (no confirmation, for automation)
|
|
80
|
+
smoltrace-cleanup --older-than 7d --no-dry-run --yes
|
|
81
|
+
|
|
82
|
+
For more information, see: https://github.com/Mandark-droid/SMOLTRACE#dataset-cleanup
|
|
83
|
+
""",
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
# Filtering options
|
|
87
|
+
filter_group = parser.add_mutually_exclusive_group()
|
|
88
|
+
filter_group.add_argument(
|
|
89
|
+
"--older-than", type=str, help="Delete datasets older than N days (e.g., 7d, 30d, 1w, 1m)"
|
|
90
|
+
)
|
|
91
|
+
filter_group.add_argument(
|
|
92
|
+
"--keep-recent",
|
|
93
|
+
type=int,
|
|
94
|
+
metavar="N",
|
|
95
|
+
help="Keep only N most recent evaluations, delete the rest",
|
|
96
|
+
)
|
|
97
|
+
filter_group.add_argument(
|
|
98
|
+
"--incomplete-only",
|
|
99
|
+
action="store_true",
|
|
100
|
+
help="Delete only incomplete runs (missing traces or metrics)",
|
|
101
|
+
)
|
|
102
|
+
filter_group.add_argument(
|
|
103
|
+
"--all",
|
|
104
|
+
action="store_true",
|
|
105
|
+
help="Delete ALL SMOLTRACE datasets (use with extreme caution!)",
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
# Dataset type selection
|
|
109
|
+
parser.add_argument(
|
|
110
|
+
"--only",
|
|
111
|
+
choices=["results", "traces", "metrics"],
|
|
112
|
+
help="Delete only specific dataset type (default: all)",
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
# Safety options
|
|
116
|
+
parser.add_argument(
|
|
117
|
+
"--dry-run",
|
|
118
|
+
action="store_true",
|
|
119
|
+
default=False,
|
|
120
|
+
help="Show what would be deleted without actually deleting (default: enabled unless --no-dry-run)",
|
|
121
|
+
)
|
|
122
|
+
parser.add_argument(
|
|
123
|
+
"--no-dry-run",
|
|
124
|
+
action="store_true",
|
|
125
|
+
help="Actually delete datasets (required for real deletion)",
|
|
126
|
+
)
|
|
127
|
+
parser.add_argument(
|
|
128
|
+
"--yes", "-y", action="store_true", help="Skip confirmation prompts (use with caution!)"
|
|
129
|
+
)
|
|
130
|
+
parser.add_argument(
|
|
131
|
+
"--preserve-leaderboard",
|
|
132
|
+
action="store_true",
|
|
133
|
+
default=True,
|
|
134
|
+
help="Preserve leaderboard dataset (default: enabled)",
|
|
135
|
+
)
|
|
136
|
+
parser.add_argument(
|
|
137
|
+
"--delete-leaderboard",
|
|
138
|
+
action="store_true",
|
|
139
|
+
help="Also delete leaderboard dataset (use with extreme caution!)",
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
# Other options
|
|
143
|
+
parser.add_argument("--token", help="HuggingFace token (or set HF_TOKEN environment variable)")
|
|
144
|
+
|
|
145
|
+
args = parser.parse_args()
|
|
146
|
+
|
|
147
|
+
# Determine dry-run mode
|
|
148
|
+
# Default is dry-run=True unless --no-dry-run is specified
|
|
149
|
+
if args.no_dry_run:
|
|
150
|
+
dry_run = False
|
|
151
|
+
else:
|
|
152
|
+
dry_run = True
|
|
153
|
+
|
|
154
|
+
# Parse older_than if provided
|
|
155
|
+
older_than_days = None
|
|
156
|
+
if args.older_than:
|
|
157
|
+
try:
|
|
158
|
+
older_than_days = parse_older_than(args.older_than)
|
|
159
|
+
except ValueError as e:
|
|
160
|
+
print(f"Error: {e}")
|
|
161
|
+
sys.exit(1)
|
|
162
|
+
|
|
163
|
+
# Check if at least one filter is provided (unless --all)
|
|
164
|
+
if not any([args.older_than, args.keep_recent, args.incomplete_only, args.all]):
|
|
165
|
+
print("Error: Please specify a filter option:")
|
|
166
|
+
print(" --older-than DAYS")
|
|
167
|
+
print(" --keep-recent N")
|
|
168
|
+
print(" --incomplete-only")
|
|
169
|
+
print(" --all")
|
|
170
|
+
print("\nRun 'smoltrace-cleanup --help' for more information.")
|
|
171
|
+
sys.exit(1)
|
|
172
|
+
|
|
173
|
+
# Warn if --all is used
|
|
174
|
+
if args.all and not dry_run:
|
|
175
|
+
print("\n⚠️ WARNING: --all will delete ALL SMOLTRACE datasets!")
|
|
176
|
+
print("This includes all results, traces, and metrics from all evaluation runs.")
|
|
177
|
+
if not args.yes:
|
|
178
|
+
response = input("Are you absolutely sure? Type 'YES DELETE ALL' to confirm: ")
|
|
179
|
+
if response != "YES DELETE ALL":
|
|
180
|
+
print("\n[CANCELLED] No datasets were deleted.")
|
|
181
|
+
sys.exit(0)
|
|
182
|
+
|
|
183
|
+
# Get token
|
|
184
|
+
token = args.token or os.getenv("HF_TOKEN")
|
|
185
|
+
if not token:
|
|
186
|
+
print("Error: HuggingFace token required.")
|
|
187
|
+
print("Either set HF_TOKEN environment variable or use --token argument.")
|
|
188
|
+
sys.exit(1)
|
|
189
|
+
|
|
190
|
+
# Execute cleanup
|
|
191
|
+
try:
|
|
192
|
+
result = cleanup_datasets(
|
|
193
|
+
older_than_days=older_than_days,
|
|
194
|
+
keep_recent=args.keep_recent,
|
|
195
|
+
incomplete_only=args.incomplete_only,
|
|
196
|
+
delete_all=args.all,
|
|
197
|
+
only=args.only,
|
|
198
|
+
dry_run=dry_run,
|
|
199
|
+
confirm=not args.yes, # Skip confirmation if --yes
|
|
200
|
+
preserve_leaderboard=not args.delete_leaderboard,
|
|
201
|
+
hf_token=token,
|
|
202
|
+
)
|
|
203
|
+
|
|
204
|
+
# Exit code based on result
|
|
205
|
+
if result["failed"]:
|
|
206
|
+
sys.exit(1) # Some deletions failed
|
|
207
|
+
else:
|
|
208
|
+
sys.exit(0) # Success
|
|
209
|
+
|
|
210
|
+
except Exception as e:
|
|
211
|
+
print(f"\nError: {e}")
|
|
212
|
+
import traceback
|
|
213
|
+
|
|
214
|
+
traceback.print_exc()
|
|
215
|
+
sys.exit(1)
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
if __name__ == "__main__":
|
|
219
|
+
main()
|
smoltrace/cli.py
ADDED
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
# smoltrace/cli.py
|
|
2
|
+
"""CLI for running smoltrace evaluations."""
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
|
|
6
|
+
from dotenv import load_dotenv
|
|
7
|
+
|
|
8
|
+
from .main import run_evaluation_flow
|
|
9
|
+
|
|
10
|
+
# Load .env file at startup
|
|
11
|
+
load_dotenv()
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def main():
|
|
15
|
+
"""Main entry point for the smoltrace CLI."""
|
|
16
|
+
parser = argparse.ArgumentParser(
|
|
17
|
+
description="Run agent evaluations with enhanced dataset management"
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
# Core arguments
|
|
21
|
+
parser.add_argument("--model", type=str, required=True, help="Model ID")
|
|
22
|
+
parser.add_argument(
|
|
23
|
+
"--provider",
|
|
24
|
+
type=str,
|
|
25
|
+
choices=["litellm", "transformers", "ollama"],
|
|
26
|
+
default="litellm",
|
|
27
|
+
help="Model provider: litellm (API models), transformers (HF GPU models), ollama (local)",
|
|
28
|
+
)
|
|
29
|
+
parser.add_argument(
|
|
30
|
+
"--hf-token",
|
|
31
|
+
type=str,
|
|
32
|
+
help="HuggingFace token (can also be set with HF_TOKEN env var)",
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
# Agent configuration
|
|
36
|
+
parser.add_argument(
|
|
37
|
+
"--agent-type",
|
|
38
|
+
type=str,
|
|
39
|
+
choices=["tool", "code", "both"],
|
|
40
|
+
default="both",
|
|
41
|
+
help="Type of agent to evaluate",
|
|
42
|
+
)
|
|
43
|
+
parser.add_argument("--prompt-yml", type=str, help="Path to prompt configuration YAML file")
|
|
44
|
+
parser.add_argument("--mcp-server-url", type=str, help="MCP server URL for MCP tools")
|
|
45
|
+
|
|
46
|
+
# Test configuration
|
|
47
|
+
parser.add_argument(
|
|
48
|
+
"--difficulty",
|
|
49
|
+
type=str,
|
|
50
|
+
choices=["easy", "medium", "hard"],
|
|
51
|
+
help="Filter tests by difficulty",
|
|
52
|
+
)
|
|
53
|
+
parser.add_argument(
|
|
54
|
+
"--dataset-name",
|
|
55
|
+
type=str,
|
|
56
|
+
default="kshitijthakkar/smoalagent-tasks",
|
|
57
|
+
help="HF dataset for tasks",
|
|
58
|
+
)
|
|
59
|
+
parser.add_argument("--split", type=str, default="train", help="Dataset split to use")
|
|
60
|
+
|
|
61
|
+
# Options
|
|
62
|
+
parser.add_argument("--private", action="store_true", help="Make result datasets private")
|
|
63
|
+
parser.add_argument("--enable-otel", action="store_true", help="Enable OTEL tracing")
|
|
64
|
+
parser.add_argument(
|
|
65
|
+
"--disable-gpu-metrics",
|
|
66
|
+
action="store_true",
|
|
67
|
+
help="Disable GPU metrics collection (enabled by default for local models: transformers, ollama)",
|
|
68
|
+
)
|
|
69
|
+
parser.add_argument(
|
|
70
|
+
"--run-id",
|
|
71
|
+
type=str,
|
|
72
|
+
default=None,
|
|
73
|
+
help="Optional unique run identifier (UUID format). Generated automatically if not provided. Use this to filter results in the leaderboard.",
|
|
74
|
+
)
|
|
75
|
+
parser.add_argument(
|
|
76
|
+
"--output-format",
|
|
77
|
+
type=str,
|
|
78
|
+
choices=["hub", "json"],
|
|
79
|
+
default="hub",
|
|
80
|
+
help="Output format: 'hub' (push to HuggingFace) or 'json' (save locally)",
|
|
81
|
+
)
|
|
82
|
+
parser.add_argument(
|
|
83
|
+
"--output-dir",
|
|
84
|
+
type=str,
|
|
85
|
+
default="./smoltrace_results",
|
|
86
|
+
help="Directory for local JSON output (when --output-format=json)",
|
|
87
|
+
)
|
|
88
|
+
parser.add_argument("--quiet", action="store_true", help="Reduce output verbosity")
|
|
89
|
+
parser.add_argument("--debug", action="store_true", help="Enable debug output")
|
|
90
|
+
|
|
91
|
+
args = parser.parse_args()
|
|
92
|
+
|
|
93
|
+
# Run evaluation
|
|
94
|
+
run_evaluation_flow(args)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
if __name__ == "__main__": # pragma: no cover
|
|
98
|
+
main()
|