smoltrace 0.0.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
smoltrace/__init__.py ADDED
@@ -0,0 +1,18 @@
1
+ """
2
+ SMOLTRACE - Comprehensive benchmarking and evaluation framework for smolagents.
3
+ """
4
+
5
+ __version__ = "0.0.1"
6
+
7
+ # Export main functions
8
+ from .core import run_evaluation
9
+ from .utils import (cleanup_datasets, discover_smoltrace_datasets, filter_runs,
10
+ group_datasets_by_run)
11
+
12
+ __all__ = [
13
+ "run_evaluation",
14
+ "cleanup_datasets",
15
+ "discover_smoltrace_datasets",
16
+ "group_datasets_by_run",
17
+ "filter_runs",
18
+ ]
smoltrace/cleanup.py ADDED
@@ -0,0 +1,219 @@
1
+ # smoltrace/cleanup.py
2
+ """
3
+ CLI command for cleaning up SMOLTRACE datasets from HuggingFace Hub.
4
+
5
+ Usage:
6
+ smoltrace-cleanup --dry-run
7
+ smoltrace-cleanup --older-than 7 --no-dry-run
8
+ smoltrace-cleanup --keep-recent 5 --no-dry-run
9
+ smoltrace-cleanup --incomplete-only --no-dry-run
10
+ """
11
+
12
+ import argparse
13
+ import os
14
+ import sys
15
+
16
+ from .utils import cleanup_datasets
17
+
18
+
19
+ def parse_older_than(value: str) -> int:
20
+ """
21
+ Parse --older-than argument.
22
+
23
+ Supports formats:
24
+ - "7d" or "7" → 7 days
25
+ - "30d" → 30 days
26
+ - "1w" → 7 days
27
+ - "1m" → 30 days
28
+
29
+ Args:
30
+ value: String value to parse
31
+
32
+ Returns:
33
+ Number of days
34
+ """
35
+ value = value.strip().lower()
36
+
37
+ # Handle "Nd" format
38
+ if value.endswith("d"):
39
+ return int(value[:-1])
40
+
41
+ # Handle "Nw" format (weeks)
42
+ if value.endswith("w"):
43
+ return int(value[:-1]) * 7
44
+
45
+ # Handle "Nm" format (months - approximate as 30 days)
46
+ if value.endswith("m"):
47
+ return int(value[:-1]) * 30
48
+
49
+ # Handle just a number (assume days)
50
+ try:
51
+ return int(value)
52
+ except ValueError:
53
+ raise ValueError(f"Invalid --older-than format: {value}. Use format like: 7d, 30d, 1w, 1m")
54
+
55
+
56
+ def main():
57
+ """Main entry point for smoltrace-cleanup CLI command."""
58
+ parser = argparse.ArgumentParser(
59
+ prog="smoltrace-cleanup",
60
+ description="Cleanup SMOLTRACE datasets from HuggingFace Hub",
61
+ formatter_class=argparse.RawDescriptionHelpFormatter,
62
+ epilog="""
63
+ Examples:
64
+ # Dry-run: Show what would be deleted (safe, default)
65
+ smoltrace-cleanup
66
+
67
+ # Delete datasets older than 7 days
68
+ smoltrace-cleanup --older-than 7d --no-dry-run
69
+
70
+ # Keep only 5 most recent evaluations
71
+ smoltrace-cleanup --keep-recent 5 --no-dry-run
72
+
73
+ # Delete incomplete runs (missing traces or metrics)
74
+ smoltrace-cleanup --incomplete-only --no-dry-run
75
+
76
+ # Delete only results datasets, keep traces and metrics
77
+ smoltrace-cleanup --only results --older-than 30d --no-dry-run
78
+
79
+ # Batch mode (no confirmation, for automation)
80
+ smoltrace-cleanup --older-than 7d --no-dry-run --yes
81
+
82
+ For more information, see: https://github.com/Mandark-droid/SMOLTRACE#dataset-cleanup
83
+ """,
84
+ )
85
+
86
+ # Filtering options
87
+ filter_group = parser.add_mutually_exclusive_group()
88
+ filter_group.add_argument(
89
+ "--older-than", type=str, help="Delete datasets older than N days (e.g., 7d, 30d, 1w, 1m)"
90
+ )
91
+ filter_group.add_argument(
92
+ "--keep-recent",
93
+ type=int,
94
+ metavar="N",
95
+ help="Keep only N most recent evaluations, delete the rest",
96
+ )
97
+ filter_group.add_argument(
98
+ "--incomplete-only",
99
+ action="store_true",
100
+ help="Delete only incomplete runs (missing traces or metrics)",
101
+ )
102
+ filter_group.add_argument(
103
+ "--all",
104
+ action="store_true",
105
+ help="Delete ALL SMOLTRACE datasets (use with extreme caution!)",
106
+ )
107
+
108
+ # Dataset type selection
109
+ parser.add_argument(
110
+ "--only",
111
+ choices=["results", "traces", "metrics"],
112
+ help="Delete only specific dataset type (default: all)",
113
+ )
114
+
115
+ # Safety options
116
+ parser.add_argument(
117
+ "--dry-run",
118
+ action="store_true",
119
+ default=False,
120
+ help="Show what would be deleted without actually deleting (default: enabled unless --no-dry-run)",
121
+ )
122
+ parser.add_argument(
123
+ "--no-dry-run",
124
+ action="store_true",
125
+ help="Actually delete datasets (required for real deletion)",
126
+ )
127
+ parser.add_argument(
128
+ "--yes", "-y", action="store_true", help="Skip confirmation prompts (use with caution!)"
129
+ )
130
+ parser.add_argument(
131
+ "--preserve-leaderboard",
132
+ action="store_true",
133
+ default=True,
134
+ help="Preserve leaderboard dataset (default: enabled)",
135
+ )
136
+ parser.add_argument(
137
+ "--delete-leaderboard",
138
+ action="store_true",
139
+ help="Also delete leaderboard dataset (use with extreme caution!)",
140
+ )
141
+
142
+ # Other options
143
+ parser.add_argument("--token", help="HuggingFace token (or set HF_TOKEN environment variable)")
144
+
145
+ args = parser.parse_args()
146
+
147
+ # Determine dry-run mode
148
+ # Default is dry-run=True unless --no-dry-run is specified
149
+ if args.no_dry_run:
150
+ dry_run = False
151
+ else:
152
+ dry_run = True
153
+
154
+ # Parse older_than if provided
155
+ older_than_days = None
156
+ if args.older_than:
157
+ try:
158
+ older_than_days = parse_older_than(args.older_than)
159
+ except ValueError as e:
160
+ print(f"Error: {e}")
161
+ sys.exit(1)
162
+
163
+ # Check if at least one filter is provided (unless --all)
164
+ if not any([args.older_than, args.keep_recent, args.incomplete_only, args.all]):
165
+ print("Error: Please specify a filter option:")
166
+ print(" --older-than DAYS")
167
+ print(" --keep-recent N")
168
+ print(" --incomplete-only")
169
+ print(" --all")
170
+ print("\nRun 'smoltrace-cleanup --help' for more information.")
171
+ sys.exit(1)
172
+
173
+ # Warn if --all is used
174
+ if args.all and not dry_run:
175
+ print("\n⚠️ WARNING: --all will delete ALL SMOLTRACE datasets!")
176
+ print("This includes all results, traces, and metrics from all evaluation runs.")
177
+ if not args.yes:
178
+ response = input("Are you absolutely sure? Type 'YES DELETE ALL' to confirm: ")
179
+ if response != "YES DELETE ALL":
180
+ print("\n[CANCELLED] No datasets were deleted.")
181
+ sys.exit(0)
182
+
183
+ # Get token
184
+ token = args.token or os.getenv("HF_TOKEN")
185
+ if not token:
186
+ print("Error: HuggingFace token required.")
187
+ print("Either set HF_TOKEN environment variable or use --token argument.")
188
+ sys.exit(1)
189
+
190
+ # Execute cleanup
191
+ try:
192
+ result = cleanup_datasets(
193
+ older_than_days=older_than_days,
194
+ keep_recent=args.keep_recent,
195
+ incomplete_only=args.incomplete_only,
196
+ delete_all=args.all,
197
+ only=args.only,
198
+ dry_run=dry_run,
199
+ confirm=not args.yes, # Skip confirmation if --yes
200
+ preserve_leaderboard=not args.delete_leaderboard,
201
+ hf_token=token,
202
+ )
203
+
204
+ # Exit code based on result
205
+ if result["failed"]:
206
+ sys.exit(1) # Some deletions failed
207
+ else:
208
+ sys.exit(0) # Success
209
+
210
+ except Exception as e:
211
+ print(f"\nError: {e}")
212
+ import traceback
213
+
214
+ traceback.print_exc()
215
+ sys.exit(1)
216
+
217
+
218
+ if __name__ == "__main__":
219
+ main()
smoltrace/cli.py ADDED
@@ -0,0 +1,98 @@
1
+ # smoltrace/cli.py
2
+ """CLI for running smoltrace evaluations."""
3
+
4
+ import argparse
5
+
6
+ from dotenv import load_dotenv
7
+
8
+ from .main import run_evaluation_flow
9
+
10
+ # Load .env file at startup
11
+ load_dotenv()
12
+
13
+
14
+ def main():
15
+ """Main entry point for the smoltrace CLI."""
16
+ parser = argparse.ArgumentParser(
17
+ description="Run agent evaluations with enhanced dataset management"
18
+ )
19
+
20
+ # Core arguments
21
+ parser.add_argument("--model", type=str, required=True, help="Model ID")
22
+ parser.add_argument(
23
+ "--provider",
24
+ type=str,
25
+ choices=["litellm", "transformers", "ollama"],
26
+ default="litellm",
27
+ help="Model provider: litellm (API models), transformers (HF GPU models), ollama (local)",
28
+ )
29
+ parser.add_argument(
30
+ "--hf-token",
31
+ type=str,
32
+ help="HuggingFace token (can also be set with HF_TOKEN env var)",
33
+ )
34
+
35
+ # Agent configuration
36
+ parser.add_argument(
37
+ "--agent-type",
38
+ type=str,
39
+ choices=["tool", "code", "both"],
40
+ default="both",
41
+ help="Type of agent to evaluate",
42
+ )
43
+ parser.add_argument("--prompt-yml", type=str, help="Path to prompt configuration YAML file")
44
+ parser.add_argument("--mcp-server-url", type=str, help="MCP server URL for MCP tools")
45
+
46
+ # Test configuration
47
+ parser.add_argument(
48
+ "--difficulty",
49
+ type=str,
50
+ choices=["easy", "medium", "hard"],
51
+ help="Filter tests by difficulty",
52
+ )
53
+ parser.add_argument(
54
+ "--dataset-name",
55
+ type=str,
56
+ default="kshitijthakkar/smoalagent-tasks",
57
+ help="HF dataset for tasks",
58
+ )
59
+ parser.add_argument("--split", type=str, default="train", help="Dataset split to use")
60
+
61
+ # Options
62
+ parser.add_argument("--private", action="store_true", help="Make result datasets private")
63
+ parser.add_argument("--enable-otel", action="store_true", help="Enable OTEL tracing")
64
+ parser.add_argument(
65
+ "--disable-gpu-metrics",
66
+ action="store_true",
67
+ help="Disable GPU metrics collection (enabled by default for local models: transformers, ollama)",
68
+ )
69
+ parser.add_argument(
70
+ "--run-id",
71
+ type=str,
72
+ default=None,
73
+ help="Optional unique run identifier (UUID format). Generated automatically if not provided. Use this to filter results in the leaderboard.",
74
+ )
75
+ parser.add_argument(
76
+ "--output-format",
77
+ type=str,
78
+ choices=["hub", "json"],
79
+ default="hub",
80
+ help="Output format: 'hub' (push to HuggingFace) or 'json' (save locally)",
81
+ )
82
+ parser.add_argument(
83
+ "--output-dir",
84
+ type=str,
85
+ default="./smoltrace_results",
86
+ help="Directory for local JSON output (when --output-format=json)",
87
+ )
88
+ parser.add_argument("--quiet", action="store_true", help="Reduce output verbosity")
89
+ parser.add_argument("--debug", action="store_true", help="Enable debug output")
90
+
91
+ args = parser.parse_args()
92
+
93
+ # Run evaluation
94
+ run_evaluation_flow(args)
95
+
96
+
97
+ if __name__ == "__main__": # pragma: no cover
98
+ main()