piqc 1.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- piqc/__init__.py +10 -0
- piqc/__main__.py +11 -0
- piqc/cli/__init__.py +1 -0
- piqc/cli/commands.py +566 -0
- piqc/collectors/__init__.py +1 -0
- piqc/collectors/amd/__init__.py +14 -0
- piqc/collectors/config_collector.py +238 -0
- piqc/collectors/gpu_collector.py +313 -0
- piqc/collectors/vllm_api_client.py +674 -0
- piqc/collectors/vllm_collector.py +328 -0
- piqc/core/__init__.py +1 -0
- piqc/core/aggregator.py +302 -0
- piqc/core/confidence.py +225 -0
- piqc/core/discovery.py +563 -0
- piqc/core/k8s_client.py +604 -0
- piqc/core/llm_d/__init__.py +16 -0
- piqc/core/modes.py +153 -0
- piqc/core/orchestrator.py +829 -0
- piqc/generators/__init__.py +1 -0
- piqc/generators/json_generator.py +139 -0
- piqc/generators/piqc_generator.py +1010 -0
- piqc/generators/table_generator.py +797 -0
- piqc/generators/yaml_generator.py +183 -0
- piqc/models/__init__.py +31 -0
- piqc/models/modelspec.py +419 -0
- piqc/models/piqc_schema.py +246 -0
- piqc/parsers/__init__.py +1 -0
- piqc/parsers/vllm_parser.py +227 -0
- piqc/telemetry.py +156 -0
- piqc/utils/__init__.py +1 -0
- piqc/utils/exceptions.py +223 -0
- piqc/utils/logger.py +183 -0
- piqc-1.2.0.dist-info/METADATA +618 -0
- piqc-1.2.0.dist-info/RECORD +37 -0
- piqc-1.2.0.dist-info/WHEEL +4 -0
- piqc-1.2.0.dist-info/entry_points.txt +3 -0
- piqc-1.2.0.dist-info/licenses/LICENSE +25 -0
piqc/__init__.py
ADDED
piqc/__main__.py
ADDED
piqc/cli/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""CLI module for piqc."""
|
piqc/cli/commands.py
ADDED
|
@@ -0,0 +1,566 @@
|
|
|
1
|
+
"""
|
|
2
|
+
CLI commands for piqc.
|
|
3
|
+
|
|
4
|
+
Provides the command-line interface for scanning Kubernetes clusters
|
|
5
|
+
and collecting facts about vLLM inference deployments.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import sys
|
|
9
|
+
from typing import Optional
|
|
10
|
+
|
|
11
|
+
import click
|
|
12
|
+
from rich.console import Console
|
|
13
|
+
|
|
14
|
+
from piqc import __version__
|
|
15
|
+
from piqc.core.k8s_client import K8sClient
|
|
16
|
+
from piqc.core.orchestrator import ScanOrchestrator
|
|
17
|
+
from piqc.generators.json_generator import JSONGenerator
|
|
18
|
+
from piqc.generators.piqc_generator import PIQCGenerator
|
|
19
|
+
from piqc.generators.table_generator import TableGenerator, format_duration
|
|
20
|
+
from piqc.generators.yaml_generator import YAMLGenerator
|
|
21
|
+
from piqc.utils.exceptions import (
|
|
22
|
+
KubernetesConnectionError,
|
|
23
|
+
RBACPermissionError,
|
|
24
|
+
)
|
|
25
|
+
from piqc.utils.logger import setup_logging, get_logger
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
# Use a wide fixed width when running without a TTY (e.g. kubectl logs, CI)
|
|
29
|
+
console = Console() if sys.stdout.isatty() else Console(width=220, force_terminal=True)
|
|
30
|
+
logger = get_logger(__name__)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def print_header() -> None:
|
|
34
|
+
"""Print application header."""
|
|
35
|
+
console.print(f"piqc v{__version__}")
|
|
36
|
+
console.print("=" * 40)
|
|
37
|
+
console.print()
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def print_info(message: str) -> None:
|
|
41
|
+
"""Print an info message."""
|
|
42
|
+
console.print(f"[INFO] {message}")
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def print_warning(message: str) -> None:
|
|
46
|
+
"""Print a warning message."""
|
|
47
|
+
console.print(f"[WARN] {message}", style="yellow")
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def print_error(message: str) -> None:
|
|
51
|
+
"""Print an error message."""
|
|
52
|
+
console.print(f"[ERROR] {message}", style="red")
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _push_bundle(
|
|
56
|
+
piqc_file: str,
|
|
57
|
+
push_url: str,
|
|
58
|
+
cluster_id: str,
|
|
59
|
+
console: Console,
|
|
60
|
+
debug: bool = False,
|
|
61
|
+
) -> None:
|
|
62
|
+
"""Read the generated piqc-facts.json and POST it to the platform ingest endpoint."""
|
|
63
|
+
import json
|
|
64
|
+
import requests as _requests
|
|
65
|
+
|
|
66
|
+
try:
|
|
67
|
+
with open(piqc_file, "r", encoding="utf-8") as f:
|
|
68
|
+
bundle = json.load(f)
|
|
69
|
+
except Exception as exc:
|
|
70
|
+
console.print(f"[red][ERROR] Could not read piqc bundle for push: {exc}[/red]")
|
|
71
|
+
return
|
|
72
|
+
|
|
73
|
+
# Translate PIQCBundle format (uses 'objects') to ingest API format (uses 'workloads').
|
|
74
|
+
workloads = []
|
|
75
|
+
for obj in bundle.get("objects", []):
|
|
76
|
+
workloads.append({
|
|
77
|
+
"workloadId": obj.get("workloadId", ""),
|
|
78
|
+
"facts": obj.get("facts", {}),
|
|
79
|
+
})
|
|
80
|
+
|
|
81
|
+
payload = {
|
|
82
|
+
"schemaVersion": bundle.get("schemaVersion", "piqc-scan.v0.1"),
|
|
83
|
+
"cluster_id": cluster_id,
|
|
84
|
+
"workloads": workloads,
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
ingest_url = push_url.rstrip("/") + "/v1/ingest"
|
|
88
|
+
console.print()
|
|
89
|
+
console.print(f"[INFO] Pushing {len(workloads)} workload(s) to {ingest_url} ...")
|
|
90
|
+
|
|
91
|
+
try:
|
|
92
|
+
resp = _requests.post(ingest_url, json=payload, timeout=30)
|
|
93
|
+
resp.raise_for_status()
|
|
94
|
+
results = resp.json()
|
|
95
|
+
total_recs = sum(len(r.get("recommendations", [])) for r in results)
|
|
96
|
+
console.print(f"[green] Push succeeded — {total_recs} recommendation(s) generated[/green]")
|
|
97
|
+
except _requests.HTTPError as exc:
|
|
98
|
+
console.print(f"[red][ERROR] Platform returned {exc.response.status_code}: {exc.response.text[:200]}[/red]")
|
|
99
|
+
if debug:
|
|
100
|
+
console.print_exception()
|
|
101
|
+
except _requests.RequestException as exc:
|
|
102
|
+
console.print(f"[red][ERROR] Could not reach platform at {ingest_url}: {exc}[/red]")
|
|
103
|
+
if debug:
|
|
104
|
+
console.print_exception()
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
@click.group()
|
|
108
|
+
@click.version_option(version=__version__, prog_name="piqc")
|
|
109
|
+
def main() -> None:
|
|
110
|
+
"""
|
|
111
|
+
piqc - vLLM-native fact collector for Kubernetes inference fleets.
|
|
112
|
+
|
|
113
|
+
Discovers vLLM inference deployments in Kubernetes clusters and
|
|
114
|
+
collects facts on GPU waste, idle capacity, and tier misplacement.
|
|
115
|
+
"""
|
|
116
|
+
pass
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
@main.command()
|
|
120
|
+
@click.option(
|
|
121
|
+
"--kubeconfig",
|
|
122
|
+
type=click.Path(exists=True),
|
|
123
|
+
default=None,
|
|
124
|
+
help="Path to kubeconfig file. Default: ~/.kube/config",
|
|
125
|
+
)
|
|
126
|
+
@click.option(
|
|
127
|
+
"--context",
|
|
128
|
+
type=str,
|
|
129
|
+
default=None,
|
|
130
|
+
help="Kubernetes context to use. Default: current context",
|
|
131
|
+
)
|
|
132
|
+
@click.option(
|
|
133
|
+
"--namespace",
|
|
134
|
+
"-n",
|
|
135
|
+
type=str,
|
|
136
|
+
default=None,
|
|
137
|
+
help="Specific namespace to scan. Default: all namespaces",
|
|
138
|
+
)
|
|
139
|
+
@click.option(
|
|
140
|
+
"--format",
|
|
141
|
+
"output_format",
|
|
142
|
+
type=click.Choice(["yaml", "json", "table"]),
|
|
143
|
+
default="table",
|
|
144
|
+
help="Output format. Default: table",
|
|
145
|
+
)
|
|
146
|
+
@click.option(
|
|
147
|
+
"--output",
|
|
148
|
+
"-o",
|
|
149
|
+
type=click.Path(),
|
|
150
|
+
default="./output",
|
|
151
|
+
help="Output directory for generated files. Default: ./output",
|
|
152
|
+
)
|
|
153
|
+
@click.option(
|
|
154
|
+
"--timeout",
|
|
155
|
+
type=int,
|
|
156
|
+
default=30,
|
|
157
|
+
help="Operation timeout in seconds. Default: 30",
|
|
158
|
+
)
|
|
159
|
+
@click.option(
|
|
160
|
+
"--no-exec",
|
|
161
|
+
is_flag=True,
|
|
162
|
+
default=False,
|
|
163
|
+
help="Disable pod exec (skip GPU metrics collection)",
|
|
164
|
+
)
|
|
165
|
+
@click.option(
|
|
166
|
+
"--no-logs",
|
|
167
|
+
is_flag=True,
|
|
168
|
+
default=False,
|
|
169
|
+
help="Disable log reading",
|
|
170
|
+
)
|
|
171
|
+
@click.option(
|
|
172
|
+
"--workers",
|
|
173
|
+
type=int,
|
|
174
|
+
default=10,
|
|
175
|
+
help="Number of parallel workers. Default: 10",
|
|
176
|
+
)
|
|
177
|
+
@click.option(
|
|
178
|
+
"--verbose",
|
|
179
|
+
"-v",
|
|
180
|
+
is_flag=True,
|
|
181
|
+
default=False,
|
|
182
|
+
help="Enable verbose output",
|
|
183
|
+
)
|
|
184
|
+
@click.option(
|
|
185
|
+
"--debug",
|
|
186
|
+
is_flag=True,
|
|
187
|
+
default=False,
|
|
188
|
+
help="Enable debug mode with detailed trace",
|
|
189
|
+
)
|
|
190
|
+
@click.option(
|
|
191
|
+
"--combined",
|
|
192
|
+
is_flag=True,
|
|
193
|
+
default=False,
|
|
194
|
+
help="Generate a single combined output file instead of per-deployment files",
|
|
195
|
+
)
|
|
196
|
+
@click.option(
|
|
197
|
+
"--collect-runtime/--no-collect-runtime",
|
|
198
|
+
default=True,
|
|
199
|
+
help="Collect runtime metrics via vLLM API. Default: enabled",
|
|
200
|
+
)
|
|
201
|
+
@click.option(
|
|
202
|
+
"--aggregate/--no-aggregate",
|
|
203
|
+
default=True,
|
|
204
|
+
help="Aggregate metrics across pod replicas. Default: aggregate",
|
|
205
|
+
)
|
|
206
|
+
@click.option(
|
|
207
|
+
"--mode",
|
|
208
|
+
type=click.Choice(["auto", "remote", "incluster", "dry-run"]),
|
|
209
|
+
default="auto",
|
|
210
|
+
help="Execution mode. Default: auto-detect",
|
|
211
|
+
)
|
|
212
|
+
@click.option(
|
|
213
|
+
"--output-piqc",
|
|
214
|
+
is_flag=True,
|
|
215
|
+
default=False,
|
|
216
|
+
help="Generate piqc-facts.json output (PIQC scan v0.1 schema)",
|
|
217
|
+
)
|
|
218
|
+
@click.option(
|
|
219
|
+
"--gpu-cost",
|
|
220
|
+
type=float,
|
|
221
|
+
default=None,
|
|
222
|
+
metavar="DOLLARS",
|
|
223
|
+
help="Override GPU cost in $/GPU/hr (e.g. 3.50). Default: auto-detect by GPU type.",
|
|
224
|
+
)
|
|
225
|
+
@click.option(
|
|
226
|
+
"--node-cost",
|
|
227
|
+
type=float,
|
|
228
|
+
default=None,
|
|
229
|
+
metavar="DOLLARS",
|
|
230
|
+
help="Override cost in $/GPU/hr for unallocated nodes. Defaults to --gpu-cost if set.",
|
|
231
|
+
)
|
|
232
|
+
@click.option(
|
|
233
|
+
"--contribute-benchmarks",
|
|
234
|
+
is_flag=True,
|
|
235
|
+
default=False,
|
|
236
|
+
help="Contribute anonymized GPU/model performance data to the ParallelIQ benchmark dataset.",
|
|
237
|
+
)
|
|
238
|
+
@click.option(
|
|
239
|
+
"--push-url",
|
|
240
|
+
type=str,
|
|
241
|
+
default=None,
|
|
242
|
+
metavar="URL",
|
|
243
|
+
help="Push piqc facts to a ParallelIQ platform API gateway (e.g. http://localhost:8000). Implies --output-piqc.",
|
|
244
|
+
)
|
|
245
|
+
@click.option(
|
|
246
|
+
"--cluster-id",
|
|
247
|
+
type=str,
|
|
248
|
+
default=None,
|
|
249
|
+
metavar="ID",
|
|
250
|
+
help="Cluster identifier sent with --push-url. Defaults to the Kubernetes cluster name.",
|
|
251
|
+
)
|
|
252
|
+
def scan(
|
|
253
|
+
kubeconfig: Optional[str],
|
|
254
|
+
context: Optional[str],
|
|
255
|
+
namespace: Optional[str],
|
|
256
|
+
output_format: str,
|
|
257
|
+
output: str,
|
|
258
|
+
timeout: int,
|
|
259
|
+
no_exec: bool,
|
|
260
|
+
no_logs: bool,
|
|
261
|
+
workers: int,
|
|
262
|
+
verbose: bool,
|
|
263
|
+
debug: bool,
|
|
264
|
+
combined: bool,
|
|
265
|
+
collect_runtime: bool,
|
|
266
|
+
aggregate: bool,
|
|
267
|
+
mode: str,
|
|
268
|
+
output_piqc: bool,
|
|
269
|
+
gpu_cost: Optional[float],
|
|
270
|
+
node_cost: Optional[float],
|
|
271
|
+
contribute_benchmarks: bool,
|
|
272
|
+
push_url: Optional[str],
|
|
273
|
+
cluster_id: Optional[str],
|
|
274
|
+
) -> None:
|
|
275
|
+
"""
|
|
276
|
+
Scan Kubernetes cluster for vLLM model deployments.
|
|
277
|
+
|
|
278
|
+
Discovers vLLM inference workloads and collects facts about them,
|
|
279
|
+
including hardware, configuration, and optional runtime metrics.
|
|
280
|
+
|
|
281
|
+
\b
|
|
282
|
+
Examples:
|
|
283
|
+
# Scan entire cluster
|
|
284
|
+
piqc scan
|
|
285
|
+
|
|
286
|
+
# Scan specific namespace
|
|
287
|
+
piqc scan -n production
|
|
288
|
+
|
|
289
|
+
# Collect runtime metrics via vLLM API
|
|
290
|
+
piqc scan --collect-runtime
|
|
291
|
+
|
|
292
|
+
# JSON output without aggregation
|
|
293
|
+
piqc scan --format json --no-aggregate
|
|
294
|
+
|
|
295
|
+
# Table output (no files generated)
|
|
296
|
+
piqc scan --format table
|
|
297
|
+
|
|
298
|
+
# Skip GPU metrics (faster)
|
|
299
|
+
piqc scan --no-exec
|
|
300
|
+
|
|
301
|
+
# Push facts to the ParallelIQ platform (closed-loop demo)
|
|
302
|
+
piqc scan --push-url http://localhost:8000
|
|
303
|
+
|
|
304
|
+
# Push with an explicit cluster ID
|
|
305
|
+
piqc scan --push-url https://api.paralleliq.ai --cluster-id prod-cluster-1
|
|
306
|
+
"""
|
|
307
|
+
# Setup logging
|
|
308
|
+
setup_logging(verbose=verbose, debug=debug)
|
|
309
|
+
|
|
310
|
+
# Print header
|
|
311
|
+
print_header()
|
|
312
|
+
|
|
313
|
+
# Connect to cluster
|
|
314
|
+
print_info("Connecting to cluster...")
|
|
315
|
+
|
|
316
|
+
try:
|
|
317
|
+
k8s_client = K8sClient(
|
|
318
|
+
kubeconfig_path=kubeconfig,
|
|
319
|
+
context=context,
|
|
320
|
+
timeout=timeout,
|
|
321
|
+
)
|
|
322
|
+
k8s_client.test_connection()
|
|
323
|
+
|
|
324
|
+
conn_info = k8s_client.get_connection_info()
|
|
325
|
+
console.print(f" Context: {conn_info['context']}")
|
|
326
|
+
if conn_info['cluster'] != "unknown":
|
|
327
|
+
console.print(f" Cluster: {conn_info['cluster']}")
|
|
328
|
+
console.print()
|
|
329
|
+
|
|
330
|
+
except KubernetesConnectionError as e:
|
|
331
|
+
print_error(str(e))
|
|
332
|
+
if e.details:
|
|
333
|
+
console.print(f" {e.details}")
|
|
334
|
+
sys.exit(1)
|
|
335
|
+
except RBACPermissionError as e:
|
|
336
|
+
print_error(str(e))
|
|
337
|
+
console.print(" See RBAC documentation for required permissions.")
|
|
338
|
+
sys.exit(1)
|
|
339
|
+
except Exception as e:
|
|
340
|
+
print_error(f"Unexpected error: {e}")
|
|
341
|
+
if debug:
|
|
342
|
+
console.print_exception()
|
|
343
|
+
sys.exit(1)
|
|
344
|
+
|
|
345
|
+
# Initialize orchestrator
|
|
346
|
+
orchestrator = ScanOrchestrator(
|
|
347
|
+
k8s_client=k8s_client,
|
|
348
|
+
enable_exec=not no_exec,
|
|
349
|
+
enable_logs=not no_logs,
|
|
350
|
+
enable_runtime_collection=collect_runtime,
|
|
351
|
+
workers=workers,
|
|
352
|
+
timeout=timeout,
|
|
353
|
+
)
|
|
354
|
+
|
|
355
|
+
# Log mode if not auto
|
|
356
|
+
if mode != "auto":
|
|
357
|
+
print_info(f"Execution mode: {mode}")
|
|
358
|
+
|
|
359
|
+
# Execute scan
|
|
360
|
+
print_info("Scanning namespaces...")
|
|
361
|
+
|
|
362
|
+
namespaces = [namespace] if namespace else None
|
|
363
|
+
|
|
364
|
+
try:
|
|
365
|
+
result = orchestrator.scan(namespaces=namespaces)
|
|
366
|
+
except Exception as e:
|
|
367
|
+
print_error(f"Scan failed: {e}")
|
|
368
|
+
if debug:
|
|
369
|
+
console.print_exception()
|
|
370
|
+
sys.exit(1)
|
|
371
|
+
|
|
372
|
+
# Print scan summary
|
|
373
|
+
console.print(f" Discovered: {result.namespaces_scanned} namespace(s)")
|
|
374
|
+
console.print()
|
|
375
|
+
|
|
376
|
+
print_info("Detecting inference workloads...")
|
|
377
|
+
console.print(f" Pods analyzed: {result.pods_analyzed}")
|
|
378
|
+
console.print(f" Inference deployments found: {result.deployments_found}")
|
|
379
|
+
console.print()
|
|
380
|
+
|
|
381
|
+
# Print framework distribution
|
|
382
|
+
if result.modelspecs:
|
|
383
|
+
table_gen = TableGenerator(console)
|
|
384
|
+
table_gen.print_framework_summary(result.modelspecs)
|
|
385
|
+
console.print()
|
|
386
|
+
|
|
387
|
+
# Print warnings
|
|
388
|
+
if result.warnings:
|
|
389
|
+
print_info("Warnings:")
|
|
390
|
+
for warning in result.warnings[:10]: # Limit to first 10
|
|
391
|
+
console.print(f" - {warning}", style="yellow")
|
|
392
|
+
if len(result.warnings) > 10:
|
|
393
|
+
console.print(f" ... and {len(result.warnings) - 10} more")
|
|
394
|
+
console.print()
|
|
395
|
+
|
|
396
|
+
# Generate output
|
|
397
|
+
if result.modelspecs:
|
|
398
|
+
# Show hint for unknown-runtime deployments
|
|
399
|
+
unknown = [s for s in result.modelspecs if s.engine.name == "unknown"]
|
|
400
|
+
if unknown:
|
|
401
|
+
console.print(
|
|
402
|
+
f"[yellow] {len(unknown)} unknown-runtime pod(s) detected "
|
|
403
|
+
f"(TGI / Triton / other). Run with --verbose to inspect.[/yellow]"
|
|
404
|
+
)
|
|
405
|
+
console.print()
|
|
406
|
+
|
|
407
|
+
if output_format == "table":
|
|
408
|
+
# Table output to console
|
|
409
|
+
table_gen = TableGenerator(console)
|
|
410
|
+
table_gen.generate_summary_table(
|
|
411
|
+
result.modelspecs,
|
|
412
|
+
gpu_cost_override=gpu_cost,
|
|
413
|
+
unallocated_nodes=result.unallocated_nodes,
|
|
414
|
+
node_cost_override=node_cost,
|
|
415
|
+
fragmented_nodes=result.fragmented_nodes,
|
|
416
|
+
pending_gpu_pods=result.pending_gpu_pods,
|
|
417
|
+
)
|
|
418
|
+
|
|
419
|
+
elif output_format == "yaml":
|
|
420
|
+
yaml_gen = YAMLGenerator()
|
|
421
|
+
if combined:
|
|
422
|
+
output_file = f"{output}/modelspecs.yaml"
|
|
423
|
+
yaml_gen.generate_combined(result.modelspecs, output_file)
|
|
424
|
+
print_info(f"Output generated:")
|
|
425
|
+
console.print(f" File: {output_file}")
|
|
426
|
+
else:
|
|
427
|
+
files = yaml_gen.generate_multi(result.modelspecs, output)
|
|
428
|
+
print_info(f"Output generated:")
|
|
429
|
+
console.print(f" Directory: {output}")
|
|
430
|
+
console.print(f" Files: {len(files)} ModelSpec file(s)")
|
|
431
|
+
|
|
432
|
+
elif output_format == "json":
|
|
433
|
+
json_gen = JSONGenerator()
|
|
434
|
+
if combined:
|
|
435
|
+
output_file = f"{output}/modelspecs.json"
|
|
436
|
+
json_gen.generate_combined(result.modelspecs, output_file)
|
|
437
|
+
print_info(f"Output generated:")
|
|
438
|
+
console.print(f" File: {output_file}")
|
|
439
|
+
else:
|
|
440
|
+
files = json_gen.generate_multi(result.modelspecs, output)
|
|
441
|
+
print_info(f"Output generated:")
|
|
442
|
+
console.print(f" Directory: {output}")
|
|
443
|
+
console.print(f" Files: {len(files)} ModelSpec file(s)")
|
|
444
|
+
|
|
445
|
+
# Generate PIQC facts bundle if requested (also implied by --push-url)
|
|
446
|
+
if output_piqc or push_url:
|
|
447
|
+
piqc_gen = PIQCGenerator()
|
|
448
|
+
piqc_file = piqc_gen.generate(
|
|
449
|
+
modelspecs=result.modelspecs,
|
|
450
|
+
output_path=output,
|
|
451
|
+
cluster_context=conn_info.get('context'),
|
|
452
|
+
cluster_name=conn_info.get('cluster'),
|
|
453
|
+
namespaces=namespaces,
|
|
454
|
+
unallocated_nodes=result.unallocated_nodes,
|
|
455
|
+
)
|
|
456
|
+
print_info(f"PIQC facts bundle generated:")
|
|
457
|
+
console.print(f" File: {piqc_file}")
|
|
458
|
+
|
|
459
|
+
if push_url:
|
|
460
|
+
_push_bundle(
|
|
461
|
+
piqc_file=piqc_file,
|
|
462
|
+
push_url=push_url,
|
|
463
|
+
cluster_id=cluster_id or conn_info.get('cluster') or conn_info.get('context') or 'unknown',
|
|
464
|
+
console=console,
|
|
465
|
+
debug=debug,
|
|
466
|
+
)
|
|
467
|
+
|
|
468
|
+
console.print()
|
|
469
|
+
else:
|
|
470
|
+
print_info("No inference deployments found.")
|
|
471
|
+
console.print()
|
|
472
|
+
|
|
473
|
+
# Contribute anonymized benchmarks if requested
|
|
474
|
+
if contribute_benchmarks and result.modelspecs:
|
|
475
|
+
from piqc import __version__
|
|
476
|
+
from piqc.telemetry import contribute
|
|
477
|
+
sent, records = contribute(result.modelspecs, __version__)
|
|
478
|
+
if sent > 0:
|
|
479
|
+
console.print(f"[dim] Contributed {sent} anonymized benchmark record(s) to paralleliq.ai[/dim]")
|
|
480
|
+
if verbose:
|
|
481
|
+
import json
|
|
482
|
+
console.print("[dim] Payload (no identifying info):[/dim]")
|
|
483
|
+
for r in records:
|
|
484
|
+
console.print(f"[dim] {json.dumps(r)}[/dim]")
|
|
485
|
+
else:
|
|
486
|
+
console.print("[dim] Benchmark contribution failed (non-fatal) — check connectivity[/dim]")
|
|
487
|
+
console.print()
|
|
488
|
+
|
|
489
|
+
# Print timing
|
|
490
|
+
print_info(f"Scan completed in {format_duration(result.duration_seconds)}")
|
|
491
|
+
|
|
492
|
+
# Exit with appropriate code
|
|
493
|
+
if result.errors:
|
|
494
|
+
sys.exit(1)
|
|
495
|
+
|
|
496
|
+
|
|
497
|
+
@main.command()
|
|
498
|
+
@click.option(
|
|
499
|
+
"--kubeconfig",
|
|
500
|
+
type=click.Path(exists=True),
|
|
501
|
+
default=None,
|
|
502
|
+
help="Path to kubeconfig file",
|
|
503
|
+
)
|
|
504
|
+
@click.option(
|
|
505
|
+
"--context",
|
|
506
|
+
type=str,
|
|
507
|
+
default=None,
|
|
508
|
+
help="Kubernetes context to use",
|
|
509
|
+
)
|
|
510
|
+
def test_connection(
|
|
511
|
+
kubeconfig: Optional[str],
|
|
512
|
+
context: Optional[str],
|
|
513
|
+
) -> None:
|
|
514
|
+
"""
|
|
515
|
+
Test connection to Kubernetes cluster.
|
|
516
|
+
|
|
517
|
+
Verifies that the tool can connect to the cluster and has
|
|
518
|
+
necessary permissions for scanning.
|
|
519
|
+
"""
|
|
520
|
+
print_header()
|
|
521
|
+
print_info("Testing cluster connection...")
|
|
522
|
+
|
|
523
|
+
try:
|
|
524
|
+
k8s_client = K8sClient(
|
|
525
|
+
kubeconfig_path=kubeconfig,
|
|
526
|
+
context=context,
|
|
527
|
+
)
|
|
528
|
+
k8s_client.test_connection()
|
|
529
|
+
|
|
530
|
+
conn_info = k8s_client.get_connection_info()
|
|
531
|
+
|
|
532
|
+
console.print()
|
|
533
|
+
console.print("[green]Connection successful[/green]")
|
|
534
|
+
console.print()
|
|
535
|
+
console.print(f"Context: {conn_info['context']}")
|
|
536
|
+
console.print(f"Cluster: {conn_info['cluster']}")
|
|
537
|
+
|
|
538
|
+
# Test namespace listing
|
|
539
|
+
print_info("Testing namespace access...")
|
|
540
|
+
namespaces = k8s_client.list_namespaces()
|
|
541
|
+
console.print(f" Accessible namespaces: {len(namespaces)}")
|
|
542
|
+
|
|
543
|
+
console.print()
|
|
544
|
+
console.print("[green]All checks passed[/green]")
|
|
545
|
+
|
|
546
|
+
except KubernetesConnectionError as e:
|
|
547
|
+
print_error(str(e))
|
|
548
|
+
if e.details:
|
|
549
|
+
console.print(f" {e.details}")
|
|
550
|
+
sys.exit(1)
|
|
551
|
+
except RBACPermissionError as e:
|
|
552
|
+
print_error(str(e))
|
|
553
|
+
sys.exit(1)
|
|
554
|
+
except Exception as e:
|
|
555
|
+
print_error(f"Connection test failed: {e}")
|
|
556
|
+
sys.exit(1)
|
|
557
|
+
|
|
558
|
+
|
|
559
|
+
@main.command()
|
|
560
|
+
def version() -> None:
|
|
561
|
+
"""Display version information."""
|
|
562
|
+
console.print(f"piqc v{__version__}")
|
|
563
|
+
|
|
564
|
+
|
|
565
|
+
if __name__ == "__main__":
|
|
566
|
+
main()
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Collectors for gathering deployment information."""
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""
|
|
2
|
+
AMD GPU Metrics Collection (Coming Soon)
|
|
3
|
+
|
|
4
|
+
This module will provide support for AMD GPU metrics collection
|
|
5
|
+
using rocm-smi for AMD Instinct and Radeon GPUs.
|
|
6
|
+
|
|
7
|
+
Planned Features:
|
|
8
|
+
- ROCm SMI integration for AMD GPU metrics
|
|
9
|
+
- AMD Instinct MI250X/MI300X support
|
|
10
|
+
- GPU utilization, memory, temperature, and power metrics
|
|
11
|
+
- Seamless integration with existing GPU collection framework
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
# Implementation coming in a future release
|