@minato-aqukin/autodl-cli 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,948 @@
1
+ import { z } from 'zod';
2
+ import { McpServer } from '@modelcontextprotocol/sdk/server/mcp.js';
3
+ import { Client } from 'ssh2';
4
+ import { Writable } from 'node:stream';
5
+
6
+ interface StoredConfig {
7
+ token?: string;
8
+ baseUrl?: string;
9
+ lang?: "zh" | "en";
10
+ defaults?: {
11
+ gpu?: string;
12
+ image?: string;
13
+ regions?: string[];
14
+ ttl?: string;
15
+ minBalanceYuan?: number;
16
+ };
17
+ }
18
+ declare function readConfig(): StoredConfig;
19
+ /** Write config with 0600 so a shared machine can't read the token. */
20
+ declare function writeConfig(config: StoredConfig): void;
21
+ declare function updateConfig(patch: Partial<StoredConfig>): StoredConfig;
22
+ interface TokenResolution {
23
+ token: string;
24
+ source: "flag" | "env" | "config";
25
+ }
26
+ /**
27
+ * Token precedence: explicit flag > AUTODL_TOKEN > config file.
28
+ * Throws a AUTH_MISSING error (exit 3) with setup instructions when nothing is found.
29
+ */
30
+ declare function resolveToken(explicit?: string): TokenResolution;
31
+
32
+ /**
33
+ * Error taxonomy for AutoDL-cli.
34
+ *
35
+ * Every failure that reaches the CLI surface maps to exactly one exit code so that
36
+ * agents can branch on the result without parsing prose. The codes are part of the
37
+ * public contract — changing one is a breaking change.
38
+ */
39
+ declare const ExitCode: {
40
+ readonly OK: 0;
41
+ readonly GENERIC: 1;
42
+ readonly USAGE: 2;
43
+ readonly AUTH: 3;
44
+ readonly NOT_FOUND: 4;
45
+ readonly BUDGET: 5;
46
+ readonly NO_STOCK: 6;
47
+ readonly TIMEOUT: 7;
48
+ readonly SSH: 8;
49
+ };
50
+ type ExitCode = (typeof ExitCode)[keyof typeof ExitCode];
51
+ /** Machine-readable error codes surfaced in `--json` output. */
52
+ type ErrorCode = "GENERIC" | "USAGE" | "AUTH_MISSING" | "AUTH_INVALID" | "NOT_FOUND" | "INSUFFICIENT_BALANCE" | "GUARD_BLOCKED" | "NO_STOCK" | "TIMEOUT" | "SSH_FAILED" | "API_ERROR" | "NETWORK";
53
+ interface AutoDLErrorOptions {
54
+ code?: ErrorCode;
55
+ exitCode?: ExitCode;
56
+ /** Actionable next step shown to humans and handed to agents verbatim. */
57
+ hint?: string;
58
+ /** AutoDL's own `request_id`, invaluable when asking support about a failure. */
59
+ requestId?: string;
60
+ cause?: unknown;
61
+ details?: Record<string, unknown>;
62
+ }
63
+ declare class AutoDLError extends Error {
64
+ readonly code: ErrorCode;
65
+ readonly exitCode: ExitCode;
66
+ readonly hint?: string;
67
+ readonly requestId?: string;
68
+ readonly details?: Record<string, unknown>;
69
+ constructor(message: string, options?: AutoDLErrorOptions);
70
+ toJSON(): {
71
+ code: ErrorCode;
72
+ message: string;
73
+ hint?: string;
74
+ requestId?: string;
75
+ };
76
+ }
77
+ declare class AuthError extends AutoDLError {
78
+ constructor(message: string, options?: AutoDLErrorOptions);
79
+ }
80
+ declare class NotFoundError extends AutoDLError {
81
+ constructor(message: string, options?: AutoDLErrorOptions);
82
+ }
83
+ declare class UsageError extends AutoDLError {
84
+ constructor(message: string, options?: AutoDLErrorOptions);
85
+ }
86
+ declare class BudgetError extends AutoDLError {
87
+ constructor(message: string, options?: AutoDLErrorOptions);
88
+ }
89
+ declare class NoStockError extends AutoDLError {
90
+ constructor(message: string, options?: AutoDLErrorOptions);
91
+ }
92
+ declare class TimeoutError extends AutoDLError {
93
+ constructor(message: string, options?: AutoDLErrorOptions);
94
+ }
95
+ declare class SSHError extends AutoDLError {
96
+ constructor(message: string, options?: AutoDLErrorOptions);
97
+ }
98
+ /** Coerce anything thrown into an AutoDLError so the CLI always has an exit code. */
99
+ declare function toAutoDLError(err: unknown): AutoDLError;
100
+
101
+ declare const DEFAULT_BASE_URL = "https://api.autodl.com";
102
+ interface ClientOptions {
103
+ token: string;
104
+ baseUrl?: string;
105
+ /** Per-request timeout in ms. */
106
+ timeoutMs?: number;
107
+ /** Retry attempts for transient failures (network, 429, 5xx). */
108
+ maxRetries?: number;
109
+ /** Base delay for exponential backoff. Lowered in tests to keep the suite fast. */
110
+ retryBaseDelayMs?: number;
111
+ /** Called with sanitised request/response summaries when --verbose is on. */
112
+ onDebug?: (message: string) => void;
113
+ fetchImpl?: typeof fetch;
114
+ }
115
+ interface RequestOptions {
116
+ method: "GET" | "POST" | "PUT" | "DELETE";
117
+ path: string;
118
+ body?: Record<string, unknown>;
119
+ /** Override the retry budget for a single call (e.g. 0 for non-idempotent creates). */
120
+ maxRetries?: number;
121
+ timeoutMs?: number;
122
+ }
123
+ /** Never let a token reach a log line intact. */
124
+ declare function redactToken(token: string): string;
125
+ declare class AutoDLClient {
126
+ private readonly token;
127
+ private readonly baseUrl;
128
+ private readonly timeoutMs;
129
+ private readonly maxRetries;
130
+ private readonly retryBaseDelayMs;
131
+ private readonly onDebug;
132
+ private readonly fetchImpl;
133
+ constructor(options: ClientOptions);
134
+ get maskedToken(): string;
135
+ get<T>(path: string, body?: Record<string, unknown>, extra?: Partial<RequestOptions>): Promise<T>;
136
+ post<T>(path: string, body?: Record<string, unknown>, extra?: Partial<RequestOptions>): Promise<T>;
137
+ put<T>(path: string, body?: Record<string, unknown>, extra?: Partial<RequestOptions>): Promise<T>;
138
+ delete<T>(path: string, body?: Record<string, unknown>, extra?: Partial<RequestOptions>): Promise<T>;
139
+ request<T>(options: RequestOptions): Promise<T>;
140
+ private debug;
141
+ private attempt;
142
+ private fetchJson;
143
+ /** GET with a JSON body — legal HTTP/1.1, but fetch refuses, so drop to node:https. */
144
+ private rawGetWithBody;
145
+ private readEnvelope;
146
+ }
147
+
148
+ /**
149
+ * Minimal message catalogue. AutoDL's user base is overwhelmingly Chinese-speaking,
150
+ * so zh is the default; `--lang en` / `AUTODL_LANG=en` switches.
151
+ */
152
+ type Lang = "zh" | "en";
153
+
154
+ interface GlobalOptions {
155
+ token?: string;
156
+ baseUrl?: string;
157
+ json?: boolean;
158
+ color?: boolean;
159
+ verbose?: boolean;
160
+ lang?: string;
161
+ /**
162
+ * Whether to run the opportunistic TTL sweep. Commander's `--no-sweep` sets this to
163
+ * `false` (it does NOT produce a `noSweep` key), so the check below tests for an
164
+ * explicit `false` rather than for truthiness of a negated name.
165
+ */
166
+ sweep?: boolean;
167
+ }
168
+ interface Context {
169
+ client: AutoDLClient;
170
+ lang: Lang;
171
+ }
172
+ /**
173
+ * Build the client every command shares.
174
+ *
175
+ * Kept separate from the commander wiring so the MCP server can construct an identical
176
+ * context — the two entry points must never drift in how they resolve tokens, honour
177
+ * env vars, or apply guards.
178
+ */
179
+ declare function createContext(options?: GlobalOptions): Context;
180
+
181
+ /**
182
+ * Static reference data.
183
+ *
184
+ * AutoDL's open API exposes no catalogue or stock endpoint for Pro instances, so the
185
+ * GPU specs, region codes and public base images have to be baked in. They can drift
186
+ * when the platform changes — `autodl gpus` / `regions` / `images --base` print this
187
+ * table, and the README asks users to open an issue when something is stale.
188
+ *
189
+ * Sources: https://www.autodl.com/docs/instance_pro_api/ and /docs/esd_api_doc/
190
+ * Last verified: 2026-08.
191
+ */
192
+ interface GpuSpec {
193
+ /** Value for `gpu_spec_uuid` in the create payload. */
194
+ id: string;
195
+ /** Name as shown in the AutoDL console. */
196
+ displayName: string;
197
+ /** AutoDL's own tiering: 通用型 vs 性能型. */
198
+ tier: "general" | "performance";
199
+ vramGb: number;
200
+ /** Convenience aliases so `--gpu 4090` resolves without the exact spec id. */
201
+ aliases: string[];
202
+ /**
203
+ * The name this GPU goes by in the stock endpoint, which uses a different naming
204
+ * scheme from `gpu_spec_uuid`. Verified against live data: westDC3 returns both
205
+ * `vGPU-48GB` and `vGPU-48GB-350W`, matching the `v-48g` / `v-48g-350w` split
206
+ * exactly, which is what makes this mapping safe to hardcode.
207
+ */
208
+ stockName: string;
209
+ }
210
+ declare const GPU_SPECS: readonly GpuSpec[];
211
+ /**
212
+ * A region in the `data_center_list` / `gpu_stock` namespace.
213
+ *
214
+ * AutoDL uses a *different* namespace in instance responses — an instance in 北京B区
215
+ * reports `region_sign: "bj-B2"`, not `beijingDC2`. Never feed an instance's reported
216
+ * region back into these APIs; see `assertStockRegion`.
217
+ *
218
+ * Names verified against the official elastic-deployment API docs, 2026-08.
219
+ */
220
+ interface Region {
221
+ /** Value for `data_center_list` entries. */
222
+ id: string;
223
+ displayName: string;
224
+ aliases: string[];
225
+ /**
226
+ * Whether Pro instance creation accepts this region in `data_center_list`.
227
+ *
228
+ * Only two do. Every other region is elastic-deployment only and makes
229
+ * `instance/pro/create` fail with `RequestParameterIsWrong`. Verified empirically
230
+ * against the live API on 2026-08-23 by probing all eleven.
231
+ */
232
+ proCreate: boolean;
233
+ }
234
+ declare const REGIONS: readonly Region[];
235
+ interface BaseImage {
236
+ uuid: string;
237
+ framework: string;
238
+ /** Full image tag as AutoDL names it. */
239
+ tag: string;
240
+ cuda: string;
241
+ python: string;
242
+ }
243
+ declare const BASE_IMAGES: readonly BaseImage[];
244
+ /** The image `autodl create` picks when the user doesn't name one. */
245
+ declare const DEFAULT_BASE_IMAGE = "base-image-l2t43iu6uk";
246
+ /** Resolve a user-supplied GPU string to a spec id, accepting display names and aliases. */
247
+ declare function resolveGpuSpec(input: string): GpuSpec | undefined;
248
+ declare function resolveRegion(input: string): Region | undefined;
249
+ /**
250
+ * Guard the stock endpoint against a region id from the wrong namespace.
251
+ *
252
+ * `gpu_stock` answers an unknown region with `{"code":"Success","data":[]}` — a typo and
253
+ * a genuinely sold-out region are indistinguishable in the response. Without this check
254
+ * `--region bj-B2` would silently report "no stock" forever.
255
+ */
256
+ declare function assertStockRegion(input: string): Region;
257
+ declare function findBaseImage(input: string): BaseImage | undefined;
258
+ /** `11.8` / `118` / `11` -> the integer form AutoDL's `cuda_v_from` expects. */
259
+ declare function parseCudaVersion(input: string | number): number;
260
+ /** Inverse of parseCudaVersion, for display: `118` -> `11.8`. */
261
+ declare function formatCudaVersion(value: number): string;
262
+
263
+ /**
264
+ * Parse a human duration like `30m`, `2h`, `1.5h`, `90` (bare = minutes).
265
+ * Used by `--ttl`, `--timeout` and the idle guard.
266
+ */
267
+ declare function parseDuration(input: string, { bareUnit }?: {
268
+ bareUnit?: string | undefined;
269
+ }): number;
270
+ /** Seconds -> compact human string, e.g. `2h30m`. */
271
+ declare function formatDuration(seconds: number): string;
272
+
273
+ declare const instanceStatusSchema: z.ZodEnum<["creating", "starting", "running", "shutting_down", "shutdown", "releasing", "released", "failed"]>;
274
+ type InstanceStatus = z.infer<typeof instanceStatusSchema> | (string & {});
275
+ /** Public, stable shape returned by `autodl ls --json`. Breaking changes need a major. */
276
+ interface Instance {
277
+ uuid: string;
278
+ name: string | null;
279
+ status: InstanceStatus;
280
+ subStatus: string | null;
281
+ machineId: string | null;
282
+ regionSign: string | null;
283
+ regionName: string | null;
284
+ chargeType: string | null;
285
+ startMode: string | null;
286
+ gpuSpec: string | null;
287
+ gpuNum: number;
288
+ createdAt: string | null;
289
+ startedAt: string | null;
290
+ stoppedAt: string | null;
291
+ expiredAt: string | null;
292
+ timedShutdownAt: string | null;
293
+ }
294
+ declare function normalizeInstance(raw: unknown): Instance;
295
+ interface ServiceEndpoint {
296
+ port: number;
297
+ domain: string;
298
+ protocol: string | null;
299
+ }
300
+ /** Public shape for `autodl info --json`. `rootPassword` is sensitive — see redaction. */
301
+ interface InstanceSnapshot {
302
+ regionSign: string | null;
303
+ gpuAlias: string | null;
304
+ chipCorp: string | null;
305
+ cpuArch: string | null;
306
+ /** Pay-as-you-go rate in yuan per hour. */
307
+ priceYuanPerHour: number;
308
+ originalPriceYuanPerHour: number;
309
+ ssh: {
310
+ command: string | null;
311
+ host: string | null;
312
+ port: number | null;
313
+ user: "root";
314
+ password: string | null;
315
+ };
316
+ jupyter: {
317
+ token: string | null;
318
+ domain: string | null;
319
+ };
320
+ services: ServiceEndpoint[];
321
+ disk: {
322
+ systemInitBytes: number;
323
+ systemExpandBytes: number;
324
+ };
325
+ usage: {
326
+ cpuPercent: number | null;
327
+ memPercent: number | null;
328
+ memUsedBytes: number | null;
329
+ memLimitBytes: number | null;
330
+ rootFsUsedBytes: number | null;
331
+ rootFsTotalBytes: number | null;
332
+ dataDiskUsedBytes: number | null;
333
+ dataDiskTotalBytes: number | null;
334
+ };
335
+ }
336
+ declare function normalizeSnapshot(raw: unknown): InstanceSnapshot;
337
+ interface Balance {
338
+ /** Spendable balance in yuan. */
339
+ balanceYuan: number;
340
+ /** Lifetime spend in yuan. */
341
+ accumulatedYuan: number;
342
+ /** Voucher balance in yuan. */
343
+ voucherYuan: number;
344
+ }
345
+ declare function normalizeBalance(raw: unknown): Balance;
346
+ interface PrivateImage {
347
+ imageUuid: string;
348
+ name: string;
349
+ status: string | null;
350
+ sizeBytes: number;
351
+ createdAt: string | null;
352
+ }
353
+ interface Pagination {
354
+ pageIndex: number;
355
+ pageSize: number;
356
+ maxPage: number;
357
+ total: number;
358
+ }
359
+ /** Strip secrets before printing a snapshot to a shared surface (logs, MCP previews). */
360
+ declare function redactSnapshot(snapshot: InstanceSnapshot): InstanceSnapshot;
361
+
362
+ /** Wallet balance, lifetime spend and voucher balance — the only account endpoint AutoDL exposes. */
363
+ declare function getBalance(client: AutoDLClient): Promise<Balance>;
364
+ /**
365
+ * Mount or unmount the exclusive NFS / file storage for a region.
366
+ * `mountable` is AutoDL's own 1 / -1 convention.
367
+ */
368
+ declare function setNfsMount(client: AutoDLClient, dataCenter: string, mount: boolean): Promise<void>;
369
+
370
+ /** Snapshot a running instance into a reusable private image. */
371
+ declare function saveImage(client: AutoDLClient, instanceUuid: string, imageName: string): Promise<string>;
372
+ declare function listPrivateImages(client: AutoDLClient, options?: {
373
+ pageIndex?: number;
374
+ pageSize?: number;
375
+ }): Promise<{
376
+ images: PrivateImage[];
377
+ pagination: Pagination;
378
+ }>;
379
+
380
+ interface CreateInstanceInput {
381
+ /** `gpu_spec_uuid`, e.g. `pro6000-p`. Resolve aliases before calling. */
382
+ gpuSpec: string;
383
+ /** 1-4, enforced by AutoDL. */
384
+ gpuNum: number;
385
+ imageUuid: string;
386
+ /** Integer CUDA floor, e.g. 118 for CUDA >= 11.8. */
387
+ cudaFrom: number;
388
+ /** 0-500 GB of extra system disk, fixed at creation time. */
389
+ expandSystemDiskGb?: number;
390
+ /** Region preference list; AutoDL picks the first with capacity. */
391
+ regions?: string[];
392
+ name?: string;
393
+ /** Shell command run after boot. Also how the TTL guard arms itself. */
394
+ startCommand?: string;
395
+ }
396
+ /** Create a pay-as-you-go Pro instance. Returns the new instance uuid. */
397
+ declare function createInstance(client: AutoDLClient, input: CreateInstanceInput): Promise<string>;
398
+ interface ListInstancesOptions {
399
+ pageIndex?: number;
400
+ pageSize?: number;
401
+ }
402
+ declare function listInstancesPage(client: AutoDLClient, options?: ListInstancesOptions): Promise<{
403
+ instances: Instance[];
404
+ pagination: Pagination;
405
+ }>;
406
+ /** Walk every page so callers never have to think about pagination. */
407
+ declare function listAllInstances(client: AutoDLClient): Promise<Instance[]>;
408
+ declare function getInstanceStatus(client: AutoDLClient, uuid: string): Promise<string>;
409
+ declare function getInstanceSnapshot(client: AutoDLClient, uuid: string): Promise<InstanceSnapshot>;
410
+ /** Find one instance in the list endpoint (there is no per-instance GET). */
411
+ declare function findInstance(client: AutoDLClient, uuid: string): Promise<Instance>;
412
+ interface PowerOnOptions {
413
+ /**
414
+ * `payload` is fixed to "gpu" because nothing else works — for now.
415
+ *
416
+ * AutoDL documents it as "gpu:有卡开机, 暂不支持API以无卡模式开机": *not yet*
417
+ * supported, not never. Probed live 2026-08-24 on a shut-down instance: "cpu",
418
+ * "no_gpu", "nogpu", "cpu_only", "cpu-only" and "none" each returned
419
+ * `不支持的启动模式`, and an empty string was accepted but came back with
420
+ * `start_mode: "gpu"` — a default, not a CPU mode.
421
+ *
422
+ * When AutoDL enables it, the change is small: add `mode?: "gpu" | "<their value>"`
423
+ * to PowerOnOptions, thread it into the payload below, and surface it as
424
+ * `autodl start --no-gpu` plus a `mode` field on the MCP power-on tool.
425
+ *
426
+ * Re-checking costs nothing: from any already shut-down instance, a power_on with a
427
+ * rejected payload is refused without booting anything.
428
+ */
429
+ startCommand?: string;
430
+ }
431
+ declare function powerOnInstance(client: AutoDLClient, uuid: string, options?: PowerOnOptions): Promise<void>;
432
+ declare function powerOffInstance(client: AutoDLClient, uuid: string): Promise<void>;
433
+ /** AutoDL requires the instance to be shut down first; this is irreversible. */
434
+ declare function releaseInstance(client: AutoDLClient, uuid: string): Promise<void>;
435
+
436
+ /**
437
+ * GPU stock, the only capacity-visibility endpoint AutoDL exposes.
438
+ *
439
+ * Officially titled "获取弹性部署GPU库存" — it belongs to the elastic-deployment product
440
+ * line, not to Pro container instances. The returned names include Pro-only variants
441
+ * (`vGPU-48GB-350W`, `RTX PRO 6000`, `H800`), which strongly suggests a shared physical
442
+ * pool, but AutoDL makes no guarantee that "ESD has stock" means "creating a Pro
443
+ * instance will succeed". Callers must treat this as advisory ranking, never as a gate.
444
+ */
445
+ interface GpuStockEntry {
446
+ /** Stock-namespace GPU name, e.g. `RTX PRO 6000`. See `GpuSpec.stockName`. */
447
+ gpuName: string;
448
+ idle: number;
449
+ total: number;
450
+ /** Present in live responses though absent from the published docs. */
451
+ chipCorp: string | null;
452
+ cpuArch: string | null;
453
+ }
454
+ interface StockQuery {
455
+ /** Region id in the `data_center_list` namespace, e.g. `beijingDC2`. */
456
+ regionSign: string;
457
+ /** Restrict to these stock-namespace GPU names. */
458
+ gpuNames?: string[];
459
+ cudaFrom?: number;
460
+ cudaTo?: number;
461
+ cpuNumFrom?: number;
462
+ cpuNumTo?: number;
463
+ memorySizeFrom?: number;
464
+ memorySizeTo?: number;
465
+ /** Price bounds in yuan per hour; converted to AutoDL's milliyuan on the way out. */
466
+ priceFromYuan?: number;
467
+ priceToYuan?: number;
468
+ }
469
+ declare function getRegionGpuStock(client: AutoDLClient, query: StockQuery): Promise<GpuStockEntry[]>;
470
+
471
+ /**
472
+ * AutoDL expresses every monetary value in "milliyuan" (毫元) — integer thousandths
473
+ * of a CNY. We normalise to plain yuan at the edge of `core/` so nothing above this
474
+ * layer ever has to remember the factor.
475
+ */
476
+ /** Milliyuan integer -> yuan number, rounded to fen (2dp) to avoid float noise. */
477
+ declare function milliToYuan(milli: number | null | undefined): number;
478
+ /** Yuan -> milliyuan integer, for request payloads that take price bounds. */
479
+ declare function yuanToMilli(yuan: number): number;
480
+ /** Human display, e.g. `¥1.97`. */
481
+ declare function formatYuan(yuan: number): string;
482
+ /** Human display for a per-hour rate, e.g. `¥1.97/时`. */
483
+ declare function formatRate(yuanPerHour: number, unit?: string): string;
484
+ /**
485
+ * Estimate cost for a duration at a per-hour rate. AutoDL bills per second with a
486
+ * ¥0.01 floor, so this mirrors that rather than rounding up to whole hours.
487
+ */
488
+ declare function estimateCost(yuanPerHour: number, seconds: number): number;
489
+
490
+ /**
491
+ * Parsing and safe assembly of code-hosting URLs.
492
+ *
493
+ * The token handling here is the security-sensitive part: a credential embedded in a
494
+ * clone URL must never reach a log line, an error message, `--json` output, or the
495
+ * remote's stored git config. Everything that builds a URL with a token also provides
496
+ * a redacted twin for display.
497
+ */
498
+ interface ParsedRepo {
499
+ /** Host as given, e.g. `github.com`. */
500
+ host: string;
501
+ /** `owner/name` path, without a trailing `.git`. */
502
+ path: string;
503
+ /** Last path segment — the default directory name. */
504
+ name: string;
505
+ /** Normalised https clone URL, never containing credentials. */
506
+ cloneUrl: string;
507
+ /** True when AutoDL's academic proxy covers this host. */
508
+ needsAcceleration: boolean;
509
+ }
510
+ /**
511
+ * Accept the forms people actually paste: https URLs, `git@host:owner/repo.git`,
512
+ * and bare `owner/repo` (assumed GitHub, matching how most tools behave).
513
+ */
514
+ declare function parseRepo(input: string): ParsedRepo;
515
+ /**
516
+ * Clone URL with a token embedded, plus the redacted form to show instead.
517
+ *
518
+ * Callers must use `display` for every user-visible surface — the raw URL is only ever
519
+ * allowed inside the command string sent over SSH.
520
+ */
521
+ declare function withCredentials(repo: ParsedRepo, token?: string): {
522
+ url: string;
523
+ display: string;
524
+ };
525
+ /** Strip any embedded credential from arbitrary text before it is displayed or stored. */
526
+ declare function redactCredentials(text: string): string;
527
+ /** Resolve a git token from the flag, then the usual environment variables. */
528
+ declare function resolveGitToken(explicit?: string): string | undefined;
529
+
530
+ /**
531
+ * Turn raw per-region stock into a ranked list of places to try.
532
+ *
533
+ * The stock endpoint belongs to elastic deployment, so a positive number is evidence
534
+ * rather than a promise — measured 2026-08-23, it reported 140 idle RTX 4090D in a
535
+ * region where Pro creation answered "暂无库存". Everything here is ordering advice
536
+ * only; nothing refuses to create because a region looks empty.
537
+ */
538
+ interface RegionStock {
539
+ regionId: string;
540
+ regionName: string;
541
+ idle: number;
542
+ total: number;
543
+ }
544
+ interface StockSnapshot {
545
+ regionId: string;
546
+ regionName: string;
547
+ entries: GpuStockEntry[];
548
+ }
549
+ /** Query several regions at once; one region failing must not sink the rest. */
550
+ declare function getStockByRegion(client: AutoDLClient, options?: {
551
+ regions?: string[];
552
+ gpuNames?: string[];
553
+ }): Promise<{
554
+ snapshots: StockSnapshot[];
555
+ failures: {
556
+ regionId: string;
557
+ reason: string;
558
+ }[];
559
+ }>;
560
+ /**
561
+ * Regions that currently have the given GPU free, best first.
562
+ *
563
+ * Returns regions with `idle === 0` too (at the end), so callers can tell "queried and
564
+ * everything is busy" apart from "the query failed" — those are different situations
565
+ * and only one of them warrants a warning.
566
+ */
567
+ declare function findRegionsWithStock(client: AutoDLClient, spec: GpuSpec, options?: {
568
+ regions?: string[];
569
+ }): Promise<{
570
+ ranked: RegionStock[];
571
+ failures: {
572
+ regionId: string;
573
+ reason: string;
574
+ }[];
575
+ }>;
576
+ /** Look up a spec by its stock-namespace name (inverse of `GpuSpec.stockName`). */
577
+ declare function specForStockName(gpuName: string): GpuSpec | undefined;
578
+ interface RegionChoice {
579
+ /** Value for `data_center_list`; empty means "let AutoDL schedule anywhere". */
580
+ regions: string[];
581
+ ranked: RegionStock[];
582
+ /** True when stock was consulted and every candidate region reported zero idle. */
583
+ allBusy: boolean;
584
+ }
585
+ /**
586
+ * Decide what to put in `data_center_list`.
587
+ *
588
+ * Deliberately does NOT narrow an unconstrained request. Two live findings drove this:
589
+ *
590
+ * - Pro creation accepts only westDC3 and beijingDC2; the other nine regions are
591
+ * elastic-deployment only and hard-fail with `RequestParameterIsWrong`.
592
+ * - Stock numbers come from the elastic-deployment pool and do not track Pro
593
+ * availability. Measured 2026-08-23: stock reported 140 idle RTX 4090D in westDC3
594
+ * while Pro creation there answered "暂无库存" — and the same request with no region
595
+ * constraint succeeded, landing in beijingDC2.
596
+ *
597
+ * So an empty list (let AutoDL schedule) beats any list we could infer. When the user
598
+ * does name regions we keep their choice, ordering by stock and warning if it looks
599
+ * empty — a weak signal, but the only one available.
600
+ */
601
+ declare function chooseRegions(client: AutoDLClient, spec: GpuSpec, requested?: string[]): Promise<RegionChoice>;
602
+
603
+ interface WaitOptions {
604
+ /** Give up after this many ms. */
605
+ timeoutMs?: number;
606
+ /** Gap between status polls. */
607
+ intervalMs?: number;
608
+ /** Progress callback, e.g. to drive a spinner. */
609
+ onPoll?: (status: string, elapsedMs: number) => void;
610
+ signal?: AbortSignal;
611
+ }
612
+ /**
613
+ * Poll `instance/pro/status` until it reaches one of `targets`.
614
+ *
615
+ * Used before every SSH attempt: AutoDL reassigns the SSH port and root password on
616
+ * each power cycle, and the snapshot is only trustworthy once the instance is running.
617
+ */
618
+ declare function waitForStatus(client: AutoDLClient, uuid: string, targets: string[], options?: WaitOptions): Promise<string>;
619
+ declare const waitForRunning: (client: AutoDLClient, uuid: string, options?: WaitOptions) => Promise<string>;
620
+ declare const waitForShutdown: (client: AutoDLClient, uuid: string, options?: WaitOptions) => Promise<string>;
621
+
622
+ declare function resolveMinBalance(explicit?: number): number;
623
+ /**
624
+ * Refuse to create an instance when the wallet is nearly empty.
625
+ *
626
+ * AutoDL doesn't reclaim instances the moment the balance hits zero — it keeps them
627
+ * around to protect data — so a low balance turns into a stuck, unusable instance
628
+ * rather than a clean failure. Better to stop before renting.
629
+ */
630
+ declare function assertBudget(client: AutoDLClient, minYuan?: number): Promise<number>;
631
+
632
+ /**
633
+ * Idle detection.
634
+ *
635
+ * The TTL guard handles "this job should never run longer than N hours". This handles
636
+ * the other half: a job that finished (or crashed) an hour ago while the meter kept
637
+ * running. We sample GPU utilisation over SSH and power off after a sustained lull.
638
+ */
639
+ interface IdleOptions {
640
+ /** Utilisation percentage at or below which a sample counts as idle. */
641
+ thresholdPercent?: number;
642
+ /** Consecutive idle samples required before shutting down. */
643
+ samples?: number;
644
+ /** Seconds between samples. */
645
+ intervalSeconds?: number;
646
+ /** Report but don't actually power off. */
647
+ dryRun?: boolean;
648
+ signal?: AbortSignal;
649
+ onSample?: (utilisation: number, consecutiveIdle: number) => void;
650
+ }
651
+ interface IdleResult {
652
+ stopped: boolean;
653
+ samplesTaken: number;
654
+ lastUtilisation: number | null;
655
+ reason: "idle" | "aborted" | "dry-run";
656
+ }
657
+ /** Average utilisation across all GPUs, or null when nvidia-smi gave us nothing usable. */
658
+ declare function parseUtilisation(stdout: string): number | null;
659
+ /**
660
+ * Watch an instance and power it off once the GPU has been quiet long enough.
661
+ * Blocks until it shuts the instance down or the signal aborts.
662
+ */
663
+ declare function watchIdle(client: AutoDLClient, uuid: string, options?: IdleOptions): Promise<IdleResult>;
664
+
665
+ /**
666
+ * Shell snippet that arms the in-instance timer.
667
+ *
668
+ * Deliberately quote-free: it is embedded in AutoDL's `start_command` field, and we
669
+ * can't see how that string is re-parsed on their side. A subshell with `&&` expresses
670
+ * "wait then shut down" without a single quote character.
671
+ */
672
+ declare function buildTTLSnippet(seconds: number): string;
673
+ /** Compose the boot command: arm the timer first, then run whatever the user asked for. */
674
+ declare function composeStartCommand(ttlSeconds: number | undefined, userCommand: string | undefined): string | undefined;
675
+ /**
676
+ * Arm (or re-arm) the timer on an already-running instance over SSH.
677
+ * Used by `autodl start --ttl`, where there is no create payload to piggyback on.
678
+ */
679
+ declare function armTTLOverSSH(client: AutoDLClient, uuid: string, seconds: number): Promise<boolean>;
680
+ /** Cancel a previously armed in-instance timer. */
681
+ declare function disarmTTLOverSSH(client: AutoDLClient, uuid: string): Promise<boolean>;
682
+ interface RecordTTLInput {
683
+ uuid: string;
684
+ name?: string;
685
+ ttlSeconds: number;
686
+ inInstanceTimer: boolean;
687
+ }
688
+ /** Add the instance to the local ledger so the sweep can catch it later. */
689
+ declare function recordTTL(input: RecordTTLInput): void;
690
+ interface SweepResult {
691
+ stopped: string[];
692
+ alreadyStopped: string[];
693
+ failed: {
694
+ uuid: string;
695
+ reason: string;
696
+ }[];
697
+ }
698
+ /**
699
+ * Power off any tracked instance past its TTL.
700
+ *
701
+ * Runs opportunistically before every command, so an agent that forgot to clean up
702
+ * gets caught the next time anything touches the CLI. Failures here are reported but
703
+ * never abort the command the user actually asked for.
704
+ */
705
+ declare function sweepExpired(client: AutoDLClient): Promise<SweepResult>;
706
+
707
+ declare function buildServer(context: Context): McpServer;
708
+ /** Start the stdio MCP server. Blocks until the transport closes. */
709
+ declare function startMcpServer(options?: {
710
+ token?: string;
711
+ baseUrl?: string;
712
+ }): Promise<void>;
713
+
714
+ interface SSHCredentials {
715
+ uuid: string;
716
+ host: string;
717
+ port: number;
718
+ user: "root";
719
+ password: string;
720
+ }
721
+ interface CredentialOptions {
722
+ /** Power the instance on (and wait) when it isn't running. */
723
+ autoStart?: boolean;
724
+ /** How long to wait for `running` when auto-starting. */
725
+ waitTimeoutMs?: number;
726
+ /** Gap between status polls while waiting. */
727
+ pollIntervalMs?: number;
728
+ signal?: AbortSignal;
729
+ }
730
+ /**
731
+ * Fetch live SSH credentials for an instance.
732
+ *
733
+ * AutoDL may reassign `ssh_port` and `root_password` on any power cycle — the instance
734
+ * can be rescheduled onto a different machine. It does not always happen (a real
735
+ * stop/start was observed keeping both identical), which is exactly why caching is
736
+ * unsafe: the stale value works often enough to hide the bug. Nothing in this codebase
737
+ * may cache credentials across calls — always come back through here.
738
+ */
739
+ declare function getCredentials(client: AutoDLClient, uuid: string, options?: CredentialOptions): Promise<SSHCredentials>;
740
+ interface ConnectOptions extends CredentialOptions {
741
+ /** Socket-level connect timeout. */
742
+ connectTimeoutMs?: number;
743
+ keepaliveIntervalMs?: number;
744
+ /** Connection attempts before giving up. Each one re-reads the credentials. */
745
+ connectAttempts?: number;
746
+ }
747
+ /**
748
+ * The single funnel for every SSH operation.
749
+ *
750
+ * Each attempt re-reads the credentials, which covers both ways a connection can fail:
751
+ * the port may have been reassigned since the snapshot was taken, and sshd may simply
752
+ * not be up yet on a freshly booted instance. The first is fixed by the refresh, the
753
+ * second only by waiting — so attempts are spaced out rather than fired back to back.
754
+ */
755
+ declare function withSSH<T>(client: AutoDLClient, uuid: string, fn: (conn: Client, creds: SSHCredentials) => Promise<T>, options?: ConnectOptions): Promise<T>;
756
+
757
+ /**
758
+ * Interactive login hands off to the system `ssh` binary rather than ssh2: the user
759
+ * gets their own terminal handling, agent forwarding, ProxyJump, ~/.ssh/config, etc.
760
+ */
761
+ declare function connectInteractive(client: AutoDLClient, uuid: string, options?: CredentialOptions & {
762
+ extraArgs?: string[];
763
+ }): Promise<number>;
764
+ /** The connection details, for printing or for feeding another tool. */
765
+ declare function formatSSHCommand(creds: SSHCredentials): string;
766
+
767
+ interface ExecResult {
768
+ /** Remote process exit code. `null` when the process was killed by a signal. */
769
+ exitCode: number | null;
770
+ signal: string | null;
771
+ stdout: string;
772
+ stderr: string;
773
+ }
774
+ interface ExecOptions extends ConnectOptions {
775
+ /** Mirror remote output to these streams as it arrives. */
776
+ stdout?: Writable;
777
+ stderr?: Writable;
778
+ /** Capture output into the result. Off for huge jobs to bound memory. */
779
+ capture?: boolean;
780
+ /** Kill the remote command after this many ms. */
781
+ timeoutMs?: number;
782
+ /** Working directory on the remote host. */
783
+ cwd?: string;
784
+ /** Extra environment variables exported before the command runs. */
785
+ env?: Record<string, string>;
786
+ /** Request a PTY — needed for programs that check isatty (e.g. progress bars). */
787
+ pty?: boolean;
788
+ /**
789
+ * Run through a login shell (default true).
790
+ *
791
+ * AutoDL images put python/pip/conda in `/root/miniconda3/bin`, which reaches PATH
792
+ * only via the login profile. A non-interactive `ssh host "cmd"` gets a bare PATH —
793
+ * `.bashrc` bails out at the standard "If not running interactively, don't do
794
+ * anything" guard — so `pip install` there fails with exit 127. A login shell matches
795
+ * what the user sees when they `autodl ssh` in by hand.
796
+ */
797
+ loginShell?: boolean;
798
+ }
799
+ /** Run one command over an already-open connection. */
800
+ declare function execOnConnection(conn: Client, command: string, options?: ExecOptions): Promise<ExecResult>;
801
+ /** Connect (refreshing credentials as needed) and run a single command. */
802
+ declare function execCommand(client: AutoDLClient, uuid: string, command: string, options?: ExecOptions): Promise<ExecResult>;
803
+
804
+ interface TransferProgress {
805
+ file: string;
806
+ index: number;
807
+ total: number;
808
+ bytes: number;
809
+ }
810
+ interface TransferOptions extends ConnectOptions {
811
+ /** Glob-ish ignore patterns; `.autodlignore` then `.gitignore` are used by default. */
812
+ ignore?: string[];
813
+ onProgress?: (progress: TransferProgress) => void;
814
+ /** Skip reading ignore files from disk (used by tests and explicit single-file copies). */
815
+ noIgnoreFiles?: boolean;
816
+ }
817
+ interface TransferSummary {
818
+ files: number;
819
+ bytes: number;
820
+ }
821
+ /** Upload a local file or directory to the instance. */
822
+ declare function push(client: AutoDLClient, uuid: string, localPath: string, remotePath: string, options?: TransferOptions): Promise<TransferSummary>;
823
+ /** Download a remote file or directory from the instance. */
824
+ declare function pull(client: AutoDLClient, uuid: string, remotePath: string, localPath: string, options?: TransferOptions): Promise<TransferSummary>;
825
+
826
+ declare const VERSION: string;
827
+
828
+ /**
829
+ * Deploy a hosted git project onto an AutoDL instance.
830
+ *
831
+ * The defining difference from `runWorkflow`: this **stops** the instance at the end
832
+ * rather than releasing it. A stopped container instance keeps its disks, so the next
833
+ * deploy can power the same box back on and `git pull` instead of rebuilding from
834
+ * scratch. Releasing would throw all of that away.
835
+ */
836
+ interface DeployOptions {
837
+ repo: string;
838
+ branch?: string;
839
+ /** Reuse an existing instance instead of creating one. */
840
+ instanceUuid?: string;
841
+ gpu?: string;
842
+ gpuNum?: number;
843
+ image?: string;
844
+ regions?: string[];
845
+ diskGb?: number;
846
+ name?: string;
847
+ /** Remote checkout directory. Defaults to /root/autodl-tmp/<repo>. */
848
+ dir?: string;
849
+ /** Overrides dependency auto-detection. */
850
+ setup?: string;
851
+ noSetup?: boolean;
852
+ /** Command that runs the project once dependencies are in place. */
853
+ start?: string;
854
+ /** Background the start command and return immediately. */
855
+ detach?: boolean;
856
+ gitToken?: string;
857
+ /** Disable `source /etc/network_turbo` even for accelerated hosts. */
858
+ noAcceleration?: boolean;
859
+ ttlSeconds: number;
860
+ onFinish?: "poweroff" | "release" | "keep";
861
+ commandTimeoutMs?: number;
862
+ minBalanceYuan?: number;
863
+ stockCheck?: boolean;
864
+ env?: Record<string, string>;
865
+ signal?: AbortSignal;
866
+ }
867
+ interface DeployResult {
868
+ instanceUuid: string;
869
+ /** True when this call created the instance rather than reusing one. */
870
+ created: boolean;
871
+ repo: {
872
+ host: string;
873
+ path: string;
874
+ name: string;
875
+ branch: string | null;
876
+ };
877
+ dir: string;
878
+ setupCommand: string | null;
879
+ startCommand: string | null;
880
+ detached: boolean;
881
+ exitCode: number | null;
882
+ stdout: string;
883
+ stderr: string;
884
+ /** How to reach a detached service. */
885
+ access: {
886
+ publicUrls: string[];
887
+ tunnelHint: string | null;
888
+ logHint: string | null;
889
+ };
890
+ finalAction: "poweroff" | "release" | "keep";
891
+ durationMs: number;
892
+ }
893
+ declare function deployWorkflow(client: AutoDLClient, options: DeployOptions): Promise<DeployResult>;
894
+
895
+ interface RunOptions {
896
+ command: string;
897
+ gpu: string;
898
+ gpuNum?: number;
899
+ image?: string;
900
+ cudaFrom?: string | number;
901
+ regions?: string[];
902
+ diskGb?: number;
903
+ name?: string;
904
+ ttlSeconds: number;
905
+ /** Local directory uploaded before the command runs. */
906
+ sync?: string;
907
+ /** Remote working directory; also the sync destination. */
908
+ workdir?: string;
909
+ /** Remote path copied back after the command finishes. */
910
+ pullFrom?: string;
911
+ /** Local destination for `pullFrom`. */
912
+ pullTo?: string;
913
+ /** What to do with the instance afterwards. */
914
+ onFinish?: "poweroff" | "release" | "keep";
915
+ /** Kill the remote command after this many ms. */
916
+ commandTimeoutMs?: number;
917
+ minBalanceYuan?: number;
918
+ env?: Record<string, string>;
919
+ /** Set false to skip the pre-create stock lookup. */
920
+ stockCheck?: boolean;
921
+ signal?: AbortSignal;
922
+ }
923
+ interface RunResult {
924
+ instanceUuid: string;
925
+ exitCode: number | null;
926
+ stdout: string;
927
+ stderr: string;
928
+ uploaded: {
929
+ files: number;
930
+ bytes: number;
931
+ } | null;
932
+ downloaded: {
933
+ files: number;
934
+ bytes: number;
935
+ } | null;
936
+ finalAction: "poweroff" | "release" | "keep";
937
+ durationMs: number;
938
+ }
939
+ /**
940
+ * The single verb an agent reaches for: rent a box, put code on it, run something,
941
+ * bring the results back, and give the box back.
942
+ *
943
+ * Every exit path — success, remote failure, Ctrl-C, a thrown error — routes through
944
+ * the same cleanup, because the one unacceptable outcome is leaving a GPU powered on.
945
+ */
946
+ declare function runWorkflow(client: AutoDLClient, options: RunOptions): Promise<RunResult>;
947
+
948
+ export { AuthError, AutoDLClient, AutoDLError, BASE_IMAGES, type Balance, type BaseImage, BudgetError, type ClientOptions, type Context, type CreateInstanceInput, DEFAULT_BASE_IMAGE, DEFAULT_BASE_URL, type DeployOptions, type DeployResult, type ErrorCode, type ExecOptions, type ExecResult, ExitCode, GPU_SPECS, type GlobalOptions, type GpuSpec, type GpuStockEntry, type IdleOptions, type IdleResult, type Instance, type InstanceSnapshot, type InstanceStatus, NoStockError, NotFoundError, type Pagination, type ParsedRepo, type PrivateImage, REGIONS, type Region, type RegionChoice, type RegionStock, type RunOptions, type RunResult, type SSHCredentials, SSHError, type ServiceEndpoint, type StockQuery, type StockSnapshot, type StoredConfig, TimeoutError, type TransferOptions, type TransferSummary, UsageError, VERSION, type WaitOptions, armTTLOverSSH, assertBudget, assertStockRegion, buildServer, buildTTLSnippet, chooseRegions, composeStartCommand, connectInteractive, createContext, createInstance, deployWorkflow, disarmTTLOverSSH, estimateCost, execCommand, execOnConnection, findBaseImage, findInstance, findRegionsWithStock, formatCudaVersion, formatDuration, formatRate, formatSSHCommand, formatYuan, getBalance, getCredentials, getInstanceSnapshot, getInstanceStatus, getRegionGpuStock, getStockByRegion, listAllInstances, listInstancesPage, listPrivateImages, milliToYuan, normalizeBalance, normalizeInstance, normalizeSnapshot, parseCudaVersion, parseDuration, parseRepo, parseUtilisation, powerOffInstance, powerOnInstance, pull, push, readConfig, recordTTL, redactCredentials, redactSnapshot, redactToken, releaseInstance, resolveGitToken, resolveGpuSpec, resolveMinBalance, resolveRegion, resolveToken, runWorkflow, saveImage, setNfsMount, specForStockName, startMcpServer, sweepExpired, toAutoDLError, updateConfig, waitForRunning, waitForShutdown, waitForStatus, watchIdle, withCredentials, withSSH, writeConfig, yuanToMilli };