bitfab 0.62.5 → 0.62.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/lib/bitfab/cloudReplay.py +501 -82
- data/lib/bitfab/cloud_replay_cli.rb +1 -1
- data/lib/bitfab/replay_registry.rb +16 -0
- data/lib/bitfab/version.rb +1 -1
- metadata +1 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: b300b5a65bacd9fd60231bcab3ba96751c111ae7e5f12075a6149c163927cf65
|
|
4
|
+
data.tar.gz: b3b61e0146d500bebc4719ca485e42e175bd0e8f78053d3fa451bb256d494407
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: f16224c7b475fdfea0e847780b18d7e503b100250ec66523382a3f1bfa45e8b60281d52042de6dd0ac5f5b90bfea4a5585481fdaaa94500d019ce346ec81cb1d
|
|
7
|
+
data.tar.gz: 836b9a8bdcfd24612213eefef798713e758b337b283ee1ff89ee62b1d17e60f12e8ba883b3532cd2a47430ee3968637e96375f9912f3c35d7df8290f9a7f460d
|
data/lib/bitfab/cloudReplay.py
CHANGED
|
@@ -47,11 +47,55 @@ SHA = re.compile(r"^[0-9a-f]{40}$")
|
|
|
47
47
|
PIPELINE = re.compile(r"^[\w.][\w.-]*$")
|
|
48
48
|
# Every SDK's replay prints this to stderr as soon as the server creates the experiment.
|
|
49
49
|
EXPERIMENT_LINE = re.compile(rb"^\[replay\] Experiment ([0-9a-f-]{36}):")
|
|
50
|
+
ITEM_ID_FIELDS = ("originalTraceId", "original_trace_id", "traceId", "trace_id")
|
|
51
|
+
ITEM_ERROR_LINES = 20
|
|
52
|
+
ITEM_ERROR_LENGTH = 500
|
|
53
|
+
OUTPUT_TAIL_LENGTH = 4000
|
|
54
|
+
ENV_NAME = re.compile(r"[A-Za-z_][A-Za-z0-9_]*")
|
|
55
|
+
CLOUD_VALUE_FLAGS = (
|
|
56
|
+
"--registry",
|
|
57
|
+
"--trace-ids",
|
|
58
|
+
"--max-concurrency",
|
|
59
|
+
"--cloud-status",
|
|
60
|
+
"--cloud-watch",
|
|
61
|
+
"--cloud-cancel",
|
|
62
|
+
"--cloud-cleanup",
|
|
63
|
+
"--cloud-request-id",
|
|
64
|
+
"--cloud-include",
|
|
65
|
+
"--cloud-timeout",
|
|
66
|
+
)
|
|
67
|
+
CLOUD_SWITCHES = (
|
|
68
|
+
"--cloud",
|
|
69
|
+
"--cloud-dry-run",
|
|
70
|
+
"--cloud-detach",
|
|
71
|
+
"--cloud-check",
|
|
72
|
+
"--fail-on-error",
|
|
73
|
+
"-h",
|
|
74
|
+
"--help",
|
|
75
|
+
)
|
|
76
|
+
RUNNER_OWNED_FLAGS = {
|
|
77
|
+
"--dry-run": "use --cloud-dry-run to review the snapshot or --cloud-check to resolve traces on the runner",
|
|
78
|
+
"--seed": "seed locally",
|
|
79
|
+
"--cases": "seed locally",
|
|
80
|
+
"--from-trace": "seed locally",
|
|
81
|
+
"--run": "seed locally",
|
|
82
|
+
}
|
|
83
|
+
PATH_FLAGS = ("--params", "--code-change")
|
|
84
|
+
SELECTION_FLAGS = ("--dataset-ids", "--dataset-id", "--resume")
|
|
50
85
|
HELP = """GitHub cloud replay (requires git, gh login, and Python 3.10+).
|
|
51
86
|
--cloud PIPELINE --trace-ids UUID[,UUID] [--registry PATH]
|
|
52
87
|
[--max-concurrency 1..32] [--cloud-request-id UUID]
|
|
53
88
|
[--cloud-include PATH ...] [--cloud-dry-run] [--cloud-detach]
|
|
54
|
-
[--cloud-timeout MINUTES]
|
|
89
|
+
[--cloud-timeout MINUTES] [--cloud-check] [--fail-on-error] [REPLAY OPTIONS]
|
|
90
|
+
Every other replay option after the pipeline, such as --name, --dataset-ids,
|
|
91
|
+
--attempts, --mock, or --resume, is passed to the replay on the runner as
|
|
92
|
+
given. --params and --code-change files must be in the snapshot.
|
|
93
|
+
--registry comes from .bitfab/cloud.json; --dry-run and seeding stay local.
|
|
94
|
+
Select traces with --trace-ids (1..100 UUIDs), --dataset-ids, or --resume.
|
|
95
|
+
--fail-on-error exits 1 locally when any replayed item errored.
|
|
96
|
+
--cloud-check runs on GitHub without replaying anything: it checks that every
|
|
97
|
+
configured secret has a value, runs cloud.json's checkCommand when set, and
|
|
98
|
+
resolves the traces with the replay's --dry-run to load the registry.
|
|
55
99
|
--cloud-status UUID | --cloud-watch UUID | --cloud-cancel UUID
|
|
56
100
|
--cloud-cleanup UUID
|
|
57
101
|
--cloud-init [--config SPEC] (creates a setup, or brings an existing one up to date)
|
|
@@ -64,6 +108,12 @@ continue on GitHub; watch/status/cleanup can recover them using the printed UUID
|
|
|
64
108
|
"""
|
|
65
109
|
|
|
66
110
|
|
|
111
|
+
class CommandError(RuntimeError):
|
|
112
|
+
def __init__(self, message, http_status=None):
|
|
113
|
+
super().__init__(message)
|
|
114
|
+
self.http_status = http_status
|
|
115
|
+
|
|
116
|
+
|
|
67
117
|
def command(args, *, cwd=None, env=None, timeout=60, input=None):
|
|
68
118
|
result = subprocess.run(
|
|
69
119
|
args,
|
|
@@ -77,8 +127,13 @@ def command(args, *, cwd=None, env=None, timeout=60, input=None):
|
|
|
77
127
|
)
|
|
78
128
|
if result.returncode:
|
|
79
129
|
# Child errors can contain credential-bearing URLs or application output.
|
|
80
|
-
|
|
81
|
-
|
|
130
|
+
status = re.search(r"\bHTTP (\d{3})\b", result.stderr or "")
|
|
131
|
+
http_status = int(status[1]) if status else None
|
|
132
|
+
raise CommandError(
|
|
133
|
+
f"{args[0]} {args[1]} failed (exit {result.returncode}"
|
|
134
|
+
+ (f", HTTP {http_status}" if http_status else "")
|
|
135
|
+
+ "); check authentication and permissions",
|
|
136
|
+
http_status,
|
|
82
137
|
)
|
|
83
138
|
return result.stdout.strip()
|
|
84
139
|
|
|
@@ -213,10 +268,10 @@ def configure_secrets(argv):
|
|
|
213
268
|
args = parser.parse_args(argv)
|
|
214
269
|
root = root_directory()
|
|
215
270
|
config = configuration(root)
|
|
216
|
-
|
|
217
|
-
names = args.names or
|
|
218
|
-
if not all(
|
|
219
|
-
raise ValueError("Secret names must be
|
|
271
|
+
targets = secret_targets(config)
|
|
272
|
+
names = args.names or list(targets) or ["BITFAB_API_KEY"]
|
|
273
|
+
if not all(ENV_NAME.fullmatch(name) for name in names):
|
|
274
|
+
raise ValueError("Secret names must be environment variable names")
|
|
220
275
|
values = {}
|
|
221
276
|
for path in args.env_file:
|
|
222
277
|
for key, value in parse_environment_file(
|
|
@@ -228,14 +283,15 @@ def configure_secrets(argv):
|
|
|
228
283
|
missing = []
|
|
229
284
|
empty = []
|
|
230
285
|
for name in names:
|
|
231
|
-
|
|
286
|
+
target = targets.get(name, config.get("secretPrefix", "") + name)
|
|
287
|
+
renamed = "secret" in config.get("env", {}).get(name, {})
|
|
288
|
+
value = values.get(target, values.get(name)) if renamed else values.get(name)
|
|
232
289
|
if value is None:
|
|
233
290
|
missing.append(name)
|
|
234
291
|
continue
|
|
235
292
|
if not value:
|
|
236
293
|
empty.append(name)
|
|
237
294
|
continue
|
|
238
|
-
target = prefix + name
|
|
239
295
|
if not args.dry_run:
|
|
240
296
|
command(
|
|
241
297
|
[
|
|
@@ -261,6 +317,48 @@ def configure_secrets(argv):
|
|
|
261
317
|
}
|
|
262
318
|
|
|
263
319
|
|
|
320
|
+
def secret_targets(config):
|
|
321
|
+
prefix = config.get("secretPrefix", "")
|
|
322
|
+
targets = {name: prefix + name for name in config.get("secrets", [])}
|
|
323
|
+
for name, source in config.get("env", {}).items():
|
|
324
|
+
if "secret" in source:
|
|
325
|
+
targets[name] = source["secret"]
|
|
326
|
+
return targets
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
def runner_env(config):
|
|
330
|
+
env = {
|
|
331
|
+
name: "${{ secrets." + target + " }}"
|
|
332
|
+
for name, target in secret_targets(config).items()
|
|
333
|
+
}
|
|
334
|
+
for name, source in config.get("env", {}).items():
|
|
335
|
+
if "variable" in source:
|
|
336
|
+
env[name] = "${{ vars." + source["variable"] + " }}"
|
|
337
|
+
return env
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
def validate_env(mapping, secrets):
|
|
341
|
+
if not isinstance(mapping, dict):
|
|
342
|
+
raise ValueError(
|
|
343
|
+
'env must map environment variable names to {"secret": NAME} or {"variable": NAME}'
|
|
344
|
+
)
|
|
345
|
+
for name, source in mapping.items():
|
|
346
|
+
if not ENV_NAME.fullmatch(name) or name in RUNNER_ENV:
|
|
347
|
+
raise ValueError(f"env cannot set {name}")
|
|
348
|
+
if name in secrets:
|
|
349
|
+
raise ValueError(f"{name} is in both secrets and env; list it once")
|
|
350
|
+
if (
|
|
351
|
+
not isinstance(source, dict)
|
|
352
|
+
or len(source) != 1
|
|
353
|
+
or next(iter(source)) not in ("secret", "variable")
|
|
354
|
+
or not isinstance(next(iter(source.values())), str)
|
|
355
|
+
or not ENV_NAME.fullmatch(next(iter(source.values())))
|
|
356
|
+
):
|
|
357
|
+
raise ValueError(
|
|
358
|
+
f'env.{name} must be {{"secret": "GITHUB_SECRET_NAME"}} or {{"variable": "GITHUB_VARIABLE_NAME"}}'
|
|
359
|
+
)
|
|
360
|
+
|
|
361
|
+
|
|
264
362
|
def validate_config(config):
|
|
265
363
|
if config.get("version") != 1 or config.get("provider") != "github":
|
|
266
364
|
raise ValueError("Run bitfab:setup cloud to configure the GitHub provider")
|
|
@@ -304,6 +402,18 @@ def validate_config(config):
|
|
|
304
402
|
for name in names
|
|
305
403
|
):
|
|
306
404
|
raise ValueError("secrets must be uppercase environment variable names")
|
|
405
|
+
validate_env(config.get("env", {}), names)
|
|
406
|
+
if "checkCommand" in config:
|
|
407
|
+
check = config["checkCommand"]
|
|
408
|
+
if (
|
|
409
|
+
not isinstance(check, list)
|
|
410
|
+
or not check
|
|
411
|
+
or not all(isinstance(v, str) and v and "\x00" not in v for v in check)
|
|
412
|
+
or any(v.startswith("--cloud") for v in check)
|
|
413
|
+
):
|
|
414
|
+
raise ValueError(
|
|
415
|
+
"checkCommand must be a nonempty JSON argument array that runs locally"
|
|
416
|
+
)
|
|
307
417
|
|
|
308
418
|
|
|
309
419
|
def validate_cli_command(cli_command):
|
|
@@ -355,7 +465,53 @@ def api(repo, suffix, *, method="GET", payload=None):
|
|
|
355
465
|
return json.loads(result) if result else None
|
|
356
466
|
|
|
357
467
|
|
|
468
|
+
def split_replay_options(argv):
|
|
469
|
+
cloud, options = [], []
|
|
470
|
+
index = 0
|
|
471
|
+
while index < len(argv):
|
|
472
|
+
token = argv[index]
|
|
473
|
+
name = token.split("=", 1)[0]
|
|
474
|
+
if name in CLOUD_VALUE_FLAGS:
|
|
475
|
+
step = 1 if "=" in token else 2
|
|
476
|
+
cloud += argv[index : index + step]
|
|
477
|
+
index += step
|
|
478
|
+
elif name in CLOUD_SWITCHES:
|
|
479
|
+
cloud.append(token)
|
|
480
|
+
index += 1
|
|
481
|
+
elif token.startswith("-"):
|
|
482
|
+
if name in RUNNER_OWNED_FLAGS:
|
|
483
|
+
raise ValueError(
|
|
484
|
+
f"{name} is not sent to the runner; {RUNNER_OWNED_FLAGS[name]}"
|
|
485
|
+
)
|
|
486
|
+
options += token.split("=", 1) if token.startswith("--") else [token]
|
|
487
|
+
index += 1
|
|
488
|
+
following = argv[index] if index < len(argv) else None
|
|
489
|
+
if (
|
|
490
|
+
"=" not in token
|
|
491
|
+
and following is not None
|
|
492
|
+
and not following.startswith("-")
|
|
493
|
+
):
|
|
494
|
+
options.append(following)
|
|
495
|
+
index += 1
|
|
496
|
+
elif options:
|
|
497
|
+
raise ValueError(
|
|
498
|
+
"Put the pipeline right after --cloud, before replay options"
|
|
499
|
+
)
|
|
500
|
+
else:
|
|
501
|
+
cloud.append(token)
|
|
502
|
+
index += 1
|
|
503
|
+
for value in options:
|
|
504
|
+
if not value or "\x00" in value or "\n" in value or len(value) > 1000:
|
|
505
|
+
raise ValueError(
|
|
506
|
+
"Replay options must be single-line values of at most 1000 characters"
|
|
507
|
+
)
|
|
508
|
+
if sum(len(value) for value in options) > 16000:
|
|
509
|
+
raise ValueError("Replay options exceed 16000 characters")
|
|
510
|
+
return cloud, options
|
|
511
|
+
|
|
512
|
+
|
|
358
513
|
def parse(argv):
|
|
514
|
+
argv, options = split_replay_options(argv)
|
|
359
515
|
flags = [value.split("=", 1)[0] for value in argv if value.startswith("--")]
|
|
360
516
|
if any(flags.count(flag) > 1 for flag in flags if flag != "--cloud-include"):
|
|
361
517
|
raise ValueError("Duplicate cloud option")
|
|
@@ -376,25 +532,35 @@ def parse(argv):
|
|
|
376
532
|
parser.add_argument("--cloud-dry-run", action="store_true")
|
|
377
533
|
parser.add_argument("--cloud-detach", action="store_true")
|
|
378
534
|
parser.add_argument("--cloud-timeout", type=int)
|
|
535
|
+
parser.add_argument("--cloud-check", action="store_true")
|
|
536
|
+
parser.add_argument("--fail-on-error", action="store_true")
|
|
379
537
|
args = parser.parse_args(argv)
|
|
538
|
+
args.options = options
|
|
380
539
|
operations = [
|
|
381
540
|
name
|
|
382
541
|
for name in ("status", "watch", "cancel", "cleanup")
|
|
383
542
|
if getattr(args, "cloud_" + name)
|
|
384
543
|
]
|
|
385
544
|
if operations:
|
|
386
|
-
if len(operations) != 1 or len(argv) != 2:
|
|
545
|
+
if len(operations) != 1 or len(argv) != 2 or options:
|
|
387
546
|
raise ValueError("Cloud lifecycle commands take only their execution UUID")
|
|
388
547
|
operation = operations[0]
|
|
389
548
|
execution_id = getattr(args, "cloud_" + operation)
|
|
390
549
|
else:
|
|
391
|
-
if not args.cloud or not args.pipeline
|
|
550
|
+
if not args.cloud or not args.pipeline:
|
|
392
551
|
raise ValueError(HELP)
|
|
552
|
+
if not args.trace_ids and not any(flag in options for flag in SELECTION_FLAGS):
|
|
553
|
+
raise ValueError(
|
|
554
|
+
"Select traces with --trace-ids, --dataset-ids, or --resume"
|
|
555
|
+
)
|
|
393
556
|
operation = "submit"
|
|
394
557
|
execution_id = args.cloud_request_id or str(uuid.uuid4())
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
558
|
+
if args.trace_ids is not None:
|
|
559
|
+
traces = args.trace_ids.split(",")
|
|
560
|
+
if not 1 <= len(traces) <= 100 or not all(
|
|
561
|
+
UUID.fullmatch(t) for t in traces
|
|
562
|
+
):
|
|
563
|
+
raise ValueError("Supply 1..100 explicit trace UUIDs")
|
|
398
564
|
if not 1 <= args.max_concurrency <= 32:
|
|
399
565
|
raise ValueError("--max-concurrency must be 1..32")
|
|
400
566
|
# Self-hosted runners allow jobs of up to 5 days; GitHub-hosted ones stop at 6 hours.
|
|
@@ -405,6 +571,28 @@ def parse(argv):
|
|
|
405
571
|
return args, operation, execution_id
|
|
406
572
|
|
|
407
573
|
|
|
574
|
+
def option_paths(options):
|
|
575
|
+
return [value for flag, value in zip(options, options[1:]) if flag in PATH_FLAGS]
|
|
576
|
+
|
|
577
|
+
|
|
578
|
+
def map_option_paths(options, convert):
|
|
579
|
+
mapped = list(options)
|
|
580
|
+
for index, flag in enumerate(options[:-1]):
|
|
581
|
+
if flag in PATH_FLAGS:
|
|
582
|
+
mapped[index + 1] = convert(options[index + 1])
|
|
583
|
+
return mapped
|
|
584
|
+
|
|
585
|
+
|
|
586
|
+
def repository_path(root, value):
|
|
587
|
+
try:
|
|
588
|
+
path = Path(value).resolve().relative_to(root).as_posix()
|
|
589
|
+
except ValueError as error:
|
|
590
|
+
raise ValueError(f"{value} must be inside the repository") from error
|
|
591
|
+
if sensitive(path):
|
|
592
|
+
raise ValueError(f"Refusing credential-like file: {value}")
|
|
593
|
+
return path
|
|
594
|
+
|
|
595
|
+
|
|
408
596
|
def sensitive(path):
|
|
409
597
|
parts = Path(path).parts
|
|
410
598
|
name = Path(path).name.lower()
|
|
@@ -428,7 +616,7 @@ def sensitive(path):
|
|
|
428
616
|
)
|
|
429
617
|
|
|
430
618
|
|
|
431
|
-
def snapshot(root, config, args, execution_id):
|
|
619
|
+
def snapshot(root, config, args, execution_id, files=()):
|
|
432
620
|
if git(root, "ls-files", "-u"):
|
|
433
621
|
raise ValueError("Resolve merge conflicts before snapshotting")
|
|
434
622
|
head = git(root, "rev-parse", "HEAD")
|
|
@@ -465,7 +653,7 @@ def snapshot(root, config, args, execution_id):
|
|
|
465
653
|
raise ValueError(
|
|
466
654
|
f"Refusing credential-like tracked file in snapshot: {path}"
|
|
467
655
|
)
|
|
468
|
-
required = [CONFIG, f".github/workflows/{config['workflow']}"]
|
|
656
|
+
required = [CONFIG, f".github/workflows/{config['workflow']}", *files]
|
|
469
657
|
if config.get("registry") is not None:
|
|
470
658
|
required.append(config["registry"])
|
|
471
659
|
tracked = set(
|
|
@@ -585,6 +773,7 @@ def status(record, *, fetch_result=True):
|
|
|
585
773
|
fetch_result
|
|
586
774
|
and record["state"] == "completed"
|
|
587
775
|
and not record.get("testRunId")
|
|
776
|
+
and not record.get("check")
|
|
588
777
|
and not record.get("resultChecked")
|
|
589
778
|
):
|
|
590
779
|
success = record["conclusion"] == "success"
|
|
@@ -599,9 +788,18 @@ def status(record, *, fetch_result=True):
|
|
|
599
788
|
if (
|
|
600
789
|
result.get("executionId") != record["id"]
|
|
601
790
|
or result.get("commitSha") != record["sha"]
|
|
602
|
-
or not UUID.fullmatch(result.get("testRunId", ""))
|
|
603
791
|
):
|
|
604
792
|
raise ValueError("Replay result does not match this execution")
|
|
793
|
+
if record["request"].get("check"):
|
|
794
|
+
if result.get("check") != "passed" or not isinstance(
|
|
795
|
+
result.get("resolved"), int
|
|
796
|
+
):
|
|
797
|
+
raise ValueError("Cloud check result does not match this execution")
|
|
798
|
+
record["check"] = "passed"
|
|
799
|
+
record["resolved"] = result["resolved"]
|
|
800
|
+
return record
|
|
801
|
+
if not UUID.fullmatch(result.get("testRunId", "")):
|
|
802
|
+
raise ValueError("Replay result does not match this execution")
|
|
605
803
|
counts = replay_counts(result)
|
|
606
804
|
record["testRunId"] = result["testRunId"]
|
|
607
805
|
if isinstance(result.get("stoppedEarly"), str):
|
|
@@ -688,16 +886,35 @@ def cleanup(root, record):
|
|
|
688
886
|
record["cleaned"] = True
|
|
689
887
|
|
|
690
888
|
|
|
889
|
+
def http_status(error):
|
|
890
|
+
status = getattr(error, "http_status", None)
|
|
891
|
+
return f" (HTTP {status})" if status else ""
|
|
892
|
+
|
|
893
|
+
|
|
691
894
|
def preflight(repo, config):
|
|
692
|
-
|
|
693
|
-
|
|
895
|
+
try:
|
|
896
|
+
command(["gh", "auth", "status", "--hostname", "github.com"])
|
|
897
|
+
except RuntimeError as error:
|
|
898
|
+
raise ValueError(
|
|
899
|
+
"gh is not logged in to github.com; run gh auth login"
|
|
900
|
+
) from error
|
|
901
|
+
try:
|
|
902
|
+
api(repo, "")
|
|
903
|
+
except RuntimeError as error:
|
|
904
|
+
raise ValueError(
|
|
905
|
+
f"The gh account cannot read {repo}{http_status(error)}; log in with an account that has write access to it"
|
|
906
|
+
) from error
|
|
907
|
+
path = f".github/workflows/{config['workflow']}"
|
|
908
|
+
try:
|
|
909
|
+
workflow = api(repo, f"actions/workflows/{config['workflow']}")
|
|
910
|
+
except RuntimeError as error:
|
|
911
|
+
raise ValueError(
|
|
912
|
+
f"GitHub Actions has not registered {path}{http_status(error)}; merging it to the default branch once registers it, and after that each replay runs the copy in its own snapshot"
|
|
913
|
+
) from error
|
|
694
914
|
if workflow.get("state") != "active":
|
|
695
|
-
raise ValueError(
|
|
696
|
-
|
|
697
|
-
|
|
698
|
-
f"contents/.github/workflows/{config['workflow']}?"
|
|
699
|
-
+ urlencode({"ref": info["default_branch"]}),
|
|
700
|
-
)
|
|
915
|
+
raise ValueError(
|
|
916
|
+
f"{path} is {workflow.get('state')}; enable it in the repository's Actions tab"
|
|
917
|
+
)
|
|
701
918
|
|
|
702
919
|
|
|
703
920
|
def run_cli(argv):
|
|
@@ -717,19 +934,27 @@ def run_cli(argv):
|
|
|
717
934
|
registry = Path(args.registry).resolve().relative_to(root).as_posix()
|
|
718
935
|
if registry != config.get("registry"):
|
|
719
936
|
raise ValueError("Registry does not match .bitfab/cloud.json")
|
|
937
|
+
options = map_option_paths(
|
|
938
|
+
args.options, lambda value: repository_path(root, value)
|
|
939
|
+
)
|
|
720
940
|
request = {
|
|
721
941
|
"id": execution_id,
|
|
722
942
|
"pipeline": args.pipeline,
|
|
723
|
-
"traceIds": args.trace_ids.split(","),
|
|
724
943
|
"maxConcurrency": args.max_concurrency,
|
|
725
944
|
}
|
|
945
|
+
if args.trace_ids is not None:
|
|
946
|
+
request["traceIds"] = args.trace_ids.split(",")
|
|
947
|
+
if options:
|
|
948
|
+
request["options"] = options
|
|
949
|
+
if args.cloud_check:
|
|
950
|
+
request["check"] = True
|
|
726
951
|
if args.cloud_timeout is not None:
|
|
727
952
|
request["timeoutMinutes"] = args.cloud_timeout
|
|
728
953
|
if args.cloud_dry_run:
|
|
729
954
|
return {
|
|
730
955
|
"dryRun": True,
|
|
731
956
|
"repository": repo,
|
|
732
|
-
**snapshot(root, config, args, execution_id),
|
|
957
|
+
**snapshot(root, config, args, execution_id, option_paths(options)),
|
|
733
958
|
}
|
|
734
959
|
if path.exists():
|
|
735
960
|
record = json.loads(path.read_text())
|
|
@@ -747,9 +972,10 @@ def run_cli(argv):
|
|
|
747
972
|
"Submission stopped before dispatch. Use --cloud-cleanup, then submit a new execution UUID"
|
|
748
973
|
)
|
|
749
974
|
else:
|
|
750
|
-
command(["gh", "auth", "status", "--hostname", "github.com"])
|
|
751
975
|
preflight(repo, config)
|
|
752
|
-
source = snapshot(
|
|
976
|
+
source = snapshot(
|
|
977
|
+
root, config, args, execution_id, option_paths(options)
|
|
978
|
+
)
|
|
753
979
|
record = {
|
|
754
980
|
"id": execution_id,
|
|
755
981
|
"repository": repo,
|
|
@@ -759,6 +985,8 @@ def run_cli(argv):
|
|
|
759
985
|
"state": "prepared",
|
|
760
986
|
**source,
|
|
761
987
|
}
|
|
988
|
+
if args.fail_on_error:
|
|
989
|
+
record["failOnError"] = True
|
|
762
990
|
save(path, record)
|
|
763
991
|
print(
|
|
764
992
|
f"Cloud execution {execution_id}. Recover with --cloud-status {execution_id}",
|
|
@@ -766,26 +994,39 @@ def run_cli(argv):
|
|
|
766
994
|
flush=True,
|
|
767
995
|
)
|
|
768
996
|
ref = "refs/heads/" + record["branch"]
|
|
769
|
-
|
|
770
|
-
|
|
771
|
-
|
|
772
|
-
|
|
773
|
-
|
|
774
|
-
|
|
775
|
-
|
|
776
|
-
|
|
997
|
+
try:
|
|
998
|
+
# Empty lease asserts that the temporary remote branch does not exist.
|
|
999
|
+
git(
|
|
1000
|
+
root,
|
|
1001
|
+
"push",
|
|
1002
|
+
f"--force-with-lease={ref}:",
|
|
1003
|
+
"origin",
|
|
1004
|
+
f"{record['sha']}:{ref}",
|
|
1005
|
+
)
|
|
1006
|
+
except RuntimeError as error:
|
|
1007
|
+
raise ValueError(
|
|
1008
|
+
f"Pushing the snapshot branch {record['branch']} to origin failed; check that git can push to {repo}"
|
|
1009
|
+
) from error
|
|
777
1010
|
record["state"] = "dispatch_unknown"
|
|
778
1011
|
save(path, record)
|
|
779
1012
|
encoded = base64.b64encode(json.dumps(request).encode()).decode()
|
|
780
|
-
|
|
781
|
-
|
|
782
|
-
|
|
783
|
-
|
|
784
|
-
|
|
785
|
-
|
|
786
|
-
|
|
787
|
-
|
|
788
|
-
|
|
1013
|
+
try:
|
|
1014
|
+
response = api(
|
|
1015
|
+
repo,
|
|
1016
|
+
f"actions/workflows/{config['workflow']}/dispatches",
|
|
1017
|
+
method="POST",
|
|
1018
|
+
payload={
|
|
1019
|
+
"ref": record["branch"],
|
|
1020
|
+
"inputs": {
|
|
1021
|
+
"execution_id": execution_id,
|
|
1022
|
+
"request": encoded,
|
|
1023
|
+
},
|
|
1024
|
+
},
|
|
1025
|
+
)
|
|
1026
|
+
except RuntimeError as error:
|
|
1027
|
+
raise ValueError(
|
|
1028
|
+
f"Dispatching {config['workflow']} on {record['branch']} failed{http_status(error)}; the gh account needs write access to Actions, and the registered workflow must accept workflow_dispatch. Check the Actions tab, then resume with --cloud-status {execution_id}"
|
|
1029
|
+
) from error
|
|
789
1030
|
if isinstance(response, dict) and response.get("workflow_run_id"):
|
|
790
1031
|
record["runId"] = response["workflow_run_id"]
|
|
791
1032
|
record["url"] = response.get("html_url")
|
|
@@ -853,30 +1094,44 @@ def execute():
|
|
|
853
1094
|
if request["id"] != os.environ["BITFAB_EXECUTION_ID"]:
|
|
854
1095
|
raise ValueError("Replay request does not match the configured execution")
|
|
855
1096
|
timeout = request.get("timeoutMinutes")
|
|
856
|
-
|
|
1097
|
+
traces = (
|
|
1098
|
+
["--trace-ids", ",".join(request["traceIds"])] if "traceIds" in request else []
|
|
1099
|
+
)
|
|
1100
|
+
options = request.get("options", [])
|
|
1101
|
+
parsed, _, _ = parse(
|
|
857
1102
|
[
|
|
858
1103
|
"--cloud",
|
|
859
1104
|
request["pipeline"],
|
|
860
|
-
|
|
861
|
-
",".join(request["traceIds"]),
|
|
1105
|
+
*traces,
|
|
862
1106
|
"--max-concurrency",
|
|
863
1107
|
str(request["maxConcurrency"]),
|
|
864
1108
|
"--cloud-request-id",
|
|
865
1109
|
request["id"],
|
|
866
1110
|
*([] if timeout is None else ["--cloud-timeout", str(timeout)]),
|
|
1111
|
+
*(["--cloud-check"] if request.get("check") else []),
|
|
1112
|
+
*options,
|
|
867
1113
|
]
|
|
868
1114
|
)
|
|
869
|
-
if
|
|
870
|
-
raise ValueError("
|
|
1115
|
+
if parsed.options != options:
|
|
1116
|
+
raise ValueError("Replay options were not in the form the SDK sends")
|
|
1117
|
+
check_secrets(config)
|
|
871
1118
|
args = [
|
|
872
1119
|
*config["command"],
|
|
873
1120
|
request["pipeline"],
|
|
874
|
-
|
|
875
|
-
",".join(request["traceIds"]),
|
|
1121
|
+
*traces,
|
|
876
1122
|
"--max-concurrency",
|
|
877
1123
|
str(request["maxConcurrency"]),
|
|
878
|
-
|
|
1124
|
+
*map_option_paths(options, lambda value: snapshot_file(root, value)),
|
|
1125
|
+
*(
|
|
1126
|
+
[]
|
|
1127
|
+
if "--code-change" in options or "--no-code-change" in options
|
|
1128
|
+
else ["--no-code-change"]
|
|
1129
|
+
),
|
|
879
1130
|
]
|
|
1131
|
+
if request.get("check"):
|
|
1132
|
+
summary = run_check(root, config, request, args)
|
|
1133
|
+
write_result(summary)
|
|
1134
|
+
return summary
|
|
880
1135
|
experiment = {}
|
|
881
1136
|
# GitHub cancels and times out a job with SIGINT then SIGTERM; turn SIGTERM into the
|
|
882
1137
|
# same interrupt so the replay is stopped cleanly and its experiment is still reported.
|
|
@@ -900,6 +1155,96 @@ def execute():
|
|
|
900
1155
|
return summary
|
|
901
1156
|
|
|
902
1157
|
|
|
1158
|
+
def snapshot_file(root, value):
|
|
1159
|
+
path = within(root, value)
|
|
1160
|
+
if sensitive(value) or not path.is_file():
|
|
1161
|
+
raise ValueError(f"{value} is not a file in the replay snapshot")
|
|
1162
|
+
return str(path)
|
|
1163
|
+
|
|
1164
|
+
|
|
1165
|
+
def check_secrets(config):
|
|
1166
|
+
targets = {"BITFAB_API_KEY": "BITFAB_API_KEY", **secret_targets(config)}
|
|
1167
|
+
empty = [name for name in targets if not os.environ.get(name)]
|
|
1168
|
+
if empty:
|
|
1169
|
+
raise ValueError(
|
|
1170
|
+
"These replay environment variables are empty on the runner: "
|
|
1171
|
+
+ ", ".join(f"{name} (secret {targets[name]})" for name in empty)
|
|
1172
|
+
+ ". GitHub passes a secret that does not exist as an empty string. Create each one under Settings, Secrets and variables, Actions, in the repository or in the job's Environment"
|
|
1173
|
+
)
|
|
1174
|
+
|
|
1175
|
+
|
|
1176
|
+
def run_check(root, config, request, args):
|
|
1177
|
+
directory = within(root, config["workingDirectory"])
|
|
1178
|
+
if config.get("checkCommand"):
|
|
1179
|
+
print(f"Running checkCommand: {shlex.join(config['checkCommand'])}", flush=True)
|
|
1180
|
+
code = subprocess.run(
|
|
1181
|
+
config["checkCommand"], cwd=directory, stdin=subprocess.DEVNULL, check=False
|
|
1182
|
+
).returncode
|
|
1183
|
+
if code:
|
|
1184
|
+
raise ValueError(f"checkCommand exited {code}; its output is above")
|
|
1185
|
+
result = run_command(root, config, [*args, "--dry-run"], None, {})
|
|
1186
|
+
items = result.get("items")
|
|
1187
|
+
if not isinstance(items, list):
|
|
1188
|
+
raise ValueError("The replay dry run did not return its resolved items")
|
|
1189
|
+
errors = item_errors(items)
|
|
1190
|
+
report_item_errors(errors)
|
|
1191
|
+
if errors:
|
|
1192
|
+
raise ValueError(
|
|
1193
|
+
f"{len(errors)} of {len(items)} traces failed to resolve; the errors are above"
|
|
1194
|
+
)
|
|
1195
|
+
return {
|
|
1196
|
+
"executionId": request["id"],
|
|
1197
|
+
"commitSha": os.environ["GITHUB_SHA"],
|
|
1198
|
+
"check": "passed",
|
|
1199
|
+
"resolved": len(items),
|
|
1200
|
+
}
|
|
1201
|
+
|
|
1202
|
+
|
|
1203
|
+
def item_errors(items):
|
|
1204
|
+
errors = []
|
|
1205
|
+
for item in items:
|
|
1206
|
+
if not item_errored(item):
|
|
1207
|
+
continue
|
|
1208
|
+
error = next(
|
|
1209
|
+
item[field] for field in ITEM_ERROR_FIELDS if item.get(field) is not None
|
|
1210
|
+
)
|
|
1211
|
+
if isinstance(error, dict):
|
|
1212
|
+
error = error.get("message") or json.dumps(error)
|
|
1213
|
+
trace = next(
|
|
1214
|
+
(
|
|
1215
|
+
item[field]
|
|
1216
|
+
for field in ITEM_ID_FIELDS
|
|
1217
|
+
if isinstance(item.get(field), str)
|
|
1218
|
+
),
|
|
1219
|
+
"unknown trace",
|
|
1220
|
+
)
|
|
1221
|
+
text = " ".join(str(error).split())
|
|
1222
|
+
if len(text) > ITEM_ERROR_LENGTH:
|
|
1223
|
+
text = text[:ITEM_ERROR_LENGTH] + "..."
|
|
1224
|
+
errors.append((trace, text))
|
|
1225
|
+
return errors
|
|
1226
|
+
|
|
1227
|
+
|
|
1228
|
+
def report_item_errors(errors):
|
|
1229
|
+
if not errors:
|
|
1230
|
+
return
|
|
1231
|
+
shown = errors[:ITEM_ERROR_LINES]
|
|
1232
|
+
lines = [f"trace {trace}: {text}" for trace, text in shown]
|
|
1233
|
+
if len(errors) > len(shown):
|
|
1234
|
+
lines.append(f"and {len(errors) - len(shown)} more errored items")
|
|
1235
|
+
print("Errored items:", file=sys.stderr)
|
|
1236
|
+
for line in lines:
|
|
1237
|
+
print(" " + line, file=sys.stderr)
|
|
1238
|
+
sys.stderr.flush()
|
|
1239
|
+
summary = os.environ.get("GITHUB_STEP_SUMMARY")
|
|
1240
|
+
if summary:
|
|
1241
|
+
with Path(summary).open("a") as file:
|
|
1242
|
+
file.write(
|
|
1243
|
+
"\n#### Errored items\n\n"
|
|
1244
|
+
+ "".join(f"- `{line.replace('`', chr(39))}`\n" for line in lines)
|
|
1245
|
+
)
|
|
1246
|
+
|
|
1247
|
+
|
|
903
1248
|
def raise_interrupt(signum, frame):
|
|
904
1249
|
raise KeyboardInterrupt
|
|
905
1250
|
|
|
@@ -919,6 +1264,12 @@ def write_result(summary):
|
|
|
919
1264
|
# annotations, so a JSON credential alone turns each { and } into ***.
|
|
920
1265
|
message = base64.b64encode(json.dumps(summary).encode()).decode()
|
|
921
1266
|
print(f"::notice title={RESULT_TITLE}::{message}", flush=True)
|
|
1267
|
+
if "check" in summary:
|
|
1268
|
+
with Path(os.environ["GITHUB_STEP_SUMMARY"]).open("a") as file:
|
|
1269
|
+
file.write(
|
|
1270
|
+
f"### Bitfab cloud check\n\nPassed: every secret has a value and {summary['resolved']} traces resolved. Commit: `{summary['commitSha']}`\n"
|
|
1271
|
+
)
|
|
1272
|
+
return
|
|
922
1273
|
lines = [
|
|
923
1274
|
f"Test run: `{summary['testRunId']}`",
|
|
924
1275
|
f"Commit: `{summary['commitSha']}`",
|
|
@@ -934,6 +1285,43 @@ def write_result(summary):
|
|
|
934
1285
|
|
|
935
1286
|
|
|
936
1287
|
def run_replay(root, config, request, args, timeout, experiment):
|
|
1288
|
+
result = run_command(root, config, args, timeout, experiment)
|
|
1289
|
+
test_run = result.get("testRunId", result.get("test_run_id"))
|
|
1290
|
+
if not isinstance(test_run, str) or not UUID.fullmatch(test_run):
|
|
1291
|
+
raise ValueError("Replay did not return a valid persisted test run UUID")
|
|
1292
|
+
items = result.get("items")
|
|
1293
|
+
if not isinstance(items, list):
|
|
1294
|
+
raise ValueError("Replay did not return its replayed items")
|
|
1295
|
+
replayed = [item for item in items if not carried_over(item)]
|
|
1296
|
+
errors = item_errors(replayed)
|
|
1297
|
+
report_item_errors(errors)
|
|
1298
|
+
return {
|
|
1299
|
+
"executionId": request["id"],
|
|
1300
|
+
"commitSha": os.environ["GITHUB_SHA"],
|
|
1301
|
+
"testRunId": test_run,
|
|
1302
|
+
"replayed": len(replayed),
|
|
1303
|
+
"errored": len(errors),
|
|
1304
|
+
}
|
|
1305
|
+
|
|
1306
|
+
|
|
1307
|
+
def carried_over(item):
|
|
1308
|
+
return isinstance(item, dict) and (
|
|
1309
|
+
item.get("carriedOver") is True or item.get("carried_over") is True
|
|
1310
|
+
)
|
|
1311
|
+
|
|
1312
|
+
|
|
1313
|
+
def print_output_tail(text):
|
|
1314
|
+
tail = text[-OUTPUT_TAIL_LENGTH:].strip()
|
|
1315
|
+
if tail:
|
|
1316
|
+
print(
|
|
1317
|
+
"Last replay output:\n"
|
|
1318
|
+
+ "\n".join(" " + line for line in tail.splitlines()),
|
|
1319
|
+
file=sys.stderr,
|
|
1320
|
+
flush=True,
|
|
1321
|
+
)
|
|
1322
|
+
|
|
1323
|
+
|
|
1324
|
+
def run_command(root, config, args, timeout, experiment):
|
|
937
1325
|
with tempfile.TemporaryFile() as output:
|
|
938
1326
|
with subprocess.Popen(
|
|
939
1327
|
args,
|
|
@@ -974,9 +1362,10 @@ def run_replay(root, config, request, args, timeout, experiment):
|
|
|
974
1362
|
if output.tell() > 16 * 1024 * 1024:
|
|
975
1363
|
raise ValueError("Replay output exceeded 16 MiB")
|
|
976
1364
|
output.seek(0)
|
|
977
|
-
text = output.read().decode()
|
|
1365
|
+
text = output.read().decode(errors="replace")
|
|
978
1366
|
if code:
|
|
979
|
-
|
|
1367
|
+
print_output_tail(text)
|
|
1368
|
+
raise ValueError(f"Replay command exited {code}; its output is above")
|
|
980
1369
|
decoder = json.JSONDecoder()
|
|
981
1370
|
result = None
|
|
982
1371
|
for offset in [0, *[i + 1 for i, value in enumerate(text) if value == "\n"]]:
|
|
@@ -987,22 +1376,10 @@ def run_replay(root, config, request, args, timeout, experiment):
|
|
|
987
1376
|
result = value
|
|
988
1377
|
except json.JSONDecodeError:
|
|
989
1378
|
pass
|
|
990
|
-
|
|
991
|
-
|
|
992
|
-
|
|
993
|
-
|
|
994
|
-
raise ValueError("Replay did not return a valid persisted test run UUID")
|
|
995
|
-
items = result.get("items")
|
|
996
|
-
if not isinstance(items, list):
|
|
997
|
-
raise ValueError("Replay did not return its replayed items")
|
|
998
|
-
summary = {
|
|
999
|
-
"executionId": request["id"],
|
|
1000
|
-
"commitSha": os.environ["GITHUB_SHA"],
|
|
1001
|
-
"testRunId": test_run,
|
|
1002
|
-
"replayed": len(items),
|
|
1003
|
-
"errored": sum(1 for item in items if item_errored(item)),
|
|
1004
|
-
}
|
|
1005
|
-
return summary
|
|
1379
|
+
if result is None:
|
|
1380
|
+
print_output_tail(text)
|
|
1381
|
+
raise ValueError("Replay did not print a JSON result; its output is above")
|
|
1382
|
+
return result
|
|
1006
1383
|
|
|
1007
1384
|
|
|
1008
1385
|
def item_errored(item):
|
|
@@ -1105,7 +1482,7 @@ def initialize(argv):
|
|
|
1105
1482
|
)
|
|
1106
1483
|
parser.add_argument(
|
|
1107
1484
|
"--config",
|
|
1108
|
-
help=
|
|
1485
|
+
help='JSON file with cloud config, cliCommand, setupSteps, optional pipeline (omit or null to allow every registry pipeline), secrets, variables, env (runner variable names mapped to {"secret": NAME} or {"variable": NAME} for renamed secrets and repository variables), checkCommand, environment, services, runsOn. Required for a new setup; for an existing one only cliCommand is read, and only when the workflow does not already show it',
|
|
1109
1486
|
)
|
|
1110
1487
|
args = parser.parse_args(argv)
|
|
1111
1488
|
spec = json.loads(Path(args.config).read_text()) if args.config else None
|
|
@@ -1134,6 +1511,15 @@ def initialize(argv):
|
|
|
1134
1511
|
secret_prefix = DEFAULT_SECRET_PREFIX
|
|
1135
1512
|
config["secrets"] = secrets
|
|
1136
1513
|
config["secretPrefix"] = secret_prefix
|
|
1514
|
+
if not isinstance(variables, list):
|
|
1515
|
+
raise ValueError("variables must be a list of names")
|
|
1516
|
+
env = dict(spec.get("env") or {})
|
|
1517
|
+
for name in variables:
|
|
1518
|
+
env.setdefault(name, {"variable": name})
|
|
1519
|
+
if env:
|
|
1520
|
+
config["env"] = env
|
|
1521
|
+
if spec.get("checkCommand") is not None:
|
|
1522
|
+
config["checkCommand"] = spec["checkCommand"]
|
|
1137
1523
|
validate_config(config)
|
|
1138
1524
|
validate_cli_command(spec.get("cliCommand"))
|
|
1139
1525
|
if set(secrets) & set(variables):
|
|
@@ -1151,10 +1537,8 @@ def initialize(argv):
|
|
|
1151
1537
|
or not all(isinstance(step, dict) for step in steps)
|
|
1152
1538
|
):
|
|
1153
1539
|
raise ValueError("setupSteps must contain reviewed GitHub Actions setup steps")
|
|
1154
|
-
env = {key: "${{ secrets." + secret_prefix + key + " }}" for key in secrets}
|
|
1155
|
-
env.update({key: "${{ vars." + key + " }}" for key in variables})
|
|
1156
1540
|
workflow = workflow_document(
|
|
1157
|
-
spec, steps, replay_step(config, spec["cliCommand"],
|
|
1541
|
+
spec, steps, replay_step(config, spec["cliCommand"], runner_env(config))
|
|
1158
1542
|
)
|
|
1159
1543
|
outputs = {
|
|
1160
1544
|
CONFIG: json.dumps(config, indent=2) + "\n",
|
|
@@ -1164,9 +1548,13 @@ def initialize(argv):
|
|
|
1164
1548
|
write_new_files(root, outputs)
|
|
1165
1549
|
return {
|
|
1166
1550
|
"files": list(outputs),
|
|
1167
|
-
"requiredSecrets":
|
|
1168
|
-
"requiredVariables":
|
|
1169
|
-
|
|
1551
|
+
"requiredSecrets": sorted(set(secret_targets(config).values())),
|
|
1552
|
+
"requiredVariables": sorted(
|
|
1553
|
+
source["variable"]
|
|
1554
|
+
for source in config.get("env", {}).values()
|
|
1555
|
+
if "variable" in source
|
|
1556
|
+
),
|
|
1557
|
+
"next": "Configure secrets securely, review push triggers, get the workflow registered with GitHub Actions (merging it to the default branch once does that), run --cloud-dry-run, then --cloud-check",
|
|
1170
1558
|
}
|
|
1171
1559
|
|
|
1172
1560
|
|
|
@@ -1225,6 +1613,12 @@ def update_existing(root, spec):
|
|
|
1225
1613
|
for key, value in (step.get("env") or {}).items()
|
|
1226
1614
|
if key not in RUNNER_ENV
|
|
1227
1615
|
}
|
|
1616
|
+
declared = runner_env(config)
|
|
1617
|
+
conflicts = sorted(
|
|
1618
|
+
name for name in declared if name in env and env[name] != declared[name]
|
|
1619
|
+
)
|
|
1620
|
+
undeclared = sorted(name for name in env if name not in declared)
|
|
1621
|
+
env = {**declared, **env}
|
|
1228
1622
|
extra = {
|
|
1229
1623
|
key: value
|
|
1230
1624
|
for key, value in step.items()
|
|
@@ -1251,11 +1645,23 @@ def update_existing(root, spec):
|
|
|
1251
1645
|
# Setups before the SDK carried the script left a copy that nothing reads now.
|
|
1252
1646
|
old_script = ".bitfab/cloudReplay.py"
|
|
1253
1647
|
removable = [old_script] if within(root, old_script).exists() else []
|
|
1648
|
+
mismatches = {}
|
|
1649
|
+
if conflicts:
|
|
1650
|
+
mismatches["conflicts"] = conflicts
|
|
1651
|
+
if undeclared:
|
|
1652
|
+
mismatches["undeclared"] = undeclared
|
|
1653
|
+
if mismatches:
|
|
1654
|
+
mismatches["mismatchNext"] = (
|
|
1655
|
+
"The Replay step sets these differently from, or in addition to, .bitfab/cloud.json. "
|
|
1656
|
+
'Describe each one in cloud.json "env" as {"secret": NAME} or {"variable": NAME} '
|
|
1657
|
+
"so --cloud-secrets and the runner's empty-secret check see it; the step was left as it is"
|
|
1658
|
+
)
|
|
1254
1659
|
if json.dumps([config, workflow], sort_keys=True) == before:
|
|
1255
1660
|
return {
|
|
1256
1661
|
"files": [],
|
|
1257
1662
|
"updated": False,
|
|
1258
1663
|
"removable": removable,
|
|
1664
|
+
**mismatches,
|
|
1259
1665
|
"next": "Already up to date",
|
|
1260
1666
|
}
|
|
1261
1667
|
outputs = {
|
|
@@ -1268,7 +1674,8 @@ def update_existing(root, spec):
|
|
|
1268
1674
|
"files": list(outputs),
|
|
1269
1675
|
"updated": True,
|
|
1270
1676
|
"removable": removable,
|
|
1271
|
-
|
|
1677
|
+
**mismatches,
|
|
1678
|
+
"next": "Review the diff, then run --cloud-dry-run; replays use the workflow in their own snapshot, so the change applies without merging",
|
|
1272
1679
|
}
|
|
1273
1680
|
|
|
1274
1681
|
|
|
@@ -1284,14 +1691,26 @@ def main():
|
|
|
1284
1691
|
result = run_cli(sys.argv[1:])
|
|
1285
1692
|
print(json.dumps(result, indent=2))
|
|
1286
1693
|
if result.get("state") == "completed" and result.get("conclusion") != "success":
|
|
1287
|
-
if result.get("
|
|
1694
|
+
if result.get("request", {}).get("check"):
|
|
1695
|
+
print(
|
|
1696
|
+
f"Cloud check failed. The Replay step log names what is missing: {result.get('url')}",
|
|
1697
|
+
file=sys.stderr,
|
|
1698
|
+
)
|
|
1699
|
+
elif result.get("testRunId"):
|
|
1288
1700
|
print(
|
|
1289
1701
|
f"Cloud replay stopped early ({result.get('stoppedEarly', result.get('conclusion'))}). Traces that finished are saved in test run {result['testRunId']}.",
|
|
1290
1702
|
file=sys.stderr,
|
|
1291
1703
|
)
|
|
1292
1704
|
return 1
|
|
1293
1705
|
if result.get("state") == "completed":
|
|
1294
|
-
|
|
1706
|
+
code = report_errored_items(result)
|
|
1707
|
+
if result.get("failOnError") and result.get("errored"):
|
|
1708
|
+
print(
|
|
1709
|
+
"Cloud replay: exiting 1 because of --fail-on-error",
|
|
1710
|
+
file=sys.stderr,
|
|
1711
|
+
)
|
|
1712
|
+
return 1
|
|
1713
|
+
return code
|
|
1295
1714
|
return 0
|
|
1296
1715
|
except KeyboardInterrupt:
|
|
1297
1716
|
print(
|
|
@@ -6,7 +6,7 @@ require "tempfile"
|
|
|
6
6
|
|
|
7
7
|
module Bitfab
|
|
8
8
|
module CloudReplayCli
|
|
9
|
-
HELP = "Direct GitHub replay: --cloud PIPELINE --trace-ids UUID[,UUID] [--registry PATH] [--max-concurrency 1..32] [--cloud-include FILE] [--cloud-dry-run] [--cloud-detach] [--cloud-request-id UUID] [--cloud-timeout MINUTES]. Lifecycle: --cloud-status|--cloud-watch|--cloud-cancel|--cloud-cleanup UUID. Setup: --cloud-init [--config FILE] (creates a setup or updates it in place), --cloud-secrets --env-file FILE [NAME ...]. Requires git, gh auth login, Python 3.10+, and bitfab:setup cloud."
|
|
9
|
+
HELP = "Direct GitHub replay: --cloud PIPELINE --trace-ids UUID[,UUID] [--registry PATH] [--max-concurrency 1..32] [--cloud-include FILE] [--cloud-dry-run] [--cloud-detach] [--cloud-request-id UUID] [--cloud-timeout MINUTES] [--cloud-check] [--fail-on-error] [replay options such as --name, --dataset-ids, --attempts, or --resume, passed to the replay on the runner]. Lifecycle: --cloud-status|--cloud-watch|--cloud-cancel|--cloud-cleanup UUID. Setup: --cloud-init [--config FILE] (creates a setup or updates it in place), --cloud-secrets --env-file FILE [NAME ...]. Requires git, gh auth login, Python 3.10+, and bitfab:setup cloud."
|
|
10
10
|
|
|
11
11
|
HELPER = File.expand_path("cloudReplay.py", __dir__)
|
|
12
12
|
|
|
@@ -89,6 +89,8 @@ module Bitfab
|
|
|
89
89
|
module ReplayCli
|
|
90
90
|
MAX_PINNED_TRACE_IDS = 100
|
|
91
91
|
|
|
92
|
+
class ItemsErroredError < StandardError; end
|
|
93
|
+
|
|
92
94
|
module_function
|
|
93
95
|
|
|
94
96
|
# get_assertions returns every state so a reviewer can see drafts, but only an
|
|
@@ -266,9 +268,20 @@ module Bitfab
|
|
|
266
268
|
render_summary(args.fetch(:pipeline), result, stderr)
|
|
267
269
|
end
|
|
268
270
|
stdout.puts Bitfab.serialize_replay_result(result)
|
|
271
|
+
fail_if_items_errored(result, dry_run: options[:dry_run]) if args[:fail_on_error]
|
|
269
272
|
result
|
|
270
273
|
end
|
|
271
274
|
|
|
275
|
+
def fail_if_items_errored(result, dry_run:)
|
|
276
|
+
items = result[:items].reject { |item| item[:carried_over] }
|
|
277
|
+
errored = items.count { |item| item[:error] }
|
|
278
|
+
return if errored.zero?
|
|
279
|
+
|
|
280
|
+
noun = dry_run ? "resolved" : "replayed"
|
|
281
|
+
raise ItemsErroredError,
|
|
282
|
+
"[replay] #{errored} of #{items.length} #{noun} items errored; exiting 1 because of --fail-on-error"
|
|
283
|
+
end
|
|
284
|
+
|
|
272
285
|
def parse(registry, argv)
|
|
273
286
|
if argv.include?("--db-branch") && argv.include?("--no-db-branch")
|
|
274
287
|
raise OptionParser::InvalidArgument, "--db-branch and --no-db-branch cannot be used together"
|
|
@@ -315,6 +328,9 @@ module Bitfab
|
|
|
315
328
|
options.on("--judge-assertions",
|
|
316
329
|
"Judge each replay's approved assertions as it finishes (costs model calls)") { args[:judge_assertions] = true }
|
|
317
330
|
options.on("--dry-run") { args[:dry_run] = true }
|
|
331
|
+
options.on("--fail-on-error",
|
|
332
|
+
"Exit 1 after printing the result when any replayed item errored. " \
|
|
333
|
+
"Under --dry-run, items whose inputs failed to resolve count.") { args[:fail_on_error] = true }
|
|
318
334
|
options.on("--primitive TYPE", %w[async process]) { |value| args[:primitive] = value }
|
|
319
335
|
options.on("--[no-]memory-throttle") { |value| args[:memory_throttle] = value }
|
|
320
336
|
options.on("--mock STRATEGY", %w[none all marked]) { |value| args[:mock] = value }
|
data/lib/bitfab/version.rb
CHANGED