calkit-python 0.47.1__py3-none-any.whl → 0.47.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- calkit/cli/check.py +1 -4
- calkit/cli/main/core.py +1 -5
- calkit/cli/new.py +157 -30
- calkit/environments.py +14 -0
- calkit/invenio.py +51 -8
- calkit/models/core.py +3 -0
- calkit/pipeline.py +21 -0
- calkit/releases.py +190 -7
- calkit/tests/cli/test_new.py +212 -0
- calkit/tests/test_environments.py +28 -0
- calkit/tests/test_invenio.py +79 -0
- calkit/tests/test_releases.py +78 -5
- {calkit_python-0.47.1.dist-info → calkit_python-0.47.2.dist-info}/METADATA +1 -1
- {calkit_python-0.47.1.dist-info → calkit_python-0.47.2.dist-info}/RECORD +31 -31
- {calkit_python-0.47.1.data → calkit_python-0.47.2.data}/data/etc/jupyter/jupyter_server_config.d/calkit.json +0 -0
- {calkit_python-0.47.1.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/package.json +0 -0
- {calkit_python-0.47.1.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/schemas/calkit/package.json.orig +0 -0
- {calkit_python-0.47.1.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/schemas/calkit/plugin.json +0 -0
- {calkit_python-0.47.1.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/502.9a2c5772a15466e923ef.js +0 -0
- {calkit_python-0.47.1.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/695.2c41003a452d43d2b358.js +0 -0
- {calkit_python-0.47.1.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/867.a42a046aa5108f54f8fb.js +0 -0
- {calkit_python-0.47.1.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/909.6d8285ce7c45878ac508.js +0 -0
- {calkit_python-0.47.1.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/946.050af2abf7845cfbdbd2.js +0 -0
- {calkit_python-0.47.1.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/946.050af2abf7845cfbdbd2.js.LICENSE.txt +0 -0
- {calkit_python-0.47.1.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/b2f1c3efe70cb539d121.png +0 -0
- {calkit_python-0.47.1.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/remoteEntry.d7a43c7948f690d37d19.js +0 -0
- {calkit_python-0.47.1.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/style.js +0 -0
- {calkit_python-0.47.1.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/third-party-licenses.json +0 -0
- {calkit_python-0.47.1.dist-info → calkit_python-0.47.2.dist-info}/WHEEL +0 -0
- {calkit_python-0.47.1.dist-info → calkit_python-0.47.2.dist-info}/entry_points.txt +0 -0
- {calkit_python-0.47.1.dist-info → calkit_python-0.47.2.dist-info}/licenses/LICENSE +0 -0
calkit/cli/check.py
CHANGED
|
@@ -1802,10 +1802,7 @@ def check_venv(
|
|
|
1802
1802
|
if verbose:
|
|
1803
1803
|
typer.echo(f"Using legacy lock file: {legacy_fpath}")
|
|
1804
1804
|
break
|
|
1805
|
-
|
|
1806
|
-
activate_cmd = f"{prefix}\\Scripts\\activate"
|
|
1807
|
-
else:
|
|
1808
|
-
activate_cmd = f". {prefix}/bin/activate"
|
|
1805
|
+
activate_cmd = calkit.environments.get_venv_activate_cmd(prefix)
|
|
1809
1806
|
|
|
1810
1807
|
def pip_install_and_freeze(reqs_arg: str) -> None:
|
|
1811
1808
|
check_cmd = (
|
calkit/cli/main/core.py
CHANGED
|
@@ -6,7 +6,6 @@ import csv
|
|
|
6
6
|
import json
|
|
7
7
|
import logging
|
|
8
8
|
import os
|
|
9
|
-
import platform as _platform
|
|
10
9
|
import posixpath
|
|
11
10
|
import shlex
|
|
12
11
|
import shutil
|
|
@@ -3543,10 +3542,7 @@ def run_in_env(
|
|
|
3543
3542
|
envs, path, env_name
|
|
3544
3543
|
)
|
|
3545
3544
|
shell_cmd = _to_shell_cmd(cmd)
|
|
3546
|
-
|
|
3547
|
-
activate_cmd = f"{prefix}\\Scripts\\activate"
|
|
3548
|
-
else:
|
|
3549
|
-
activate_cmd = f". {prefix}/bin/activate"
|
|
3545
|
+
activate_cmd = calkit.environments.get_venv_activate_cmd(prefix)
|
|
3550
3546
|
if verbose:
|
|
3551
3547
|
typer.echo(f"Raw command: {cmd}")
|
|
3552
3548
|
typer.echo(f"Shell command: {shell_cmd}")
|
calkit/cli/new.py
CHANGED
|
@@ -3434,6 +3434,18 @@ def new_release(
|
|
|
3434
3434
|
str | None,
|
|
3435
3435
|
typer.Option("--date", help="Release date. Will default to today."),
|
|
3436
3436
|
] = None,
|
|
3437
|
+
include_pipeline: Annotated[
|
|
3438
|
+
bool,
|
|
3439
|
+
typer.Option(
|
|
3440
|
+
"--pipeline",
|
|
3441
|
+
help=(
|
|
3442
|
+
"Include everything needed to reproduce the released path, "
|
|
3443
|
+
"i.e., the pipeline, its lock file, and the stages, inputs, "
|
|
3444
|
+
"and environments the path depends on. Stages unrelated to "
|
|
3445
|
+
"the path are left out."
|
|
3446
|
+
),
|
|
3447
|
+
),
|
|
3448
|
+
] = False,
|
|
3437
3449
|
no_docker_images: Annotated[
|
|
3438
3450
|
bool,
|
|
3439
3451
|
typer.Option(
|
|
@@ -3521,7 +3533,6 @@ def new_release(
|
|
|
3521
3533
|
] = False,
|
|
3522
3534
|
):
|
|
3523
3535
|
"""Create a new release."""
|
|
3524
|
-
import bibtexparser
|
|
3525
3536
|
import dotenv
|
|
3526
3537
|
|
|
3527
3538
|
import calkit.pipeline
|
|
@@ -3536,6 +3547,20 @@ def new_release(
|
|
|
3536
3547
|
repo = calkit.git.get_repo()
|
|
3537
3548
|
if name in repo.tags:
|
|
3538
3549
|
raise_error(f"Git tag with name '{name}' already exists")
|
|
3550
|
+
# A release commits to calkit.yaml and pushes the branch it's on, neither
|
|
3551
|
+
# of which works from a detached HEAD. Check before anything is uploaded,
|
|
3552
|
+
# so a release can't get published and then fail on the way out.
|
|
3553
|
+
will_push = (
|
|
3554
|
+
not dry_run and not no_push and not no_commit and not draft_only
|
|
3555
|
+
)
|
|
3556
|
+
if repo.head.is_detached and will_push:
|
|
3557
|
+
# Suggest creating a branch rather than checking one out, since in a
|
|
3558
|
+
# worktree the branch they'd want may be checked out elsewhere
|
|
3559
|
+
raise_error(
|
|
3560
|
+
"HEAD is detached, so there is no branch to commit the release "
|
|
3561
|
+
"record to and push; create a branch at this revision first, "
|
|
3562
|
+
"e.g., with `git switch -c <branch>`"
|
|
3563
|
+
)
|
|
3539
3564
|
# Detect the release kind from the path unless it was given with --kind. A
|
|
3540
3565
|
# "." path is always a project release; otherwise prefer a declared
|
|
3541
3566
|
# artifact in calkit.yaml, falling back to auto-detection from the path
|
|
@@ -3589,10 +3614,24 @@ def new_release(
|
|
|
3589
3614
|
# that produces the released artifact when releasing a single path.
|
|
3590
3615
|
typer.echo("Checking pipeline is up-to-date for release")
|
|
3591
3616
|
targets = None
|
|
3617
|
+
# The stage that builds the released path, whose upstream stages define
|
|
3618
|
+
# what a --pipeline release carries
|
|
3619
|
+
pipeline_stage = ""
|
|
3592
3620
|
if path != ".":
|
|
3593
3621
|
stage_name = calkit.pipeline.get_stage_for_output(path, ck_info)
|
|
3594
|
-
if stage_name is
|
|
3622
|
+
if stage_name is None:
|
|
3623
|
+
if include_pipeline:
|
|
3624
|
+
raise_error(
|
|
3625
|
+
f"No pipeline stage produces '{path}', "
|
|
3626
|
+
"so there is no pipeline to release along with it"
|
|
3627
|
+
)
|
|
3628
|
+
else:
|
|
3595
3629
|
targets = [stage_name]
|
|
3630
|
+
pipeline_stage = stage_name
|
|
3631
|
+
elif include_pipeline:
|
|
3632
|
+
# A project release carries the whole pipeline already
|
|
3633
|
+
typer.echo("Project releases already include the pipeline")
|
|
3634
|
+
include_pipeline = False
|
|
3596
3635
|
status = calkit.pipeline.get_status(
|
|
3597
3636
|
ck_info=ck_info,
|
|
3598
3637
|
targets=targets,
|
|
@@ -3617,6 +3656,15 @@ def new_release(
|
|
|
3617
3656
|
release_date = str(calkit.utcnow().date())
|
|
3618
3657
|
typer.echo(f"Using release date: {release_date}")
|
|
3619
3658
|
git_rev = repo.git.rev_parse(["--short", "HEAD"])
|
|
3659
|
+
# This goes both beside the archive, which is the copy the archival
|
|
3660
|
+
# service displays, and inside it, so an extracted copy still says what
|
|
3661
|
+
# produced it. Rebuilt below once a more specific title is known.
|
|
3662
|
+
release_readme = calkit.releases.create_release_readme(
|
|
3663
|
+
release_kind=release_kind,
|
|
3664
|
+
name=name,
|
|
3665
|
+
git_rev=git_rev,
|
|
3666
|
+
title=ck_info.get("title"),
|
|
3667
|
+
)
|
|
3620
3668
|
# Fields below are populated only for external (archival) releases;
|
|
3621
3669
|
# internal releases leave them empty.
|
|
3622
3670
|
doi = None
|
|
@@ -3636,9 +3684,15 @@ def new_release(
|
|
|
3636
3684
|
stored_filename = f"{project_name}-{name}.zip"
|
|
3637
3685
|
is_zip = True
|
|
3638
3686
|
elif os.path.isfile(path):
|
|
3639
|
-
|
|
3640
|
-
|
|
3641
|
-
|
|
3687
|
+
# Releasing the pipeline along with the artifact means shipping
|
|
3688
|
+
# more than one file, so the artifact gets zipped up with it
|
|
3689
|
+
if include_pipeline:
|
|
3690
|
+
stored_filename = f"{project_name}-{name}.zip"
|
|
3691
|
+
is_zip = True
|
|
3692
|
+
else:
|
|
3693
|
+
_, ext = os.path.splitext(os.path.basename(path))
|
|
3694
|
+
stored_filename = f"{project_name}-{name}{ext}"
|
|
3695
|
+
is_zip = False
|
|
3642
3696
|
else:
|
|
3643
3697
|
raise_error(f"Release path '{path}' does not exist")
|
|
3644
3698
|
stored_path = os.path.join(release_dir, stored_filename)
|
|
@@ -3648,10 +3702,38 @@ def new_release(
|
|
|
3648
3702
|
typer.echo(f"Would {action} {path} to {stored_path_posix}")
|
|
3649
3703
|
else:
|
|
3650
3704
|
os.makedirs(release_dir, exist_ok=True)
|
|
3651
|
-
|
|
3705
|
+
overrides: dict[str, str] = {}
|
|
3706
|
+
if include_pipeline:
|
|
3707
|
+
typer.echo(f"Pruning project to what builds {path}")
|
|
3708
|
+
try:
|
|
3709
|
+
overrides, paths = calkit.releases.prune_for_stage(
|
|
3710
|
+
ck_info, pipeline_stage
|
|
3711
|
+
)
|
|
3712
|
+
except Exception as e:
|
|
3713
|
+
raise_error(
|
|
3714
|
+
f"Failed to prune project for stage "
|
|
3715
|
+
f"'{pipeline_stage}': {e}"
|
|
3716
|
+
)
|
|
3717
|
+
elif is_zip:
|
|
3652
3718
|
paths = calkit.releases.ls_files() if path == "." else [path]
|
|
3719
|
+
else:
|
|
3720
|
+
paths = []
|
|
3721
|
+
if is_zip:
|
|
3653
3722
|
typer.echo(f"Archiving {path} to {stored_path_posix}")
|
|
3654
|
-
calkit.releases.zip_paths(
|
|
3723
|
+
calkit.releases.zip_paths(
|
|
3724
|
+
stored_path,
|
|
3725
|
+
paths,
|
|
3726
|
+
overrides=overrides
|
|
3727
|
+
| {"CALKIT-RELEASE.md": release_readme},
|
|
3728
|
+
)
|
|
3729
|
+
if include_pipeline:
|
|
3730
|
+
typer.echo("Checking extracted release archive")
|
|
3731
|
+
try:
|
|
3732
|
+
calkit.releases.check_project_release_archive(
|
|
3733
|
+
stored_path, verbose=verbose
|
|
3734
|
+
)
|
|
3735
|
+
except Exception as e:
|
|
3736
|
+
raise_error(str(e))
|
|
3655
3737
|
else:
|
|
3656
3738
|
typer.echo(f"Copying {path} to {stored_path_posix}")
|
|
3657
3739
|
shutil.copy2(path, stored_path)
|
|
@@ -3679,10 +3761,30 @@ def new_release(
|
|
|
3679
3761
|
if path == ".":
|
|
3680
3762
|
if release_kind is None:
|
|
3681
3763
|
release_kind = "project"
|
|
3764
|
+
# Settle the title before building the archive, since the README
|
|
3765
|
+
# that goes inside it is headed with the title
|
|
3766
|
+
title = ck_info.get("title")
|
|
3767
|
+
if title is None:
|
|
3768
|
+
warn("Project has no title")
|
|
3769
|
+
title = typer.prompt("Enter a title for the project")
|
|
3770
|
+
ck_info["title"] = title
|
|
3771
|
+
if not dry_run:
|
|
3772
|
+
with open("calkit.yaml", "w") as f:
|
|
3773
|
+
calkit.ryaml.dump(ck_info, f)
|
|
3774
|
+
release_readme = calkit.releases.create_release_readme(
|
|
3775
|
+
release_kind=release_kind,
|
|
3776
|
+
name=name,
|
|
3777
|
+
git_rev=git_rev,
|
|
3778
|
+
title=title,
|
|
3779
|
+
)
|
|
3682
3780
|
zip_path = release_files_dir + "/archive.zip"
|
|
3683
3781
|
all_paths = calkit.releases.ls_files()
|
|
3684
3782
|
typer.echo(f"Adding files to {zip_path}")
|
|
3685
|
-
calkit.releases.zip_paths(
|
|
3783
|
+
calkit.releases.zip_paths(
|
|
3784
|
+
zip_path,
|
|
3785
|
+
all_paths,
|
|
3786
|
+
overrides={"CALKIT-RELEASE.md": release_readme},
|
|
3787
|
+
)
|
|
3686
3788
|
typer.echo("Checking extracted project release archive")
|
|
3687
3789
|
try:
|
|
3688
3790
|
calkit.releases.check_project_release_archive(
|
|
@@ -3690,14 +3792,6 @@ def new_release(
|
|
|
3690
3792
|
)
|
|
3691
3793
|
except Exception as e:
|
|
3692
3794
|
raise_error(str(e))
|
|
3693
|
-
title = ck_info.get("title")
|
|
3694
|
-
if title is None:
|
|
3695
|
-
warn("Project has no title")
|
|
3696
|
-
title = typer.prompt("Enter a title for the project")
|
|
3697
|
-
ck_info["title"] = title
|
|
3698
|
-
if not dry_run:
|
|
3699
|
-
with open("calkit.yaml", "w") as f:
|
|
3700
|
-
calkit.ryaml.dump(ck_info, f)
|
|
3701
3795
|
else:
|
|
3702
3796
|
# TODO: Handle directories, e.g., datasets
|
|
3703
3797
|
if not os.path.isfile(path):
|
|
@@ -3724,9 +3818,46 @@ def new_release(
|
|
|
3724
3818
|
)
|
|
3725
3819
|
if title is None:
|
|
3726
3820
|
raise_error(f"{release_kind} at {path} has no title")
|
|
3821
|
+
release_readme = calkit.releases.create_release_readme(
|
|
3822
|
+
release_kind=release_kind,
|
|
3823
|
+
name=name,
|
|
3824
|
+
git_rev=git_rev,
|
|
3825
|
+
title=title,
|
|
3826
|
+
)
|
|
3827
|
+
# Ship the artifact's provenance beside it: the stages that
|
|
3828
|
+
# build it, their inputs and environments, and a pipeline and
|
|
3829
|
+
# lock file pruned to match
|
|
3830
|
+
if include_pipeline:
|
|
3831
|
+
zip_path = release_files_dir + "/archive.zip"
|
|
3832
|
+
typer.echo(f"Pruning project to what builds {path}")
|
|
3833
|
+
try:
|
|
3834
|
+
overrides, all_paths = calkit.releases.prune_for_stage(
|
|
3835
|
+
ck_info, pipeline_stage
|
|
3836
|
+
)
|
|
3837
|
+
except Exception as e:
|
|
3838
|
+
raise_error(
|
|
3839
|
+
f"Failed to prune project for stage "
|
|
3840
|
+
f"'{pipeline_stage}': {e}"
|
|
3841
|
+
)
|
|
3842
|
+
typer.echo(f"Adding files to {zip_path}")
|
|
3843
|
+
calkit.releases.zip_paths(
|
|
3844
|
+
zip_path,
|
|
3845
|
+
all_paths,
|
|
3846
|
+
overrides=overrides
|
|
3847
|
+
| {"CALKIT-RELEASE.md": release_readme},
|
|
3848
|
+
)
|
|
3849
|
+
typer.echo("Checking extracted project release archive")
|
|
3850
|
+
try:
|
|
3851
|
+
calkit.releases.check_project_release_archive(
|
|
3852
|
+
zip_path, verbose=verbose
|
|
3853
|
+
)
|
|
3854
|
+
except Exception as e:
|
|
3855
|
+
raise_error(str(e))
|
|
3727
3856
|
# Save a metadata file with each DVC file's MD5 checksum
|
|
3728
3857
|
dvc_md5s = calkit.releases.make_dvc_md5s(
|
|
3729
|
-
zipfile=
|
|
3858
|
+
zipfile=(
|
|
3859
|
+
"archive.zip" if path == "." or include_pipeline else None
|
|
3860
|
+
),
|
|
3730
3861
|
paths=None if path == "." else [path],
|
|
3731
3862
|
)
|
|
3732
3863
|
dvc_md5s_path = release_dir + "/dvc-md5s.yaml"
|
|
@@ -3738,7 +3869,7 @@ def new_release(
|
|
|
3738
3869
|
# Archive the project's Docker images, so reproducing it doesn't
|
|
3739
3870
|
# depend on a registry keeping them around, and leave breadcrumbs
|
|
3740
3871
|
# behind so the environment check can fetch them back
|
|
3741
|
-
if path == "." and not no_docker_images:
|
|
3872
|
+
if (path == "." or include_pipeline) and not no_docker_images:
|
|
3742
3873
|
typer.echo("Archiving Docker images")
|
|
3743
3874
|
docker_images = calkit.releases.save_docker_images(
|
|
3744
3875
|
release_files_dir
|
|
@@ -3752,16 +3883,11 @@ def new_release(
|
|
|
3752
3883
|
calkit.ryaml.dump(docker_images, f)
|
|
3753
3884
|
if not dry_run:
|
|
3754
3885
|
repo.git.add(docker_images_path)
|
|
3755
|
-
#
|
|
3756
|
-
|
|
3757
|
-
git_rev = repo.git.rev_parse(["--short", "HEAD"])
|
|
3758
|
-
readme_txt += (
|
|
3759
|
-
f"\nThis is a {release_kind} release ({name}) generated with "
|
|
3760
|
-
f"Calkit v{calkit.__version__} from Git rev {git_rev}.\n"
|
|
3761
|
-
)
|
|
3886
|
+
# Write the same README beside the archive, since this is the copy
|
|
3887
|
+
# the archival service renders on the record page
|
|
3762
3888
|
readme_path = release_files_dir + "/README.md"
|
|
3763
3889
|
with open(readme_path, "w") as f:
|
|
3764
|
-
f.write(
|
|
3890
|
+
f.write(release_readme)
|
|
3765
3891
|
# Check size of files dir
|
|
3766
3892
|
size = calkit.get_size(release_files_dir)
|
|
3767
3893
|
typer.echo(f"Release size: {(size / 1e6):.1f} MB")
|
|
@@ -4114,6 +4240,7 @@ def new_release(
|
|
|
4114
4240
|
description=release_description,
|
|
4115
4241
|
internal=internal_release,
|
|
4116
4242
|
stored_path=stored_path_posix,
|
|
4243
|
+
includes_pipeline=include_pipeline,
|
|
4117
4244
|
).model_dump()
|
|
4118
4245
|
releases[name] = release
|
|
4119
4246
|
ck_info["releases"] = releases
|
|
@@ -4167,7 +4294,7 @@ def new_release(
|
|
|
4167
4294
|
record_id=record_id, # type: ignore
|
|
4168
4295
|
service=to, # type: ignore
|
|
4169
4296
|
)
|
|
4170
|
-
new_entries =
|
|
4297
|
+
new_entries = calkit.releases.parse_bibtex(invenio_bibtex)
|
|
4171
4298
|
if not new_entries:
|
|
4172
4299
|
raise ValueError("Failed to parse generated BibTeX entry")
|
|
4173
4300
|
new_entry = new_entries[0]
|
|
@@ -4179,9 +4306,9 @@ def new_release(
|
|
|
4179
4306
|
replace_ids = []
|
|
4180
4307
|
if new_doi:
|
|
4181
4308
|
try:
|
|
4182
|
-
existing_entries =
|
|
4309
|
+
existing_entries = calkit.releases.parse_bibtex(
|
|
4183
4310
|
existing_text
|
|
4184
|
-
)
|
|
4311
|
+
)
|
|
4185
4312
|
except Exception as e:
|
|
4186
4313
|
warn(f"Could not parse existing references to dedupe: {e}")
|
|
4187
4314
|
existing_entries = []
|
|
@@ -4220,7 +4347,7 @@ def new_release(
|
|
|
4220
4347
|
if not dry_run and calkit.git.get_staged_files() and not no_commit:
|
|
4221
4348
|
repo.git.commit(["-m", f"Create new {release_kind} release {name}"])
|
|
4222
4349
|
# Push with Git
|
|
4223
|
-
if
|
|
4350
|
+
if will_push:
|
|
4224
4351
|
repo.git.push(["origin", repo.active_branch.name, "--tags"])
|
|
4225
4352
|
# Now create GitHub release (external releases only)
|
|
4226
4353
|
if not internal_release and not no_github_release:
|
calkit/environments.py
CHANGED
|
@@ -1290,6 +1290,20 @@ def get_default_venv_prefix(envs: dict, path: str, name: str) -> str:
|
|
|
1290
1290
|
return Path(base).as_posix()
|
|
1291
1291
|
|
|
1292
1292
|
|
|
1293
|
+
def get_venv_activate_cmd(prefix: str, system: str | None = None) -> str:
|
|
1294
|
+
"""Get the shell command that activates the virtualenv at ``prefix``.
|
|
1295
|
+
|
|
1296
|
+
Prefixes are kept POSIX-style, but cmd reads a forward slash as the start
|
|
1297
|
+
of a switch, so it takes ``.calkit/envs/x/.venv`` for a command named
|
|
1298
|
+
``.calkit``. Hand Windows native separators instead.
|
|
1299
|
+
"""
|
|
1300
|
+
if system is None:
|
|
1301
|
+
system = platform.system()
|
|
1302
|
+
if system == "Windows":
|
|
1303
|
+
return prefix.replace("/", "\\") + "\\Scripts\\activate"
|
|
1304
|
+
return f". {prefix}/bin/activate"
|
|
1305
|
+
|
|
1306
|
+
|
|
1293
1307
|
def env_from_name_or_path(
|
|
1294
1308
|
name_or_path: str | None = None,
|
|
1295
1309
|
ck_info: dict | None = None,
|
calkit/invenio.py
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
"""Functionality for working with InvenioRDM instances like Zenodo."""
|
|
2
2
|
|
|
3
3
|
import os
|
|
4
|
+
import time
|
|
4
5
|
from functools import partial
|
|
5
6
|
from typing import Literal
|
|
6
7
|
|
|
@@ -57,6 +58,24 @@ def get_base_url(service: ServiceName = DEFAULT_SERVICE) -> str:
|
|
|
57
58
|
raise ValueError(f"Unknown archival service '{service}'")
|
|
58
59
|
|
|
59
60
|
|
|
61
|
+
# Pushing a release can mean sending hundreds of megabytes over a slow or
|
|
62
|
+
# distant link, so give a response plenty of time to arrive rather than
|
|
63
|
+
# letting requests wait forever with no timeout at all.
|
|
64
|
+
CONNECT_TIMEOUT = 30
|
|
65
|
+
READ_TIMEOUT = 600
|
|
66
|
+
# Gateways in front of InvenioRDM return these when they're busy, which says
|
|
67
|
+
# nothing about whether the request itself was valid, so it's worth sending
|
|
68
|
+
# again after a pause.
|
|
69
|
+
RETRY_STATUS_CODES = {429, 500, 502, 503, 504}
|
|
70
|
+
MAX_ATTEMPTS = 4
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def get_timeout() -> tuple[float, float]:
|
|
74
|
+
"""Get the connect and read timeouts to use for requests."""
|
|
75
|
+
read = os.getenv("CALKIT_INVENIO_TIMEOUT")
|
|
76
|
+
return CONNECT_TIMEOUT, float(read) if read else READ_TIMEOUT
|
|
77
|
+
|
|
78
|
+
|
|
60
79
|
def _request(
|
|
61
80
|
kind: Literal["get", "post", "put", "patch", "delete"],
|
|
62
81
|
path: str,
|
|
@@ -73,15 +92,31 @@ def _request(
|
|
|
73
92
|
params = {}
|
|
74
93
|
if auth and "access_token" not in params:
|
|
75
94
|
params = params | {"access_token": get_token(service=service)}
|
|
95
|
+
kwargs.setdefault("timeout", get_timeout())
|
|
76
96
|
func = getattr(requests, kind)
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
97
|
+
# A POST that timed out may still have been carried out on the far end,
|
|
98
|
+
# e.g., leaving a draft record behind, so only repeat requests that are
|
|
99
|
+
# safe to send twice
|
|
100
|
+
max_attempts = 1 if kind == "post" else MAX_ATTEMPTS
|
|
101
|
+
for attempt in range(1, max_attempts + 1):
|
|
102
|
+
try:
|
|
103
|
+
resp = func(
|
|
104
|
+
get_base_url(service=service) + path,
|
|
105
|
+
params=params,
|
|
106
|
+
json=json,
|
|
107
|
+
data=data,
|
|
108
|
+
headers=headers,
|
|
109
|
+
**kwargs,
|
|
110
|
+
)
|
|
111
|
+
except (requests.ConnectionError, requests.Timeout):
|
|
112
|
+
if attempt == max_attempts:
|
|
113
|
+
raise
|
|
114
|
+
time.sleep(2**attempt)
|
|
115
|
+
continue
|
|
116
|
+
if resp.status_code in RETRY_STATUS_CODES and attempt < max_attempts:
|
|
117
|
+
time.sleep(2**attempt)
|
|
118
|
+
continue
|
|
119
|
+
break
|
|
85
120
|
if resp.status_code >= 400:
|
|
86
121
|
msg = f"{resp.status_code}: "
|
|
87
122
|
try:
|
|
@@ -92,6 +127,14 @@ def _request(
|
|
|
92
127
|
msg += f"\nErrors:\n{resp_json['errors']}"
|
|
93
128
|
except ValueError:
|
|
94
129
|
msg += resp.text
|
|
130
|
+
if kind == "post" and resp.status_code in RETRY_STATUS_CODES:
|
|
131
|
+
# The far end may have done the work anyway, and a blind retry
|
|
132
|
+
# would duplicate it, so say so rather than papering over it
|
|
133
|
+
msg += (
|
|
134
|
+
f"\nThis request was not retried automatically, since "
|
|
135
|
+
f"{service} may have carried it out despite the error. "
|
|
136
|
+
"Check for a leftover draft record before trying again."
|
|
137
|
+
)
|
|
95
138
|
raise HTTPError(msg)
|
|
96
139
|
resp.raise_for_status()
|
|
97
140
|
if as_json:
|
calkit/models/core.py
CHANGED
|
@@ -1554,6 +1554,9 @@ class Release(BaseModel):
|
|
|
1554
1554
|
# ".calkit/releases/v0/my-project-slides-v0.pdf". Only set for internal
|
|
1555
1555
|
# releases, which store the artifact in the repo rather than ignoring it.
|
|
1556
1556
|
stored_path: str | None = None
|
|
1557
|
+
# Whether the release bundles the pipeline and inputs needed to
|
|
1558
|
+
# rebuild its path, rather than just the artifact itself.
|
|
1559
|
+
includes_pipeline: bool = False
|
|
1557
1560
|
|
|
1558
1561
|
|
|
1559
1562
|
class StaticHtmlApp(BaseModel):
|
calkit/pipeline.py
CHANGED
|
@@ -2328,3 +2328,24 @@ def translate_run_targets(
|
|
|
2328
2328
|
else:
|
|
2329
2329
|
parent_targets.append(f"{sp}/dvc.yaml")
|
|
2330
2330
|
return parent_targets, isolated_sp_targets
|
|
2331
|
+
|
|
2332
|
+
|
|
2333
|
+
def get_upstream_stages(target: str, wdir: str | None = None) -> set[str]:
|
|
2334
|
+
"""Get a stage's name plus the names of all stages it depends on.
|
|
2335
|
+
|
|
2336
|
+
DVC already knows how to walk its own graph, including foreach/matrix
|
|
2337
|
+
groups and stages defined in subdirectories, so we let it collect the
|
|
2338
|
+
target with its dependencies rather than reimplementing the traversal.
|
|
2339
|
+
Names of generated foreach stages come back as ``stage@item``, and both
|
|
2340
|
+
``calkit.yaml`` and ``dvc.yaml`` key off the base name, so we keep both.
|
|
2341
|
+
"""
|
|
2342
|
+
import calkit.dvc
|
|
2343
|
+
|
|
2344
|
+
repo = calkit.dvc.get_dvc_repo(wdir)
|
|
2345
|
+
names = set()
|
|
2346
|
+
for stage in repo.stage.collect(target, with_deps=True):
|
|
2347
|
+
name = getattr(stage, "name", None)
|
|
2348
|
+
if name:
|
|
2349
|
+
names.add(name)
|
|
2350
|
+
names.add(name.split("@")[0])
|
|
2351
|
+
return names
|