calkit-python 0.47.0__py3-none-any.whl → 0.47.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- calkit/cli/check.py +22 -13
- calkit/cli/main/core.py +1 -5
- calkit/cli/new.py +157 -30
- calkit/detect.py +93 -11
- calkit/docker.py +8 -6
- calkit/environments.py +14 -0
- calkit/invenio.py +51 -8
- calkit/models/core.py +38 -0
- calkit/pipeline.py +94 -0
- calkit/questions.py +313 -48
- calkit/releases.py +190 -7
- calkit/resources/devcontainer/Dockerfile +6 -8
- calkit/tests/cli/test_check.py +30 -8
- calkit/tests/cli/test_new.py +212 -0
- calkit/tests/test_detect.py +74 -1
- calkit/tests/test_environments.py +28 -0
- calkit/tests/test_invenio.py +79 -0
- calkit/tests/test_questions.py +182 -8
- calkit/tests/test_releases.py +78 -5
- {calkit_python-0.47.0.dist-info → calkit_python-0.47.2.dist-info}/METADATA +2 -1
- {calkit_python-0.47.0.dist-info → calkit_python-0.47.2.dist-info}/RECORD +38 -38
- {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/etc/jupyter/jupyter_server_config.d/calkit.json +0 -0
- {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/package.json +0 -0
- {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/schemas/calkit/package.json.orig +0 -0
- {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/schemas/calkit/plugin.json +0 -0
- {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/502.9a2c5772a15466e923ef.js +0 -0
- {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/695.2c41003a452d43d2b358.js +0 -0
- {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/867.a42a046aa5108f54f8fb.js +0 -0
- {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/909.6d8285ce7c45878ac508.js +0 -0
- {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/946.050af2abf7845cfbdbd2.js +0 -0
- {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/946.050af2abf7845cfbdbd2.js.LICENSE.txt +0 -0
- {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/b2f1c3efe70cb539d121.png +0 -0
- {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/remoteEntry.d7a43c7948f690d37d19.js +0 -0
- {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/style.js +0 -0
- {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/third-party-licenses.json +0 -0
- {calkit_python-0.47.0.dist-info → calkit_python-0.47.2.dist-info}/WHEEL +0 -0
- {calkit_python-0.47.0.dist-info → calkit_python-0.47.2.dist-info}/entry_points.txt +0 -0
- {calkit_python-0.47.0.dist-info → calkit_python-0.47.2.dist-info}/licenses/LICENSE +0 -0
calkit/cli/check.py
CHANGED
|
@@ -1802,10 +1802,7 @@ def check_venv(
|
|
|
1802
1802
|
if verbose:
|
|
1803
1803
|
typer.echo(f"Using legacy lock file: {legacy_fpath}")
|
|
1804
1804
|
break
|
|
1805
|
-
|
|
1806
|
-
activate_cmd = f"{prefix}\\Scripts\\activate"
|
|
1807
|
-
else:
|
|
1808
|
-
activate_cmd = f". {prefix}/bin/activate"
|
|
1805
|
+
activate_cmd = calkit.environments.get_venv_activate_cmd(prefix)
|
|
1809
1806
|
|
|
1810
1807
|
def pip_install_and_freeze(reqs_arg: str) -> None:
|
|
1811
1808
|
check_cmd = (
|
|
@@ -2060,20 +2057,32 @@ def check_questions(
|
|
|
2060
2057
|
json_output: Annotated[
|
|
2061
2058
|
bool, typer.Option("--json", help="Output the report as JSON.")
|
|
2062
2059
|
] = False,
|
|
2060
|
+
no_pipeline: Annotated[
|
|
2061
|
+
bool,
|
|
2062
|
+
typer.Option(
|
|
2063
|
+
"--no-pipeline",
|
|
2064
|
+
help="Skip asking DVC which stages are out of date, which is "
|
|
2065
|
+
"the slowest part of the check.",
|
|
2066
|
+
),
|
|
2067
|
+
] = False,
|
|
2063
2068
|
) -> None:
|
|
2064
|
-
"""Check that answered questions are
|
|
2065
|
-
|
|
2066
|
-
|
|
2067
|
-
|
|
2068
|
-
|
|
2069
|
-
|
|
2070
|
-
|
|
2071
|
-
|
|
2069
|
+
"""Check that answered questions are backed by current evidence.
|
|
2070
|
+
|
|
2071
|
+
Reports, worst first: evidence that isn't there (never run, never
|
|
2072
|
+
pushed, or pinned to a Git ref that doesn't exist); broken references
|
|
2073
|
+
(a key that doesn't resolve, a placeholder that names no evidence, a
|
|
2074
|
+
label missing from the LaTeX); evidence the pipeline would rebuild;
|
|
2075
|
+
and evidence from a frozen stage, or downstream of one, which nothing
|
|
2076
|
+
will ever report out of date unless the citation pins a git_ref.
|
|
2077
|
+
|
|
2078
|
+
Evidence pinned with a git_ref is checked at that ref rather than in
|
|
2079
|
+
the working tree. Exits with an error if any answered question is
|
|
2080
|
+
missing evidence, broken, or out of date with the pipeline.
|
|
2072
2081
|
"""
|
|
2073
2082
|
from calkit.questions import check_questions as _check_questions
|
|
2074
2083
|
from calkit.questions import format_status
|
|
2075
2084
|
|
|
2076
|
-
status = _check_questions(wdir=wdir)
|
|
2085
|
+
status = _check_questions(wdir=wdir, check_pipeline=not no_pipeline)
|
|
2077
2086
|
if json_output:
|
|
2078
2087
|
calkit.echo(json.dumps(status.model_dump(mode="json"), indent=2))
|
|
2079
2088
|
else:
|
calkit/cli/main/core.py
CHANGED
|
@@ -6,7 +6,6 @@ import csv
|
|
|
6
6
|
import json
|
|
7
7
|
import logging
|
|
8
8
|
import os
|
|
9
|
-
import platform as _platform
|
|
10
9
|
import posixpath
|
|
11
10
|
import shlex
|
|
12
11
|
import shutil
|
|
@@ -3543,10 +3542,7 @@ def run_in_env(
|
|
|
3543
3542
|
envs, path, env_name
|
|
3544
3543
|
)
|
|
3545
3544
|
shell_cmd = _to_shell_cmd(cmd)
|
|
3546
|
-
|
|
3547
|
-
activate_cmd = f"{prefix}\\Scripts\\activate"
|
|
3548
|
-
else:
|
|
3549
|
-
activate_cmd = f". {prefix}/bin/activate"
|
|
3545
|
+
activate_cmd = calkit.environments.get_venv_activate_cmd(prefix)
|
|
3550
3546
|
if verbose:
|
|
3551
3547
|
typer.echo(f"Raw command: {cmd}")
|
|
3552
3548
|
typer.echo(f"Shell command: {shell_cmd}")
|
calkit/cli/new.py
CHANGED
|
@@ -3434,6 +3434,18 @@ def new_release(
|
|
|
3434
3434
|
str | None,
|
|
3435
3435
|
typer.Option("--date", help="Release date. Will default to today."),
|
|
3436
3436
|
] = None,
|
|
3437
|
+
include_pipeline: Annotated[
|
|
3438
|
+
bool,
|
|
3439
|
+
typer.Option(
|
|
3440
|
+
"--pipeline",
|
|
3441
|
+
help=(
|
|
3442
|
+
"Include everything needed to reproduce the released path, "
|
|
3443
|
+
"i.e., the pipeline, its lock file, and the stages, inputs, "
|
|
3444
|
+
"and environments the path depends on. Stages unrelated to "
|
|
3445
|
+
"the path are left out."
|
|
3446
|
+
),
|
|
3447
|
+
),
|
|
3448
|
+
] = False,
|
|
3437
3449
|
no_docker_images: Annotated[
|
|
3438
3450
|
bool,
|
|
3439
3451
|
typer.Option(
|
|
@@ -3521,7 +3533,6 @@ def new_release(
|
|
|
3521
3533
|
] = False,
|
|
3522
3534
|
):
|
|
3523
3535
|
"""Create a new release."""
|
|
3524
|
-
import bibtexparser
|
|
3525
3536
|
import dotenv
|
|
3526
3537
|
|
|
3527
3538
|
import calkit.pipeline
|
|
@@ -3536,6 +3547,20 @@ def new_release(
|
|
|
3536
3547
|
repo = calkit.git.get_repo()
|
|
3537
3548
|
if name in repo.tags:
|
|
3538
3549
|
raise_error(f"Git tag with name '{name}' already exists")
|
|
3550
|
+
# A release commits to calkit.yaml and pushes the branch it's on, neither
|
|
3551
|
+
# of which works from a detached HEAD. Check before anything is uploaded,
|
|
3552
|
+
# so a release can't get published and then fail on the way out.
|
|
3553
|
+
will_push = (
|
|
3554
|
+
not dry_run and not no_push and not no_commit and not draft_only
|
|
3555
|
+
)
|
|
3556
|
+
if repo.head.is_detached and will_push:
|
|
3557
|
+
# Suggest creating a branch rather than checking one out, since in a
|
|
3558
|
+
# worktree the branch they'd want may be checked out elsewhere
|
|
3559
|
+
raise_error(
|
|
3560
|
+
"HEAD is detached, so there is no branch to commit the release "
|
|
3561
|
+
"record to and push; create a branch at this revision first, "
|
|
3562
|
+
"e.g., with `git switch -c <branch>`"
|
|
3563
|
+
)
|
|
3539
3564
|
# Detect the release kind from the path unless it was given with --kind. A
|
|
3540
3565
|
# "." path is always a project release; otherwise prefer a declared
|
|
3541
3566
|
# artifact in calkit.yaml, falling back to auto-detection from the path
|
|
@@ -3589,10 +3614,24 @@ def new_release(
|
|
|
3589
3614
|
# that produces the released artifact when releasing a single path.
|
|
3590
3615
|
typer.echo("Checking pipeline is up-to-date for release")
|
|
3591
3616
|
targets = None
|
|
3617
|
+
# The stage that builds the released path, whose upstream stages define
|
|
3618
|
+
# what a --pipeline release carries
|
|
3619
|
+
pipeline_stage = ""
|
|
3592
3620
|
if path != ".":
|
|
3593
3621
|
stage_name = calkit.pipeline.get_stage_for_output(path, ck_info)
|
|
3594
|
-
if stage_name is
|
|
3622
|
+
if stage_name is None:
|
|
3623
|
+
if include_pipeline:
|
|
3624
|
+
raise_error(
|
|
3625
|
+
f"No pipeline stage produces '{path}', "
|
|
3626
|
+
"so there is no pipeline to release along with it"
|
|
3627
|
+
)
|
|
3628
|
+
else:
|
|
3595
3629
|
targets = [stage_name]
|
|
3630
|
+
pipeline_stage = stage_name
|
|
3631
|
+
elif include_pipeline:
|
|
3632
|
+
# A project release carries the whole pipeline already
|
|
3633
|
+
typer.echo("Project releases already include the pipeline")
|
|
3634
|
+
include_pipeline = False
|
|
3596
3635
|
status = calkit.pipeline.get_status(
|
|
3597
3636
|
ck_info=ck_info,
|
|
3598
3637
|
targets=targets,
|
|
@@ -3617,6 +3656,15 @@ def new_release(
|
|
|
3617
3656
|
release_date = str(calkit.utcnow().date())
|
|
3618
3657
|
typer.echo(f"Using release date: {release_date}")
|
|
3619
3658
|
git_rev = repo.git.rev_parse(["--short", "HEAD"])
|
|
3659
|
+
# This goes both beside the archive, which is the copy the archival
|
|
3660
|
+
# service displays, and inside it, so an extracted copy still says what
|
|
3661
|
+
# produced it. Rebuilt below once a more specific title is known.
|
|
3662
|
+
release_readme = calkit.releases.create_release_readme(
|
|
3663
|
+
release_kind=release_kind,
|
|
3664
|
+
name=name,
|
|
3665
|
+
git_rev=git_rev,
|
|
3666
|
+
title=ck_info.get("title"),
|
|
3667
|
+
)
|
|
3620
3668
|
# Fields below are populated only for external (archival) releases;
|
|
3621
3669
|
# internal releases leave them empty.
|
|
3622
3670
|
doi = None
|
|
@@ -3636,9 +3684,15 @@ def new_release(
|
|
|
3636
3684
|
stored_filename = f"{project_name}-{name}.zip"
|
|
3637
3685
|
is_zip = True
|
|
3638
3686
|
elif os.path.isfile(path):
|
|
3639
|
-
|
|
3640
|
-
|
|
3641
|
-
|
|
3687
|
+
# Releasing the pipeline along with the artifact means shipping
|
|
3688
|
+
# more than one file, so the artifact gets zipped up with it
|
|
3689
|
+
if include_pipeline:
|
|
3690
|
+
stored_filename = f"{project_name}-{name}.zip"
|
|
3691
|
+
is_zip = True
|
|
3692
|
+
else:
|
|
3693
|
+
_, ext = os.path.splitext(os.path.basename(path))
|
|
3694
|
+
stored_filename = f"{project_name}-{name}{ext}"
|
|
3695
|
+
is_zip = False
|
|
3642
3696
|
else:
|
|
3643
3697
|
raise_error(f"Release path '{path}' does not exist")
|
|
3644
3698
|
stored_path = os.path.join(release_dir, stored_filename)
|
|
@@ -3648,10 +3702,38 @@ def new_release(
|
|
|
3648
3702
|
typer.echo(f"Would {action} {path} to {stored_path_posix}")
|
|
3649
3703
|
else:
|
|
3650
3704
|
os.makedirs(release_dir, exist_ok=True)
|
|
3651
|
-
|
|
3705
|
+
overrides: dict[str, str] = {}
|
|
3706
|
+
if include_pipeline:
|
|
3707
|
+
typer.echo(f"Pruning project to what builds {path}")
|
|
3708
|
+
try:
|
|
3709
|
+
overrides, paths = calkit.releases.prune_for_stage(
|
|
3710
|
+
ck_info, pipeline_stage
|
|
3711
|
+
)
|
|
3712
|
+
except Exception as e:
|
|
3713
|
+
raise_error(
|
|
3714
|
+
f"Failed to prune project for stage "
|
|
3715
|
+
f"'{pipeline_stage}': {e}"
|
|
3716
|
+
)
|
|
3717
|
+
elif is_zip:
|
|
3652
3718
|
paths = calkit.releases.ls_files() if path == "." else [path]
|
|
3719
|
+
else:
|
|
3720
|
+
paths = []
|
|
3721
|
+
if is_zip:
|
|
3653
3722
|
typer.echo(f"Archiving {path} to {stored_path_posix}")
|
|
3654
|
-
calkit.releases.zip_paths(
|
|
3723
|
+
calkit.releases.zip_paths(
|
|
3724
|
+
stored_path,
|
|
3725
|
+
paths,
|
|
3726
|
+
overrides=overrides
|
|
3727
|
+
| {"CALKIT-RELEASE.md": release_readme},
|
|
3728
|
+
)
|
|
3729
|
+
if include_pipeline:
|
|
3730
|
+
typer.echo("Checking extracted release archive")
|
|
3731
|
+
try:
|
|
3732
|
+
calkit.releases.check_project_release_archive(
|
|
3733
|
+
stored_path, verbose=verbose
|
|
3734
|
+
)
|
|
3735
|
+
except Exception as e:
|
|
3736
|
+
raise_error(str(e))
|
|
3655
3737
|
else:
|
|
3656
3738
|
typer.echo(f"Copying {path} to {stored_path_posix}")
|
|
3657
3739
|
shutil.copy2(path, stored_path)
|
|
@@ -3679,10 +3761,30 @@ def new_release(
|
|
|
3679
3761
|
if path == ".":
|
|
3680
3762
|
if release_kind is None:
|
|
3681
3763
|
release_kind = "project"
|
|
3764
|
+
# Settle the title before building the archive, since the README
|
|
3765
|
+
# that goes inside it is headed with the title
|
|
3766
|
+
title = ck_info.get("title")
|
|
3767
|
+
if title is None:
|
|
3768
|
+
warn("Project has no title")
|
|
3769
|
+
title = typer.prompt("Enter a title for the project")
|
|
3770
|
+
ck_info["title"] = title
|
|
3771
|
+
if not dry_run:
|
|
3772
|
+
with open("calkit.yaml", "w") as f:
|
|
3773
|
+
calkit.ryaml.dump(ck_info, f)
|
|
3774
|
+
release_readme = calkit.releases.create_release_readme(
|
|
3775
|
+
release_kind=release_kind,
|
|
3776
|
+
name=name,
|
|
3777
|
+
git_rev=git_rev,
|
|
3778
|
+
title=title,
|
|
3779
|
+
)
|
|
3682
3780
|
zip_path = release_files_dir + "/archive.zip"
|
|
3683
3781
|
all_paths = calkit.releases.ls_files()
|
|
3684
3782
|
typer.echo(f"Adding files to {zip_path}")
|
|
3685
|
-
calkit.releases.zip_paths(
|
|
3783
|
+
calkit.releases.zip_paths(
|
|
3784
|
+
zip_path,
|
|
3785
|
+
all_paths,
|
|
3786
|
+
overrides={"CALKIT-RELEASE.md": release_readme},
|
|
3787
|
+
)
|
|
3686
3788
|
typer.echo("Checking extracted project release archive")
|
|
3687
3789
|
try:
|
|
3688
3790
|
calkit.releases.check_project_release_archive(
|
|
@@ -3690,14 +3792,6 @@ def new_release(
|
|
|
3690
3792
|
)
|
|
3691
3793
|
except Exception as e:
|
|
3692
3794
|
raise_error(str(e))
|
|
3693
|
-
title = ck_info.get("title")
|
|
3694
|
-
if title is None:
|
|
3695
|
-
warn("Project has no title")
|
|
3696
|
-
title = typer.prompt("Enter a title for the project")
|
|
3697
|
-
ck_info["title"] = title
|
|
3698
|
-
if not dry_run:
|
|
3699
|
-
with open("calkit.yaml", "w") as f:
|
|
3700
|
-
calkit.ryaml.dump(ck_info, f)
|
|
3701
3795
|
else:
|
|
3702
3796
|
# TODO: Handle directories, e.g., datasets
|
|
3703
3797
|
if not os.path.isfile(path):
|
|
@@ -3724,9 +3818,46 @@ def new_release(
|
|
|
3724
3818
|
)
|
|
3725
3819
|
if title is None:
|
|
3726
3820
|
raise_error(f"{release_kind} at {path} has no title")
|
|
3821
|
+
release_readme = calkit.releases.create_release_readme(
|
|
3822
|
+
release_kind=release_kind,
|
|
3823
|
+
name=name,
|
|
3824
|
+
git_rev=git_rev,
|
|
3825
|
+
title=title,
|
|
3826
|
+
)
|
|
3827
|
+
# Ship the artifact's provenance beside it: the stages that
|
|
3828
|
+
# build it, their inputs and environments, and a pipeline and
|
|
3829
|
+
# lock file pruned to match
|
|
3830
|
+
if include_pipeline:
|
|
3831
|
+
zip_path = release_files_dir + "/archive.zip"
|
|
3832
|
+
typer.echo(f"Pruning project to what builds {path}")
|
|
3833
|
+
try:
|
|
3834
|
+
overrides, all_paths = calkit.releases.prune_for_stage(
|
|
3835
|
+
ck_info, pipeline_stage
|
|
3836
|
+
)
|
|
3837
|
+
except Exception as e:
|
|
3838
|
+
raise_error(
|
|
3839
|
+
f"Failed to prune project for stage "
|
|
3840
|
+
f"'{pipeline_stage}': {e}"
|
|
3841
|
+
)
|
|
3842
|
+
typer.echo(f"Adding files to {zip_path}")
|
|
3843
|
+
calkit.releases.zip_paths(
|
|
3844
|
+
zip_path,
|
|
3845
|
+
all_paths,
|
|
3846
|
+
overrides=overrides
|
|
3847
|
+
| {"CALKIT-RELEASE.md": release_readme},
|
|
3848
|
+
)
|
|
3849
|
+
typer.echo("Checking extracted project release archive")
|
|
3850
|
+
try:
|
|
3851
|
+
calkit.releases.check_project_release_archive(
|
|
3852
|
+
zip_path, verbose=verbose
|
|
3853
|
+
)
|
|
3854
|
+
except Exception as e:
|
|
3855
|
+
raise_error(str(e))
|
|
3727
3856
|
# Save a metadata file with each DVC file's MD5 checksum
|
|
3728
3857
|
dvc_md5s = calkit.releases.make_dvc_md5s(
|
|
3729
|
-
zipfile=
|
|
3858
|
+
zipfile=(
|
|
3859
|
+
"archive.zip" if path == "." or include_pipeline else None
|
|
3860
|
+
),
|
|
3730
3861
|
paths=None if path == "." else [path],
|
|
3731
3862
|
)
|
|
3732
3863
|
dvc_md5s_path = release_dir + "/dvc-md5s.yaml"
|
|
@@ -3738,7 +3869,7 @@ def new_release(
|
|
|
3738
3869
|
# Archive the project's Docker images, so reproducing it doesn't
|
|
3739
3870
|
# depend on a registry keeping them around, and leave breadcrumbs
|
|
3740
3871
|
# behind so the environment check can fetch them back
|
|
3741
|
-
if path == "." and not no_docker_images:
|
|
3872
|
+
if (path == "." or include_pipeline) and not no_docker_images:
|
|
3742
3873
|
typer.echo("Archiving Docker images")
|
|
3743
3874
|
docker_images = calkit.releases.save_docker_images(
|
|
3744
3875
|
release_files_dir
|
|
@@ -3752,16 +3883,11 @@ def new_release(
|
|
|
3752
3883
|
calkit.ryaml.dump(docker_images, f)
|
|
3753
3884
|
if not dry_run:
|
|
3754
3885
|
repo.git.add(docker_images_path)
|
|
3755
|
-
#
|
|
3756
|
-
|
|
3757
|
-
git_rev = repo.git.rev_parse(["--short", "HEAD"])
|
|
3758
|
-
readme_txt += (
|
|
3759
|
-
f"\nThis is a {release_kind} release ({name}) generated with "
|
|
3760
|
-
f"Calkit v{calkit.__version__} from Git rev {git_rev}.\n"
|
|
3761
|
-
)
|
|
3886
|
+
# Write the same README beside the archive, since this is the copy
|
|
3887
|
+
# the archival service renders on the record page
|
|
3762
3888
|
readme_path = release_files_dir + "/README.md"
|
|
3763
3889
|
with open(readme_path, "w") as f:
|
|
3764
|
-
f.write(
|
|
3890
|
+
f.write(release_readme)
|
|
3765
3891
|
# Check size of files dir
|
|
3766
3892
|
size = calkit.get_size(release_files_dir)
|
|
3767
3893
|
typer.echo(f"Release size: {(size / 1e6):.1f} MB")
|
|
@@ -4114,6 +4240,7 @@ def new_release(
|
|
|
4114
4240
|
description=release_description,
|
|
4115
4241
|
internal=internal_release,
|
|
4116
4242
|
stored_path=stored_path_posix,
|
|
4243
|
+
includes_pipeline=include_pipeline,
|
|
4117
4244
|
).model_dump()
|
|
4118
4245
|
releases[name] = release
|
|
4119
4246
|
ck_info["releases"] = releases
|
|
@@ -4167,7 +4294,7 @@ def new_release(
|
|
|
4167
4294
|
record_id=record_id, # type: ignore
|
|
4168
4295
|
service=to, # type: ignore
|
|
4169
4296
|
)
|
|
4170
|
-
new_entries =
|
|
4297
|
+
new_entries = calkit.releases.parse_bibtex(invenio_bibtex)
|
|
4171
4298
|
if not new_entries:
|
|
4172
4299
|
raise ValueError("Failed to parse generated BibTeX entry")
|
|
4173
4300
|
new_entry = new_entries[0]
|
|
@@ -4179,9 +4306,9 @@ def new_release(
|
|
|
4179
4306
|
replace_ids = []
|
|
4180
4307
|
if new_doi:
|
|
4181
4308
|
try:
|
|
4182
|
-
existing_entries =
|
|
4309
|
+
existing_entries = calkit.releases.parse_bibtex(
|
|
4183
4310
|
existing_text
|
|
4184
|
-
)
|
|
4311
|
+
)
|
|
4185
4312
|
except Exception as e:
|
|
4186
4313
|
warn(f"Could not parse existing references to dedupe: {e}")
|
|
4187
4314
|
existing_entries = []
|
|
@@ -4220,7 +4347,7 @@ def new_release(
|
|
|
4220
4347
|
if not dry_run and calkit.git.get_staged_files() and not no_commit:
|
|
4221
4348
|
repo.git.commit(["-m", f"Create new {release_kind} release {name}"])
|
|
4222
4349
|
# Push with Git
|
|
4223
|
-
if
|
|
4350
|
+
if will_push:
|
|
4224
4351
|
repo.git.push(["origin", repo.active_branch.name, "--tags"])
|
|
4225
4352
|
# Now create GitHub release (external releases only)
|
|
4226
4353
|
if not internal_release and not no_github_release:
|
calkit/detect.py
CHANGED
|
@@ -1773,40 +1773,119 @@ def detect_r_dependencies(
|
|
|
1773
1773
|
def detect_julia_dependencies(
|
|
1774
1774
|
script_path: str | None = None,
|
|
1775
1775
|
code: str | None = None,
|
|
1776
|
+
script_dir: str | None = None,
|
|
1777
|
+
project_dir: str = ".",
|
|
1776
1778
|
) -> list[str]:
|
|
1777
1779
|
"""Detect package dependencies from a Julia script or code string.
|
|
1778
1780
|
|
|
1781
|
+
Julia's ``include`` splices a file in as source text, so any package used
|
|
1782
|
+
by an included file must be declared by the project that includes it.
|
|
1783
|
+
Includes with a literal path that resolve inside the project are therefore
|
|
1784
|
+
followed. Ones pointing outside it are not, since that code declares its
|
|
1785
|
+
dependencies in its own project file, e.g., a package's own source in the
|
|
1786
|
+
depot reached via ``pkgdir``.
|
|
1787
|
+
|
|
1779
1788
|
Parameters
|
|
1780
1789
|
----------
|
|
1781
1790
|
script_path : str | None
|
|
1782
1791
|
Path to Julia script. Either this or code must be provided.
|
|
1783
1792
|
code : str | None
|
|
1784
1793
|
Julia code string. Either this or script_path must be provided.
|
|
1794
|
+
script_dir : str | None
|
|
1795
|
+
Directory the code came from, against which its includes resolve.
|
|
1796
|
+
Only used with ``code``; defaults to ``project_dir``.
|
|
1797
|
+
project_dir : str
|
|
1798
|
+
Project root, outside of which includes are not followed.
|
|
1785
1799
|
|
|
1786
1800
|
Returns
|
|
1787
1801
|
-------
|
|
1788
1802
|
list[str]
|
|
1789
1803
|
List of Julia package names.
|
|
1790
1804
|
"""
|
|
1805
|
+
|
|
1806
|
+
def parse_dependencies(code: str) -> set[str]:
|
|
1807
|
+
deps = set()
|
|
1808
|
+
# Both `using` and `import` load a package, either can start a line or
|
|
1809
|
+
# follow a semicolon, and either can be prefixed by macros, e.g.,
|
|
1810
|
+
# `@everywhere using Foo`
|
|
1811
|
+
clauses = re.findall(
|
|
1812
|
+
r"(?:^|;)[ \t]*(?:@[A-Za-z_][A-Za-z0-9_!]*[ \t]+)*"
|
|
1813
|
+
r"(?:using|import)[ \t]+([^\n;]+)",
|
|
1814
|
+
code,
|
|
1815
|
+
flags=re.MULTILINE,
|
|
1816
|
+
)
|
|
1817
|
+
for clause in clauses:
|
|
1818
|
+
# In `using Foo: bar, baz` only what precedes the colon is a
|
|
1819
|
+
# package
|
|
1820
|
+
clause = clause.split(":")[0]
|
|
1821
|
+
for part in clause.split(","):
|
|
1822
|
+
# Drop an `as` alias, e.g., `import Foo as F`
|
|
1823
|
+
name = re.split(r"\s+as\s+", part.strip())[0].strip()
|
|
1824
|
+
# A leading dot means a module local to this file, not a
|
|
1825
|
+
# package
|
|
1826
|
+
if not name or name.startswith("."):
|
|
1827
|
+
continue
|
|
1828
|
+
# Submodules like `Foo.Bar` come from the `Foo` package
|
|
1829
|
+
name = name.split(".")[0]
|
|
1830
|
+
if not re.fullmatch(r"[A-Za-z_][A-Za-z0-9_!]*", name):
|
|
1831
|
+
continue
|
|
1832
|
+
if name in ("Base", "Core", "Main"):
|
|
1833
|
+
continue
|
|
1834
|
+
deps.add(name)
|
|
1835
|
+
return deps
|
|
1836
|
+
|
|
1837
|
+
def find_includes(code: str, from_dir: str) -> list[str]:
|
|
1838
|
+
paths = []
|
|
1839
|
+
for match in re.findall(
|
|
1840
|
+
r'include\s*\(\s*["\']([^"\']+\.jl)["\']\s*\)', code
|
|
1841
|
+
):
|
|
1842
|
+
if os.path.isabs(match):
|
|
1843
|
+
continue
|
|
1844
|
+
path = os.path.realpath(os.path.join(from_dir, match))
|
|
1845
|
+
if not path.startswith(root + os.sep) or not os.path.isfile(path):
|
|
1846
|
+
continue
|
|
1847
|
+
paths.append(path)
|
|
1848
|
+
return paths
|
|
1849
|
+
|
|
1850
|
+
def read(path: str) -> str | None:
|
|
1851
|
+
try:
|
|
1852
|
+
with open(path, "r", encoding="utf-8") as f:
|
|
1853
|
+
return f.read()
|
|
1854
|
+
except (UnicodeDecodeError, IOError):
|
|
1855
|
+
return None
|
|
1856
|
+
|
|
1791
1857
|
if script_path is None and code is None:
|
|
1792
1858
|
raise ValueError("Either script_path or code must be provided")
|
|
1859
|
+
root = os.path.realpath(project_dir)
|
|
1793
1860
|
if code is None:
|
|
1794
1861
|
assert script_path is not None # Type guard
|
|
1795
1862
|
if not os.path.exists(script_path):
|
|
1796
1863
|
return []
|
|
1797
|
-
|
|
1798
|
-
|
|
1799
|
-
code = f.read()
|
|
1800
|
-
except (UnicodeDecodeError, IOError):
|
|
1864
|
+
code = read(script_path)
|
|
1865
|
+
if code is None:
|
|
1801
1866
|
return []
|
|
1802
|
-
|
|
1803
|
-
|
|
1867
|
+
from_dir = os.path.dirname(script_path) or "."
|
|
1868
|
+
seen = {os.path.realpath(script_path)}
|
|
1869
|
+
else:
|
|
1870
|
+
from_dir = script_dir if script_dir is not None else project_dir
|
|
1871
|
+
seen = set()
|
|
1804
1872
|
# Remove comments
|
|
1805
1873
|
code = re.sub(r"#.*$", "", code, flags=re.MULTILINE)
|
|
1806
|
-
|
|
1807
|
-
|
|
1808
|
-
|
|
1809
|
-
|
|
1874
|
+
dependencies = parse_dependencies(code)
|
|
1875
|
+
# Walk the include tree, resolving each file's includes against its own
|
|
1876
|
+
# directory the way Julia does
|
|
1877
|
+
queue = find_includes(code, from_dir)
|
|
1878
|
+
while queue:
|
|
1879
|
+
path = queue.pop(0)
|
|
1880
|
+
if path in seen:
|
|
1881
|
+
continue
|
|
1882
|
+
seen.add(path)
|
|
1883
|
+
included = read(path)
|
|
1884
|
+
if included is None:
|
|
1885
|
+
continue
|
|
1886
|
+
included = re.sub(r"#.*$", "", included, flags=re.MULTILINE)
|
|
1887
|
+
dependencies |= parse_dependencies(included)
|
|
1888
|
+
queue += find_includes(included, os.path.dirname(path))
|
|
1810
1889
|
return sorted(list(dependencies))
|
|
1811
1890
|
|
|
1812
1891
|
|
|
@@ -1855,7 +1934,10 @@ def detect_dependencies_from_notebook(
|
|
|
1855
1934
|
if language == "python":
|
|
1856
1935
|
return detect_python_dependencies(code=combined_code)
|
|
1857
1936
|
elif language == "julia":
|
|
1858
|
-
return detect_julia_dependencies(
|
|
1937
|
+
return detect_julia_dependencies(
|
|
1938
|
+
code=combined_code,
|
|
1939
|
+
script_dir=os.path.dirname(notebook_path) or ".",
|
|
1940
|
+
)
|
|
1859
1941
|
elif language == "r":
|
|
1860
1942
|
return detect_r_dependencies(code=combined_code)
|
|
1861
1943
|
return []
|
calkit/docker.py
CHANGED
|
@@ -14,7 +14,7 @@ from pydantic import BaseModel
|
|
|
14
14
|
MINIFORGE_LAYER_TXT = r"""
|
|
15
15
|
# Install Miniforge
|
|
16
16
|
ARG MINIFORGE_NAME=Miniforge3
|
|
17
|
-
ARG MINIFORGE_VERSION=
|
|
17
|
+
ARG MINIFORGE_VERSION=26.7.2-0
|
|
18
18
|
ARG TARGETPLATFORM
|
|
19
19
|
|
|
20
20
|
ENV CONDA_DIR=/opt/conda
|
|
@@ -51,24 +51,26 @@ RUN apt-get update > /dev/null && \
|
|
|
51
51
|
echo ". ${CONDA_DIR}/etc/profile.d/conda.sh && conda activate base" >> ~/.bashrc
|
|
52
52
|
""".strip()
|
|
53
53
|
|
|
54
|
+
# foamPy ships only an sdist whose setup.py imports numpy, so it needs the
|
|
55
|
+
# surrounding environment rather than an isolated build one
|
|
54
56
|
FOAMPY_LAYER_TEXT = r"""
|
|
55
57
|
RUN pip install --no-cache-dir numpy pandas matplotlib h5py \
|
|
56
58
|
&& pip install --no-cache-dir scipy \
|
|
57
|
-
&& pip install --no-cache-dir foampy
|
|
59
|
+
&& pip install --no-cache-dir --no-build-isolation foampy
|
|
58
60
|
""".strip()
|
|
59
61
|
|
|
60
62
|
UV_LAYER_TEXT = """
|
|
61
|
-
COPY --from=ghcr.io/astral-sh/uv:0.
|
|
63
|
+
COPY --from=ghcr.io/astral-sh/uv:0.12.11 /uv /uvx /bin/
|
|
62
64
|
"""
|
|
63
65
|
|
|
64
66
|
JULIA_LAYER_TEXT = """
|
|
65
67
|
# Install Julia
|
|
66
|
-
# Ensure base image is a
|
|
67
|
-
COPY --from=julia:1.11.
|
|
68
|
+
# Ensure base image is a bookworm distribution
|
|
69
|
+
COPY --from=julia:1.11.9-bookworm /usr/local/julia /usr/local/julia
|
|
68
70
|
ENV JULIA_PATH=/usr/local/julia \
|
|
69
71
|
PATH=$PATH:/usr/local/julia/bin \
|
|
70
72
|
JULIA_GPG=3673DF529D9049477F76B37566E3C7DC03D6E495 \
|
|
71
|
-
JULIA_VERSION=1.11.
|
|
73
|
+
JULIA_VERSION=1.11.9
|
|
72
74
|
"""
|
|
73
75
|
|
|
74
76
|
LAYERS = {
|
calkit/environments.py
CHANGED
|
@@ -1290,6 +1290,20 @@ def get_default_venv_prefix(envs: dict, path: str, name: str) -> str:
|
|
|
1290
1290
|
return Path(base).as_posix()
|
|
1291
1291
|
|
|
1292
1292
|
|
|
1293
|
+
def get_venv_activate_cmd(prefix: str, system: str | None = None) -> str:
|
|
1294
|
+
"""Get the shell command that activates the virtualenv at ``prefix``.
|
|
1295
|
+
|
|
1296
|
+
Prefixes are kept POSIX-style, but cmd reads a forward slash as the start
|
|
1297
|
+
of a switch, so it takes ``.calkit/envs/x/.venv`` for a command named
|
|
1298
|
+
``.calkit``. Hand Windows native separators instead.
|
|
1299
|
+
"""
|
|
1300
|
+
if system is None:
|
|
1301
|
+
system = platform.system()
|
|
1302
|
+
if system == "Windows":
|
|
1303
|
+
return prefix.replace("/", "\\") + "\\Scripts\\activate"
|
|
1304
|
+
return f". {prefix}/bin/activate"
|
|
1305
|
+
|
|
1306
|
+
|
|
1293
1307
|
def env_from_name_or_path(
|
|
1294
1308
|
name_or_path: str | None = None,
|
|
1295
1309
|
ck_info: dict | None = None,
|
calkit/invenio.py
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
"""Functionality for working with InvenioRDM instances like Zenodo."""
|
|
2
2
|
|
|
3
3
|
import os
|
|
4
|
+
import time
|
|
4
5
|
from functools import partial
|
|
5
6
|
from typing import Literal
|
|
6
7
|
|
|
@@ -57,6 +58,24 @@ def get_base_url(service: ServiceName = DEFAULT_SERVICE) -> str:
|
|
|
57
58
|
raise ValueError(f"Unknown archival service '{service}'")
|
|
58
59
|
|
|
59
60
|
|
|
61
|
+
# Pushing a release can mean sending hundreds of megabytes over a slow or
|
|
62
|
+
# distant link, so give a response plenty of time to arrive rather than
|
|
63
|
+
# letting requests wait forever with no timeout at all.
|
|
64
|
+
CONNECT_TIMEOUT = 30
|
|
65
|
+
READ_TIMEOUT = 600
|
|
66
|
+
# Gateways in front of InvenioRDM return these when they're busy, which says
|
|
67
|
+
# nothing about whether the request itself was valid, so it's worth sending
|
|
68
|
+
# again after a pause.
|
|
69
|
+
RETRY_STATUS_CODES = {429, 500, 502, 503, 504}
|
|
70
|
+
MAX_ATTEMPTS = 4
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def get_timeout() -> tuple[float, float]:
|
|
74
|
+
"""Get the connect and read timeouts to use for requests."""
|
|
75
|
+
read = os.getenv("CALKIT_INVENIO_TIMEOUT")
|
|
76
|
+
return CONNECT_TIMEOUT, float(read) if read else READ_TIMEOUT
|
|
77
|
+
|
|
78
|
+
|
|
60
79
|
def _request(
|
|
61
80
|
kind: Literal["get", "post", "put", "patch", "delete"],
|
|
62
81
|
path: str,
|
|
@@ -73,15 +92,31 @@ def _request(
|
|
|
73
92
|
params = {}
|
|
74
93
|
if auth and "access_token" not in params:
|
|
75
94
|
params = params | {"access_token": get_token(service=service)}
|
|
95
|
+
kwargs.setdefault("timeout", get_timeout())
|
|
76
96
|
func = getattr(requests, kind)
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
97
|
+
# A POST that timed out may still have been carried out on the far end,
|
|
98
|
+
# e.g., leaving a draft record behind, so only repeat requests that are
|
|
99
|
+
# safe to send twice
|
|
100
|
+
max_attempts = 1 if kind == "post" else MAX_ATTEMPTS
|
|
101
|
+
for attempt in range(1, max_attempts + 1):
|
|
102
|
+
try:
|
|
103
|
+
resp = func(
|
|
104
|
+
get_base_url(service=service) + path,
|
|
105
|
+
params=params,
|
|
106
|
+
json=json,
|
|
107
|
+
data=data,
|
|
108
|
+
headers=headers,
|
|
109
|
+
**kwargs,
|
|
110
|
+
)
|
|
111
|
+
except (requests.ConnectionError, requests.Timeout):
|
|
112
|
+
if attempt == max_attempts:
|
|
113
|
+
raise
|
|
114
|
+
time.sleep(2**attempt)
|
|
115
|
+
continue
|
|
116
|
+
if resp.status_code in RETRY_STATUS_CODES and attempt < max_attempts:
|
|
117
|
+
time.sleep(2**attempt)
|
|
118
|
+
continue
|
|
119
|
+
break
|
|
85
120
|
if resp.status_code >= 400:
|
|
86
121
|
msg = f"{resp.status_code}: "
|
|
87
122
|
try:
|
|
@@ -92,6 +127,14 @@ def _request(
|
|
|
92
127
|
msg += f"\nErrors:\n{resp_json['errors']}"
|
|
93
128
|
except ValueError:
|
|
94
129
|
msg += resp.text
|
|
130
|
+
if kind == "post" and resp.status_code in RETRY_STATUS_CODES:
|
|
131
|
+
# The far end may have done the work anyway, and a blind retry
|
|
132
|
+
# would duplicate it, so say so rather than papering over it
|
|
133
|
+
msg += (
|
|
134
|
+
f"\nThis request was not retried automatically, since "
|
|
135
|
+
f"{service} may have carried it out despite the error. "
|
|
136
|
+
"Check for a leftover draft record before trying again."
|
|
137
|
+
)
|
|
95
138
|
raise HTTPError(msg)
|
|
96
139
|
resp.raise_for_status()
|
|
97
140
|
if as_json:
|