calkit-python 0.47.0__py3-none-any.whl → 0.47.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. calkit/cli/check.py +22 -13
  2. calkit/cli/main/core.py +1 -5
  3. calkit/cli/new.py +157 -30
  4. calkit/detect.py +93 -11
  5. calkit/docker.py +8 -6
  6. calkit/environments.py +14 -0
  7. calkit/invenio.py +51 -8
  8. calkit/models/core.py +38 -0
  9. calkit/pipeline.py +94 -0
  10. calkit/questions.py +313 -48
  11. calkit/releases.py +190 -7
  12. calkit/resources/devcontainer/Dockerfile +6 -8
  13. calkit/tests/cli/test_check.py +30 -8
  14. calkit/tests/cli/test_new.py +212 -0
  15. calkit/tests/test_detect.py +74 -1
  16. calkit/tests/test_environments.py +28 -0
  17. calkit/tests/test_invenio.py +79 -0
  18. calkit/tests/test_questions.py +182 -8
  19. calkit/tests/test_releases.py +78 -5
  20. {calkit_python-0.47.0.dist-info → calkit_python-0.47.2.dist-info}/METADATA +2 -1
  21. {calkit_python-0.47.0.dist-info → calkit_python-0.47.2.dist-info}/RECORD +38 -38
  22. {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/etc/jupyter/jupyter_server_config.d/calkit.json +0 -0
  23. {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/package.json +0 -0
  24. {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/schemas/calkit/package.json.orig +0 -0
  25. {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/schemas/calkit/plugin.json +0 -0
  26. {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/502.9a2c5772a15466e923ef.js +0 -0
  27. {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/695.2c41003a452d43d2b358.js +0 -0
  28. {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/867.a42a046aa5108f54f8fb.js +0 -0
  29. {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/909.6d8285ce7c45878ac508.js +0 -0
  30. {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/946.050af2abf7845cfbdbd2.js +0 -0
  31. {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/946.050af2abf7845cfbdbd2.js.LICENSE.txt +0 -0
  32. {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/b2f1c3efe70cb539d121.png +0 -0
  33. {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/remoteEntry.d7a43c7948f690d37d19.js +0 -0
  34. {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/style.js +0 -0
  35. {calkit_python-0.47.0.data → calkit_python-0.47.2.data}/data/share/jupyter/labextensions/calkit/static/third-party-licenses.json +0 -0
  36. {calkit_python-0.47.0.dist-info → calkit_python-0.47.2.dist-info}/WHEEL +0 -0
  37. {calkit_python-0.47.0.dist-info → calkit_python-0.47.2.dist-info}/entry_points.txt +0 -0
  38. {calkit_python-0.47.0.dist-info → calkit_python-0.47.2.dist-info}/licenses/LICENSE +0 -0
calkit/cli/check.py CHANGED
@@ -1802,10 +1802,7 @@ def check_venv(
1802
1802
  if verbose:
1803
1803
  typer.echo(f"Using legacy lock file: {legacy_fpath}")
1804
1804
  break
1805
- if _platform.system() == "Windows":
1806
- activate_cmd = f"{prefix}\\Scripts\\activate"
1807
- else:
1808
- activate_cmd = f". {prefix}/bin/activate"
1805
+ activate_cmd = calkit.environments.get_venv_activate_cmd(prefix)
1809
1806
 
1810
1807
  def pip_install_and_freeze(reqs_arg: str) -> None:
1811
1808
  check_cmd = (
@@ -2060,20 +2057,32 @@ def check_questions(
2060
2057
  json_output: Annotated[
2061
2058
  bool, typer.Option("--json", help="Output the report as JSON.")
2062
2059
  ] = False,
2060
+ no_pipeline: Annotated[
2061
+ bool,
2062
+ typer.Option(
2063
+ "--no-pipeline",
2064
+ help="Skip asking DVC which stages are out of date, which is "
2065
+ "the slowest part of the check.",
2066
+ ),
2067
+ ] = False,
2063
2068
  ) -> None:
2064
- """Check that answered questions are consistent with their evidence.
2065
-
2066
- A question is stale if any of its evidence changed after the commit
2067
- that last edited the question, in Git history for Git-tracked outputs
2068
- or in dvc.lock for DVC-tracked ones. Evidence paths must exist, value
2069
- keys must resolve, every placeholder in the text must render, and a
2070
- publication label must still be present in the LaTeX source. Exits
2071
- with an error if any answered question is stale or broken.
2069
+ """Check that answered questions are backed by current evidence.
2070
+
2071
+ Reports, worst first: evidence that isn't there (never run, never
2072
+ pushed, or pinned to a Git ref that doesn't exist); broken references
2073
+ (a key that doesn't resolve, a placeholder that names no evidence, a
2074
+ label missing from the LaTeX); evidence the pipeline would rebuild;
2075
+ and evidence from a frozen stage, or downstream of one, which nothing
2076
+ will ever report out of date unless the citation pins a git_ref.
2077
+
2078
+ Evidence pinned with a git_ref is checked at that ref rather than in
2079
+ the working tree. Exits with an error if any answered question is
2080
+ missing evidence, broken, or out of date with the pipeline.
2072
2081
  """
2073
2082
  from calkit.questions import check_questions as _check_questions
2074
2083
  from calkit.questions import format_status
2075
2084
 
2076
- status = _check_questions(wdir=wdir)
2085
+ status = _check_questions(wdir=wdir, check_pipeline=not no_pipeline)
2077
2086
  if json_output:
2078
2087
  calkit.echo(json.dumps(status.model_dump(mode="json"), indent=2))
2079
2088
  else:
calkit/cli/main/core.py CHANGED
@@ -6,7 +6,6 @@ import csv
6
6
  import json
7
7
  import logging
8
8
  import os
9
- import platform as _platform
10
9
  import posixpath
11
10
  import shlex
12
11
  import shutil
@@ -3543,10 +3542,7 @@ def run_in_env(
3543
3542
  envs, path, env_name
3544
3543
  )
3545
3544
  shell_cmd = _to_shell_cmd(cmd)
3546
- if _platform.system() == "Windows":
3547
- activate_cmd = f"{prefix}\\Scripts\\activate"
3548
- else:
3549
- activate_cmd = f". {prefix}/bin/activate"
3545
+ activate_cmd = calkit.environments.get_venv_activate_cmd(prefix)
3550
3546
  if verbose:
3551
3547
  typer.echo(f"Raw command: {cmd}")
3552
3548
  typer.echo(f"Shell command: {shell_cmd}")
calkit/cli/new.py CHANGED
@@ -3434,6 +3434,18 @@ def new_release(
3434
3434
  str | None,
3435
3435
  typer.Option("--date", help="Release date. Will default to today."),
3436
3436
  ] = None,
3437
+ include_pipeline: Annotated[
3438
+ bool,
3439
+ typer.Option(
3440
+ "--pipeline",
3441
+ help=(
3442
+ "Include everything needed to reproduce the released path, "
3443
+ "i.e., the pipeline, its lock file, and the stages, inputs, "
3444
+ "and environments the path depends on. Stages unrelated to "
3445
+ "the path are left out."
3446
+ ),
3447
+ ),
3448
+ ] = False,
3437
3449
  no_docker_images: Annotated[
3438
3450
  bool,
3439
3451
  typer.Option(
@@ -3521,7 +3533,6 @@ def new_release(
3521
3533
  ] = False,
3522
3534
  ):
3523
3535
  """Create a new release."""
3524
- import bibtexparser
3525
3536
  import dotenv
3526
3537
 
3527
3538
  import calkit.pipeline
@@ -3536,6 +3547,20 @@ def new_release(
3536
3547
  repo = calkit.git.get_repo()
3537
3548
  if name in repo.tags:
3538
3549
  raise_error(f"Git tag with name '{name}' already exists")
3550
+ # A release commits to calkit.yaml and pushes the branch it's on, neither
3551
+ # of which works from a detached HEAD. Check before anything is uploaded,
3552
+ # so a release can't get published and then fail on the way out.
3553
+ will_push = (
3554
+ not dry_run and not no_push and not no_commit and not draft_only
3555
+ )
3556
+ if repo.head.is_detached and will_push:
3557
+ # Suggest creating a branch rather than checking one out, since in a
3558
+ # worktree the branch they'd want may be checked out elsewhere
3559
+ raise_error(
3560
+ "HEAD is detached, so there is no branch to commit the release "
3561
+ "record to and push; create a branch at this revision first, "
3562
+ "e.g., with `git switch -c <branch>`"
3563
+ )
3539
3564
  # Detect the release kind from the path unless it was given with --kind. A
3540
3565
  # "." path is always a project release; otherwise prefer a declared
3541
3566
  # artifact in calkit.yaml, falling back to auto-detection from the path
@@ -3589,10 +3614,24 @@ def new_release(
3589
3614
  # that produces the released artifact when releasing a single path.
3590
3615
  typer.echo("Checking pipeline is up-to-date for release")
3591
3616
  targets = None
3617
+ # The stage that builds the released path, whose upstream stages define
3618
+ # what a --pipeline release carries
3619
+ pipeline_stage = ""
3592
3620
  if path != ".":
3593
3621
  stage_name = calkit.pipeline.get_stage_for_output(path, ck_info)
3594
- if stage_name is not None:
3622
+ if stage_name is None:
3623
+ if include_pipeline:
3624
+ raise_error(
3625
+ f"No pipeline stage produces '{path}', "
3626
+ "so there is no pipeline to release along with it"
3627
+ )
3628
+ else:
3595
3629
  targets = [stage_name]
3630
+ pipeline_stage = stage_name
3631
+ elif include_pipeline:
3632
+ # A project release carries the whole pipeline already
3633
+ typer.echo("Project releases already include the pipeline")
3634
+ include_pipeline = False
3596
3635
  status = calkit.pipeline.get_status(
3597
3636
  ck_info=ck_info,
3598
3637
  targets=targets,
@@ -3617,6 +3656,15 @@ def new_release(
3617
3656
  release_date = str(calkit.utcnow().date())
3618
3657
  typer.echo(f"Using release date: {release_date}")
3619
3658
  git_rev = repo.git.rev_parse(["--short", "HEAD"])
3659
+ # This goes both beside the archive, which is the copy the archival
3660
+ # service displays, and inside it, so an extracted copy still says what
3661
+ # produced it. Rebuilt below once a more specific title is known.
3662
+ release_readme = calkit.releases.create_release_readme(
3663
+ release_kind=release_kind,
3664
+ name=name,
3665
+ git_rev=git_rev,
3666
+ title=ck_info.get("title"),
3667
+ )
3620
3668
  # Fields below are populated only for external (archival) releases;
3621
3669
  # internal releases leave them empty.
3622
3670
  doi = None
@@ -3636,9 +3684,15 @@ def new_release(
3636
3684
  stored_filename = f"{project_name}-{name}.zip"
3637
3685
  is_zip = True
3638
3686
  elif os.path.isfile(path):
3639
- _, ext = os.path.splitext(os.path.basename(path))
3640
- stored_filename = f"{project_name}-{name}{ext}"
3641
- is_zip = False
3687
+ # Releasing the pipeline along with the artifact means shipping
3688
+ # more than one file, so the artifact gets zipped up with it
3689
+ if include_pipeline:
3690
+ stored_filename = f"{project_name}-{name}.zip"
3691
+ is_zip = True
3692
+ else:
3693
+ _, ext = os.path.splitext(os.path.basename(path))
3694
+ stored_filename = f"{project_name}-{name}{ext}"
3695
+ is_zip = False
3642
3696
  else:
3643
3697
  raise_error(f"Release path '{path}' does not exist")
3644
3698
  stored_path = os.path.join(release_dir, stored_filename)
@@ -3648,10 +3702,38 @@ def new_release(
3648
3702
  typer.echo(f"Would {action} {path} to {stored_path_posix}")
3649
3703
  else:
3650
3704
  os.makedirs(release_dir, exist_ok=True)
3651
- if is_zip:
3705
+ overrides: dict[str, str] = {}
3706
+ if include_pipeline:
3707
+ typer.echo(f"Pruning project to what builds {path}")
3708
+ try:
3709
+ overrides, paths = calkit.releases.prune_for_stage(
3710
+ ck_info, pipeline_stage
3711
+ )
3712
+ except Exception as e:
3713
+ raise_error(
3714
+ f"Failed to prune project for stage "
3715
+ f"'{pipeline_stage}': {e}"
3716
+ )
3717
+ elif is_zip:
3652
3718
  paths = calkit.releases.ls_files() if path == "." else [path]
3719
+ else:
3720
+ paths = []
3721
+ if is_zip:
3653
3722
  typer.echo(f"Archiving {path} to {stored_path_posix}")
3654
- calkit.releases.zip_paths(stored_path, paths)
3723
+ calkit.releases.zip_paths(
3724
+ stored_path,
3725
+ paths,
3726
+ overrides=overrides
3727
+ | {"CALKIT-RELEASE.md": release_readme},
3728
+ )
3729
+ if include_pipeline:
3730
+ typer.echo("Checking extracted release archive")
3731
+ try:
3732
+ calkit.releases.check_project_release_archive(
3733
+ stored_path, verbose=verbose
3734
+ )
3735
+ except Exception as e:
3736
+ raise_error(str(e))
3655
3737
  else:
3656
3738
  typer.echo(f"Copying {path} to {stored_path_posix}")
3657
3739
  shutil.copy2(path, stored_path)
@@ -3679,10 +3761,30 @@ def new_release(
3679
3761
  if path == ".":
3680
3762
  if release_kind is None:
3681
3763
  release_kind = "project"
3764
+ # Settle the title before building the archive, since the README
3765
+ # that goes inside it is headed with the title
3766
+ title = ck_info.get("title")
3767
+ if title is None:
3768
+ warn("Project has no title")
3769
+ title = typer.prompt("Enter a title for the project")
3770
+ ck_info["title"] = title
3771
+ if not dry_run:
3772
+ with open("calkit.yaml", "w") as f:
3773
+ calkit.ryaml.dump(ck_info, f)
3774
+ release_readme = calkit.releases.create_release_readme(
3775
+ release_kind=release_kind,
3776
+ name=name,
3777
+ git_rev=git_rev,
3778
+ title=title,
3779
+ )
3682
3780
  zip_path = release_files_dir + "/archive.zip"
3683
3781
  all_paths = calkit.releases.ls_files()
3684
3782
  typer.echo(f"Adding files to {zip_path}")
3685
- calkit.releases.zip_paths(zip_path, all_paths)
3783
+ calkit.releases.zip_paths(
3784
+ zip_path,
3785
+ all_paths,
3786
+ overrides={"CALKIT-RELEASE.md": release_readme},
3787
+ )
3686
3788
  typer.echo("Checking extracted project release archive")
3687
3789
  try:
3688
3790
  calkit.releases.check_project_release_archive(
@@ -3690,14 +3792,6 @@ def new_release(
3690
3792
  )
3691
3793
  except Exception as e:
3692
3794
  raise_error(str(e))
3693
- title = ck_info.get("title")
3694
- if title is None:
3695
- warn("Project has no title")
3696
- title = typer.prompt("Enter a title for the project")
3697
- ck_info["title"] = title
3698
- if not dry_run:
3699
- with open("calkit.yaml", "w") as f:
3700
- calkit.ryaml.dump(ck_info, f)
3701
3795
  else:
3702
3796
  # TODO: Handle directories, e.g., datasets
3703
3797
  if not os.path.isfile(path):
@@ -3724,9 +3818,46 @@ def new_release(
3724
3818
  )
3725
3819
  if title is None:
3726
3820
  raise_error(f"{release_kind} at {path} has no title")
3821
+ release_readme = calkit.releases.create_release_readme(
3822
+ release_kind=release_kind,
3823
+ name=name,
3824
+ git_rev=git_rev,
3825
+ title=title,
3826
+ )
3827
+ # Ship the artifact's provenance beside it: the stages that
3828
+ # build it, their inputs and environments, and a pipeline and
3829
+ # lock file pruned to match
3830
+ if include_pipeline:
3831
+ zip_path = release_files_dir + "/archive.zip"
3832
+ typer.echo(f"Pruning project to what builds {path}")
3833
+ try:
3834
+ overrides, all_paths = calkit.releases.prune_for_stage(
3835
+ ck_info, pipeline_stage
3836
+ )
3837
+ except Exception as e:
3838
+ raise_error(
3839
+ f"Failed to prune project for stage "
3840
+ f"'{pipeline_stage}': {e}"
3841
+ )
3842
+ typer.echo(f"Adding files to {zip_path}")
3843
+ calkit.releases.zip_paths(
3844
+ zip_path,
3845
+ all_paths,
3846
+ overrides=overrides
3847
+ | {"CALKIT-RELEASE.md": release_readme},
3848
+ )
3849
+ typer.echo("Checking extracted project release archive")
3850
+ try:
3851
+ calkit.releases.check_project_release_archive(
3852
+ zip_path, verbose=verbose
3853
+ )
3854
+ except Exception as e:
3855
+ raise_error(str(e))
3727
3856
  # Save a metadata file with each DVC file's MD5 checksum
3728
3857
  dvc_md5s = calkit.releases.make_dvc_md5s(
3729
- zipfile="archive.zip" if path == "." else None,
3858
+ zipfile=(
3859
+ "archive.zip" if path == "." or include_pipeline else None
3860
+ ),
3730
3861
  paths=None if path == "." else [path],
3731
3862
  )
3732
3863
  dvc_md5s_path = release_dir + "/dvc-md5s.yaml"
@@ -3738,7 +3869,7 @@ def new_release(
3738
3869
  # Archive the project's Docker images, so reproducing it doesn't
3739
3870
  # depend on a registry keeping them around, and leave breadcrumbs
3740
3871
  # behind so the environment check can fetch them back
3741
- if path == "." and not no_docker_images:
3872
+ if (path == "." or include_pipeline) and not no_docker_images:
3742
3873
  typer.echo("Archiving Docker images")
3743
3874
  docker_images = calkit.releases.save_docker_images(
3744
3875
  release_files_dir
@@ -3752,16 +3883,11 @@ def new_release(
3752
3883
  calkit.ryaml.dump(docker_images, f)
3753
3884
  if not dry_run:
3754
3885
  repo.git.add(docker_images_path)
3755
- # Create a README for the Zenodo release
3756
- readme_txt = f"# {title}\n"
3757
- git_rev = repo.git.rev_parse(["--short", "HEAD"])
3758
- readme_txt += (
3759
- f"\nThis is a {release_kind} release ({name}) generated with "
3760
- f"Calkit v{calkit.__version__} from Git rev {git_rev}.\n"
3761
- )
3886
+ # Write the same README beside the archive, since this is the copy
3887
+ # the archival service renders on the record page
3762
3888
  readme_path = release_files_dir + "/README.md"
3763
3889
  with open(readme_path, "w") as f:
3764
- f.write(readme_txt)
3890
+ f.write(release_readme)
3765
3891
  # Check size of files dir
3766
3892
  size = calkit.get_size(release_files_dir)
3767
3893
  typer.echo(f"Release size: {(size / 1e6):.1f} MB")
@@ -4114,6 +4240,7 @@ def new_release(
4114
4240
  description=release_description,
4115
4241
  internal=internal_release,
4116
4242
  stored_path=stored_path_posix,
4243
+ includes_pipeline=include_pipeline,
4117
4244
  ).model_dump()
4118
4245
  releases[name] = release
4119
4246
  ck_info["releases"] = releases
@@ -4167,7 +4294,7 @@ def new_release(
4167
4294
  record_id=record_id, # type: ignore
4168
4295
  service=to, # type: ignore
4169
4296
  )
4170
- new_entries = bibtexparser.loads(invenio_bibtex).entries
4297
+ new_entries = calkit.releases.parse_bibtex(invenio_bibtex)
4171
4298
  if not new_entries:
4172
4299
  raise ValueError("Failed to parse generated BibTeX entry")
4173
4300
  new_entry = new_entries[0]
@@ -4179,9 +4306,9 @@ def new_release(
4179
4306
  replace_ids = []
4180
4307
  if new_doi:
4181
4308
  try:
4182
- existing_entries = bibtexparser.loads(
4309
+ existing_entries = calkit.releases.parse_bibtex(
4183
4310
  existing_text
4184
- ).entries
4311
+ )
4185
4312
  except Exception as e:
4186
4313
  warn(f"Could not parse existing references to dedupe: {e}")
4187
4314
  existing_entries = []
@@ -4220,7 +4347,7 @@ def new_release(
4220
4347
  if not dry_run and calkit.git.get_staged_files() and not no_commit:
4221
4348
  repo.git.commit(["-m", f"Create new {release_kind} release {name}"])
4222
4349
  # Push with Git
4223
- if not dry_run and not no_push and not no_commit and not draft_only:
4350
+ if will_push:
4224
4351
  repo.git.push(["origin", repo.active_branch.name, "--tags"])
4225
4352
  # Now create GitHub release (external releases only)
4226
4353
  if not internal_release and not no_github_release:
calkit/detect.py CHANGED
@@ -1773,40 +1773,119 @@ def detect_r_dependencies(
1773
1773
  def detect_julia_dependencies(
1774
1774
  script_path: str | None = None,
1775
1775
  code: str | None = None,
1776
+ script_dir: str | None = None,
1777
+ project_dir: str = ".",
1776
1778
  ) -> list[str]:
1777
1779
  """Detect package dependencies from a Julia script or code string.
1778
1780
 
1781
+ Julia's ``include`` splices a file in as source text, so any package used
1782
+ by an included file must be declared by the project that includes it.
1783
+ Includes with a literal path that resolve inside the project are therefore
1784
+ followed. Ones pointing outside it are not, since that code declares its
1785
+ dependencies in its own project file, e.g., a package's own source in the
1786
+ depot reached via ``pkgdir``.
1787
+
1779
1788
  Parameters
1780
1789
  ----------
1781
1790
  script_path : str | None
1782
1791
  Path to Julia script. Either this or code must be provided.
1783
1792
  code : str | None
1784
1793
  Julia code string. Either this or script_path must be provided.
1794
+ script_dir : str | None
1795
+ Directory the code came from, against which its includes resolve.
1796
+ Only used with ``code``; defaults to ``project_dir``.
1797
+ project_dir : str
1798
+ Project root, outside of which includes are not followed.
1785
1799
 
1786
1800
  Returns
1787
1801
  -------
1788
1802
  list[str]
1789
1803
  List of Julia package names.
1790
1804
  """
1805
+
1806
+ def parse_dependencies(code: str) -> set[str]:
1807
+ deps = set()
1808
+ # Both `using` and `import` load a package, either can start a line or
1809
+ # follow a semicolon, and either can be prefixed by macros, e.g.,
1810
+ # `@everywhere using Foo`
1811
+ clauses = re.findall(
1812
+ r"(?:^|;)[ \t]*(?:@[A-Za-z_][A-Za-z0-9_!]*[ \t]+)*"
1813
+ r"(?:using|import)[ \t]+([^\n;]+)",
1814
+ code,
1815
+ flags=re.MULTILINE,
1816
+ )
1817
+ for clause in clauses:
1818
+ # In `using Foo: bar, baz` only what precedes the colon is a
1819
+ # package
1820
+ clause = clause.split(":")[0]
1821
+ for part in clause.split(","):
1822
+ # Drop an `as` alias, e.g., `import Foo as F`
1823
+ name = re.split(r"\s+as\s+", part.strip())[0].strip()
1824
+ # A leading dot means a module local to this file, not a
1825
+ # package
1826
+ if not name or name.startswith("."):
1827
+ continue
1828
+ # Submodules like `Foo.Bar` come from the `Foo` package
1829
+ name = name.split(".")[0]
1830
+ if not re.fullmatch(r"[A-Za-z_][A-Za-z0-9_!]*", name):
1831
+ continue
1832
+ if name in ("Base", "Core", "Main"):
1833
+ continue
1834
+ deps.add(name)
1835
+ return deps
1836
+
1837
+ def find_includes(code: str, from_dir: str) -> list[str]:
1838
+ paths = []
1839
+ for match in re.findall(
1840
+ r'include\s*\(\s*["\']([^"\']+\.jl)["\']\s*\)', code
1841
+ ):
1842
+ if os.path.isabs(match):
1843
+ continue
1844
+ path = os.path.realpath(os.path.join(from_dir, match))
1845
+ if not path.startswith(root + os.sep) or not os.path.isfile(path):
1846
+ continue
1847
+ paths.append(path)
1848
+ return paths
1849
+
1850
+ def read(path: str) -> str | None:
1851
+ try:
1852
+ with open(path, "r", encoding="utf-8") as f:
1853
+ return f.read()
1854
+ except (UnicodeDecodeError, IOError):
1855
+ return None
1856
+
1791
1857
  if script_path is None and code is None:
1792
1858
  raise ValueError("Either script_path or code must be provided")
1859
+ root = os.path.realpath(project_dir)
1793
1860
  if code is None:
1794
1861
  assert script_path is not None # Type guard
1795
1862
  if not os.path.exists(script_path):
1796
1863
  return []
1797
- try:
1798
- with open(script_path, "r", encoding="utf-8") as f:
1799
- code = f.read()
1800
- except (UnicodeDecodeError, IOError):
1864
+ code = read(script_path)
1865
+ if code is None:
1801
1866
  return []
1802
- assert code is not None # Type guard
1803
- dependencies = set()
1867
+ from_dir = os.path.dirname(script_path) or "."
1868
+ seen = {os.path.realpath(script_path)}
1869
+ else:
1870
+ from_dir = script_dir if script_dir is not None else project_dir
1871
+ seen = set()
1804
1872
  # Remove comments
1805
1873
  code = re.sub(r"#.*$", "", code, flags=re.MULTILINE)
1806
- # Pattern for using statements
1807
- pattern = r"using\s+([a-zA-Z0-9._]+)"
1808
- matches = re.findall(pattern, code)
1809
- dependencies.update(matches)
1874
+ dependencies = parse_dependencies(code)
1875
+ # Walk the include tree, resolving each file's includes against its own
1876
+ # directory the way Julia does
1877
+ queue = find_includes(code, from_dir)
1878
+ while queue:
1879
+ path = queue.pop(0)
1880
+ if path in seen:
1881
+ continue
1882
+ seen.add(path)
1883
+ included = read(path)
1884
+ if included is None:
1885
+ continue
1886
+ included = re.sub(r"#.*$", "", included, flags=re.MULTILINE)
1887
+ dependencies |= parse_dependencies(included)
1888
+ queue += find_includes(included, os.path.dirname(path))
1810
1889
  return sorted(list(dependencies))
1811
1890
 
1812
1891
 
@@ -1855,7 +1934,10 @@ def detect_dependencies_from_notebook(
1855
1934
  if language == "python":
1856
1935
  return detect_python_dependencies(code=combined_code)
1857
1936
  elif language == "julia":
1858
- return detect_julia_dependencies(code=combined_code)
1937
+ return detect_julia_dependencies(
1938
+ code=combined_code,
1939
+ script_dir=os.path.dirname(notebook_path) or ".",
1940
+ )
1859
1941
  elif language == "r":
1860
1942
  return detect_r_dependencies(code=combined_code)
1861
1943
  return []
calkit/docker.py CHANGED
@@ -14,7 +14,7 @@ from pydantic import BaseModel
14
14
  MINIFORGE_LAYER_TXT = r"""
15
15
  # Install Miniforge
16
16
  ARG MINIFORGE_NAME=Miniforge3
17
- ARG MINIFORGE_VERSION=24.9.2-0
17
+ ARG MINIFORGE_VERSION=26.7.2-0
18
18
  ARG TARGETPLATFORM
19
19
 
20
20
  ENV CONDA_DIR=/opt/conda
@@ -51,24 +51,26 @@ RUN apt-get update > /dev/null && \
51
51
  echo ". ${CONDA_DIR}/etc/profile.d/conda.sh && conda activate base" >> ~/.bashrc
52
52
  """.strip()
53
53
 
54
+ # foamPy ships only an sdist whose setup.py imports numpy, so it needs the
55
+ # surrounding environment rather than an isolated build one
54
56
  FOAMPY_LAYER_TEXT = r"""
55
57
  RUN pip install --no-cache-dir numpy pandas matplotlib h5py \
56
58
  && pip install --no-cache-dir scipy \
57
- && pip install --no-cache-dir foampy
59
+ && pip install --no-cache-dir --no-build-isolation foampy
58
60
  """.strip()
59
61
 
60
62
  UV_LAYER_TEXT = """
61
- COPY --from=ghcr.io/astral-sh/uv:0.8.5 /uv /uvx /bin/
63
+ COPY --from=ghcr.io/astral-sh/uv:0.12.11 /uv /uvx /bin/
62
64
  """
63
65
 
64
66
  JULIA_LAYER_TEXT = """
65
67
  # Install Julia
66
- # Ensure base image is a bullseye distribution
67
- COPY --from=julia:1.11.6-bullseye /usr/local/julia /usr/local/julia
68
+ # Ensure base image is a bookworm distribution
69
+ COPY --from=julia:1.11.9-bookworm /usr/local/julia /usr/local/julia
68
70
  ENV JULIA_PATH=/usr/local/julia \
69
71
  PATH=$PATH:/usr/local/julia/bin \
70
72
  JULIA_GPG=3673DF529D9049477F76B37566E3C7DC03D6E495 \
71
- JULIA_VERSION=1.11.6
73
+ JULIA_VERSION=1.11.9
72
74
  """
73
75
 
74
76
  LAYERS = {
calkit/environments.py CHANGED
@@ -1290,6 +1290,20 @@ def get_default_venv_prefix(envs: dict, path: str, name: str) -> str:
1290
1290
  return Path(base).as_posix()
1291
1291
 
1292
1292
 
1293
+ def get_venv_activate_cmd(prefix: str, system: str | None = None) -> str:
1294
+ """Get the shell command that activates the virtualenv at ``prefix``.
1295
+
1296
+ Prefixes are kept POSIX-style, but cmd reads a forward slash as the start
1297
+ of a switch, so it takes ``.calkit/envs/x/.venv`` for a command named
1298
+ ``.calkit``. Hand Windows native separators instead.
1299
+ """
1300
+ if system is None:
1301
+ system = platform.system()
1302
+ if system == "Windows":
1303
+ return prefix.replace("/", "\\") + "\\Scripts\\activate"
1304
+ return f". {prefix}/bin/activate"
1305
+
1306
+
1293
1307
  def env_from_name_or_path(
1294
1308
  name_or_path: str | None = None,
1295
1309
  ck_info: dict | None = None,
calkit/invenio.py CHANGED
@@ -1,6 +1,7 @@
1
1
  """Functionality for working with InvenioRDM instances like Zenodo."""
2
2
 
3
3
  import os
4
+ import time
4
5
  from functools import partial
5
6
  from typing import Literal
6
7
 
@@ -57,6 +58,24 @@ def get_base_url(service: ServiceName = DEFAULT_SERVICE) -> str:
57
58
  raise ValueError(f"Unknown archival service '{service}'")
58
59
 
59
60
 
61
+ # Pushing a release can mean sending hundreds of megabytes over a slow or
62
+ # distant link, so give a response plenty of time to arrive rather than
63
+ # letting requests wait forever with no timeout at all.
64
+ CONNECT_TIMEOUT = 30
65
+ READ_TIMEOUT = 600
66
+ # Gateways in front of InvenioRDM return these when they're busy, which says
67
+ # nothing about whether the request itself was valid, so it's worth sending
68
+ # again after a pause.
69
+ RETRY_STATUS_CODES = {429, 500, 502, 503, 504}
70
+ MAX_ATTEMPTS = 4
71
+
72
+
73
+ def get_timeout() -> tuple[float, float]:
74
+ """Get the connect and read timeouts to use for requests."""
75
+ read = os.getenv("CALKIT_INVENIO_TIMEOUT")
76
+ return CONNECT_TIMEOUT, float(read) if read else READ_TIMEOUT
77
+
78
+
60
79
  def _request(
61
80
  kind: Literal["get", "post", "put", "patch", "delete"],
62
81
  path: str,
@@ -73,15 +92,31 @@ def _request(
73
92
  params = {}
74
93
  if auth and "access_token" not in params:
75
94
  params = params | {"access_token": get_token(service=service)}
95
+ kwargs.setdefault("timeout", get_timeout())
76
96
  func = getattr(requests, kind)
77
- resp = func(
78
- get_base_url(service=service) + path,
79
- params=params,
80
- json=json,
81
- data=data,
82
- headers=headers,
83
- **kwargs,
84
- )
97
+ # A POST that timed out may still have been carried out on the far end,
98
+ # e.g., leaving a draft record behind, so only repeat requests that are
99
+ # safe to send twice
100
+ max_attempts = 1 if kind == "post" else MAX_ATTEMPTS
101
+ for attempt in range(1, max_attempts + 1):
102
+ try:
103
+ resp = func(
104
+ get_base_url(service=service) + path,
105
+ params=params,
106
+ json=json,
107
+ data=data,
108
+ headers=headers,
109
+ **kwargs,
110
+ )
111
+ except (requests.ConnectionError, requests.Timeout):
112
+ if attempt == max_attempts:
113
+ raise
114
+ time.sleep(2**attempt)
115
+ continue
116
+ if resp.status_code in RETRY_STATUS_CODES and attempt < max_attempts:
117
+ time.sleep(2**attempt)
118
+ continue
119
+ break
85
120
  if resp.status_code >= 400:
86
121
  msg = f"{resp.status_code}: "
87
122
  try:
@@ -92,6 +127,14 @@ def _request(
92
127
  msg += f"\nErrors:\n{resp_json['errors']}"
93
128
  except ValueError:
94
129
  msg += resp.text
130
+ if kind == "post" and resp.status_code in RETRY_STATUS_CODES:
131
+ # The far end may have done the work anyway, and a blind retry
132
+ # would duplicate it, so say so rather than papering over it
133
+ msg += (
134
+ f"\nThis request was not retried automatically, since "
135
+ f"{service} may have carried it out despite the error. "
136
+ "Check for a leftover draft record before trying again."
137
+ )
95
138
  raise HTTPError(msg)
96
139
  resp.raise_for_status()
97
140
  if as_json: