reader-workbench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- reader_workbench/__init__.py +22 -0
- reader_workbench/__main__.py +4 -0
- reader_workbench/_version.py +17 -0
- reader_workbench/api/__init__.py +74 -0
- reader_workbench/api/_record_reads.py +75 -0
- reader_workbench/api/artifacts.py +79 -0
- reader_workbench/api/facade.py +538 -0
- reader_workbench/api/models.py +285 -0
- reader_workbench/api/notebooks.py +63 -0
- reader_workbench/contracts/__init__.py +18 -0
- reader_workbench/contracts/builtins/__init__.py +36 -0
- reader_workbench/contracts/builtins/cytometry.py +140 -0
- reader_workbench/contracts/builtins/four_state_event_window.py +214 -0
- reader_workbench/contracts/builtins/generic.py +21 -0
- reader_workbench/contracts/builtins/logic.py +149 -0
- reader_workbench/contracts/builtins/plate_reader.py +47 -0
- reader_workbench/contracts/catalog.py +257 -0
- reader_workbench/contracts/model.py +109 -0
- reader_workbench/domains/__init__.py +1 -0
- reader_workbench/domains/cytometry/__init__.py +3 -0
- reader_workbench/domains/cytometry/analysis/__init__.py +29 -0
- reader_workbench/domains/cytometry/analysis/events.py +182 -0
- reader_workbench/domains/cytometry/analysis/gating.py +175 -0
- reader_workbench/domains/cytometry/analysis/workflow.py +248 -0
- reader_workbench/domains/cytometry/io/__init__.py +3 -0
- reader_workbench/domains/cytometry/io/fcs.py +135 -0
- reader_workbench/domains/cytometry/plots/__init__.py +5 -0
- reader_workbench/domains/cytometry/plots/diagnostic.py +155 -0
- reader_workbench/domains/logic/__init__.py +3 -0
- reader_workbench/domains/logic/crosstalk/__init__.py +3 -0
- reader_workbench/domains/logic/crosstalk/pairs.py +661 -0
- reader_workbench/domains/logic/four_state_vector/__init__.py +7 -0
- reader_workbench/domains/logic/four_state_vector/builder.py +321 -0
- reader_workbench/domains/logic/four_state_vector/collection/__init__.py +20 -0
- reader_workbench/domains/logic/four_state_vector/collection/checks.py +49 -0
- reader_workbench/domains/logic/four_state_vector/collection/constants.py +29 -0
- reader_workbench/domains/logic/four_state_vector/collection/model.py +19 -0
- reader_workbench/domains/logic/four_state_vector/collection/render.py +383 -0
- reader_workbench/domains/logic/four_state_vector/collection/sources.py +185 -0
- reader_workbench/domains/logic/four_state_vector/config.py +214 -0
- reader_workbench/domains/logic/four_state_vector/diagnostic.py +361 -0
- reader_workbench/domains/logic/four_state_vector/heatmap.py +86 -0
- reader_workbench/domains/logic/four_state_vector/math.py +191 -0
- reader_workbench/domains/logic/four_state_vector/reference.py +85 -0
- reader_workbench/domains/logic/four_state_vector/selection.py +228 -0
- reader_workbench/domains/logic/four_state_vector/treatment_semantics.py +51 -0
- reader_workbench/domains/logic/four_state_vector/validation.py +19 -0
- reader_workbench/domains/logic/logic_symmetry/__init__.py +3 -0
- reader_workbench/domains/logic/logic_symmetry/encodings.py +93 -0
- reader_workbench/domains/logic/logic_symmetry/extract_corners.py +192 -0
- reader_workbench/domains/logic/logic_symmetry/main.py +236 -0
- reader_workbench/domains/logic/logic_symmetry/metrics.py +96 -0
- reader_workbench/domains/logic/logic_symmetry/overlay.py +129 -0
- reader_workbench/domains/logic/logic_symmetry/prep.py +138 -0
- reader_workbench/domains/logic/logic_symmetry/render.py +356 -0
- reader_workbench/domains/logic/treatment_columns.py +42 -0
- reader_workbench/domains/plate_reader/__init__.py +1 -0
- reader_workbench/domains/plate_reader/analysis/__init__.py +14 -0
- reader_workbench/domains/plate_reader/analysis/fold_change.py +474 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/__init__.py +21 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/aggregation.py +191 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/contract_fields.py +50 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/contracts.py +320 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/design_dispositions.py +54 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/disposition_records.py +140 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/event_sensitivity.py +27 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/materialize.py +274 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/observation_resampling.py +97 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/reduction.py +62 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/seeds.py +15 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/sources.py +308 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/well_exclusion_validation.py +94 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/well_exclusions.py +54 -0
- reader_workbench/domains/plate_reader/analysis/timepoints.py +76 -0
- reader_workbench/domains/plate_reader/io/__init__.py +6 -0
- reader_workbench/domains/plate_reader/io/sample_map.py +65 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/__init__.py +6 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_kinetic.py +141 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_parser.py +295 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_shared.py +195 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_snapshot.py +130 -0
- reader_workbench/domains/plate_reader/ordering.py +59 -0
- reader_workbench/domains/plate_reader/plots/__init__.py +15 -0
- reader_workbench/domains/plate_reader/plots/_data.py +29 -0
- reader_workbench/domains/plate_reader/plots/common.py +346 -0
- reader_workbench/domains/plate_reader/plots/distributions.py +324 -0
- reader_workbench/domains/plate_reader/plots/dual_reporter_triptych.py +525 -0
- reader_workbench/domains/plate_reader/plots/dual_reporter_triptych_render.py +195 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/__init__.py +26 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic.py +295 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_components.py +149 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_render.py +308 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_style.py +51 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/schema.py +8 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/summary.py +140 -0
- reader_workbench/domains/plate_reader/plots/grouping.py +53 -0
- reader_workbench/domains/plate_reader/plots/panels/__init__.py +12 -0
- reader_workbench/domains/plate_reader/plots/panels/snapshot.py +161 -0
- reader_workbench/domains/plate_reader/plots/panels/snapshot_data.py +91 -0
- reader_workbench/domains/plate_reader/plots/panels/time_series.py +296 -0
- reader_workbench/domains/plate_reader/plots/single_reporter_diagnostic.py +435 -0
- reader_workbench/domains/plate_reader/plots/single_reporter_diagnostic_render.py +300 -0
- reader_workbench/domains/plate_reader/plots/snapshot_barplot/__init__.py +315 -0
- reader_workbench/domains/plate_reader/plots/snapshot_barplot/planning.py +168 -0
- reader_workbench/domains/plate_reader/plots/snapshot_heatmap/__init__.py +205 -0
- reader_workbench/domains/plate_reader/plots/snapshot_heatmap/inputs.py +103 -0
- reader_workbench/domains/plate_reader/plots/time_series.py +317 -0
- reader_workbench/domains/plate_reader/plots/ts_and_snap/__init__.py +479 -0
- reader_workbench/domains/plate_reader/plots/ts_and_snap/planning.py +283 -0
- reader_workbench/domains/time_series/__init__.py +29 -0
- reader_workbench/domains/time_series/aggregation.py +60 -0
- reader_workbench/domains/time_series/contracts.py +368 -0
- reader_workbench/domains/time_series/reduction.py +395 -0
- reader_workbench/errors.py +55 -0
- reader_workbench/maintenance/__init__.py +6 -0
- reader_workbench/maintenance/docs.py +335 -0
- reader_workbench/maintenance/model.py +28 -0
- reader_workbench/maintenance/release.py +39 -0
- reader_workbench/maintenance/skills.py +124 -0
- reader_workbench/plotting/__init__.py +20 -0
- reader_workbench/plotting/mpl.py +56 -0
- reader_workbench/plotting/sinks.py +69 -0
- reader_workbench/plotting/style.py +175 -0
- reader_workbench/plotting/utils.py +27 -0
- reader_workbench/plugins/__init__.py +1 -0
- reader_workbench/plugins/catalog.py +33 -0
- reader_workbench/plugins/export/__init__.py +0 -0
- reader_workbench/plugins/export/_paths.py +21 -0
- reader_workbench/plugins/export/csv.py +41 -0
- reader_workbench/plugins/export/xlsx.py +44 -0
- reader_workbench/plugins/ingest/__init__.py +0 -0
- reader_workbench/plugins/ingest/_discovery.py +58 -0
- reader_workbench/plugins/ingest/discovery_policy.py +66 -0
- reader_workbench/plugins/ingest/flow_cytometer.py +139 -0
- reader_workbench/plugins/ingest/synergy_h1.py +234 -0
- reader_workbench/plugins/manifests/__init__.py +1 -0
- reader_workbench/plugins/manifests/export.py +29 -0
- reader_workbench/plugins/manifests/ingest.py +29 -0
- reader_workbench/plugins/manifests/plot.py +161 -0
- reader_workbench/plugins/manifests/transform.py +172 -0
- reader_workbench/plugins/manifests/validator.py +18 -0
- reader_workbench/plugins/plot/__init__.py +0 -0
- reader_workbench/plugins/plot/_shared.py +55 -0
- reader_workbench/plugins/plot/cytometry_diagnostic.py +54 -0
- reader_workbench/plugins/plot/distributions.py +58 -0
- reader_workbench/plugins/plot/dual_reporter_triptych.py +224 -0
- reader_workbench/plugins/plot/four_state_event_window_diagnostic.py +133 -0
- reader_workbench/plugins/plot/four_state_event_window_summary.py +53 -0
- reader_workbench/plugins/plot/four_state_vector_collection.py +45 -0
- reader_workbench/plugins/plot/four_state_vector_diagnostic.py +105 -0
- reader_workbench/plugins/plot/four_state_vector_heatmap.py +63 -0
- reader_workbench/plugins/plot/logic_symmetry.py +56 -0
- reader_workbench/plugins/plot/single_reporter_diagnostic.py +279 -0
- reader_workbench/plugins/plot/snapshot_barplot.py +65 -0
- reader_workbench/plugins/plot/snapshot_heatmap.py +104 -0
- reader_workbench/plugins/plot/time_series.py +114 -0
- reader_workbench/plugins/plot/ts_and_snap.py +210 -0
- reader_workbench/plugins/transform/__init__.py +0 -0
- reader_workbench/plugins/transform/_four_state_vector.py +204 -0
- reader_workbench/plugins/transform/_labeling.py +109 -0
- reader_workbench/plugins/transform/alias.py +70 -0
- reader_workbench/plugins/transform/assay_labels.py +62 -0
- reader_workbench/plugins/transform/blank.py +79 -0
- reader_workbench/plugins/transform/crosstalk_pairs.py +180 -0
- reader_workbench/plugins/transform/cytometry_gating.py +120 -0
- reader_workbench/plugins/transform/fold_change.py +79 -0
- reader_workbench/plugins/transform/four_state_event_window.py +93 -0
- reader_workbench/plugins/transform/four_state_vector.py +62 -0
- reader_workbench/plugins/transform/four_state_vector_collection.py +41 -0
- reader_workbench/plugins/transform/logic_symmetry.py +67 -0
- reader_workbench/plugins/transform/outlier_filter.py +60 -0
- reader_workbench/plugins/transform/overflow.py +197 -0
- reader_workbench/plugins/transform/ratio.py +237 -0
- reader_workbench/plugins/transform/sample_map.py +170 -0
- reader_workbench/plugins/transform/sample_metadata.py +94 -0
- reader_workbench/plugins/validator/__init__.py +1 -0
- reader_workbench/plugins/validator/to_tidy_plus_map.py +155 -0
- reader_workbench/protocols/__init__.py +80 -0
- reader_workbench/protocols/_builtins_plate_reader_growth.py +179 -0
- reader_workbench/protocols/_builtins_plate_reader_variants.py +274 -0
- reader_workbench/protocols/builtins.py +1656 -0
- reader_workbench/protocols/compiler.py +22 -0
- reader_workbench/protocols/compilers/__init__.py +1 -0
- reader_workbench/protocols/compilers/common.py +100 -0
- reader_workbench/protocols/compilers/cytometry.py +87 -0
- reader_workbench/protocols/compilers/generic.py +14 -0
- reader_workbench/protocols/compilers/logic.py +245 -0
- reader_workbench/protocols/compilers/plate_reader.py +937 -0
- reader_workbench/protocols/compilers/plate_reader_pipeline.py +197 -0
- reader_workbench/protocols/model.py +1486 -0
- reader_workbench/protocols/semantic_coverage.py +234 -0
- reader_workbench/runtime/__init__.py +12 -0
- reader_workbench/runtime/builtin.py +23 -0
- reader_workbench/runtime/model.py +42 -0
- reader_workbench/workbench/__init__.py +60 -0
- reader_workbench/workbench/assets/__init__.py +22 -0
- reader_workbench/workbench/assets/types.py +118 -0
- reader_workbench/workbench/audit/__init__.py +5 -0
- reader_workbench/workbench/audit/experiments.py +307 -0
- reader_workbench/workbench/audit/staging.py +187 -0
- reader_workbench/workbench/cli/__init__.py +51 -0
- reader_workbench/workbench/cli/_lazy.py +9 -0
- reader_workbench/workbench/cli/_records_view.py +150 -0
- reader_workbench/workbench/cli/_surface_execution.py +443 -0
- reader_workbench/workbench/cli/audit.py +95 -0
- reader_workbench/workbench/cli/automation.py +229 -0
- reader_workbench/workbench/cli/demo.py +46 -0
- reader_workbench/workbench/cli/dop.py +91 -0
- reader_workbench/workbench/cli/experiments.py +635 -0
- reader_workbench/workbench/cli/helpers.py +232 -0
- reader_workbench/workbench/cli/main.py +59 -0
- reader_workbench/workbench/cli/maintenance.py +82 -0
- reader_workbench/workbench/cli/notebooks.py +260 -0
- reader_workbench/workbench/cli/pagination.py +117 -0
- reader_workbench/workbench/cli/protocols.py +336 -0
- reader_workbench/workbench/cli/shared.py +309 -0
- reader_workbench/workbench/cli/surfaces.py +534 -0
- reader_workbench/workbench/cli/verification.py +128 -0
- reader_workbench/workbench/commands.py +10 -0
- reader_workbench/workbench/config/__init__.py +47 -0
- reader_workbench/workbench/config/identity.py +13 -0
- reader_workbench/workbench/config/load.py +405 -0
- reader_workbench/workbench/config/model.py +274 -0
- reader_workbench/workbench/context.py +26 -0
- reader_workbench/workbench/decl/__init__.py +31 -0
- reader_workbench/workbench/decl/build.py +190 -0
- reader_workbench/workbench/decl/model.py +81 -0
- reader_workbench/workbench/dop/__init__.py +12 -0
- reader_workbench/workbench/dop/builtins.py +261 -0
- reader_workbench/workbench/dop/model.py +209 -0
- reader_workbench/workbench/engine/__init__.py +42 -0
- reader_workbench/workbench/engine/_shared.py +76 -0
- reader_workbench/workbench/engine/contracts.py +283 -0
- reader_workbench/workbench/engine/execution.py +326 -0
- reader_workbench/workbench/engine/file_outputs.py +260 -0
- reader_workbench/workbench/engine/inputs.py +161 -0
- reader_workbench/workbench/engine/invocations.py +507 -0
- reader_workbench/workbench/engine/planning.py +72 -0
- reader_workbench/workbench/engine/runtime.py +464 -0
- reader_workbench/workbench/engine/setup.py +149 -0
- reader_workbench/workbench/engine/validation.py +684 -0
- reader_workbench/workbench/experiment/__init__.py +47 -0
- reader_workbench/workbench/experiment/model.py +381 -0
- reader_workbench/workbench/experiments.py +133 -0
- reader_workbench/workbench/graph/__init__.py +47 -0
- reader_workbench/workbench/graph/nodes.py +102 -0
- reader_workbench/workbench/graph/normalize.py +177 -0
- reader_workbench/workbench/graph/refs.py +148 -0
- reader_workbench/workbench/input_discovery.py +19 -0
- reader_workbench/workbench/inspection/__init__.py +3 -0
- reader_workbench/workbench/inspection/catalogs.py +128 -0
- reader_workbench/workbench/inspection/common.py +92 -0
- reader_workbench/workbench/inspection/dop.py +64 -0
- reader_workbench/workbench/inspection/experiments.py +449 -0
- reader_workbench/workbench/inspection/inventory.py +68 -0
- reader_workbench/workbench/inspection/protocols.py +368 -0
- reader_workbench/workbench/inspection/readiness.py +333 -0
- reader_workbench/workbench/inspection/reports.py +367 -0
- reader_workbench/workbench/inspection/results.py +166 -0
- reader_workbench/workbench/inspection/runtime.py +287 -0
- reader_workbench/workbench/inspection/semantics.py +192 -0
- reader_workbench/workbench/inspection/validation.py +30 -0
- reader_workbench/workbench/notebooks/__init__.py +17 -0
- reader_workbench/workbench/notebooks/_launch_registry.py +112 -0
- reader_workbench/workbench/notebooks/_launch_runtime.py +104 -0
- reader_workbench/workbench/notebooks/components/__init__.py +21 -0
- reader_workbench/workbench/notebooks/components/deliverables.py +403 -0
- reader_workbench/workbench/notebooks/components/overview.py +119 -0
- reader_workbench/workbench/notebooks/eda.marimo.py.txt +153 -0
- reader_workbench/workbench/notebooks/launch.py +274 -0
- reader_workbench/workbench/notebooks/presentation.py +136 -0
- reader_workbench/workbench/notebooks/scaffold.py +60 -0
- reader_workbench/workbench/ontology.py +78 -0
- reader_workbench/workbench/paths.py +44 -0
- reader_workbench/workbench/ports/__init__.py +31 -0
- reader_workbench/workbench/ports/model.py +168 -0
- reader_workbench/workbench/records/__init__.py +44 -0
- reader_workbench/workbench/records/epoch.py +329 -0
- reader_workbench/workbench/records/evidence.py +247 -0
- reader_workbench/workbench/records/identity.py +87 -0
- reader_workbench/workbench/records/locking.py +185 -0
- reader_workbench/workbench/records/model.py +711 -0
- reader_workbench/workbench/records/sources.py +73 -0
- reader_workbench/workbench/records/store.py +1022 -0
- reader_workbench/workbench/records/verification.py +998 -0
- reader_workbench/workbench/registry.py +333 -0
- reader_workbench/workbench/spec_overrides.py +215 -0
- reader_workbench-1.0.0.dist-info/METADATA +91 -0
- reader_workbench-1.0.0.dist-info/RECORD +293 -0
- reader_workbench-1.0.0.dist-info/WHEEL +5 -0
- reader_workbench-1.0.0.dist-info/entry_points.txt +2 -0
- reader_workbench-1.0.0.dist-info/licenses/LICENSE +21 -0
- reader_workbench-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1022 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
import stat
|
|
6
|
+
from collections.abc import Iterable, Iterator, Mapping
|
|
7
|
+
from contextlib import contextmanager, suppress
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from datetime import UTC, datetime
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from tempfile import NamedTemporaryFile
|
|
12
|
+
from typing import Any
|
|
13
|
+
from uuid import UUID, uuid4
|
|
14
|
+
|
|
15
|
+
import pandas as pd
|
|
16
|
+
|
|
17
|
+
from reader_workbench.contracts import ContractCatalog, ContractId
|
|
18
|
+
from reader_workbench.errors import ProvenanceEpochChangedError, RecordError
|
|
19
|
+
from reader_workbench.workbench.experiments import ExperimentCatalog, ExperimentLocation
|
|
20
|
+
from reader_workbench.workbench.graph import (
|
|
21
|
+
FileRef,
|
|
22
|
+
ProvenanceInput,
|
|
23
|
+
RecipeSource,
|
|
24
|
+
RecordRef,
|
|
25
|
+
ResourceRef,
|
|
26
|
+
SourceRecordRef,
|
|
27
|
+
)
|
|
28
|
+
from reader_workbench.workbench.ontology import WorkbenchProducerKind, WorkbenchRecordKind
|
|
29
|
+
from reader_workbench.workbench.paths import resolve_confined_sink_root
|
|
30
|
+
from reader_workbench.workbench.records.model import (
|
|
31
|
+
DataFrameArtifactRecord,
|
|
32
|
+
FileBundleRecord,
|
|
33
|
+
PathDescription,
|
|
34
|
+
RecordProducer,
|
|
35
|
+
RecordRecipeSource,
|
|
36
|
+
normalize_file_bundle_metadata,
|
|
37
|
+
record_from_dict,
|
|
38
|
+
record_to_dict,
|
|
39
|
+
sha256_file,
|
|
40
|
+
verify_record_artifact_integrity,
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
from .epoch import assert_no_interrupted_epoch, replace_generated_epoch, validate_generated_epoch_boundary
|
|
44
|
+
from .evidence import RecordInputEvidence, SourceExperimentResolver, capture_artifact_evidence
|
|
45
|
+
from .identity import BuildIdentity, current_build_identity, digest_json
|
|
46
|
+
from .locking import ProvenanceFileLock, provenance_lock_scope
|
|
47
|
+
from .model import record_revision_digest
|
|
48
|
+
|
|
49
|
+
RECORD_CATALOG_SCHEMA_VERSION = 4
|
|
50
|
+
_RECORD_CATALOG_FIELDS = frozenset({"schema_version", "provenance_epoch_id", "latest", "history"})
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _is_canonical_uuid4(value: object) -> bool:
|
|
54
|
+
if not isinstance(value, str):
|
|
55
|
+
return False
|
|
56
|
+
try:
|
|
57
|
+
parsed = UUID(value)
|
|
58
|
+
except ValueError:
|
|
59
|
+
return False
|
|
60
|
+
return parsed.version == 4 and str(parsed) == value
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _empty_catalog() -> dict[str, Any]:
|
|
64
|
+
return {
|
|
65
|
+
"schema_version": RECORD_CATALOG_SCHEMA_VERSION,
|
|
66
|
+
"provenance_epoch_id": str(uuid4()),
|
|
67
|
+
"latest": {},
|
|
68
|
+
"history": {},
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _selected_latest_record_ids(
|
|
73
|
+
latest: Mapping[str, Any],
|
|
74
|
+
*,
|
|
75
|
+
config_digest: str | None,
|
|
76
|
+
record_ids: frozenset[str] | None,
|
|
77
|
+
) -> frozenset[str]:
|
|
78
|
+
candidates = set(latest)
|
|
79
|
+
if record_ids is not None:
|
|
80
|
+
candidates &= record_ids
|
|
81
|
+
if config_digest is None:
|
|
82
|
+
return frozenset(candidates)
|
|
83
|
+
return frozenset(
|
|
84
|
+
record_id
|
|
85
|
+
for record_id in candidates
|
|
86
|
+
if not isinstance(latest[record_id], dict) or latest[record_id].get("config_digest") == config_digest
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
@dataclass(frozen=True)
|
|
91
|
+
class RecordCatalogSnapshot:
|
|
92
|
+
schema_version: int
|
|
93
|
+
provenance_epoch_id: str
|
|
94
|
+
active_invocation_ledger: Path
|
|
95
|
+
latest_records: tuple[DataFrameArtifactRecord | FileBundleRecord, ...]
|
|
96
|
+
revision_counts: dict[str, int]
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
class RecordStore:
|
|
100
|
+
def __init__(
|
|
101
|
+
self,
|
|
102
|
+
outputs_dir: Path,
|
|
103
|
+
*,
|
|
104
|
+
contracts: ContractCatalog,
|
|
105
|
+
plots_subdir: str | None = "plots",
|
|
106
|
+
exports_subdir: str | None = "exports",
|
|
107
|
+
experiment_root: Path | None = None,
|
|
108
|
+
create: bool = True,
|
|
109
|
+
) -> None:
|
|
110
|
+
self.root = Path(outputs_dir).expanduser().absolute()
|
|
111
|
+
self.contracts = contracts
|
|
112
|
+
self.artifacts_dir = self.root / "artifacts"
|
|
113
|
+
self.manifests_dir = self.root / "manifests"
|
|
114
|
+
self.records_path = self.manifests_dir / "records.json"
|
|
115
|
+
self._catalog_lock_path = self.manifests_dir / ".records.lock"
|
|
116
|
+
self._catalog_lock = ProvenanceFileLock(self._catalog_lock_path, timeout=30)
|
|
117
|
+
self._bound_provenance_epoch_id: str | None = None
|
|
118
|
+
self.experiment_root = Path(experiment_root or self.root.parent).expanduser().absolute()
|
|
119
|
+
if plots_subdir in (None, "", ".", "./"):
|
|
120
|
+
self.plots_dir = self.root
|
|
121
|
+
else:
|
|
122
|
+
self.plots_dir = self.root / plots_subdir
|
|
123
|
+
if exports_subdir in (None, "", ".", "./"):
|
|
124
|
+
self.exports_dir = self.root
|
|
125
|
+
else:
|
|
126
|
+
self.exports_dir = self.root / exports_subdir
|
|
127
|
+
self._validate_sink_roots()
|
|
128
|
+
if create:
|
|
129
|
+
self.ensure_layout()
|
|
130
|
+
|
|
131
|
+
def _validate_sink_roots(self) -> None:
|
|
132
|
+
try:
|
|
133
|
+
resolve_confined_sink_root(self.root, root=self.experiment_root, label="outputs")
|
|
134
|
+
for label, path in (
|
|
135
|
+
("artifacts", self.artifacts_dir),
|
|
136
|
+
("manifests", self.manifests_dir),
|
|
137
|
+
("plots", self.plots_dir),
|
|
138
|
+
("exports", self.exports_dir),
|
|
139
|
+
):
|
|
140
|
+
resolve_confined_sink_root(path, root=self.root, label=label)
|
|
141
|
+
except ValueError as exc:
|
|
142
|
+
raise RecordError(str(exc)) from exc
|
|
143
|
+
|
|
144
|
+
def _validate_catalog_roots(self) -> None:
|
|
145
|
+
try:
|
|
146
|
+
resolve_confined_sink_root(self.root, root=self.experiment_root, label="outputs")
|
|
147
|
+
resolve_confined_sink_root(self.manifests_dir, root=self.root, label="manifests")
|
|
148
|
+
except ValueError as exc:
|
|
149
|
+
raise RecordError(str(exc)) from exc
|
|
150
|
+
|
|
151
|
+
def ensure_layout(self) -> None:
|
|
152
|
+
self._validate_sink_roots()
|
|
153
|
+
self.artifacts_dir.mkdir(parents=True, exist_ok=True)
|
|
154
|
+
self.manifests_dir.mkdir(parents=True, exist_ok=True)
|
|
155
|
+
if self.plots_dir != self.root:
|
|
156
|
+
self.plots_dir.mkdir(parents=True, exist_ok=True)
|
|
157
|
+
if self.exports_dir != self.root:
|
|
158
|
+
self.exports_dir.mkdir(parents=True, exist_ok=True)
|
|
159
|
+
if not self.records_path.exists():
|
|
160
|
+
self._write_catalog(_empty_catalog(), create_only=True)
|
|
161
|
+
|
|
162
|
+
def catalog_exists(self) -> bool:
|
|
163
|
+
return self.records_path.exists() or self.records_path.is_symlink()
|
|
164
|
+
|
|
165
|
+
def provenance_lock_exists(self) -> bool:
|
|
166
|
+
"""Return whether the non-followed writer coordination entry already exists."""
|
|
167
|
+
|
|
168
|
+
return self._catalog_lock_path.exists() or self._catalog_lock_path.is_symlink()
|
|
169
|
+
|
|
170
|
+
def reset_generated_epoch(self, *, preserved_paths: Iterable[Path] = ()) -> str:
|
|
171
|
+
"""Begin a fresh epoch for catalog, artifacts, plots, and exports."""
|
|
172
|
+
|
|
173
|
+
if self._bound_provenance_epoch_id is not None:
|
|
174
|
+
raise RecordError("A RecordStore bound to a provenance epoch cannot reset generated outputs")
|
|
175
|
+
|
|
176
|
+
def _initialize() -> str:
|
|
177
|
+
self.ensure_layout()
|
|
178
|
+
return self.provenance_epoch_id()
|
|
179
|
+
|
|
180
|
+
with self._catalog_lock_scope(operation="reset the generated-output epoch"):
|
|
181
|
+
return replace_generated_epoch(
|
|
182
|
+
outputs_root=self.root,
|
|
183
|
+
artifacts_root=self.artifacts_dir,
|
|
184
|
+
manifests_root=self.manifests_dir,
|
|
185
|
+
plots_root=self.plots_dir,
|
|
186
|
+
exports_root=self.exports_dir,
|
|
187
|
+
log_path=self.root / "reader.log",
|
|
188
|
+
preserved_paths=preserved_paths,
|
|
189
|
+
lock_path=self._catalog_lock_path,
|
|
190
|
+
initialize=_initialize,
|
|
191
|
+
)
|
|
192
|
+
|
|
193
|
+
def validate_generated_epoch_reset(self, *, preserved_paths: Iterable[Path] = ()) -> None:
|
|
194
|
+
"""Fail before mutation when generated-output ownership is unsafe to reset."""
|
|
195
|
+
|
|
196
|
+
validate_generated_epoch_boundary(
|
|
197
|
+
outputs_root=self.root,
|
|
198
|
+
artifacts_root=self.artifacts_dir,
|
|
199
|
+
manifests_root=self.manifests_dir,
|
|
200
|
+
plots_root=self.plots_dir,
|
|
201
|
+
exports_root=self.exports_dir,
|
|
202
|
+
log_path=self.root / "reader.log",
|
|
203
|
+
preserved_paths=preserved_paths,
|
|
204
|
+
lock_path=self._catalog_lock_path,
|
|
205
|
+
)
|
|
206
|
+
|
|
207
|
+
def validate_no_interrupted_epoch(self) -> None:
|
|
208
|
+
"""Fail while recovery evidence from an interrupted reset remains."""
|
|
209
|
+
|
|
210
|
+
assert_no_interrupted_epoch(self.root)
|
|
211
|
+
|
|
212
|
+
def provenance_epoch_id(self) -> str:
|
|
213
|
+
"""Return the catalog-owned identity for the active provenance epoch."""
|
|
214
|
+
|
|
215
|
+
return str(self._read_catalog()["provenance_epoch_id"])
|
|
216
|
+
|
|
217
|
+
def invocation_ledger_path(self) -> Path:
|
|
218
|
+
"""Return the active epoch's deterministic invocation-ledger path."""
|
|
219
|
+
|
|
220
|
+
epoch_id = self.provenance_epoch_id()
|
|
221
|
+
return self.manifests_dir / "invocations" / f"{epoch_id}.jsonl"
|
|
222
|
+
|
|
223
|
+
def catalog_snapshot(
|
|
224
|
+
self,
|
|
225
|
+
*,
|
|
226
|
+
current_config_digest: str | None = None,
|
|
227
|
+
current_record_ids: frozenset[str] | None = None,
|
|
228
|
+
) -> RecordCatalogSnapshot:
|
|
229
|
+
"""Read one internally consistent catalog projection for inspection surfaces.
|
|
230
|
+
|
|
231
|
+
A current-config projection validates the full catalog structure and
|
|
232
|
+
every serialized payload, but resolves live external source locations
|
|
233
|
+
only for selected latest records. Historical and unselected source
|
|
234
|
+
resolution remains the responsibility of an unscoped snapshot.
|
|
235
|
+
"""
|
|
236
|
+
|
|
237
|
+
source_experiment_resolver = self._source_experiment_resolver()
|
|
238
|
+
with self._catalog_lock_scope(operation="inspect the record catalog"):
|
|
239
|
+
catalog = self._read_catalog(
|
|
240
|
+
source_experiment_resolver=source_experiment_resolver,
|
|
241
|
+
materialize_config_digest=current_config_digest,
|
|
242
|
+
materialize_record_ids=current_record_ids,
|
|
243
|
+
)
|
|
244
|
+
epoch_id = str(catalog["provenance_epoch_id"])
|
|
245
|
+
selected_record_ids = _selected_latest_record_ids(
|
|
246
|
+
catalog["latest"],
|
|
247
|
+
config_digest=current_config_digest,
|
|
248
|
+
record_ids=current_record_ids,
|
|
249
|
+
)
|
|
250
|
+
records = tuple(
|
|
251
|
+
self._materialize(
|
|
252
|
+
record_id,
|
|
253
|
+
payload,
|
|
254
|
+
source_experiment_resolver=source_experiment_resolver,
|
|
255
|
+
)
|
|
256
|
+
for record_id, payload in sorted(catalog["latest"].items())
|
|
257
|
+
if record_id in selected_record_ids
|
|
258
|
+
)
|
|
259
|
+
revision_counts = {
|
|
260
|
+
record_id: len(catalog["history"][record_id]) for record_id in sorted(selected_record_ids)
|
|
261
|
+
}
|
|
262
|
+
return RecordCatalogSnapshot(
|
|
263
|
+
schema_version=RECORD_CATALOG_SCHEMA_VERSION,
|
|
264
|
+
provenance_epoch_id=epoch_id,
|
|
265
|
+
active_invocation_ledger=self.manifests_dir / "invocations" / f"{epoch_id}.jsonl",
|
|
266
|
+
latest_records=records,
|
|
267
|
+
revision_counts=revision_counts,
|
|
268
|
+
)
|
|
269
|
+
|
|
270
|
+
def assert_provenance_epoch(self, expected_epoch_id: str) -> None:
|
|
271
|
+
"""Fail if another catalog epoch has replaced the caller's active epoch."""
|
|
272
|
+
|
|
273
|
+
active_epoch_id = self.provenance_epoch_id()
|
|
274
|
+
if active_epoch_id != expected_epoch_id:
|
|
275
|
+
raise ProvenanceEpochChangedError(
|
|
276
|
+
"The record catalog provenance epoch changed during this operation; "
|
|
277
|
+
"discard the stale result and start a new Reader invocation."
|
|
278
|
+
)
|
|
279
|
+
|
|
280
|
+
def bind_provenance_epoch(self, expected_epoch_id: str) -> None:
|
|
281
|
+
"""Bind subsequent reads and writes to one invocation's catalog epoch."""
|
|
282
|
+
|
|
283
|
+
self.assert_provenance_epoch(expected_epoch_id)
|
|
284
|
+
self._bound_provenance_epoch_id = expected_epoch_id
|
|
285
|
+
|
|
286
|
+
@property
|
|
287
|
+
def provenance_lock(self) -> ProvenanceFileLock:
|
|
288
|
+
"""Reentrant writer lease shared by operations, catalog commits, and ledger appends."""
|
|
289
|
+
|
|
290
|
+
return self._catalog_lock
|
|
291
|
+
|
|
292
|
+
@contextmanager
|
|
293
|
+
def _catalog_lock_scope(self, *, operation: str) -> Iterator[None]:
|
|
294
|
+
"""Normalize lock failures without swallowing errors from the protected operation."""
|
|
295
|
+
|
|
296
|
+
with provenance_lock_scope(
|
|
297
|
+
self._catalog_lock,
|
|
298
|
+
acquire_error=RecordError(f"Could not {operation}: the record catalog provenance lock is unavailable"),
|
|
299
|
+
release_error=RecordError(f"Could not {operation}: the record catalog provenance lock could not release"),
|
|
300
|
+
release_note="Reader also could not release the record catalog lock",
|
|
301
|
+
):
|
|
302
|
+
yield
|
|
303
|
+
|
|
304
|
+
def assert_catalog_snapshot(self, *, provenance_epoch_id: str, catalog_digest: str) -> None:
|
|
305
|
+
"""Fail unless the catalog still matches a previously read snapshot."""
|
|
306
|
+
|
|
307
|
+
with self._catalog_lock_scope(operation="validate the record catalog snapshot"):
|
|
308
|
+
self.assert_provenance_epoch(provenance_epoch_id)
|
|
309
|
+
if digest_json(self._read_catalog()) != catalog_digest:
|
|
310
|
+
raise RecordError("The record catalog changed concurrently; retry from the current catalog state.")
|
|
311
|
+
|
|
312
|
+
def _source_experiment_resolver(self) -> SourceExperimentResolver:
|
|
313
|
+
experiment_catalog: ExperimentCatalog | None = None
|
|
314
|
+
|
|
315
|
+
def resolve(experiment_id: str):
|
|
316
|
+
nonlocal experiment_catalog
|
|
317
|
+
if experiment_catalog is None:
|
|
318
|
+
experiment_catalog = ExperimentCatalog.from_experiment_root(self.experiment_root)
|
|
319
|
+
return experiment_catalog.resolve(experiment_id)
|
|
320
|
+
|
|
321
|
+
return resolve
|
|
322
|
+
|
|
323
|
+
def _source_identity_resolver(self) -> SourceExperimentResolver:
|
|
324
|
+
"""Validate serialized source identity without requiring its live workspace."""
|
|
325
|
+
|
|
326
|
+
def preserve(experiment_id: str) -> ExperimentLocation:
|
|
327
|
+
if Path(experiment_id).name != experiment_id or experiment_id in {".", ".."}:
|
|
328
|
+
raise RecordError(f"Source experiment id must be one safe path segment: {experiment_id!r}")
|
|
329
|
+
return ExperimentLocation(
|
|
330
|
+
id=experiment_id,
|
|
331
|
+
root=self.experiment_root,
|
|
332
|
+
config_path=self.experiment_root / "config.yaml",
|
|
333
|
+
outputs_dir=self.root,
|
|
334
|
+
)
|
|
335
|
+
|
|
336
|
+
return preserve
|
|
337
|
+
|
|
338
|
+
def _read_catalog(
|
|
339
|
+
self,
|
|
340
|
+
*,
|
|
341
|
+
source_experiment_resolver: SourceExperimentResolver | None = None,
|
|
342
|
+
materialize_config_digest: str | None = None,
|
|
343
|
+
materialize_record_ids: frozenset[str] | None = None,
|
|
344
|
+
) -> dict[str, Any]:
|
|
345
|
+
self._validate_catalog_roots()
|
|
346
|
+
if self.records_path.is_symlink():
|
|
347
|
+
raise RecordError("records.json must not be a symlink")
|
|
348
|
+
if not self.records_path.exists():
|
|
349
|
+
raise RecordError("records.json is missing")
|
|
350
|
+
descriptor = -1
|
|
351
|
+
try:
|
|
352
|
+
flags = os.O_RDONLY | getattr(os, "O_CLOEXEC", 0) | getattr(os, "O_NOFOLLOW", 0)
|
|
353
|
+
flags |= getattr(os, "O_NONBLOCK", 0)
|
|
354
|
+
descriptor = os.open(self.records_path, flags)
|
|
355
|
+
catalog_stat = os.fstat(descriptor)
|
|
356
|
+
if not stat.S_ISREG(catalog_stat.st_mode) or catalog_stat.st_nlink != 1:
|
|
357
|
+
raise RecordError("records.json must be a regular file with a single link")
|
|
358
|
+
with os.fdopen(descriptor, mode="r", encoding="utf-8") as stream:
|
|
359
|
+
descriptor = -1
|
|
360
|
+
content = stream.read()
|
|
361
|
+
except UnicodeError as exc:
|
|
362
|
+
raise RecordError(f"records.json is not valid UTF-8: {exc}") from exc
|
|
363
|
+
except RecordError:
|
|
364
|
+
raise
|
|
365
|
+
except OSError as exc:
|
|
366
|
+
raise RecordError(f"Could not read records.json: {exc}") from exc
|
|
367
|
+
finally:
|
|
368
|
+
if descriptor >= 0:
|
|
369
|
+
os.close(descriptor)
|
|
370
|
+
try:
|
|
371
|
+
data = json.loads(content)
|
|
372
|
+
except json.JSONDecodeError as exc:
|
|
373
|
+
raise RecordError(f"records.json is not valid JSON: {exc}") from exc
|
|
374
|
+
if not isinstance(data, dict):
|
|
375
|
+
raise RecordError("records.json must be a JSON object")
|
|
376
|
+
schema_version = data.get("schema_version")
|
|
377
|
+
if schema_version != RECORD_CATALOG_SCHEMA_VERSION:
|
|
378
|
+
raise RecordError(
|
|
379
|
+
f"records.json schema_version must be {RECORD_CATALOG_SCHEMA_VERSION} (got {schema_version!r})"
|
|
380
|
+
)
|
|
381
|
+
if set(data) != _RECORD_CATALOG_FIELDS:
|
|
382
|
+
raise RecordError(
|
|
383
|
+
"records.json must contain exactly schema_version, provenance_epoch_id, latest, and history"
|
|
384
|
+
)
|
|
385
|
+
if not _is_canonical_uuid4(data.get("provenance_epoch_id")):
|
|
386
|
+
raise RecordError("records.json provenance_epoch_id must be a canonical UUID4")
|
|
387
|
+
if "latest" not in data or "history" not in data:
|
|
388
|
+
raise RecordError("records.json must include 'latest' and 'history' objects")
|
|
389
|
+
if not isinstance(data["latest"], dict) or not isinstance(data["history"], dict):
|
|
390
|
+
raise RecordError("records.json 'latest' and 'history' must be JSON objects")
|
|
391
|
+
selected_record_ids = _selected_latest_record_ids(
|
|
392
|
+
data["latest"],
|
|
393
|
+
config_digest=materialize_config_digest,
|
|
394
|
+
record_ids=materialize_record_ids,
|
|
395
|
+
)
|
|
396
|
+
self._validate_catalog_lineage(
|
|
397
|
+
data,
|
|
398
|
+
source_experiment_resolver=source_experiment_resolver or self._source_experiment_resolver(),
|
|
399
|
+
materialize_latest_ids=selected_record_ids,
|
|
400
|
+
materialize_history=materialize_config_digest is None and materialize_record_ids is None,
|
|
401
|
+
)
|
|
402
|
+
if (
|
|
403
|
+
self._bound_provenance_epoch_id is not None
|
|
404
|
+
and data["provenance_epoch_id"] != self._bound_provenance_epoch_id
|
|
405
|
+
):
|
|
406
|
+
raise ProvenanceEpochChangedError(
|
|
407
|
+
"The record catalog provenance epoch changed during this operation; "
|
|
408
|
+
"discard the stale result and start a new Reader invocation."
|
|
409
|
+
)
|
|
410
|
+
return data
|
|
411
|
+
|
|
412
|
+
def _validate_catalog_lineage(
|
|
413
|
+
self,
|
|
414
|
+
catalog: dict[str, Any],
|
|
415
|
+
*,
|
|
416
|
+
source_experiment_resolver: SourceExperimentResolver,
|
|
417
|
+
materialize_latest_ids: frozenset[str],
|
|
418
|
+
materialize_history: bool,
|
|
419
|
+
) -> None:
|
|
420
|
+
latest = catalog["latest"]
|
|
421
|
+
history = catalog["history"]
|
|
422
|
+
latest_ids = set(latest)
|
|
423
|
+
history_ids = set(history)
|
|
424
|
+
if latest_ids != history_ids:
|
|
425
|
+
missing_history = sorted(latest_ids - history_ids)
|
|
426
|
+
orphan_history = sorted(history_ids - latest_ids)
|
|
427
|
+
details = []
|
|
428
|
+
if missing_history:
|
|
429
|
+
details.append("latest records missing history: " + ", ".join(missing_history))
|
|
430
|
+
if orphan_history:
|
|
431
|
+
details.append("history records missing latest: " + ", ".join(orphan_history))
|
|
432
|
+
raise RecordError("records.json latest/history lineage is inconsistent: " + "; ".join(details))
|
|
433
|
+
identity_resolver = self._source_identity_resolver()
|
|
434
|
+
for record_id in sorted(latest_ids):
|
|
435
|
+
revisions = history[record_id]
|
|
436
|
+
if not isinstance(revisions, list):
|
|
437
|
+
raise RecordError(f"records.json history for {record_id!r} must be a list")
|
|
438
|
+
if not revisions:
|
|
439
|
+
raise RecordError(f"records.json history for latest record {record_id!r} must be non-empty")
|
|
440
|
+
if not materialize_history:
|
|
441
|
+
for payload in revisions:
|
|
442
|
+
self._materialize(
|
|
443
|
+
record_id,
|
|
444
|
+
payload,
|
|
445
|
+
source_experiment_resolver=identity_resolver,
|
|
446
|
+
)
|
|
447
|
+
self._materialize(
|
|
448
|
+
record_id,
|
|
449
|
+
latest[record_id],
|
|
450
|
+
source_experiment_resolver=identity_resolver,
|
|
451
|
+
)
|
|
452
|
+
if materialize_history:
|
|
453
|
+
for payload in revisions:
|
|
454
|
+
self._materialize(
|
|
455
|
+
record_id,
|
|
456
|
+
payload,
|
|
457
|
+
source_experiment_resolver=source_experiment_resolver,
|
|
458
|
+
)
|
|
459
|
+
if record_id in materialize_latest_ids:
|
|
460
|
+
self._materialize(
|
|
461
|
+
record_id,
|
|
462
|
+
latest[record_id],
|
|
463
|
+
source_experiment_resolver=source_experiment_resolver,
|
|
464
|
+
)
|
|
465
|
+
if revisions[-1] != latest[record_id]:
|
|
466
|
+
raise RecordError(
|
|
467
|
+
f"records.json history for {record_id!r} must end with the exact latest record payload"
|
|
468
|
+
)
|
|
469
|
+
|
|
470
|
+
def _write_catalog(
|
|
471
|
+
self,
|
|
472
|
+
payload: dict[str, Any],
|
|
473
|
+
*,
|
|
474
|
+
expected_provenance_epoch_id: str | None = None,
|
|
475
|
+
expected_catalog_digest: str | None = None,
|
|
476
|
+
create_only: bool = False,
|
|
477
|
+
) -> None:
|
|
478
|
+
self._validate_catalog_roots()
|
|
479
|
+
self.manifests_dir.mkdir(parents=True, exist_ok=True)
|
|
480
|
+
staged_path: Path | None = None
|
|
481
|
+
try:
|
|
482
|
+
with NamedTemporaryFile(
|
|
483
|
+
mode="w",
|
|
484
|
+
encoding="utf-8",
|
|
485
|
+
dir=self.manifests_dir,
|
|
486
|
+
prefix=f".{self.records_path.name}.",
|
|
487
|
+
suffix=".tmp",
|
|
488
|
+
delete=False,
|
|
489
|
+
) as staged:
|
|
490
|
+
staged_path = Path(staged.name)
|
|
491
|
+
json.dump(payload, staged, indent=2, sort_keys=True)
|
|
492
|
+
staged.flush()
|
|
493
|
+
os.fsync(staged.fileno())
|
|
494
|
+
try:
|
|
495
|
+
with self._catalog_lock_scope(operation="commit the record catalog"):
|
|
496
|
+
self._validate_catalog_roots()
|
|
497
|
+
if create_only and self.records_path.exists():
|
|
498
|
+
return
|
|
499
|
+
if expected_provenance_epoch_id is not None:
|
|
500
|
+
self.assert_provenance_epoch(expected_provenance_epoch_id)
|
|
501
|
+
if expected_catalog_digest is not None:
|
|
502
|
+
current_digest = digest_json(self._read_catalog())
|
|
503
|
+
if current_digest != expected_catalog_digest:
|
|
504
|
+
raise RecordError(
|
|
505
|
+
"The record catalog changed concurrently; retry from the current catalog state."
|
|
506
|
+
)
|
|
507
|
+
staged_path.replace(self.records_path)
|
|
508
|
+
except OSError as exc:
|
|
509
|
+
raise RecordError(f"Could not atomically replace records.json: {exc}") from exc
|
|
510
|
+
finally:
|
|
511
|
+
if staged_path is not None:
|
|
512
|
+
staged_path.unlink(missing_ok=True)
|
|
513
|
+
|
|
514
|
+
def _materialize(
|
|
515
|
+
self,
|
|
516
|
+
record_id: str,
|
|
517
|
+
payload: dict[str, Any],
|
|
518
|
+
*,
|
|
519
|
+
source_experiment_resolver: SourceExperimentResolver | None = None,
|
|
520
|
+
) -> DataFrameArtifactRecord | FileBundleRecord:
|
|
521
|
+
record = record_from_dict(
|
|
522
|
+
payload,
|
|
523
|
+
outputs_dir=self.root,
|
|
524
|
+
experiment_root=self.experiment_root,
|
|
525
|
+
source_experiment_resolver=source_experiment_resolver or self._source_experiment_resolver(),
|
|
526
|
+
)
|
|
527
|
+
if record.record_id != record_id:
|
|
528
|
+
raise RecordError(
|
|
529
|
+
f"records.json entry key {record_id!r} does not match payload record_id {record.record_id!r}"
|
|
530
|
+
)
|
|
531
|
+
return record
|
|
532
|
+
|
|
533
|
+
def capture_inputs(
|
|
534
|
+
self,
|
|
535
|
+
inputs: Iterable[ProvenanceInput],
|
|
536
|
+
*,
|
|
537
|
+
resolved_inputs: Mapping[str, Any] | None = None,
|
|
538
|
+
) -> tuple[RecordInputEvidence, ...]:
|
|
539
|
+
"""Bind immutable evidence to the exact inputs resolved for one computation."""
|
|
540
|
+
|
|
541
|
+
resolved_by_label = dict(resolved_inputs or {})
|
|
542
|
+
from .sources import ResolvedSourceRecord, SourceRecordCollection # noqa: PLC0415
|
|
543
|
+
|
|
544
|
+
resolved_sources: dict[SourceRecordRef, ResolvedSourceRecord] = {}
|
|
545
|
+
for value in resolved_by_label.values():
|
|
546
|
+
if isinstance(value, ResolvedSourceRecord):
|
|
547
|
+
candidates = (value,)
|
|
548
|
+
elif isinstance(value, SourceRecordCollection):
|
|
549
|
+
candidates = value.records
|
|
550
|
+
else:
|
|
551
|
+
continue
|
|
552
|
+
for candidate in candidates:
|
|
553
|
+
previous = resolved_sources.get(candidate.ref)
|
|
554
|
+
if previous is not None and previous.revision_digest != candidate.revision_digest:
|
|
555
|
+
raise RecordError(
|
|
556
|
+
f"Resolved source record {candidate.ref.experiment_id}:{candidate.ref.record_id} "
|
|
557
|
+
"has conflicting revisions"
|
|
558
|
+
)
|
|
559
|
+
resolved_sources[candidate.ref] = candidate
|
|
560
|
+
|
|
561
|
+
evidence: list[RecordInputEvidence] = []
|
|
562
|
+
for item in inputs:
|
|
563
|
+
if isinstance(item.ref, RecordRef):
|
|
564
|
+
upstream = resolved_by_label.get(item.label)
|
|
565
|
+
if upstream is None:
|
|
566
|
+
upstream = self.latest_record(item.ref.record_id)
|
|
567
|
+
if upstream is None:
|
|
568
|
+
raise RecordError(
|
|
569
|
+
f"Input record {item.ref.record_id!r} is missing; produce it before persisting this record."
|
|
570
|
+
)
|
|
571
|
+
if not isinstance(upstream, (DataFrameArtifactRecord, FileBundleRecord)):
|
|
572
|
+
raise RecordError(f"Resolved input {item.label!r} must be a Reader record")
|
|
573
|
+
if upstream.record_id != item.ref.record_id:
|
|
574
|
+
raise RecordError(
|
|
575
|
+
f"Resolved input {item.label!r} is record {upstream.record_id!r}, "
|
|
576
|
+
f"expected {item.ref.record_id!r}"
|
|
577
|
+
)
|
|
578
|
+
verify_record_artifact_integrity(upstream, outputs_dir=self.root)
|
|
579
|
+
evidence.append(
|
|
580
|
+
RecordInputEvidence(
|
|
581
|
+
label=item.label,
|
|
582
|
+
ref=item.ref,
|
|
583
|
+
discovery_policy="record",
|
|
584
|
+
record_revision_digest=record_revision_digest(upstream, outputs_dir=self.root),
|
|
585
|
+
)
|
|
586
|
+
)
|
|
587
|
+
continue
|
|
588
|
+
if isinstance(item.ref, SourceRecordRef):
|
|
589
|
+
from .sources import resolve_source_record # noqa: PLC0415
|
|
590
|
+
|
|
591
|
+
upstream = resolved_by_label.get(item.label)
|
|
592
|
+
if upstream is None:
|
|
593
|
+
upstream = resolved_sources.get(item.ref)
|
|
594
|
+
if upstream is None:
|
|
595
|
+
upstream = resolve_source_record(item.ref, contracts=self.contracts)
|
|
596
|
+
if not isinstance(upstream, ResolvedSourceRecord):
|
|
597
|
+
raise RecordError(f"Resolved input {item.label!r} must be a Reader source record")
|
|
598
|
+
if upstream.ref != item.ref:
|
|
599
|
+
raise RecordError(
|
|
600
|
+
f"Resolved input {item.label!r} is source record "
|
|
601
|
+
f"{upstream.ref.experiment_id}:{upstream.ref.record_id}, expected "
|
|
602
|
+
f"{item.ref.experiment_id}:{item.ref.record_id}"
|
|
603
|
+
)
|
|
604
|
+
upstream.verify_artifact_integrity()
|
|
605
|
+
evidence.append(
|
|
606
|
+
RecordInputEvidence(
|
|
607
|
+
label=item.label,
|
|
608
|
+
ref=item.ref,
|
|
609
|
+
discovery_policy="source_record",
|
|
610
|
+
record_revision_digest=upstream.revision_digest,
|
|
611
|
+
)
|
|
612
|
+
)
|
|
613
|
+
continue
|
|
614
|
+
if isinstance(item.ref, ResourceRef):
|
|
615
|
+
default_policy = "declared_resource"
|
|
616
|
+
elif isinstance(item.ref, FileRef):
|
|
617
|
+
default_policy = "declared_file"
|
|
618
|
+
else:
|
|
619
|
+
raise RecordError(f"Unsupported input reference for {item.label!r}")
|
|
620
|
+
evidence.append(
|
|
621
|
+
RecordInputEvidence(
|
|
622
|
+
label=item.label,
|
|
623
|
+
ref=item.ref,
|
|
624
|
+
discovery_policy=item.discovery_policy or default_policy,
|
|
625
|
+
artifact=capture_artifact_evidence(item.ref.path, root=self.experiment_root),
|
|
626
|
+
)
|
|
627
|
+
)
|
|
628
|
+
return tuple(evidence)
|
|
629
|
+
|
|
630
|
+
def _require_captured_inputs(self, inputs: Iterable[RecordInputEvidence]) -> tuple[RecordInputEvidence, ...]:
|
|
631
|
+
captured = tuple(inputs)
|
|
632
|
+
if any(not isinstance(item, RecordInputEvidence) for item in captured):
|
|
633
|
+
raise RecordError(
|
|
634
|
+
"Record persistence requires pre-captured RecordInputEvidence; "
|
|
635
|
+
"capture inputs before computation with RecordStore.capture_inputs()."
|
|
636
|
+
)
|
|
637
|
+
self._assert_captured_inputs_current(captured)
|
|
638
|
+
return captured
|
|
639
|
+
|
|
640
|
+
def _assert_captured_inputs_current(self, captured: Iterable[RecordInputEvidence]) -> None:
|
|
641
|
+
for item in captured:
|
|
642
|
+
if isinstance(item.ref, RecordRef):
|
|
643
|
+
current = self.latest_record(item.ref.record_id)
|
|
644
|
+
if current is None:
|
|
645
|
+
raise RecordError(
|
|
646
|
+
f"Input record {item.ref.record_id!r} changed after input evidence was captured: "
|
|
647
|
+
"the current revision is missing."
|
|
648
|
+
)
|
|
649
|
+
current_revision = record_revision_digest(current, outputs_dir=self.root)
|
|
650
|
+
if current_revision != item.record_revision_digest:
|
|
651
|
+
raise RecordError(f"Input record {item.ref.record_id!r} changed after input evidence was captured.")
|
|
652
|
+
try:
|
|
653
|
+
verify_record_artifact_integrity(current, outputs_dir=self.root)
|
|
654
|
+
except RecordError as exc:
|
|
655
|
+
raise RecordError(
|
|
656
|
+
f"Input record {item.ref.record_id!r} changed after input evidence was captured: {exc}"
|
|
657
|
+
) from exc
|
|
658
|
+
continue
|
|
659
|
+
if isinstance(item.ref, SourceRecordRef):
|
|
660
|
+
from .sources import resolve_source_record # noqa: PLC0415
|
|
661
|
+
|
|
662
|
+
try:
|
|
663
|
+
current = resolve_source_record(item.ref, contracts=self.contracts)
|
|
664
|
+
current.verify_artifact_integrity()
|
|
665
|
+
except RecordError as exc:
|
|
666
|
+
raise RecordError(
|
|
667
|
+
f"Source record {item.ref.experiment_id}:{item.ref.record_id} changed after input "
|
|
668
|
+
f"evidence was captured: {exc}"
|
|
669
|
+
) from exc
|
|
670
|
+
if current.revision_digest != item.record_revision_digest:
|
|
671
|
+
raise RecordError(
|
|
672
|
+
f"Source record {item.ref.experiment_id}:{item.ref.record_id} changed after input "
|
|
673
|
+
"evidence was captured."
|
|
674
|
+
)
|
|
675
|
+
continue
|
|
676
|
+
try:
|
|
677
|
+
current_artifact = capture_artifact_evidence(item.ref.path, root=self.experiment_root)
|
|
678
|
+
except (OSError, RecordError) as exc:
|
|
679
|
+
raise RecordError(
|
|
680
|
+
f"Input file for {item.label!r} changed after input evidence was captured: {exc}"
|
|
681
|
+
) from exc
|
|
682
|
+
if current_artifact != item.artifact:
|
|
683
|
+
raise RecordError(f"Input file for {item.label!r} changed after input evidence was captured.")
|
|
684
|
+
|
|
685
|
+
def iter_latest_records(
|
|
686
|
+
self,
|
|
687
|
+
*,
|
|
688
|
+
kind: WorkbenchRecordKind | None = None,
|
|
689
|
+
producer_kind: WorkbenchProducerKind | None = None,
|
|
690
|
+
) -> tuple[DataFrameArtifactRecord | FileBundleRecord, ...]:
|
|
691
|
+
source_experiment_resolver = self._source_experiment_resolver()
|
|
692
|
+
catalog = self._read_catalog(source_experiment_resolver=source_experiment_resolver)
|
|
693
|
+
out: list[DataFrameArtifactRecord | FileBundleRecord] = []
|
|
694
|
+
for record_id, payload in sorted(catalog["latest"].items()):
|
|
695
|
+
record = self._materialize(
|
|
696
|
+
record_id,
|
|
697
|
+
payload,
|
|
698
|
+
source_experiment_resolver=source_experiment_resolver,
|
|
699
|
+
)
|
|
700
|
+
if kind is not None and record.kind != kind:
|
|
701
|
+
continue
|
|
702
|
+
if producer_kind is not None and record.producer.kind != producer_kind:
|
|
703
|
+
continue
|
|
704
|
+
out.append(record)
|
|
705
|
+
return tuple(out)
|
|
706
|
+
|
|
707
|
+
def record_history(self, record_id: str) -> tuple[DataFrameArtifactRecord | FileBundleRecord, ...]:
|
|
708
|
+
source_experiment_resolver = self._source_experiment_resolver()
|
|
709
|
+
catalog = self._read_catalog(source_experiment_resolver=source_experiment_resolver)
|
|
710
|
+
history = catalog["history"].get(record_id, [])
|
|
711
|
+
if not isinstance(history, list):
|
|
712
|
+
raise RecordError(f"records.json history for {record_id!r} must be a list")
|
|
713
|
+
return tuple(
|
|
714
|
+
self._materialize(
|
|
715
|
+
record_id,
|
|
716
|
+
payload,
|
|
717
|
+
source_experiment_resolver=source_experiment_resolver,
|
|
718
|
+
)
|
|
719
|
+
for payload in history
|
|
720
|
+
)
|
|
721
|
+
|
|
722
|
+
def revision_counts(self, record_ids: Iterable[str] | None = None) -> dict[str, int]:
|
|
723
|
+
catalog = self._read_catalog()
|
|
724
|
+
requested = None if record_ids is None else {str(record_id) for record_id in record_ids}
|
|
725
|
+
counts: dict[str, int] = {}
|
|
726
|
+
for record_id, history in catalog["history"].items():
|
|
727
|
+
if requested is not None and record_id not in requested:
|
|
728
|
+
continue
|
|
729
|
+
if not isinstance(history, list):
|
|
730
|
+
raise RecordError(f"records.json history for {record_id!r} must be a list")
|
|
731
|
+
counts[record_id] = len(history)
|
|
732
|
+
if requested is not None:
|
|
733
|
+
for record_id in requested:
|
|
734
|
+
counts.setdefault(record_id, 0)
|
|
735
|
+
return counts
|
|
736
|
+
|
|
737
|
+
def latest_dataframe(self, record_id: str) -> DataFrameArtifactRecord | None:
|
|
738
|
+
source_experiment_resolver = self._source_experiment_resolver()
|
|
739
|
+
catalog = self._read_catalog(source_experiment_resolver=source_experiment_resolver)
|
|
740
|
+
payload = catalog["latest"].get(record_id)
|
|
741
|
+
if payload is None:
|
|
742
|
+
return None
|
|
743
|
+
record = self._materialize(
|
|
744
|
+
record_id,
|
|
745
|
+
payload,
|
|
746
|
+
source_experiment_resolver=source_experiment_resolver,
|
|
747
|
+
)
|
|
748
|
+
if not isinstance(record, DataFrameArtifactRecord):
|
|
749
|
+
raise RecordError(f"Record {record_id!r} exists but is not a dataframe artifact")
|
|
750
|
+
return record
|
|
751
|
+
|
|
752
|
+
def latest_record(self, record_id: str) -> DataFrameArtifactRecord | FileBundleRecord | None:
|
|
753
|
+
source_experiment_resolver = self._source_experiment_resolver()
|
|
754
|
+
catalog = self._read_catalog(source_experiment_resolver=source_experiment_resolver)
|
|
755
|
+
payload = catalog["latest"].get(record_id)
|
|
756
|
+
if payload is None:
|
|
757
|
+
return None
|
|
758
|
+
return self._materialize(
|
|
759
|
+
record_id,
|
|
760
|
+
payload,
|
|
761
|
+
source_experiment_resolver=source_experiment_resolver,
|
|
762
|
+
)
|
|
763
|
+
|
|
764
|
+
def read_dataframe(self, record_id: str) -> DataFrameArtifactRecord:
|
|
765
|
+
record = self.latest_dataframe(record_id)
|
|
766
|
+
if record is None:
|
|
767
|
+
raise RecordError(f"Dataframe record '{record_id}' is missing; produce it in an earlier step.")
|
|
768
|
+
return record
|
|
769
|
+
|
|
770
|
+
def read_record(self, record_id: str) -> DataFrameArtifactRecord | FileBundleRecord:
|
|
771
|
+
record = self.latest_record(record_id)
|
|
772
|
+
if record is None:
|
|
773
|
+
raise RecordError(f"Record '{record_id}' is missing; produce it in an earlier step.")
|
|
774
|
+
return record
|
|
775
|
+
|
|
776
|
+
def _revision_dir(self, step_dir: Path) -> Path:
|
|
777
|
+
index = 1
|
|
778
|
+
while True:
|
|
779
|
+
candidate = step_dir if index == 1 else step_dir.with_name(f"{step_dir.name}__r{index}")
|
|
780
|
+
if not candidate.exists():
|
|
781
|
+
return candidate
|
|
782
|
+
index += 1
|
|
783
|
+
|
|
784
|
+
def persist_dataframe(
|
|
785
|
+
self,
|
|
786
|
+
*,
|
|
787
|
+
producer_id: str,
|
|
788
|
+
producer_plugin: str,
|
|
789
|
+
out_name: str,
|
|
790
|
+
record_id: str,
|
|
791
|
+
df: pd.DataFrame,
|
|
792
|
+
contract_id: ContractId,
|
|
793
|
+
inputs: Iterable[RecordInputEvidence],
|
|
794
|
+
config_digest: str,
|
|
795
|
+
producer_config_digest: str | None = None,
|
|
796
|
+
code_digest: str | None = None,
|
|
797
|
+
build_identity: BuildIdentity | None = None,
|
|
798
|
+
producer_kind: WorkbenchProducerKind = "pipeline",
|
|
799
|
+
source_recipe: RecipeSource | None = None,
|
|
800
|
+
) -> DataFrameArtifactRecord:
|
|
801
|
+
self.contracts.validate(df, contract_id=contract_id, where=f"{producer_id}:{out_name}")
|
|
802
|
+
producer = RecordProducer(
|
|
803
|
+
kind=producer_kind,
|
|
804
|
+
id=producer_id,
|
|
805
|
+
plugin=producer_plugin,
|
|
806
|
+
source_recipe=(
|
|
807
|
+
RecordRecipeSource(recipe=source_recipe.recipe, with_=dict(source_recipe.with_ or {}))
|
|
808
|
+
if source_recipe is not None
|
|
809
|
+
else None
|
|
810
|
+
),
|
|
811
|
+
)
|
|
812
|
+
record_inputs = self._require_captured_inputs(inputs)
|
|
813
|
+
effective_build_identity = build_identity or current_build_identity()
|
|
814
|
+
effective_code_digest = code_digest or effective_build_identity.source_digest
|
|
815
|
+
effective_producer_config_digest = producer_config_digest or config_digest
|
|
816
|
+
self.ensure_layout()
|
|
817
|
+
base_name = f"{producer_id}.{producer_plugin.replace('/', '_')}"
|
|
818
|
+
base_step_dir = self.artifacts_dir / base_name
|
|
819
|
+
source_experiment_resolver = self._source_experiment_resolver()
|
|
820
|
+
catalog = self._read_catalog(source_experiment_resolver=source_experiment_resolver)
|
|
821
|
+
provenance_epoch_id = str(catalog["provenance_epoch_id"])
|
|
822
|
+
catalog_digest = digest_json(catalog)
|
|
823
|
+
previous_payload = catalog["latest"].get(record_id)
|
|
824
|
+
previous_record = None
|
|
825
|
+
if previous_payload is not None:
|
|
826
|
+
previous_record = self._materialize(
|
|
827
|
+
record_id,
|
|
828
|
+
previous_payload,
|
|
829
|
+
source_experiment_resolver=source_experiment_resolver,
|
|
830
|
+
)
|
|
831
|
+
if not isinstance(previous_record, DataFrameArtifactRecord):
|
|
832
|
+
raise RecordError(
|
|
833
|
+
f"Record id {record_id!r} is already used by a non-dataframe record; choose a unique id."
|
|
834
|
+
)
|
|
835
|
+
filename = f"{out_name}.parquet"
|
|
836
|
+
staged_path: Path | None = None
|
|
837
|
+
revision_dir: Path | None = None
|
|
838
|
+
data_path: Path | None = None
|
|
839
|
+
try:
|
|
840
|
+
with NamedTemporaryFile(
|
|
841
|
+
dir=self.artifacts_dir,
|
|
842
|
+
prefix=f".{base_name}.",
|
|
843
|
+
suffix=".parquet",
|
|
844
|
+
delete=False,
|
|
845
|
+
) as staged:
|
|
846
|
+
staged_path = Path(staged.name)
|
|
847
|
+
df.to_parquet(staged_path, index=False)
|
|
848
|
+
content_digest = sha256_file(staged_path)
|
|
849
|
+
if (
|
|
850
|
+
previous_record is not None
|
|
851
|
+
and previous_record.config_digest == config_digest
|
|
852
|
+
and previous_record.content_digest == content_digest
|
|
853
|
+
and previous_record.contract_id == contract_id
|
|
854
|
+
and previous_record.code_digest == effective_code_digest
|
|
855
|
+
and previous_record.producer_config_digest == effective_producer_config_digest
|
|
856
|
+
and previous_record.build_identity == effective_build_identity
|
|
857
|
+
and previous_record.producer == producer
|
|
858
|
+
and previous_record.inputs == record_inputs
|
|
859
|
+
and previous_record.path.name == filename
|
|
860
|
+
):
|
|
861
|
+
previous_record.verify_content_digest()
|
|
862
|
+
self.assert_catalog_snapshot(
|
|
863
|
+
provenance_epoch_id=provenance_epoch_id,
|
|
864
|
+
catalog_digest=catalog_digest,
|
|
865
|
+
)
|
|
866
|
+
return previous_record
|
|
867
|
+
|
|
868
|
+
revision_dir = self._revision_dir(base_step_dir)
|
|
869
|
+
revision_dir.mkdir(parents=True)
|
|
870
|
+
data_path = revision_dir / filename
|
|
871
|
+
staged_path.replace(data_path)
|
|
872
|
+
staged_path = None
|
|
873
|
+
record = DataFrameArtifactRecord(
|
|
874
|
+
record_id=record_id,
|
|
875
|
+
kind="dataframe_artifact",
|
|
876
|
+
producer=producer,
|
|
877
|
+
created_at=datetime.now(UTC).isoformat(),
|
|
878
|
+
inputs=record_inputs,
|
|
879
|
+
config_digest=config_digest,
|
|
880
|
+
contract_id=contract_id,
|
|
881
|
+
path=data_path,
|
|
882
|
+
content_digest=content_digest,
|
|
883
|
+
code_digest=effective_code_digest,
|
|
884
|
+
producer_config_digest=effective_producer_config_digest,
|
|
885
|
+
build_identity=effective_build_identity,
|
|
886
|
+
size_bytes=data_path.stat().st_size,
|
|
887
|
+
)
|
|
888
|
+
payload = record_to_dict(record, outputs_dir=self.root)
|
|
889
|
+
self._assert_captured_inputs_current(record_inputs)
|
|
890
|
+
catalog["latest"][record_id] = payload
|
|
891
|
+
catalog["history"].setdefault(record_id, []).append(payload)
|
|
892
|
+
self._write_catalog(
|
|
893
|
+
catalog,
|
|
894
|
+
expected_provenance_epoch_id=provenance_epoch_id,
|
|
895
|
+
expected_catalog_digest=catalog_digest,
|
|
896
|
+
)
|
|
897
|
+
return record
|
|
898
|
+
except Exception:
|
|
899
|
+
if data_path is not None:
|
|
900
|
+
data_path.unlink(missing_ok=True)
|
|
901
|
+
if revision_dir is not None:
|
|
902
|
+
with suppress(OSError):
|
|
903
|
+
revision_dir.rmdir()
|
|
904
|
+
raise
|
|
905
|
+
finally:
|
|
906
|
+
if staged_path is not None:
|
|
907
|
+
staged_path.unlink(missing_ok=True)
|
|
908
|
+
|
|
909
|
+
def append_file_bundle(
|
|
910
|
+
self,
|
|
911
|
+
*,
|
|
912
|
+
producer_kind: WorkbenchProducerKind,
|
|
913
|
+
producer_id: str,
|
|
914
|
+
producer_plugin: str,
|
|
915
|
+
record_id: str,
|
|
916
|
+
inputs: Iterable[RecordInputEvidence],
|
|
917
|
+
config_digest: str,
|
|
918
|
+
producer_config_digest: str | None = None,
|
|
919
|
+
files: list[Path],
|
|
920
|
+
description: str,
|
|
921
|
+
path_descriptions: tuple[PathDescription, ...] = (),
|
|
922
|
+
source_recipe: RecipeSource | None = None,
|
|
923
|
+
build_identity: BuildIdentity | None = None,
|
|
924
|
+
) -> FileBundleRecord:
|
|
925
|
+
producer = RecordProducer(
|
|
926
|
+
kind=producer_kind,
|
|
927
|
+
id=producer_id,
|
|
928
|
+
plugin=producer_plugin,
|
|
929
|
+
source_recipe=(
|
|
930
|
+
RecordRecipeSource(recipe=source_recipe.recipe, with_=dict(source_recipe.with_ or {}))
|
|
931
|
+
if source_recipe is not None
|
|
932
|
+
else None
|
|
933
|
+
),
|
|
934
|
+
)
|
|
935
|
+
return self._append_file_bundle_record(
|
|
936
|
+
producer=producer,
|
|
937
|
+
record_id=record_id,
|
|
938
|
+
inputs=inputs,
|
|
939
|
+
config_digest=config_digest,
|
|
940
|
+
producer_config_digest=producer_config_digest,
|
|
941
|
+
files=files,
|
|
942
|
+
description=description,
|
|
943
|
+
path_descriptions=path_descriptions,
|
|
944
|
+
build_identity=build_identity,
|
|
945
|
+
)
|
|
946
|
+
|
|
947
|
+
def _append_file_bundle_record(
|
|
948
|
+
self,
|
|
949
|
+
*,
|
|
950
|
+
producer: RecordProducer,
|
|
951
|
+
record_id: str,
|
|
952
|
+
inputs: Iterable[RecordInputEvidence],
|
|
953
|
+
config_digest: str,
|
|
954
|
+
producer_config_digest: str | None,
|
|
955
|
+
files: list[Path],
|
|
956
|
+
description: str,
|
|
957
|
+
path_descriptions: tuple[PathDescription, ...],
|
|
958
|
+
build_identity: BuildIdentity | None,
|
|
959
|
+
) -> FileBundleRecord:
|
|
960
|
+
normalized_files = tuple(sorted(dict.fromkeys(files), key=str))
|
|
961
|
+
# Validate bundle semantics before touching the filesystem so malformed
|
|
962
|
+
# descriptions and path mappings fail independently of artifact state.
|
|
963
|
+
description, path_descriptions = normalize_file_bundle_metadata(
|
|
964
|
+
producer_kind=producer.kind,
|
|
965
|
+
files=normalized_files,
|
|
966
|
+
description=description,
|
|
967
|
+
path_descriptions=path_descriptions,
|
|
968
|
+
)
|
|
969
|
+
materialized_paths = tuple(path if path.is_absolute() else self.root / path for path in normalized_files)
|
|
970
|
+
missing = [path for path in materialized_paths if not path.exists()]
|
|
971
|
+
if missing:
|
|
972
|
+
raise RecordError(
|
|
973
|
+
f"File-bundle record {record_id!r} cannot persist missing files: "
|
|
974
|
+
+ ", ".join(str(path) for path in missing)
|
|
975
|
+
)
|
|
976
|
+
non_files = [path for path in materialized_paths if not path.is_file()]
|
|
977
|
+
if non_files:
|
|
978
|
+
raise RecordError(
|
|
979
|
+
f"File-bundle record {record_id!r} cannot persist non-file paths: "
|
|
980
|
+
+ ", ".join(str(path) for path in non_files)
|
|
981
|
+
)
|
|
982
|
+
file_evidence = tuple(capture_artifact_evidence(path, root=self.root) for path in materialized_paths)
|
|
983
|
+
effective_build_identity = build_identity or current_build_identity()
|
|
984
|
+
record_inputs = self._require_captured_inputs(inputs)
|
|
985
|
+
self.ensure_layout()
|
|
986
|
+
source_experiment_resolver = self._source_experiment_resolver()
|
|
987
|
+
catalog = self._read_catalog(source_experiment_resolver=source_experiment_resolver)
|
|
988
|
+
provenance_epoch_id = str(catalog["provenance_epoch_id"])
|
|
989
|
+
catalog_digest = digest_json(catalog)
|
|
990
|
+
previous_payload = catalog["latest"].get(record_id)
|
|
991
|
+
if previous_payload is not None:
|
|
992
|
+
previous_record = self._materialize(
|
|
993
|
+
record_id,
|
|
994
|
+
previous_payload,
|
|
995
|
+
source_experiment_resolver=source_experiment_resolver,
|
|
996
|
+
)
|
|
997
|
+
if isinstance(previous_record, DataFrameArtifactRecord):
|
|
998
|
+
raise RecordError(f"Record id {record_id!r} is already used by a dataframe record; choose a unique id.")
|
|
999
|
+
record = FileBundleRecord(
|
|
1000
|
+
record_id=record_id,
|
|
1001
|
+
kind="file_bundle",
|
|
1002
|
+
producer=producer,
|
|
1003
|
+
created_at=datetime.now(UTC).isoformat(),
|
|
1004
|
+
inputs=record_inputs,
|
|
1005
|
+
config_digest=config_digest,
|
|
1006
|
+
files=normalized_files,
|
|
1007
|
+
description=description,
|
|
1008
|
+
path_descriptions=path_descriptions,
|
|
1009
|
+
file_evidence=file_evidence,
|
|
1010
|
+
producer_config_digest=producer_config_digest or config_digest,
|
|
1011
|
+
build_identity=effective_build_identity,
|
|
1012
|
+
)
|
|
1013
|
+
payload = record_to_dict(record, outputs_dir=self.root)
|
|
1014
|
+
self._assert_captured_inputs_current(record_inputs)
|
|
1015
|
+
catalog["latest"][record_id] = payload
|
|
1016
|
+
catalog["history"].setdefault(record_id, []).append(payload)
|
|
1017
|
+
self._write_catalog(
|
|
1018
|
+
catalog,
|
|
1019
|
+
expected_provenance_epoch_id=provenance_epoch_id,
|
|
1020
|
+
expected_catalog_digest=catalog_digest,
|
|
1021
|
+
)
|
|
1022
|
+
return record
|