mostlyright-data 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mostlyright/data_harness/__init__.py +158 -0
- mostlyright/data_harness/acquisition/__init__.py +55 -0
- mostlyright/data_harness/acquisition/http.py +2773 -0
- mostlyright/data_harness/acquisition/parsing.py +809 -0
- mostlyright/data_harness/acquisition/ranges.py +495 -0
- mostlyright/data_harness/acquisition/result_download.py +360 -0
- mostlyright/data_harness/acquisition/retention_admission.py +248 -0
- mostlyright/data_harness/acquisition/sandbox.py +4888 -0
- mostlyright/data_harness/acquisition/url_policy.py +530 -0
- mostlyright/data_harness/agent_runtime.py +2743 -0
- mostlyright/data_harness/assets/logo-ink.svg +31 -0
- mostlyright/data_harness/backends/__init__.py +28 -0
- mostlyright/data_harness/backends/pandas_backend.py +350 -0
- mostlyright/data_harness/backends/polars_backend.py +366 -0
- mostlyright/data_harness/backends/protocol.py +124 -0
- mostlyright/data_harness/backends/reference.py +83 -0
- mostlyright/data_harness/backends/registry.py +55 -0
- mostlyright/data_harness/backends/restrictions.py +126 -0
- mostlyright/data_harness/canonical.py +333 -0
- mostlyright/data_harness/catalog_job.py +625 -0
- mostlyright/data_harness/cli.py +5398 -0
- mostlyright/data_harness/contracts.py +53 -0
- mostlyright/data_harness/coordinator.py +1307 -0
- mostlyright/data_harness/deploy.py +924 -0
- mostlyright/data_harness/deploy_target.py +312 -0
- mostlyright/data_harness/deployment_evidence.py +1067 -0
- mostlyright/data_harness/event_presentation.py +576 -0
- mostlyright/data_harness/events.py +2152 -0
- mostlyright/data_harness/fast_delimited.py +239 -0
- mostlyright/data_harness/fleet.py +237 -0
- mostlyright/data_harness/formats.py +236 -0
- mostlyright/data_harness/governors.py +1163 -0
- mostlyright/data_harness/hosted_bootstrap.py +972 -0
- mostlyright/data_harness/hosted_crawler.py +1115 -0
- mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
- mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
- mostlyright/data_harness/hosted_crawler_job.py +1277 -0
- mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
- mostlyright/data_harness/hosted_dataset.py +1500 -0
- mostlyright/data_harness/hosted_deploy.py +3037 -0
- mostlyright/data_harness/hosted_handoff.py +62 -0
- mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
- mostlyright/data_harness/hosted_ingestion_job.py +356 -0
- mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
- mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
- mostlyright/data_harness/hosted_session_worker.py +3554 -0
- mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
- mostlyright/data_harness/hosted_worker.py +6784 -0
- mostlyright/data_harness/ingestion/__init__.py +56 -0
- mostlyright/data_harness/ingestion/contracts.py +461 -0
- mostlyright/data_harness/ingestion/faults.py +42 -0
- mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
- mostlyright/data_harness/ingestion/spool.py +130 -0
- mostlyright/data_harness/ingestion/store.py +885 -0
- mostlyright/data_harness/key_seam.py +434 -0
- mostlyright/data_harness/linux_process_boundary.py +262 -0
- mostlyright/data_harness/local_contracts.py +2880 -0
- mostlyright/data_harness/local_search/__init__.py +5 -0
- mostlyright/data_harness/local_search/build_index.py +1087 -0
- mostlyright/data_harness/local_search/contracts.py +920 -0
- mostlyright/data_harness/local_search/query_trace.py +266 -0
- mostlyright/data_harness/local_search/retrieval.py +700 -0
- mostlyright/data_harness/local_search/sealed.py +474 -0
- mostlyright/data_harness/local_search/service.py +784 -0
- mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
- mostlyright/data_harness/nbrender/__init__.py +12 -0
- mostlyright/data_harness/nbrender/chrome.py +359 -0
- mostlyright/data_harness/nbrender/code_body.py +266 -0
- mostlyright/data_harness/nbrender/document.py +407 -0
- mostlyright/data_harness/nbrender/frame.py +275 -0
- mostlyright/data_harness/nbrender/interactive.py +337 -0
- mostlyright/data_harness/nbrender/markdown_body.py +477 -0
- mostlyright/data_harness/nbrender/mr_components.py +134 -0
- mostlyright/data_harness/nbrender/outputs_data.py +595 -0
- mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
- mostlyright/data_harness/nbrender/outputs_source.py +260 -0
- mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
- mostlyright/data_harness/nbrender/outputs_text.py +400 -0
- mostlyright/data_harness/nbrender/parse.py +394 -0
- mostlyright/data_harness/nbrender/status.py +40 -0
- mostlyright/data_harness/nbrender/tokens.py +1295 -0
- mostlyright/data_harness/notebook.py +1710 -0
- mostlyright/data_harness/offline.py +2049 -0
- mostlyright/data_harness/operation_registry.py +1007 -0
- mostlyright/data_harness/operator_setup.py +239 -0
- mostlyright/data_harness/pipeline.py +6428 -0
- mostlyright/data_harness/plan_graph.py +2026 -0
- mostlyright/data_harness/preparation/__init__.py +104 -0
- mostlyright/data_harness/preparation/contracts.py +1017 -0
- mostlyright/data_harness/preparation/engine.py +221 -0
- mostlyright/data_harness/preparation/errors.py +14 -0
- mostlyright/data_harness/preparation/gates.py +751 -0
- mostlyright/data_harness/preparation/joins.py +574 -0
- mostlyright/data_harness/preparation/profile.py +384 -0
- mostlyright/data_harness/preparation/table.py +217 -0
- mostlyright/data_harness/preparation/transforms.py +568 -0
- mostlyright/data_harness/progress_events.py +534 -0
- mostlyright/data_harness/readers/__init__.py +46 -0
- mostlyright/data_harness/readers/containers.py +963 -0
- mostlyright/data_harness/readers/contracts.py +542 -0
- mostlyright/data_harness/readers/delimited.py +257 -0
- mostlyright/data_harness/readers/grib2/__init__.py +33 -0
- mostlyright/data_harness/readers/grib2/admission.py +722 -0
- mostlyright/data_harness/readers/grib2/decode.py +1009 -0
- mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
- mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
- mostlyright/data_harness/readers/json_tabular.py +485 -0
- mostlyright/data_harness/readers/registry.py +514 -0
- mostlyright/data_harness/readers/samples/README.md +110 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples.py +582 -0
- mostlyright/data_harness/readers/spreadsheet.py +803 -0
- mostlyright/data_harness/readers/tabular.py +510 -0
- mostlyright/data_harness/recipe.py +5321 -0
- mostlyright/data_harness/repair/__init__.py +78 -0
- mostlyright/data_harness/repair/adapters.py +274 -0
- mostlyright/data_harness/repair/contracts.py +872 -0
- mostlyright/data_harness/repair/coordinator.py +1099 -0
- mostlyright/data_harness/repair/errors.py +16 -0
- mostlyright/data_harness/review.py +2533 -0
- mostlyright/data_harness/rowset.py +283 -0
- mostlyright/data_harness/serving.py +1975 -0
- mostlyright/data_harness/serving_edge.py +590 -0
- mostlyright/data_harness/serving_http.py +1031 -0
- mostlyright/data_harness/session_probes.py +759 -0
- mostlyright/data_harness/signing.py +101 -0
- mostlyright/data_harness/source_discovery.py +898 -0
- mostlyright/data_harness/sources/__init__.py +209 -0
- mostlyright/data_harness/sources/_adapter_steps.py +213 -0
- mostlyright/data_harness/sources/adapters.py +1214 -0
- mostlyright/data_harness/sources/cadence.py +1428 -0
- mostlyright/data_harness/sources/cadence_emission.py +453 -0
- mostlyright/data_harness/sources/cadence_history.py +546 -0
- mostlyright/data_harness/sources/catalog/__init__.py +17 -0
- mostlyright/data_harness/sources/catalog/admission.py +477 -0
- mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
- mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
- mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
- mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
- mostlyright/data_harness/sources/catalog/channel.py +523 -0
- mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
- mostlyright/data_harness/sources/catalog/contracts.py +825 -0
- mostlyright/data_harness/sources/catalog/coverage.py +137 -0
- mostlyright/data_harness/sources/catalog/delta.py +1340 -0
- mostlyright/data_harness/sources/catalog/embedding.py +532 -0
- mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
- mostlyright/data_harness/sources/catalog/fill.py +3889 -0
- mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
- mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
- mostlyright/data_harness/sources/catalog/gating.py +374 -0
- mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
- mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
- mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
- mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
- mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
- mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
- mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
- mostlyright/data_harness/sources/catalog/health.py +447 -0
- mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
- mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
- mostlyright/data_harness/sources/catalog/neural.py +1618 -0
- mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
- mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
- mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
- mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
- mostlyright/data_harness/sources/catalog/recommend.py +171 -0
- mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
- mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
- mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
- mostlyright/data_harness/sources/catalog/sealed.py +560 -0
- mostlyright/data_harness/sources/catalog/search.py +230 -0
- mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
- mostlyright/data_harness/sources/catalog/update.py +891 -0
- mostlyright/data_harness/sources/collections.py +815 -0
- mostlyright/data_harness/sources/contracts.py +2223 -0
- mostlyright/data_harness/sources/deletion.py +761 -0
- mostlyright/data_harness/sources/fitness.py +162 -0
- mostlyright/data_harness/sources/governance.py +163 -0
- mostlyright/data_harness/sources/hosted.py +173 -0
- mostlyright/data_harness/sources/integration.py +218 -0
- mostlyright/data_harness/sources/range_reader.py +418 -0
- mostlyright/data_harness/sources/registry.py +514 -0
- mostlyright/data_harness/sources/rights_rule.py +59 -0
- mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
- mostlyright/data_harness/sources/sports.py +521 -0
- mostlyright/data_harness/sources/stream.py +524 -0
- mostlyright/data_harness/sources/stream_connector.py +418 -0
- mostlyright/data_harness/sources/stream_recorder.py +1404 -0
- mostlyright/data_harness/studio_boundary.py +2019 -0
- mostlyright/data_harness/thin/__init__.py +37 -0
- mostlyright/data_harness/thin/acquire.py +1137 -0
- mostlyright/data_harness/thin/acquire_cancel.py +579 -0
- mostlyright/data_harness/thin/approvals.py +617 -0
- mostlyright/data_harness/thin/commands.py +406 -0
- mostlyright/data_harness/thin/download.py +194 -0
- mostlyright/data_harness/thin/narrative.py +589 -0
- mostlyright/data_harness/thin/parity.py +1070 -0
- mostlyright/data_harness/thin/propose.py +2759 -0
- mostlyright/data_harness/thin/research.py +1663 -0
- mostlyright/data_harness/thin/router.py +924 -0
- mostlyright/data_harness/thin/runs.py +519 -0
- mostlyright/data_harness/thin/session.py +281 -0
- mostlyright/data_harness/thin/stream.py +501 -0
- mostlyright/data_harness/thin/transport.py +187 -0
- mostlyright/data_harness/thin/vocabulary.py +368 -0
- mostlyright/data_harness/thin/workers.py +164 -0
- mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
- mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
- mostlyright/data_harness/unit_flow.py +927 -0
- mostlyright/data_harness/units.py +572 -0
- mostlyright/data_harness/ux/__init__.py +9 -0
- mostlyright/data_harness/ux/approve.py +485 -0
- mostlyright/data_harness/ux/author_yaml.py +597 -0
- mostlyright/data_harness/ux/cloud_auth.py +447 -0
- mostlyright/data_harness/ux/commands/__init__.py +260 -0
- mostlyright/data_harness/ux/commands/approve.py +136 -0
- mostlyright/data_harness/ux/commands/auth.py +744 -0
- mostlyright/data_harness/ux/commands/author.py +79 -0
- mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
- mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
- mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
- mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
- mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
- mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
- mostlyright/data_harness/ux/commands/deploy.py +134 -0
- mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
- mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
- mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
- mostlyright/data_harness/ux/commands/diff.py +74 -0
- mostlyright/data_harness/ux/commands/index.py +84 -0
- mostlyright/data_harness/ux/commands/inventory.py +47 -0
- mostlyright/data_harness/ux/commands/list_builds.py +143 -0
- mostlyright/data_harness/ux/commands/login.py +63 -0
- mostlyright/data_harness/ux/commands/peek.py +236 -0
- mostlyright/data_harness/ux/commands/plan_check.py +90 -0
- mostlyright/data_harness/ux/commands/preflight.py +97 -0
- mostlyright/data_harness/ux/commands/record.py +107 -0
- mostlyright/data_harness/ux/commands/review_setup.py +47 -0
- mostlyright/data_harness/ux/commands/search.py +440 -0
- mostlyright/data_harness/ux/commands/show.py +61 -0
- mostlyright/data_harness/ux/commands/whoami.py +37 -0
- mostlyright/data_harness/ux/credential_native.py +551 -0
- mostlyright/data_harness/ux/credential_store.py +1055 -0
- mostlyright/data_harness/ux/credentials.py +631 -0
- mostlyright/data_harness/ux/diffing.py +444 -0
- mostlyright/data_harness/ux/headline.py +671 -0
- mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
- mostlyright/data_harness/ux/hosted_run_status.py +619 -0
- mostlyright/data_harness/ux/inventory.py +427 -0
- mostlyright/data_harness/ux/local_review.py +375 -0
- mostlyright/data_harness/ux/login.py +691 -0
- mostlyright/data_harness/ux/path_kind.py +147 -0
- mostlyright/data_harness/ux/peek.py +1000 -0
- mostlyright/data_harness/ux/plain_file.py +178 -0
- mostlyright/data_harness/ux/plan_check.py +311 -0
- mostlyright/data_harness/ux/preflight.py +918 -0
- mostlyright/data_harness/ux/readers.py +1124 -0
- mostlyright/data_harness/ux/remediation.py +2195 -0
- mostlyright/data_harness/ux/render.py +657 -0
- mostlyright/data_harness/ux/workload.py +1077 -0
- mostlyright/data_harness/viewer.py +3713 -0
- mostlyright/data_harness/visual_run/__init__.py +83 -0
- mostlyright/data_harness/visual_run/authoring.py +235 -0
- mostlyright/data_harness/visual_run/contracts.py +673 -0
- mostlyright/data_harness/visual_run/materialize.py +486 -0
- mostlyright/data_harness/visual_run/observations.py +874 -0
- mostlyright/data_harness/visual_run/query.py +259 -0
- mostlyright/data_harness/visual_run/reducer.py +280 -0
- mostlyright/data_harness/visual_run/sdk.py +892 -0
- mostlyright/data_harness/visual_run/store.py +584 -0
- mostlyright/data_harness/visual_run/transport.py +239 -0
- mostlyright/data_harness/watch.py +2999 -0
- mostlyright_data-0.9.0.dist-info/METADATA +607 -0
- mostlyright_data-0.9.0.dist-info/RECORD +314 -0
- mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
- mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
|
@@ -0,0 +1,3037 @@
|
|
|
1
|
+
"""Crash-safe deployment of one exact reviewed Build through Cloud and Studio."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import fcntl
|
|
6
|
+
import hashlib
|
|
7
|
+
import importlib
|
|
8
|
+
import ipaddress
|
|
9
|
+
import os
|
|
10
|
+
import re
|
|
11
|
+
import ssl
|
|
12
|
+
import stat
|
|
13
|
+
import time
|
|
14
|
+
from collections.abc import Callable, Mapping
|
|
15
|
+
from dataclasses import dataclass
|
|
16
|
+
from datetime import UTC, datetime
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Any, Protocol
|
|
19
|
+
from urllib.error import HTTPError
|
|
20
|
+
from urllib.parse import urlsplit
|
|
21
|
+
from urllib.request import HTTPRedirectHandler, HTTPSHandler, ProxyHandler, Request, build_opener
|
|
22
|
+
from uuid import NAMESPACE_URL, UUID, uuid5
|
|
23
|
+
|
|
24
|
+
from croniter import CroniterBadCronError, CroniterBadDateError, croniter
|
|
25
|
+
|
|
26
|
+
from mostlyright.data_harness import canonical, deployment_evidence, pipeline
|
|
27
|
+
from mostlyright.data_harness.deploy import DeploymentRequest
|
|
28
|
+
from mostlyright.data_harness.hosted_crawler_protocol import (
|
|
29
|
+
OPENLIGADB_EGRESS_POLICY_ATTESTATION,
|
|
30
|
+
PUBLIC_HTTPS_EGRESS_POLICY_ATTESTATION,
|
|
31
|
+
)
|
|
32
|
+
from mostlyright.data_harness.studio_boundary import StudioBoundaryError, client_contract_identity
|
|
33
|
+
from mostlyright.data_harness.ux.credentials import ResolvedCloudCredentials
|
|
34
|
+
|
|
35
|
+
CLOUD_TOKEN_PATH = "/api/cli/studio-token"
|
|
36
|
+
CLOUD_DATASET_BINDINGS_PATH = "/api/cli/table-bindings"
|
|
37
|
+
CLOUD_TABLE_BINDING_SCHEMA_VERSION = "mostlyright-cloud-table-binding.v1"
|
|
38
|
+
CLOUD_TABLE_BINDING_CONTRACT_SHA256 = (
|
|
39
|
+
"f408d3ce54994624778e74186077557228bfd89f95ab882a152efa1b5af111fd"
|
|
40
|
+
)
|
|
41
|
+
DEPLOYMENT_STATE_SCHEMA = "mostlyright-hosted-deployment-state.v3"
|
|
42
|
+
DEPLOYMENT_RECEIPT_SCHEMA = "mostlyright-hosted-deployment-receipt.v3"
|
|
43
|
+
STUDIO_SCHEMA_VERSION = "3.0.0"
|
|
44
|
+
REQUEST_TIMEOUT_SECONDS = 30.0
|
|
45
|
+
#: How many EXTRA attempts one transient Cloud 503 earns before this deployment
|
|
46
|
+
#: refuses.
|
|
47
|
+
#:
|
|
48
|
+
#: ⚠ WHY THIS EXISTS AT ALL. Studio runs behind Cloud and scales to zero; a cold
|
|
49
|
+
#: start takes ~20s to become servable. Cloud provisions this workspace's Studio
|
|
50
|
+
#: tenancy on EVERY token exchange, so the request that WAKES Studio is the same
|
|
51
|
+
#: request that has to wait for it. Before 2026-08-20 that request simply
|
|
52
|
+
#: answered 503 and this function raised DEPLOY_STUDIO_UNAVAILABLE immediately --
|
|
53
|
+
#: which meant `mr-data` could never deploy after any idle period, because every
|
|
54
|
+
#: attempt woke Studio, gave up on it, and left it to scale back to zero. Eight
|
|
55
|
+
#: such failures were recorded against production on 2026-08-20 alone.
|
|
56
|
+
#:
|
|
57
|
+
#: Two attempts is deliberate rather than generous: Cloud's own provisioning
|
|
58
|
+
#: budget already clears the observed cold start, so a retry here is the second
|
|
59
|
+
#: line, covering the case where Studio was not merely cold but genuinely
|
|
60
|
+
#: restarting.
|
|
61
|
+
TOKEN_EXCHANGE_RETRY_ATTEMPTS = 2
|
|
62
|
+
#: Ceiling on any single retry wait, however large a `retry-after` Cloud sends.
|
|
63
|
+
#: A deployment must stay a bounded, interruptible foreground operation.
|
|
64
|
+
TOKEN_EXCHANGE_MAX_RETRY_SECONDS = 30.0
|
|
65
|
+
MAX_TOKEN_RESPONSE_BYTES = 32 * 1024
|
|
66
|
+
MAX_BINDING_RESPONSE_BYTES = 32 * 1024
|
|
67
|
+
MAX_STATE_BYTES = 8 * 1024 * 1024
|
|
68
|
+
STATE_RELATIVE_PATH = Path("evidence") / "deployment" / "state.json"
|
|
69
|
+
|
|
70
|
+
_DIGEST = re.compile(r"^[0-9a-f]{64}$")
|
|
71
|
+
_PREFIXED_DIGEST = re.compile(r"^sha256:[0-9a-f]{64}$")
|
|
72
|
+
_IMMUTABLE_IMAGE = re.compile(r"^[a-z0-9][a-z0-9._/-]{0,254}@sha256:[0-9a-f]{64}$")
|
|
73
|
+
_OBJECT_GENERATION = re.compile(r"^[1-9][0-9]{0,31}$")
|
|
74
|
+
_CLI_KEY = re.compile(r"^mr_cli_[A-Za-z0-9_-]{64}$")
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class HostedDeployError(RuntimeError):
|
|
78
|
+
"""A typed, credential-safe deployment refusal."""
|
|
79
|
+
|
|
80
|
+
def __init__(self, code: str, detail: str) -> None:
|
|
81
|
+
self.code = code
|
|
82
|
+
self.detail = detail
|
|
83
|
+
super().__init__(f"{code}: {detail}")
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
@dataclass(frozen=True)
|
|
87
|
+
class StudioToken:
|
|
88
|
+
studio_base_url: str
|
|
89
|
+
workspace_id: UUID
|
|
90
|
+
token: str
|
|
91
|
+
expires_at: datetime
|
|
92
|
+
|
|
93
|
+
def __repr__(self) -> str:
|
|
94
|
+
return (
|
|
95
|
+
"StudioToken("
|
|
96
|
+
f"studio_base_url={self.studio_base_url!r}, workspace_id={self.workspace_id!r}, "
|
|
97
|
+
f"token='<redacted>', expires_at={self.expires_at!r})"
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
@dataclass(frozen=True)
|
|
102
|
+
class StudioResponse:
|
|
103
|
+
status_code: int
|
|
104
|
+
body: Mapping[str, Any]
|
|
105
|
+
etag: str | None = None
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
class TokenExchangeTransport(Protocol):
|
|
109
|
+
def request(
|
|
110
|
+
self,
|
|
111
|
+
method: str,
|
|
112
|
+
url: str,
|
|
113
|
+
headers: Mapping[str, str],
|
|
114
|
+
body: bytes | None,
|
|
115
|
+
maximum: int,
|
|
116
|
+
) -> tuple[int, bytes, Mapping[str, str]]: ...
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
class StudioDeploymentClient(Protocol):
|
|
120
|
+
def call(
|
|
121
|
+
self,
|
|
122
|
+
operation: str,
|
|
123
|
+
body: Mapping[str, Any] | None,
|
|
124
|
+
*,
|
|
125
|
+
idempotency_key: str | None = None,
|
|
126
|
+
resource_id: UUID | None = None,
|
|
127
|
+
if_match: str | None = None,
|
|
128
|
+
) -> StudioResponse: ...
|
|
129
|
+
|
|
130
|
+
def preflight(self, body: Mapping[str, Any]) -> StudioResponse: ...
|
|
131
|
+
|
|
132
|
+
def read(self, operation: str, *, resource_id: UUID | None = None) -> StudioResponse: ...
|
|
133
|
+
|
|
134
|
+
def upload(self, session: Mapping[str, Any], content: bytes) -> str: ...
|
|
135
|
+
|
|
136
|
+
def close(self) -> None: ...
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
class DeploymentStateStore(Protocol):
|
|
140
|
+
@property
|
|
141
|
+
def path(self) -> Path: ...
|
|
142
|
+
|
|
143
|
+
def exists(self) -> bool: ...
|
|
144
|
+
|
|
145
|
+
def load(self) -> dict[str, Any]: ...
|
|
146
|
+
|
|
147
|
+
def save(self, state: Mapping[str, Any]) -> None: ...
|
|
148
|
+
|
|
149
|
+
def close(self) -> None: ...
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
class _NoRedirect(HTTPRedirectHandler):
|
|
153
|
+
def redirect_request(
|
|
154
|
+
self,
|
|
155
|
+
request: Request,
|
|
156
|
+
file_pointer: Any,
|
|
157
|
+
code: int,
|
|
158
|
+
message: str,
|
|
159
|
+
headers: Any,
|
|
160
|
+
new_url: str,
|
|
161
|
+
) -> None:
|
|
162
|
+
return None
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
class UrlLibTokenExchangeTransport:
|
|
166
|
+
"""Bounded Cloud token exchange with ambient proxies and redirects disabled."""
|
|
167
|
+
|
|
168
|
+
def __init__(self) -> None:
|
|
169
|
+
self._opener = build_opener(
|
|
170
|
+
ProxyHandler({}), HTTPSHandler(context=ssl.create_default_context()), _NoRedirect()
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
def request(
|
|
174
|
+
self,
|
|
175
|
+
method: str,
|
|
176
|
+
url: str,
|
|
177
|
+
headers: Mapping[str, str],
|
|
178
|
+
body: bytes | None,
|
|
179
|
+
maximum: int,
|
|
180
|
+
) -> tuple[int, bytes, Mapping[str, str]]:
|
|
181
|
+
request = Request(url, data=body, headers=dict(headers), method=method)
|
|
182
|
+
try:
|
|
183
|
+
response = self._opener.open(request, timeout=REQUEST_TIMEOUT_SECONDS)
|
|
184
|
+
except HTTPError as error:
|
|
185
|
+
response = error
|
|
186
|
+
with response:
|
|
187
|
+
raw = response.read(maximum + 1)
|
|
188
|
+
status = int(response.status)
|
|
189
|
+
response_headers = {key.lower(): value for key, value in response.headers.items()}
|
|
190
|
+
if len(raw) > maximum:
|
|
191
|
+
raise HostedDeployError(
|
|
192
|
+
"DEPLOY_TOKEN_EXCHANGE_FAILED", "the Cloud token response exceeds its byte limit"
|
|
193
|
+
)
|
|
194
|
+
return status, raw, response_headers
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _transient_retry_seconds(response_headers: Mapping[str, str]) -> float | None:
|
|
198
|
+
"""Seconds Cloud asked this deployment to wait, or ``None`` if it asked at all.
|
|
199
|
+
|
|
200
|
+
⚠ THE ABSENCE OF THE HEADER IS THE SIGNAL, not a missing detail to paper
|
|
201
|
+
over. Cloud answers a 503 in exactly two shapes, and they mean opposite
|
|
202
|
+
things:
|
|
203
|
+
|
|
204
|
+
* ``serviceUnavailable()`` -- "our failure, not yours; please retry" --
|
|
205
|
+
stamps ``retry-after``. This is the cold-start case, and it is worth
|
|
206
|
+
retrying.
|
|
207
|
+
* ``studioUnconfigured()`` -- Studio is not configured on that deployment
|
|
208
|
+
at all -- stamps NO ``retry-after``. Retrying it would burn a bounded
|
|
209
|
+
deployment budget on a condition no amount of waiting can change.
|
|
210
|
+
|
|
211
|
+
Both bodies are deliberately generic (Cloud does not leak which of its
|
|
212
|
+
environment variables is at fault), so this header is the ONLY thing that
|
|
213
|
+
distinguishes them from out here. Returning ``None`` for a malformed or
|
|
214
|
+
out-of-range value fails toward the safe reading -- refuse now rather than
|
|
215
|
+
sleep on a number we do not trust.
|
|
216
|
+
"""
|
|
217
|
+
|
|
218
|
+
raw = response_headers.get("retry-after", "")
|
|
219
|
+
if not raw.isdecimal():
|
|
220
|
+
return None
|
|
221
|
+
seconds = int(raw)
|
|
222
|
+
if not 1 <= seconds <= 3600:
|
|
223
|
+
return None
|
|
224
|
+
return float(min(seconds, TOKEN_EXCHANGE_MAX_RETRY_SECONDS))
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def exchange_studio_token(
|
|
228
|
+
credentials: ResolvedCloudCredentials,
|
|
229
|
+
*,
|
|
230
|
+
transport: TokenExchangeTransport | None = None,
|
|
231
|
+
now: Callable[[], datetime] = lambda: datetime.now(UTC),
|
|
232
|
+
sleep: Callable[[float], None] = time.sleep,
|
|
233
|
+
) -> StudioToken:
|
|
234
|
+
"""Exchange the selected Cloud credential for a short-lived human Studio token."""
|
|
235
|
+
|
|
236
|
+
_require_cli_credential(credentials)
|
|
237
|
+
cloud_url = _service_url(credentials.cloud_url, "Cloud URL", allow_loopback_http=False)
|
|
238
|
+
selected = transport or UrlLibTokenExchangeTransport()
|
|
239
|
+
attempts_remaining = TOKEN_EXCHANGE_RETRY_ATTEMPTS
|
|
240
|
+
while True:
|
|
241
|
+
try:
|
|
242
|
+
status, raw, response_headers = selected.request(
|
|
243
|
+
"POST",
|
|
244
|
+
f"{cloud_url}{CLOUD_TOKEN_PATH}",
|
|
245
|
+
{"Accept": "application/json", "x-api-key": credentials.raw_key},
|
|
246
|
+
b"",
|
|
247
|
+
MAX_TOKEN_RESPONSE_BYTES,
|
|
248
|
+
)
|
|
249
|
+
except HostedDeployError:
|
|
250
|
+
raise
|
|
251
|
+
except Exception as error:
|
|
252
|
+
raise HostedDeployError(
|
|
253
|
+
"DEPLOY_TOKEN_EXCHANGE_FAILED", "the Cloud token exchange could not be reached"
|
|
254
|
+
) from error
|
|
255
|
+
if status != 503 or attempts_remaining == 0:
|
|
256
|
+
break
|
|
257
|
+
delay = _transient_retry_seconds(response_headers)
|
|
258
|
+
if delay is None:
|
|
259
|
+
break
|
|
260
|
+
attempts_remaining -= 1
|
|
261
|
+
sleep(delay)
|
|
262
|
+
try:
|
|
263
|
+
parsed = canonical.parse_json(raw)
|
|
264
|
+
except canonical.CanonicalJSONError:
|
|
265
|
+
parsed = None
|
|
266
|
+
if status == 401:
|
|
267
|
+
raise HostedDeployError(
|
|
268
|
+
"DEPLOY_AUTHENTICATION_FAILED", "the selected Cloud credential was rejected"
|
|
269
|
+
)
|
|
270
|
+
if status == 402:
|
|
271
|
+
raise HostedDeployError(
|
|
272
|
+
"DEPLOY_SUBSCRIPTION_REQUIRED", "this workspace has no hosted deployment access"
|
|
273
|
+
)
|
|
274
|
+
if status == 429:
|
|
275
|
+
retry_after = response_headers.get("retry-after", "")
|
|
276
|
+
suffix = (
|
|
277
|
+
f" after {retry_after} seconds"
|
|
278
|
+
if retry_after.isdecimal() and 1 <= int(retry_after) <= 3600
|
|
279
|
+
else ""
|
|
280
|
+
)
|
|
281
|
+
raise HostedDeployError(
|
|
282
|
+
"DEPLOY_RATE_LIMITED", f"Cloud asked this deployment to retry{suffix}"
|
|
283
|
+
)
|
|
284
|
+
if status == 503:
|
|
285
|
+
raise HostedDeployError(
|
|
286
|
+
"DEPLOY_STUDIO_UNAVAILABLE",
|
|
287
|
+
"Studio is not configured, or stayed unavailable across every attempt",
|
|
288
|
+
)
|
|
289
|
+
if status != 200 or not isinstance(parsed, dict):
|
|
290
|
+
raise HostedDeployError(
|
|
291
|
+
"DEPLOY_TOKEN_EXCHANGE_FAILED", f"the Cloud token exchange failed (HTTP {status})"
|
|
292
|
+
)
|
|
293
|
+
base_url = _required_text(parsed, "studio_base_url")
|
|
294
|
+
workspace_id = _uuid(_required_text(parsed, "workspace_id"), "workspace_id")
|
|
295
|
+
token = _required_text(parsed, "token")
|
|
296
|
+
expires_at = _timestamp(_required_text(parsed, "expires_at"), "expires_at")
|
|
297
|
+
if expires_at <= now():
|
|
298
|
+
raise HostedDeployError(
|
|
299
|
+
"DEPLOY_TOKEN_EXCHANGE_FAILED", "the Cloud returned an expired Studio token"
|
|
300
|
+
)
|
|
301
|
+
return StudioToken(
|
|
302
|
+
studio_base_url=_service_url(base_url, "Studio base URL", allow_loopback_http=True),
|
|
303
|
+
workspace_id=workspace_id,
|
|
304
|
+
token=token,
|
|
305
|
+
expires_at=expires_at,
|
|
306
|
+
)
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
_OPERATIONS: dict[str, tuple[str, str]] = {
|
|
310
|
+
"create_dataset": ("ContractCreateDatasetRequest", "create_dataset"),
|
|
311
|
+
"create_question": ("ContractCreateQuestionRequest", "create_question"),
|
|
312
|
+
"put_requirements": ("ContractUpsertCommand", "put_question_requirements"),
|
|
313
|
+
"register_source": ("ContractRegisterCommand", "register_source"),
|
|
314
|
+
"create_plan": ("ContractCreateCommand", "create_table_plan"),
|
|
315
|
+
"create_source_staging": ("CreateSourceStagingCommand", "create_source_staging"),
|
|
316
|
+
"complete_artifact_upload": (
|
|
317
|
+
"ContractCompleteArtifactUploadCommand",
|
|
318
|
+
"complete_artifact_upload",
|
|
319
|
+
),
|
|
320
|
+
"finalize_source_staging": ("FinalizeSourceStagingCommand", "finalize_source_staging"),
|
|
321
|
+
"register_connector": (
|
|
322
|
+
"ConnectorConfigurationRegistrationCommand",
|
|
323
|
+
"register_connector_configuration",
|
|
324
|
+
),
|
|
325
|
+
"create_proposal": ("ContractProposalCommand", "create_recipe_proposal"),
|
|
326
|
+
"request_approval": ("ContractApprovalCommand", "request_recipe_approval"),
|
|
327
|
+
# This is intentionally the dedicated V3 Table-recipe operation. The coordinated Studio client
|
|
328
|
+
# exposes only confirmation of the exact Table Recipe approval the Editor just requested, not
|
|
329
|
+
# generic approval decisions. Never map this to the generic V3 approval route or the retired V2
|
|
330
|
+
# recipe confirmation route.
|
|
331
|
+
"confirm_table_recipe_approval": (
|
|
332
|
+
"ContractDecisionCommand",
|
|
333
|
+
"confirm_table_recipe_approval",
|
|
334
|
+
),
|
|
335
|
+
"activate_recipe": ("ContractActivateRecipeCommand", "activate_recipe"),
|
|
336
|
+
}
|
|
337
|
+
|
|
338
|
+
#: The read operations, as the generated method each one resolves to and the keyword it names its
|
|
339
|
+
#: resource with.
|
|
340
|
+
#:
|
|
341
|
+
#: ⚠ A READ IS A DIFFERENT SHAPE, NOT A RARER WRITE. Every entry in `_OPERATIONS` above is built
|
|
342
|
+
#: the same way: a request model from a body, an idempotency key, and a journaled plan written
|
|
343
|
+
#: before the request is sent, because each of those operations changes something in Studio and
|
|
344
|
+
#: has to be replayable. A read carries no body and no idempotency key, so there is nothing for
|
|
345
|
+
#: that path to build and nothing for the journal to replay -- which is why `get_run` is here
|
|
346
|
+
#: rather than as a thirteenth entry above, and why nothing on this path can mutate a Run by
|
|
347
|
+
#: accident: the method names reachable through `read` are exactly these.
|
|
348
|
+
_READ_OPERATIONS: dict[str, tuple[str, str]] = {
|
|
349
|
+
"get_proposal": ("get_recipe_proposal", "recipe_proposal_id"),
|
|
350
|
+
"get_run": ("get_run", "run_id"),
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
# The generated V3 distribution exports operations as modules. Keep the deployment surface
|
|
354
|
+
# closed by resolving only this audited inventory; no caller-provided operation name reaches an
|
|
355
|
+
# import path.
|
|
356
|
+
_GENERATED_OPERATION_MODULES = {
|
|
357
|
+
"create_dataset": ("datasets", "create_dataset"),
|
|
358
|
+
"create_question": ("questions", "create_question"),
|
|
359
|
+
"put_question_requirements": ("questions", "put_question_requirements"),
|
|
360
|
+
"register_source": ("sources", "register_source"),
|
|
361
|
+
"create_table_plan": ("plans", "create_table_plan"),
|
|
362
|
+
"create_source_staging": ("artifacts", "create_source_staging"),
|
|
363
|
+
"complete_artifact_upload": ("artifacts", "complete_artifact_upload"),
|
|
364
|
+
"finalize_source_staging": ("artifacts", "finalize_source_staging"),
|
|
365
|
+
"register_connector_configuration": ("connectors", "register_connector_configuration"),
|
|
366
|
+
"create_recipe_proposal": ("recipes", "create_recipe_proposal"),
|
|
367
|
+
"create_worker_policy_preflight": ("recipes", "create_worker_policy_preflight"),
|
|
368
|
+
"request_recipe_approval": ("recipes", "request_recipe_approval"),
|
|
369
|
+
"activate_recipe": ("recipes", "activate_recipe"),
|
|
370
|
+
"get_recipe_proposal": ("recipes", "get_recipe_proposal"),
|
|
371
|
+
"get_run": ("runs", "get_run"),
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
def build_editor_client(token: StudioToken) -> tuple[Any, Any]:
|
|
376
|
+
"""Return the pinned V3 Table-recipe confirmation facade and its transport.
|
|
377
|
+
|
|
378
|
+
Written once because two lanes need the same thing: the deployment writes through this facade
|
|
379
|
+
and the dataset handoff reads through it, and a second copy of the construction would be a
|
|
380
|
+
second place for the contract pin, the redirect policy, or the timeout to be got wrong.
|
|
381
|
+
|
|
382
|
+
Raises:
|
|
383
|
+
HostedDeployError: when the pinned client is absent or incompatible.
|
|
384
|
+
"""
|
|
385
|
+
|
|
386
|
+
try:
|
|
387
|
+
client_contract_identity()
|
|
388
|
+
client_module = importlib.import_module("mostlyright_studio.client")
|
|
389
|
+
authority_module = importlib.import_module("mostlyright_studio.authority")
|
|
390
|
+
authenticated = client_module.AuthenticatedClient(
|
|
391
|
+
base_url=token.studio_base_url,
|
|
392
|
+
token=token.token,
|
|
393
|
+
timeout=REQUEST_TIMEOUT_SECONDS,
|
|
394
|
+
follow_redirects=False,
|
|
395
|
+
)
|
|
396
|
+
return authenticated, authority_module.StudioV3RecipeConfirmationClient(authenticated)
|
|
397
|
+
except (ImportError, AttributeError, StudioBoundaryError) as error:
|
|
398
|
+
raise HostedDeployError("DEPLOY_STUDIO_CLIENT_INVALID", str(error)) from error
|
|
399
|
+
|
|
400
|
+
|
|
401
|
+
class GeneratedStudioDeploymentClient:
|
|
402
|
+
"""All remote deployment operations through the pinned generated editor facade."""
|
|
403
|
+
|
|
404
|
+
def __init__(self, token: StudioToken) -> None:
|
|
405
|
+
self._raw_client, self._editor_client = build_editor_client(token)
|
|
406
|
+
try:
|
|
407
|
+
self._models = importlib.import_module("mostlyright_studio.models")
|
|
408
|
+
except ImportError as error:
|
|
409
|
+
raise HostedDeployError("DEPLOY_STUDIO_CLIENT_INVALID", str(error)) from error
|
|
410
|
+
confirmation = getattr(self._editor_client, "confirm_table_recipe_approval", None)
|
|
411
|
+
if confirmation is None or not callable(getattr(confirmation, "sync_detailed", None)):
|
|
412
|
+
raise HostedDeployError(
|
|
413
|
+
"DEPLOY_STUDIO_CLIENT_INVALID",
|
|
414
|
+
"the pinned Studio client does not expose V3 Table-recipe confirmation; "
|
|
415
|
+
"install the coordinated regenerated Studio client before deploying",
|
|
416
|
+
)
|
|
417
|
+
try:
|
|
418
|
+
preflight = self._operation("create_worker_policy_preflight")
|
|
419
|
+
except ImportError as error:
|
|
420
|
+
raise HostedDeployError(
|
|
421
|
+
"DEPLOY_STUDIO_CLIENT_INVALID",
|
|
422
|
+
"the installed generated Studio client does not expose "
|
|
423
|
+
"createWorkerPolicyPreflight; merge Studio #80 and install its regenerated client "
|
|
424
|
+
"before deploying",
|
|
425
|
+
) from error
|
|
426
|
+
if not callable(getattr(preflight, "sync_detailed", None)):
|
|
427
|
+
raise HostedDeployError(
|
|
428
|
+
"DEPLOY_STUDIO_CLIENT_INVALID",
|
|
429
|
+
"the installed generated Studio client has no callable "
|
|
430
|
+
"createWorkerPolicyPreflight operation; merge Studio #80 and install its "
|
|
431
|
+
"regenerated client before deploying",
|
|
432
|
+
)
|
|
433
|
+
self._base_url = token.studio_base_url
|
|
434
|
+
|
|
435
|
+
def call(
|
|
436
|
+
self,
|
|
437
|
+
operation: str,
|
|
438
|
+
body: Mapping[str, Any] | None,
|
|
439
|
+
*,
|
|
440
|
+
idempotency_key: str | None = None,
|
|
441
|
+
resource_id: UUID | None = None,
|
|
442
|
+
if_match: str | None = None,
|
|
443
|
+
) -> StudioResponse:
|
|
444
|
+
if operation in _READ_OPERATIONS:
|
|
445
|
+
return self.read(operation, resource_id=resource_id)
|
|
446
|
+
coordinate = _OPERATIONS.get(operation)
|
|
447
|
+
if coordinate is None or body is None or idempotency_key is None:
|
|
448
|
+
raise HostedDeployError("DEPLOY_REQUEST_INVALID", "deployment operation is incomplete")
|
|
449
|
+
model_name, method_name = coordinate
|
|
450
|
+
model = self._model(model_name, body)
|
|
451
|
+
kwargs: dict[str, Any] = {"body": model, "idempotency_key": idempotency_key}
|
|
452
|
+
if operation == "put_requirements":
|
|
453
|
+
kwargs.update(question_id=_required_uuid(resource_id, "question_id"), if_match=if_match)
|
|
454
|
+
elif operation in {"finalize_source_staging"}:
|
|
455
|
+
kwargs["source_staging_id"] = _required_uuid(resource_id, "source_staging_id")
|
|
456
|
+
elif operation == "complete_artifact_upload":
|
|
457
|
+
kwargs["artifact_id"] = _required_uuid(resource_id, "artifact_id")
|
|
458
|
+
elif operation in {"request_approval", "activate_recipe"}:
|
|
459
|
+
kwargs.update(
|
|
460
|
+
recipe_proposal_id=_required_uuid(resource_id, "recipe_proposal_id"),
|
|
461
|
+
if_match=if_match,
|
|
462
|
+
)
|
|
463
|
+
elif operation == "confirm_table_recipe_approval":
|
|
464
|
+
kwargs.update(
|
|
465
|
+
approval_request_id=_required_uuid(resource_id, "approval_request_id"),
|
|
466
|
+
if_match=if_match,
|
|
467
|
+
)
|
|
468
|
+
if operation == "confirm_table_recipe_approval":
|
|
469
|
+
confirmation = getattr(self._editor_client, method_name, None)
|
|
470
|
+
if confirmation is None or not callable(getattr(confirmation, "sync_detailed", None)):
|
|
471
|
+
raise HostedDeployError(
|
|
472
|
+
"DEPLOY_STUDIO_CLIENT_INVALID",
|
|
473
|
+
"the installed generated Studio client does not expose "
|
|
474
|
+
"StudioV3RecipeConfirmationClient.confirm_table_recipe_approval; pin the "
|
|
475
|
+
"coordinated "
|
|
476
|
+
"regenerated Studio client before deployment",
|
|
477
|
+
)
|
|
478
|
+
response = confirmation.sync_detailed(**kwargs)
|
|
479
|
+
else:
|
|
480
|
+
response = self._operation(method_name).sync_detailed(client=self._raw_client, **kwargs)
|
|
481
|
+
return _generated_response(response)
|
|
482
|
+
|
|
483
|
+
def preflight(self, body: Mapping[str, Any]) -> StudioResponse:
|
|
484
|
+
"""Resolve Studio's current worker policy without journaling it as a mutation."""
|
|
485
|
+
|
|
486
|
+
command = self._model("ContractWorkerPolicyPreflightCommand", body)
|
|
487
|
+
response = self._operation("create_worker_policy_preflight").sync_detailed(
|
|
488
|
+
client=self._raw_client,
|
|
489
|
+
body=command,
|
|
490
|
+
)
|
|
491
|
+
return _generated_response(response)
|
|
492
|
+
|
|
493
|
+
def read(self, operation: str, *, resource_id: UUID | None = None) -> StudioResponse:
|
|
494
|
+
"""One bounded read of one Studio resource, sending no body and changing nothing.
|
|
495
|
+
|
|
496
|
+
Transport failures are deliberately not translated here. What a caller should say about an
|
|
497
|
+
unreachable Studio depends on what it was reading -- a status check that could not be made
|
|
498
|
+
is a retryable answer, an approved proposal that could not be fetched stops a deployment --
|
|
499
|
+
so the refusal is written where that is known rather than flattened into one code here.
|
|
500
|
+
"""
|
|
501
|
+
|
|
502
|
+
coordinate = _READ_OPERATIONS.get(operation)
|
|
503
|
+
if coordinate is None:
|
|
504
|
+
raise HostedDeployError("DEPLOY_REQUEST_INVALID", "deployment operation is incomplete")
|
|
505
|
+
method_name, parameter = coordinate
|
|
506
|
+
response = self._operation(method_name).sync_detailed(
|
|
507
|
+
client=self._raw_client,
|
|
508
|
+
**{parameter: _required_uuid(resource_id, parameter)},
|
|
509
|
+
)
|
|
510
|
+
return _generated_response(response)
|
|
511
|
+
|
|
512
|
+
@staticmethod
|
|
513
|
+
def _operation(method_name: str) -> Any:
|
|
514
|
+
coordinate = _GENERATED_OPERATION_MODULES.get(method_name)
|
|
515
|
+
if coordinate is None:
|
|
516
|
+
raise HostedDeployError("DEPLOY_REQUEST_INVALID", "deployment operation is incomplete")
|
|
517
|
+
tag, module = coordinate
|
|
518
|
+
return importlib.import_module(f"mostlyright_studio.api.{tag}.{module}")
|
|
519
|
+
|
|
520
|
+
def upload(self, session: Mapping[str, Any], content: bytes) -> str:
|
|
521
|
+
expected_digest = "sha256:" + hashlib.sha256(content).hexdigest()
|
|
522
|
+
if (
|
|
523
|
+
session.get("direction") != "upload"
|
|
524
|
+
or session.get("method") != "PUT"
|
|
525
|
+
or session.get("route_authority") != "source_staging_editor"
|
|
526
|
+
or session.get("expected_content_digest") != expected_digest
|
|
527
|
+
or session.get("expected_size_bytes") != len(content)
|
|
528
|
+
):
|
|
529
|
+
raise HostedDeployError(
|
|
530
|
+
"DEPLOY_SIGNED_SESSION_INVALID", "Studio returned a mismatched upload capability"
|
|
531
|
+
)
|
|
532
|
+
try:
|
|
533
|
+
import httpx
|
|
534
|
+
except ImportError as error:
|
|
535
|
+
raise HostedDeployError(
|
|
536
|
+
"DEPLOY_STUDIO_CLIENT_INVALID", "the hosted extra does not contain httpx"
|
|
537
|
+
) from error
|
|
538
|
+
headers = _signed_headers(session, size=len(content))
|
|
539
|
+
try:
|
|
540
|
+
with httpx.Client(
|
|
541
|
+
timeout=REQUEST_TIMEOUT_SECONDS, follow_redirects=False, trust_env=False
|
|
542
|
+
) as client:
|
|
543
|
+
response = client.put(
|
|
544
|
+
_transfer_url(_required_text(session, "signed_url"), self._base_url),
|
|
545
|
+
headers=headers,
|
|
546
|
+
content=content,
|
|
547
|
+
)
|
|
548
|
+
except httpx.RequestError as error:
|
|
549
|
+
raise HostedDeployError(
|
|
550
|
+
"DEPLOY_SOURCE_UPLOAD_FAILED", "the signed source upload could not be reached"
|
|
551
|
+
) from error
|
|
552
|
+
if response.status_code not in {200, 201}:
|
|
553
|
+
if 400 <= response.status_code < 500:
|
|
554
|
+
raise HostedDeployError(
|
|
555
|
+
"DEPLOY_SOURCE_UPLOAD_SESSION_REJECTED",
|
|
556
|
+
"the signed source upload capability was rejected and must be renewed",
|
|
557
|
+
)
|
|
558
|
+
raise HostedDeployError(
|
|
559
|
+
"DEPLOY_SOURCE_UPLOAD_FAILED",
|
|
560
|
+
f"the signed source upload returned HTTP {response.status_code}",
|
|
561
|
+
)
|
|
562
|
+
generation = response.headers.get(
|
|
563
|
+
"x-mostlyright-object-generation", response.headers.get("x-goog-generation", "")
|
|
564
|
+
)
|
|
565
|
+
if _OBJECT_GENERATION.fullmatch(generation) is None:
|
|
566
|
+
raise HostedDeployError(
|
|
567
|
+
"DEPLOY_SOURCE_UPLOAD_FAILED", "the upload response omitted object generation"
|
|
568
|
+
)
|
|
569
|
+
return generation
|
|
570
|
+
|
|
571
|
+
def close(self) -> None:
|
|
572
|
+
client = getattr(self._raw_client, "_client", None)
|
|
573
|
+
if client is not None:
|
|
574
|
+
client.close()
|
|
575
|
+
|
|
576
|
+
def _model(self, name: str, body: Mapping[str, Any]) -> Any:
|
|
577
|
+
model_type = getattr(self._models, name, None)
|
|
578
|
+
if model_type is None:
|
|
579
|
+
raise HostedDeployError(
|
|
580
|
+
"DEPLOY_STUDIO_CLIENT_INVALID", f"the generated Studio client lacks {name}"
|
|
581
|
+
)
|
|
582
|
+
try:
|
|
583
|
+
model = model_type.from_dict(dict(body))
|
|
584
|
+
except (KeyError, TypeError, ValueError) as error:
|
|
585
|
+
raise HostedDeployError(
|
|
586
|
+
"DEPLOY_REQUEST_INVALID", f"the deployment handoff is not a valid {name}"
|
|
587
|
+
) from error
|
|
588
|
+
if model.to_dict() != dict(body):
|
|
589
|
+
raise HostedDeployError(
|
|
590
|
+
"DEPLOY_REQUEST_INVALID", f"the deployment handoff is not the exact {name} shape"
|
|
591
|
+
)
|
|
592
|
+
return model
|
|
593
|
+
|
|
594
|
+
|
|
595
|
+
class FileDeploymentStateStore:
|
|
596
|
+
"""Canonical, atomically replaced non-secret journal under the reviewed run directory."""
|
|
597
|
+
|
|
598
|
+
def __init__(self, run_dir: Path) -> None:
|
|
599
|
+
self._run_dir = Path(run_dir).absolute()
|
|
600
|
+
self._path = self._run_dir / STATE_RELATIVE_PATH
|
|
601
|
+
self._run_handle = None
|
|
602
|
+
self._directory_descriptors: list[int] = []
|
|
603
|
+
self._directory_bindings: list[tuple[int, str, int, int, int]] = []
|
|
604
|
+
try:
|
|
605
|
+
self._run_handle = pipeline._open_candidate_run_handle(self._run_dir)
|
|
606
|
+
evidence = self._open_directory(self._run_handle.run_fd, "evidence")
|
|
607
|
+
deployment = self._open_directory(evidence, "deployment")
|
|
608
|
+
self._deployment_fd = deployment
|
|
609
|
+
fcntl.flock(self._deployment_fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
610
|
+
self._validate()
|
|
611
|
+
except (OSError, pipeline.BuildError) as error:
|
|
612
|
+
self.close()
|
|
613
|
+
raise HostedDeployError(
|
|
614
|
+
"DEPLOY_STATE_UNAVAILABLE",
|
|
615
|
+
"the deployment journal directory is unavailable, unsafe, or already in use",
|
|
616
|
+
) from error
|
|
617
|
+
|
|
618
|
+
@property
|
|
619
|
+
def path(self) -> Path:
|
|
620
|
+
return self._path
|
|
621
|
+
|
|
622
|
+
def exists(self) -> bool:
|
|
623
|
+
self._validate()
|
|
624
|
+
try:
|
|
625
|
+
os.stat("state.json", dir_fd=self._deployment_fd, follow_symlinks=False)
|
|
626
|
+
except FileNotFoundError:
|
|
627
|
+
return False
|
|
628
|
+
except OSError as error:
|
|
629
|
+
raise HostedDeployError(
|
|
630
|
+
"DEPLOY_STATE_INVALID", "the deployment journal is unavailable or unsafe"
|
|
631
|
+
) from error
|
|
632
|
+
return True
|
|
633
|
+
|
|
634
|
+
def load(self) -> dict[str, Any]:
|
|
635
|
+
try:
|
|
636
|
+
descriptor = os.open(
|
|
637
|
+
"state.json",
|
|
638
|
+
os.O_RDONLY
|
|
639
|
+
| getattr(os, "O_NOFOLLOW", 0)
|
|
640
|
+
| getattr(os, "O_NONBLOCK", 0)
|
|
641
|
+
| getattr(os, "O_CLOEXEC", 0),
|
|
642
|
+
dir_fd=self._deployment_fd,
|
|
643
|
+
)
|
|
644
|
+
except OSError as error:
|
|
645
|
+
raise HostedDeployError(
|
|
646
|
+
"DEPLOY_STATE_INVALID", "the deployment journal is unavailable or unsafe"
|
|
647
|
+
) from error
|
|
648
|
+
try:
|
|
649
|
+
info = os.fstat(descriptor)
|
|
650
|
+
if (
|
|
651
|
+
not stat.S_ISREG(info.st_mode)
|
|
652
|
+
or not 0 < info.st_size <= MAX_STATE_BYTES
|
|
653
|
+
or info.st_nlink != 1
|
|
654
|
+
):
|
|
655
|
+
raise HostedDeployError(
|
|
656
|
+
"DEPLOY_STATE_INVALID", "the deployment journal is not a bounded regular file"
|
|
657
|
+
)
|
|
658
|
+
chunks: list[bytes] = []
|
|
659
|
+
remaining = MAX_STATE_BYTES + 1
|
|
660
|
+
while remaining:
|
|
661
|
+
chunk = os.read(descriptor, min(64 * 1024, remaining))
|
|
662
|
+
if not chunk:
|
|
663
|
+
break
|
|
664
|
+
chunks.append(chunk)
|
|
665
|
+
remaining -= len(chunk)
|
|
666
|
+
raw = b"".join(chunks)
|
|
667
|
+
after = os.fstat(descriptor)
|
|
668
|
+
named = os.stat("state.json", dir_fd=self._deployment_fd, follow_symlinks=False)
|
|
669
|
+
if _file_identity(after) != _file_identity(info) or _file_identity(
|
|
670
|
+
named
|
|
671
|
+
) != _file_identity(info):
|
|
672
|
+
raise HostedDeployError(
|
|
673
|
+
"DEPLOY_STATE_INVALID", "the deployment journal changed while being read"
|
|
674
|
+
)
|
|
675
|
+
finally:
|
|
676
|
+
os.close(descriptor)
|
|
677
|
+
self._validate()
|
|
678
|
+
try:
|
|
679
|
+
value = canonical.parse_canonical_json(raw)
|
|
680
|
+
except canonical.CanonicalJSONError as error:
|
|
681
|
+
raise HostedDeployError(
|
|
682
|
+
"DEPLOY_STATE_INVALID", "the deployment journal is not canonical JSON"
|
|
683
|
+
) from error
|
|
684
|
+
if not isinstance(value, dict):
|
|
685
|
+
raise HostedDeployError(
|
|
686
|
+
"DEPLOY_STATE_INVALID", "the deployment journal must be one object"
|
|
687
|
+
)
|
|
688
|
+
return value
|
|
689
|
+
|
|
690
|
+
def save(self, state: Mapping[str, Any]) -> None:
|
|
691
|
+
raw = canonical.canonical_json_bytes(dict(state))
|
|
692
|
+
if len(raw) > MAX_STATE_BYTES:
|
|
693
|
+
raise HostedDeployError("DEPLOY_STATE_INVALID", "the deployment journal is too large")
|
|
694
|
+
self._validate()
|
|
695
|
+
temporary = f".state.{os.getpid()}.{os.urandom(8).hex()}.partial"
|
|
696
|
+
descriptor = -1
|
|
697
|
+
try:
|
|
698
|
+
descriptor = os.open(
|
|
699
|
+
temporary,
|
|
700
|
+
os.O_WRONLY
|
|
701
|
+
| os.O_CREAT
|
|
702
|
+
| os.O_EXCL
|
|
703
|
+
| getattr(os, "O_NOFOLLOW", 0)
|
|
704
|
+
| getattr(os, "O_CLOEXEC", 0),
|
|
705
|
+
0o600,
|
|
706
|
+
dir_fd=self._deployment_fd,
|
|
707
|
+
)
|
|
708
|
+
offset = 0
|
|
709
|
+
while offset < len(raw):
|
|
710
|
+
offset += os.write(descriptor, raw[offset:])
|
|
711
|
+
os.fsync(descriptor)
|
|
712
|
+
os.close(descriptor)
|
|
713
|
+
descriptor = -1
|
|
714
|
+
self._validate()
|
|
715
|
+
os.replace(
|
|
716
|
+
temporary,
|
|
717
|
+
"state.json",
|
|
718
|
+
src_dir_fd=self._deployment_fd,
|
|
719
|
+
dst_dir_fd=self._deployment_fd,
|
|
720
|
+
)
|
|
721
|
+
os.fsync(self._deployment_fd)
|
|
722
|
+
self._validate()
|
|
723
|
+
except Exception:
|
|
724
|
+
if descriptor >= 0:
|
|
725
|
+
os.close(descriptor)
|
|
726
|
+
try:
|
|
727
|
+
os.unlink(temporary, dir_fd=self._deployment_fd)
|
|
728
|
+
except FileNotFoundError:
|
|
729
|
+
pass
|
|
730
|
+
raise
|
|
731
|
+
|
|
732
|
+
def close(self) -> None:
|
|
733
|
+
descriptors = list(reversed(getattr(self, "_directory_descriptors", [])))
|
|
734
|
+
self._directory_descriptors = []
|
|
735
|
+
for descriptor in descriptors:
|
|
736
|
+
try:
|
|
737
|
+
os.close(descriptor)
|
|
738
|
+
except OSError:
|
|
739
|
+
pass
|
|
740
|
+
handle = getattr(self, "_run_handle", None)
|
|
741
|
+
self._run_handle = None
|
|
742
|
+
if handle is not None:
|
|
743
|
+
handle.close()
|
|
744
|
+
|
|
745
|
+
def _open_directory(self, parent_fd: int, name: str) -> int:
|
|
746
|
+
try:
|
|
747
|
+
os.mkdir(name, 0o700, dir_fd=parent_fd)
|
|
748
|
+
except FileExistsError:
|
|
749
|
+
pass
|
|
750
|
+
descriptor = os.open(
|
|
751
|
+
name,
|
|
752
|
+
os.O_RDONLY
|
|
753
|
+
| getattr(os, "O_DIRECTORY", 0)
|
|
754
|
+
| getattr(os, "O_NOFOLLOW", 0)
|
|
755
|
+
| getattr(os, "O_CLOEXEC", 0),
|
|
756
|
+
dir_fd=parent_fd,
|
|
757
|
+
)
|
|
758
|
+
opened = os.fstat(descriptor)
|
|
759
|
+
if not stat.S_ISDIR(opened.st_mode):
|
|
760
|
+
os.close(descriptor)
|
|
761
|
+
raise OSError("deployment state component is not a directory")
|
|
762
|
+
self._directory_descriptors.append(descriptor)
|
|
763
|
+
self._directory_bindings.append((parent_fd, name, descriptor, opened.st_dev, opened.st_ino))
|
|
764
|
+
return descriptor
|
|
765
|
+
|
|
766
|
+
def _validate(self) -> None:
|
|
767
|
+
if self._run_handle is None:
|
|
768
|
+
raise HostedDeployError("DEPLOY_STATE_INVALID", "the deployment journal is closed")
|
|
769
|
+
try:
|
|
770
|
+
pipeline._validate_candidate_run_handle(self._run_handle)
|
|
771
|
+
for parent_fd, name, descriptor, device, inode in self._directory_bindings:
|
|
772
|
+
opened = os.fstat(descriptor)
|
|
773
|
+
named = os.stat(name, dir_fd=parent_fd, follow_symlinks=False)
|
|
774
|
+
expected = (device, inode)
|
|
775
|
+
if (
|
|
776
|
+
not stat.S_ISDIR(opened.st_mode)
|
|
777
|
+
or not stat.S_ISDIR(named.st_mode)
|
|
778
|
+
or (opened.st_dev, opened.st_ino) != expected
|
|
779
|
+
or (named.st_dev, named.st_ino) != expected
|
|
780
|
+
):
|
|
781
|
+
raise OSError("deployment state directory changed")
|
|
782
|
+
except (OSError, pipeline.BuildError) as error:
|
|
783
|
+
raise HostedDeployError(
|
|
784
|
+
"DEPLOY_STATE_INVALID", "the deployment journal path changed during deployment"
|
|
785
|
+
) from error
|
|
786
|
+
|
|
787
|
+
|
|
788
|
+
def journaled_reviewed_build(run_dir: Path) -> Mapping[str, Any] | None:
|
|
789
|
+
"""Return a legacy reviewed-Build binding recorded in a deployment journal.
|
|
790
|
+
|
|
791
|
+
This compatibility reader does not participate in the ordinary Editor deploy path. That path
|
|
792
|
+
always rebuilds and re-verifies the local Build, then confirms its internal Studio approval in
|
|
793
|
+
the same invocation; ``--resume`` only replays an interrupted mutation chain.
|
|
794
|
+
|
|
795
|
+
Raises:
|
|
796
|
+
HostedDeployError: when there is no journal to resume, or it cannot be read safely.
|
|
797
|
+
"""
|
|
798
|
+
|
|
799
|
+
store = FileDeploymentStateStore(run_dir)
|
|
800
|
+
try:
|
|
801
|
+
if not store.exists():
|
|
802
|
+
raise HostedDeployError(
|
|
803
|
+
"DEPLOY_STATE_MISSING", f"no deployment journal exists at {store.path}"
|
|
804
|
+
)
|
|
805
|
+
state = store.load()
|
|
806
|
+
finally:
|
|
807
|
+
store.close()
|
|
808
|
+
resources = state.get("resources")
|
|
809
|
+
proposal = resources.get("proposal") if isinstance(resources, dict) else None
|
|
810
|
+
evidence = proposal.get("bootstrap_evidence") if isinstance(proposal, dict) else None
|
|
811
|
+
reviewed = evidence.get("reviewed_build") if isinstance(evidence, dict) else None
|
|
812
|
+
return reviewed if isinstance(reviewed, Mapping) else None
|
|
813
|
+
|
|
814
|
+
|
|
815
|
+
def _legacy_local_build_bootstrap_table_digest(bootstrap: Mapping[str, Any]) -> str:
|
|
816
|
+
"""Return the local-Build activation digest or refuse a Studio-owned first build.
|
|
817
|
+
|
|
818
|
+
``deploy_managed_dataset`` predates hosted-first Recipes. It owns the local Build it has just
|
|
819
|
+
staged, confirmed, and then activates by passing that Build's digest to Studio. A
|
|
820
|
+
``hosted_first_build`` has no such local Build: Studio/Cloud owns its atomic approval and
|
|
821
|
+
first-run transition. Treating that object as local evidence would either leak a raw
|
|
822
|
+
``KeyError`` or activate the same Recipe a second time.
|
|
823
|
+
|
|
824
|
+
Omitted ``origin`` remains the historical local shape for backwards compatibility.
|
|
825
|
+
"""
|
|
826
|
+
|
|
827
|
+
origin = bootstrap.get("origin", "local_build")
|
|
828
|
+
if origin == "hosted_first_build":
|
|
829
|
+
raise HostedDeployError(
|
|
830
|
+
"DEPLOY_HOSTED_FIRST_STUDIO_OWNED",
|
|
831
|
+
"this Recipe's hosted first build is owned by Studio/Cloud; approve and start it "
|
|
832
|
+
"there instead of using the legacy local-Build deploy command",
|
|
833
|
+
)
|
|
834
|
+
if origin != "local_build":
|
|
835
|
+
raise HostedDeployError(
|
|
836
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID",
|
|
837
|
+
"Studio returned an unknown Recipe bootstrap origin",
|
|
838
|
+
)
|
|
839
|
+
digest = bootstrap.get("bootstrap_table_digest")
|
|
840
|
+
if not isinstance(digest, str) or _DIGEST.fullmatch(digest) is None:
|
|
841
|
+
raise HostedDeployError(
|
|
842
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID",
|
|
843
|
+
"Studio's local-Build Recipe bootstrap omitted a valid bootstrap_table_digest",
|
|
844
|
+
)
|
|
845
|
+
return digest
|
|
846
|
+
|
|
847
|
+
|
|
848
|
+
def deploy_managed_dataset(
|
|
849
|
+
request: DeploymentRequest,
|
|
850
|
+
*,
|
|
851
|
+
run_dir: Path,
|
|
852
|
+
previous_run_dir: Path | None,
|
|
853
|
+
token: StudioToken,
|
|
854
|
+
credentials: ResolvedCloudCredentials,
|
|
855
|
+
resume: bool,
|
|
856
|
+
cron: str = "0 0 * * *",
|
|
857
|
+
max_concurrent_runs: int = 1,
|
|
858
|
+
state_store: DeploymentStateStore | None = None,
|
|
859
|
+
client_factory: Callable[
|
|
860
|
+
[StudioToken], StudioDeploymentClient
|
|
861
|
+
] = GeneratedStudioDeploymentClient,
|
|
862
|
+
cloud_transport: TokenExchangeTransport | None = None,
|
|
863
|
+
) -> dict[str, Any]:
|
|
864
|
+
"""Prepare and activate one exact Dataset, or resume it after interruption.
|
|
865
|
+
|
|
866
|
+
``previous_run_dir`` is the run directory whose own deployment published the Dataset this one
|
|
867
|
+
updates, and naming it is what makes this a successor rather than a first deployment. Its
|
|
868
|
+
*absence* is refused here, by :func:`_deployment_predecessor`, and only for the one shape this
|
|
869
|
+
layer can judge alone: a Build whose Recipe is a successor version, which exists because a
|
|
870
|
+
Dataset it follows exists. The opposite rule -- that a Build executed as an ``initial`` run
|
|
871
|
+
must not name a predecessor, because it never read one -- is enforced where the request is
|
|
872
|
+
built, in :func:`deploy.build_deployment_request`, which is the only layer that can see the
|
|
873
|
+
execution mode. So this function trusts its caller for that half; every command in this
|
|
874
|
+
repository builds its request through that function first.
|
|
875
|
+
"""
|
|
876
|
+
|
|
877
|
+
_deployment_options(request.dataset_name, cron, max_concurrent_runs)
|
|
878
|
+
if token.workspace_id is None:
|
|
879
|
+
raise HostedDeployError("DEPLOY_TOKEN_EXCHANGE_FAILED", "Studio workspace is missing")
|
|
880
|
+
predecessor = _deployment_predecessor(request, previous_run_dir)
|
|
881
|
+
owns_store = state_store is None
|
|
882
|
+
store = state_store or FileDeploymentStateStore(run_dir)
|
|
883
|
+
cloud_url = _service_url(credentials.cloud_url, "Cloud URL", allow_loopback_http=False)
|
|
884
|
+
identity = canonical.canonical_sha256(
|
|
885
|
+
{
|
|
886
|
+
"request_digest": request.digest(),
|
|
887
|
+
"workspace_id": str(token.workspace_id),
|
|
888
|
+
"cloud_url": cloud_url,
|
|
889
|
+
"cron": cron,
|
|
890
|
+
"max_concurrent_runs": max_concurrent_runs,
|
|
891
|
+
# Which live Dataset this updates is an input like the schedule is, and it is the one
|
|
892
|
+
# a resume can get wrong without changing the request at all. The key is present only
|
|
893
|
+
# when there is a predecessor, so every deployment that could already be in flight --
|
|
894
|
+
# all of them, until this release gave the command a flag to name one -- keeps the
|
|
895
|
+
# exact identity it was staged under.
|
|
896
|
+
**({} if predecessor is None else {"previous_run_dir": predecessor}),
|
|
897
|
+
}
|
|
898
|
+
)
|
|
899
|
+
inputs = _identity_inputs(
|
|
900
|
+
request,
|
|
901
|
+
workspace_id=token.workspace_id,
|
|
902
|
+
cloud_url=cloud_url,
|
|
903
|
+
cron=cron,
|
|
904
|
+
max_concurrent_runs=max_concurrent_runs,
|
|
905
|
+
previous_run_dir=predecessor,
|
|
906
|
+
)
|
|
907
|
+
if store.exists():
|
|
908
|
+
state = store.load()
|
|
909
|
+
_validate_state(
|
|
910
|
+
state,
|
|
911
|
+
identity,
|
|
912
|
+
token.workspace_id,
|
|
913
|
+
inputs=inputs,
|
|
914
|
+
request_digest=request.digest(),
|
|
915
|
+
)
|
|
916
|
+
_refuse_legacy_approval_journal(state)
|
|
917
|
+
if not resume and state.get("status") != "initialized":
|
|
918
|
+
raise HostedDeployError(
|
|
919
|
+
"DEPLOY_STATE_EXISTS",
|
|
920
|
+
f"deployment journal already exists at {store.path}; rerun with --resume",
|
|
921
|
+
)
|
|
922
|
+
else:
|
|
923
|
+
if resume:
|
|
924
|
+
raise HostedDeployError(
|
|
925
|
+
"DEPLOY_STATE_MISSING", f"no deployment journal exists at {store.path}"
|
|
926
|
+
)
|
|
927
|
+
state = {
|
|
928
|
+
"schema_version": DEPLOYMENT_STATE_SCHEMA,
|
|
929
|
+
"deployment_identity": identity,
|
|
930
|
+
"request_digest": request.digest(),
|
|
931
|
+
# The inputs behind that identity, recorded so a refusal hours or days later can name
|
|
932
|
+
# the one that moved. The identity itself is unchanged and stays the only thing any
|
|
933
|
+
# check compares; this is what turns its refusal into a sentence.
|
|
934
|
+
"identity_inputs": inputs,
|
|
935
|
+
"workspace_id": str(token.workspace_id),
|
|
936
|
+
"status": "initialized",
|
|
937
|
+
"schedule": {
|
|
938
|
+
"cron": cron,
|
|
939
|
+
"timezone": "UTC",
|
|
940
|
+
"mode": "incremental_refresh",
|
|
941
|
+
"max_concurrent_runs": max_concurrent_runs,
|
|
942
|
+
},
|
|
943
|
+
"operations": {},
|
|
944
|
+
"resources": {},
|
|
945
|
+
}
|
|
946
|
+
store.save(state)
|
|
947
|
+
|
|
948
|
+
client: StudioDeploymentClient | None = None
|
|
949
|
+
try:
|
|
950
|
+
client = client_factory(token)
|
|
951
|
+
proposal = _prepare_and_decide_approval(
|
|
952
|
+
request,
|
|
953
|
+
state,
|
|
954
|
+
store,
|
|
955
|
+
client,
|
|
956
|
+
run_dir=Path(run_dir),
|
|
957
|
+
previous_run_dir=previous_run_dir,
|
|
958
|
+
workspace_id=token.workspace_id,
|
|
959
|
+
)
|
|
960
|
+
proposal_id = _uuid(proposal["recipe_proposal_id"], "recipe_proposal_id")
|
|
961
|
+
fetched = client.call("get_proposal", None, resource_id=proposal_id)
|
|
962
|
+
fetched_body = _success(fetched, {200}, "DEPLOY_PROPOSAL_READ_FAILED")
|
|
963
|
+
_same(fetched_body, "recipe_digest", request.recipe_digest)
|
|
964
|
+
activation_intent = {
|
|
965
|
+
"table_name": request.dataset_name,
|
|
966
|
+
"schedule": state["schedule"],
|
|
967
|
+
}
|
|
968
|
+
_same(fetched_body, "activation_intent", activation_intent)
|
|
969
|
+
if fetched_body.get("status") not in {"approved", "activation_started"}:
|
|
970
|
+
raise HostedDeployError(
|
|
971
|
+
"DEPLOY_APPROVAL_DECISION_INVALID",
|
|
972
|
+
"Studio has not recorded this authenticated Editor's approval for the exact "
|
|
973
|
+
"Recipe proposal",
|
|
974
|
+
)
|
|
975
|
+
etag = _etag(fetched, "proposal read")
|
|
976
|
+
approval_request_id = _uuid_field(fetched_body, "approval_request_id")
|
|
977
|
+
approval_decision_id = _uuid_field(fetched_body, "approval_decision_id")
|
|
978
|
+
bootstrap = fetched_body.get("bootstrap_evidence")
|
|
979
|
+
if not isinstance(bootstrap, dict):
|
|
980
|
+
raise HostedDeployError(
|
|
981
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID", "the approved proposal omitted Build evidence"
|
|
982
|
+
)
|
|
983
|
+
bootstrap_table_digest = _legacy_local_build_bootstrap_table_digest(bootstrap)
|
|
984
|
+
activation_body = {
|
|
985
|
+
"schema_version": STUDIO_SCHEMA_VERSION,
|
|
986
|
+
"workspace_id": str(token.workspace_id),
|
|
987
|
+
"recipe_proposal_id": str(proposal_id),
|
|
988
|
+
"table_recipe_id": request.recipe_id,
|
|
989
|
+
"recipe_version": request.recipe_version,
|
|
990
|
+
"recipe_digest": request.recipe_digest,
|
|
991
|
+
"source_authority_bindings": fetched_body["source_authority_bindings"],
|
|
992
|
+
"source_authority_bindings_digest": fetched_body["source_authority_bindings_digest"],
|
|
993
|
+
"approval_request_id": str(approval_request_id),
|
|
994
|
+
"approval_decision_id": str(approval_decision_id),
|
|
995
|
+
"bootstrap_table_digest": bootstrap_table_digest,
|
|
996
|
+
"table_name": activation_intent["table_name"],
|
|
997
|
+
"schedule": activation_intent["schedule"],
|
|
998
|
+
}
|
|
999
|
+
activation = _mutate(
|
|
1000
|
+
state,
|
|
1001
|
+
store,
|
|
1002
|
+
client,
|
|
1003
|
+
name="activation",
|
|
1004
|
+
operation="activate_recipe",
|
|
1005
|
+
body=activation_body,
|
|
1006
|
+
expected={202},
|
|
1007
|
+
resource_id=proposal_id,
|
|
1008
|
+
if_match=etag,
|
|
1009
|
+
)
|
|
1010
|
+
state["resources"]["activation"] = dict(activation)
|
|
1011
|
+
state["status"] = "activation_queued"
|
|
1012
|
+
store.save(state)
|
|
1013
|
+
binding = _bind_cloud_dataset(
|
|
1014
|
+
request,
|
|
1015
|
+
state,
|
|
1016
|
+
store,
|
|
1017
|
+
credentials=credentials,
|
|
1018
|
+
transport=cloud_transport,
|
|
1019
|
+
)
|
|
1020
|
+
state["resources"]["cloud_binding"] = dict(binding)
|
|
1021
|
+
state["status"] = "cloud_dataset_bound"
|
|
1022
|
+
store.save(state)
|
|
1023
|
+
return _activation_receipt(state, store, cloud_url)
|
|
1024
|
+
finally:
|
|
1025
|
+
if client is not None:
|
|
1026
|
+
client.close()
|
|
1027
|
+
if owns_store:
|
|
1028
|
+
store.close()
|
|
1029
|
+
|
|
1030
|
+
|
|
1031
|
+
def _worker_policy_preflight(
|
|
1032
|
+
request: DeploymentRequest,
|
|
1033
|
+
recipe_document: Mapping[str, Any],
|
|
1034
|
+
proposal: Mapping[str, Any],
|
|
1035
|
+
client: StudioDeploymentClient,
|
|
1036
|
+
) -> dict[str, Any]:
|
|
1037
|
+
"""Require Studio's selected fleet to match this runtime before approval."""
|
|
1038
|
+
|
|
1039
|
+
command_fields = (
|
|
1040
|
+
"workspace_id",
|
|
1041
|
+
"dataset_id",
|
|
1042
|
+
"table_plan_id",
|
|
1043
|
+
"table_recipe_id",
|
|
1044
|
+
"recipe_version",
|
|
1045
|
+
"recipe_digest",
|
|
1046
|
+
"recipe_proposal_id",
|
|
1047
|
+
)
|
|
1048
|
+
try:
|
|
1049
|
+
command = {field: proposal[field] for field in command_fields}
|
|
1050
|
+
except KeyError as error:
|
|
1051
|
+
raise HostedDeployError(
|
|
1052
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID",
|
|
1053
|
+
"Studio's recipe proposal omitted worker-policy preflight scope",
|
|
1054
|
+
) from error
|
|
1055
|
+
if "table_id" in proposal:
|
|
1056
|
+
command["table_id"] = proposal["table_id"]
|
|
1057
|
+
response = client.preflight(command)
|
|
1058
|
+
preflight = dict(_success(response, {200}, "DEPLOY_WORKER_POLICY_PREFLIGHT_FAILED"))
|
|
1059
|
+
|
|
1060
|
+
required = {
|
|
1061
|
+
"schema_version",
|
|
1062
|
+
"workspace_id",
|
|
1063
|
+
"dataset_id",
|
|
1064
|
+
"table_plan_id",
|
|
1065
|
+
"table_recipe_id",
|
|
1066
|
+
"recipe_version",
|
|
1067
|
+
"recipe_digest",
|
|
1068
|
+
"builder_image_digest",
|
|
1069
|
+
"verifier_image_digest",
|
|
1070
|
+
"validation_policy_digest",
|
|
1071
|
+
"selection_fence",
|
|
1072
|
+
"selection_id",
|
|
1073
|
+
"policy_mode",
|
|
1074
|
+
"validation_policy_digests",
|
|
1075
|
+
"policy_set_digest",
|
|
1076
|
+
}
|
|
1077
|
+
expected_fields = required | ({"table_id"} if "table_id" in command else set())
|
|
1078
|
+
if (
|
|
1079
|
+
set(preflight) != expected_fields
|
|
1080
|
+
or preflight.get("schema_version") != STUDIO_SCHEMA_VERSION
|
|
1081
|
+
):
|
|
1082
|
+
raise HostedDeployError(
|
|
1083
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID",
|
|
1084
|
+
"Studio returned a non-exact worker-policy preflight",
|
|
1085
|
+
)
|
|
1086
|
+
for field in (
|
|
1087
|
+
"workspace_id",
|
|
1088
|
+
"dataset_id",
|
|
1089
|
+
"table_plan_id",
|
|
1090
|
+
"table_id",
|
|
1091
|
+
"table_recipe_id",
|
|
1092
|
+
"recipe_version",
|
|
1093
|
+
"recipe_digest",
|
|
1094
|
+
):
|
|
1095
|
+
if field in command and preflight.get(field) != command[field]:
|
|
1096
|
+
raise HostedDeployError(
|
|
1097
|
+
"DEPLOY_WORKER_POLICY_MISMATCH",
|
|
1098
|
+
f"Studio's worker-policy preflight {field} differs from the frozen Recipe proposal",
|
|
1099
|
+
)
|
|
1100
|
+
|
|
1101
|
+
transform_plan = recipe_document.get("transform_plan")
|
|
1102
|
+
plan_schema = (
|
|
1103
|
+
transform_plan.get("schema_version") if isinstance(transform_plan, Mapping) else None
|
|
1104
|
+
)
|
|
1105
|
+
modes = {
|
|
1106
|
+
"local-table-plan.v1": "table",
|
|
1107
|
+
"local-graph-table-plan.v1": "graph",
|
|
1108
|
+
}
|
|
1109
|
+
policy_mode = modes.get(plan_schema)
|
|
1110
|
+
if policy_mode is None:
|
|
1111
|
+
raise HostedDeployError(
|
|
1112
|
+
"DEPLOY_REQUEST_INVALID",
|
|
1113
|
+
"the frozen Recipe has no supported table or graph transform plan",
|
|
1114
|
+
)
|
|
1115
|
+
local_policies = {
|
|
1116
|
+
"table": "sha256:" + pipeline.validation_policy_digest(graph=False),
|
|
1117
|
+
"graph": "sha256:" + pipeline.validation_policy_digest(graph=True),
|
|
1118
|
+
}
|
|
1119
|
+
local_policy_set_digest = "sha256:" + canonical.canonical_sha256(local_policies)
|
|
1120
|
+
selected_policies = preflight.get("validation_policy_digests")
|
|
1121
|
+
selected_policy = preflight.get("validation_policy_digest")
|
|
1122
|
+
if (
|
|
1123
|
+
selected_policies != local_policies
|
|
1124
|
+
or preflight.get("policy_set_digest") != local_policy_set_digest
|
|
1125
|
+
or preflight.get("policy_mode") != policy_mode
|
|
1126
|
+
or selected_policy != local_policies[policy_mode]
|
|
1127
|
+
):
|
|
1128
|
+
raise HostedDeployError(
|
|
1129
|
+
"DEPLOY_WORKER_POLICY_MISMATCH",
|
|
1130
|
+
"your CLI and the hosted workers disagree about the validation policy: "
|
|
1131
|
+
f"Studio selected {selected_policy} for {preflight.get('policy_mode')}, while this "
|
|
1132
|
+
f"Recipe runtime computes {local_policies[policy_mode]} for {policy_mode}; re-export "
|
|
1133
|
+
"with 'mr-data recipe-export', or wait for the hosted fleet to be republished",
|
|
1134
|
+
)
|
|
1135
|
+
|
|
1136
|
+
builder_image = preflight.get("builder_image_digest")
|
|
1137
|
+
verifier_image = preflight.get("verifier_image_digest")
|
|
1138
|
+
selection_fence = preflight.get("selection_fence")
|
|
1139
|
+
selection_id = preflight.get("selection_id")
|
|
1140
|
+
if (
|
|
1141
|
+
not isinstance(builder_image, str)
|
|
1142
|
+
or _IMMUTABLE_IMAGE.fullmatch(builder_image) is None
|
|
1143
|
+
or not isinstance(verifier_image, str)
|
|
1144
|
+
or _IMMUTABLE_IMAGE.fullmatch(verifier_image) is None
|
|
1145
|
+
or not isinstance(selection_fence, str)
|
|
1146
|
+
or _PREFIXED_DIGEST.fullmatch(selection_fence) is None
|
|
1147
|
+
or not isinstance(selection_id, str)
|
|
1148
|
+
or _DIGEST.fullmatch(selection_id) is None
|
|
1149
|
+
):
|
|
1150
|
+
raise HostedDeployError(
|
|
1151
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID",
|
|
1152
|
+
"Studio returned an invalid worker-policy selection identity",
|
|
1153
|
+
)
|
|
1154
|
+
selection = {
|
|
1155
|
+
"schema_version": "mostlyright-worker-policy-binding.v2",
|
|
1156
|
+
"producer_image_digest": builder_image,
|
|
1157
|
+
"verifier_image_digest": verifier_image,
|
|
1158
|
+
"validation_policy_digests": local_policies,
|
|
1159
|
+
"policy_set_digest": local_policy_set_digest,
|
|
1160
|
+
"release_revision_coordinate": selection_fence,
|
|
1161
|
+
}
|
|
1162
|
+
if canonical.canonical_sha256(selection) != selection_id:
|
|
1163
|
+
raise HostedDeployError(
|
|
1164
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID",
|
|
1165
|
+
"Studio's worker-policy selection ID does not bind its exact images, policy set, and "
|
|
1166
|
+
"release fence",
|
|
1167
|
+
)
|
|
1168
|
+
if request.recipe_digest != preflight["recipe_digest"]:
|
|
1169
|
+
raise HostedDeployError(
|
|
1170
|
+
"DEPLOY_WORKER_POLICY_MISMATCH",
|
|
1171
|
+
"Studio's worker-policy preflight does not bind the exact local Recipe",
|
|
1172
|
+
)
|
|
1173
|
+
return preflight
|
|
1174
|
+
|
|
1175
|
+
|
|
1176
|
+
def _prepare_and_decide_approval(
|
|
1177
|
+
request: DeploymentRequest,
|
|
1178
|
+
state: dict[str, Any],
|
|
1179
|
+
store: DeploymentStateStore,
|
|
1180
|
+
client: StudioDeploymentClient,
|
|
1181
|
+
*,
|
|
1182
|
+
run_dir: Path,
|
|
1183
|
+
previous_run_dir: Path | None,
|
|
1184
|
+
workspace_id: UUID,
|
|
1185
|
+
) -> Mapping[str, Any]:
|
|
1186
|
+
recipe_document = canonical.parse_canonical_json(request.canonical_recipe_json.encode("utf-8"))
|
|
1187
|
+
if not isinstance(recipe_document, dict):
|
|
1188
|
+
raise HostedDeployError("DEPLOY_REQUEST_INVALID", "Recipe must be one object")
|
|
1189
|
+
resources = state["resources"]
|
|
1190
|
+
# Whether this deployment founds a Dataset or updates one is decided by whether it named a
|
|
1191
|
+
# predecessor, which is the same question `recipe.verify_recipe_candidate` asks -- a Recipe
|
|
1192
|
+
# execution names its predecessor exactly when its mode is not `initial`, and `deploy.py`
|
|
1193
|
+
# refuses both shapes where the two would disagree. Branching on `recipe_version` instead
|
|
1194
|
+
# looked equivalent and is not: a refresh of an unchanged Recipe is a successor whose Recipe
|
|
1195
|
+
# version has not moved, and it would arrive here to have its project, question, plan and
|
|
1196
|
+
# Dataset created a second time.
|
|
1197
|
+
if previous_run_dir is None:
|
|
1198
|
+
project = _mutate(
|
|
1199
|
+
state,
|
|
1200
|
+
store,
|
|
1201
|
+
client,
|
|
1202
|
+
name="project",
|
|
1203
|
+
operation="create_dataset",
|
|
1204
|
+
body={
|
|
1205
|
+
"workspace_id": str(workspace_id),
|
|
1206
|
+
"name": request.dataset_name,
|
|
1207
|
+
"description": "Managed from an exact reviewed local Recipe Build.",
|
|
1208
|
+
},
|
|
1209
|
+
expected={201},
|
|
1210
|
+
)
|
|
1211
|
+
dataset_id = _uuid_field(project, "dataset_id")
|
|
1212
|
+
question = _mutate(
|
|
1213
|
+
state,
|
|
1214
|
+
store,
|
|
1215
|
+
client,
|
|
1216
|
+
name="question",
|
|
1217
|
+
operation="create_question",
|
|
1218
|
+
body={
|
|
1219
|
+
"workspace_id": str(workspace_id),
|
|
1220
|
+
"dataset_id": str(dataset_id),
|
|
1221
|
+
"question": recipe_document["question"]["text"],
|
|
1222
|
+
},
|
|
1223
|
+
expected={201},
|
|
1224
|
+
)
|
|
1225
|
+
question_id = _uuid_field(question, "question_id")
|
|
1226
|
+
requirements_document = recipe_document["requirements"]
|
|
1227
|
+
requirements = _mutate(
|
|
1228
|
+
state,
|
|
1229
|
+
store,
|
|
1230
|
+
client,
|
|
1231
|
+
name="requirements",
|
|
1232
|
+
operation="put_requirements",
|
|
1233
|
+
body={
|
|
1234
|
+
"schema_version": STUDIO_SCHEMA_VERSION,
|
|
1235
|
+
"workspace_id": str(workspace_id),
|
|
1236
|
+
"population": requirements_document["population"],
|
|
1237
|
+
"time_range": requirements_document["time_range"],
|
|
1238
|
+
"output_grain": requirements_document["output_grain"]["columns"],
|
|
1239
|
+
"required_fields": requirements_document["required_fields"],
|
|
1240
|
+
"target_policy": requirements_document["target_policy"],
|
|
1241
|
+
"success_criteria": requirements_document["success_criteria"],
|
|
1242
|
+
"feasibility": _studio_feasibility(requirements_document["feasibility"]),
|
|
1243
|
+
**(
|
|
1244
|
+
{"prediction_cutoff": requirements_document["prediction_cutoff"]}
|
|
1245
|
+
if "prediction_cutoff" in requirements_document
|
|
1246
|
+
else {}
|
|
1247
|
+
),
|
|
1248
|
+
},
|
|
1249
|
+
expected={200, 201},
|
|
1250
|
+
resource_id=question_id,
|
|
1251
|
+
# The requirements upsert is preconditioned on the REQUIREMENTS
|
|
1252
|
+
# subresource's own version, not the question's: Studio's
|
|
1253
|
+
# put_requirements allows the initial upsert only as If-Match: "*"
|
|
1254
|
+
# (allow_initial), and the question ETag can never satisfy it.
|
|
1255
|
+
if_match="*",
|
|
1256
|
+
)
|
|
1257
|
+
requirements_id = _uuid_field(requirements, "requirements_id")
|
|
1258
|
+
source_registry_ids = _register_sources(
|
|
1259
|
+
request,
|
|
1260
|
+
recipe_document,
|
|
1261
|
+
state,
|
|
1262
|
+
store,
|
|
1263
|
+
client,
|
|
1264
|
+
run_dir=run_dir,
|
|
1265
|
+
previous_run_dir=previous_run_dir,
|
|
1266
|
+
dataset_id=dataset_id,
|
|
1267
|
+
workspace_id=workspace_id,
|
|
1268
|
+
)
|
|
1269
|
+
policy = recipe_document.get("backfill_policy") or _default_backfill_policy(
|
|
1270
|
+
requirements_document["time_range"]
|
|
1271
|
+
)
|
|
1272
|
+
plan = _mutate(
|
|
1273
|
+
state,
|
|
1274
|
+
store,
|
|
1275
|
+
client,
|
|
1276
|
+
name="plan",
|
|
1277
|
+
operation="create_plan",
|
|
1278
|
+
body={
|
|
1279
|
+
"schema_version": STUDIO_SCHEMA_VERSION,
|
|
1280
|
+
"workspace_id": str(workspace_id),
|
|
1281
|
+
"dataset_id": str(dataset_id),
|
|
1282
|
+
"question_id": str(question_id),
|
|
1283
|
+
"requirements_id": str(requirements_id),
|
|
1284
|
+
"source_ids": [str(value) for value in source_registry_ids],
|
|
1285
|
+
"output_grain": requirements_document["output_grain"]["columns"],
|
|
1286
|
+
"transformation_contract_version": "1.0.0",
|
|
1287
|
+
"execution_plan_digest": "sha256:" + request.execution_plan_digest,
|
|
1288
|
+
"validation_policy_digest": "sha256:"
|
|
1289
|
+
+ request.bootstrap_evidence["bootstrap_verification_digest"],
|
|
1290
|
+
"approved_backfill_policy": policy,
|
|
1291
|
+
"backfill_policy_digest": "sha256:" + canonical.canonical_sha256(policy),
|
|
1292
|
+
},
|
|
1293
|
+
expected={201},
|
|
1294
|
+
)
|
|
1295
|
+
table_plan_id = _uuid_field(plan, "table_plan_id")
|
|
1296
|
+
else:
|
|
1297
|
+
previous = _previous_deployment_resources(previous_run_dir)
|
|
1298
|
+
dataset_id = _uuid(previous["dataset_id"], "dataset_id")
|
|
1299
|
+
table_plan_id = _uuid(previous["table_plan_id"], "table_plan_id")
|
|
1300
|
+
resources["previous_dataset_id"] = previous["dataset_id"]
|
|
1301
|
+
store.save(state)
|
|
1302
|
+
|
|
1303
|
+
authorities = _source_authorities(
|
|
1304
|
+
request,
|
|
1305
|
+
recipe_document,
|
|
1306
|
+
state,
|
|
1307
|
+
store,
|
|
1308
|
+
client,
|
|
1309
|
+
run_dir=run_dir,
|
|
1310
|
+
previous_run_dir=previous_run_dir,
|
|
1311
|
+
workspace_id=workspace_id,
|
|
1312
|
+
)
|
|
1313
|
+
inventory_digest = canonical.canonical_sha256(
|
|
1314
|
+
[
|
|
1315
|
+
{
|
|
1316
|
+
"source_id": item["source_id"],
|
|
1317
|
+
"source_authority_digest": item["source_authority_digest"],
|
|
1318
|
+
}
|
|
1319
|
+
for item in authorities
|
|
1320
|
+
]
|
|
1321
|
+
)
|
|
1322
|
+
proposal_body = {
|
|
1323
|
+
"schema_version": STUDIO_SCHEMA_VERSION,
|
|
1324
|
+
"workspace_id": str(workspace_id),
|
|
1325
|
+
"dataset_id": str(dataset_id),
|
|
1326
|
+
"table_plan_id": str(table_plan_id),
|
|
1327
|
+
"recipe_media_type": "application/vnd.mostlyright.frozen-recipe+json",
|
|
1328
|
+
"recipe_schema_version": request.recipe_schema_version,
|
|
1329
|
+
"table_recipe_id": request.recipe_id,
|
|
1330
|
+
"recipe_version": request.recipe_version,
|
|
1331
|
+
"predecessor_recipe_digest": recipe_document.get("predecessor_recipe_digest"),
|
|
1332
|
+
"canonical_recipe_json": request.canonical_recipe_json,
|
|
1333
|
+
"recipe_digest": request.recipe_digest,
|
|
1334
|
+
"source_authority_bindings": authorities,
|
|
1335
|
+
"source_authority_bindings_digest": inventory_digest,
|
|
1336
|
+
"bootstrap_evidence": request.bootstrap_evidence,
|
|
1337
|
+
"activation_intent": {
|
|
1338
|
+
"table_name": request.dataset_name,
|
|
1339
|
+
"schedule": state["schedule"],
|
|
1340
|
+
},
|
|
1341
|
+
**({} if previous_run_dir is None else {"dataset_id": resources["previous_dataset_id"]}),
|
|
1342
|
+
}
|
|
1343
|
+
proposal = _mutate(
|
|
1344
|
+
state,
|
|
1345
|
+
store,
|
|
1346
|
+
client,
|
|
1347
|
+
name="proposal",
|
|
1348
|
+
operation="create_proposal",
|
|
1349
|
+
body=proposal_body,
|
|
1350
|
+
expected={201},
|
|
1351
|
+
# The approval request below conditions on this ETag.
|
|
1352
|
+
requires_etag=True,
|
|
1353
|
+
)
|
|
1354
|
+
proposal_id = _uuid_field(proposal, "recipe_proposal_id")
|
|
1355
|
+
_same(proposal, "recipe_digest", request.recipe_digest)
|
|
1356
|
+
activation_operation = state["operations"].get("activation")
|
|
1357
|
+
if (
|
|
1358
|
+
state.get("status") == "cloud_dataset_bound"
|
|
1359
|
+
and isinstance(resources.get("activation"), dict)
|
|
1360
|
+
and isinstance(activation_operation, dict)
|
|
1361
|
+
and activation_operation.get("status") == "completed"
|
|
1362
|
+
):
|
|
1363
|
+
return proposal
|
|
1364
|
+
worker_policy_preflight = _worker_policy_preflight(
|
|
1365
|
+
request,
|
|
1366
|
+
recipe_document,
|
|
1367
|
+
proposal,
|
|
1368
|
+
client,
|
|
1369
|
+
)
|
|
1370
|
+
proposal_etag = _resource_etag(state, "proposal")
|
|
1371
|
+
approval = _mutate(
|
|
1372
|
+
state,
|
|
1373
|
+
store,
|
|
1374
|
+
client,
|
|
1375
|
+
name="approval_request",
|
|
1376
|
+
operation="request_approval",
|
|
1377
|
+
body={
|
|
1378
|
+
"schema_version": STUDIO_SCHEMA_VERSION,
|
|
1379
|
+
"workspace_id": str(workspace_id),
|
|
1380
|
+
"recipe_proposal_id": str(proposal_id),
|
|
1381
|
+
"recipe_digest": request.recipe_digest,
|
|
1382
|
+
"worker_policy_preflight": worker_policy_preflight,
|
|
1383
|
+
},
|
|
1384
|
+
expected={201},
|
|
1385
|
+
resource_id=proposal_id,
|
|
1386
|
+
if_match=proposal_etag,
|
|
1387
|
+
# The confirmation below conditions on this ETag, and reads it from the journal.
|
|
1388
|
+
requires_etag=True,
|
|
1389
|
+
)
|
|
1390
|
+
approval_request_id = _uuid_field(approval, "approval_request_id")
|
|
1391
|
+
_same(approval, "status", "pending")
|
|
1392
|
+
_same(approval, "subject_digest", request.recipe_digest)
|
|
1393
|
+
approval_version = _positive_int_field(approval, "version")
|
|
1394
|
+
approval_etag = _resource_etag(state, "approval_request")
|
|
1395
|
+
decision = _mutate(
|
|
1396
|
+
state,
|
|
1397
|
+
store,
|
|
1398
|
+
client,
|
|
1399
|
+
name="approval_confirmation",
|
|
1400
|
+
operation="confirm_table_recipe_approval",
|
|
1401
|
+
body={
|
|
1402
|
+
"schema_version": STUDIO_SCHEMA_VERSION,
|
|
1403
|
+
"workspace_id": str(workspace_id),
|
|
1404
|
+
"decision": "approved",
|
|
1405
|
+
"expected_request_version": approval_version,
|
|
1406
|
+
"subject_digest": request.recipe_digest,
|
|
1407
|
+
},
|
|
1408
|
+
expected={200},
|
|
1409
|
+
resource_id=approval_request_id,
|
|
1410
|
+
if_match=approval_etag,
|
|
1411
|
+
)
|
|
1412
|
+
_same(decision, "approval_request_id", str(approval_request_id))
|
|
1413
|
+
_same(decision, "decision", "approved")
|
|
1414
|
+
_same(decision, "subject_digest", request.recipe_digest)
|
|
1415
|
+
resources["approval_request_id"] = str(approval_request_id)
|
|
1416
|
+
resources["approval_decision_id"] = str(_uuid_field(decision, "approval_decision_id"))
|
|
1417
|
+
resources["dataset_id"] = str(dataset_id)
|
|
1418
|
+
resources["table_plan_id"] = str(table_plan_id)
|
|
1419
|
+
state["status"] = "approval_decided"
|
|
1420
|
+
store.save(state)
|
|
1421
|
+
return proposal
|
|
1422
|
+
|
|
1423
|
+
|
|
1424
|
+
def _acquisition_bundle(
|
|
1425
|
+
request: DeploymentRequest,
|
|
1426
|
+
*,
|
|
1427
|
+
run_dir: Path,
|
|
1428
|
+
previous_run_dir: Path | None,
|
|
1429
|
+
) -> Mapping[str, Any]:
|
|
1430
|
+
"""Return the closed acquisition-evidence bundle this request names, from either home.
|
|
1431
|
+
|
|
1432
|
+
A request that carries a reviewed-Build binding carries the bundle document inside it, and that
|
|
1433
|
+
copy is used unchanged -- it is the exact document the binding's signature covers, so reading
|
|
1434
|
+
it from anywhere else would check a different object than the one that was signed.
|
|
1435
|
+
|
|
1436
|
+
A request built for a dataset the caller publishes itself carries no binding, so the bundle is
|
|
1437
|
+
read back out of the run directory under ``request.acquisition_bundle_digest``, which
|
|
1438
|
+
:func:`deployment_evidence.load_acquisition_bundle_document` authenticates to this exact Build
|
|
1439
|
+
before returning it. That digest is a field of the request, so it is inside the request digest,
|
|
1440
|
+
so it is inside the deployment identity the journal is held to. This is the same route
|
|
1441
|
+
``_source_authorities`` already takes for the source bytes themselves.
|
|
1442
|
+
"""
|
|
1443
|
+
|
|
1444
|
+
reviewed = request.bootstrap_evidence.get("reviewed_build")
|
|
1445
|
+
if isinstance(reviewed, Mapping):
|
|
1446
|
+
bundle = reviewed["acquisition_bundle"]
|
|
1447
|
+
if not isinstance(bundle, Mapping):
|
|
1448
|
+
raise HostedDeployError(
|
|
1449
|
+
"DEPLOY_REQUEST_INVALID", "the reviewed Build carries no acquisition bundle"
|
|
1450
|
+
)
|
|
1451
|
+
return bundle
|
|
1452
|
+
try:
|
|
1453
|
+
return deployment_evidence.load_acquisition_bundle_document(
|
|
1454
|
+
run_dir,
|
|
1455
|
+
request.acquisition_bundle_digest,
|
|
1456
|
+
previous_run_dir=previous_run_dir,
|
|
1457
|
+
)
|
|
1458
|
+
except deployment_evidence.DeploymentEvidenceError as error:
|
|
1459
|
+
raise HostedDeployError(
|
|
1460
|
+
"DEPLOY_REQUEST_INVALID",
|
|
1461
|
+
f"the acquisition evidence for {run_dir} could not be read [{error.code}]",
|
|
1462
|
+
) from error
|
|
1463
|
+
|
|
1464
|
+
|
|
1465
|
+
def _register_sources(
|
|
1466
|
+
request: DeploymentRequest,
|
|
1467
|
+
recipe_document: Mapping[str, Any],
|
|
1468
|
+
state: dict[str, Any],
|
|
1469
|
+
store: DeploymentStateStore,
|
|
1470
|
+
client: StudioDeploymentClient,
|
|
1471
|
+
*,
|
|
1472
|
+
run_dir: Path,
|
|
1473
|
+
previous_run_dir: Path | None,
|
|
1474
|
+
dataset_id: UUID,
|
|
1475
|
+
workspace_id: UUID,
|
|
1476
|
+
) -> list[UUID]:
|
|
1477
|
+
proposals = {item["source_id"]: item for item in recipe_document["source_proposals"]}
|
|
1478
|
+
recipe_sources = {item["source_id"]: item for item in recipe_document["sources"]}
|
|
1479
|
+
bundle_sources = {
|
|
1480
|
+
item["source_id"]: item
|
|
1481
|
+
for item in _acquisition_bundle(
|
|
1482
|
+
request, run_dir=run_dir, previous_run_dir=previous_run_dir
|
|
1483
|
+
)["sources"]
|
|
1484
|
+
}
|
|
1485
|
+
result: list[UUID] = []
|
|
1486
|
+
for source_id in sorted(recipe_sources):
|
|
1487
|
+
proposal = proposals[source_id]
|
|
1488
|
+
source = recipe_sources[source_id]
|
|
1489
|
+
bundle = bundle_sources[source_id]
|
|
1490
|
+
locator = _source_locator(source, proposal, bundle)
|
|
1491
|
+
response = _mutate(
|
|
1492
|
+
state,
|
|
1493
|
+
store,
|
|
1494
|
+
client,
|
|
1495
|
+
name=f"source_registry:{source_id}",
|
|
1496
|
+
operation="register_source",
|
|
1497
|
+
body={
|
|
1498
|
+
"schema_version": STUDIO_SCHEMA_VERSION,
|
|
1499
|
+
"workspace_id": str(workspace_id),
|
|
1500
|
+
"dataset_id": str(dataset_id),
|
|
1501
|
+
"name": proposal["display_name"],
|
|
1502
|
+
"source_class": proposal["source_class"],
|
|
1503
|
+
"locator": locator,
|
|
1504
|
+
"credential_reference_ids": [],
|
|
1505
|
+
"data_classification": source["declared_classification"],
|
|
1506
|
+
"rights_claim": {
|
|
1507
|
+
"claimed_basis": (
|
|
1508
|
+
"permission_asserted"
|
|
1509
|
+
if proposal.get("rights_status") == "approved"
|
|
1510
|
+
else "unknown"
|
|
1511
|
+
),
|
|
1512
|
+
"claim_evidence_digest": "sha256:"
|
|
1513
|
+
+ canonical.canonical_sha256(proposal["evidence"]),
|
|
1514
|
+
"claim_note": source["lawful_basis"],
|
|
1515
|
+
},
|
|
1516
|
+
"retention_policy": {
|
|
1517
|
+
"raw_days": 30,
|
|
1518
|
+
"derived_days": 365,
|
|
1519
|
+
"tombstone_required": False,
|
|
1520
|
+
},
|
|
1521
|
+
},
|
|
1522
|
+
expected={201},
|
|
1523
|
+
)
|
|
1524
|
+
result.append(_uuid_field(response, "source_id"))
|
|
1525
|
+
return result
|
|
1526
|
+
|
|
1527
|
+
|
|
1528
|
+
def _source_authorities(
|
|
1529
|
+
request: DeploymentRequest,
|
|
1530
|
+
recipe_document: Mapping[str, Any],
|
|
1531
|
+
state: dict[str, Any],
|
|
1532
|
+
store: DeploymentStateStore,
|
|
1533
|
+
client: StudioDeploymentClient,
|
|
1534
|
+
*,
|
|
1535
|
+
run_dir: Path,
|
|
1536
|
+
previous_run_dir: Path | None,
|
|
1537
|
+
workspace_id: UUID,
|
|
1538
|
+
) -> list[dict[str, Any]]:
|
|
1539
|
+
bundle = _acquisition_bundle(request, run_dir=run_dir, previous_run_dir=previous_run_dir)
|
|
1540
|
+
recipe_sources = {item["source_id"]: item for item in recipe_document["sources"]}
|
|
1541
|
+
resources = state["resources"]
|
|
1542
|
+
result: list[dict[str, Any]] = []
|
|
1543
|
+
for entry in bundle["sources"]:
|
|
1544
|
+
source_id = entry["source_id"]
|
|
1545
|
+
if entry["authority_kind"] == "connector":
|
|
1546
|
+
source = recipe_sources[source_id]
|
|
1547
|
+
if source["adapter_id"] not in {"public.https", "external.openligadb"}:
|
|
1548
|
+
raise HostedDeployError(
|
|
1549
|
+
"DEPLOY_CONNECTOR_UNSUPPORTED",
|
|
1550
|
+
f"{source_id} does not use a deployed refresh connector",
|
|
1551
|
+
)
|
|
1552
|
+
egress_attestation = (
|
|
1553
|
+
PUBLIC_HTTPS_EGRESS_POLICY_ATTESTATION
|
|
1554
|
+
if source["adapter_id"] == "public.https"
|
|
1555
|
+
else OPENLIGADB_EGRESS_POLICY_ATTESTATION
|
|
1556
|
+
)
|
|
1557
|
+
body = {
|
|
1558
|
+
"schema_version": STUDIO_SCHEMA_VERSION,
|
|
1559
|
+
"workspace_id": str(workspace_id),
|
|
1560
|
+
"source_id": source_id,
|
|
1561
|
+
"adapter_id": source["adapter_id"],
|
|
1562
|
+
"credential_mode": "none",
|
|
1563
|
+
"crawler_egress_policy_attestation": egress_attestation,
|
|
1564
|
+
**(
|
|
1565
|
+
{"query": source["query_template"]}
|
|
1566
|
+
if source["adapter_id"] == "public.https"
|
|
1567
|
+
else {}
|
|
1568
|
+
),
|
|
1569
|
+
}
|
|
1570
|
+
configuration = _mutate(
|
|
1571
|
+
state,
|
|
1572
|
+
store,
|
|
1573
|
+
client,
|
|
1574
|
+
name=f"connector:{source_id}",
|
|
1575
|
+
operation="register_connector",
|
|
1576
|
+
body=body,
|
|
1577
|
+
expected={201},
|
|
1578
|
+
)
|
|
1579
|
+
authority = {
|
|
1580
|
+
"source_id": source_id,
|
|
1581
|
+
"authority_kind": "connector",
|
|
1582
|
+
"source_authority_digest": "",
|
|
1583
|
+
"adapter_id": source["adapter_id"],
|
|
1584
|
+
"connector_configuration_id": configuration["connector_configuration_id"],
|
|
1585
|
+
"connector_configuration_digest": configuration["configuration_digest"],
|
|
1586
|
+
"credential_mode": "none",
|
|
1587
|
+
"crawler_egress_policy_attestation": egress_attestation,
|
|
1588
|
+
}
|
|
1589
|
+
authority["source_authority_digest"] = canonical.canonical_sha256(
|
|
1590
|
+
{key: value for key, value in authority.items() if key != "source_authority_digest"}
|
|
1591
|
+
)
|
|
1592
|
+
result.append(authority)
|
|
1593
|
+
continue
|
|
1594
|
+
payload = deployment_evidence.load_acquisition_source_payload(
|
|
1595
|
+
run_dir,
|
|
1596
|
+
request.acquisition_bundle_digest,
|
|
1597
|
+
source_id,
|
|
1598
|
+
previous_run_dir=previous_run_dir,
|
|
1599
|
+
)
|
|
1600
|
+
descriptors = {
|
|
1601
|
+
role: {
|
|
1602
|
+
"media_type": reference["media_type"],
|
|
1603
|
+
"expected_content_digest": "sha256:" + reference["content_sha256"],
|
|
1604
|
+
"expected_size_bytes": reference["size_bytes"],
|
|
1605
|
+
}
|
|
1606
|
+
for role, reference in (
|
|
1607
|
+
("source_data", entry["source_data"]),
|
|
1608
|
+
("acquisition_receipt", entry["acquisition_receipt"]),
|
|
1609
|
+
("source_observation", entry["source_observation"]),
|
|
1610
|
+
)
|
|
1611
|
+
}
|
|
1612
|
+
content_by_role = {
|
|
1613
|
+
"source_data": payload.source_data,
|
|
1614
|
+
"acquisition_receipt": payload.acquisition_receipt,
|
|
1615
|
+
"source_observation": payload.source_observation,
|
|
1616
|
+
}
|
|
1617
|
+
attempts = state.setdefault("source_staging_attempts", {})
|
|
1618
|
+
attempt = attempts.get(source_id, 0)
|
|
1619
|
+
if isinstance(attempt, bool) or not isinstance(attempt, int) or attempt < 0:
|
|
1620
|
+
raise HostedDeployError(
|
|
1621
|
+
"DEPLOY_STATE_INVALID", f"source staging attempt for {source_id} is invalid"
|
|
1622
|
+
)
|
|
1623
|
+
renewed = False
|
|
1624
|
+
while True:
|
|
1625
|
+
suffix = "" if attempt == 0 else f":renew:{attempt}"
|
|
1626
|
+
staging_id = _source_staging_id(request.digest(), workspace_id, source_id, attempt)
|
|
1627
|
+
staging_command = {
|
|
1628
|
+
"schema_version": STUDIO_SCHEMA_VERSION,
|
|
1629
|
+
"source_staging_id": str(staging_id),
|
|
1630
|
+
"workspace_id": str(workspace_id),
|
|
1631
|
+
"source_id": source_id,
|
|
1632
|
+
**descriptors,
|
|
1633
|
+
}
|
|
1634
|
+
staging = _mutate(
|
|
1635
|
+
state,
|
|
1636
|
+
store,
|
|
1637
|
+
client,
|
|
1638
|
+
name=f"source_staging:{source_id}{suffix}",
|
|
1639
|
+
operation="create_source_staging",
|
|
1640
|
+
body=staging_command,
|
|
1641
|
+
expected={201},
|
|
1642
|
+
)
|
|
1643
|
+
artifacts = _validate_source_staging(
|
|
1644
|
+
staging,
|
|
1645
|
+
command=staging_command,
|
|
1646
|
+
staging_id=staging_id,
|
|
1647
|
+
workspace_id=workspace_id,
|
|
1648
|
+
source_id=source_id,
|
|
1649
|
+
)
|
|
1650
|
+
if staging.get("status") == "finalized":
|
|
1651
|
+
break
|
|
1652
|
+
# A staging Studio has already sealed is replayed from its journaled finalize
|
|
1653
|
+
# response instead of being walked again.
|
|
1654
|
+
#
|
|
1655
|
+
# ⚠ WHY THIS EXISTS. The create response journals signed upload sessions that live for
|
|
1656
|
+
# minutes, and the whole point of a resume is that it arrives hours or days later,
|
|
1657
|
+
# after a human has approved in Cloud. Walking the uploads again therefore always found
|
|
1658
|
+
# those sessions expired, took the renewal path, and minted a SECOND staging: every
|
|
1659
|
+
# source byte uploaded to Studio a second time, new artifact ids, and new source
|
|
1660
|
+
# authorities -- which are inputs to the Recipe proposal the human already approved.
|
|
1661
|
+
# The proposal short-circuit hid it, so the rebuilt bindings were discarded in silence
|
|
1662
|
+
# and the resume looked fine. Reading the journaled request digest is what made it
|
|
1663
|
+
# visible: the rebuilt proposal no longer matched the one Studio holds, because this
|
|
1664
|
+
# invocation had just moved it.
|
|
1665
|
+
#
|
|
1666
|
+
# The sealed objects, their artifact ids and the source authority all come back in the
|
|
1667
|
+
# finalize response, which is journaled. Replaying it keeps the authorities -- and the
|
|
1668
|
+
# proposal built from them -- exactly what the approval was given against, and stops a
|
|
1669
|
+
# resume re-uploading a Build that is already staged.
|
|
1670
|
+
if isinstance(resources.get(f"source_finalize:{source_id}{suffix}"), dict):
|
|
1671
|
+
staging = _mutate(
|
|
1672
|
+
state,
|
|
1673
|
+
store,
|
|
1674
|
+
client,
|
|
1675
|
+
name=f"source_finalize:{source_id}{suffix}",
|
|
1676
|
+
operation="finalize_source_staging",
|
|
1677
|
+
body={
|
|
1678
|
+
"schema_version": STUDIO_SCHEMA_VERSION,
|
|
1679
|
+
"workspace_id": str(workspace_id),
|
|
1680
|
+
},
|
|
1681
|
+
expected={200},
|
|
1682
|
+
resource_id=staging_id,
|
|
1683
|
+
)
|
|
1684
|
+
if staging.get("status") != "finalized":
|
|
1685
|
+
raise HostedDeployError(
|
|
1686
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID",
|
|
1687
|
+
f"the journaled finalize for {source_id} does not report a sealed staging",
|
|
1688
|
+
)
|
|
1689
|
+
artifacts = _validate_source_staging(
|
|
1690
|
+
staging,
|
|
1691
|
+
command=staging_command,
|
|
1692
|
+
staging_id=staging_id,
|
|
1693
|
+
workspace_id=workspace_id,
|
|
1694
|
+
source_id=source_id,
|
|
1695
|
+
)
|
|
1696
|
+
# The journaled finalize is what reconciled any transfer whose response was lost,
|
|
1697
|
+
# so a replay records that here exactly as the live finalize does. Skipping it
|
|
1698
|
+
# would leave the journal describing a transfer Studio has sealed as still planned.
|
|
1699
|
+
_reconcile_lost_uploads(state, store, source_id)
|
|
1700
|
+
break
|
|
1701
|
+
try:
|
|
1702
|
+
for role, artifact in artifacts.items():
|
|
1703
|
+
session = artifact["upload_session"]
|
|
1704
|
+
_validate_source_upload_session(
|
|
1705
|
+
session,
|
|
1706
|
+
role=role,
|
|
1707
|
+
artifact_id=_uuid(artifact["artifact_id"], "artifact_id"),
|
|
1708
|
+
staging_id=staging_id,
|
|
1709
|
+
workspace_id=workspace_id,
|
|
1710
|
+
descriptor=descriptors[role],
|
|
1711
|
+
)
|
|
1712
|
+
generation = _upload_once(
|
|
1713
|
+
state,
|
|
1714
|
+
store,
|
|
1715
|
+
client,
|
|
1716
|
+
name=f"source_upload:{source_id}:{role}{suffix}",
|
|
1717
|
+
session=session,
|
|
1718
|
+
content=content_by_role[role],
|
|
1719
|
+
)
|
|
1720
|
+
if generation is not None:
|
|
1721
|
+
_mutate(
|
|
1722
|
+
state,
|
|
1723
|
+
store,
|
|
1724
|
+
client,
|
|
1725
|
+
name=f"source_complete:{source_id}:{role}{suffix}",
|
|
1726
|
+
operation="complete_artifact_upload",
|
|
1727
|
+
body={
|
|
1728
|
+
"schema_version": STUDIO_SCHEMA_VERSION,
|
|
1729
|
+
"workspace_id": str(workspace_id),
|
|
1730
|
+
"reservation_id": session["reservation_id"],
|
|
1731
|
+
"reservation_digest": session["reservation_digest"],
|
|
1732
|
+
"upload_session_id": session["session_id"],
|
|
1733
|
+
"object_generation": generation,
|
|
1734
|
+
"observed_size_bytes": len(content_by_role[role]),
|
|
1735
|
+
"observed_content_digest": "sha256:"
|
|
1736
|
+
+ hashlib.sha256(content_by_role[role]).hexdigest(),
|
|
1737
|
+
},
|
|
1738
|
+
expected={201},
|
|
1739
|
+
resource_id=_uuid(artifact["artifact_id"], "artifact_id"),
|
|
1740
|
+
)
|
|
1741
|
+
except HostedDeployError as error:
|
|
1742
|
+
if error.code != "DEPLOY_SOURCE_UPLOAD_SESSION_REJECTED" or renewed:
|
|
1743
|
+
raise
|
|
1744
|
+
attempt += 1
|
|
1745
|
+
attempts[source_id] = attempt
|
|
1746
|
+
store.save(state)
|
|
1747
|
+
renewed = True
|
|
1748
|
+
continue
|
|
1749
|
+
staging = _mutate(
|
|
1750
|
+
state,
|
|
1751
|
+
store,
|
|
1752
|
+
client,
|
|
1753
|
+
name=f"source_finalize:{source_id}{suffix}",
|
|
1754
|
+
operation="finalize_source_staging",
|
|
1755
|
+
body={"schema_version": STUDIO_SCHEMA_VERSION, "workspace_id": str(workspace_id)},
|
|
1756
|
+
expected={200},
|
|
1757
|
+
resource_id=staging_id,
|
|
1758
|
+
)
|
|
1759
|
+
artifacts = _validate_source_staging(
|
|
1760
|
+
staging,
|
|
1761
|
+
command=staging_command,
|
|
1762
|
+
staging_id=staging_id,
|
|
1763
|
+
workspace_id=workspace_id,
|
|
1764
|
+
source_id=source_id,
|
|
1765
|
+
)
|
|
1766
|
+
_reconcile_lost_uploads(state, store, source_id)
|
|
1767
|
+
break
|
|
1768
|
+
authority = _validate_source_authority(
|
|
1769
|
+
staging.get("authority"),
|
|
1770
|
+
source_id=source_id,
|
|
1771
|
+
workspace_id=workspace_id,
|
|
1772
|
+
artifacts=artifacts,
|
|
1773
|
+
descriptors=descriptors,
|
|
1774
|
+
)
|
|
1775
|
+
result.append(authority)
|
|
1776
|
+
if [item["source_id"] for item in result] != sorted(recipe_sources):
|
|
1777
|
+
raise HostedDeployError(
|
|
1778
|
+
"DEPLOY_SOURCE_INVENTORY_MISMATCH", "Studio source authority inventory is not exact"
|
|
1779
|
+
)
|
|
1780
|
+
return result
|
|
1781
|
+
|
|
1782
|
+
|
|
1783
|
+
def _reconcile_lost_uploads(
|
|
1784
|
+
state: dict[str, Any], store: DeploymentStateStore, source_id: str
|
|
1785
|
+
) -> None:
|
|
1786
|
+
"""Retire this source's still-planned transfers once Studio has sealed the staging.
|
|
1787
|
+
|
|
1788
|
+
A transfer can commit remotely while its response is lost, and :func:`_upload_once` leaves that
|
|
1789
|
+
operation ``planned`` on purpose. The finalize response is what settles it: Studio names the
|
|
1790
|
+
exact sealed object, so the plan is neither outstanding nor a failure. Recording it is not
|
|
1791
|
+
bookkeeping for its own sake -- the journal is the evidence of what this deployment did.
|
|
1792
|
+
"""
|
|
1793
|
+
|
|
1794
|
+
for name, operation in list(state["operations"].items()):
|
|
1795
|
+
if (
|
|
1796
|
+
name.startswith(f"source_upload:{source_id}:")
|
|
1797
|
+
and isinstance(operation, dict)
|
|
1798
|
+
and operation.get("status") == "planned"
|
|
1799
|
+
):
|
|
1800
|
+
state["operations"][name] = {**operation, "status": "reconciled"}
|
|
1801
|
+
store.save(state)
|
|
1802
|
+
|
|
1803
|
+
|
|
1804
|
+
def _source_staging_id(
|
|
1805
|
+
request_digest: str,
|
|
1806
|
+
workspace_id: UUID,
|
|
1807
|
+
source_id: str,
|
|
1808
|
+
attempt: int,
|
|
1809
|
+
) -> UUID:
|
|
1810
|
+
return uuid5(
|
|
1811
|
+
NAMESPACE_URL,
|
|
1812
|
+
f"mostlyright:source-staging:{request_digest}:{workspace_id}:{source_id}:attempt:{attempt}",
|
|
1813
|
+
)
|
|
1814
|
+
|
|
1815
|
+
|
|
1816
|
+
_SOURCE_STAGING_ROLES = {
|
|
1817
|
+
"source_data": ("raw_snapshot", "source_staging_source_data"),
|
|
1818
|
+
"acquisition_receipt": (
|
|
1819
|
+
"candidate_evidence",
|
|
1820
|
+
"source_staging_acquisition_receipt",
|
|
1821
|
+
),
|
|
1822
|
+
"source_observation": (
|
|
1823
|
+
"candidate_evidence",
|
|
1824
|
+
"source_staging_source_observation",
|
|
1825
|
+
),
|
|
1826
|
+
}
|
|
1827
|
+
|
|
1828
|
+
|
|
1829
|
+
def _validate_source_staging(
|
|
1830
|
+
staging: Mapping[str, Any],
|
|
1831
|
+
*,
|
|
1832
|
+
command: Mapping[str, Any],
|
|
1833
|
+
staging_id: UUID,
|
|
1834
|
+
workspace_id: UUID,
|
|
1835
|
+
source_id: str,
|
|
1836
|
+
) -> dict[str, Mapping[str, Any]]:
|
|
1837
|
+
if (
|
|
1838
|
+
staging.get("schema_version") != STUDIO_SCHEMA_VERSION
|
|
1839
|
+
or staging.get("source_staging_id") != str(staging_id)
|
|
1840
|
+
or staging.get("workspace_id") != str(workspace_id)
|
|
1841
|
+
or staging.get("source_id") != source_id
|
|
1842
|
+
or staging.get("requested_artifacts")
|
|
1843
|
+
!= {role: command[role] for role in _SOURCE_STAGING_ROLES}
|
|
1844
|
+
or staging.get("status") not in {"preparing", "uploading", "finalized"}
|
|
1845
|
+
):
|
|
1846
|
+
raise HostedDeployError(
|
|
1847
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID",
|
|
1848
|
+
f"Studio returned a source staging outside the request for {source_id}",
|
|
1849
|
+
)
|
|
1850
|
+
raw = staging.get("artifacts")
|
|
1851
|
+
if not isinstance(raw, list):
|
|
1852
|
+
raise HostedDeployError(
|
|
1853
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID",
|
|
1854
|
+
f"Studio returned an invalid source staging inventory for {source_id}",
|
|
1855
|
+
)
|
|
1856
|
+
artifacts: dict[str, Mapping[str, Any]] = {}
|
|
1857
|
+
for item in raw:
|
|
1858
|
+
if not isinstance(item, dict):
|
|
1859
|
+
raise HostedDeployError(
|
|
1860
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID",
|
|
1861
|
+
f"Studio returned an invalid source staging inventory for {source_id}",
|
|
1862
|
+
)
|
|
1863
|
+
role = item.get("role")
|
|
1864
|
+
if role not in _SOURCE_STAGING_ROLES or role in artifacts:
|
|
1865
|
+
raise HostedDeployError(
|
|
1866
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID",
|
|
1867
|
+
f"Studio returned a duplicate or unknown source staging role for {source_id}",
|
|
1868
|
+
)
|
|
1869
|
+
_uuid(item.get("artifact_id"), "artifact_id")
|
|
1870
|
+
if not isinstance(item.get("upload_session"), dict):
|
|
1871
|
+
raise HostedDeployError(
|
|
1872
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID",
|
|
1873
|
+
f"Studio omitted a source staging session for {source_id}",
|
|
1874
|
+
)
|
|
1875
|
+
artifacts[role] = item
|
|
1876
|
+
if set(artifacts) != set(_SOURCE_STAGING_ROLES):
|
|
1877
|
+
raise HostedDeployError(
|
|
1878
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID",
|
|
1879
|
+
f"Studio returned an incomplete source staging for {source_id}",
|
|
1880
|
+
)
|
|
1881
|
+
return artifacts
|
|
1882
|
+
|
|
1883
|
+
|
|
1884
|
+
def _validate_source_upload_session(
|
|
1885
|
+
session: Mapping[str, Any],
|
|
1886
|
+
*,
|
|
1887
|
+
role: str,
|
|
1888
|
+
artifact_id: UUID,
|
|
1889
|
+
staging_id: UUID,
|
|
1890
|
+
workspace_id: UUID,
|
|
1891
|
+
descriptor: Mapping[str, Any],
|
|
1892
|
+
) -> None:
|
|
1893
|
+
kind, purpose = _SOURCE_STAGING_ROLES[role]
|
|
1894
|
+
forbidden = {"attempt_id", "producer_generation", "fence", "verification_generation"}
|
|
1895
|
+
if (
|
|
1896
|
+
session.get("schema_version") != STUDIO_SCHEMA_VERSION
|
|
1897
|
+
or session.get("workspace_id") != str(workspace_id)
|
|
1898
|
+
or session.get("run_id") != str(staging_id)
|
|
1899
|
+
or session.get("artifact_id") != str(artifact_id)
|
|
1900
|
+
or session.get("kind") != kind
|
|
1901
|
+
or session.get("purpose") != purpose
|
|
1902
|
+
or session.get("classification") != "restricted"
|
|
1903
|
+
or session.get("media_type") != descriptor["media_type"]
|
|
1904
|
+
or session.get("contract_version") != STUDIO_SCHEMA_VERSION
|
|
1905
|
+
or session.get("direction") != "upload"
|
|
1906
|
+
or session.get("route_authority") != "source_staging_editor"
|
|
1907
|
+
or session.get("method") != "PUT"
|
|
1908
|
+
or session.get("expected_content_digest") != descriptor["expected_content_digest"]
|
|
1909
|
+
or session.get("expected_size_bytes") != descriptor["expected_size_bytes"]
|
|
1910
|
+
or session.get("single_use") is not True
|
|
1911
|
+
or any(field in session for field in forbidden)
|
|
1912
|
+
):
|
|
1913
|
+
raise HostedDeployError(
|
|
1914
|
+
"DEPLOY_SIGNED_SESSION_INVALID",
|
|
1915
|
+
f"Studio returned an upload capability outside the {role} reservation",
|
|
1916
|
+
)
|
|
1917
|
+
for field in ("session_id", "reservation_id"):
|
|
1918
|
+
_uuid(session.get(field), field)
|
|
1919
|
+
for field in ("reservation_digest", "object_key_digest"):
|
|
1920
|
+
value = session.get(field)
|
|
1921
|
+
if (
|
|
1922
|
+
not isinstance(value, str)
|
|
1923
|
+
or not value.startswith("sha256:")
|
|
1924
|
+
or _DIGEST.fullmatch(value.removeprefix("sha256:")) is None
|
|
1925
|
+
):
|
|
1926
|
+
raise HostedDeployError(
|
|
1927
|
+
"DEPLOY_SIGNED_SESSION_INVALID", f"Studio returned an invalid {field}"
|
|
1928
|
+
)
|
|
1929
|
+
issued_at = _timestamp(_required_text(session, "issued_at"), "issued_at")
|
|
1930
|
+
expires_at = _timestamp(_required_text(session, "expires_at"), "expires_at")
|
|
1931
|
+
if expires_at <= issued_at:
|
|
1932
|
+
raise HostedDeployError(
|
|
1933
|
+
"DEPLOY_SIGNED_SESSION_INVALID", "Studio returned an invalid upload session lifetime"
|
|
1934
|
+
)
|
|
1935
|
+
if expires_at <= datetime.now(UTC):
|
|
1936
|
+
raise HostedDeployError(
|
|
1937
|
+
"DEPLOY_SOURCE_UPLOAD_SESSION_REJECTED",
|
|
1938
|
+
"the signed source upload session expired before it could be used",
|
|
1939
|
+
)
|
|
1940
|
+
|
|
1941
|
+
|
|
1942
|
+
def _validate_source_authority(
|
|
1943
|
+
raw: Any,
|
|
1944
|
+
*,
|
|
1945
|
+
source_id: str,
|
|
1946
|
+
workspace_id: UUID,
|
|
1947
|
+
artifacts: Mapping[str, Mapping[str, Any]],
|
|
1948
|
+
descriptors: Mapping[str, Mapping[str, Any]],
|
|
1949
|
+
) -> dict[str, Any]:
|
|
1950
|
+
if not isinstance(raw, dict):
|
|
1951
|
+
raise HostedDeployError(
|
|
1952
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID", f"Studio omitted source authority for {source_id}"
|
|
1953
|
+
)
|
|
1954
|
+
authority = dict(raw)
|
|
1955
|
+
reference_fields = {
|
|
1956
|
+
"source_data": "source_artifact",
|
|
1957
|
+
"acquisition_receipt": "acquisition_receipt_artifact",
|
|
1958
|
+
"source_observation": "source_observation_artifact",
|
|
1959
|
+
}
|
|
1960
|
+
if (
|
|
1961
|
+
authority.get("source_id") != source_id
|
|
1962
|
+
or authority.get("authority_kind") != "sealed_artifact"
|
|
1963
|
+
):
|
|
1964
|
+
raise HostedDeployError(
|
|
1965
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID",
|
|
1966
|
+
f"Studio returned mismatched authority for {source_id}",
|
|
1967
|
+
)
|
|
1968
|
+
for role, field in reference_fields.items():
|
|
1969
|
+
reference = authority.get(field)
|
|
1970
|
+
if (
|
|
1971
|
+
not isinstance(reference, dict)
|
|
1972
|
+
or reference.get("workspace_id") != str(workspace_id)
|
|
1973
|
+
or reference.get("artifact_id") != artifacts[role].get("artifact_id")
|
|
1974
|
+
or reference.get("contract_version") != STUDIO_SCHEMA_VERSION
|
|
1975
|
+
or reference.get("content_digest") != descriptors[role]["expected_content_digest"]
|
|
1976
|
+
):
|
|
1977
|
+
raise HostedDeployError(
|
|
1978
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID",
|
|
1979
|
+
f"Studio returned a mismatched {role} authority for {source_id}",
|
|
1980
|
+
)
|
|
1981
|
+
digest = authority.get("source_authority_digest")
|
|
1982
|
+
expected = canonical.canonical_sha256(
|
|
1983
|
+
{key: value for key, value in authority.items() if key != "source_authority_digest"}
|
|
1984
|
+
)
|
|
1985
|
+
if digest != expected:
|
|
1986
|
+
raise HostedDeployError(
|
|
1987
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID",
|
|
1988
|
+
f"Studio returned a mismatched authority digest for {source_id}",
|
|
1989
|
+
)
|
|
1990
|
+
return authority
|
|
1991
|
+
|
|
1992
|
+
|
|
1993
|
+
#: Where a journaled plan records one digest per top-level field of the request body.
|
|
1994
|
+
#:
|
|
1995
|
+
#: The whole-body digest says *that* a replayed request differs. This says *which* field does, and
|
|
1996
|
+
#: it is the difference between a refusal a person can act on and one they can only stare at. It is
|
|
1997
|
+
#: derived from the same body ``request_digest`` already covers, so it adds naming and never
|
|
1998
|
+
#: authority: no check reads it to decide whether a request matches, only to describe how it does
|
|
1999
|
+
#: not.
|
|
2000
|
+
_REQUEST_FIELDS = "request_field_digests"
|
|
2001
|
+
|
|
2002
|
+
|
|
2003
|
+
def _request_field_digests(body: Mapping[str, Any]) -> dict[str, str]:
|
|
2004
|
+
"""One digest per top-level field of a request body, so a replay can name what moved."""
|
|
2005
|
+
|
|
2006
|
+
return {str(field): canonical.canonical_sha256(value) for field, value in body.items()}
|
|
2007
|
+
|
|
2008
|
+
|
|
2009
|
+
def _journaled_request_digest(journaled: Any) -> str | None:
|
|
2010
|
+
"""The request digest a journaled operation records, or ``None`` when it records none."""
|
|
2011
|
+
|
|
2012
|
+
if not isinstance(journaled, Mapping):
|
|
2013
|
+
return None
|
|
2014
|
+
recorded = journaled.get("request_digest")
|
|
2015
|
+
if not isinstance(recorded, str) or _DIGEST.fullmatch(recorded) is None:
|
|
2016
|
+
return None
|
|
2017
|
+
return recorded
|
|
2018
|
+
|
|
2019
|
+
|
|
2020
|
+
def _drifted_request_fields(journaled: Any, fields: Mapping[str, str]) -> str:
|
|
2021
|
+
"""The request fields that differ from the journaled plan, named for a person to read.
|
|
2022
|
+
|
|
2023
|
+
A plan journaled before per-field digests existed can only say that the request as a whole
|
|
2024
|
+
differs, and that is exactly what this says about it. Naming a field the journal cannot see
|
|
2025
|
+
would be a guess, and a guess in a refusal sends an operator to fix the wrong input.
|
|
2026
|
+
"""
|
|
2027
|
+
|
|
2028
|
+
recorded = journaled.get(_REQUEST_FIELDS) if isinstance(journaled, Mapping) else None
|
|
2029
|
+
if not isinstance(recorded, Mapping):
|
|
2030
|
+
return "the request"
|
|
2031
|
+
drifted = sorted(
|
|
2032
|
+
field for field in {*recorded, *fields} if recorded.get(field) != fields.get(field)
|
|
2033
|
+
)
|
|
2034
|
+
return ", ".join(drifted) if drifted else "the request"
|
|
2035
|
+
|
|
2036
|
+
|
|
2037
|
+
def _same_plan(journaled: Any, planned: Mapping[str, Any]) -> bool:
|
|
2038
|
+
"""Whether a journaled plan is the same plan, including one journaled before field naming.
|
|
2039
|
+
|
|
2040
|
+
``request_field_digests`` is derived from the body ``request_digest`` already covers, so a plan
|
|
2041
|
+
journaled without it is held to exactly the evidence it does carry, and one that carries it is
|
|
2042
|
+
compared whole. A deployment interrupted before this field existed stays resumable rather than
|
|
2043
|
+
having a recoverable lost response turned into a refusal by an upgrade.
|
|
2044
|
+
"""
|
|
2045
|
+
|
|
2046
|
+
if not isinstance(journaled, Mapping):
|
|
2047
|
+
return False
|
|
2048
|
+
if _REQUEST_FIELDS in journaled:
|
|
2049
|
+
return dict(journaled) == dict(planned)
|
|
2050
|
+
return dict(journaled) == {
|
|
2051
|
+
field: value for field, value in planned.items() if field != _REQUEST_FIELDS
|
|
2052
|
+
}
|
|
2053
|
+
|
|
2054
|
+
|
|
2055
|
+
def _mutate(
|
|
2056
|
+
state: dict[str, Any],
|
|
2057
|
+
store: DeploymentStateStore,
|
|
2058
|
+
client: StudioDeploymentClient,
|
|
2059
|
+
*,
|
|
2060
|
+
name: str,
|
|
2061
|
+
operation: str,
|
|
2062
|
+
body: Mapping[str, Any],
|
|
2063
|
+
expected: set[int],
|
|
2064
|
+
resource_id: UUID | None = None,
|
|
2065
|
+
if_match: str | None = None,
|
|
2066
|
+
requires_etag: bool = False,
|
|
2067
|
+
) -> Mapping[str, Any]:
|
|
2068
|
+
operations = state["operations"]
|
|
2069
|
+
resources = state["resources"]
|
|
2070
|
+
request_digest = canonical.canonical_sha256(dict(body))
|
|
2071
|
+
fields = _request_field_digests(body)
|
|
2072
|
+
existing = resources.get(name)
|
|
2073
|
+
if isinstance(existing, dict):
|
|
2074
|
+
# A completed operation short-circuits to its journaled response, and the request this
|
|
2075
|
+
# invocation rebuilt is discarded rather than sent. So a local input that moved since
|
|
2076
|
+
# staging is not transmitted -- it is IGNORED, silently, and the operator is handed the
|
|
2077
|
+
# result of the request they no longer have. The journal has recorded the digest of what
|
|
2078
|
+
# was actually sent since v3 for exactly this comparison; until now nothing read it.
|
|
2079
|
+
journaled = operations.get(name)
|
|
2080
|
+
recorded = _journaled_request_digest(journaled)
|
|
2081
|
+
if recorded is None:
|
|
2082
|
+
raise HostedDeployError(
|
|
2083
|
+
"DEPLOY_STATE_INVALID",
|
|
2084
|
+
f"deployment operation {name} recorded a result without the request that "
|
|
2085
|
+
"produced it",
|
|
2086
|
+
)
|
|
2087
|
+
if recorded != request_digest:
|
|
2088
|
+
raise HostedDeployError(
|
|
2089
|
+
"DEPLOY_STATE_MISMATCH",
|
|
2090
|
+
f"deployment operation {name} already ran, and "
|
|
2091
|
+
f"{_drifted_request_fields(journaled, fields)} here no longer matches what it "
|
|
2092
|
+
"sent; this run would keep the recorded result and ignore the change, so start a "
|
|
2093
|
+
"new deployment for it",
|
|
2094
|
+
)
|
|
2095
|
+
return existing
|
|
2096
|
+
idempotency_key = _idempotency(name, state["deployment_identity"])
|
|
2097
|
+
operation_state = operations.get(name)
|
|
2098
|
+
planned = {
|
|
2099
|
+
"operation": operation,
|
|
2100
|
+
"request_digest": request_digest,
|
|
2101
|
+
_REQUEST_FIELDS: fields,
|
|
2102
|
+
"idempotency_key": idempotency_key,
|
|
2103
|
+
"status": "planned",
|
|
2104
|
+
}
|
|
2105
|
+
if operation_state is None:
|
|
2106
|
+
operations[name] = planned
|
|
2107
|
+
store.save(state)
|
|
2108
|
+
elif not _same_plan(operation_state, planned):
|
|
2109
|
+
raise HostedDeployError(
|
|
2110
|
+
"DEPLOY_STATE_MISMATCH",
|
|
2111
|
+
f"deployment operation {name} changed after journaling: "
|
|
2112
|
+
f"{_drifted_request_fields(operation_state, fields)} is not what was journaled",
|
|
2113
|
+
)
|
|
2114
|
+
response = client.call(
|
|
2115
|
+
operation,
|
|
2116
|
+
body,
|
|
2117
|
+
idempotency_key=idempotency_key,
|
|
2118
|
+
resource_id=resource_id,
|
|
2119
|
+
if_match=if_match,
|
|
2120
|
+
)
|
|
2121
|
+
result = dict(_success(response, expected, "DEPLOY_STUDIO_MUTATION_FAILED"))
|
|
2122
|
+
if requires_etag and response.etag is None:
|
|
2123
|
+
# Refuse to journal this as completed. A later step conditions its write on this
|
|
2124
|
+
# response's ETag and reads it back from the journal, never from the wire -- so
|
|
2125
|
+
# recording "completed" without one produces a deployment that cannot be finished
|
|
2126
|
+
# AND cannot be resumed, because a completed operation is never re-issued. The
|
|
2127
|
+
# proposal it already created still holds the Recipe digest, so the Build cannot be
|
|
2128
|
+
# retried either.
|
|
2129
|
+
#
|
|
2130
|
+
# Leaving the operation "planned" is what makes this recoverable: a resume re-sends
|
|
2131
|
+
# it under the same idempotency key, and a server that has learned to return the
|
|
2132
|
+
# ETag answers the replay with one. That is exactly how this was recovered when
|
|
2133
|
+
# Studio omitted it (mostlyrightmd/mostlyright-studio#57).
|
|
2134
|
+
raise HostedDeployError(
|
|
2135
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID",
|
|
2136
|
+
f"Studio omitted the ETag after {name}, which the next step must write against; "
|
|
2137
|
+
"nothing was journaled as complete, so rerunning with --resume will ask again",
|
|
2138
|
+
)
|
|
2139
|
+
resources[name] = result
|
|
2140
|
+
if response.etag is not None:
|
|
2141
|
+
state.setdefault("etags", {})[name] = response.etag
|
|
2142
|
+
operations[name] = {**planned, "status": "completed"}
|
|
2143
|
+
store.save(state)
|
|
2144
|
+
return result
|
|
2145
|
+
|
|
2146
|
+
|
|
2147
|
+
def _upload_once(
|
|
2148
|
+
state: dict[str, Any],
|
|
2149
|
+
store: DeploymentStateStore,
|
|
2150
|
+
client: StudioDeploymentClient,
|
|
2151
|
+
*,
|
|
2152
|
+
name: str,
|
|
2153
|
+
session: Mapping[str, Any],
|
|
2154
|
+
content: bytes,
|
|
2155
|
+
) -> str | None:
|
|
2156
|
+
operations = state["operations"]
|
|
2157
|
+
completed = operations.get(name, {})
|
|
2158
|
+
status = completed.get("status")
|
|
2159
|
+
content_digest = hashlib.sha256(content).hexdigest()
|
|
2160
|
+
if status in {"completed", "reconciled"}:
|
|
2161
|
+
# Both of these short-circuit an object Studio has already sealed, and neither one sends
|
|
2162
|
+
# the bytes this invocation just read. Local source evidence that moved since staging is
|
|
2163
|
+
# therefore ignored rather than uploaded, so it is compared against the digest the journal
|
|
2164
|
+
# recorded when the transfer actually happened.
|
|
2165
|
+
recorded = _journaled_request_digest(completed)
|
|
2166
|
+
if recorded is None:
|
|
2167
|
+
raise HostedDeployError(
|
|
2168
|
+
"DEPLOY_STATE_INVALID",
|
|
2169
|
+
f"source upload {name} recorded a result without the content it sent",
|
|
2170
|
+
)
|
|
2171
|
+
if recorded != content_digest:
|
|
2172
|
+
raise HostedDeployError(
|
|
2173
|
+
"DEPLOY_STATE_MISMATCH",
|
|
2174
|
+
f"source upload {name} already ran, and the content here no longer matches what "
|
|
2175
|
+
"it sent; this run would keep the sealed object and ignore the change, so start a "
|
|
2176
|
+
"new deployment for it",
|
|
2177
|
+
)
|
|
2178
|
+
if status == "reconciled":
|
|
2179
|
+
return None
|
|
2180
|
+
generation = completed.get("object_generation")
|
|
2181
|
+
if not isinstance(generation, str) or _OBJECT_GENERATION.fullmatch(generation) is None:
|
|
2182
|
+
raise HostedDeployError(
|
|
2183
|
+
"DEPLOY_STATE_INVALID", f"source upload {name} has no object generation"
|
|
2184
|
+
)
|
|
2185
|
+
return generation
|
|
2186
|
+
planned = {
|
|
2187
|
+
"operation": "signed_upload",
|
|
2188
|
+
"request_digest": content_digest,
|
|
2189
|
+
"session_id": session.get("session_id"),
|
|
2190
|
+
"status": "planned",
|
|
2191
|
+
}
|
|
2192
|
+
current = operations.get(name)
|
|
2193
|
+
if current is None:
|
|
2194
|
+
operations[name] = planned
|
|
2195
|
+
store.save(state)
|
|
2196
|
+
elif current != planned:
|
|
2197
|
+
raise HostedDeployError(
|
|
2198
|
+
"DEPLOY_STATE_MISMATCH", f"source upload {name} changed after journaling"
|
|
2199
|
+
)
|
|
2200
|
+
try:
|
|
2201
|
+
generation = client.upload(session, content)
|
|
2202
|
+
except HostedDeployError as error:
|
|
2203
|
+
if error.code != "DEPLOY_SOURCE_UPLOAD_FAILED":
|
|
2204
|
+
raise
|
|
2205
|
+
# A transfer can commit remotely while its response is lost. Leave the upload planned;
|
|
2206
|
+
# finalization or the next idempotent replay will reconcile the exact sealed object.
|
|
2207
|
+
return None
|
|
2208
|
+
operations[name] = {**planned, "status": "completed", "object_generation": generation}
|
|
2209
|
+
store.save(state)
|
|
2210
|
+
return generation
|
|
2211
|
+
|
|
2212
|
+
|
|
2213
|
+
def _approval_receipt(
|
|
2214
|
+
state: Mapping[str, Any], store: DeploymentStateStore, cloud_url: str
|
|
2215
|
+
) -> dict[str, Any]:
|
|
2216
|
+
proposal = state["resources"]["proposal"]
|
|
2217
|
+
return {
|
|
2218
|
+
"schema_version": DEPLOYMENT_RECEIPT_SCHEMA,
|
|
2219
|
+
"status": "deployment_approval_required",
|
|
2220
|
+
"workspace_id": state["workspace_id"],
|
|
2221
|
+
"recipe_proposal_id": proposal["recipe_proposal_id"],
|
|
2222
|
+
"approval_request_id": state["resources"]["approval_request_id"],
|
|
2223
|
+
"state": str(store.path),
|
|
2224
|
+
"dashboard_url": (
|
|
2225
|
+
f"{_service_url(cloud_url, 'Cloud URL', allow_loopback_http=False)}/dashboard/approvals"
|
|
2226
|
+
),
|
|
2227
|
+
"next": (
|
|
2228
|
+
"approve the exact Recipe in Cloud Approvals, then rerun this command with --resume"
|
|
2229
|
+
),
|
|
2230
|
+
}
|
|
2231
|
+
|
|
2232
|
+
|
|
2233
|
+
def _activation_receipt(
|
|
2234
|
+
state: Mapping[str, Any], store: DeploymentStateStore, cloud_url: str
|
|
2235
|
+
) -> dict[str, Any]:
|
|
2236
|
+
activation = state["resources"]["activation"]
|
|
2237
|
+
binding = state["resources"]["cloud_binding"]
|
|
2238
|
+
table_id = _uuid_field(activation, "table_id")
|
|
2239
|
+
run_id = _uuid_field(activation, "run_id")
|
|
2240
|
+
return {
|
|
2241
|
+
"schema_version": DEPLOYMENT_RECEIPT_SCHEMA,
|
|
2242
|
+
"status": "dataset_activation_queued",
|
|
2243
|
+
"workspace_id": state["workspace_id"],
|
|
2244
|
+
"dataset_id": state["resources"]["dataset_id"],
|
|
2245
|
+
"table_plan_id": state["resources"]["table_plan_id"],
|
|
2246
|
+
"table_id": str(table_id),
|
|
2247
|
+
"cloud_dataset_id": _required_text(binding, "cloud_dataset_id"),
|
|
2248
|
+
"current_path": _required_text(binding, "current_path"),
|
|
2249
|
+
"run_id": str(run_id),
|
|
2250
|
+
# The reviewed schedule is the coordinate that says when this Dataset refreshes itself,
|
|
2251
|
+
# and it is the one activation coordinate the receipt used to omit -- so the operator who
|
|
2252
|
+
# just deployed had no way to read back the cadence Studio activated without reopening
|
|
2253
|
+
# the journal. It is the exact object the activation body carried, held in ``state`` from
|
|
2254
|
+
# the moment the deployment was initialized.
|
|
2255
|
+
"schedule": dict(state["schedule"]),
|
|
2256
|
+
"recipe_proposal_id": state["resources"]["proposal"]["recipe_proposal_id"],
|
|
2257
|
+
"state": str(store.path),
|
|
2258
|
+
"dashboard_url": (
|
|
2259
|
+
f"{_service_url(cloud_url, 'Cloud URL', allow_loopback_http=False)}/dashboard/datasets"
|
|
2260
|
+
),
|
|
2261
|
+
# The queued Run is not a released version, and this receipt is where somebody learns that
|
|
2262
|
+
# for the first time. Its sibling above has always ended by naming the next command; this
|
|
2263
|
+
# one ended at a dashboard URL, which is exactly the trip to Cloud the status command now
|
|
2264
|
+
# removes.
|
|
2265
|
+
"next": "run mr-data deploy-status RUN_DIR to follow the run this queued",
|
|
2266
|
+
}
|
|
2267
|
+
|
|
2268
|
+
|
|
2269
|
+
def cloud_dashboard_url(cloud_url: str, page: str) -> str:
|
|
2270
|
+
"""The Cloud page an operator opens about this deployment, from one validated origin.
|
|
2271
|
+
|
|
2272
|
+
Every receipt that points somebody at Cloud goes through here, so the origin is checked the
|
|
2273
|
+
same way each time and the three surfaces cannot drift into three spellings of one URL.
|
|
2274
|
+
"""
|
|
2275
|
+
|
|
2276
|
+
origin = _service_url(cloud_url, "Cloud URL", allow_loopback_http=False)
|
|
2277
|
+
return f"{origin}/dashboard/workspace/{page}"
|
|
2278
|
+
|
|
2279
|
+
|
|
2280
|
+
def _bind_cloud_dataset(
|
|
2281
|
+
request: DeploymentRequest,
|
|
2282
|
+
state: dict[str, Any],
|
|
2283
|
+
store: DeploymentStateStore,
|
|
2284
|
+
*,
|
|
2285
|
+
credentials: ResolvedCloudCredentials,
|
|
2286
|
+
transport: TokenExchangeTransport | None,
|
|
2287
|
+
) -> Mapping[str, Any]:
|
|
2288
|
+
_require_cli_credential(credentials)
|
|
2289
|
+
resources = state["resources"]
|
|
2290
|
+
activation = resources.get("activation")
|
|
2291
|
+
if not isinstance(activation, dict):
|
|
2292
|
+
raise HostedDeployError(
|
|
2293
|
+
"DEPLOY_STATE_INVALID", "Cloud Dataset binding requires a completed activation"
|
|
2294
|
+
)
|
|
2295
|
+
body = {
|
|
2296
|
+
"schema_version": CLOUD_TABLE_BINDING_SCHEMA_VERSION,
|
|
2297
|
+
"studio_workspace_id": state["workspace_id"],
|
|
2298
|
+
"studio_dataset_id": _required_text(resources, "dataset_id"),
|
|
2299
|
+
"studio_table_id": str(_uuid_field(activation, "table_id")),
|
|
2300
|
+
"table_recipe": {
|
|
2301
|
+
"table_recipe_id": request.recipe_id,
|
|
2302
|
+
"recipe_version": request.recipe_version,
|
|
2303
|
+
"recipe_digest": request.recipe_digest,
|
|
2304
|
+
},
|
|
2305
|
+
}
|
|
2306
|
+
request_digest = canonical.canonical_sha256(body)
|
|
2307
|
+
fields = _request_field_digests(body)
|
|
2308
|
+
operations = state["operations"]
|
|
2309
|
+
existing = resources.get("cloud_binding")
|
|
2310
|
+
if isinstance(existing, dict):
|
|
2311
|
+
# The binding Cloud already persisted is returned without contacting Cloud again, so a
|
|
2312
|
+
# rebuilt body that moved is ignored rather than sent. The body is built before this
|
|
2313
|
+
# short-circuit for that comparison alone.
|
|
2314
|
+
journaled = operations.get("cloud_binding")
|
|
2315
|
+
recorded = _journaled_request_digest(journaled)
|
|
2316
|
+
if recorded is None:
|
|
2317
|
+
raise HostedDeployError(
|
|
2318
|
+
"DEPLOY_STATE_INVALID",
|
|
2319
|
+
"the Cloud Dataset binding recorded a result without the request that produced it",
|
|
2320
|
+
)
|
|
2321
|
+
if recorded != request_digest:
|
|
2322
|
+
raise HostedDeployError(
|
|
2323
|
+
"DEPLOY_STATE_MISMATCH",
|
|
2324
|
+
"the Cloud Dataset binding already ran, and "
|
|
2325
|
+
f"{_drifted_request_fields(journaled, fields)} here no longer matches what it "
|
|
2326
|
+
"sent; this run would keep the recorded binding and ignore the change, so start a "
|
|
2327
|
+
"new deployment for it",
|
|
2328
|
+
)
|
|
2329
|
+
return existing
|
|
2330
|
+
planned = {
|
|
2331
|
+
"operation": "bind_cloud_dataset",
|
|
2332
|
+
"request_digest": request_digest,
|
|
2333
|
+
_REQUEST_FIELDS: fields,
|
|
2334
|
+
"status": "planned",
|
|
2335
|
+
}
|
|
2336
|
+
current = operations.get("cloud_binding")
|
|
2337
|
+
if current is None:
|
|
2338
|
+
operations["cloud_binding"] = planned
|
|
2339
|
+
store.save(state)
|
|
2340
|
+
elif not _same_plan(current, planned):
|
|
2341
|
+
raise HostedDeployError(
|
|
2342
|
+
"DEPLOY_STATE_MISMATCH",
|
|
2343
|
+
"the Cloud Dataset binding changed after journaling: "
|
|
2344
|
+
f"{_drifted_request_fields(current, fields)} is not what was journaled",
|
|
2345
|
+
)
|
|
2346
|
+
selected = transport or UrlLibTokenExchangeTransport()
|
|
2347
|
+
cloud_url = _service_url(credentials.cloud_url, "Cloud URL", allow_loopback_http=False)
|
|
2348
|
+
try:
|
|
2349
|
+
status, raw, _response_headers = selected.request(
|
|
2350
|
+
"POST",
|
|
2351
|
+
f"{cloud_url}{CLOUD_DATASET_BINDINGS_PATH}",
|
|
2352
|
+
{
|
|
2353
|
+
"Accept": "application/json",
|
|
2354
|
+
"Content-Type": "application/json",
|
|
2355
|
+
"x-api-key": credentials.raw_key,
|
|
2356
|
+
"Idempotency-Key": _idempotency(
|
|
2357
|
+
"cloud-table-binding", state["deployment_identity"]
|
|
2358
|
+
),
|
|
2359
|
+
},
|
|
2360
|
+
canonical.canonical_json_bytes(body),
|
|
2361
|
+
MAX_BINDING_RESPONSE_BYTES,
|
|
2362
|
+
)
|
|
2363
|
+
except HostedDeployError:
|
|
2364
|
+
raise
|
|
2365
|
+
except Exception as error:
|
|
2366
|
+
raise HostedDeployError(
|
|
2367
|
+
"DEPLOY_CLOUD_BINDING_FAILED",
|
|
2368
|
+
"the Cloud Dataset binding could not be reached; rerun with --resume",
|
|
2369
|
+
) from error
|
|
2370
|
+
try:
|
|
2371
|
+
parsed = canonical.parse_json(raw)
|
|
2372
|
+
except canonical.CanonicalJSONError:
|
|
2373
|
+
parsed = None
|
|
2374
|
+
if status == 401:
|
|
2375
|
+
raise HostedDeployError(
|
|
2376
|
+
"DEPLOY_AUTHENTICATION_FAILED", "the selected CLI credential was rejected"
|
|
2377
|
+
)
|
|
2378
|
+
if status == 402:
|
|
2379
|
+
raise HostedDeployError(
|
|
2380
|
+
"DEPLOY_SUBSCRIPTION_REQUIRED", "this workspace has no hosted deployment access"
|
|
2381
|
+
)
|
|
2382
|
+
if status == 409:
|
|
2383
|
+
raise HostedDeployError(
|
|
2384
|
+
"DEPLOY_CLOUD_BINDING_CONFLICT",
|
|
2385
|
+
"Cloud could not bind the exact accepted Studio activation",
|
|
2386
|
+
)
|
|
2387
|
+
if status in {429, 503}:
|
|
2388
|
+
raise HostedDeployError(
|
|
2389
|
+
"DEPLOY_CLOUD_BINDING_UNAVAILABLE",
|
|
2390
|
+
"Cloud could not finish Dataset binding; rerun with --resume",
|
|
2391
|
+
)
|
|
2392
|
+
if status != 200 or not isinstance(parsed, dict):
|
|
2393
|
+
raise HostedDeployError(
|
|
2394
|
+
"DEPLOY_CLOUD_BINDING_FAILED",
|
|
2395
|
+
f"the Cloud Dataset binding failed (HTTP {status})",
|
|
2396
|
+
)
|
|
2397
|
+
if set(parsed) != {
|
|
2398
|
+
"schema_version",
|
|
2399
|
+
"cloud_dataset_id",
|
|
2400
|
+
"cloud_table_id",
|
|
2401
|
+
"studio_dataset_id",
|
|
2402
|
+
"studio_table_id",
|
|
2403
|
+
"current_path",
|
|
2404
|
+
}:
|
|
2405
|
+
raise HostedDeployError(
|
|
2406
|
+
"DEPLOY_CLOUD_BINDING_FAILED", "Cloud returned an invalid Dataset binding"
|
|
2407
|
+
)
|
|
2408
|
+
if parsed["schema_version"] != CLOUD_TABLE_BINDING_SCHEMA_VERSION:
|
|
2409
|
+
raise HostedDeployError(
|
|
2410
|
+
"DEPLOY_CLOUD_BINDING_FAILED", "Cloud returned an invalid binding schema"
|
|
2411
|
+
)
|
|
2412
|
+
cloud_dataset_id = _uuid(_required_text(parsed, "cloud_dataset_id"), "cloud_dataset_id")
|
|
2413
|
+
cloud_table_id = _uuid(_required_text(parsed, "cloud_table_id"), "cloud_table_id")
|
|
2414
|
+
studio_dataset_id = _uuid(_required_text(parsed, "studio_dataset_id"), "studio_dataset_id")
|
|
2415
|
+
studio_table_id = _uuid(_required_text(parsed, "studio_table_id"), "studio_table_id")
|
|
2416
|
+
expected_studio_dataset_id = _uuid(_required_text(resources, "dataset_id"), "dataset_id")
|
|
2417
|
+
expected_studio_table_id = _uuid_field(activation, "table_id")
|
|
2418
|
+
current_path = _required_text(parsed, "current_path")
|
|
2419
|
+
if (
|
|
2420
|
+
studio_dataset_id != expected_studio_dataset_id
|
|
2421
|
+
or studio_table_id != expected_studio_table_id
|
|
2422
|
+
or current_path != f"/api/v2/tables/{cloud_table_id}/current"
|
|
2423
|
+
):
|
|
2424
|
+
raise HostedDeployError(
|
|
2425
|
+
"DEPLOY_CLOUD_BINDING_FAILED", "Cloud returned a mismatched Dataset binding"
|
|
2426
|
+
)
|
|
2427
|
+
result = {
|
|
2428
|
+
"cloud_dataset_id": str(cloud_dataset_id),
|
|
2429
|
+
"cloud_table_id": str(cloud_table_id),
|
|
2430
|
+
"studio_dataset_id": str(studio_dataset_id),
|
|
2431
|
+
"studio_table_id": str(studio_table_id),
|
|
2432
|
+
"current_path": current_path,
|
|
2433
|
+
}
|
|
2434
|
+
resources["cloud_binding"] = result
|
|
2435
|
+
operations["cloud_binding"] = {**planned, "status": "completed"}
|
|
2436
|
+
store.save(state)
|
|
2437
|
+
return result
|
|
2438
|
+
|
|
2439
|
+
|
|
2440
|
+
def _deployment_predecessor(
|
|
2441
|
+
request: DeploymentRequest, previous_run_dir: Path | None
|
|
2442
|
+
) -> str | None:
|
|
2443
|
+
"""Settle which live Dataset this deployment updates, before it writes or sends anything of
|
|
2444
|
+
its own.
|
|
2445
|
+
|
|
2446
|
+
``mr-data deploy`` has one predecessor, not two. ``--previous-run-dir`` names the run directory
|
|
2447
|
+
whose deployment published the Dataset this version updates, and the same directory is what
|
|
2448
|
+
:func:`recipe.verify_recipe_candidate` authenticates the version chain out of. The two layers
|
|
2449
|
+
ask one question because :func:`deploy.build_deployment_request` refuses both shapes where they
|
|
2450
|
+
could differ: a Recipe execution whose mode is not ``initial`` must name its predecessor, and
|
|
2451
|
+
one whose mode is ``initial`` must not. So by the time a request reaches here, "a predecessor
|
|
2452
|
+
was named" and "this Build follows a sealed one" are the same fact, and that -- not the Recipe
|
|
2453
|
+
version -- is what :func:`_prepare_and_request_approval` branches on. A refresh of an unchanged
|
|
2454
|
+
Recipe is a successor whose Recipe version has not moved.
|
|
2455
|
+
|
|
2456
|
+
What the Recipe version still decides is the one shape it alone can see: a successor Recipe
|
|
2457
|
+
exists because a Dataset it follows exists, so deploying one that names no predecessor at all
|
|
2458
|
+
would found a second Dataset and strand the first. That is refused here rather than at the
|
|
2459
|
+
Studio call, so it leaves no journal behind -- ``DEPLOY_PREDECESSOR_REQUIRED`` names a flag,
|
|
2460
|
+
and the rerun that supplies it has to be able to stage cleanly instead of meeting the identity
|
|
2461
|
+
gate holding the journal of the run that refused.
|
|
2462
|
+
|
|
2463
|
+
Returns:
|
|
2464
|
+
The predecessor coordinate this deployment's identity is computed from, or None when there
|
|
2465
|
+
is no predecessor.
|
|
2466
|
+
"""
|
|
2467
|
+
|
|
2468
|
+
if previous_run_dir is None:
|
|
2469
|
+
if request.recipe_version > 1:
|
|
2470
|
+
# `_previous_deployment_resources` owns this sentence; calling it with nothing is what
|
|
2471
|
+
# moves its refusal ahead of the journal this run would otherwise have written.
|
|
2472
|
+
_previous_deployment_resources(previous_run_dir)
|
|
2473
|
+
return None
|
|
2474
|
+
# Resolved, because a resume is run from whatever directory the operator happens to be standing
|
|
2475
|
+
# in and the predecessor it names is the same one either way.
|
|
2476
|
+
coordinate = str(Path(previous_run_dir).resolve())
|
|
2477
|
+
# Reads the predecessor's own activation receipt, so one that never finished activating is
|
|
2478
|
+
# refused here rather than at the Studio call. This does open a journal store on the
|
|
2479
|
+
# predecessor, which creates its `evidence/deployment` directory and takes a lock there; what
|
|
2480
|
+
# is not yet written is anything belonging to the deployment being started.
|
|
2481
|
+
_previous_deployment_resources(previous_run_dir)
|
|
2482
|
+
return coordinate
|
|
2483
|
+
|
|
2484
|
+
|
|
2485
|
+
def _previous_deployment_resources(previous_run_dir: Path | None) -> Mapping[str, str]:
|
|
2486
|
+
if previous_run_dir is None:
|
|
2487
|
+
raise HostedDeployError(
|
|
2488
|
+
"DEPLOY_PREDECESSOR_REQUIRED", "a successor deployment needs --previous-run-dir"
|
|
2489
|
+
)
|
|
2490
|
+
store = FileDeploymentStateStore(previous_run_dir)
|
|
2491
|
+
try:
|
|
2492
|
+
state = store.load()
|
|
2493
|
+
finally:
|
|
2494
|
+
store.close()
|
|
2495
|
+
resources = state.get("resources")
|
|
2496
|
+
activation = resources.get("activation") if isinstance(resources, dict) else None
|
|
2497
|
+
if not isinstance(resources, dict) or not isinstance(activation, dict):
|
|
2498
|
+
raise HostedDeployError(
|
|
2499
|
+
"DEPLOY_PREDECESSOR_INVALID", "the previous Build has no completed activation receipt"
|
|
2500
|
+
)
|
|
2501
|
+
return {
|
|
2502
|
+
"dataset_id": _required_text(resources, "dataset_id"),
|
|
2503
|
+
"table_plan_id": _required_text(resources, "table_plan_id"),
|
|
2504
|
+
"table_id": str(_uuid_field(activation, "table_id")),
|
|
2505
|
+
}
|
|
2506
|
+
|
|
2507
|
+
|
|
2508
|
+
def _source_locator(
|
|
2509
|
+
source: Mapping[str, Any], proposal: Mapping[str, Any], bundle: Mapping[str, Any]
|
|
2510
|
+
) -> dict[str, Any]:
|
|
2511
|
+
if bundle["authority_kind"] == "sealed_artifact":
|
|
2512
|
+
return {
|
|
2513
|
+
"kind": "artifact",
|
|
2514
|
+
"display_locator": "urn:sha256:" + bundle["source_data"]["content_sha256"],
|
|
2515
|
+
}
|
|
2516
|
+
if source["adapter_id"] == "public.https":
|
|
2517
|
+
return {
|
|
2518
|
+
"kind": "https_url",
|
|
2519
|
+
"display_locator": source["query_template"]["url"],
|
|
2520
|
+
"connector_contract_version": source["adapter_version"],
|
|
2521
|
+
}
|
|
2522
|
+
return {
|
|
2523
|
+
"kind": "sdk_connector",
|
|
2524
|
+
"display_locator": proposal["locator"],
|
|
2525
|
+
"connector_contract_version": source["adapter_version"],
|
|
2526
|
+
}
|
|
2527
|
+
|
|
2528
|
+
|
|
2529
|
+
def _studio_feasibility(value: Mapping[str, Any]) -> dict[str, Any]:
|
|
2530
|
+
mapping = {
|
|
2531
|
+
"sufficient_evidence": "SUFFICIENT_EVIDENCE",
|
|
2532
|
+
"limited_coverage": "LIMITED_COVERAGE",
|
|
2533
|
+
"unacceptable_delay": "UNACCEPTABLE_DELAY",
|
|
2534
|
+
"rights_unclear": "RIGHTS_UNCLEAR",
|
|
2535
|
+
"no_reliable_source": "NO_RELIABLE_SOURCE",
|
|
2536
|
+
"target_not_observable": "TARGET_NOT_OBSERVABLE",
|
|
2537
|
+
}
|
|
2538
|
+
return {
|
|
2539
|
+
"decision": value["decision"],
|
|
2540
|
+
"reason_codes": [mapping[item] for item in value["reason_codes"]],
|
|
2541
|
+
"narrative": value["narrative"],
|
|
2542
|
+
}
|
|
2543
|
+
|
|
2544
|
+
|
|
2545
|
+
def _default_backfill_policy(time_range: Mapping[str, Any]) -> dict[str, Any]:
|
|
2546
|
+
start = _timestamp(_required_text(time_range, "start_inclusive"), "start_inclusive")
|
|
2547
|
+
end = _timestamp(_required_text(time_range, "end_exclusive"), "end_exclusive")
|
|
2548
|
+
seconds = int((end - start).total_seconds())
|
|
2549
|
+
if seconds < 1:
|
|
2550
|
+
raise HostedDeployError("DEPLOY_REQUEST_INVALID", "Recipe time range is empty")
|
|
2551
|
+
return {
|
|
2552
|
+
"schema_version": "local-backfill-policy.v1",
|
|
2553
|
+
"coverage_envelope": dict(time_range),
|
|
2554
|
+
"max_window_seconds": seconds,
|
|
2555
|
+
"overlap_policy": "disallow_released_coverage",
|
|
2556
|
+
"partition_scope": {"fields": [], "allowed_values": {}, "max_partitions_per_run": 1},
|
|
2557
|
+
}
|
|
2558
|
+
|
|
2559
|
+
|
|
2560
|
+
#: Every operator-visible input the deployment identity is computed from, and what to call it when
|
|
2561
|
+
#: it moves. A resume is by definition something a person comes back to hours later, after an
|
|
2562
|
+
#: approval, with the original command long gone from their scrollback, so getting one of these
|
|
2563
|
+
#: wrong is the expected case rather than the exceptional one.
|
|
2564
|
+
#:
|
|
2565
|
+
#: Two of them are not arguments at all. The Cloud URL comes from the environment, and the review
|
|
2566
|
+
#: authority comes from whether this shell is inside a coordinator review session -- which is why a
|
|
2567
|
+
#: refusal that names the review authority must not offer an argument list as the fix.
|
|
2568
|
+
_IDENTITY_INPUT_LABELS: dict[str, str] = {
|
|
2569
|
+
"dataset_name": "--dataset-name",
|
|
2570
|
+
"cron": "--cron",
|
|
2571
|
+
"previous_run_dir": "--previous-run-dir",
|
|
2572
|
+
"max_runs_per_month": "--max-runs-per-month",
|
|
2573
|
+
"max_concurrent_runs": "--max-concurrent-runs",
|
|
2574
|
+
"max_monthly_cost_units": "--max-monthly-cost-units",
|
|
2575
|
+
"recipe_id": "the Recipe id",
|
|
2576
|
+
"recipe_version": "the Recipe version",
|
|
2577
|
+
"recipe_digest": "the Recipe",
|
|
2578
|
+
"candidate_digest": "the reviewed Build",
|
|
2579
|
+
"cloud_url": "the Cloud URL",
|
|
2580
|
+
"workspace_id": "the workspace",
|
|
2581
|
+
"review_authority": "the review authority",
|
|
2582
|
+
}
|
|
2583
|
+
|
|
2584
|
+
#: The journal key under which a staged deployment records the inputs above.
|
|
2585
|
+
_IDENTITY_INPUTS = "identity_inputs"
|
|
2586
|
+
|
|
2587
|
+
#: Which authority the staged request carried. A request carries the signed reviewed-Build binding
|
|
2588
|
+
#: or it carries none, and which one it is moves the request digest; nothing on the command line
|
|
2589
|
+
#: says so.
|
|
2590
|
+
_REVIEW_SESSION = "review_session"
|
|
2591
|
+
_UNATTESTED = "unattested"
|
|
2592
|
+
|
|
2593
|
+
#: What can still move a request digest once every named input matches. None of it is an argument,
|
|
2594
|
+
#: which is what makes listing it honest rather than unhelpful: it tells an operator that no rerun
|
|
2595
|
+
#: of the same command will converge.
|
|
2596
|
+
_UNNAMED_REQUEST_INPUTS = (
|
|
2597
|
+
"the sealed Build evidence the request carries",
|
|
2598
|
+
"the release of this command that staged it",
|
|
2599
|
+
)
|
|
2600
|
+
|
|
2601
|
+
|
|
2602
|
+
def _identity_inputs(
|
|
2603
|
+
request: DeploymentRequest,
|
|
2604
|
+
*,
|
|
2605
|
+
workspace_id: UUID,
|
|
2606
|
+
cloud_url: str,
|
|
2607
|
+
cron: str,
|
|
2608
|
+
max_concurrent_runs: int,
|
|
2609
|
+
previous_run_dir: str | None,
|
|
2610
|
+
) -> dict[str, Any]:
|
|
2611
|
+
"""The named inputs behind one deployment identity, in the shape the journal records them.
|
|
2612
|
+
|
|
2613
|
+
``previous_run_dir`` is recorded on every deployment, including the initial ones where it is
|
|
2614
|
+
None. The identity digest carries it only when there is one -- an absent key there would move
|
|
2615
|
+
every identity staged before this release -- but the journal has no such constraint, and
|
|
2616
|
+
recording the absence is what lets a resume that adds a predecessor be named as the input that
|
|
2617
|
+
moved rather than listed as one of the inputs the journal could not resolve.
|
|
2618
|
+
"""
|
|
2619
|
+
|
|
2620
|
+
return {
|
|
2621
|
+
"candidate_digest": request.candidate_digest,
|
|
2622
|
+
"cloud_url": cloud_url,
|
|
2623
|
+
"cron": cron,
|
|
2624
|
+
"table_name": request.dataset_name,
|
|
2625
|
+
"max_concurrent_runs": max_concurrent_runs,
|
|
2626
|
+
"max_monthly_cost_units": request.caps.max_monthly_cost_units,
|
|
2627
|
+
"max_runs_per_month": request.caps.max_runs_per_month,
|
|
2628
|
+
"previous_run_dir": previous_run_dir,
|
|
2629
|
+
"recipe_digest": request.recipe_digest,
|
|
2630
|
+
"table_recipe_id": request.recipe_id,
|
|
2631
|
+
"recipe_version": request.recipe_version,
|
|
2632
|
+
"review_authority": (
|
|
2633
|
+
_REVIEW_SESSION if request.review_decision_digest is not None else _UNATTESTED
|
|
2634
|
+
),
|
|
2635
|
+
"workspace_id": str(workspace_id),
|
|
2636
|
+
}
|
|
2637
|
+
|
|
2638
|
+
|
|
2639
|
+
def _journaled_identity_inputs(state: Mapping[str, Any]) -> dict[str, Any]:
|
|
2640
|
+
"""What this journal can say about the inputs it was staged with.
|
|
2641
|
+
|
|
2642
|
+
A journal written since staging began recording them answers for every input. An older one
|
|
2643
|
+
answers only for what it happened to need: the schedule it sent to Studio, and -- once Studio
|
|
2644
|
+
held a Recipe proposal -- the Dataset name inside the activation intent. Whatever it cannot
|
|
2645
|
+
answer is left out here, so a refusal lists it as a possibility rather than saying it matched.
|
|
2646
|
+
"""
|
|
2647
|
+
|
|
2648
|
+
recorded = state.get(_IDENTITY_INPUTS)
|
|
2649
|
+
if isinstance(recorded, Mapping):
|
|
2650
|
+
return {name: recorded[name] for name in _IDENTITY_INPUT_LABELS if name in recorded}
|
|
2651
|
+
inputs: dict[str, Any] = {}
|
|
2652
|
+
schedule = state.get("schedule")
|
|
2653
|
+
if isinstance(schedule, Mapping):
|
|
2654
|
+
for name in ("cron", "max_concurrent_runs"):
|
|
2655
|
+
if name in schedule:
|
|
2656
|
+
inputs[name] = schedule[name]
|
|
2657
|
+
resources = state.get("resources")
|
|
2658
|
+
proposal = resources.get("proposal") if isinstance(resources, Mapping) else None
|
|
2659
|
+
intent = proposal.get("activation_intent") if isinstance(proposal, Mapping) else None
|
|
2660
|
+
if isinstance(intent, Mapping) and "dataset_name" in intent:
|
|
2661
|
+
inputs["table_name"] = intent["table_name"]
|
|
2662
|
+
return inputs
|
|
2663
|
+
|
|
2664
|
+
|
|
2665
|
+
def _drifted_identity_inputs(journaled: Mapping[str, Any], inputs: Mapping[str, Any]) -> list[str]:
|
|
2666
|
+
"""The named inputs this journal can prove moved, in a stable order."""
|
|
2667
|
+
|
|
2668
|
+
return sorted(name for name, value in journaled.items() if inputs.get(name) != value)
|
|
2669
|
+
|
|
2670
|
+
|
|
2671
|
+
def _identity_drift_sentence(
|
|
2672
|
+
journaled: Mapping[str, Any], inputs: Mapping[str, Any], names: list[str]
|
|
2673
|
+
) -> str:
|
|
2674
|
+
"""Each differing input, as the value it was staged with and the value it has here."""
|
|
2675
|
+
|
|
2676
|
+
return "; ".join(
|
|
2677
|
+
f"{_IDENTITY_INPUT_LABELS[name]} was staged as {journaled[name]} and is "
|
|
2678
|
+
f"{inputs.get(name)} here"
|
|
2679
|
+
for name in names
|
|
2680
|
+
)
|
|
2681
|
+
|
|
2682
|
+
|
|
2683
|
+
def _validate_state(
|
|
2684
|
+
state: Mapping[str, Any],
|
|
2685
|
+
identity: str,
|
|
2686
|
+
workspace_id: UUID,
|
|
2687
|
+
*,
|
|
2688
|
+
inputs: Mapping[str, Any],
|
|
2689
|
+
request_digest: str,
|
|
2690
|
+
) -> None:
|
|
2691
|
+
"""Hold one journal to one exact handoff, and say which input moved when it does not.
|
|
2692
|
+
|
|
2693
|
+
The identity comparison is unchanged: one digest over the request, the workspace, the Cloud
|
|
2694
|
+
URL, the schedule and the concurrency cap, and it either names this handoff or it does not.
|
|
2695
|
+
What changes is what a person is told when it does not. The journal is holding the answer --
|
|
2696
|
+
it recorded the inputs it was staged with -- and until now the refusal said only that the two
|
|
2697
|
+
disagreed, which is the one thing the operator already knew.
|
|
2698
|
+
"""
|
|
2699
|
+
|
|
2700
|
+
if (
|
|
2701
|
+
state.get("schema_version") != DEPLOYMENT_STATE_SCHEMA
|
|
2702
|
+
or not isinstance(state.get("operations"), dict)
|
|
2703
|
+
or not isinstance(state.get("resources"), dict)
|
|
2704
|
+
):
|
|
2705
|
+
raise HostedDeployError(
|
|
2706
|
+
"DEPLOY_STATE_MISMATCH", "the deployment journal belongs to another exact handoff"
|
|
2707
|
+
)
|
|
2708
|
+
staged_workspace = state.get("workspace_id")
|
|
2709
|
+
if staged_workspace != str(workspace_id):
|
|
2710
|
+
raise HostedDeployError(
|
|
2711
|
+
"DEPLOY_STATE_MISMATCH",
|
|
2712
|
+
f"this deployment journal was staged in workspace {staged_workspace} and this run is "
|
|
2713
|
+
f"authenticated to workspace {workspace_id}; sign in to the one it was staged in, or "
|
|
2714
|
+
"start a new deployment",
|
|
2715
|
+
)
|
|
2716
|
+
if state.get("deployment_identity") == identity:
|
|
2717
|
+
return
|
|
2718
|
+
journaled = _journaled_identity_inputs(state)
|
|
2719
|
+
drifted = _drifted_identity_inputs(journaled, inputs)
|
|
2720
|
+
if "review_authority" in drifted:
|
|
2721
|
+
# The sixth input, and the only one that is invisible from the command line: `mr-data
|
|
2722
|
+
# deploy` reads the review authority off the ambient shell, so a session-staged deployment
|
|
2723
|
+
# resumed from an ordinary terminal rebuilds an unattested request, moves the digest, and
|
|
2724
|
+
# lands here with no differing flag to name. Printing an argument list for this one would
|
|
2725
|
+
# send an operator to retype a command that already matches.
|
|
2726
|
+
if journaled["review_authority"] == _REVIEW_SESSION:
|
|
2727
|
+
raise HostedDeployError(
|
|
2728
|
+
"DEPLOY_STATE_MISMATCH",
|
|
2729
|
+
"this deployment was staged inside a coordinator review session and this run has "
|
|
2730
|
+
"none, so it rebuilt a different request; no argument list can bridge that -- "
|
|
2731
|
+
"resume it from a review session, or start a new deployment",
|
|
2732
|
+
)
|
|
2733
|
+
raise HostedDeployError(
|
|
2734
|
+
"DEPLOY_STATE_MISMATCH",
|
|
2735
|
+
"this deployment was staged outside a coordinator review session and this run is "
|
|
2736
|
+
"inside one, so it rebuilt a different request; no argument list can bridge that -- "
|
|
2737
|
+
"resume it outside the session, or start a new deployment",
|
|
2738
|
+
)
|
|
2739
|
+
if drifted:
|
|
2740
|
+
raise HostedDeployError(
|
|
2741
|
+
"DEPLOY_STATE_MISMATCH",
|
|
2742
|
+
"this deployment journal was staged with different inputs: "
|
|
2743
|
+
f"{_identity_drift_sentence(journaled, inputs, drifted)}; rerun it with the values it "
|
|
2744
|
+
"was staged with, or start a new deployment",
|
|
2745
|
+
)
|
|
2746
|
+
if state.get("request_digest") != request_digest:
|
|
2747
|
+
candidates = [
|
|
2748
|
+
*sorted(
|
|
2749
|
+
_IDENTITY_INPUT_LABELS[name]
|
|
2750
|
+
for name in _IDENTITY_INPUT_LABELS
|
|
2751
|
+
if name not in journaled and name != "workspace_id"
|
|
2752
|
+
),
|
|
2753
|
+
*_UNNAMED_REQUEST_INPUTS,
|
|
2754
|
+
]
|
|
2755
|
+
raise HostedDeployError(
|
|
2756
|
+
"DEPLOY_STATE_MISMATCH",
|
|
2757
|
+
"this deployment journal was staged from a different request, and every input it "
|
|
2758
|
+
f"records still matches, so what moved is one of: {', '.join(candidates)} -- rerun it "
|
|
2759
|
+
"the way it was staged, or start a new deployment",
|
|
2760
|
+
)
|
|
2761
|
+
raise HostedDeployError(
|
|
2762
|
+
"DEPLOY_STATE_MISMATCH", "the deployment journal belongs to another exact handoff"
|
|
2763
|
+
)
|
|
2764
|
+
|
|
2765
|
+
|
|
2766
|
+
def _refuse_legacy_approval_journal(state: Mapping[str, Any]) -> None:
|
|
2767
|
+
"""Refuse every old human-gate pause rather than inventing a direct confirmation.
|
|
2768
|
+
|
|
2769
|
+
The direct Editor flow never writes ``approval_required``. Even an old journal that happens to
|
|
2770
|
+
carry a complete approval-request response and ETag has no journaled Editor confirmation, so
|
|
2771
|
+
that status alone is sufficient to identify a handoff that cannot safely cross into activation.
|
|
2772
|
+
"""
|
|
2773
|
+
|
|
2774
|
+
if state.get("status") != "approval_required":
|
|
2775
|
+
return
|
|
2776
|
+
raise HostedDeployError(
|
|
2777
|
+
"DEPLOY_LEGACY_APPROVAL_JOURNAL",
|
|
2778
|
+
"this deployment journal was paused by the former human-approval flow and has no "
|
|
2779
|
+
"journaled Editor confirmation; it cannot be activated safely, so start a new deployment "
|
|
2780
|
+
"from a fresh verified Build directory",
|
|
2781
|
+
)
|
|
2782
|
+
|
|
2783
|
+
|
|
2784
|
+
def _resource_etag(state: Mapping[str, Any], name: str) -> str:
|
|
2785
|
+
value = state.get("etags", {}).get(name) if isinstance(state.get("etags"), dict) else None
|
|
2786
|
+
if not isinstance(value, str) or not value.startswith('"') or not value.endswith('"'):
|
|
2787
|
+
raise HostedDeployError(
|
|
2788
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID", f"Studio omitted the ETag after {name}"
|
|
2789
|
+
)
|
|
2790
|
+
return value
|
|
2791
|
+
|
|
2792
|
+
|
|
2793
|
+
def _generated_response(response: Any) -> StudioResponse:
|
|
2794
|
+
parsed = getattr(response, "parsed", None)
|
|
2795
|
+
body = parsed.to_dict() if parsed is not None and hasattr(parsed, "to_dict") else {}
|
|
2796
|
+
headers = getattr(response, "headers", {})
|
|
2797
|
+
return StudioResponse(
|
|
2798
|
+
status_code=int(response.status_code),
|
|
2799
|
+
body=body,
|
|
2800
|
+
etag=headers.get("ETag") if hasattr(headers, "get") else None,
|
|
2801
|
+
)
|
|
2802
|
+
|
|
2803
|
+
|
|
2804
|
+
def _success(response: StudioResponse, expected: set[int], code: str) -> Mapping[str, Any]:
|
|
2805
|
+
if response.status_code not in expected:
|
|
2806
|
+
api_code = response.body.get("code")
|
|
2807
|
+
suffix = f" ({api_code})" if isinstance(api_code, str) else ""
|
|
2808
|
+
raise HostedDeployError(
|
|
2809
|
+
code, f"Studio refused the operation (HTTP {response.status_code}){suffix}"
|
|
2810
|
+
)
|
|
2811
|
+
return response.body
|
|
2812
|
+
|
|
2813
|
+
|
|
2814
|
+
def _etag(response: StudioResponse, operation: str) -> str:
|
|
2815
|
+
value = response.etag
|
|
2816
|
+
if not isinstance(value, str) or not value.startswith('"') or not value.endswith('"'):
|
|
2817
|
+
raise HostedDeployError(
|
|
2818
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID", f"Studio omitted the ETag after {operation}"
|
|
2819
|
+
)
|
|
2820
|
+
return value
|
|
2821
|
+
|
|
2822
|
+
|
|
2823
|
+
def _idempotency(operation: str, identity: str) -> str:
|
|
2824
|
+
digest = hashlib.sha256(f"{operation}:{identity}".encode()).hexdigest()
|
|
2825
|
+
return f"mr-data.deploy.v3:{digest}"
|
|
2826
|
+
|
|
2827
|
+
|
|
2828
|
+
def _same(value: Mapping[str, Any], field: str, expected: Any) -> None:
|
|
2829
|
+
if value.get(field) != expected:
|
|
2830
|
+
raise HostedDeployError(
|
|
2831
|
+
"DEPLOY_STATE_MISMATCH", f"{field} differs from the authenticated deployment scope"
|
|
2832
|
+
)
|
|
2833
|
+
|
|
2834
|
+
|
|
2835
|
+
def _required_text(value: Mapping[str, Any], field: str) -> str:
|
|
2836
|
+
selected = value.get(field)
|
|
2837
|
+
if not isinstance(selected, str) or not selected:
|
|
2838
|
+
raise HostedDeployError("DEPLOY_RESPONSE_INVALID", f"the response omitted {field}")
|
|
2839
|
+
return selected
|
|
2840
|
+
|
|
2841
|
+
|
|
2842
|
+
def _required_uuid(value: UUID | None, field: str) -> UUID:
|
|
2843
|
+
if value is None:
|
|
2844
|
+
raise HostedDeployError("DEPLOY_REQUEST_INVALID", f"{field} is missing")
|
|
2845
|
+
return value
|
|
2846
|
+
|
|
2847
|
+
|
|
2848
|
+
def _uuid(value: Any, field: str) -> UUID:
|
|
2849
|
+
try:
|
|
2850
|
+
parsed = UUID(value) if isinstance(value, str) else None
|
|
2851
|
+
except ValueError:
|
|
2852
|
+
parsed = None
|
|
2853
|
+
if parsed is None or str(parsed) != value:
|
|
2854
|
+
raise HostedDeployError("DEPLOY_RESPONSE_INVALID", f"{field} is not a canonical UUID")
|
|
2855
|
+
return parsed
|
|
2856
|
+
|
|
2857
|
+
|
|
2858
|
+
def _uuid_field(value: Mapping[str, Any], field: str) -> UUID:
|
|
2859
|
+
try:
|
|
2860
|
+
return _uuid(value.get(field), field)
|
|
2861
|
+
except HostedDeployError as error:
|
|
2862
|
+
raise HostedDeployError(
|
|
2863
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID", f"Studio omitted a valid {field}"
|
|
2864
|
+
) from error
|
|
2865
|
+
|
|
2866
|
+
|
|
2867
|
+
def _positive_int_field(value: Mapping[str, Any], field: str) -> int:
|
|
2868
|
+
selected = value.get(field)
|
|
2869
|
+
if isinstance(selected, bool) or not isinstance(selected, int) or selected < 1:
|
|
2870
|
+
raise HostedDeployError(
|
|
2871
|
+
"DEPLOY_STUDIO_RESPONSE_INVALID", f"Studio omitted a positive {field}"
|
|
2872
|
+
)
|
|
2873
|
+
return selected
|
|
2874
|
+
|
|
2875
|
+
|
|
2876
|
+
def _timestamp(value: str, field: str) -> datetime:
|
|
2877
|
+
try:
|
|
2878
|
+
parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
|
|
2879
|
+
except ValueError as error:
|
|
2880
|
+
raise HostedDeployError("DEPLOY_RESPONSE_INVALID", f"{field} is not a timestamp") from error
|
|
2881
|
+
if parsed.tzinfo is None or parsed.utcoffset() != UTC.utcoffset(parsed):
|
|
2882
|
+
raise HostedDeployError("DEPLOY_RESPONSE_INVALID", f"{field} is not UTC")
|
|
2883
|
+
return parsed
|
|
2884
|
+
|
|
2885
|
+
|
|
2886
|
+
def _deployment_options(dataset_name: Any, cron: Any, maximum: Any) -> None:
|
|
2887
|
+
if (
|
|
2888
|
+
not isinstance(dataset_name, str)
|
|
2889
|
+
or not 1 <= len(dataset_name) <= 200
|
|
2890
|
+
or dataset_name.strip() != dataset_name
|
|
2891
|
+
or not dataset_name
|
|
2892
|
+
):
|
|
2893
|
+
raise HostedDeployError(
|
|
2894
|
+
"DEPLOY_REQUEST_INVALID", "dataset name must be 1-200 trimmed nonblank characters"
|
|
2895
|
+
)
|
|
2896
|
+
if (
|
|
2897
|
+
not isinstance(cron, str)
|
|
2898
|
+
or not 9 <= len(cron) <= 100
|
|
2899
|
+
or cron.strip() != cron
|
|
2900
|
+
or len(cron.split()) != 5
|
|
2901
|
+
or "\n" in cron
|
|
2902
|
+
or "\r" in cron
|
|
2903
|
+
):
|
|
2904
|
+
raise HostedDeployError(
|
|
2905
|
+
"DEPLOY_REQUEST_INVALID", "cron must be one satisfiable five-field UTC expression"
|
|
2906
|
+
)
|
|
2907
|
+
try:
|
|
2908
|
+
croniter(cron, datetime(2000, 1, 1, tzinfo=UTC)).get_next(datetime)
|
|
2909
|
+
except (CroniterBadCronError, CroniterBadDateError, ValueError) as error:
|
|
2910
|
+
raise HostedDeployError(
|
|
2911
|
+
"DEPLOY_REQUEST_INVALID", "cron must be one satisfiable five-field UTC expression"
|
|
2912
|
+
) from error
|
|
2913
|
+
if isinstance(maximum, bool) or not isinstance(maximum, int) or not 1 <= maximum <= 10:
|
|
2914
|
+
raise HostedDeployError(
|
|
2915
|
+
"DEPLOY_REQUEST_INVALID", "max concurrent runs must be between 1 and 10"
|
|
2916
|
+
)
|
|
2917
|
+
|
|
2918
|
+
|
|
2919
|
+
def _service_url(value: str, label: str, *, allow_loopback_http: bool) -> str:
|
|
2920
|
+
if not isinstance(value, str) or not 1 <= len(value) <= 2048 or value.endswith("/"):
|
|
2921
|
+
raise HostedDeployError("DEPLOY_CONFIG_INVALID", f"{label} is not a usable origin")
|
|
2922
|
+
parsed = urlsplit(value)
|
|
2923
|
+
loopback = False
|
|
2924
|
+
if parsed.hostname:
|
|
2925
|
+
try:
|
|
2926
|
+
loopback = ipaddress.ip_address(parsed.hostname).is_loopback
|
|
2927
|
+
except ValueError:
|
|
2928
|
+
loopback = parsed.hostname == "localhost"
|
|
2929
|
+
if (
|
|
2930
|
+
parsed.scheme not in ({"https", "http"} if allow_loopback_http else {"https"})
|
|
2931
|
+
or (parsed.scheme == "http" and not loopback)
|
|
2932
|
+
or not parsed.hostname
|
|
2933
|
+
or parsed.username is not None
|
|
2934
|
+
or parsed.password is not None
|
|
2935
|
+
or parsed.fragment
|
|
2936
|
+
or parsed.path not in {"", "/"}
|
|
2937
|
+
or parsed.query
|
|
2938
|
+
):
|
|
2939
|
+
raise HostedDeployError(
|
|
2940
|
+
"DEPLOY_CONFIG_INVALID",
|
|
2941
|
+
f"{label} must be an HTTPS origin without user information, path, query, or fragment",
|
|
2942
|
+
)
|
|
2943
|
+
return value
|
|
2944
|
+
|
|
2945
|
+
|
|
2946
|
+
def _require_cli_credential(credentials: ResolvedCloudCredentials) -> None:
|
|
2947
|
+
if _CLI_KEY.fullmatch(credentials.raw_key) is None:
|
|
2948
|
+
raise HostedDeployError(
|
|
2949
|
+
"DEPLOY_AUTHENTICATION_FAILED",
|
|
2950
|
+
"managed deployment requires an mr_cli_ credential; "
|
|
2951
|
+
"a live data key has read-only scope",
|
|
2952
|
+
)
|
|
2953
|
+
|
|
2954
|
+
|
|
2955
|
+
def _transfer_url(url: str, base_url: str) -> str:
|
|
2956
|
+
parsed = urlsplit(url)
|
|
2957
|
+
base = urlsplit(base_url)
|
|
2958
|
+
if (
|
|
2959
|
+
parsed.scheme not in {"https", "http"}
|
|
2960
|
+
or not parsed.hostname
|
|
2961
|
+
or parsed.username is not None
|
|
2962
|
+
or parsed.password is not None
|
|
2963
|
+
or parsed.fragment
|
|
2964
|
+
):
|
|
2965
|
+
raise HostedDeployError("DEPLOY_SIGNED_SESSION_INVALID", "signed upload URL is invalid")
|
|
2966
|
+
if parsed.scheme == "http" and (
|
|
2967
|
+
base.scheme != "http" or parsed.hostname != base.hostname or parsed.port != base.port
|
|
2968
|
+
):
|
|
2969
|
+
raise HostedDeployError(
|
|
2970
|
+
"DEPLOY_SIGNED_SESSION_INVALID",
|
|
2971
|
+
"insecure signed upload URL differs from the local Studio origin",
|
|
2972
|
+
)
|
|
2973
|
+
return url
|
|
2974
|
+
|
|
2975
|
+
|
|
2976
|
+
def _signed_headers(session: Mapping[str, Any], *, size: int) -> dict[str, str]:
|
|
2977
|
+
result: dict[str, str] = {}
|
|
2978
|
+
required = session.get("required_headers")
|
|
2979
|
+
if not isinstance(required, list):
|
|
2980
|
+
raise HostedDeployError("DEPLOY_SIGNED_SESSION_INVALID", "signed upload headers are absent")
|
|
2981
|
+
for item in required:
|
|
2982
|
+
if not isinstance(item, dict):
|
|
2983
|
+
raise HostedDeployError(
|
|
2984
|
+
"DEPLOY_SIGNED_SESSION_INVALID", "signed upload header is invalid"
|
|
2985
|
+
)
|
|
2986
|
+
name = item.get("name")
|
|
2987
|
+
value = item.get("value")
|
|
2988
|
+
if (
|
|
2989
|
+
not isinstance(name, str)
|
|
2990
|
+
or not isinstance(value, str)
|
|
2991
|
+
or name.lower() in {"authorization", "cookie", "host", "proxy-authorization"}
|
|
2992
|
+
or re.fullmatch(r"[A-Za-z0-9-]{1,128}", name) is None
|
|
2993
|
+
or "\r" in value
|
|
2994
|
+
or "\n" in value
|
|
2995
|
+
or name.lower() in result
|
|
2996
|
+
):
|
|
2997
|
+
raise HostedDeployError(
|
|
2998
|
+
"DEPLOY_SIGNED_SESSION_INVALID", "signed upload header is unsafe"
|
|
2999
|
+
)
|
|
3000
|
+
result[name.lower()] = value
|
|
3001
|
+
media_type = _required_text(session, "media_type")
|
|
3002
|
+
if result.get("content-type", media_type) != media_type:
|
|
3003
|
+
raise HostedDeployError(
|
|
3004
|
+
"DEPLOY_SIGNED_SESSION_INVALID", "signed upload content type differs"
|
|
3005
|
+
)
|
|
3006
|
+
result["content-type"] = media_type
|
|
3007
|
+
result["content-length"] = str(size)
|
|
3008
|
+
return result
|
|
3009
|
+
|
|
3010
|
+
|
|
3011
|
+
def _file_identity(value: os.stat_result) -> tuple[int, int, int, int, int, int, int]:
|
|
3012
|
+
return (
|
|
3013
|
+
value.st_dev,
|
|
3014
|
+
value.st_ino,
|
|
3015
|
+
value.st_mode,
|
|
3016
|
+
value.st_nlink,
|
|
3017
|
+
value.st_size,
|
|
3018
|
+
value.st_mtime_ns,
|
|
3019
|
+
value.st_ctime_ns,
|
|
3020
|
+
)
|
|
3021
|
+
|
|
3022
|
+
|
|
3023
|
+
__all__ = [
|
|
3024
|
+
"DEPLOYMENT_RECEIPT_SCHEMA",
|
|
3025
|
+
"DEPLOYMENT_STATE_SCHEMA",
|
|
3026
|
+
"FileDeploymentStateStore",
|
|
3027
|
+
"GeneratedStudioDeploymentClient",
|
|
3028
|
+
"HostedDeployError",
|
|
3029
|
+
"StudioDeploymentClient",
|
|
3030
|
+
"StudioResponse",
|
|
3031
|
+
"StudioToken",
|
|
3032
|
+
"TokenExchangeTransport",
|
|
3033
|
+
"build_editor_client",
|
|
3034
|
+
"deploy_managed_dataset",
|
|
3035
|
+
"exchange_studio_token",
|
|
3036
|
+
"journaled_reviewed_build",
|
|
3037
|
+
]
|