mostlyright-data 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mostlyright/data_harness/__init__.py +158 -0
- mostlyright/data_harness/acquisition/__init__.py +55 -0
- mostlyright/data_harness/acquisition/http.py +2773 -0
- mostlyright/data_harness/acquisition/parsing.py +809 -0
- mostlyright/data_harness/acquisition/ranges.py +495 -0
- mostlyright/data_harness/acquisition/result_download.py +360 -0
- mostlyright/data_harness/acquisition/retention_admission.py +248 -0
- mostlyright/data_harness/acquisition/sandbox.py +4888 -0
- mostlyright/data_harness/acquisition/url_policy.py +530 -0
- mostlyright/data_harness/agent_runtime.py +2743 -0
- mostlyright/data_harness/assets/logo-ink.svg +31 -0
- mostlyright/data_harness/backends/__init__.py +28 -0
- mostlyright/data_harness/backends/pandas_backend.py +350 -0
- mostlyright/data_harness/backends/polars_backend.py +366 -0
- mostlyright/data_harness/backends/protocol.py +124 -0
- mostlyright/data_harness/backends/reference.py +83 -0
- mostlyright/data_harness/backends/registry.py +55 -0
- mostlyright/data_harness/backends/restrictions.py +126 -0
- mostlyright/data_harness/canonical.py +333 -0
- mostlyright/data_harness/catalog_job.py +625 -0
- mostlyright/data_harness/cli.py +5398 -0
- mostlyright/data_harness/contracts.py +53 -0
- mostlyright/data_harness/coordinator.py +1307 -0
- mostlyright/data_harness/deploy.py +924 -0
- mostlyright/data_harness/deploy_target.py +312 -0
- mostlyright/data_harness/deployment_evidence.py +1067 -0
- mostlyright/data_harness/event_presentation.py +576 -0
- mostlyright/data_harness/events.py +2152 -0
- mostlyright/data_harness/fast_delimited.py +239 -0
- mostlyright/data_harness/fleet.py +237 -0
- mostlyright/data_harness/formats.py +236 -0
- mostlyright/data_harness/governors.py +1163 -0
- mostlyright/data_harness/hosted_bootstrap.py +972 -0
- mostlyright/data_harness/hosted_crawler.py +1115 -0
- mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
- mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
- mostlyright/data_harness/hosted_crawler_job.py +1277 -0
- mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
- mostlyright/data_harness/hosted_dataset.py +1500 -0
- mostlyright/data_harness/hosted_deploy.py +3037 -0
- mostlyright/data_harness/hosted_handoff.py +62 -0
- mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
- mostlyright/data_harness/hosted_ingestion_job.py +356 -0
- mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
- mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
- mostlyright/data_harness/hosted_session_worker.py +3554 -0
- mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
- mostlyright/data_harness/hosted_worker.py +6784 -0
- mostlyright/data_harness/ingestion/__init__.py +56 -0
- mostlyright/data_harness/ingestion/contracts.py +461 -0
- mostlyright/data_harness/ingestion/faults.py +42 -0
- mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
- mostlyright/data_harness/ingestion/spool.py +130 -0
- mostlyright/data_harness/ingestion/store.py +885 -0
- mostlyright/data_harness/key_seam.py +434 -0
- mostlyright/data_harness/linux_process_boundary.py +262 -0
- mostlyright/data_harness/local_contracts.py +2880 -0
- mostlyright/data_harness/local_search/__init__.py +5 -0
- mostlyright/data_harness/local_search/build_index.py +1087 -0
- mostlyright/data_harness/local_search/contracts.py +920 -0
- mostlyright/data_harness/local_search/query_trace.py +266 -0
- mostlyright/data_harness/local_search/retrieval.py +700 -0
- mostlyright/data_harness/local_search/sealed.py +474 -0
- mostlyright/data_harness/local_search/service.py +784 -0
- mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
- mostlyright/data_harness/nbrender/__init__.py +12 -0
- mostlyright/data_harness/nbrender/chrome.py +359 -0
- mostlyright/data_harness/nbrender/code_body.py +266 -0
- mostlyright/data_harness/nbrender/document.py +407 -0
- mostlyright/data_harness/nbrender/frame.py +275 -0
- mostlyright/data_harness/nbrender/interactive.py +337 -0
- mostlyright/data_harness/nbrender/markdown_body.py +477 -0
- mostlyright/data_harness/nbrender/mr_components.py +134 -0
- mostlyright/data_harness/nbrender/outputs_data.py +595 -0
- mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
- mostlyright/data_harness/nbrender/outputs_source.py +260 -0
- mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
- mostlyright/data_harness/nbrender/outputs_text.py +400 -0
- mostlyright/data_harness/nbrender/parse.py +394 -0
- mostlyright/data_harness/nbrender/status.py +40 -0
- mostlyright/data_harness/nbrender/tokens.py +1295 -0
- mostlyright/data_harness/notebook.py +1710 -0
- mostlyright/data_harness/offline.py +2049 -0
- mostlyright/data_harness/operation_registry.py +1007 -0
- mostlyright/data_harness/operator_setup.py +239 -0
- mostlyright/data_harness/pipeline.py +6428 -0
- mostlyright/data_harness/plan_graph.py +2026 -0
- mostlyright/data_harness/preparation/__init__.py +104 -0
- mostlyright/data_harness/preparation/contracts.py +1017 -0
- mostlyright/data_harness/preparation/engine.py +221 -0
- mostlyright/data_harness/preparation/errors.py +14 -0
- mostlyright/data_harness/preparation/gates.py +751 -0
- mostlyright/data_harness/preparation/joins.py +574 -0
- mostlyright/data_harness/preparation/profile.py +384 -0
- mostlyright/data_harness/preparation/table.py +217 -0
- mostlyright/data_harness/preparation/transforms.py +568 -0
- mostlyright/data_harness/progress_events.py +534 -0
- mostlyright/data_harness/readers/__init__.py +46 -0
- mostlyright/data_harness/readers/containers.py +963 -0
- mostlyright/data_harness/readers/contracts.py +542 -0
- mostlyright/data_harness/readers/delimited.py +257 -0
- mostlyright/data_harness/readers/grib2/__init__.py +33 -0
- mostlyright/data_harness/readers/grib2/admission.py +722 -0
- mostlyright/data_harness/readers/grib2/decode.py +1009 -0
- mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
- mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
- mostlyright/data_harness/readers/json_tabular.py +485 -0
- mostlyright/data_harness/readers/registry.py +514 -0
- mostlyright/data_harness/readers/samples/README.md +110 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples.py +582 -0
- mostlyright/data_harness/readers/spreadsheet.py +803 -0
- mostlyright/data_harness/readers/tabular.py +510 -0
- mostlyright/data_harness/recipe.py +5321 -0
- mostlyright/data_harness/repair/__init__.py +78 -0
- mostlyright/data_harness/repair/adapters.py +274 -0
- mostlyright/data_harness/repair/contracts.py +872 -0
- mostlyright/data_harness/repair/coordinator.py +1099 -0
- mostlyright/data_harness/repair/errors.py +16 -0
- mostlyright/data_harness/review.py +2533 -0
- mostlyright/data_harness/rowset.py +283 -0
- mostlyright/data_harness/serving.py +1975 -0
- mostlyright/data_harness/serving_edge.py +590 -0
- mostlyright/data_harness/serving_http.py +1031 -0
- mostlyright/data_harness/session_probes.py +759 -0
- mostlyright/data_harness/signing.py +101 -0
- mostlyright/data_harness/source_discovery.py +898 -0
- mostlyright/data_harness/sources/__init__.py +209 -0
- mostlyright/data_harness/sources/_adapter_steps.py +213 -0
- mostlyright/data_harness/sources/adapters.py +1214 -0
- mostlyright/data_harness/sources/cadence.py +1428 -0
- mostlyright/data_harness/sources/cadence_emission.py +453 -0
- mostlyright/data_harness/sources/cadence_history.py +546 -0
- mostlyright/data_harness/sources/catalog/__init__.py +17 -0
- mostlyright/data_harness/sources/catalog/admission.py +477 -0
- mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
- mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
- mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
- mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
- mostlyright/data_harness/sources/catalog/channel.py +523 -0
- mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
- mostlyright/data_harness/sources/catalog/contracts.py +825 -0
- mostlyright/data_harness/sources/catalog/coverage.py +137 -0
- mostlyright/data_harness/sources/catalog/delta.py +1340 -0
- mostlyright/data_harness/sources/catalog/embedding.py +532 -0
- mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
- mostlyright/data_harness/sources/catalog/fill.py +3889 -0
- mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
- mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
- mostlyright/data_harness/sources/catalog/gating.py +374 -0
- mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
- mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
- mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
- mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
- mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
- mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
- mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
- mostlyright/data_harness/sources/catalog/health.py +447 -0
- mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
- mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
- mostlyright/data_harness/sources/catalog/neural.py +1618 -0
- mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
- mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
- mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
- mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
- mostlyright/data_harness/sources/catalog/recommend.py +171 -0
- mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
- mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
- mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
- mostlyright/data_harness/sources/catalog/sealed.py +560 -0
- mostlyright/data_harness/sources/catalog/search.py +230 -0
- mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
- mostlyright/data_harness/sources/catalog/update.py +891 -0
- mostlyright/data_harness/sources/collections.py +815 -0
- mostlyright/data_harness/sources/contracts.py +2223 -0
- mostlyright/data_harness/sources/deletion.py +761 -0
- mostlyright/data_harness/sources/fitness.py +162 -0
- mostlyright/data_harness/sources/governance.py +163 -0
- mostlyright/data_harness/sources/hosted.py +173 -0
- mostlyright/data_harness/sources/integration.py +218 -0
- mostlyright/data_harness/sources/range_reader.py +418 -0
- mostlyright/data_harness/sources/registry.py +514 -0
- mostlyright/data_harness/sources/rights_rule.py +59 -0
- mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
- mostlyright/data_harness/sources/sports.py +521 -0
- mostlyright/data_harness/sources/stream.py +524 -0
- mostlyright/data_harness/sources/stream_connector.py +418 -0
- mostlyright/data_harness/sources/stream_recorder.py +1404 -0
- mostlyright/data_harness/studio_boundary.py +2019 -0
- mostlyright/data_harness/thin/__init__.py +37 -0
- mostlyright/data_harness/thin/acquire.py +1137 -0
- mostlyright/data_harness/thin/acquire_cancel.py +579 -0
- mostlyright/data_harness/thin/approvals.py +617 -0
- mostlyright/data_harness/thin/commands.py +406 -0
- mostlyright/data_harness/thin/download.py +194 -0
- mostlyright/data_harness/thin/narrative.py +589 -0
- mostlyright/data_harness/thin/parity.py +1070 -0
- mostlyright/data_harness/thin/propose.py +2759 -0
- mostlyright/data_harness/thin/research.py +1663 -0
- mostlyright/data_harness/thin/router.py +924 -0
- mostlyright/data_harness/thin/runs.py +519 -0
- mostlyright/data_harness/thin/session.py +281 -0
- mostlyright/data_harness/thin/stream.py +501 -0
- mostlyright/data_harness/thin/transport.py +187 -0
- mostlyright/data_harness/thin/vocabulary.py +368 -0
- mostlyright/data_harness/thin/workers.py +164 -0
- mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
- mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
- mostlyright/data_harness/unit_flow.py +927 -0
- mostlyright/data_harness/units.py +572 -0
- mostlyright/data_harness/ux/__init__.py +9 -0
- mostlyright/data_harness/ux/approve.py +485 -0
- mostlyright/data_harness/ux/author_yaml.py +597 -0
- mostlyright/data_harness/ux/cloud_auth.py +447 -0
- mostlyright/data_harness/ux/commands/__init__.py +260 -0
- mostlyright/data_harness/ux/commands/approve.py +136 -0
- mostlyright/data_harness/ux/commands/auth.py +744 -0
- mostlyright/data_harness/ux/commands/author.py +79 -0
- mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
- mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
- mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
- mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
- mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
- mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
- mostlyright/data_harness/ux/commands/deploy.py +134 -0
- mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
- mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
- mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
- mostlyright/data_harness/ux/commands/diff.py +74 -0
- mostlyright/data_harness/ux/commands/index.py +84 -0
- mostlyright/data_harness/ux/commands/inventory.py +47 -0
- mostlyright/data_harness/ux/commands/list_builds.py +143 -0
- mostlyright/data_harness/ux/commands/login.py +63 -0
- mostlyright/data_harness/ux/commands/peek.py +236 -0
- mostlyright/data_harness/ux/commands/plan_check.py +90 -0
- mostlyright/data_harness/ux/commands/preflight.py +97 -0
- mostlyright/data_harness/ux/commands/record.py +107 -0
- mostlyright/data_harness/ux/commands/review_setup.py +47 -0
- mostlyright/data_harness/ux/commands/search.py +440 -0
- mostlyright/data_harness/ux/commands/show.py +61 -0
- mostlyright/data_harness/ux/commands/whoami.py +37 -0
- mostlyright/data_harness/ux/credential_native.py +551 -0
- mostlyright/data_harness/ux/credential_store.py +1055 -0
- mostlyright/data_harness/ux/credentials.py +631 -0
- mostlyright/data_harness/ux/diffing.py +444 -0
- mostlyright/data_harness/ux/headline.py +671 -0
- mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
- mostlyright/data_harness/ux/hosted_run_status.py +619 -0
- mostlyright/data_harness/ux/inventory.py +427 -0
- mostlyright/data_harness/ux/local_review.py +375 -0
- mostlyright/data_harness/ux/login.py +691 -0
- mostlyright/data_harness/ux/path_kind.py +147 -0
- mostlyright/data_harness/ux/peek.py +1000 -0
- mostlyright/data_harness/ux/plain_file.py +178 -0
- mostlyright/data_harness/ux/plan_check.py +311 -0
- mostlyright/data_harness/ux/preflight.py +918 -0
- mostlyright/data_harness/ux/readers.py +1124 -0
- mostlyright/data_harness/ux/remediation.py +2195 -0
- mostlyright/data_harness/ux/render.py +657 -0
- mostlyright/data_harness/ux/workload.py +1077 -0
- mostlyright/data_harness/viewer.py +3713 -0
- mostlyright/data_harness/visual_run/__init__.py +83 -0
- mostlyright/data_harness/visual_run/authoring.py +235 -0
- mostlyright/data_harness/visual_run/contracts.py +673 -0
- mostlyright/data_harness/visual_run/materialize.py +486 -0
- mostlyright/data_harness/visual_run/observations.py +874 -0
- mostlyright/data_harness/visual_run/query.py +259 -0
- mostlyright/data_harness/visual_run/reducer.py +280 -0
- mostlyright/data_harness/visual_run/sdk.py +892 -0
- mostlyright/data_harness/visual_run/store.py +584 -0
- mostlyright/data_harness/visual_run/transport.py +239 -0
- mostlyright/data_harness/watch.py +2999 -0
- mostlyright_data-0.9.0.dist-info/METADATA +607 -0
- mostlyright_data-0.9.0.dist-info/RECORD +314 -0
- mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
- mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
|
@@ -0,0 +1,619 @@
|
|
|
1
|
+
"""Studio's own reading of one activated Run, reported exactly as Studio reports it.
|
|
2
|
+
|
|
3
|
+
`mr-data deploy --resume` queues a Run and returns its `run_id`. Until now that identifier was the
|
|
4
|
+
end of the conversation: nothing in this repository ever asked Studio how the Run was doing, so the
|
|
5
|
+
only way to learn whether a Dataset had built was to open the Cloud dashboard. This module asks,
|
|
6
|
+
through the same pinned generated client the deployment path already builds, and reports the answer
|
|
7
|
+
without embellishing it.
|
|
8
|
+
|
|
9
|
+
Three rules decide everything below.
|
|
10
|
+
|
|
11
|
+
**The state vocabulary is Studio's.** `RecipeBoundRunStatus` has thirteen values, and every one of
|
|
12
|
+
them is in :data:`RUN_STATES` with what it means and what the operator should do about it, each
|
|
13
|
+
classified from Studio's own state machines rather than from how its name reads. The build is
|
|
14
|
+
where a fourteenth value is caught: `tests/test_hosted_run_status.py` enumerates the enum out of
|
|
15
|
+
the pinned client wheel, so a state added upstream fails there rather than reaching a person as a
|
|
16
|
+
shrug. At runtime such a value is refused rather than rendered -- the pinned client will not parse
|
|
17
|
+
it, and `DEPLOY_RUN_STATUS_UNKNOWN` guards the table for any answer that gets past it -- and in no
|
|
18
|
+
case is it shown as a state, a terminal answer, or a released version.
|
|
19
|
+
|
|
20
|
+
**Nothing displayed is derived from a clock.** Every fact in the payload comes out of one Studio
|
|
21
|
+
response or out of the deployment journal. The one duration reported is the difference between two
|
|
22
|
+
timestamps Studio persisted, and it is absent whenever either is. No percentage, no estimate, and
|
|
23
|
+
no elapsed counter is manufactured here, because a number nobody measured is worse than no number.
|
|
24
|
+
|
|
25
|
+
**A poll is a bounded request.** With no `--watch-seconds` this is exactly one read. With one, it
|
|
26
|
+
is a bounded loop that stops the moment Studio's answer stops changing on its own -- at a terminal
|
|
27
|
+
state, or at one of the two states that will not move until somebody acts elsewhere -- and, if the
|
|
28
|
+
bound arrives first, says so as its own answer rather than claiming the Run ended.
|
|
29
|
+
|
|
30
|
+
The polling shape, the backoff, and the two retryable refusals follow
|
|
31
|
+
:mod:`mostlyright.data_harness.ux.hosted_acquisition`, which already polls a hosted Studio session
|
|
32
|
+
to a terminal state; the binding check before an answer is believed follows its
|
|
33
|
+
``_validate_initial_session``. Nothing here streams events, downloads an artifact, cancels, or
|
|
34
|
+
retries a Run: this module reads one resource and prints what it says.
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
from __future__ import annotations
|
|
38
|
+
|
|
39
|
+
import time
|
|
40
|
+
from collections.abc import Callable, Mapping
|
|
41
|
+
from dataclasses import dataclass
|
|
42
|
+
from datetime import UTC, datetime
|
|
43
|
+
from pathlib import Path
|
|
44
|
+
from typing import Any
|
|
45
|
+
from uuid import UUID
|
|
46
|
+
|
|
47
|
+
from mostlyright.data_harness.hosted_deploy import (
|
|
48
|
+
STUDIO_SCHEMA_VERSION,
|
|
49
|
+
FileDeploymentStateStore,
|
|
50
|
+
GeneratedStudioDeploymentClient,
|
|
51
|
+
HostedDeployError,
|
|
52
|
+
StudioDeploymentClient,
|
|
53
|
+
StudioToken,
|
|
54
|
+
cloud_dashboard_url,
|
|
55
|
+
)
|
|
56
|
+
from mostlyright.data_harness.ux.credentials import ResolvedCloudCredentials
|
|
57
|
+
|
|
58
|
+
RUN_STATUS_SCHEMA = "mostlyright-hosted-run-status.v1"
|
|
59
|
+
|
|
60
|
+
#: The longest a `--watch-seconds` may ask this command to wait, and the same half hour
|
|
61
|
+
#: `hosted_acquisition.POLL_TIMEOUT_SECONDS` bounds a hosted acquisition by. A watch is a
|
|
62
|
+
#: foreground operation somebody is sitting in front of; it stays interruptible and it always ends.
|
|
63
|
+
MAX_WATCH_SECONDS = 1800
|
|
64
|
+
|
|
65
|
+
#: The first gap between reads, and the ceiling the doubling stops at. A hosted Run is minutes to
|
|
66
|
+
#: hours long, so the gap grows quickly and then stays at half a minute rather than asking Studio
|
|
67
|
+
#: about a Run that has not started yet twice a second.
|
|
68
|
+
FIRST_POLL_DELAY_SECONDS = 2.0
|
|
69
|
+
MAX_POLL_DELAY_SECONDS = 30.0
|
|
70
|
+
|
|
71
|
+
#: What the harness does next, given a state. `watching` is the only value that keeps a watch
|
|
72
|
+
#: going: `released`, `stopped` and `needs_a_person` are all places the answer stops changing
|
|
73
|
+
#: without something happening outside this command.
|
|
74
|
+
WATCHING = "watching"
|
|
75
|
+
RELEASED = "released"
|
|
76
|
+
STOPPED = "stopped"
|
|
77
|
+
NEEDS_A_PERSON = "needs_a_person"
|
|
78
|
+
|
|
79
|
+
_WAIT = "nothing to do here; the run continues on its own"
|
|
80
|
+
_DEPLOY_AGAIN = (
|
|
81
|
+
"deploy again when you want another version, passing --previous-run-dir so it updates this "
|
|
82
|
+
"Dataset instead of founding a second one"
|
|
83
|
+
)
|
|
84
|
+
#: What to do about a run that ended without a version, said once. Both states that reach it need
|
|
85
|
+
#: the same two things: a corrected Build, and the flag that keeps it on the same Dataset.
|
|
86
|
+
_DEPLOY_CORRECTED = (
|
|
87
|
+
"correct the Recipe or the sources it names, then deploy the corrected Build with "
|
|
88
|
+
"--previous-run-dir pointing at this run directory"
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
@dataclass(frozen=True)
|
|
93
|
+
class RunState:
|
|
94
|
+
"""One of Studio's thirteen Run states, and what it means for whoever deployed.
|
|
95
|
+
|
|
96
|
+
``disposition`` is what this command does about the state, not a coarser name for it: the
|
|
97
|
+
state itself is always reported verbatim beside it. ``terminal`` says Studio has finished with
|
|
98
|
+
this Run; a state that is neither terminal nor watched is one waiting on a person, which is a
|
|
99
|
+
third thing and is never folded into either of the other two.
|
|
100
|
+
"""
|
|
101
|
+
|
|
102
|
+
disposition: str
|
|
103
|
+
terminal: bool
|
|
104
|
+
means: str
|
|
105
|
+
next: str
|
|
106
|
+
|
|
107
|
+
@property
|
|
108
|
+
def keep_watching(self) -> bool:
|
|
109
|
+
return self.disposition == WATCHING
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
#: Every value of the pinned V3 contract's `RecipeBoundRunStatus`, with the sentence a person reads.
|
|
113
|
+
#:
|
|
114
|
+
#: ⚠ THIS TABLE IS THE CONTRACT, and it is checked against the pinned client rather than
|
|
115
|
+
#: remembered. Five of these states -- the two retry states, the two that wait on a person, and
|
|
116
|
+
#: `releasable` -- have no place at all in the five-step ladder this surface was first sketched as,
|
|
117
|
+
#: and folding them into "building" or "failed" would have hidden the two situations where somebody
|
|
118
|
+
#: has to go and do something. So each state says what it is and what to do, and a state that needs
|
|
119
|
+
#: a person says so in the disposition as well as in the words.
|
|
120
|
+
#:
|
|
121
|
+
#: ⚠ WHICH STATES THOSE ARE IS STUDIO'S ANSWER, NOT A GUESS FROM THEIR NAMES. Every classification
|
|
122
|
+
#: below is read off Studio's own machines -- `contracts/state-machines/run.json` and
|
|
123
|
+
#: `run-orchestration-v2.json` in mostlyrightmd/mostlyright-studio -- because two of them read
|
|
124
|
+
#: backwards from the outside. `awaiting_candidate_selection` sounds like a person choosing and is
|
|
125
|
+
#: an ordinary step Studio takes itself (`awaiting_candidate_selection -> verifying`, actor
|
|
126
|
+
#: `studio_orchestrator`) on the way through EVERY healthy run; parking a watch there would stop
|
|
127
|
+
#: the common case dead and send the operator to do something nobody is permitted to do.
|
|
128
|
+
#: `human_direction_required` sounds like a question somebody can answer and is listed in that
|
|
129
|
+
#: machine's `terminal_states` with one way in (`repair_required`, at the fix-it cycle limit) and
|
|
130
|
+
#: no way out at all, so the only move left is a corrected Build.
|
|
131
|
+
#:
|
|
132
|
+
#: ⚠ TWO OF STUDIO'S RUN STATES ARE NOT IN THIS TABLE, and that is Studio's doing rather than an
|
|
133
|
+
#: omission here. The two machines together declare fifteen; `RecipeBoundRunStatus`, the vocabulary
|
|
134
|
+
#: the pinned v2 read contract actually answers with, has thirteen. `reviewing` and
|
|
135
|
+
#: `awaiting_approval` exist in `run.json` and in the `listRuns` filter, and have no value in the
|
|
136
|
+
#: enum `get_run` returns, so no answer this module can receive is in either of them. They are
|
|
137
|
+
#: written down here rather than left out silently, and
|
|
138
|
+
#: `test_the_states_studio_cannot_report_are_named_here` fails the day one gains an enum value.
|
|
139
|
+
STATES_STUDIO_CANNOT_REPORT: frozenset[str] = frozenset({"reviewing", "awaiting_approval"})
|
|
140
|
+
|
|
141
|
+
RUN_STATES: dict[str, RunState] = {
|
|
142
|
+
"queued": RunState(
|
|
143
|
+
disposition=WATCHING,
|
|
144
|
+
terminal=False,
|
|
145
|
+
means="Studio has accepted this run and has not started it yet.",
|
|
146
|
+
next=_WAIT,
|
|
147
|
+
),
|
|
148
|
+
"running": RunState(
|
|
149
|
+
disposition=WATCHING,
|
|
150
|
+
terminal=False,
|
|
151
|
+
means="A Builder is building this version of the Dataset.",
|
|
152
|
+
next=_WAIT,
|
|
153
|
+
),
|
|
154
|
+
"verifying": RunState(
|
|
155
|
+
disposition=WATCHING,
|
|
156
|
+
terminal=False,
|
|
157
|
+
means="A Checker is re-running this version's own checks against it.",
|
|
158
|
+
next=_WAIT,
|
|
159
|
+
),
|
|
160
|
+
"releasable": RunState(
|
|
161
|
+
disposition=WATCHING,
|
|
162
|
+
terminal=False,
|
|
163
|
+
means="Every check passed. Studio has not released the version yet.",
|
|
164
|
+
next=_WAIT,
|
|
165
|
+
),
|
|
166
|
+
"released": RunState(
|
|
167
|
+
disposition=RELEASED,
|
|
168
|
+
terminal=True,
|
|
169
|
+
means="The new version passed and is live. This is the state a deployment is aiming at.",
|
|
170
|
+
next="read the Dataset at the current path this deployment recorded",
|
|
171
|
+
),
|
|
172
|
+
"cancelled": RunState(
|
|
173
|
+
disposition=STOPPED,
|
|
174
|
+
terminal=True,
|
|
175
|
+
means="Somebody stopped this run before it produced a version.",
|
|
176
|
+
next=_DEPLOY_AGAIN,
|
|
177
|
+
),
|
|
178
|
+
"failed": RunState(
|
|
179
|
+
disposition=STOPPED,
|
|
180
|
+
terminal=True,
|
|
181
|
+
means="The run stopped without producing a version.",
|
|
182
|
+
next=_DEPLOY_CORRECTED,
|
|
183
|
+
),
|
|
184
|
+
"abandoned": RunState(
|
|
185
|
+
disposition=STOPPED,
|
|
186
|
+
terminal=True,
|
|
187
|
+
means="Studio stopped attempting this run. Nothing further will happen to it.",
|
|
188
|
+
next=_DEPLOY_AGAIN,
|
|
189
|
+
),
|
|
190
|
+
"awaiting_candidate_selection": RunState(
|
|
191
|
+
disposition=WATCHING,
|
|
192
|
+
terminal=False,
|
|
193
|
+
means="The Builder finished and Studio has not started checking its result yet.",
|
|
194
|
+
next=_WAIT,
|
|
195
|
+
),
|
|
196
|
+
"human_direction_required": RunState(
|
|
197
|
+
disposition=NEEDS_A_PERSON,
|
|
198
|
+
terminal=True,
|
|
199
|
+
means=(
|
|
200
|
+
"Studio reached its limit for fix-it attempts on this run and stopped. Nothing moves "
|
|
201
|
+
"it from here, and no answer given in Cloud restarts it."
|
|
202
|
+
),
|
|
203
|
+
next=f"{_DEPLOY_CORRECTED}; this run itself will not go any further",
|
|
204
|
+
),
|
|
205
|
+
"producer_retry_pending": RunState(
|
|
206
|
+
disposition=WATCHING,
|
|
207
|
+
terminal=False,
|
|
208
|
+
means="A build attempt did not finish, and Studio has queued another one.",
|
|
209
|
+
next=_WAIT,
|
|
210
|
+
),
|
|
211
|
+
"verifier_retry_pending": RunState(
|
|
212
|
+
disposition=WATCHING,
|
|
213
|
+
terminal=False,
|
|
214
|
+
means="A checking attempt did not finish, and Studio has queued another one.",
|
|
215
|
+
next=_WAIT,
|
|
216
|
+
),
|
|
217
|
+
"repair_required": RunState(
|
|
218
|
+
disposition=NEEDS_A_PERSON,
|
|
219
|
+
terminal=False,
|
|
220
|
+
means=(
|
|
221
|
+
"A check did not pass, and Studio has opened a fix-it task for it. The run waits "
|
|
222
|
+
"until that task is carried out and produces a corrected run."
|
|
223
|
+
),
|
|
224
|
+
next=(
|
|
225
|
+
"follow that fix-it task in Cloud; this command displays the state and cannot carry "
|
|
226
|
+
"the task out"
|
|
227
|
+
),
|
|
228
|
+
),
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
@dataclass(frozen=True)
|
|
233
|
+
class ActivatedRun:
|
|
234
|
+
"""The coordinates one deployment journal already holds about the Run it queued.
|
|
235
|
+
|
|
236
|
+
Nothing here is invented and nothing is asked of Studio to build it: the deployment wrote all
|
|
237
|
+
of it down before this command existed, which is exactly why a killed process, a slept laptop,
|
|
238
|
+
or a fresh terminal can pick the same Run back up.
|
|
239
|
+
"""
|
|
240
|
+
|
|
241
|
+
run_id: UUID
|
|
242
|
+
dataset_id: UUID
|
|
243
|
+
table_id: UUID
|
|
244
|
+
workspace_id: UUID
|
|
245
|
+
table_recipe_id: str
|
|
246
|
+
recipe: Mapping[str, Any]
|
|
247
|
+
schedule: Mapping[str, Any]
|
|
248
|
+
cloud_dataset_id: str | None
|
|
249
|
+
current_path: str | None
|
|
250
|
+
state_path: Path
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def journaled_activation(run_dir: Path) -> ActivatedRun:
|
|
254
|
+
"""Read the queued Run's coordinates out of one deployment journal, writing no state.
|
|
255
|
+
|
|
256
|
+
Nothing recorded is changed, and no request leaves the machine. The one thing this does touch
|
|
257
|
+
is the journal's own directory: opening it goes through the deployment's file store, which
|
|
258
|
+
creates ``evidence/deployment/`` when it is absent and takes its lock, so a run directory that
|
|
259
|
+
has never been deployed gains two empty directories before being refused.
|
|
260
|
+
|
|
261
|
+
Raises:
|
|
262
|
+
HostedDeployError: when there is no journal, when it cannot be read safely, or when it
|
|
263
|
+
records no queued Run yet.
|
|
264
|
+
"""
|
|
265
|
+
|
|
266
|
+
store = FileDeploymentStateStore(run_dir)
|
|
267
|
+
try:
|
|
268
|
+
if not store.exists():
|
|
269
|
+
raise HostedDeployError(
|
|
270
|
+
"DEPLOY_STATE_MISSING", f"no deployment journal exists at {store.path}"
|
|
271
|
+
)
|
|
272
|
+
state = store.load()
|
|
273
|
+
path = store.path
|
|
274
|
+
finally:
|
|
275
|
+
store.close()
|
|
276
|
+
resources = state.get("resources")
|
|
277
|
+
activation = resources.get("activation") if isinstance(resources, Mapping) else None
|
|
278
|
+
if not isinstance(activation, Mapping):
|
|
279
|
+
raise HostedDeployError(
|
|
280
|
+
"DEPLOY_RUN_NOT_ACTIVATED",
|
|
281
|
+
"this deployment has not queued a run yet; recover the exact deployment with "
|
|
282
|
+
"mr-data deploy --resume",
|
|
283
|
+
)
|
|
284
|
+
if state.get("schema_version") != "mostlyright-hosted-deployment-state.v3":
|
|
285
|
+
raise HostedDeployError(
|
|
286
|
+
"DEPLOY_STATE_INVALID",
|
|
287
|
+
"the deployment journal is not a V3 Dataset/Table deployment",
|
|
288
|
+
)
|
|
289
|
+
binding = resources.get("cloud_binding")
|
|
290
|
+
binding = binding if isinstance(binding, Mapping) else {}
|
|
291
|
+
inputs = state.get("identity_inputs")
|
|
292
|
+
inputs = inputs if isinstance(inputs, Mapping) else {}
|
|
293
|
+
schedule = state.get("schedule")
|
|
294
|
+
return ActivatedRun(
|
|
295
|
+
run_id=_journaled_uuid(activation, "run_id"),
|
|
296
|
+
dataset_id=_journaled_uuid(resources, "dataset_id"),
|
|
297
|
+
table_id=_journaled_uuid(activation, "table_id"),
|
|
298
|
+
workspace_id=_journaled_uuid(state, "workspace_id"),
|
|
299
|
+
table_recipe_id=_journaled_identifier(inputs, "table_recipe_id"),
|
|
300
|
+
# The V3 deployment journal records the exact TableRecipe coordinate before activation.
|
|
301
|
+
# Older V2 journals are deliberately not aliased: their dataset_id named the child object.
|
|
302
|
+
recipe={name: inputs[name] for name in ("recipe_version", "recipe_digest")},
|
|
303
|
+
schedule=dict(schedule) if isinstance(schedule, Mapping) else {},
|
|
304
|
+
cloud_dataset_id=_optional_text(binding, "cloud_dataset_id"),
|
|
305
|
+
current_path=_optional_text(binding, "current_path"),
|
|
306
|
+
state_path=path,
|
|
307
|
+
)
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
def report_run_status(
|
|
311
|
+
run_dir: Path,
|
|
312
|
+
*,
|
|
313
|
+
token: StudioToken,
|
|
314
|
+
credentials: ResolvedCloudCredentials,
|
|
315
|
+
watch_seconds: int | None = None,
|
|
316
|
+
client_factory: Callable[[StudioToken], StudioDeploymentClient] | None = None,
|
|
317
|
+
clock: Callable[[], float] = time.monotonic,
|
|
318
|
+
sleep: Callable[[float], None] = time.sleep,
|
|
319
|
+
) -> dict[str, Any]:
|
|
320
|
+
"""Report the real state of the Run this deployment queued, reading Studio and nothing else.
|
|
321
|
+
|
|
322
|
+
With ``watch_seconds`` unset this makes exactly one read. With it set, it reads on a bounded
|
|
323
|
+
backoff until the state stops changing on its own or the bound arrives, whichever is first,
|
|
324
|
+
and each distinct state it saw is listed once in the answer.
|
|
325
|
+
|
|
326
|
+
Raises:
|
|
327
|
+
HostedDeployError: when the journal cannot be read, when it names no queued Run, when the
|
|
328
|
+
authenticated workspace is not the one that deployed, or when Studio's answer cannot
|
|
329
|
+
be reached, is refused, or is not bound to this deployment. An answer naming a state
|
|
330
|
+
this release does not know is refused too, though the pinned client refuses to parse
|
|
331
|
+
one before it can reach that check.
|
|
332
|
+
"""
|
|
333
|
+
|
|
334
|
+
if watch_seconds is not None and (
|
|
335
|
+
isinstance(watch_seconds, bool)
|
|
336
|
+
or not isinstance(watch_seconds, int)
|
|
337
|
+
or not 1 <= watch_seconds <= MAX_WATCH_SECONDS
|
|
338
|
+
):
|
|
339
|
+
raise HostedDeployError(
|
|
340
|
+
"DEPLOY_REQUEST_INVALID",
|
|
341
|
+
f"watch seconds must be between 1 and {MAX_WATCH_SECONDS}",
|
|
342
|
+
)
|
|
343
|
+
activated = journaled_activation(run_dir)
|
|
344
|
+
if token.workspace_id != activated.workspace_id:
|
|
345
|
+
raise HostedDeployError(
|
|
346
|
+
"DEPLOY_STATE_MISMATCH",
|
|
347
|
+
"this credential authenticates a different workspace from the one that deployed",
|
|
348
|
+
)
|
|
349
|
+
client = (client_factory or GeneratedStudioDeploymentClient)(token)
|
|
350
|
+
try:
|
|
351
|
+
run, observed, timed_out = _watch(
|
|
352
|
+
client,
|
|
353
|
+
activated,
|
|
354
|
+
watch_seconds=watch_seconds,
|
|
355
|
+
clock=clock,
|
|
356
|
+
sleep=sleep,
|
|
357
|
+
)
|
|
358
|
+
finally:
|
|
359
|
+
client.close()
|
|
360
|
+
return _payload(
|
|
361
|
+
activated,
|
|
362
|
+
run,
|
|
363
|
+
observed,
|
|
364
|
+
timed_out=timed_out,
|
|
365
|
+
watch_seconds=watch_seconds,
|
|
366
|
+
cloud_url=credentials.cloud_url,
|
|
367
|
+
)
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
def _watch(
|
|
371
|
+
client: StudioDeploymentClient,
|
|
372
|
+
activated: ActivatedRun,
|
|
373
|
+
*,
|
|
374
|
+
watch_seconds: int | None,
|
|
375
|
+
clock: Callable[[], float],
|
|
376
|
+
sleep: Callable[[float], None],
|
|
377
|
+
) -> tuple[Mapping[str, Any], list[str], bool]:
|
|
378
|
+
"""Read until the state stops moving on its own, the bound arrives, or once when unwatched."""
|
|
379
|
+
|
|
380
|
+
# The bound starts before the first read, not after it. A read is itself bounded -- the client
|
|
381
|
+
# carries `REQUEST_TIMEOUT_SECONDS` -- but starting the clock afterwards would hand every watch
|
|
382
|
+
# one free request beyond the number the caller asked for.
|
|
383
|
+
deadline = None if watch_seconds is None else clock() + float(watch_seconds)
|
|
384
|
+
run = _read_run(client, activated)
|
|
385
|
+
observed = [str(run["status"])]
|
|
386
|
+
if deadline is None:
|
|
387
|
+
return run, observed, False
|
|
388
|
+
delay = FIRST_POLL_DELAY_SECONDS
|
|
389
|
+
while RUN_STATES[str(run["status"])].keep_watching:
|
|
390
|
+
remaining = deadline - clock()
|
|
391
|
+
if remaining <= 0:
|
|
392
|
+
return run, observed, True
|
|
393
|
+
# Never sleep past the bound the caller asked for: a watch that overshoots its own
|
|
394
|
+
# deadline is no longer the bounded foreground operation it was described as.
|
|
395
|
+
sleep(min(delay, remaining))
|
|
396
|
+
delay = min(delay * 2, MAX_POLL_DELAY_SECONDS)
|
|
397
|
+
run = _read_run(client, activated)
|
|
398
|
+
status = str(run["status"])
|
|
399
|
+
if status not in observed:
|
|
400
|
+
observed.append(status)
|
|
401
|
+
return run, observed, False
|
|
402
|
+
|
|
403
|
+
|
|
404
|
+
def _read_run(client: StudioDeploymentClient, activated: ActivatedRun) -> Mapping[str, Any]:
|
|
405
|
+
"""One read of one Run, believed only after it is bound to this deployment."""
|
|
406
|
+
|
|
407
|
+
try:
|
|
408
|
+
response = client.read("get_run", resource_id=activated.run_id)
|
|
409
|
+
except HostedDeployError:
|
|
410
|
+
raise
|
|
411
|
+
except Exception as error:
|
|
412
|
+
# ⚠ EVERY failure of the call itself is one refusal, and deliberately so.
|
|
413
|
+
#
|
|
414
|
+
# It is tempting to single out `ValueError` here and call it "Studio named a state this
|
|
415
|
+
# release does not know", because that is one thing it can mean: the pinned model builds
|
|
416
|
+
# `RecipeBoundRunStatus(...)` while parsing, so an unknown state raises before a body ever
|
|
417
|
+
# reaches this module. Doing that was WRONG, and measurably so. The generated operation
|
|
418
|
+
# parses the error body of every non-200 through `StableAPIError.from_dict`, which builds
|
|
419
|
+
# `StableAPIErrorCode(...)` -- an enum with no generic 5xx or not-found member -- and
|
|
420
|
+
# `response.json()` raises `JSONDecodeError`, itself a `ValueError`, on any HTML body. So a
|
|
421
|
+
# cold-start 503 behind Cloud, a 500 with no body, and a 404 for a run that is not there
|
|
422
|
+
# all raise `ValueError` too, and singling it out reported the commonest transient failure
|
|
423
|
+
# this deployment path has as a contract this release cannot read: the one refusal an
|
|
424
|
+
# operator must not retry, printed for the three they must.
|
|
425
|
+
#
|
|
426
|
+
# Nothing here can tell those apart -- the status code is inside the exception's own
|
|
427
|
+
# stack, not on it -- so this says the true and useful thing for all of them, and the run
|
|
428
|
+
# state vocabulary is held by the build-time gate in `tests/test_hosted_run_status.py`
|
|
429
|
+
# instead. `DEPLOY_RUN_STATUS_UNKNOWN` still guards the table below, for any answer that
|
|
430
|
+
# gets that far.
|
|
431
|
+
raise HostedDeployError(
|
|
432
|
+
"DEPLOY_RUN_POLL_UNAVAILABLE", "the run status could not be checked"
|
|
433
|
+
) from error
|
|
434
|
+
if response.status_code != 200:
|
|
435
|
+
raise HostedDeployError(
|
|
436
|
+
"DEPLOY_RUN_POLL_REFUSED",
|
|
437
|
+
f"Studio refused the run status check (HTTP {response.status_code})",
|
|
438
|
+
)
|
|
439
|
+
_validate_run(response.body, activated)
|
|
440
|
+
return response.body
|
|
441
|
+
|
|
442
|
+
|
|
443
|
+
def _validate_run(run: Mapping[str, Any], activated: ActivatedRun) -> None:
|
|
444
|
+
"""Refuse an answer that is not this deployment's Run before a word of it is believed.
|
|
445
|
+
|
|
446
|
+
An answer is trusted for exactly one reason: every coordinate the deployment journaled is in
|
|
447
|
+
it and matches. Without this a substituted or stale response -- another workspace's Run, the
|
|
448
|
+
predecessor Dataset's Run, a replayed body from a different deployment -- would be rendered as
|
|
449
|
+
this Run's own state, and a `released` that belongs to something else is the one wrong answer
|
|
450
|
+
this surface must never give.
|
|
451
|
+
"""
|
|
452
|
+
|
|
453
|
+
expected: dict[str, Any] = {
|
|
454
|
+
"schema_version": STUDIO_SCHEMA_VERSION,
|
|
455
|
+
"run_id": str(activated.run_id),
|
|
456
|
+
"workspace_id": str(activated.workspace_id),
|
|
457
|
+
"dataset_id": str(activated.dataset_id),
|
|
458
|
+
"table_id": str(activated.table_id),
|
|
459
|
+
"table_recipe_id": activated.table_recipe_id,
|
|
460
|
+
**dict(activated.recipe),
|
|
461
|
+
}
|
|
462
|
+
drifted = sorted(name for name, value in expected.items() if run.get(name) != value)
|
|
463
|
+
if drifted:
|
|
464
|
+
raise HostedDeployError(
|
|
465
|
+
"DEPLOY_RUN_BINDING",
|
|
466
|
+
"the run status answer is not bound to this deployment: " + ", ".join(drifted),
|
|
467
|
+
)
|
|
468
|
+
status = run.get("status")
|
|
469
|
+
if not isinstance(status, str) or status not in RUN_STATES:
|
|
470
|
+
raise HostedDeployError(
|
|
471
|
+
"DEPLOY_RUN_STATUS_UNKNOWN",
|
|
472
|
+
f"Studio reported a run state this release does not know: {status!r}",
|
|
473
|
+
)
|
|
474
|
+
for name in ("created_at", "started_at", "completed_at"):
|
|
475
|
+
if name in run:
|
|
476
|
+
_utc(run, name)
|
|
477
|
+
|
|
478
|
+
|
|
479
|
+
def _payload(
|
|
480
|
+
activated: ActivatedRun,
|
|
481
|
+
run: Mapping[str, Any],
|
|
482
|
+
observed: list[str],
|
|
483
|
+
*,
|
|
484
|
+
timed_out: bool,
|
|
485
|
+
watch_seconds: int | None,
|
|
486
|
+
cloud_url: str,
|
|
487
|
+
) -> dict[str, Any]:
|
|
488
|
+
"""One answer, every fact in it read from Studio's response or from the journal."""
|
|
489
|
+
|
|
490
|
+
status = str(run["status"])
|
|
491
|
+
state = RUN_STATES[status]
|
|
492
|
+
duration = _duration_seconds(run)
|
|
493
|
+
payload: dict[str, Any] = {
|
|
494
|
+
"schema_version": RUN_STATUS_SCHEMA,
|
|
495
|
+
"status": "run_watch_timed_out" if timed_out else "run_status_reported",
|
|
496
|
+
"run_status": status,
|
|
497
|
+
"run_disposition": state.disposition,
|
|
498
|
+
"terminal": state.terminal,
|
|
499
|
+
"means": state.means,
|
|
500
|
+
"next": _next_action(state, activated, timed_out=timed_out, watch_seconds=watch_seconds),
|
|
501
|
+
"workspace_id": str(activated.workspace_id),
|
|
502
|
+
"dataset_id": str(activated.dataset_id),
|
|
503
|
+
"table_id": str(activated.table_id),
|
|
504
|
+
"table_recipe_id": activated.table_recipe_id,
|
|
505
|
+
"run_id": str(activated.run_id),
|
|
506
|
+
"run_kind": run.get("kind"),
|
|
507
|
+
"states_seen": list(observed),
|
|
508
|
+
"created_at": run.get("created_at"),
|
|
509
|
+
}
|
|
510
|
+
for name in ("started_at", "completed_at", "failure_code", "repair_task_id"):
|
|
511
|
+
if run.get(name) is not None:
|
|
512
|
+
payload[name] = run[name]
|
|
513
|
+
if duration is not None:
|
|
514
|
+
# The only number here, and it is a subtraction of two timestamps Studio persisted. It is
|
|
515
|
+
# absent whenever either of them is, because the alternative -- measuring from this
|
|
516
|
+
# command's own clock -- would report how long somebody had been watching as though it
|
|
517
|
+
# were how long the run had taken.
|
|
518
|
+
payload["run_seconds"] = duration
|
|
519
|
+
if watch_seconds is not None:
|
|
520
|
+
payload["watch_limit_seconds"] = watch_seconds
|
|
521
|
+
if activated.cloud_dataset_id is not None:
|
|
522
|
+
payload["cloud_dataset_id"] = activated.cloud_dataset_id
|
|
523
|
+
if activated.current_path is not None:
|
|
524
|
+
payload["current_path"] = activated.current_path
|
|
525
|
+
payload["schedule"] = dict(activated.schedule)
|
|
526
|
+
payload["state"] = str(activated.state_path)
|
|
527
|
+
payload["dashboard_url"] = cloud_dashboard_url(cloud_url, "datasets")
|
|
528
|
+
return payload
|
|
529
|
+
|
|
530
|
+
|
|
531
|
+
def _next_action(
|
|
532
|
+
state: RunState,
|
|
533
|
+
activated: ActivatedRun,
|
|
534
|
+
*,
|
|
535
|
+
timed_out: bool,
|
|
536
|
+
watch_seconds: int | None,
|
|
537
|
+
) -> str:
|
|
538
|
+
"""What to do about this state, given what this deployment actually recorded.
|
|
539
|
+
|
|
540
|
+
Only two cases differ from the state's own written sentence, and both are cases where that
|
|
541
|
+
sentence would name something the reader does not have: a watch that ran out has not reached
|
|
542
|
+
the state its answer describes, and a released version cannot be read at a path this journal
|
|
543
|
+
never got as far as recording.
|
|
544
|
+
"""
|
|
545
|
+
|
|
546
|
+
if timed_out:
|
|
547
|
+
# The limit, named as the limit. How long the watch really took is a measurement nobody
|
|
548
|
+
# here made, and this module does not print those.
|
|
549
|
+
return (
|
|
550
|
+
f"this watch reached its limit of {watch_seconds} seconds and the run had not "
|
|
551
|
+
"finished; run this command again to pick the same run back up"
|
|
552
|
+
)
|
|
553
|
+
if state.disposition == RELEASED and activated.current_path is None:
|
|
554
|
+
return "open this Dataset in Cloud; this deployment recorded no current path for it"
|
|
555
|
+
return state.next
|
|
556
|
+
|
|
557
|
+
|
|
558
|
+
def _duration_seconds(run: Mapping[str, Any]) -> int | None:
|
|
559
|
+
"""How long the Run took, from Studio's two persisted timestamps and from nothing else.
|
|
560
|
+
|
|
561
|
+
A pair that runs backwards is no measurement, so it is reported as none at all rather than as
|
|
562
|
+
a negative number of seconds -- the same rule as an absent timestamp, for the same reason.
|
|
563
|
+
"""
|
|
564
|
+
|
|
565
|
+
if run.get("started_at") is None or run.get("completed_at") is None:
|
|
566
|
+
return None
|
|
567
|
+
seconds = (_utc(run, "completed_at") - _utc(run, "started_at")).total_seconds()
|
|
568
|
+
return None if seconds < 0 else round(seconds)
|
|
569
|
+
|
|
570
|
+
|
|
571
|
+
def _utc(run: Mapping[str, Any], field: str) -> datetime:
|
|
572
|
+
value = run.get(field)
|
|
573
|
+
try:
|
|
574
|
+
parsed = datetime.fromisoformat(str(value).replace("Z", "+00:00"))
|
|
575
|
+
except ValueError as error:
|
|
576
|
+
raise HostedDeployError(
|
|
577
|
+
"DEPLOY_RUN_BINDING", f"the run status answer carries an unreadable {field}"
|
|
578
|
+
) from error
|
|
579
|
+
if parsed.tzinfo is None or parsed.utcoffset() != UTC.utcoffset(parsed):
|
|
580
|
+
raise HostedDeployError(
|
|
581
|
+
"DEPLOY_RUN_BINDING", f"the run status answer carries a {field} that is not UTC"
|
|
582
|
+
)
|
|
583
|
+
return parsed
|
|
584
|
+
|
|
585
|
+
|
|
586
|
+
def _journaled_uuid(value: Any, field: str) -> UUID:
|
|
587
|
+
selected = value.get(field) if isinstance(value, Mapping) else None
|
|
588
|
+
try:
|
|
589
|
+
parsed = UUID(selected) if isinstance(selected, str) else None
|
|
590
|
+
except ValueError:
|
|
591
|
+
parsed = None
|
|
592
|
+
if parsed is None or str(parsed) != selected:
|
|
593
|
+
raise HostedDeployError(
|
|
594
|
+
"DEPLOY_STATE_INVALID", f"the deployment journal records no usable {field}"
|
|
595
|
+
)
|
|
596
|
+
return parsed
|
|
597
|
+
|
|
598
|
+
|
|
599
|
+
def _journaled_identifier(value: Any, field: str) -> str:
|
|
600
|
+
selected = value.get(field) if isinstance(value, Mapping) else None
|
|
601
|
+
if (
|
|
602
|
+
not isinstance(selected, str)
|
|
603
|
+
or not selected
|
|
604
|
+
or not selected[0].isalnum()
|
|
605
|
+
or len(selected) > 128
|
|
606
|
+
or any(
|
|
607
|
+
not (character.isascii() and (character.isalnum() or character in "_.-"))
|
|
608
|
+
for character in selected
|
|
609
|
+
)
|
|
610
|
+
):
|
|
611
|
+
raise HostedDeployError(
|
|
612
|
+
"DEPLOY_STATE_INVALID", f"the deployment journal records no usable {field}"
|
|
613
|
+
)
|
|
614
|
+
return selected
|
|
615
|
+
|
|
616
|
+
|
|
617
|
+
def _optional_text(value: Mapping[str, Any], field: str) -> str | None:
|
|
618
|
+
selected = value.get(field)
|
|
619
|
+
return selected if isinstance(selected, str) and selected else None
|