mostlyright-data 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mostlyright/data_harness/__init__.py +158 -0
- mostlyright/data_harness/acquisition/__init__.py +55 -0
- mostlyright/data_harness/acquisition/http.py +2773 -0
- mostlyright/data_harness/acquisition/parsing.py +809 -0
- mostlyright/data_harness/acquisition/ranges.py +495 -0
- mostlyright/data_harness/acquisition/result_download.py +360 -0
- mostlyright/data_harness/acquisition/retention_admission.py +248 -0
- mostlyright/data_harness/acquisition/sandbox.py +4888 -0
- mostlyright/data_harness/acquisition/url_policy.py +530 -0
- mostlyright/data_harness/agent_runtime.py +2743 -0
- mostlyright/data_harness/assets/logo-ink.svg +31 -0
- mostlyright/data_harness/backends/__init__.py +28 -0
- mostlyright/data_harness/backends/pandas_backend.py +350 -0
- mostlyright/data_harness/backends/polars_backend.py +366 -0
- mostlyright/data_harness/backends/protocol.py +124 -0
- mostlyright/data_harness/backends/reference.py +83 -0
- mostlyright/data_harness/backends/registry.py +55 -0
- mostlyright/data_harness/backends/restrictions.py +126 -0
- mostlyright/data_harness/canonical.py +333 -0
- mostlyright/data_harness/catalog_job.py +625 -0
- mostlyright/data_harness/cli.py +5398 -0
- mostlyright/data_harness/contracts.py +53 -0
- mostlyright/data_harness/coordinator.py +1307 -0
- mostlyright/data_harness/deploy.py +924 -0
- mostlyright/data_harness/deploy_target.py +312 -0
- mostlyright/data_harness/deployment_evidence.py +1067 -0
- mostlyright/data_harness/event_presentation.py +576 -0
- mostlyright/data_harness/events.py +2152 -0
- mostlyright/data_harness/fast_delimited.py +239 -0
- mostlyright/data_harness/fleet.py +237 -0
- mostlyright/data_harness/formats.py +236 -0
- mostlyright/data_harness/governors.py +1163 -0
- mostlyright/data_harness/hosted_bootstrap.py +972 -0
- mostlyright/data_harness/hosted_crawler.py +1115 -0
- mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
- mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
- mostlyright/data_harness/hosted_crawler_job.py +1277 -0
- mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
- mostlyright/data_harness/hosted_dataset.py +1500 -0
- mostlyright/data_harness/hosted_deploy.py +3037 -0
- mostlyright/data_harness/hosted_handoff.py +62 -0
- mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
- mostlyright/data_harness/hosted_ingestion_job.py +356 -0
- mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
- mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
- mostlyright/data_harness/hosted_session_worker.py +3554 -0
- mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
- mostlyright/data_harness/hosted_worker.py +6784 -0
- mostlyright/data_harness/ingestion/__init__.py +56 -0
- mostlyright/data_harness/ingestion/contracts.py +461 -0
- mostlyright/data_harness/ingestion/faults.py +42 -0
- mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
- mostlyright/data_harness/ingestion/spool.py +130 -0
- mostlyright/data_harness/ingestion/store.py +885 -0
- mostlyright/data_harness/key_seam.py +434 -0
- mostlyright/data_harness/linux_process_boundary.py +262 -0
- mostlyright/data_harness/local_contracts.py +2880 -0
- mostlyright/data_harness/local_search/__init__.py +5 -0
- mostlyright/data_harness/local_search/build_index.py +1087 -0
- mostlyright/data_harness/local_search/contracts.py +920 -0
- mostlyright/data_harness/local_search/query_trace.py +266 -0
- mostlyright/data_harness/local_search/retrieval.py +700 -0
- mostlyright/data_harness/local_search/sealed.py +474 -0
- mostlyright/data_harness/local_search/service.py +784 -0
- mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
- mostlyright/data_harness/nbrender/__init__.py +12 -0
- mostlyright/data_harness/nbrender/chrome.py +359 -0
- mostlyright/data_harness/nbrender/code_body.py +266 -0
- mostlyright/data_harness/nbrender/document.py +407 -0
- mostlyright/data_harness/nbrender/frame.py +275 -0
- mostlyright/data_harness/nbrender/interactive.py +337 -0
- mostlyright/data_harness/nbrender/markdown_body.py +477 -0
- mostlyright/data_harness/nbrender/mr_components.py +134 -0
- mostlyright/data_harness/nbrender/outputs_data.py +595 -0
- mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
- mostlyright/data_harness/nbrender/outputs_source.py +260 -0
- mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
- mostlyright/data_harness/nbrender/outputs_text.py +400 -0
- mostlyright/data_harness/nbrender/parse.py +394 -0
- mostlyright/data_harness/nbrender/status.py +40 -0
- mostlyright/data_harness/nbrender/tokens.py +1295 -0
- mostlyright/data_harness/notebook.py +1710 -0
- mostlyright/data_harness/offline.py +2049 -0
- mostlyright/data_harness/operation_registry.py +1007 -0
- mostlyright/data_harness/operator_setup.py +239 -0
- mostlyright/data_harness/pipeline.py +6428 -0
- mostlyright/data_harness/plan_graph.py +2026 -0
- mostlyright/data_harness/preparation/__init__.py +104 -0
- mostlyright/data_harness/preparation/contracts.py +1017 -0
- mostlyright/data_harness/preparation/engine.py +221 -0
- mostlyright/data_harness/preparation/errors.py +14 -0
- mostlyright/data_harness/preparation/gates.py +751 -0
- mostlyright/data_harness/preparation/joins.py +574 -0
- mostlyright/data_harness/preparation/profile.py +384 -0
- mostlyright/data_harness/preparation/table.py +217 -0
- mostlyright/data_harness/preparation/transforms.py +568 -0
- mostlyright/data_harness/progress_events.py +534 -0
- mostlyright/data_harness/readers/__init__.py +46 -0
- mostlyright/data_harness/readers/containers.py +963 -0
- mostlyright/data_harness/readers/contracts.py +542 -0
- mostlyright/data_harness/readers/delimited.py +257 -0
- mostlyright/data_harness/readers/grib2/__init__.py +33 -0
- mostlyright/data_harness/readers/grib2/admission.py +722 -0
- mostlyright/data_harness/readers/grib2/decode.py +1009 -0
- mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
- mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
- mostlyright/data_harness/readers/json_tabular.py +485 -0
- mostlyright/data_harness/readers/registry.py +514 -0
- mostlyright/data_harness/readers/samples/README.md +110 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples.py +582 -0
- mostlyright/data_harness/readers/spreadsheet.py +803 -0
- mostlyright/data_harness/readers/tabular.py +510 -0
- mostlyright/data_harness/recipe.py +5321 -0
- mostlyright/data_harness/repair/__init__.py +78 -0
- mostlyright/data_harness/repair/adapters.py +274 -0
- mostlyright/data_harness/repair/contracts.py +872 -0
- mostlyright/data_harness/repair/coordinator.py +1099 -0
- mostlyright/data_harness/repair/errors.py +16 -0
- mostlyright/data_harness/review.py +2533 -0
- mostlyright/data_harness/rowset.py +283 -0
- mostlyright/data_harness/serving.py +1975 -0
- mostlyright/data_harness/serving_edge.py +590 -0
- mostlyright/data_harness/serving_http.py +1031 -0
- mostlyright/data_harness/session_probes.py +759 -0
- mostlyright/data_harness/signing.py +101 -0
- mostlyright/data_harness/source_discovery.py +898 -0
- mostlyright/data_harness/sources/__init__.py +209 -0
- mostlyright/data_harness/sources/_adapter_steps.py +213 -0
- mostlyright/data_harness/sources/adapters.py +1214 -0
- mostlyright/data_harness/sources/cadence.py +1428 -0
- mostlyright/data_harness/sources/cadence_emission.py +453 -0
- mostlyright/data_harness/sources/cadence_history.py +546 -0
- mostlyright/data_harness/sources/catalog/__init__.py +17 -0
- mostlyright/data_harness/sources/catalog/admission.py +477 -0
- mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
- mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
- mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
- mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
- mostlyright/data_harness/sources/catalog/channel.py +523 -0
- mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
- mostlyright/data_harness/sources/catalog/contracts.py +825 -0
- mostlyright/data_harness/sources/catalog/coverage.py +137 -0
- mostlyright/data_harness/sources/catalog/delta.py +1340 -0
- mostlyright/data_harness/sources/catalog/embedding.py +532 -0
- mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
- mostlyright/data_harness/sources/catalog/fill.py +3889 -0
- mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
- mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
- mostlyright/data_harness/sources/catalog/gating.py +374 -0
- mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
- mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
- mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
- mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
- mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
- mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
- mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
- mostlyright/data_harness/sources/catalog/health.py +447 -0
- mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
- mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
- mostlyright/data_harness/sources/catalog/neural.py +1618 -0
- mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
- mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
- mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
- mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
- mostlyright/data_harness/sources/catalog/recommend.py +171 -0
- mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
- mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
- mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
- mostlyright/data_harness/sources/catalog/sealed.py +560 -0
- mostlyright/data_harness/sources/catalog/search.py +230 -0
- mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
- mostlyright/data_harness/sources/catalog/update.py +891 -0
- mostlyright/data_harness/sources/collections.py +815 -0
- mostlyright/data_harness/sources/contracts.py +2223 -0
- mostlyright/data_harness/sources/deletion.py +761 -0
- mostlyright/data_harness/sources/fitness.py +162 -0
- mostlyright/data_harness/sources/governance.py +163 -0
- mostlyright/data_harness/sources/hosted.py +173 -0
- mostlyright/data_harness/sources/integration.py +218 -0
- mostlyright/data_harness/sources/range_reader.py +418 -0
- mostlyright/data_harness/sources/registry.py +514 -0
- mostlyright/data_harness/sources/rights_rule.py +59 -0
- mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
- mostlyright/data_harness/sources/sports.py +521 -0
- mostlyright/data_harness/sources/stream.py +524 -0
- mostlyright/data_harness/sources/stream_connector.py +418 -0
- mostlyright/data_harness/sources/stream_recorder.py +1404 -0
- mostlyright/data_harness/studio_boundary.py +2019 -0
- mostlyright/data_harness/thin/__init__.py +37 -0
- mostlyright/data_harness/thin/acquire.py +1137 -0
- mostlyright/data_harness/thin/acquire_cancel.py +579 -0
- mostlyright/data_harness/thin/approvals.py +617 -0
- mostlyright/data_harness/thin/commands.py +406 -0
- mostlyright/data_harness/thin/download.py +194 -0
- mostlyright/data_harness/thin/narrative.py +589 -0
- mostlyright/data_harness/thin/parity.py +1070 -0
- mostlyright/data_harness/thin/propose.py +2759 -0
- mostlyright/data_harness/thin/research.py +1663 -0
- mostlyright/data_harness/thin/router.py +924 -0
- mostlyright/data_harness/thin/runs.py +519 -0
- mostlyright/data_harness/thin/session.py +281 -0
- mostlyright/data_harness/thin/stream.py +501 -0
- mostlyright/data_harness/thin/transport.py +187 -0
- mostlyright/data_harness/thin/vocabulary.py +368 -0
- mostlyright/data_harness/thin/workers.py +164 -0
- mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
- mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
- mostlyright/data_harness/unit_flow.py +927 -0
- mostlyright/data_harness/units.py +572 -0
- mostlyright/data_harness/ux/__init__.py +9 -0
- mostlyright/data_harness/ux/approve.py +485 -0
- mostlyright/data_harness/ux/author_yaml.py +597 -0
- mostlyright/data_harness/ux/cloud_auth.py +447 -0
- mostlyright/data_harness/ux/commands/__init__.py +260 -0
- mostlyright/data_harness/ux/commands/approve.py +136 -0
- mostlyright/data_harness/ux/commands/auth.py +744 -0
- mostlyright/data_harness/ux/commands/author.py +79 -0
- mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
- mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
- mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
- mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
- mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
- mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
- mostlyright/data_harness/ux/commands/deploy.py +134 -0
- mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
- mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
- mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
- mostlyright/data_harness/ux/commands/diff.py +74 -0
- mostlyright/data_harness/ux/commands/index.py +84 -0
- mostlyright/data_harness/ux/commands/inventory.py +47 -0
- mostlyright/data_harness/ux/commands/list_builds.py +143 -0
- mostlyright/data_harness/ux/commands/login.py +63 -0
- mostlyright/data_harness/ux/commands/peek.py +236 -0
- mostlyright/data_harness/ux/commands/plan_check.py +90 -0
- mostlyright/data_harness/ux/commands/preflight.py +97 -0
- mostlyright/data_harness/ux/commands/record.py +107 -0
- mostlyright/data_harness/ux/commands/review_setup.py +47 -0
- mostlyright/data_harness/ux/commands/search.py +440 -0
- mostlyright/data_harness/ux/commands/show.py +61 -0
- mostlyright/data_harness/ux/commands/whoami.py +37 -0
- mostlyright/data_harness/ux/credential_native.py +551 -0
- mostlyright/data_harness/ux/credential_store.py +1055 -0
- mostlyright/data_harness/ux/credentials.py +631 -0
- mostlyright/data_harness/ux/diffing.py +444 -0
- mostlyright/data_harness/ux/headline.py +671 -0
- mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
- mostlyright/data_harness/ux/hosted_run_status.py +619 -0
- mostlyright/data_harness/ux/inventory.py +427 -0
- mostlyright/data_harness/ux/local_review.py +375 -0
- mostlyright/data_harness/ux/login.py +691 -0
- mostlyright/data_harness/ux/path_kind.py +147 -0
- mostlyright/data_harness/ux/peek.py +1000 -0
- mostlyright/data_harness/ux/plain_file.py +178 -0
- mostlyright/data_harness/ux/plan_check.py +311 -0
- mostlyright/data_harness/ux/preflight.py +918 -0
- mostlyright/data_harness/ux/readers.py +1124 -0
- mostlyright/data_harness/ux/remediation.py +2195 -0
- mostlyright/data_harness/ux/render.py +657 -0
- mostlyright/data_harness/ux/workload.py +1077 -0
- mostlyright/data_harness/viewer.py +3713 -0
- mostlyright/data_harness/visual_run/__init__.py +83 -0
- mostlyright/data_harness/visual_run/authoring.py +235 -0
- mostlyright/data_harness/visual_run/contracts.py +673 -0
- mostlyright/data_harness/visual_run/materialize.py +486 -0
- mostlyright/data_harness/visual_run/observations.py +874 -0
- mostlyright/data_harness/visual_run/query.py +259 -0
- mostlyright/data_harness/visual_run/reducer.py +280 -0
- mostlyright/data_harness/visual_run/sdk.py +892 -0
- mostlyright/data_harness/visual_run/store.py +584 -0
- mostlyright/data_harness/visual_run/transport.py +239 -0
- mostlyright/data_harness/watch.py +2999 -0
- mostlyright_data-0.9.0.dist-info/METADATA +607 -0
- mostlyright_data-0.9.0.dist-info/RECORD +314 -0
- mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
- mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
|
@@ -0,0 +1,2759 @@
|
|
|
1
|
+
"""``mr-data propose``: the hosted authoring front door ADR 0021 left open.
|
|
2
|
+
|
|
3
|
+
WHAT WAS MISSING. ADR 0021 makes the product hosted-only and the client thin, but the only thing
|
|
4
|
+
that created a recipe proposal was the LOCAL lane -- ``mr-data deploy`` from a sealed local Build,
|
|
5
|
+
through :mod:`mostlyright.data_harness.hosted_deploy`, which imports pyarrow, croniter and the
|
|
6
|
+
generated client and is forbidden in the thin profile. ``mr-data build --recipe-proposal ID
|
|
7
|
+
--recipe-digest D`` therefore needed a proposal that nothing thin could make. This command makes
|
|
8
|
+
one, over the same V3 routes ``hosted_deploy`` drives, in pure Python and urllib only.
|
|
9
|
+
|
|
10
|
+
THE DOCUMENTS AN AGENT AUTHORS. A bundle directory holds the documents the hosted API needs, each
|
|
11
|
+
in the exact shape ``docs/HOSTED-AUTHORING.md`` records and each checked here, against that shape,
|
|
12
|
+
BEFORE a credential is minted or a request is sent:
|
|
13
|
+
|
|
14
|
+
* ``dataset.json`` -> ``POST /v3/datasets``
|
|
15
|
+
* ``question.json`` -> ``POST /v3/questions`` then ``PUT /v3/questions/{id}/requirements``
|
|
16
|
+
* ``sources.json`` (a list) -> one ``POST /v3/sources`` and one
|
|
17
|
+
``POST /v3/connector-configurations`` per entry
|
|
18
|
+
* ``table_plan.json`` -> ``POST /v3/table-plans``
|
|
19
|
+
* ``recipe.json`` -> the recipe document itself, whose coordinates become the proposal's and whose
|
|
20
|
+
canonical bytes are the ``recipe_digest``
|
|
21
|
+
* ``activation.json`` -> the ``activation_intent`` sealed into the proposal
|
|
22
|
+
|
|
23
|
+
Each of the first four may instead ADOPT a resource that already exists, by naming its id
|
|
24
|
+
(``{"dataset_id": "..."}``), so a second bundle can attach to what a first one registered.
|
|
25
|
+
|
|
26
|
+
WHY ``--through sources`` EXISTS. A recipe names its sources by the Studio source id, which does
|
|
27
|
+
not exist until the source is registered. So an agent runs ``propose --through sources`` FIRST to
|
|
28
|
+
register the Dataset, Question and Sources, opens a research session against them
|
|
29
|
+
(``mr-data research-open --bundle``), probes, and only then writes ``recipe.json`` referencing the
|
|
30
|
+
registered ids and re-runs ``propose`` to finish. The journal is what carries the registered ids
|
|
31
|
+
across the two runs.
|
|
32
|
+
|
|
33
|
+
WHAT THE JOURNAL PINS, AND WHEN. A document's digest is pinned the moment the first resource built
|
|
34
|
+
from it completes -- ``dataset.json`` when the Dataset exists, ``question.json`` when the
|
|
35
|
+
requirements are written, ``sources.json`` when the last connector is registered,
|
|
36
|
+
``table_plan.json`` when the plan exists, ``recipe.json`` and ``activation.json`` when the
|
|
37
|
+
proposal exists. Until then a
|
|
38
|
+
document is free to change: a refused document is corrected and re-run, not abandoned. After it,
|
|
39
|
+
a changed document is refused with the document named, because the resource Studio holds was built
|
|
40
|
+
from the earlier one and a proposal a person is asked to approve must be the proposal that was
|
|
41
|
+
authored.
|
|
42
|
+
|
|
43
|
+
WHAT THIS COMMAND NEVER DOES. It records an approval ASK, exactly as ``mr-data approve`` does; it
|
|
44
|
+
never grants one. A Table recipe is confirmed from a signed-in editor session in the dashboard,
|
|
45
|
+
which a Cloud-minted service token cannot be, so the final receipt names the proposal, the recipe
|
|
46
|
+
digest, the approval request, the dashboard address a person opens, and the two commands that
|
|
47
|
+
follow -- ``mr-data recipe-approve`` (the ask a person settles) and then ``mr-data build``. It does
|
|
48
|
+
NOT validate the transform rules inside the recipe: that is the local lane's job and imports the
|
|
49
|
+
modules this profile forbids. Studio validates what Studio validates, and the hosted worker refuses
|
|
50
|
+
a recipe its engine cannot run.
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
from __future__ import annotations
|
|
54
|
+
|
|
55
|
+
import argparse
|
|
56
|
+
import hashlib
|
|
57
|
+
import json
|
|
58
|
+
import os
|
|
59
|
+
import re
|
|
60
|
+
import secrets
|
|
61
|
+
import stat
|
|
62
|
+
from collections.abc import Callable, Mapping, Sequence
|
|
63
|
+
from pathlib import Path
|
|
64
|
+
from typing import Any
|
|
65
|
+
from uuid import UUID
|
|
66
|
+
|
|
67
|
+
from mostlyright.data_harness import package_version
|
|
68
|
+
from mostlyright.data_harness.canonical import (
|
|
69
|
+
CanonicalJSONError,
|
|
70
|
+
canonical_json_bytes,
|
|
71
|
+
canonical_sha256,
|
|
72
|
+
parse_json,
|
|
73
|
+
)
|
|
74
|
+
from mostlyright.data_harness.formats import PARSER_FORMAT_ORDER
|
|
75
|
+
|
|
76
|
+
# The externally enforced public-crawler egress policies, as their canonical attestations. Reused
|
|
77
|
+
# rather than recomputed here for the reason the acquire lane reuses `session_probes`: a second
|
|
78
|
+
# copy of a security-critical constant is a second thing to keep in step, and this one is the
|
|
79
|
+
# digest Studio's own `STUDIO_PUBLIC_HTTPS_EGRESS_POLICY_ATTESTATION` is set to at deploy time.
|
|
80
|
+
from mostlyright.data_harness.hosted_crawler_protocol import (
|
|
81
|
+
OPENLIGADB_EGRESS_POLICY_ATTESTATION,
|
|
82
|
+
PUBLIC_HTTPS_EGRESS_POLICY_ATTESTATION,
|
|
83
|
+
)
|
|
84
|
+
from mostlyright.data_harness.thin import THIN_SCHEMA_PREFIX
|
|
85
|
+
from mostlyright.data_harness.thin.approvals import GET_APPROVAL_PATH
|
|
86
|
+
from mostlyright.data_harness.thin.research import GET_SESSION_PATH, RetryAfterRefusal
|
|
87
|
+
from mostlyright.data_harness.thin.runs import (
|
|
88
|
+
GET_RECIPE_PROPOSAL_PATH,
|
|
89
|
+
StudioApiClient,
|
|
90
|
+
)
|
|
91
|
+
from mostlyright.data_harness.thin.session import (
|
|
92
|
+
StudioSession,
|
|
93
|
+
approval_dashboard_url,
|
|
94
|
+
dataset_dashboard_url,
|
|
95
|
+
open_studio_session,
|
|
96
|
+
run_dashboard_url,
|
|
97
|
+
)
|
|
98
|
+
from mostlyright.data_harness.thin.transport import ThinLaneError
|
|
99
|
+
from mostlyright.data_harness.ux.credentials import resolve_cloud_credentials
|
|
100
|
+
from mostlyright.data_harness.ux.path_kind import UNKNOWN_KIND, presence_at
|
|
101
|
+
from mostlyright.data_harness.ux.plain_file import PlainFileRefusal, read_plain_file
|
|
102
|
+
|
|
103
|
+
PROPOSE_SCHEMA = f"{THIN_SCHEMA_PREFIX}-propose.v1"
|
|
104
|
+
JOURNAL_SCHEMA = f"{THIN_SCHEMA_PREFIX}-propose-journal.v1"
|
|
105
|
+
|
|
106
|
+
#: The V3 wire this lane speaks. The creates are each bound to an idempotency key, the one PUT
|
|
107
|
+
#: upserts requirements, and the GETs either adopt a resource that already exists or read the
|
|
108
|
+
#: proposal back for the ETag its approval is pinned to. None of them approves, activates,
|
|
109
|
+
#: releases or decides. They are the exact routes :mod:`mostlyright.data_harness.hosted_deploy`
|
|
110
|
+
#: drives, written out so a reviewer diffs them against ``contracts/openapi/studio-v3.yaml``.
|
|
111
|
+
CREATE_DATASET_PATH = "/v3/datasets"
|
|
112
|
+
GET_DATASET_PATH = "/v3/datasets/{dataset_id}"
|
|
113
|
+
CREATE_QUESTION_PATH = "/v3/questions"
|
|
114
|
+
GET_QUESTION_PATH = "/v3/questions/{question_id}"
|
|
115
|
+
PUT_REQUIREMENTS_PATH = "/v3/questions/{question_id}/requirements"
|
|
116
|
+
GET_REQUIREMENTS_PATH = "/v3/questions/{question_id}/requirements"
|
|
117
|
+
REGISTER_SOURCE_PATH = "/v3/sources"
|
|
118
|
+
GET_SOURCE_PATH = "/v3/sources/{source_id}"
|
|
119
|
+
REGISTER_CONNECTOR_PATH = "/v3/connector-configurations"
|
|
120
|
+
GET_CONNECTOR_PATH = "/v3/connector-configurations/{connector_configuration_id}"
|
|
121
|
+
CREATE_TABLE_PLAN_PATH = "/v3/table-plans"
|
|
122
|
+
GET_TABLE_PLAN_PATH = "/v3/table-plans/{table_plan_id}"
|
|
123
|
+
WORKER_POLICY_PREFLIGHT_PATH = "/v3/worker-policy-preflights"
|
|
124
|
+
CREATE_RECIPE_PROPOSAL_PATH = "/v3/recipe-proposals"
|
|
125
|
+
#: ``createRecipeProposal`` answers 201 with NO ETag -- Studio declares no ``etag_kind`` on that
|
|
126
|
+
#: route, and an idempotent replay never carries one either -- so the proposal is read back,
|
|
127
|
+
#: where the ETag is, before the approval is pinned to it (``runs.GET_RECIPE_PROPOSAL_PATH``, the
|
|
128
|
+
#: read ``build`` shares). A real run against Studio is what found this: a fake that invented an
|
|
129
|
+
#: ETag on the create kept every test green over a lane that could never open its approval.
|
|
130
|
+
REQUEST_RECIPE_APPROVAL_PATH = "/v3/recipe-proposals/{recipe_proposal_id}/approval"
|
|
131
|
+
|
|
132
|
+
#: The V3 contract version every command body carries.
|
|
133
|
+
STUDIO_SCHEMA_VERSION = "3.0.0"
|
|
134
|
+
|
|
135
|
+
#: The media type a recipe travels under, and the recipe schema versions Studio's
|
|
136
|
+
#: ``recipe_schema_version`` enum admits. Held here so a document naming a version Studio would
|
|
137
|
+
#: refuse is a sentence at the terminal rather than a 422 three routes in.
|
|
138
|
+
RECIPE_MEDIA_TYPE = "application/vnd.mostlyright.frozen-recipe+json"
|
|
139
|
+
RECIPE_SCHEMA_VERSIONS = ("frozen-recipe.v1", "frozen-recipe.v3")
|
|
140
|
+
|
|
141
|
+
#: The one bootstrap origin this lane sends. There is no local Build behind a hosted-first Recipe,
|
|
142
|
+
#: so the first hosted Build is the bootstrap and the in-product approval plus the hosted
|
|
143
|
+
#: independent check are the gates. Studio admits the hosted shape only on a first Recipe
|
|
144
|
+
#: (``recipe_version`` 1 with a null predecessor), which this lane refuses locally before it can
|
|
145
|
+
#: become a 422.
|
|
146
|
+
HOSTED_BOOTSTRAP_ORIGIN = "hosted_first_build"
|
|
147
|
+
|
|
148
|
+
#: The two credential-free public connector adapters, and the egress attestation each one carries.
|
|
149
|
+
#: A source outside this map is a credentialed or sealed-artifact source, which the hosted
|
|
150
|
+
#: authoring front door does not register -- those enter through the local lane's staged uploads.
|
|
151
|
+
CONNECTOR_EGRESS_ATTESTATION = {
|
|
152
|
+
"public.https": PUBLIC_HTTPS_EGRESS_POLICY_ATTESTATION,
|
|
153
|
+
"external.openligadb": OPENLIGADB_EGRESS_POLICY_ATTESTATION,
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
#: The stages ``--through`` may stop after, in the order a bundle is built. ``approval`` is the
|
|
157
|
+
#: whole run and the default.
|
|
158
|
+
STAGES = ("dataset", "question", "sources", "plan", "proposal", "approval")
|
|
159
|
+
|
|
160
|
+
#: The bundle documents, and the stage each is first read for. A document is refused as missing
|
|
161
|
+
#: only when a stage that needs it is reached, so ``--through sources`` never asks for a
|
|
162
|
+
#: ``recipe.json`` that has not been written yet. ``recipe.json`` is read for the plan stage as
|
|
163
|
+
#: well as the proposal, because the plan's digests describe the recipe.
|
|
164
|
+
DOCUMENTS = {
|
|
165
|
+
"dataset": ("dataset.json",),
|
|
166
|
+
"question": ("question.json",),
|
|
167
|
+
"sources": ("sources.json",),
|
|
168
|
+
"plan": ("table_plan.json", "recipe.json"),
|
|
169
|
+
"proposal": ("activation.json",),
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
#: Which documents each stage pins when it completes. A pin is the statement "a resource Studio
|
|
173
|
+
#: holds was built from this exact document"; a stage that has not completed has made no such
|
|
174
|
+
#: statement, and its documents stay editable.
|
|
175
|
+
PINS = {
|
|
176
|
+
"dataset": ("dataset.json",),
|
|
177
|
+
"question": ("question.json",),
|
|
178
|
+
"sources": ("sources.json",),
|
|
179
|
+
"plan": ("table_plan.json",),
|
|
180
|
+
"proposal": ("recipe.json", "activation.json"),
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
#: A bundle document is read whole under the same open-then-fstat rule every named path is read
|
|
184
|
+
#: under, and one four times the canonical recipe ceiling is refused before it is parsed.
|
|
185
|
+
MAX_DOCUMENT_BYTES = 4 * 262_144
|
|
186
|
+
|
|
187
|
+
#: The journal embeds every resource Studio answered with -- the proposal, with its canonical
|
|
188
|
+
#: recipe bytes, among them -- so its ceiling is above a single document's.
|
|
189
|
+
MAX_JOURNAL_BYTES = 4 * MAX_DOCUMENT_BYTES
|
|
190
|
+
|
|
191
|
+
#: ``canonical_recipe_json`` is bounded by the contract at 262144 UTF-8 bytes. Checked here so a
|
|
192
|
+
#: recipe that canonicalises larger is a sentence rather than a 422.
|
|
193
|
+
MAX_CANONICAL_RECIPE_BYTES = 262_144
|
|
194
|
+
|
|
195
|
+
#: Studio's ``Retry-After`` is delta-seconds. Bounded the way the research lane bounds its own: a
|
|
196
|
+
#: window this lane does not understand is reported as no window rather than as a number nobody
|
|
197
|
+
#: checked.
|
|
198
|
+
MAX_RETRY_AFTER_SECONDS = 3_600
|
|
199
|
+
|
|
200
|
+
#: The journal directory and file, created 0700 and 0600 so a bundle checked out on a shared
|
|
201
|
+
#: machine does not leave a workspace's created-resource ids world-readable.
|
|
202
|
+
JOURNAL_DIR = ".mr-data"
|
|
203
|
+
JOURNAL_FILE = "propose.journal.json"
|
|
204
|
+
|
|
205
|
+
#: The schedule a bundle gets when ``activation.json`` leaves one out: daily at midnight UTC, one
|
|
206
|
+
#: run at a time -- the same defaults ``mr-data deploy`` applies.
|
|
207
|
+
DEFAULT_SCHEDULE = {
|
|
208
|
+
"cron": "0 0 * * *",
|
|
209
|
+
"timezone": "UTC",
|
|
210
|
+
"mode": "incremental_refresh",
|
|
211
|
+
"max_concurrent_runs": 1,
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
# The structural rules Studio's JSON Schema files declare for the bodies this lane sends, restated
|
|
215
|
+
# here so a document is refused at the terminal in the words of the document rather than three
|
|
216
|
+
# routes in with a JSON pointer. Every enum and pattern is the contract's own, and the cross-repo
|
|
217
|
+
# gate in `tests/test_thin_propose.py` holds every body built from them against the schema files.
|
|
218
|
+
_HARNESS_VERSION = re.compile(r"^[0-9]+\.[0-9]+\.[0-9]+$")
|
|
219
|
+
_UUID_TEXT = re.compile(
|
|
220
|
+
r"^[0-9a-f]{8}-[0-9a-f]{4}-[1-8][0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$"
|
|
221
|
+
)
|
|
222
|
+
_PREFIXED_DIGEST = re.compile(r"^sha256:[0-9a-f]{64}$")
|
|
223
|
+
_BARE_DIGEST = re.compile(r"^[0-9a-f]{64}$")
|
|
224
|
+
_IDENTIFIER = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_.-]{0,127}$")
|
|
225
|
+
_COLUMN = re.compile(r"^[a-z][a-z0-9_]{0,62}$")
|
|
226
|
+
_SEMVER = re.compile(r"^[1-9][0-9]*\.[0-9]+\.[0-9]+$")
|
|
227
|
+
_TIMESTAMP = re.compile(
|
|
228
|
+
r"^(?:[0-9]{4})-(?:0[1-9]|1[0-2])-(?:0[1-9]|[12][0-9]|3[01])"
|
|
229
|
+
r"T(?:[01][0-9]|2[0-3]):[0-5][0-9]:[0-5][0-9](?:\.[0-9]{1,9})?Z$"
|
|
230
|
+
)
|
|
231
|
+
_HTTPS_URL = re.compile(r"^https://[^:/?#@]+(?:(?:/[^#]*)|(?:\?[^#]*))?$")
|
|
232
|
+
_FILENAME = re.compile(r"^[^/\\\x00-\x1f\x7f]+$")
|
|
233
|
+
|
|
234
|
+
_SOURCE_CLASSES = (
|
|
235
|
+
"mostlyright_sdk",
|
|
236
|
+
"external_adapter",
|
|
237
|
+
"user_file",
|
|
238
|
+
"user_url",
|
|
239
|
+
"user_api",
|
|
240
|
+
"database_extract",
|
|
241
|
+
"webhook",
|
|
242
|
+
"stream",
|
|
243
|
+
)
|
|
244
|
+
_LOCATOR_KINDS = ("sdk_connector", "https_url", "artifact", "webhook", "stream", "database")
|
|
245
|
+
_CLASSIFICATIONS = ("public", "internal", "confidential", "restricted")
|
|
246
|
+
_CLAIMED_BASES = (
|
|
247
|
+
"unknown",
|
|
248
|
+
"prohibited",
|
|
249
|
+
"permission_asserted",
|
|
250
|
+
"public_domain_asserted",
|
|
251
|
+
"contractual_license_asserted",
|
|
252
|
+
"terms_of_service_asserted",
|
|
253
|
+
)
|
|
254
|
+
_TARGET_POLICIES = ("none", "required_if_supportable")
|
|
255
|
+
_FEASIBILITY_DECISIONS = ("supportable", "supportable_with_limits", "unsupported")
|
|
256
|
+
_REASON_CODES = (
|
|
257
|
+
"SUFFICIENT_EVIDENCE",
|
|
258
|
+
"LIMITED_COVERAGE",
|
|
259
|
+
"UNACCEPTABLE_DELAY",
|
|
260
|
+
"RIGHTS_UNCLEAR",
|
|
261
|
+
"NO_RELIABLE_SOURCE",
|
|
262
|
+
"TARGET_NOT_OBSERVABLE",
|
|
263
|
+
)
|
|
264
|
+
_OVERLAP_POLICIES = ("disallow_released_coverage", "allow_within_envelope")
|
|
265
|
+
_BACKFILL_POLICY_SCHEMA = "local-backfill-policy.v1"
|
|
266
|
+
_QUERY_LIMIT_CEILINGS = {
|
|
267
|
+
"max_source_bytes": 268_435_456,
|
|
268
|
+
"max_normalized_bytes": 4_294_967_296,
|
|
269
|
+
"max_rows": 10_000_000,
|
|
270
|
+
"max_columns": 1_024,
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
class ProposeError(ThinLaneError):
|
|
275
|
+
"""A typed refusal from the authoring front door, carrying an operator action.
|
|
276
|
+
|
|
277
|
+
Distinct from a plain :class:`ThinLaneError` only in that it always names what to do next --
|
|
278
|
+
the document to fix, or the command to run first -- so a refusal an agent reads is actionable
|
|
279
|
+
rather than merely typed.
|
|
280
|
+
"""
|
|
281
|
+
|
|
282
|
+
def __init__(self, code: str, detail: str) -> None:
|
|
283
|
+
super().__init__(code, detail)
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
# --------------------------------------------------------------------------------------------
|
|
287
|
+
# The client
|
|
288
|
+
# --------------------------------------------------------------------------------------------
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
class StudioProposeClient(StudioApiClient):
|
|
292
|
+
"""The V3 authoring calls: creates under an idempotency key, and reads. Nothing decides.
|
|
293
|
+
|
|
294
|
+
``request_recipe_approval`` is the one that opens an approval, and it opens an ASK: Studio
|
|
295
|
+
writes one immutable request row and never a decision, which is why this lane can create the
|
|
296
|
+
thing ``mr-data recipe-approve`` later points a human at without ever being able to grant it.
|
|
297
|
+
"""
|
|
298
|
+
|
|
299
|
+
def create(
|
|
300
|
+
self,
|
|
301
|
+
path: str,
|
|
302
|
+
body: Mapping[str, Any],
|
|
303
|
+
*,
|
|
304
|
+
idempotency_key: str,
|
|
305
|
+
expected: int,
|
|
306
|
+
method: str = "POST",
|
|
307
|
+
if_match: str | None = None,
|
|
308
|
+
response_headers: dict[str, str] | None = None,
|
|
309
|
+
) -> dict[str, Any]:
|
|
310
|
+
headers = {"Idempotency-Key": idempotency_key}
|
|
311
|
+
if if_match is not None:
|
|
312
|
+
headers["If-Match"] = if_match
|
|
313
|
+
captured: dict[str, str] = {}
|
|
314
|
+
try:
|
|
315
|
+
result = self._call(
|
|
316
|
+
method,
|
|
317
|
+
path,
|
|
318
|
+
body=body,
|
|
319
|
+
extra_headers=headers,
|
|
320
|
+
expected=(expected,),
|
|
321
|
+
response_headers=captured,
|
|
322
|
+
)
|
|
323
|
+
except ThinLaneError as refusal:
|
|
324
|
+
raise _maybe_retry_after(refusal, captured) from refusal
|
|
325
|
+
if response_headers is not None:
|
|
326
|
+
response_headers.update(captured)
|
|
327
|
+
return result
|
|
328
|
+
|
|
329
|
+
def read(self, path: str, *, response_headers: dict[str, str] | None = None) -> dict[str, Any]:
|
|
330
|
+
"""One bounded GET of one Studio resource, with the ETag it answered under."""
|
|
331
|
+
|
|
332
|
+
captured: dict[str, str] = {}
|
|
333
|
+
try:
|
|
334
|
+
result = self._call("GET", path, expected=(200,), response_headers=captured)
|
|
335
|
+
except ThinLaneError as refusal:
|
|
336
|
+
raise _maybe_retry_after(refusal, captured) from refusal
|
|
337
|
+
if response_headers is not None:
|
|
338
|
+
response_headers.update(captured)
|
|
339
|
+
return result
|
|
340
|
+
|
|
341
|
+
def worker_policy_preflight(self, body: Mapping[str, Any]) -> dict[str, Any]:
|
|
342
|
+
"""Studio's current worker selection for this proposal, read without journaling a write.
|
|
343
|
+
|
|
344
|
+
The preflight is not a mutation: it resolves the deployed fleet's images and validation
|
|
345
|
+
policy for the recipe's mode, and the approval request below carries it back so Studio can
|
|
346
|
+
re-check it against the same immutable TablePlan. This lane never re-derives that policy --
|
|
347
|
+
that needs the engine this profile does not carry -- so it passes Studio's own answer
|
|
348
|
+
through, and Studio re-checks it at approval and at build.
|
|
349
|
+
"""
|
|
350
|
+
|
|
351
|
+
captured: dict[str, str] = {}
|
|
352
|
+
try:
|
|
353
|
+
return self._call(
|
|
354
|
+
"POST",
|
|
355
|
+
WORKER_POLICY_PREFLIGHT_PATH,
|
|
356
|
+
body=body,
|
|
357
|
+
expected=(200,),
|
|
358
|
+
response_headers=captured,
|
|
359
|
+
)
|
|
360
|
+
except ThinLaneError as refusal:
|
|
361
|
+
raise _maybe_retry_after(refusal, captured) from refusal
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
def _maybe_retry_after(refusal: ThinLaneError, headers: Mapping[str, str]) -> ThinLaneError:
|
|
365
|
+
"""Carry a ``Retry-After`` window on a refusal Studio stamped one on, never sleep through it.
|
|
366
|
+
|
|
367
|
+
A workspace Studio is rate-limiting answers with a window rather than a queue, and the honest
|
|
368
|
+
thing is to report the window and let the caller decide, the way the research lane reports a
|
|
369
|
+
full-workspace refusal. Any refusal that arrives with a usable ``Retry-After`` becomes a
|
|
370
|
+
:class:`RetryAfterRefusal` carrying the number the router already knows how to surface.
|
|
371
|
+
"""
|
|
372
|
+
|
|
373
|
+
raw = headers.get("retry-after", "")
|
|
374
|
+
if not raw.isdecimal():
|
|
375
|
+
return refusal
|
|
376
|
+
window = int(raw)
|
|
377
|
+
if not 1 <= window <= MAX_RETRY_AFTER_SECONDS:
|
|
378
|
+
return refusal
|
|
379
|
+
windowed = RetryAfterRefusal(
|
|
380
|
+
refusal.code,
|
|
381
|
+
f"{refusal.detail}; Studio asked this command to retry after {window} seconds. "
|
|
382
|
+
"Nothing further was created, and this command did not wait",
|
|
383
|
+
retry_after_seconds=window,
|
|
384
|
+
)
|
|
385
|
+
# The facts about the answer ride along, so the applied-or-not question below can still be
|
|
386
|
+
# asked of a windowed refusal.
|
|
387
|
+
windowed.http_status = getattr(refusal, "http_status", None) # type: ignore[attr-defined]
|
|
388
|
+
windowed.studio_code = getattr(refusal, "studio_code", None) # type: ignore[attr-defined]
|
|
389
|
+
return windowed
|
|
390
|
+
|
|
391
|
+
|
|
392
|
+
#: The statuses under which Studio answered with a problem document BEFORE any write: the request
|
|
393
|
+
#: failed validation, named a resource that is not there, asserted a stale version, or carried no
|
|
394
|
+
#: authority for the route. Every one of them is raised before ``v3_idempotent`` reserves a key.
|
|
395
|
+
_REFUSED_BEFORE_WRITE = frozenset({401, 403, 404, 412, 422})
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
def _nothing_was_applied(refusal: ThinLaneError) -> bool:
|
|
399
|
+
"""Whether a refusal means Studio answered and applied nothing -- decided narrowly.
|
|
400
|
+
|
|
401
|
+
Only a problem document from Studio under a validation or conflict status says that: 401, 403,
|
|
402
|
+
404, 412 and 422, and a 409 whose code is not one of the idempotency codes. Everything else
|
|
403
|
+
keeps the plan. A 5xx may be a gateway answering after Studio applied the write; an answer
|
|
404
|
+
this lane could not read as a problem document says nothing about what happened behind it; and
|
|
405
|
+
the two ``IDEMPOTENCY_*`` 409s are Studio saying the key IS reserved -- in progress, or
|
|
406
|
+
reserved without its record because ``v3_idempotent`` writes the resource and completes the
|
|
407
|
+
record in two transactions -- so the write may well have landed. For all of those the only
|
|
408
|
+
honest resume is the identical request under the identical key, which Studio replays.
|
|
409
|
+
"""
|
|
410
|
+
|
|
411
|
+
status = getattr(refusal, "http_status", None)
|
|
412
|
+
studio_code = getattr(refusal, "studio_code", None)
|
|
413
|
+
if not isinstance(status, int) or not isinstance(studio_code, str):
|
|
414
|
+
return False
|
|
415
|
+
if status in _REFUSED_BEFORE_WRITE:
|
|
416
|
+
return True
|
|
417
|
+
return status == 409 and not studio_code.startswith("IDEMPOTENCY_")
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
# --------------------------------------------------------------------------------------------
|
|
421
|
+
# The bundle, read and checked before anything is minted
|
|
422
|
+
# --------------------------------------------------------------------------------------------
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
def _invalid(name: str, path: str, what: str) -> ProposeError:
|
|
426
|
+
where = f"{name} {path}" if path else name
|
|
427
|
+
return ProposeError("THIN_PROPOSE_DOCUMENT_INVALID", f"{where} {what}")
|
|
428
|
+
|
|
429
|
+
|
|
430
|
+
def _bundle_directory(value: Any) -> Path:
|
|
431
|
+
"""The bundle directory, refused before a credential is resolved if it is not one.
|
|
432
|
+
|
|
433
|
+
The refusals name what the open actually observed rather than asserting a kind: ``presence_at``
|
|
434
|
+
reads the mode the look found, so "there is nothing there" and "there is a file there" are
|
|
435
|
+
answers to a look, not guesses -- the same rule ``thin.acquire`` reads a named path under, and
|
|
436
|
+
the one ``scripts/path_kind_gate.py`` holds every surface to.
|
|
437
|
+
"""
|
|
438
|
+
|
|
439
|
+
if not isinstance(value, str) or not value:
|
|
440
|
+
raise ProposeError("THIN_PROPOSE_BUNDLE_INVALID", "propose names one bundle directory")
|
|
441
|
+
path = Path(value).expanduser()
|
|
442
|
+
try:
|
|
443
|
+
info = path.lstat()
|
|
444
|
+
except OSError:
|
|
445
|
+
raise ProposeError(
|
|
446
|
+
"THIN_PROPOSE_BUNDLE_INVALID",
|
|
447
|
+
f"there is {presence_at(path) or UNKNOWN_KIND} at {path}: propose reads its documents "
|
|
448
|
+
"out of a directory",
|
|
449
|
+
) from None
|
|
450
|
+
if stat.S_ISLNK(info.st_mode) or not path.is_dir():
|
|
451
|
+
raise ProposeError(
|
|
452
|
+
"THIN_PROPOSE_BUNDLE_INVALID",
|
|
453
|
+
f"there is {presence_at(path) or UNKNOWN_KIND} at {path}, not the directory propose "
|
|
454
|
+
"reads its documents out of",
|
|
455
|
+
)
|
|
456
|
+
return path
|
|
457
|
+
|
|
458
|
+
|
|
459
|
+
def _read_document(bundle: Path, name: str) -> tuple[Any, str]:
|
|
460
|
+
"""One bundle document and the digest of its bytes, parsed as STRICT JSON.
|
|
461
|
+
|
|
462
|
+
Strict, not foreign: every number in a bundle reaches a canonical digest or a Studio body that
|
|
463
|
+
is digested there, and the Harness data model admits no fractional number anywhere on that
|
|
464
|
+
path. A float is refused here, by document and position, rather than three stages in as a
|
|
465
|
+
canonicalisation error out of a request body.
|
|
466
|
+
"""
|
|
467
|
+
|
|
468
|
+
target = bundle / name
|
|
469
|
+
try:
|
|
470
|
+
raw = read_plain_file(target, max_bytes=MAX_DOCUMENT_BYTES)
|
|
471
|
+
except PlainFileRefusal as refusal:
|
|
472
|
+
raise ProposeError(
|
|
473
|
+
"THIN_PROPOSE_DOCUMENT_UNREADABLE",
|
|
474
|
+
f"{name} could not be read from the bundle ({refusal.reason}); author it at {target}",
|
|
475
|
+
) from None
|
|
476
|
+
except OSError as error:
|
|
477
|
+
raise ProposeError(
|
|
478
|
+
"THIN_PROPOSE_DOCUMENT_UNREADABLE", f"{name} could not be read: {error}"
|
|
479
|
+
) from error
|
|
480
|
+
try:
|
|
481
|
+
parsed = parse_json(raw)
|
|
482
|
+
except CanonicalJSONError as error:
|
|
483
|
+
raise ProposeError(
|
|
484
|
+
"THIN_PROPOSE_DOCUMENT_INVALID",
|
|
485
|
+
f"{name} is not strict JSON at {error.path}: {error.detail}; a bundle carries "
|
|
486
|
+
"integers, strings, booleans, null, objects and lists, and never a fractional number",
|
|
487
|
+
) from error
|
|
488
|
+
return parsed, hashlib.sha256(raw).hexdigest()
|
|
489
|
+
|
|
490
|
+
|
|
491
|
+
# -- the checks, in the vocabulary of the document --------------------------------------------
|
|
492
|
+
|
|
493
|
+
|
|
494
|
+
def _object(value: Any, name: str, path: str = "") -> dict[str, Any]:
|
|
495
|
+
if not isinstance(value, dict):
|
|
496
|
+
raise _invalid(name, path, "must be a JSON object")
|
|
497
|
+
return value
|
|
498
|
+
|
|
499
|
+
|
|
500
|
+
def _keys(
|
|
501
|
+
value: Mapping[str, Any],
|
|
502
|
+
name: str,
|
|
503
|
+
path: str,
|
|
504
|
+
required: Sequence[str],
|
|
505
|
+
optional: Sequence[str] = (),
|
|
506
|
+
) -> None:
|
|
507
|
+
missing = [key for key in required if key not in value]
|
|
508
|
+
if missing:
|
|
509
|
+
raise _invalid(name, path, f"is missing {', '.join(repr(key) for key in missing)}")
|
|
510
|
+
extra = sorted(set(value) - set(required) - set(optional))
|
|
511
|
+
if extra:
|
|
512
|
+
raise _invalid(
|
|
513
|
+
name,
|
|
514
|
+
path,
|
|
515
|
+
f"carries {', '.join(repr(key) for key in extra)}, which the shape has no room for",
|
|
516
|
+
)
|
|
517
|
+
|
|
518
|
+
|
|
519
|
+
def _string(
|
|
520
|
+
value: Any,
|
|
521
|
+
name: str,
|
|
522
|
+
path: str,
|
|
523
|
+
*,
|
|
524
|
+
minimum: int = 1,
|
|
525
|
+
maximum: int | None = None,
|
|
526
|
+
pattern: re.Pattern[str] | None = None,
|
|
527
|
+
what: str = "",
|
|
528
|
+
) -> str:
|
|
529
|
+
if not isinstance(value, str):
|
|
530
|
+
raise _invalid(name, path, "must be a string")
|
|
531
|
+
if len(value) < minimum:
|
|
532
|
+
raise _invalid(name, path, f"must be at least {minimum} character(s)")
|
|
533
|
+
if maximum is not None and len(value) > maximum:
|
|
534
|
+
raise _invalid(name, path, f"must be at most {maximum} characters")
|
|
535
|
+
if pattern is not None and pattern.fullmatch(value) is None:
|
|
536
|
+
raise _invalid(name, path, f"is not {what or 'in the shape the contract admits'}")
|
|
537
|
+
return value
|
|
538
|
+
|
|
539
|
+
|
|
540
|
+
def _enum(value: Any, name: str, path: str, allowed: Sequence[str]) -> str:
|
|
541
|
+
if value not in allowed:
|
|
542
|
+
raise _invalid(name, path, f"must be one of: {', '.join(allowed)}")
|
|
543
|
+
return value
|
|
544
|
+
|
|
545
|
+
|
|
546
|
+
def _integer(value: Any, name: str, path: str, *, minimum: int, maximum: int) -> int:
|
|
547
|
+
if type(value) is not int or not minimum <= value <= maximum:
|
|
548
|
+
raise _invalid(name, path, f"must be an integer from {minimum} to {maximum}")
|
|
549
|
+
return value
|
|
550
|
+
|
|
551
|
+
|
|
552
|
+
def _boolean(value: Any, name: str, path: str) -> bool:
|
|
553
|
+
if type(value) is not bool:
|
|
554
|
+
raise _invalid(name, path, "must be true or false")
|
|
555
|
+
return value
|
|
556
|
+
|
|
557
|
+
|
|
558
|
+
def _uuid_text(value: Any, name: str, path: str) -> str:
|
|
559
|
+
if not isinstance(value, str) or _UUID_TEXT.fullmatch(value) is None:
|
|
560
|
+
raise _invalid(name, path, "must be an identifier Studio assigned (a UUID)")
|
|
561
|
+
return value
|
|
562
|
+
|
|
563
|
+
|
|
564
|
+
def _digest(value: Any, name: str, path: str) -> str:
|
|
565
|
+
return _string(value, name, path, pattern=_PREFIXED_DIGEST, what="a sha256:-prefixed digest")
|
|
566
|
+
|
|
567
|
+
|
|
568
|
+
def _timestamp(value: Any, name: str, path: str) -> str:
|
|
569
|
+
return _string(value, name, path, pattern=_TIMESTAMP, what="a UTC timestamp ending in Z")
|
|
570
|
+
|
|
571
|
+
|
|
572
|
+
def _time_range(value: Any, name: str, path: str) -> dict[str, str]:
|
|
573
|
+
window = _object(value, name, path)
|
|
574
|
+
_keys(window, name, path, ("start_inclusive", "end_exclusive"))
|
|
575
|
+
return {
|
|
576
|
+
"start_inclusive": _timestamp(window["start_inclusive"], name, f"{path}.start_inclusive"),
|
|
577
|
+
"end_exclusive": _timestamp(window["end_exclusive"], name, f"{path}.end_exclusive"),
|
|
578
|
+
}
|
|
579
|
+
|
|
580
|
+
|
|
581
|
+
def _column_list(
|
|
582
|
+
value: Any,
|
|
583
|
+
name: str,
|
|
584
|
+
path: str,
|
|
585
|
+
*,
|
|
586
|
+
minimum_items: int = 1,
|
|
587
|
+
maximum_items: int | None = None,
|
|
588
|
+
) -> list[str]:
|
|
589
|
+
if not isinstance(value, list) or len(value) < minimum_items:
|
|
590
|
+
raise _invalid(name, path, f"must be a list of at least {minimum_items} column name(s)")
|
|
591
|
+
if maximum_items is not None and len(value) > maximum_items:
|
|
592
|
+
raise _invalid(name, path, f"must name at most {maximum_items} columns")
|
|
593
|
+
columns = [
|
|
594
|
+
_string(item, name, f"{path}[{index}]", pattern=_COLUMN, what="a lowercase column name")
|
|
595
|
+
for index, item in enumerate(value)
|
|
596
|
+
]
|
|
597
|
+
if len(set(columns)) != len(columns):
|
|
598
|
+
raise _invalid(name, path, "must not name a column twice")
|
|
599
|
+
return columns
|
|
600
|
+
|
|
601
|
+
|
|
602
|
+
def check_dataset(document: Any) -> dict[str, Any]:
|
|
603
|
+
"""``dataset.json``: a name and an optional description, or the id of a Dataset to adopt."""
|
|
604
|
+
|
|
605
|
+
value = _object(document, "dataset.json")
|
|
606
|
+
if "dataset_id" in value:
|
|
607
|
+
_keys(value, "dataset.json", "", ("dataset_id",))
|
|
608
|
+
return {"dataset_id": _uuid_text(value["dataset_id"], "dataset.json", "dataset_id")}
|
|
609
|
+
_keys(value, "dataset.json", "", ("name",), ("description",))
|
|
610
|
+
checked = {"name": _string(value["name"], "dataset.json", "name", maximum=160)}
|
|
611
|
+
if "description" in value:
|
|
612
|
+
checked["description"] = _string(
|
|
613
|
+
value["description"], "dataset.json", "description", minimum=0, maximum=4000
|
|
614
|
+
)
|
|
615
|
+
return checked
|
|
616
|
+
|
|
617
|
+
|
|
618
|
+
def check_requirements(value: Any, name: str, path: str) -> dict[str, Any]:
|
|
619
|
+
"""The ``RequirementsUpsertCommand`` content, in the contract's own enums and patterns."""
|
|
620
|
+
|
|
621
|
+
requirements = _object(value, name, path)
|
|
622
|
+
_keys(
|
|
623
|
+
requirements,
|
|
624
|
+
name,
|
|
625
|
+
path,
|
|
626
|
+
(
|
|
627
|
+
"population",
|
|
628
|
+
"time_range",
|
|
629
|
+
"output_grain",
|
|
630
|
+
"required_fields",
|
|
631
|
+
"target_policy",
|
|
632
|
+
"success_criteria",
|
|
633
|
+
"feasibility",
|
|
634
|
+
),
|
|
635
|
+
("prediction_cutoff",),
|
|
636
|
+
)
|
|
637
|
+
feasibility = _object(requirements["feasibility"], name, f"{path}.feasibility")
|
|
638
|
+
_keys(feasibility, name, f"{path}.feasibility", ("decision", "reason_codes", "narrative"))
|
|
639
|
+
codes = feasibility["reason_codes"]
|
|
640
|
+
if not isinstance(codes, list) or len(set(map(str, codes))) != len(codes):
|
|
641
|
+
raise _invalid(name, f"{path}.feasibility.reason_codes", "must be a list without repeats")
|
|
642
|
+
criteria = requirements["success_criteria"]
|
|
643
|
+
if not isinstance(criteria, list) or not criteria:
|
|
644
|
+
raise _invalid(name, f"{path}.success_criteria", "must be a non-empty list")
|
|
645
|
+
checked: dict[str, Any] = {
|
|
646
|
+
"population": _string(requirements["population"], name, f"{path}.population", maximum=2000),
|
|
647
|
+
"time_range": _time_range(requirements["time_range"], name, f"{path}.time_range"),
|
|
648
|
+
"output_grain": _column_list(requirements["output_grain"], name, f"{path}.output_grain"),
|
|
649
|
+
"required_fields": _column_list(
|
|
650
|
+
requirements["required_fields"], name, f"{path}.required_fields"
|
|
651
|
+
),
|
|
652
|
+
"target_policy": _enum(
|
|
653
|
+
requirements["target_policy"], name, f"{path}.target_policy", _TARGET_POLICIES
|
|
654
|
+
),
|
|
655
|
+
"success_criteria": [
|
|
656
|
+
_string(item, name, f"{path}.success_criteria[{index}]", maximum=1000)
|
|
657
|
+
for index, item in enumerate(criteria)
|
|
658
|
+
],
|
|
659
|
+
"feasibility": {
|
|
660
|
+
"decision": _enum(
|
|
661
|
+
feasibility["decision"],
|
|
662
|
+
name,
|
|
663
|
+
f"{path}.feasibility.decision",
|
|
664
|
+
_FEASIBILITY_DECISIONS,
|
|
665
|
+
),
|
|
666
|
+
"reason_codes": [
|
|
667
|
+
_enum(item, name, f"{path}.feasibility.reason_codes[{index}]", _REASON_CODES)
|
|
668
|
+
for index, item in enumerate(codes)
|
|
669
|
+
],
|
|
670
|
+
"narrative": _string(
|
|
671
|
+
feasibility["narrative"], name, f"{path}.feasibility.narrative", maximum=4000
|
|
672
|
+
),
|
|
673
|
+
},
|
|
674
|
+
}
|
|
675
|
+
if "prediction_cutoff" in requirements:
|
|
676
|
+
checked["prediction_cutoff"] = _timestamp(
|
|
677
|
+
requirements["prediction_cutoff"], name, f"{path}.prediction_cutoff"
|
|
678
|
+
)
|
|
679
|
+
return checked
|
|
680
|
+
|
|
681
|
+
|
|
682
|
+
def check_question(document: Any) -> dict[str, Any]:
|
|
683
|
+
"""``question.json``: the question text and its requirements, or the id of one to adopt."""
|
|
684
|
+
|
|
685
|
+
value = _object(document, "question.json")
|
|
686
|
+
if "question_id" in value:
|
|
687
|
+
_keys(value, "question.json", "", ("question_id",))
|
|
688
|
+
return {"question_id": _uuid_text(value["question_id"], "question.json", "question_id")}
|
|
689
|
+
_keys(value, "question.json", "", ("question", "requirements"))
|
|
690
|
+
return {
|
|
691
|
+
"question": _string(value["question"], "question.json", "question", maximum=8000),
|
|
692
|
+
"requirements": check_requirements(value["requirements"], "question.json", "requirements"),
|
|
693
|
+
}
|
|
694
|
+
|
|
695
|
+
|
|
696
|
+
def _check_query(value: Any, name: str, path: str) -> dict[str, Any]:
|
|
697
|
+
"""The ``public_https_query`` a ``public.https`` connector is registered with."""
|
|
698
|
+
|
|
699
|
+
query = _object(value, name, path)
|
|
700
|
+
_keys(
|
|
701
|
+
query,
|
|
702
|
+
name,
|
|
703
|
+
path,
|
|
704
|
+
("url", "data_format", "filename", "reader_pin", "resource_caps", "limits"),
|
|
705
|
+
)
|
|
706
|
+
limits = _object(query["limits"], name, f"{path}.limits")
|
|
707
|
+
_keys(limits, name, f"{path}.limits", tuple(_QUERY_LIMIT_CEILINGS))
|
|
708
|
+
pin = query["reader_pin"]
|
|
709
|
+
if pin is not None and not isinstance(pin, dict):
|
|
710
|
+
raise _invalid(name, f"{path}.reader_pin", "must be null or a Reader pin object")
|
|
711
|
+
caps = query["resource_caps"]
|
|
712
|
+
if caps is not None:
|
|
713
|
+
if pin is None:
|
|
714
|
+
raise _invalid(name, f"{path}.resource_caps", "must be null when reader_pin is null")
|
|
715
|
+
caps = _object(caps, name, f"{path}.resource_caps")
|
|
716
|
+
_keys(
|
|
717
|
+
caps,
|
|
718
|
+
name,
|
|
719
|
+
f"{path}.resource_caps",
|
|
720
|
+
("max_declared_cells", "max_container_members", "max_nesting_depth"),
|
|
721
|
+
)
|
|
722
|
+
_integer(
|
|
723
|
+
caps["max_declared_cells"],
|
|
724
|
+
name,
|
|
725
|
+
f"{path}.resource_caps.max_declared_cells",
|
|
726
|
+
minimum=1,
|
|
727
|
+
maximum=2**53,
|
|
728
|
+
)
|
|
729
|
+
_integer(
|
|
730
|
+
caps["max_container_members"],
|
|
731
|
+
name,
|
|
732
|
+
f"{path}.resource_caps.max_container_members",
|
|
733
|
+
minimum=1,
|
|
734
|
+
maximum=2**53,
|
|
735
|
+
)
|
|
736
|
+
if caps["max_nesting_depth"] != 1:
|
|
737
|
+
raise _invalid(name, f"{path}.resource_caps.max_nesting_depth", "must be 1")
|
|
738
|
+
return {
|
|
739
|
+
"url": _string(
|
|
740
|
+
query["url"],
|
|
741
|
+
name,
|
|
742
|
+
f"{path}.url",
|
|
743
|
+
maximum=2048,
|
|
744
|
+
pattern=_HTTPS_URL,
|
|
745
|
+
what="an https:// address with no credential",
|
|
746
|
+
),
|
|
747
|
+
"data_format": _enum(
|
|
748
|
+
query["data_format"], name, f"{path}.data_format", PARSER_FORMAT_ORDER
|
|
749
|
+
),
|
|
750
|
+
"filename": _string(
|
|
751
|
+
query["filename"],
|
|
752
|
+
name,
|
|
753
|
+
f"{path}.filename",
|
|
754
|
+
maximum=255,
|
|
755
|
+
pattern=_FILENAME,
|
|
756
|
+
what="a file name without a path",
|
|
757
|
+
),
|
|
758
|
+
"reader_pin": pin,
|
|
759
|
+
"resource_caps": caps,
|
|
760
|
+
"limits": {
|
|
761
|
+
key: _integer(limits[key], name, f"{path}.limits.{key}", minimum=1, maximum=ceiling)
|
|
762
|
+
for key, ceiling in _QUERY_LIMIT_CEILINGS.items()
|
|
763
|
+
},
|
|
764
|
+
}
|
|
765
|
+
|
|
766
|
+
|
|
767
|
+
def _check_connector(value: Any, name: str, path: str) -> dict[str, Any]:
|
|
768
|
+
connector = _object(value, name, path)
|
|
769
|
+
_keys(connector, name, path, ("adapter_id",), ("credential_mode", "query"))
|
|
770
|
+
adapter_id = connector["adapter_id"]
|
|
771
|
+
if adapter_id not in CONNECTOR_EGRESS_ATTESTATION:
|
|
772
|
+
allowed = ", ".join(sorted(CONNECTOR_EGRESS_ATTESTATION))
|
|
773
|
+
raise ProposeError(
|
|
774
|
+
"THIN_PROPOSE_CONNECTOR_UNSUPPORTED",
|
|
775
|
+
f"{name} {path}.adapter_id names {adapter_id!r}; the hosted authoring front door "
|
|
776
|
+
f"registers credential-free connectors ({allowed}), and a credentialed or staged-file "
|
|
777
|
+
"source enters through the local lane instead",
|
|
778
|
+
)
|
|
779
|
+
if connector.get("credential_mode", "none") != "none":
|
|
780
|
+
raise ProposeError(
|
|
781
|
+
"THIN_PROPOSE_CONNECTOR_UNSUPPORTED",
|
|
782
|
+
f"{name} {path}.credential_mode must be none: a {adapter_id} connector is "
|
|
783
|
+
"credential-free, and this lane registers no other kind",
|
|
784
|
+
)
|
|
785
|
+
checked: dict[str, Any] = {"adapter_id": adapter_id, "credential_mode": "none"}
|
|
786
|
+
if adapter_id == "public.https":
|
|
787
|
+
if "query" not in connector:
|
|
788
|
+
raise _invalid(name, path, "needs a 'query' for a public.https connector")
|
|
789
|
+
checked["query"] = _check_query(connector["query"], name, f"{path}.query")
|
|
790
|
+
elif "query" in connector:
|
|
791
|
+
raise _invalid(name, f"{path}.query", "is only carried by a public.https connector")
|
|
792
|
+
return checked
|
|
793
|
+
|
|
794
|
+
|
|
795
|
+
def check_sources(document: Any) -> list[dict[str, Any]]:
|
|
796
|
+
"""``sources.json``: one entry per source, each registering or adopting one."""
|
|
797
|
+
|
|
798
|
+
name = "sources.json"
|
|
799
|
+
if not isinstance(document, list) or not document:
|
|
800
|
+
raise _invalid(name, "", "must be a non-empty list of source entries")
|
|
801
|
+
entries: list[dict[str, Any]] = []
|
|
802
|
+
for index, item in enumerate(document):
|
|
803
|
+
path = f"[{index}]"
|
|
804
|
+
entry = _object(item, name, path)
|
|
805
|
+
if "source_id" in entry:
|
|
806
|
+
_keys(entry, name, path, ("name", "source_id", "connector_configuration_id"))
|
|
807
|
+
entries.append(
|
|
808
|
+
{
|
|
809
|
+
"name": _string(entry["name"], name, f"{path}.name", maximum=200),
|
|
810
|
+
"source_id": _uuid_text(entry["source_id"], name, f"{path}.source_id"),
|
|
811
|
+
"connector_configuration_id": _uuid_text(
|
|
812
|
+
entry["connector_configuration_id"],
|
|
813
|
+
name,
|
|
814
|
+
f"{path}.connector_configuration_id",
|
|
815
|
+
),
|
|
816
|
+
}
|
|
817
|
+
)
|
|
818
|
+
continue
|
|
819
|
+
_keys(
|
|
820
|
+
entry,
|
|
821
|
+
name,
|
|
822
|
+
path,
|
|
823
|
+
(
|
|
824
|
+
"name",
|
|
825
|
+
"source_class",
|
|
826
|
+
"locator",
|
|
827
|
+
"credential_reference_ids",
|
|
828
|
+
"data_classification",
|
|
829
|
+
"rights_claim",
|
|
830
|
+
"retention_policy",
|
|
831
|
+
"connector",
|
|
832
|
+
),
|
|
833
|
+
)
|
|
834
|
+
locator = _object(entry["locator"], name, f"{path}.locator")
|
|
835
|
+
_keys(
|
|
836
|
+
locator,
|
|
837
|
+
name,
|
|
838
|
+
f"{path}.locator",
|
|
839
|
+
("kind", "display_locator"),
|
|
840
|
+
("connector_contract_version",),
|
|
841
|
+
)
|
|
842
|
+
checked_locator = {
|
|
843
|
+
"kind": _enum(locator["kind"], name, f"{path}.locator.kind", _LOCATOR_KINDS),
|
|
844
|
+
"display_locator": _string(
|
|
845
|
+
locator["display_locator"], name, f"{path}.locator.display_locator", maximum=2048
|
|
846
|
+
),
|
|
847
|
+
}
|
|
848
|
+
if "connector_contract_version" in locator:
|
|
849
|
+
checked_locator["connector_contract_version"] = _string(
|
|
850
|
+
locator["connector_contract_version"],
|
|
851
|
+
name,
|
|
852
|
+
f"{path}.locator.connector_contract_version",
|
|
853
|
+
pattern=_SEMVER,
|
|
854
|
+
what="a version like 1.0.0",
|
|
855
|
+
)
|
|
856
|
+
if entry["credential_reference_ids"] != []:
|
|
857
|
+
raise ProposeError(
|
|
858
|
+
"THIN_PROPOSE_CONNECTOR_UNSUPPORTED",
|
|
859
|
+
f"{name} {path}.credential_reference_ids must be empty: this lane registers "
|
|
860
|
+
"credential-free sources only",
|
|
861
|
+
)
|
|
862
|
+
claim = _object(entry["rights_claim"], name, f"{path}.rights_claim")
|
|
863
|
+
_keys(
|
|
864
|
+
claim,
|
|
865
|
+
name,
|
|
866
|
+
f"{path}.rights_claim",
|
|
867
|
+
("claimed_basis", "claim_evidence_digest"),
|
|
868
|
+
("claim_note",),
|
|
869
|
+
)
|
|
870
|
+
checked_claim = {
|
|
871
|
+
"claimed_basis": _enum(
|
|
872
|
+
claim["claimed_basis"], name, f"{path}.rights_claim.claimed_basis", _CLAIMED_BASES
|
|
873
|
+
),
|
|
874
|
+
"claim_evidence_digest": _digest(
|
|
875
|
+
claim["claim_evidence_digest"], name, f"{path}.rights_claim.claim_evidence_digest"
|
|
876
|
+
),
|
|
877
|
+
}
|
|
878
|
+
if "claim_note" in claim:
|
|
879
|
+
checked_claim["claim_note"] = _string(
|
|
880
|
+
claim["claim_note"], name, f"{path}.rights_claim.claim_note", maximum=2000
|
|
881
|
+
)
|
|
882
|
+
retention = _object(entry["retention_policy"], name, f"{path}.retention_policy")
|
|
883
|
+
_keys(
|
|
884
|
+
retention,
|
|
885
|
+
name,
|
|
886
|
+
f"{path}.retention_policy",
|
|
887
|
+
("raw_days", "derived_days", "tombstone_required"),
|
|
888
|
+
)
|
|
889
|
+
entries.append(
|
|
890
|
+
{
|
|
891
|
+
"name": _string(entry["name"], name, f"{path}.name", maximum=200),
|
|
892
|
+
"source_class": _enum(
|
|
893
|
+
entry["source_class"], name, f"{path}.source_class", _SOURCE_CLASSES
|
|
894
|
+
),
|
|
895
|
+
"locator": checked_locator,
|
|
896
|
+
"credential_reference_ids": [],
|
|
897
|
+
"data_classification": _enum(
|
|
898
|
+
entry["data_classification"],
|
|
899
|
+
name,
|
|
900
|
+
f"{path}.data_classification",
|
|
901
|
+
_CLASSIFICATIONS,
|
|
902
|
+
),
|
|
903
|
+
"rights_claim": checked_claim,
|
|
904
|
+
"retention_policy": {
|
|
905
|
+
"raw_days": _integer(
|
|
906
|
+
retention["raw_days"],
|
|
907
|
+
name,
|
|
908
|
+
f"{path}.retention_policy.raw_days",
|
|
909
|
+
minimum=0,
|
|
910
|
+
maximum=3650,
|
|
911
|
+
),
|
|
912
|
+
"derived_days": _integer(
|
|
913
|
+
retention["derived_days"],
|
|
914
|
+
name,
|
|
915
|
+
f"{path}.retention_policy.derived_days",
|
|
916
|
+
minimum=0,
|
|
917
|
+
maximum=3650,
|
|
918
|
+
),
|
|
919
|
+
"tombstone_required": _boolean(
|
|
920
|
+
retention["tombstone_required"],
|
|
921
|
+
name,
|
|
922
|
+
f"{path}.retention_policy.tombstone_required",
|
|
923
|
+
),
|
|
924
|
+
},
|
|
925
|
+
"connector": _check_connector(entry["connector"], name, f"{path}.connector"),
|
|
926
|
+
}
|
|
927
|
+
)
|
|
928
|
+
names = [entry["name"] for entry in entries]
|
|
929
|
+
if len(set(names)) != len(names):
|
|
930
|
+
raise _invalid(
|
|
931
|
+
name,
|
|
932
|
+
"",
|
|
933
|
+
"must give each entry a distinct 'name'; the name is how the registered source id is "
|
|
934
|
+
"reported back for recipe.json to reference",
|
|
935
|
+
)
|
|
936
|
+
return entries
|
|
937
|
+
|
|
938
|
+
|
|
939
|
+
def check_backfill_policy(value: Any, name: str, path: str) -> dict[str, Any]:
|
|
940
|
+
policy = _object(value, name, path)
|
|
941
|
+
_keys(
|
|
942
|
+
policy,
|
|
943
|
+
name,
|
|
944
|
+
path,
|
|
945
|
+
(
|
|
946
|
+
"schema_version",
|
|
947
|
+
"coverage_envelope",
|
|
948
|
+
"max_window_seconds",
|
|
949
|
+
"overlap_policy",
|
|
950
|
+
"partition_scope",
|
|
951
|
+
),
|
|
952
|
+
)
|
|
953
|
+
if policy["schema_version"] != _BACKFILL_POLICY_SCHEMA:
|
|
954
|
+
raise _invalid(name, f"{path}.schema_version", f"must be {_BACKFILL_POLICY_SCHEMA}")
|
|
955
|
+
scope = _object(policy["partition_scope"], name, f"{path}.partition_scope")
|
|
956
|
+
_keys(
|
|
957
|
+
scope,
|
|
958
|
+
name,
|
|
959
|
+
f"{path}.partition_scope",
|
|
960
|
+
("fields", "allowed_values", "max_partitions_per_run"),
|
|
961
|
+
)
|
|
962
|
+
allowed = _object(scope["allowed_values"], name, f"{path}.partition_scope.allowed_values")
|
|
963
|
+
if len(allowed) > 16:
|
|
964
|
+
raise _invalid(
|
|
965
|
+
name, f"{path}.partition_scope.allowed_values", "must name at most 16 fields"
|
|
966
|
+
)
|
|
967
|
+
checked_allowed: dict[str, list[str]] = {}
|
|
968
|
+
for field, values in allowed.items():
|
|
969
|
+
where = f"{path}.partition_scope.allowed_values.{field}"
|
|
970
|
+
if not isinstance(values, list) or not 1 <= len(values) <= 10_000:
|
|
971
|
+
raise _invalid(name, where, "must be a list of 1 to 10000 values")
|
|
972
|
+
items = [
|
|
973
|
+
_string(item, name, f"{where}[{index}]", maximum=256)
|
|
974
|
+
for index, item in enumerate(values)
|
|
975
|
+
]
|
|
976
|
+
if len(set(items)) != len(items):
|
|
977
|
+
raise _invalid(name, where, "must not repeat a value")
|
|
978
|
+
checked_allowed[field] = items
|
|
979
|
+
return {
|
|
980
|
+
"schema_version": _BACKFILL_POLICY_SCHEMA,
|
|
981
|
+
"coverage_envelope": _time_range(
|
|
982
|
+
policy["coverage_envelope"], name, f"{path}.coverage_envelope"
|
|
983
|
+
),
|
|
984
|
+
"max_window_seconds": _integer(
|
|
985
|
+
policy["max_window_seconds"],
|
|
986
|
+
name,
|
|
987
|
+
f"{path}.max_window_seconds",
|
|
988
|
+
minimum=1,
|
|
989
|
+
maximum=315_360_000,
|
|
990
|
+
),
|
|
991
|
+
"overlap_policy": _enum(
|
|
992
|
+
policy["overlap_policy"], name, f"{path}.overlap_policy", _OVERLAP_POLICIES
|
|
993
|
+
),
|
|
994
|
+
"partition_scope": {
|
|
995
|
+
"fields": _column_list(
|
|
996
|
+
scope["fields"],
|
|
997
|
+
name,
|
|
998
|
+
f"{path}.partition_scope.fields",
|
|
999
|
+
minimum_items=0,
|
|
1000
|
+
maximum_items=16,
|
|
1001
|
+
),
|
|
1002
|
+
"allowed_values": checked_allowed,
|
|
1003
|
+
"max_partitions_per_run": _integer(
|
|
1004
|
+
scope["max_partitions_per_run"],
|
|
1005
|
+
name,
|
|
1006
|
+
f"{path}.partition_scope.max_partitions_per_run",
|
|
1007
|
+
minimum=1,
|
|
1008
|
+
maximum=10_000,
|
|
1009
|
+
),
|
|
1010
|
+
},
|
|
1011
|
+
}
|
|
1012
|
+
|
|
1013
|
+
|
|
1014
|
+
def check_table_plan(document: Any) -> dict[str, Any]:
|
|
1015
|
+
"""``table_plan.json``: the authored plan content, or the id of a plan to adopt.
|
|
1016
|
+
|
|
1017
|
+
The three digests are optional: each is derived from ``recipe.json`` when left out, and held
|
|
1018
|
+
to that derivation when written, so an author never has to compute one and cannot get one
|
|
1019
|
+
wrong. ``approved_backfill_policy`` defaults to the recipe's own ``backfill_policy``.
|
|
1020
|
+
"""
|
|
1021
|
+
|
|
1022
|
+
name = "table_plan.json"
|
|
1023
|
+
value = _object(document, name)
|
|
1024
|
+
if "table_plan_id" in value:
|
|
1025
|
+
_keys(value, name, "", ("table_plan_id",))
|
|
1026
|
+
return {"table_plan_id": _uuid_text(value["table_plan_id"], name, "table_plan_id")}
|
|
1027
|
+
_keys(
|
|
1028
|
+
value,
|
|
1029
|
+
name,
|
|
1030
|
+
"",
|
|
1031
|
+
("output_grain", "transformation_contract_version"),
|
|
1032
|
+
(
|
|
1033
|
+
"execution_plan_digest",
|
|
1034
|
+
"validation_policy_digest",
|
|
1035
|
+
"approved_backfill_policy",
|
|
1036
|
+
"backfill_policy_digest",
|
|
1037
|
+
),
|
|
1038
|
+
)
|
|
1039
|
+
checked: dict[str, Any] = {
|
|
1040
|
+
"output_grain": _column_list(value["output_grain"], name, "output_grain"),
|
|
1041
|
+
"transformation_contract_version": _string(
|
|
1042
|
+
value["transformation_contract_version"],
|
|
1043
|
+
name,
|
|
1044
|
+
"transformation_contract_version",
|
|
1045
|
+
pattern=_SEMVER,
|
|
1046
|
+
what="a version like 1.0.0",
|
|
1047
|
+
),
|
|
1048
|
+
}
|
|
1049
|
+
for field in ("execution_plan_digest", "validation_policy_digest", "backfill_policy_digest"):
|
|
1050
|
+
if field in value:
|
|
1051
|
+
checked[field] = _digest(value[field], name, field)
|
|
1052
|
+
if "approved_backfill_policy" in value:
|
|
1053
|
+
checked["approved_backfill_policy"] = check_backfill_policy(
|
|
1054
|
+
value["approved_backfill_policy"], name, "approved_backfill_policy"
|
|
1055
|
+
)
|
|
1056
|
+
return checked
|
|
1057
|
+
|
|
1058
|
+
|
|
1059
|
+
def check_activation(document: Any) -> dict[str, Any]:
|
|
1060
|
+
"""``activation.json``: the Table name and the refresh schedule, defaults filled in."""
|
|
1061
|
+
|
|
1062
|
+
name = "activation.json"
|
|
1063
|
+
value = _object(document, name)
|
|
1064
|
+
_keys(value, name, "", ("table_name",), ("schedule",))
|
|
1065
|
+
table_name = _string(value["table_name"], name, "table_name", maximum=200)
|
|
1066
|
+
if not table_name.strip():
|
|
1067
|
+
raise _invalid(name, "table_name", "must not be blank")
|
|
1068
|
+
schedule = dict(DEFAULT_SCHEDULE)
|
|
1069
|
+
if "schedule" in value:
|
|
1070
|
+
authored = _object(value["schedule"], name, "schedule")
|
|
1071
|
+
_keys(authored, name, "schedule", (), tuple(DEFAULT_SCHEDULE))
|
|
1072
|
+
schedule.update(authored)
|
|
1073
|
+
checked_schedule = {
|
|
1074
|
+
"cron": _string(schedule["cron"], name, "schedule.cron", minimum=9, maximum=100),
|
|
1075
|
+
"timezone": _enum(schedule["timezone"], name, "schedule.timezone", ("UTC",)),
|
|
1076
|
+
"mode": _enum(schedule["mode"], name, "schedule.mode", ("incremental_refresh",)),
|
|
1077
|
+
"max_concurrent_runs": _integer(
|
|
1078
|
+
schedule["max_concurrent_runs"],
|
|
1079
|
+
name,
|
|
1080
|
+
"schedule.max_concurrent_runs",
|
|
1081
|
+
minimum=1,
|
|
1082
|
+
maximum=10,
|
|
1083
|
+
),
|
|
1084
|
+
}
|
|
1085
|
+
return {"table_name": table_name, "schedule": checked_schedule}
|
|
1086
|
+
|
|
1087
|
+
|
|
1088
|
+
def check_recipe(document: Any) -> tuple[dict[str, Any], bytes, str]:
|
|
1089
|
+
"""``recipe.json``: its coordinates, its canonical bytes and their digest.
|
|
1090
|
+
|
|
1091
|
+
Only what the proposal restates is checked -- the schema version, the first-recipe coordinate,
|
|
1092
|
+
the source inventory's shape, and the two members the plan's digests derive from. The transform
|
|
1093
|
+
rules, the Reader pins and the unit semantics are NOT checked here: that is the local lane's
|
|
1094
|
+
job and imports the modules this profile forbids, and the hosted worker refuses a recipe its
|
|
1095
|
+
engine cannot run.
|
|
1096
|
+
"""
|
|
1097
|
+
|
|
1098
|
+
name = "recipe.json"
|
|
1099
|
+
recipe = _object(document, name)
|
|
1100
|
+
for field in (
|
|
1101
|
+
"schema_version",
|
|
1102
|
+
"recipe_id",
|
|
1103
|
+
"recipe_version",
|
|
1104
|
+
"predecessor_recipe_digest",
|
|
1105
|
+
"sources",
|
|
1106
|
+
"transform_plan",
|
|
1107
|
+
"validation",
|
|
1108
|
+
):
|
|
1109
|
+
if field not in recipe:
|
|
1110
|
+
raise _invalid(name, "", f"is missing {field!r}")
|
|
1111
|
+
_enum(recipe["schema_version"], name, "schema_version", RECIPE_SCHEMA_VERSIONS)
|
|
1112
|
+
_string(
|
|
1113
|
+
recipe["recipe_id"],
|
|
1114
|
+
name,
|
|
1115
|
+
"recipe_id",
|
|
1116
|
+
pattern=_IDENTIFIER,
|
|
1117
|
+
what="a recipe identifier (letters, digits, '_', '.', '-')",
|
|
1118
|
+
)
|
|
1119
|
+
if recipe["recipe_version"] != 1:
|
|
1120
|
+
raise ProposeError(
|
|
1121
|
+
"THIN_PROPOSE_RECIPE_COORDINATE",
|
|
1122
|
+
"a hosted-first recipe is a first recipe: recipe.json must carry recipe_version 1 "
|
|
1123
|
+
f"(it carries {recipe['recipe_version']!r})",
|
|
1124
|
+
)
|
|
1125
|
+
if recipe["predecessor_recipe_digest"] is not None:
|
|
1126
|
+
raise ProposeError(
|
|
1127
|
+
"THIN_PROPOSE_RECIPE_COORDINATE",
|
|
1128
|
+
"a hosted-first recipe has no predecessor: recipe.json predecessor_recipe_digest "
|
|
1129
|
+
"must be null",
|
|
1130
|
+
)
|
|
1131
|
+
sources = recipe["sources"]
|
|
1132
|
+
if not isinstance(sources, list) or not sources:
|
|
1133
|
+
raise _invalid(name, "sources", "must be a non-empty list")
|
|
1134
|
+
seen: set[str] = set()
|
|
1135
|
+
for index, item in enumerate(sources):
|
|
1136
|
+
source = _object(item, name, f"sources[{index}]")
|
|
1137
|
+
source_id = _string(
|
|
1138
|
+
source.get("source_id"),
|
|
1139
|
+
name,
|
|
1140
|
+
f"sources[{index}].source_id",
|
|
1141
|
+
pattern=_IDENTIFIER,
|
|
1142
|
+
what="a source identifier",
|
|
1143
|
+
)
|
|
1144
|
+
_string(
|
|
1145
|
+
source.get("adapter_id"),
|
|
1146
|
+
name,
|
|
1147
|
+
f"sources[{index}].adapter_id",
|
|
1148
|
+
pattern=_IDENTIFIER,
|
|
1149
|
+
what="an adapter identifier",
|
|
1150
|
+
)
|
|
1151
|
+
if source_id in seen:
|
|
1152
|
+
raise _invalid(
|
|
1153
|
+
name,
|
|
1154
|
+
f"sources[{index}].source_id",
|
|
1155
|
+
f"repeats {source_id!r}; a recipe names each source once",
|
|
1156
|
+
)
|
|
1157
|
+
seen.add(source_id)
|
|
1158
|
+
_object(recipe["transform_plan"], name, "transform_plan")
|
|
1159
|
+
validation = _object(recipe["validation"], name, "validation")
|
|
1160
|
+
_string(
|
|
1161
|
+
validation.get("policy_digest"),
|
|
1162
|
+
name,
|
|
1163
|
+
"validation.policy_digest",
|
|
1164
|
+
pattern=_BARE_DIGEST,
|
|
1165
|
+
what="a bare lowercase sha256 digest",
|
|
1166
|
+
)
|
|
1167
|
+
if "backfill_policy" in recipe:
|
|
1168
|
+
check_backfill_policy(recipe["backfill_policy"], name, "backfill_policy")
|
|
1169
|
+
try:
|
|
1170
|
+
canonical_bytes = canonical_json_bytes(recipe)
|
|
1171
|
+
except CanonicalJSONError as error:
|
|
1172
|
+
raise ProposeError(
|
|
1173
|
+
"THIN_PROPOSE_RECIPE_NONCANONICAL",
|
|
1174
|
+
f"recipe.json cannot be canonicalised at {error.path}: {error.detail}",
|
|
1175
|
+
) from error
|
|
1176
|
+
if len(canonical_bytes) > MAX_CANONICAL_RECIPE_BYTES:
|
|
1177
|
+
raise ProposeError(
|
|
1178
|
+
"THIN_PROPOSE_RECIPE_TOO_LARGE",
|
|
1179
|
+
f"recipe.json is {len(canonical_bytes)} canonical bytes, over the "
|
|
1180
|
+
f"{MAX_CANONICAL_RECIPE_BYTES}-byte contract ceiling",
|
|
1181
|
+
)
|
|
1182
|
+
return recipe, canonical_bytes, hashlib.sha256(canonical_bytes).hexdigest()
|
|
1183
|
+
|
|
1184
|
+
|
|
1185
|
+
def harness_version() -> str:
|
|
1186
|
+
"""The X.Y.Z this client records as ``harness_version``, refused when there is none.
|
|
1187
|
+
|
|
1188
|
+
Checked before the first request rather than at the proposal stage: a checkout that reports
|
|
1189
|
+
``uninstalled`` would otherwise register four stages of resources and then stop.
|
|
1190
|
+
"""
|
|
1191
|
+
|
|
1192
|
+
version = package_version()
|
|
1193
|
+
if _HARNESS_VERSION.fullmatch(version) is None:
|
|
1194
|
+
raise ProposeError(
|
|
1195
|
+
"THIN_PROPOSE_HARNESS_VERSION",
|
|
1196
|
+
f"this client reports version {version!r}, which is not the X.Y.Z a hosted bootstrap "
|
|
1197
|
+
"records; run propose from an installed release",
|
|
1198
|
+
)
|
|
1199
|
+
return version
|
|
1200
|
+
|
|
1201
|
+
|
|
1202
|
+
# -- the derived plan digests -----------------------------------------------------------------
|
|
1203
|
+
|
|
1204
|
+
|
|
1205
|
+
def execution_plan_digest(recipe: Mapping[str, Any]) -> str:
|
|
1206
|
+
"""The digest of the recipe's transform plan as the hosted worker seals it.
|
|
1207
|
+
|
|
1208
|
+
The worker writes ``plan.json`` as the plan's sorted-key compact JSON terminated by one
|
|
1209
|
+
newline and binds that member's digest; for a plan in its normalized form those bytes are the
|
|
1210
|
+
canonical bytes plus the newline, so the plan's ``execution_plan_digest`` can describe the
|
|
1211
|
+
recipe exactly without this lane carrying the engine.
|
|
1212
|
+
"""
|
|
1213
|
+
|
|
1214
|
+
return (
|
|
1215
|
+
"sha256:"
|
|
1216
|
+
+ hashlib.sha256(canonical_json_bytes(recipe["transform_plan"]) + b"\n").hexdigest()
|
|
1217
|
+
)
|
|
1218
|
+
|
|
1219
|
+
|
|
1220
|
+
def validation_policy_digest(recipe: Mapping[str, Any]) -> str:
|
|
1221
|
+
return "sha256:" + str(recipe["validation"]["policy_digest"])
|
|
1222
|
+
|
|
1223
|
+
|
|
1224
|
+
def backfill_policy_digest(policy: Mapping[str, Any]) -> str:
|
|
1225
|
+
return "sha256:" + canonical_sha256(dict(policy))
|
|
1226
|
+
|
|
1227
|
+
|
|
1228
|
+
def plan_content(plan: Mapping[str, Any], recipe: Mapping[str, Any]) -> dict[str, Any]:
|
|
1229
|
+
"""The table plan's authored content, its digests derived from the recipe and held to it."""
|
|
1230
|
+
|
|
1231
|
+
name = "table_plan.json"
|
|
1232
|
+
derived = {
|
|
1233
|
+
"execution_plan_digest": execution_plan_digest(recipe),
|
|
1234
|
+
"validation_policy_digest": validation_policy_digest(recipe),
|
|
1235
|
+
}
|
|
1236
|
+
policy = plan.get("approved_backfill_policy")
|
|
1237
|
+
if policy is None:
|
|
1238
|
+
policy = recipe.get("backfill_policy")
|
|
1239
|
+
if not isinstance(policy, dict):
|
|
1240
|
+
raise _invalid(
|
|
1241
|
+
name,
|
|
1242
|
+
"approved_backfill_policy",
|
|
1243
|
+
"is required when recipe.json carries no backfill_policy",
|
|
1244
|
+
)
|
|
1245
|
+
policy = check_backfill_policy(policy, "recipe.json", "backfill_policy")
|
|
1246
|
+
derived["backfill_policy_digest"] = backfill_policy_digest(policy)
|
|
1247
|
+
for field, expected in derived.items():
|
|
1248
|
+
written = plan.get(field)
|
|
1249
|
+
if written is not None and written != expected:
|
|
1250
|
+
raise ProposeError(
|
|
1251
|
+
"THIN_PROPOSE_PLAN_DIGEST",
|
|
1252
|
+
f"table_plan.json {field} is {written}, but recipe.json derives {expected}; leave "
|
|
1253
|
+
"the field out to have it derived, or correct it",
|
|
1254
|
+
)
|
|
1255
|
+
return {
|
|
1256
|
+
"output_grain": plan["output_grain"],
|
|
1257
|
+
"transformation_contract_version": plan["transformation_contract_version"],
|
|
1258
|
+
"approved_backfill_policy": policy,
|
|
1259
|
+
**derived,
|
|
1260
|
+
}
|
|
1261
|
+
|
|
1262
|
+
|
|
1263
|
+
# --------------------------------------------------------------------------------------------
|
|
1264
|
+
# The journal
|
|
1265
|
+
# --------------------------------------------------------------------------------------------
|
|
1266
|
+
|
|
1267
|
+
|
|
1268
|
+
class Journal:
|
|
1269
|
+
"""The record of one bundle's authoring, in ``BUNDLE_DIR/.mr-data``.
|
|
1270
|
+
|
|
1271
|
+
It carries a nonce minted when it is created -- the idempotency keys are derived from it, so
|
|
1272
|
+
two bundles with byte-identical documents can never replay each other's resources -- the
|
|
1273
|
+
digest of every document pinned at the moment the first resource built from it completed, and
|
|
1274
|
+
one entry per created or adopted resource: the request digest it was created from, the
|
|
1275
|
+
idempotency key that created it, and the resource Studio answered with. A re-run replays every
|
|
1276
|
+
completed resource from here without rebuilding it; a pinned document that changed is refused
|
|
1277
|
+
with the document named.
|
|
1278
|
+
"""
|
|
1279
|
+
|
|
1280
|
+
def __init__(self, bundle: Path) -> None:
|
|
1281
|
+
self._dir = bundle / JOURNAL_DIR
|
|
1282
|
+
self._path = self._dir / JOURNAL_FILE
|
|
1283
|
+
self.state: dict[str, Any] = {}
|
|
1284
|
+
|
|
1285
|
+
@property
|
|
1286
|
+
def path(self) -> Path:
|
|
1287
|
+
return self._path
|
|
1288
|
+
|
|
1289
|
+
def load_or_start(self, *, workspace_id: UUID) -> None:
|
|
1290
|
+
if self._path.is_file():
|
|
1291
|
+
self.state = read_journal(self._path)
|
|
1292
|
+
if self.state.get("workspace_id") != str(workspace_id):
|
|
1293
|
+
raise ProposeError(
|
|
1294
|
+
"THIN_PROPOSE_JOURNAL_SCOPE",
|
|
1295
|
+
"this bundle was proposed under a different workspace; a bundle belongs to the "
|
|
1296
|
+
"workspace that first proposed it",
|
|
1297
|
+
)
|
|
1298
|
+
return
|
|
1299
|
+
nonce = secrets.token_hex(16)
|
|
1300
|
+
self.state = {
|
|
1301
|
+
"schema_version": JOURNAL_SCHEMA,
|
|
1302
|
+
"workspace_id": str(workspace_id),
|
|
1303
|
+
"nonce": nonce,
|
|
1304
|
+
"identity": canonical_sha256({"workspace_id": str(workspace_id), "nonce": nonce}),
|
|
1305
|
+
"documents": {},
|
|
1306
|
+
"operations": {},
|
|
1307
|
+
"resources": {},
|
|
1308
|
+
"etags": {},
|
|
1309
|
+
"ids": {},
|
|
1310
|
+
"through": None,
|
|
1311
|
+
}
|
|
1312
|
+
self.save()
|
|
1313
|
+
|
|
1314
|
+
def assert_pinned_documents_unchanged(self, digests: Mapping[str, str]) -> None:
|
|
1315
|
+
pinned = self.state.get("documents", {})
|
|
1316
|
+
drifted = sorted(
|
|
1317
|
+
name for name, digest in digests.items() if name in pinned and pinned[name] != digest
|
|
1318
|
+
)
|
|
1319
|
+
if drifted:
|
|
1320
|
+
raise ProposeError(
|
|
1321
|
+
"THIN_PROPOSE_BUNDLE_DRIFTED",
|
|
1322
|
+
f"{', '.join(drifted)} changed after the resource built from it was created, so "
|
|
1323
|
+
"the resource Studio holds no longer describes the document; restore the document "
|
|
1324
|
+
"to resume, or start a new bundle that adopts the created resources by id",
|
|
1325
|
+
)
|
|
1326
|
+
|
|
1327
|
+
def pin(self, names: Sequence[str], digests: Mapping[str, str]) -> None:
|
|
1328
|
+
for name in names:
|
|
1329
|
+
if name in digests:
|
|
1330
|
+
self.state.setdefault("documents", {})[name] = digests[name]
|
|
1331
|
+
|
|
1332
|
+
def save(self) -> None:
|
|
1333
|
+
"""Write the journal whole, through a temp file and one rename, mode 0600."""
|
|
1334
|
+
|
|
1335
|
+
self._dir.mkdir(mode=0o700, exist_ok=True)
|
|
1336
|
+
rendered = json.dumps(self.state, ensure_ascii=False, sort_keys=True).encode("utf-8")
|
|
1337
|
+
staged = self._path.with_name(self._path.name + ".partial")
|
|
1338
|
+
descriptor = os.open(staged, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
|
|
1339
|
+
try:
|
|
1340
|
+
os.write(descriptor, rendered)
|
|
1341
|
+
os.fsync(descriptor)
|
|
1342
|
+
finally:
|
|
1343
|
+
os.close(descriptor)
|
|
1344
|
+
os.chmod(staged, 0o600)
|
|
1345
|
+
os.replace(staged, self._path)
|
|
1346
|
+
|
|
1347
|
+
def record_id(self, key: str, value: Any) -> None:
|
|
1348
|
+
self.state.setdefault("ids", {})[key] = value
|
|
1349
|
+
|
|
1350
|
+
def record_stage(self, stage: str) -> None:
|
|
1351
|
+
"""Record the furthest stage this bundle reached. A shorter re-run never winds it back."""
|
|
1352
|
+
|
|
1353
|
+
reached = self.state.get("through")
|
|
1354
|
+
if reached is None or STAGES.index(stage) > STAGES.index(reached):
|
|
1355
|
+
self.state["through"] = stage
|
|
1356
|
+
self.save()
|
|
1357
|
+
|
|
1358
|
+
|
|
1359
|
+
def read_journal(path: Path) -> dict[str, Any]:
|
|
1360
|
+
"""One propose journal, read under the plain-file rule, refused unless this client wrote it."""
|
|
1361
|
+
|
|
1362
|
+
try:
|
|
1363
|
+
raw = read_plain_file(path, max_bytes=MAX_JOURNAL_BYTES)
|
|
1364
|
+
parsed = parse_json(raw)
|
|
1365
|
+
except (PlainFileRefusal, OSError, CanonicalJSONError) as error:
|
|
1366
|
+
raise ProposeError(
|
|
1367
|
+
"THIN_PROPOSE_JOURNAL_UNREADABLE",
|
|
1368
|
+
f"the propose journal at {path} could not be read; move it aside to start a fresh "
|
|
1369
|
+
"bundle",
|
|
1370
|
+
) from error
|
|
1371
|
+
if not isinstance(parsed, dict) or parsed.get("schema_version") != JOURNAL_SCHEMA:
|
|
1372
|
+
raise ProposeError(
|
|
1373
|
+
"THIN_PROPOSE_JOURNAL_UNREADABLE",
|
|
1374
|
+
f"the file at {path} is not a propose journal this client wrote",
|
|
1375
|
+
)
|
|
1376
|
+
return parsed
|
|
1377
|
+
|
|
1378
|
+
|
|
1379
|
+
def journal_ids(bundle: Path | str) -> dict[str, Any]:
|
|
1380
|
+
"""The ids one bundle's journal recorded, for ``research-open --bundle`` to read.
|
|
1381
|
+
|
|
1382
|
+
A bundle with no journal has registered nothing yet, and the refusal names the command that
|
|
1383
|
+
registers it.
|
|
1384
|
+
"""
|
|
1385
|
+
|
|
1386
|
+
directory = Path(str(bundle)).expanduser()
|
|
1387
|
+
path = directory / JOURNAL_DIR / JOURNAL_FILE
|
|
1388
|
+
if not path.is_file():
|
|
1389
|
+
raise ProposeError(
|
|
1390
|
+
"THIN_PROPOSE_JOURNAL_UNREADABLE",
|
|
1391
|
+
f"--bundle names {directory}, which has no propose journal at "
|
|
1392
|
+
f"{JOURNAL_DIR}/{JOURNAL_FILE}; run mr-data propose --through sources against it first",
|
|
1393
|
+
)
|
|
1394
|
+
ids = read_journal(path).get("ids")
|
|
1395
|
+
return dict(ids) if isinstance(ids, dict) else {}
|
|
1396
|
+
|
|
1397
|
+
|
|
1398
|
+
# --------------------------------------------------------------------------------------------
|
|
1399
|
+
# One journaled mutation, and one adoption
|
|
1400
|
+
# --------------------------------------------------------------------------------------------
|
|
1401
|
+
|
|
1402
|
+
|
|
1403
|
+
def _idempotency_key(identity: str, operation: str, request_digest: str) -> str:
|
|
1404
|
+
"""A deterministic idempotency key for one request of one bundle.
|
|
1405
|
+
|
|
1406
|
+
Derived from the bundle's identity -- which carries the journal's nonce, so no other bundle
|
|
1407
|
+
can derive it -- the operation, and the request's own digest, the way ``hosted_deploy``
|
|
1408
|
+
folds the whole request into its deployment identity. A crash and resume re-sends the
|
|
1409
|
+
identical request under the identical key and Studio replays its first result; a corrected
|
|
1410
|
+
request after a refusal gets a key of its own.
|
|
1411
|
+
"""
|
|
1412
|
+
|
|
1413
|
+
digest = hashlib.sha256(f"{operation}:{identity}:{request_digest}".encode()).hexdigest()
|
|
1414
|
+
return f"mr-data.propose.v1:{digest}"
|
|
1415
|
+
|
|
1416
|
+
|
|
1417
|
+
def _drifted_fields(journaled: Mapping[str, Any], fields: Mapping[str, str]) -> str:
|
|
1418
|
+
recorded = journaled.get("request_fields")
|
|
1419
|
+
if not isinstance(recorded, Mapping):
|
|
1420
|
+
return "the request"
|
|
1421
|
+
drifted = sorted(
|
|
1422
|
+
field for field in {*recorded, *fields} if recorded.get(field) != fields.get(field)
|
|
1423
|
+
)
|
|
1424
|
+
return ", ".join(drifted) if drifted else "the request"
|
|
1425
|
+
|
|
1426
|
+
|
|
1427
|
+
def _mutate(
|
|
1428
|
+
journal: Journal,
|
|
1429
|
+
client: StudioProposeClient,
|
|
1430
|
+
*,
|
|
1431
|
+
name: str,
|
|
1432
|
+
path: str,
|
|
1433
|
+
build: Callable[[], Mapping[str, Any]],
|
|
1434
|
+
expected: int,
|
|
1435
|
+
document: str | None,
|
|
1436
|
+
method: str = "POST",
|
|
1437
|
+
if_match: str | None = None,
|
|
1438
|
+
lost_answer_recoverable: bool = False,
|
|
1439
|
+
refusal_remedy: Callable[[ThinLaneError], str | None] | None = None,
|
|
1440
|
+
) -> dict[str, Any]:
|
|
1441
|
+
"""Create one resource, or replay the one this bundle already holds.
|
|
1442
|
+
|
|
1443
|
+
``document`` names the bundle document the request is built from, or ``None`` for the two
|
|
1444
|
+
requests built from Studio's own answers (the proposal, the approval). A completed operation
|
|
1445
|
+
built from a document is REBUILT on replay and held to the request that created it: the pin
|
|
1446
|
+
on the document catches a change once its stage completed, but a stage that did not complete
|
|
1447
|
+
-- one source registered, the next refused -- has pinned nothing, and a replay that trusted
|
|
1448
|
+
the journal's position would hand the first source's id to whichever entry now sits there.
|
|
1449
|
+
A mismatch is the document drift it is, named by document and operation. The two requests
|
|
1450
|
+
built from Studio's answers are replayed without rebuilding: this client's own version and
|
|
1451
|
+
Studio's fleet selection legitimately move between runs and are nobody's drift.
|
|
1452
|
+
|
|
1453
|
+
A planned operation whose answer was lost is re-sent under its own key, so Studio replays the
|
|
1454
|
+
first result rather than creating a second resource; if the request it rebuilds is no longer
|
|
1455
|
+
the one that was sent, it is refused, because that first request may have been applied --
|
|
1456
|
+
unless ``lost_answer_recoverable`` says the caller already learned the outcome from Studio, in
|
|
1457
|
+
which case a changed re-send is safe and goes out under its own key.
|
|
1458
|
+
"""
|
|
1459
|
+
|
|
1460
|
+
operations = journal.state["operations"]
|
|
1461
|
+
resources = journal.state["resources"]
|
|
1462
|
+
existing = resources.get(name)
|
|
1463
|
+
if isinstance(existing, dict) and document is None:
|
|
1464
|
+
return existing
|
|
1465
|
+
try:
|
|
1466
|
+
body = dict(build())
|
|
1467
|
+
request_digest = canonical_sha256(body)
|
|
1468
|
+
fields = {field: canonical_sha256(value) for field, value in body.items()}
|
|
1469
|
+
except CanonicalJSONError as error:
|
|
1470
|
+
raise ProposeError(
|
|
1471
|
+
"THIN_PROPOSE_DOCUMENT_INVALID",
|
|
1472
|
+
f"the {name} request cannot be canonicalised at {error.path}: {error.detail}",
|
|
1473
|
+
) from error
|
|
1474
|
+
journaled = operations.get(name)
|
|
1475
|
+
if isinstance(existing, dict):
|
|
1476
|
+
recorded = journaled.get("request_digest") if isinstance(journaled, dict) else None
|
|
1477
|
+
if recorded != request_digest:
|
|
1478
|
+
raise ProposeError(
|
|
1479
|
+
"THIN_PROPOSE_BUNDLE_DRIFTED",
|
|
1480
|
+
f"{document} changed after {name} was created from it "
|
|
1481
|
+
f"({_drifted_fields(journaled or {}, fields)} no longer match), so the resource "
|
|
1482
|
+
"Studio holds no longer describes the document; restore the document to resume, "
|
|
1483
|
+
"or start a new bundle that adopts the created resources by id",
|
|
1484
|
+
)
|
|
1485
|
+
return existing
|
|
1486
|
+
if (
|
|
1487
|
+
isinstance(journaled, dict)
|
|
1488
|
+
and journaled.get("request_digest") != request_digest
|
|
1489
|
+
and not lost_answer_recoverable
|
|
1490
|
+
):
|
|
1491
|
+
raise ProposeError(
|
|
1492
|
+
"THIN_PROPOSE_RESUME_UNCERTAIN",
|
|
1493
|
+
f"{name} was sent before and its answer was lost, and "
|
|
1494
|
+
f"{_drifted_fields(journaled, fields)} here no longer matches what was sent; that "
|
|
1495
|
+
"request may have been applied, so restore "
|
|
1496
|
+
f"{document or 'the request'} to resume it, or start a new bundle",
|
|
1497
|
+
)
|
|
1498
|
+
idempotency_key = _idempotency_key(journal.state["identity"], name, request_digest)
|
|
1499
|
+
operations[name] = {
|
|
1500
|
+
"operation": name,
|
|
1501
|
+
"document": document,
|
|
1502
|
+
"request_digest": request_digest,
|
|
1503
|
+
"request_fields": fields,
|
|
1504
|
+
"idempotency_key": idempotency_key,
|
|
1505
|
+
"state": "planned",
|
|
1506
|
+
}
|
|
1507
|
+
journal.save()
|
|
1508
|
+
captured: dict[str, str] = {}
|
|
1509
|
+
try:
|
|
1510
|
+
result = client.create(
|
|
1511
|
+
path,
|
|
1512
|
+
body,
|
|
1513
|
+
idempotency_key=idempotency_key,
|
|
1514
|
+
expected=expected,
|
|
1515
|
+
method=method,
|
|
1516
|
+
if_match=if_match,
|
|
1517
|
+
response_headers=captured,
|
|
1518
|
+
)
|
|
1519
|
+
except ThinLaneError as refusal:
|
|
1520
|
+
if _nothing_was_applied(refusal):
|
|
1521
|
+
# Studio answered and applied nothing, so the plan is withdrawn and the document that
|
|
1522
|
+
# produced it stays editable: the corrected request goes out under its own key.
|
|
1523
|
+
del operations[name]
|
|
1524
|
+
journal.save()
|
|
1525
|
+
remedy = refusal_remedy(refusal) if refusal_remedy is not None else None
|
|
1526
|
+
if remedy is not None:
|
|
1527
|
+
refusal.detail = f"{refusal.detail}; {remedy}"
|
|
1528
|
+
refusal.args = (f"{refusal.code}: {refusal.detail}",)
|
|
1529
|
+
raise
|
|
1530
|
+
resources[name] = result
|
|
1531
|
+
etag = captured.get("etag")
|
|
1532
|
+
if isinstance(etag, str) and etag:
|
|
1533
|
+
journal.state.setdefault("etags", {})[name] = etag
|
|
1534
|
+
operations[name] = {**operations[name], "state": "completed"}
|
|
1535
|
+
journal.save()
|
|
1536
|
+
return result
|
|
1537
|
+
|
|
1538
|
+
|
|
1539
|
+
def _adopt(
|
|
1540
|
+
journal: Journal,
|
|
1541
|
+
client: StudioProposeClient,
|
|
1542
|
+
*,
|
|
1543
|
+
name: str,
|
|
1544
|
+
path: str,
|
|
1545
|
+
check: Callable[[Mapping[str, Any]], None],
|
|
1546
|
+
missing: str | None = None,
|
|
1547
|
+
) -> dict[str, Any]:
|
|
1548
|
+
"""Read one resource that already exists and hold it the way a created one is held.
|
|
1549
|
+
|
|
1550
|
+
``missing`` is the sentence a 404 becomes, in the document's words, when the id named does
|
|
1551
|
+
not resolve -- an adopted question whose requirements were never written, say -- so the
|
|
1552
|
+
refusal names what to author rather than a route.
|
|
1553
|
+
"""
|
|
1554
|
+
|
|
1555
|
+
resources = journal.state["resources"]
|
|
1556
|
+
existing = resources.get(name)
|
|
1557
|
+
if isinstance(existing, dict):
|
|
1558
|
+
journaled = journal.state["operations"].get(name, {})
|
|
1559
|
+
if journaled.get("state") == "adopted" and journaled.get("path") != path:
|
|
1560
|
+
raise ProposeError(
|
|
1561
|
+
"THIN_PROPOSE_BUNDLE_DRIFTED",
|
|
1562
|
+
f"the id {name} adopts changed after it was adopted; restore it, or start a new "
|
|
1563
|
+
"bundle",
|
|
1564
|
+
)
|
|
1565
|
+
return existing
|
|
1566
|
+
captured: dict[str, str] = {}
|
|
1567
|
+
try:
|
|
1568
|
+
record = client.read(path, response_headers=captured)
|
|
1569
|
+
except ThinLaneError as refusal:
|
|
1570
|
+
if refusal.code == "THIN_NOT_FOUND" and missing is not None:
|
|
1571
|
+
raise ProposeError("THIN_PROPOSE_ADOPT_SCOPE", missing) from refusal
|
|
1572
|
+
raise
|
|
1573
|
+
if record.get("workspace_id") != journal.state["workspace_id"]:
|
|
1574
|
+
raise ProposeError(
|
|
1575
|
+
"THIN_PROPOSE_ADOPT_SCOPE",
|
|
1576
|
+
f"the resource {name} names belongs to another workspace than this bundle's",
|
|
1577
|
+
)
|
|
1578
|
+
check(record)
|
|
1579
|
+
resources[name] = record
|
|
1580
|
+
etag = captured.get("etag")
|
|
1581
|
+
if isinstance(etag, str) and etag:
|
|
1582
|
+
journal.state.setdefault("etags", {})[name] = etag
|
|
1583
|
+
# The path the resource was adopted from is journaled beside it, so a later run that adopts a
|
|
1584
|
+
# different id under the same name is a drift rather than a silent substitution.
|
|
1585
|
+
journal.state["operations"][name] = {"operation": name, "state": "adopted", "path": path}
|
|
1586
|
+
journal.save()
|
|
1587
|
+
return record
|
|
1588
|
+
|
|
1589
|
+
|
|
1590
|
+
# --------------------------------------------------------------------------------------------
|
|
1591
|
+
# The bodies, with the contract applied here
|
|
1592
|
+
# --------------------------------------------------------------------------------------------
|
|
1593
|
+
|
|
1594
|
+
|
|
1595
|
+
def _uuid_response(record: Mapping[str, Any], field: str, name: str) -> str:
|
|
1596
|
+
value = record.get(field)
|
|
1597
|
+
if not isinstance(value, str) or _UUID_TEXT.fullmatch(value) is None:
|
|
1598
|
+
raise ProposeError(
|
|
1599
|
+
"THIN_PROPOSE_RESPONSE_INVALID", f"Studio answered {name} without a usable {field}"
|
|
1600
|
+
)
|
|
1601
|
+
return value
|
|
1602
|
+
|
|
1603
|
+
|
|
1604
|
+
def _text_response(record: Mapping[str, Any], field: str, name: str) -> str:
|
|
1605
|
+
value = record.get(field)
|
|
1606
|
+
if not isinstance(value, str) or not value:
|
|
1607
|
+
raise ProposeError(
|
|
1608
|
+
"THIN_PROPOSE_RESPONSE_INVALID", f"Studio answered {name} without a usable {field}"
|
|
1609
|
+
)
|
|
1610
|
+
return value
|
|
1611
|
+
|
|
1612
|
+
|
|
1613
|
+
def dataset_body(dataset: Mapping[str, Any], *, workspace_id: UUID) -> dict[str, Any]:
|
|
1614
|
+
return {"workspace_id": str(workspace_id), **dataset}
|
|
1615
|
+
|
|
1616
|
+
|
|
1617
|
+
def question_body(
|
|
1618
|
+
question: Mapping[str, Any], *, workspace_id: UUID, dataset_id: str
|
|
1619
|
+
) -> dict[str, Any]:
|
|
1620
|
+
return {
|
|
1621
|
+
"workspace_id": str(workspace_id),
|
|
1622
|
+
"dataset_id": dataset_id,
|
|
1623
|
+
"question": question["question"],
|
|
1624
|
+
}
|
|
1625
|
+
|
|
1626
|
+
|
|
1627
|
+
def requirements_body(question: Mapping[str, Any], *, workspace_id: UUID) -> dict[str, Any]:
|
|
1628
|
+
return {
|
|
1629
|
+
"schema_version": STUDIO_SCHEMA_VERSION,
|
|
1630
|
+
"workspace_id": str(workspace_id),
|
|
1631
|
+
**question["requirements"],
|
|
1632
|
+
}
|
|
1633
|
+
|
|
1634
|
+
|
|
1635
|
+
def source_body(entry: Mapping[str, Any], *, workspace_id: UUID, dataset_id: str) -> dict[str, Any]:
|
|
1636
|
+
return {
|
|
1637
|
+
"schema_version": STUDIO_SCHEMA_VERSION,
|
|
1638
|
+
"workspace_id": str(workspace_id),
|
|
1639
|
+
"dataset_id": dataset_id,
|
|
1640
|
+
**{key: value for key, value in entry.items() if key != "connector"},
|
|
1641
|
+
}
|
|
1642
|
+
|
|
1643
|
+
|
|
1644
|
+
def connector_body(
|
|
1645
|
+
entry: Mapping[str, Any], *, workspace_id: UUID, source_id: str
|
|
1646
|
+
) -> dict[str, Any]:
|
|
1647
|
+
connector = entry["connector"]
|
|
1648
|
+
body: dict[str, Any] = {
|
|
1649
|
+
"schema_version": STUDIO_SCHEMA_VERSION,
|
|
1650
|
+
"workspace_id": str(workspace_id),
|
|
1651
|
+
"source_id": source_id,
|
|
1652
|
+
"adapter_id": connector["adapter_id"],
|
|
1653
|
+
"credential_mode": "none",
|
|
1654
|
+
"crawler_egress_policy_attestation": CONNECTOR_EGRESS_ATTESTATION[connector["adapter_id"]],
|
|
1655
|
+
}
|
|
1656
|
+
if "query" in connector:
|
|
1657
|
+
body["query"] = connector["query"]
|
|
1658
|
+
return body
|
|
1659
|
+
|
|
1660
|
+
|
|
1661
|
+
def table_plan_body(
|
|
1662
|
+
content: Mapping[str, Any],
|
|
1663
|
+
*,
|
|
1664
|
+
workspace_id: UUID,
|
|
1665
|
+
dataset_id: str,
|
|
1666
|
+
question_id: str,
|
|
1667
|
+
requirements_id: str,
|
|
1668
|
+
source_ids: Sequence[str],
|
|
1669
|
+
) -> dict[str, Any]:
|
|
1670
|
+
return {
|
|
1671
|
+
"schema_version": STUDIO_SCHEMA_VERSION,
|
|
1672
|
+
"workspace_id": str(workspace_id),
|
|
1673
|
+
"dataset_id": dataset_id,
|
|
1674
|
+
"question_id": question_id,
|
|
1675
|
+
"requirements_id": requirements_id,
|
|
1676
|
+
"source_ids": list(source_ids),
|
|
1677
|
+
**content,
|
|
1678
|
+
}
|
|
1679
|
+
|
|
1680
|
+
|
|
1681
|
+
def bootstrap_evidence(
|
|
1682
|
+
*, version: str, research_session_id: str | None, research_run_id: str | None
|
|
1683
|
+
) -> dict[str, Any]:
|
|
1684
|
+
evidence: dict[str, Any] = {"origin": HOSTED_BOOTSTRAP_ORIGIN, "harness_version": version}
|
|
1685
|
+
if research_session_id is not None:
|
|
1686
|
+
evidence["research_session_id"] = research_session_id
|
|
1687
|
+
if research_run_id is not None:
|
|
1688
|
+
evidence["research_run_id"] = research_run_id
|
|
1689
|
+
return evidence
|
|
1690
|
+
|
|
1691
|
+
|
|
1692
|
+
def connector_authority(*, source_id: str, configuration: Mapping[str, Any]) -> dict[str, Any]:
|
|
1693
|
+
"""One connector source-authority binding, its digest derived exactly as Studio derives it.
|
|
1694
|
+
|
|
1695
|
+
Built from the connector configuration Studio holds -- adapter, digest, attestation -- so an
|
|
1696
|
+
adopted configuration binds the same way a registered one does. ``source_authority_digest``
|
|
1697
|
+
is the bare SHA-256 over the canonical binding with that field removed, the identical rule
|
|
1698
|
+
``hosted_deploy._source_authorities`` uses and ``_assert_recipe_source_authority_current``
|
|
1699
|
+
recomputes on Studio's side, so the binding is the one the hosted worker is handed and checks
|
|
1700
|
+
the recipe against at build time.
|
|
1701
|
+
"""
|
|
1702
|
+
|
|
1703
|
+
authority: dict[str, Any] = {
|
|
1704
|
+
"source_id": source_id,
|
|
1705
|
+
"authority_kind": "connector",
|
|
1706
|
+
"adapter_id": _text_response(configuration, "adapter_id", "the connector configuration"),
|
|
1707
|
+
"connector_configuration_id": _uuid_response(
|
|
1708
|
+
configuration, "connector_configuration_id", "the connector configuration"
|
|
1709
|
+
),
|
|
1710
|
+
"connector_configuration_digest": _text_response(
|
|
1711
|
+
configuration, "configuration_digest", "the connector configuration"
|
|
1712
|
+
),
|
|
1713
|
+
"credential_mode": "none",
|
|
1714
|
+
"crawler_egress_policy_attestation": _text_response(
|
|
1715
|
+
configuration, "crawler_egress_policy_attestation", "the connector configuration"
|
|
1716
|
+
),
|
|
1717
|
+
}
|
|
1718
|
+
authority["source_authority_digest"] = canonical_sha256(dict(authority))
|
|
1719
|
+
return authority
|
|
1720
|
+
|
|
1721
|
+
|
|
1722
|
+
def proposal_body(
|
|
1723
|
+
*,
|
|
1724
|
+
recipe: Mapping[str, Any],
|
|
1725
|
+
canonical_bytes: bytes,
|
|
1726
|
+
recipe_digest: str,
|
|
1727
|
+
activation: Mapping[str, Any],
|
|
1728
|
+
workspace_id: UUID,
|
|
1729
|
+
dataset_id: str,
|
|
1730
|
+
table_plan_id: str,
|
|
1731
|
+
source_authority_bindings: Sequence[Mapping[str, Any]],
|
|
1732
|
+
evidence: Mapping[str, Any],
|
|
1733
|
+
) -> dict[str, Any]:
|
|
1734
|
+
"""The ``TableRecipeProposalCommand``, every coordinate read straight off the recipe."""
|
|
1735
|
+
|
|
1736
|
+
inventory_digest = canonical_sha256(
|
|
1737
|
+
[
|
|
1738
|
+
{
|
|
1739
|
+
"source_id": item["source_id"],
|
|
1740
|
+
"source_authority_digest": item["source_authority_digest"],
|
|
1741
|
+
}
|
|
1742
|
+
for item in source_authority_bindings
|
|
1743
|
+
]
|
|
1744
|
+
)
|
|
1745
|
+
return {
|
|
1746
|
+
"schema_version": STUDIO_SCHEMA_VERSION,
|
|
1747
|
+
"workspace_id": str(workspace_id),
|
|
1748
|
+
"dataset_id": dataset_id,
|
|
1749
|
+
"table_plan_id": table_plan_id,
|
|
1750
|
+
"recipe_media_type": RECIPE_MEDIA_TYPE,
|
|
1751
|
+
"recipe_schema_version": recipe["schema_version"],
|
|
1752
|
+
"table_recipe_id": recipe["recipe_id"],
|
|
1753
|
+
"recipe_version": 1,
|
|
1754
|
+
"predecessor_recipe_digest": None,
|
|
1755
|
+
"canonical_recipe_json": canonical_bytes.decode("utf-8"),
|
|
1756
|
+
"recipe_digest": recipe_digest,
|
|
1757
|
+
"source_authority_bindings": [dict(item) for item in source_authority_bindings],
|
|
1758
|
+
"source_authority_bindings_digest": inventory_digest,
|
|
1759
|
+
"bootstrap_evidence": dict(evidence),
|
|
1760
|
+
"activation_intent": {
|
|
1761
|
+
"table_name": activation["table_name"],
|
|
1762
|
+
"schedule": dict(activation["schedule"]),
|
|
1763
|
+
},
|
|
1764
|
+
}
|
|
1765
|
+
|
|
1766
|
+
|
|
1767
|
+
def preflight_body(
|
|
1768
|
+
proposal: Mapping[str, Any], *, workspace_id: UUID, recipe_proposal_id: str
|
|
1769
|
+
) -> dict[str, Any]:
|
|
1770
|
+
return {
|
|
1771
|
+
"workspace_id": str(workspace_id),
|
|
1772
|
+
"dataset_id": proposal["dataset_id"],
|
|
1773
|
+
"table_plan_id": proposal["table_plan_id"],
|
|
1774
|
+
"table_recipe_id": proposal["table_recipe_id"],
|
|
1775
|
+
"recipe_version": proposal["recipe_version"],
|
|
1776
|
+
"recipe_digest": proposal["recipe_digest"],
|
|
1777
|
+
"recipe_proposal_id": recipe_proposal_id,
|
|
1778
|
+
}
|
|
1779
|
+
|
|
1780
|
+
|
|
1781
|
+
def approval_body(
|
|
1782
|
+
*,
|
|
1783
|
+
workspace_id: UUID,
|
|
1784
|
+
recipe_proposal_id: str,
|
|
1785
|
+
recipe_digest: str,
|
|
1786
|
+
preflight: Mapping[str, Any],
|
|
1787
|
+
) -> dict[str, Any]:
|
|
1788
|
+
return {
|
|
1789
|
+
"schema_version": STUDIO_SCHEMA_VERSION,
|
|
1790
|
+
"workspace_id": str(workspace_id),
|
|
1791
|
+
"recipe_proposal_id": recipe_proposal_id,
|
|
1792
|
+
"recipe_digest": recipe_digest,
|
|
1793
|
+
"worker_policy_preflight": dict(preflight),
|
|
1794
|
+
}
|
|
1795
|
+
|
|
1796
|
+
|
|
1797
|
+
# --------------------------------------------------------------------------------------------
|
|
1798
|
+
# The command
|
|
1799
|
+
# --------------------------------------------------------------------------------------------
|
|
1800
|
+
|
|
1801
|
+
|
|
1802
|
+
def _session(args: argparse.Namespace) -> StudioSession:
|
|
1803
|
+
return open_studio_session(resolve_cloud_credentials())
|
|
1804
|
+
|
|
1805
|
+
|
|
1806
|
+
def _client(args: argparse.Namespace) -> StudioProposeClient:
|
|
1807
|
+
return StudioProposeClient(_session(args))
|
|
1808
|
+
|
|
1809
|
+
|
|
1810
|
+
def _through(args: argparse.Namespace) -> str:
|
|
1811
|
+
stage = getattr(args, "through", None) or "approval"
|
|
1812
|
+
if stage not in STAGES:
|
|
1813
|
+
allowed = ", ".join(STAGES)
|
|
1814
|
+
raise ProposeError("THIN_PROPOSE_STAGE_INVALID", f"--through is one of: {allowed}")
|
|
1815
|
+
return stage
|
|
1816
|
+
|
|
1817
|
+
|
|
1818
|
+
def _research_session(args: argparse.Namespace, *, stage: str) -> str | None:
|
|
1819
|
+
named = getattr(args, "research_session", None)
|
|
1820
|
+
if named is None:
|
|
1821
|
+
return None
|
|
1822
|
+
try:
|
|
1823
|
+
session_id = str(UUID(str(named)))
|
|
1824
|
+
except (ValueError, AttributeError) as error:
|
|
1825
|
+
raise ProposeError(
|
|
1826
|
+
"THIN_PROPOSE_REQUEST_INVALID", f"{named!r} is not a research session identifier"
|
|
1827
|
+
) from error
|
|
1828
|
+
if STAGES.index(stage) < STAGES.index("proposal"):
|
|
1829
|
+
raise ProposeError(
|
|
1830
|
+
"THIN_PROPOSE_REQUEST_INVALID",
|
|
1831
|
+
"--research-session is sealed into the proposal, and --through "
|
|
1832
|
+
f"{stage} stops before one exists; drop the flag, or run through the proposal",
|
|
1833
|
+
)
|
|
1834
|
+
return session_id
|
|
1835
|
+
|
|
1836
|
+
|
|
1837
|
+
class _Bundle:
|
|
1838
|
+
"""Every document a run needs, read and checked before a credential is resolved."""
|
|
1839
|
+
|
|
1840
|
+
def __init__(self, directory: Path, *, stage: str) -> None:
|
|
1841
|
+
self.directory = directory
|
|
1842
|
+
self.stage = stage
|
|
1843
|
+
self.digests: dict[str, str] = {}
|
|
1844
|
+
raw: dict[str, Any] = {}
|
|
1845
|
+
for step, names in DOCUMENTS.items():
|
|
1846
|
+
if STAGES.index(step) > STAGES.index(stage):
|
|
1847
|
+
continue
|
|
1848
|
+
for name in names:
|
|
1849
|
+
raw[name], self.digests[name] = _read_document(directory, name)
|
|
1850
|
+
self.dataset = check_dataset(raw["dataset.json"])
|
|
1851
|
+
self.question = check_question(raw["question.json"]) if "question.json" in raw else None
|
|
1852
|
+
self.sources = check_sources(raw["sources.json"]) if "sources.json" in raw else None
|
|
1853
|
+
self.plan = check_table_plan(raw["table_plan.json"]) if "table_plan.json" in raw else None
|
|
1854
|
+
self.recipe: dict[str, Any] | None = None
|
|
1855
|
+
self.canonical_recipe = b""
|
|
1856
|
+
self.recipe_digest = ""
|
|
1857
|
+
if "recipe.json" in raw:
|
|
1858
|
+
self.recipe, self.canonical_recipe, self.recipe_digest = check_recipe(
|
|
1859
|
+
raw["recipe.json"]
|
|
1860
|
+
)
|
|
1861
|
+
self.activation = (
|
|
1862
|
+
check_activation(raw["activation.json"]) if "activation.json" in raw else None
|
|
1863
|
+
)
|
|
1864
|
+
self.plan_content = (
|
|
1865
|
+
plan_content(self.plan, self.recipe)
|
|
1866
|
+
if self.plan is not None
|
|
1867
|
+
and "table_plan_id" not in self.plan
|
|
1868
|
+
and self.recipe is not None
|
|
1869
|
+
else None
|
|
1870
|
+
)
|
|
1871
|
+
|
|
1872
|
+
def reaches(self, step: str) -> bool:
|
|
1873
|
+
return STAGES.index(self.stage) >= STAGES.index(step)
|
|
1874
|
+
|
|
1875
|
+
|
|
1876
|
+
def propose(
|
|
1877
|
+
args: argparse.Namespace,
|
|
1878
|
+
*,
|
|
1879
|
+
client: StudioProposeClient | None = None,
|
|
1880
|
+
) -> dict[str, Any]:
|
|
1881
|
+
"""``mr-data propose``: author one hosted recipe proposal from a bundle, and open its approval.
|
|
1882
|
+
|
|
1883
|
+
Everything a client can settle on its own -- the bundle directory, the stage, the research
|
|
1884
|
+
coordinate, every document each reached stage needs and its shape, this client's own version
|
|
1885
|
+
-- is settled before a credential is resolved, so a mistyped path or a malformed document costs
|
|
1886
|
+
a sentence rather than a token and a 422 several routes in. The journal makes every stage
|
|
1887
|
+
idempotent: a re-run replays what was created, and a pinned document that changed is refused
|
|
1888
|
+
with the document named.
|
|
1889
|
+
"""
|
|
1890
|
+
|
|
1891
|
+
directory = _bundle_directory(getattr(args, "bundle_dir", None))
|
|
1892
|
+
stage = _through(args)
|
|
1893
|
+
research_session_id = _research_session(args, stage=stage)
|
|
1894
|
+
announce = getattr(args, "json", False) is False
|
|
1895
|
+
bundle = _Bundle(directory, stage=stage)
|
|
1896
|
+
version = harness_version() if bundle.reaches("proposal") else None
|
|
1897
|
+
|
|
1898
|
+
selected = client or _client(args)
|
|
1899
|
+
workspace_id = selected.session.workspace_id
|
|
1900
|
+
journal = Journal(directory)
|
|
1901
|
+
journal.load_or_start(workspace_id=workspace_id)
|
|
1902
|
+
journal.assert_pinned_documents_unchanged(bundle.digests)
|
|
1903
|
+
|
|
1904
|
+
def note(line: str) -> None:
|
|
1905
|
+
if announce:
|
|
1906
|
+
print(line, flush=True)
|
|
1907
|
+
|
|
1908
|
+
# dataset -------------------------------------------------------------------------------
|
|
1909
|
+
adopted_dataset = "dataset_id" in bundle.dataset
|
|
1910
|
+
if adopted_dataset:
|
|
1911
|
+
dataset = _adopt(
|
|
1912
|
+
journal,
|
|
1913
|
+
selected,
|
|
1914
|
+
name="dataset",
|
|
1915
|
+
path=GET_DATASET_PATH.format(dataset_id=bundle.dataset["dataset_id"]),
|
|
1916
|
+
check=lambda record: None,
|
|
1917
|
+
missing="dataset.json adopts a Dataset this workspace does not hold",
|
|
1918
|
+
)
|
|
1919
|
+
else:
|
|
1920
|
+
dataset = _mutate(
|
|
1921
|
+
journal,
|
|
1922
|
+
selected,
|
|
1923
|
+
name="dataset",
|
|
1924
|
+
path=CREATE_DATASET_PATH,
|
|
1925
|
+
build=lambda: dataset_body(bundle.dataset, workspace_id=workspace_id),
|
|
1926
|
+
expected=201,
|
|
1927
|
+
document="dataset.json",
|
|
1928
|
+
refusal_remedy=_dataset_conflict_remedy,
|
|
1929
|
+
)
|
|
1930
|
+
dataset_id = _uuid_response(dataset, "dataset_id", "the dataset")
|
|
1931
|
+
journal.record_id("dataset_id", dataset_id)
|
|
1932
|
+
journal.pin(PINS["dataset"], bundle.digests)
|
|
1933
|
+
journal.record_stage("dataset")
|
|
1934
|
+
note(f"dataset {dataset_id} {'adopted' if adopted_dataset else 'registered'}")
|
|
1935
|
+
if stage == "dataset":
|
|
1936
|
+
return _stage_receipt(selected, journal, stage=stage)
|
|
1937
|
+
|
|
1938
|
+
# question + requirements ---------------------------------------------------------------
|
|
1939
|
+
question_document = bundle.question
|
|
1940
|
+
assert question_document is not None
|
|
1941
|
+
if "question_id" in question_document:
|
|
1942
|
+
|
|
1943
|
+
def _same_dataset(record: Mapping[str, Any]) -> None:
|
|
1944
|
+
if record.get("dataset_id") != dataset_id:
|
|
1945
|
+
raise ProposeError(
|
|
1946
|
+
"THIN_PROPOSE_ADOPT_SCOPE",
|
|
1947
|
+
"question.json adopts a question that belongs to another Dataset than this "
|
|
1948
|
+
"bundle's",
|
|
1949
|
+
)
|
|
1950
|
+
|
|
1951
|
+
question = _adopt(
|
|
1952
|
+
journal,
|
|
1953
|
+
selected,
|
|
1954
|
+
name="question",
|
|
1955
|
+
path=GET_QUESTION_PATH.format(question_id=question_document["question_id"]),
|
|
1956
|
+
check=_same_dataset,
|
|
1957
|
+
missing="question.json adopts a question this workspace does not hold",
|
|
1958
|
+
)
|
|
1959
|
+
question_id = _uuid_response(question, "question_id", "the question")
|
|
1960
|
+
requirements = _adopt(
|
|
1961
|
+
journal,
|
|
1962
|
+
selected,
|
|
1963
|
+
name="requirements",
|
|
1964
|
+
path=GET_REQUIREMENTS_PATH.format(question_id=question_id),
|
|
1965
|
+
check=lambda record: None,
|
|
1966
|
+
missing=(
|
|
1967
|
+
"question.json adopts a question whose requirements were never written; author "
|
|
1968
|
+
"the question and its requirements in question.json instead, or write the "
|
|
1969
|
+
"requirements for that question first"
|
|
1970
|
+
),
|
|
1971
|
+
)
|
|
1972
|
+
else:
|
|
1973
|
+
if adopted_dataset and "question" not in journal.state["resources"]:
|
|
1974
|
+
_refuse_a_question_the_dataset_already_holds(
|
|
1975
|
+
selected, dataset_id=dataset_id, text=question_document["question"]
|
|
1976
|
+
)
|
|
1977
|
+
question = _mutate(
|
|
1978
|
+
journal,
|
|
1979
|
+
selected,
|
|
1980
|
+
name="question",
|
|
1981
|
+
path=CREATE_QUESTION_PATH,
|
|
1982
|
+
build=lambda: question_body(
|
|
1983
|
+
question_document, workspace_id=workspace_id, dataset_id=dataset_id
|
|
1984
|
+
),
|
|
1985
|
+
expected=201,
|
|
1986
|
+
document="question.json",
|
|
1987
|
+
)
|
|
1988
|
+
question_id = _uuid_response(question, "question_id", "the question")
|
|
1989
|
+
requirements = _mutate(
|
|
1990
|
+
journal,
|
|
1991
|
+
selected,
|
|
1992
|
+
name="requirements",
|
|
1993
|
+
path=PUT_REQUIREMENTS_PATH.format(question_id=question_id),
|
|
1994
|
+
build=lambda: requirements_body(question_document, workspace_id=workspace_id),
|
|
1995
|
+
expected=200,
|
|
1996
|
+
document="question.json",
|
|
1997
|
+
method="PUT",
|
|
1998
|
+
if_match="*",
|
|
1999
|
+
)
|
|
2000
|
+
requirements_id = _uuid_response(requirements, "requirements_id", "the requirements")
|
|
2001
|
+
journal.record_id("question_id", question_id)
|
|
2002
|
+
journal.record_id("requirements_id", requirements_id)
|
|
2003
|
+
journal.pin(PINS["question"], bundle.digests)
|
|
2004
|
+
journal.record_stage("question")
|
|
2005
|
+
note(f"question {question_id} recorded, with its requirements")
|
|
2006
|
+
if stage == "question":
|
|
2007
|
+
return _stage_receipt(selected, journal, stage=stage)
|
|
2008
|
+
|
|
2009
|
+
# sources + connectors ------------------------------------------------------------------
|
|
2010
|
+
entries = bundle.sources
|
|
2011
|
+
assert entries is not None
|
|
2012
|
+
_refuse_a_registered_source_the_bundle_no_longer_names(journal, entries)
|
|
2013
|
+
if adopted_dataset:
|
|
2014
|
+
_refuse_sources_the_dataset_already_holds(
|
|
2015
|
+
selected, journal, dataset_id=dataset_id, entries=entries
|
|
2016
|
+
)
|
|
2017
|
+
source_ids: dict[str, str] = {}
|
|
2018
|
+
configurations: dict[str, dict[str, Any]] = {}
|
|
2019
|
+
for entry in entries:
|
|
2020
|
+
name = entry["name"]
|
|
2021
|
+
if "source_id" in entry:
|
|
2022
|
+
|
|
2023
|
+
def _in_dataset(record: Mapping[str, Any], *, name: str = name) -> None:
|
|
2024
|
+
if record.get("dataset_id") != dataset_id:
|
|
2025
|
+
raise ProposeError(
|
|
2026
|
+
"THIN_PROPOSE_ADOPT_SCOPE",
|
|
2027
|
+
f"sources.json {name!r} adopts a source registered under another Dataset "
|
|
2028
|
+
"than this bundle's",
|
|
2029
|
+
)
|
|
2030
|
+
|
|
2031
|
+
source = _adopt(
|
|
2032
|
+
journal,
|
|
2033
|
+
selected,
|
|
2034
|
+
name=f"source:{name}",
|
|
2035
|
+
path=GET_SOURCE_PATH.format(source_id=entry["source_id"]),
|
|
2036
|
+
check=_in_dataset,
|
|
2037
|
+
missing=f"sources.json {name!r} adopts a source this workspace does not hold",
|
|
2038
|
+
)
|
|
2039
|
+
source_id = _uuid_response(source, "source_id", f"the source {name!r}")
|
|
2040
|
+
|
|
2041
|
+
def _for_source(
|
|
2042
|
+
record: Mapping[str, Any], *, expected: str = source_id, name: str = name
|
|
2043
|
+
) -> None:
|
|
2044
|
+
if record.get("source_id") != expected:
|
|
2045
|
+
raise ProposeError(
|
|
2046
|
+
"THIN_PROPOSE_ADOPT_SCOPE",
|
|
2047
|
+
f"sources.json {name!r} adopts a connector configuration registered for "
|
|
2048
|
+
"another source",
|
|
2049
|
+
)
|
|
2050
|
+
|
|
2051
|
+
configuration = _adopt(
|
|
2052
|
+
journal,
|
|
2053
|
+
selected,
|
|
2054
|
+
name=f"connector:{name}",
|
|
2055
|
+
path=GET_CONNECTOR_PATH.format(
|
|
2056
|
+
connector_configuration_id=entry["connector_configuration_id"]
|
|
2057
|
+
),
|
|
2058
|
+
check=_for_source,
|
|
2059
|
+
missing=(
|
|
2060
|
+
f"sources.json {name!r} adopts a connector configuration this workspace "
|
|
2061
|
+
"does not hold"
|
|
2062
|
+
),
|
|
2063
|
+
)
|
|
2064
|
+
else:
|
|
2065
|
+
source = _mutate(
|
|
2066
|
+
journal,
|
|
2067
|
+
selected,
|
|
2068
|
+
name=f"source:{name}",
|
|
2069
|
+
path=REGISTER_SOURCE_PATH,
|
|
2070
|
+
build=lambda entry=entry: source_body(
|
|
2071
|
+
entry, workspace_id=workspace_id, dataset_id=dataset_id
|
|
2072
|
+
),
|
|
2073
|
+
expected=201,
|
|
2074
|
+
document="sources.json",
|
|
2075
|
+
)
|
|
2076
|
+
source_id = _uuid_response(source, "source_id", f"the source {name!r}")
|
|
2077
|
+
configuration = _mutate(
|
|
2078
|
+
journal,
|
|
2079
|
+
selected,
|
|
2080
|
+
name=f"connector:{name}",
|
|
2081
|
+
path=REGISTER_CONNECTOR_PATH,
|
|
2082
|
+
build=lambda entry=entry, source_id=source_id: connector_body(
|
|
2083
|
+
entry, workspace_id=workspace_id, source_id=source_id
|
|
2084
|
+
),
|
|
2085
|
+
expected=201,
|
|
2086
|
+
document="sources.json",
|
|
2087
|
+
)
|
|
2088
|
+
source_ids[name] = source_id
|
|
2089
|
+
configurations[source_id] = configuration
|
|
2090
|
+
note(f"source {name!r} is {source_id}, with its connector")
|
|
2091
|
+
journal.record_id("sources", source_ids)
|
|
2092
|
+
journal.record_id("source_ids", list(source_ids.values()))
|
|
2093
|
+
journal.pin(PINS["sources"], bundle.digests)
|
|
2094
|
+
journal.record_stage("sources")
|
|
2095
|
+
if stage == "sources":
|
|
2096
|
+
return _stage_receipt(selected, journal, stage=stage)
|
|
2097
|
+
|
|
2098
|
+
# the recipe against the registered sources, and the session it was explored in ---------
|
|
2099
|
+
recipe = bundle.recipe
|
|
2100
|
+
assert recipe is not None
|
|
2101
|
+
_check_recipe_sources(recipe, source_ids=source_ids, configurations=configurations)
|
|
2102
|
+
research_run_id: str | None = None
|
|
2103
|
+
if research_session_id is not None:
|
|
2104
|
+
record = _lookup_session(
|
|
2105
|
+
selected, research_session_id, dataset_id=dataset_id, question_id=question_id
|
|
2106
|
+
)
|
|
2107
|
+
research_run_id = _uuid_response(record, "run_id", "the research session")
|
|
2108
|
+
|
|
2109
|
+
# plan ----------------------------------------------------------------------------------
|
|
2110
|
+
plan_document = bundle.plan
|
|
2111
|
+
assert plan_document is not None
|
|
2112
|
+
if "table_plan_id" in plan_document:
|
|
2113
|
+
|
|
2114
|
+
def _binds_this_bundle(record: Mapping[str, Any]) -> None:
|
|
2115
|
+
if record.get("dataset_id") != dataset_id or record.get("question_id") != question_id:
|
|
2116
|
+
raise ProposeError(
|
|
2117
|
+
"THIN_PROPOSE_ADOPT_SCOPE",
|
|
2118
|
+
"table_plan.json adopts a plan that binds another Dataset or question than "
|
|
2119
|
+
"this bundle's",
|
|
2120
|
+
)
|
|
2121
|
+
if record.get("requirements_id") != requirements_id:
|
|
2122
|
+
raise ProposeError(
|
|
2123
|
+
"THIN_PROPOSE_ADOPT_SCOPE",
|
|
2124
|
+
"table_plan.json adopts a plan bound to other requirements than this "
|
|
2125
|
+
"bundle's question carries",
|
|
2126
|
+
)
|
|
2127
|
+
planned_sources = record.get("source_ids")
|
|
2128
|
+
if not isinstance(planned_sources, list) or set(planned_sources) != set(
|
|
2129
|
+
source_ids.values()
|
|
2130
|
+
):
|
|
2131
|
+
raise ProposeError(
|
|
2132
|
+
"THIN_PROPOSE_ADOPT_SCOPE",
|
|
2133
|
+
"table_plan.json adopts a plan whose sources are not the sources this bundle "
|
|
2134
|
+
"registered or adopted",
|
|
2135
|
+
)
|
|
2136
|
+
|
|
2137
|
+
plan = _adopt(
|
|
2138
|
+
journal,
|
|
2139
|
+
selected,
|
|
2140
|
+
name="plan",
|
|
2141
|
+
path=GET_TABLE_PLAN_PATH.format(table_plan_id=plan_document["table_plan_id"]),
|
|
2142
|
+
check=_binds_this_bundle,
|
|
2143
|
+
missing="table_plan.json adopts a plan this workspace does not hold",
|
|
2144
|
+
)
|
|
2145
|
+
else:
|
|
2146
|
+
content = bundle.plan_content
|
|
2147
|
+
assert content is not None
|
|
2148
|
+
plan = _mutate(
|
|
2149
|
+
journal,
|
|
2150
|
+
selected,
|
|
2151
|
+
name="plan",
|
|
2152
|
+
path=CREATE_TABLE_PLAN_PATH,
|
|
2153
|
+
build=lambda: table_plan_body(
|
|
2154
|
+
content,
|
|
2155
|
+
workspace_id=workspace_id,
|
|
2156
|
+
dataset_id=dataset_id,
|
|
2157
|
+
question_id=question_id,
|
|
2158
|
+
requirements_id=requirements_id,
|
|
2159
|
+
source_ids=list(source_ids.values()),
|
|
2160
|
+
),
|
|
2161
|
+
expected=201,
|
|
2162
|
+
# The plan's digests derive from the recipe, so a recipe that moved after the plan
|
|
2163
|
+
# was created is a drift of this operation too, and the sentence names both.
|
|
2164
|
+
document="table_plan.json or recipe.json",
|
|
2165
|
+
)
|
|
2166
|
+
table_plan_id = _uuid_response(plan, "table_plan_id", "the table plan")
|
|
2167
|
+
journal.record_id("table_plan_id", table_plan_id)
|
|
2168
|
+
journal.pin(PINS["plan"], bundle.digests)
|
|
2169
|
+
journal.record_stage("plan")
|
|
2170
|
+
note(f"table plan {table_plan_id} in place")
|
|
2171
|
+
if stage == "plan":
|
|
2172
|
+
return _stage_receipt(selected, journal, stage=stage)
|
|
2173
|
+
|
|
2174
|
+
# proposal ------------------------------------------------------------------------------
|
|
2175
|
+
activation = bundle.activation
|
|
2176
|
+
assert activation is not None and version is not None
|
|
2177
|
+
_check_plan_describes_recipe(plan, recipe)
|
|
2178
|
+
bindings = sorted(
|
|
2179
|
+
(
|
|
2180
|
+
connector_authority(source_id=source_id, configuration=configurations[source_id])
|
|
2181
|
+
for source_id in source_ids.values()
|
|
2182
|
+
),
|
|
2183
|
+
key=lambda item: item["source_id"],
|
|
2184
|
+
)
|
|
2185
|
+
command = proposal_body(
|
|
2186
|
+
recipe=recipe,
|
|
2187
|
+
canonical_bytes=bundle.canonical_recipe,
|
|
2188
|
+
recipe_digest=bundle.recipe_digest,
|
|
2189
|
+
activation=activation,
|
|
2190
|
+
workspace_id=workspace_id,
|
|
2191
|
+
dataset_id=dataset_id,
|
|
2192
|
+
table_plan_id=table_plan_id,
|
|
2193
|
+
source_authority_bindings=bindings,
|
|
2194
|
+
evidence=bootstrap_evidence(
|
|
2195
|
+
version=version,
|
|
2196
|
+
research_session_id=research_session_id,
|
|
2197
|
+
research_run_id=research_run_id,
|
|
2198
|
+
),
|
|
2199
|
+
)
|
|
2200
|
+
proposal = _mutate(
|
|
2201
|
+
journal,
|
|
2202
|
+
selected,
|
|
2203
|
+
name="proposal",
|
|
2204
|
+
path=CREATE_RECIPE_PROPOSAL_PATH,
|
|
2205
|
+
build=lambda: command,
|
|
2206
|
+
expected=201,
|
|
2207
|
+
document=None,
|
|
2208
|
+
)
|
|
2209
|
+
recipe_proposal_id = _uuid_response(proposal, "recipe_proposal_id", "the recipe proposal")
|
|
2210
|
+
recipe_digest = _text_response(proposal, "recipe_digest", "the recipe proposal")
|
|
2211
|
+
if recipe_digest != bundle.recipe_digest and "recipe_digest" not in journal.state["ids"]:
|
|
2212
|
+
raise ProposeError(
|
|
2213
|
+
"THIN_PROPOSE_RESPONSE_INVALID",
|
|
2214
|
+
"Studio stored a recipe digest other than the one computed from recipe.json",
|
|
2215
|
+
)
|
|
2216
|
+
journal.record_id("recipe_proposal_id", recipe_proposal_id)
|
|
2217
|
+
journal.record_id("recipe_digest", recipe_digest)
|
|
2218
|
+
journal.pin(PINS["proposal"], bundle.digests)
|
|
2219
|
+
journal.record_stage("proposal")
|
|
2220
|
+
note(f"recipe proposal {recipe_proposal_id} holds recipe digest {recipe_digest}")
|
|
2221
|
+
if stage == "proposal":
|
|
2222
|
+
return _proposal_receipt(selected, journal, stage=stage)
|
|
2223
|
+
|
|
2224
|
+
# approval ------------------------------------------------------------------------------
|
|
2225
|
+
# An approval this bundle already opened is replayed from the journal. Otherwise the proposal
|
|
2226
|
+
# is read back first -- for the ETag the ask is pinned to, which the create never carries,
|
|
2227
|
+
# and because the proposal is where Studio records an approval whose answer was lost: one it
|
|
2228
|
+
# already holds is adopted rather than asked for again. The preflight is a stateless resolve
|
|
2229
|
+
# that Studio's fleet can move between runs, so it is asked for only when the ask is still to
|
|
2230
|
+
# be made, and a re-send after a lost answer is safe because the outcome was just read.
|
|
2231
|
+
if not isinstance(journal.state["resources"].get("approval"), dict):
|
|
2232
|
+
proposal_etag, held = _proposal_state(selected, journal, recipe_proposal_id, recipe_digest)
|
|
2233
|
+
if held is not None:
|
|
2234
|
+
_adopt_approval(journal, selected, held)
|
|
2235
|
+
else:
|
|
2236
|
+
preflight = selected.worker_policy_preflight(
|
|
2237
|
+
preflight_body(
|
|
2238
|
+
command, workspace_id=workspace_id, recipe_proposal_id=recipe_proposal_id
|
|
2239
|
+
)
|
|
2240
|
+
)
|
|
2241
|
+
_mutate(
|
|
2242
|
+
journal,
|
|
2243
|
+
selected,
|
|
2244
|
+
name="approval",
|
|
2245
|
+
path=REQUEST_RECIPE_APPROVAL_PATH.format(recipe_proposal_id=recipe_proposal_id),
|
|
2246
|
+
build=lambda: approval_body(
|
|
2247
|
+
workspace_id=workspace_id,
|
|
2248
|
+
recipe_proposal_id=recipe_proposal_id,
|
|
2249
|
+
recipe_digest=recipe_digest,
|
|
2250
|
+
preflight=preflight,
|
|
2251
|
+
),
|
|
2252
|
+
expected=201,
|
|
2253
|
+
document=None,
|
|
2254
|
+
if_match=proposal_etag,
|
|
2255
|
+
lost_answer_recoverable=True,
|
|
2256
|
+
)
|
|
2257
|
+
approval = journal.state["resources"]["approval"]
|
|
2258
|
+
approval_request_id = _uuid_response(approval, "approval_request_id", "the approval request")
|
|
2259
|
+
journal.record_id("approval_request_id", approval_request_id)
|
|
2260
|
+
journal.record_stage("approval")
|
|
2261
|
+
note(f"approval request {approval_request_id} open")
|
|
2262
|
+
return _proposal_receipt(selected, journal, stage="approval")
|
|
2263
|
+
|
|
2264
|
+
|
|
2265
|
+
def _dataset_conflict_remedy(refusal: ThinLaneError) -> str | None:
|
|
2266
|
+
"""What to do when Studio will not register a second Dataset of this name.
|
|
2267
|
+
|
|
2268
|
+
The commonest way to meet this is a bundle whose journal is gone: the Dataset was registered
|
|
2269
|
+
by an earlier run and nothing remembers it. The remedy is the id, not a new name.
|
|
2270
|
+
"""
|
|
2271
|
+
|
|
2272
|
+
if getattr(refusal, "http_status", None) != 409:
|
|
2273
|
+
return None
|
|
2274
|
+
return (
|
|
2275
|
+
"a Dataset of this name already exists in the workspace; if an earlier run of this bundle "
|
|
2276
|
+
'registered it, adopt it by id in dataset.json ({"dataset_id": "..."}) -- '
|
|
2277
|
+
"mr-data list names the workspace's datasets -- or choose another name"
|
|
2278
|
+
)
|
|
2279
|
+
|
|
2280
|
+
|
|
2281
|
+
def _refuse_a_question_the_dataset_already_holds(
|
|
2282
|
+
client: StudioProposeClient, *, dataset_id: str, text: str
|
|
2283
|
+
) -> None:
|
|
2284
|
+
"""Under an adopted Dataset, a question whose text the Dataset already holds is adopted, not
|
|
2285
|
+
registered again.
|
|
2286
|
+
|
|
2287
|
+
This is the lost-journal case one step on: the Dataset was adopted by id, and the question
|
|
2288
|
+
that went with it would be registered a second time in silence. The list is one page of the
|
|
2289
|
+
Dataset's own questions -- the bound Studio's list route has -- which is every question a
|
|
2290
|
+
bundle could have registered.
|
|
2291
|
+
"""
|
|
2292
|
+
|
|
2293
|
+
for question in client._call_list(
|
|
2294
|
+
"/v3/questions", query={"dataset_id": dataset_id, "limit": 200}
|
|
2295
|
+
):
|
|
2296
|
+
if question.get("text") == text and isinstance(question.get("question_id"), str):
|
|
2297
|
+
raise ProposeError(
|
|
2298
|
+
"THIN_PROPOSE_ADOPT_REQUIRED",
|
|
2299
|
+
f"the adopted Dataset already holds this question as {question['question_id']}; "
|
|
2300
|
+
'adopt it in question.json ({"question_id": "..."}) rather than registering it '
|
|
2301
|
+
"again",
|
|
2302
|
+
)
|
|
2303
|
+
|
|
2304
|
+
|
|
2305
|
+
def _refuse_sources_the_dataset_already_holds(
|
|
2306
|
+
client: StudioProposeClient,
|
|
2307
|
+
journal: Journal,
|
|
2308
|
+
*,
|
|
2309
|
+
dataset_id: str,
|
|
2310
|
+
entries: Sequence[Mapping[str, Any]],
|
|
2311
|
+
) -> None:
|
|
2312
|
+
"""Under an adopted Dataset, a source whose name the Dataset already holds is adopted, not
|
|
2313
|
+
registered again -- the same lost-journal case, for the sources."""
|
|
2314
|
+
|
|
2315
|
+
authored = {
|
|
2316
|
+
entry["name"]
|
|
2317
|
+
for entry in entries
|
|
2318
|
+
if "source_id" not in entry and f"source:{entry['name']}" not in journal.state["resources"]
|
|
2319
|
+
}
|
|
2320
|
+
if not authored:
|
|
2321
|
+
return
|
|
2322
|
+
for source in client._call_list(REGISTER_SOURCE_PATH, query={"limit": 200}):
|
|
2323
|
+
if source.get("dataset_id") != dataset_id or source.get("name") not in authored:
|
|
2324
|
+
continue
|
|
2325
|
+
source_id = source.get("source_id")
|
|
2326
|
+
raise ProposeError(
|
|
2327
|
+
"THIN_PROPOSE_ADOPT_REQUIRED",
|
|
2328
|
+
f"the adopted Dataset already holds a source named {source.get('name')!r} as "
|
|
2329
|
+
f'{source_id}; adopt it in sources.json ({{"name": ..., "source_id": '
|
|
2330
|
+
f'"{source_id}", "connector_configuration_id": "..."}}) -- '
|
|
2331
|
+
f"GET /v3/connector-configurations?source_id={source_id} names its configuration -- "
|
|
2332
|
+
"rather than registering it again",
|
|
2333
|
+
)
|
|
2334
|
+
|
|
2335
|
+
|
|
2336
|
+
def _refuse_a_registered_source_the_bundle_no_longer_names(
|
|
2337
|
+
journal: Journal, entries: Sequence[Mapping[str, Any]]
|
|
2338
|
+
) -> None:
|
|
2339
|
+
"""A source this bundle registered and then stopped naming is a drift, not an orphan.
|
|
2340
|
+
|
|
2341
|
+
Sources are journaled by the name the author gave them, so reordering the list is harmless
|
|
2342
|
+
and renaming an entry after its source was registered is not: the registered source would be
|
|
2343
|
+
left behind and a second one registered under the new name. Naming the missing one is the
|
|
2344
|
+
whole of the remedy.
|
|
2345
|
+
"""
|
|
2346
|
+
|
|
2347
|
+
named = {entry["name"] for entry in entries}
|
|
2348
|
+
registered = sorted(
|
|
2349
|
+
operation[len("source:") :]
|
|
2350
|
+
for operation, journaled in journal.state["operations"].items()
|
|
2351
|
+
if operation.startswith("source:") and journaled.get("state") in {"completed", "adopted"}
|
|
2352
|
+
)
|
|
2353
|
+
missing = [name for name in registered if name not in named]
|
|
2354
|
+
if missing:
|
|
2355
|
+
raise ProposeError(
|
|
2356
|
+
"THIN_PROPOSE_BUNDLE_DRIFTED",
|
|
2357
|
+
f"sources.json no longer names {', '.join(repr(name) for name in missing)}, which this "
|
|
2358
|
+
"bundle already registered; restore the entry under that name, or start a new bundle",
|
|
2359
|
+
)
|
|
2360
|
+
|
|
2361
|
+
|
|
2362
|
+
def _check_recipe_sources(
|
|
2363
|
+
recipe: Mapping[str, Any],
|
|
2364
|
+
*,
|
|
2365
|
+
source_ids: Mapping[str, str],
|
|
2366
|
+
configurations: Mapping[str, Mapping[str, Any]],
|
|
2367
|
+
) -> None:
|
|
2368
|
+
"""The recipe names exactly the registered sources, each under its connector's adapter."""
|
|
2369
|
+
|
|
2370
|
+
registered = set(source_ids.values())
|
|
2371
|
+
named = {item["source_id"]: item for item in recipe["sources"]}
|
|
2372
|
+
if set(named) != registered:
|
|
2373
|
+
raise ProposeError(
|
|
2374
|
+
"THIN_PROPOSE_SOURCE_INVENTORY",
|
|
2375
|
+
"recipe.json sources must reference exactly the registered source ids "
|
|
2376
|
+
f"({', '.join(sorted(registered))}); author recipe.json after "
|
|
2377
|
+
"`mr-data propose --through sources` and reference the ids the journal recorded",
|
|
2378
|
+
)
|
|
2379
|
+
for source_id, source in named.items():
|
|
2380
|
+
adapter_id = configurations[source_id].get("adapter_id")
|
|
2381
|
+
if source.get("adapter_id") != adapter_id:
|
|
2382
|
+
raise ProposeError(
|
|
2383
|
+
"THIN_PROPOSE_SOURCE_INVENTORY",
|
|
2384
|
+
f"recipe.json names source {source_id} under adapter {source.get('adapter_id')!r}, "
|
|
2385
|
+
f"but its connector is registered as {adapter_id!r}",
|
|
2386
|
+
)
|
|
2387
|
+
|
|
2388
|
+
|
|
2389
|
+
def _check_plan_describes_recipe(plan: Mapping[str, Any], recipe: Mapping[str, Any]) -> None:
|
|
2390
|
+
"""The plan Studio holds -- created earlier, or adopted -- still describes this recipe."""
|
|
2391
|
+
|
|
2392
|
+
expected = {
|
|
2393
|
+
"plan_digest": execution_plan_digest(recipe),
|
|
2394
|
+
"validation_policy_digest": validation_policy_digest(recipe),
|
|
2395
|
+
}
|
|
2396
|
+
stale = sorted(field for field, value in expected.items() if plan.get(field) != value)
|
|
2397
|
+
if stale:
|
|
2398
|
+
raise ProposeError(
|
|
2399
|
+
"THIN_PROPOSE_PLAN_DIGEST",
|
|
2400
|
+
f"the table plan's {', '.join(stale)} no longer describes recipe.json: the plan was "
|
|
2401
|
+
"built from an earlier recipe; restore that recipe, or start a new bundle that adopts "
|
|
2402
|
+
"the Dataset, question and sources by id and creates a plan for this one",
|
|
2403
|
+
)
|
|
2404
|
+
|
|
2405
|
+
|
|
2406
|
+
def _lookup_session(
|
|
2407
|
+
client: StudioProposeClient, session_id: str, *, dataset_id: str, question_id: str
|
|
2408
|
+
) -> dict[str, Any]:
|
|
2409
|
+
"""The research session named as evidence, held to what Studio will hold it to.
|
|
2410
|
+
|
|
2411
|
+
Read before the plan exists rather than at the proposal, so a session that explored another
|
|
2412
|
+
Dataset, that failed, or that probed nothing is a sentence here and not a 409 after the plan
|
|
2413
|
+
has been created -- every one of these is a check Studio makes on the proposal.
|
|
2414
|
+
"""
|
|
2415
|
+
|
|
2416
|
+
record = client.read(GET_SESSION_PATH.format(session_id=session_id))
|
|
2417
|
+
if record.get("session_id") != session_id:
|
|
2418
|
+
raise ProposeError(
|
|
2419
|
+
"THIN_PROPOSE_REQUEST_INVALID",
|
|
2420
|
+
"Studio answered about a different research session than the one named as evidence",
|
|
2421
|
+
)
|
|
2422
|
+
context = record.get("context")
|
|
2423
|
+
if not isinstance(context, Mapping) or context.get("dataset_id") != dataset_id:
|
|
2424
|
+
raise ProposeError(
|
|
2425
|
+
"THIN_PROPOSE_RESEARCH_SESSION",
|
|
2426
|
+
"--research-session names a session that explored another Dataset than this bundle's",
|
|
2427
|
+
)
|
|
2428
|
+
if context.get("question_id") != question_id:
|
|
2429
|
+
raise ProposeError(
|
|
2430
|
+
"THIN_PROPOSE_RESEARCH_SESSION",
|
|
2431
|
+
"--research-session names a session that explored another question than this bundle's",
|
|
2432
|
+
)
|
|
2433
|
+
if record.get("state") == "failed":
|
|
2434
|
+
raise ProposeError(
|
|
2435
|
+
"THIN_PROPOSE_RESEARCH_SESSION",
|
|
2436
|
+
"--research-session names a session that failed; its run holds no probe results",
|
|
2437
|
+
)
|
|
2438
|
+
submitted = record.get("probes_submitted")
|
|
2439
|
+
if type(submitted) is not int or submitted < 1:
|
|
2440
|
+
raise ProposeError(
|
|
2441
|
+
"THIN_PROPOSE_RESEARCH_SESSION",
|
|
2442
|
+
"--research-session names a session that submitted no probe; it explored nothing "
|
|
2443
|
+
"this recipe could rest on",
|
|
2444
|
+
)
|
|
2445
|
+
return record
|
|
2446
|
+
|
|
2447
|
+
|
|
2448
|
+
def _proposal_state(
|
|
2449
|
+
client: StudioProposeClient, journal: Journal, recipe_proposal_id: str, recipe_digest: str
|
|
2450
|
+
) -> tuple[str, dict[str, Any] | None]:
|
|
2451
|
+
"""The proposal's current ETag, and the approval Studio already holds for it, if any.
|
|
2452
|
+
|
|
2453
|
+
Read EVERY time the approval is still to be opened, never from the journal: the proposal's
|
|
2454
|
+
version moves when an approval is requested, so an ETag journaled before a lost answer is
|
|
2455
|
+
exactly the stale one a 412 would refuse forever. And the proposal is where Studio records the
|
|
2456
|
+
approval a lost ``POST`` did open -- ``approval_request_id`` on the proposal -- so reading it
|
|
2457
|
+
is how that outcome is learned rather than asked for twice.
|
|
2458
|
+
"""
|
|
2459
|
+
|
|
2460
|
+
captured: dict[str, str] = {}
|
|
2461
|
+
record = client.recipe_proposal(recipe_proposal_id, response_headers=captured)
|
|
2462
|
+
if record.get("recipe_proposal_id") != recipe_proposal_id:
|
|
2463
|
+
raise ProposeError(
|
|
2464
|
+
"THIN_PROPOSE_RESPONSE_INVALID", "Studio answered about a different recipe proposal"
|
|
2465
|
+
)
|
|
2466
|
+
if record.get("recipe_digest") != recipe_digest:
|
|
2467
|
+
raise ProposeError(
|
|
2468
|
+
"THIN_PROPOSE_RESPONSE_INVALID",
|
|
2469
|
+
"the recipe proposal Studio holds no longer carries the recipe digest this bundle "
|
|
2470
|
+
"proposed",
|
|
2471
|
+
)
|
|
2472
|
+
etag = captured.get("etag")
|
|
2473
|
+
if not isinstance(etag, str) or not etag:
|
|
2474
|
+
raise ProposeError(
|
|
2475
|
+
"THIN_PROPOSE_RESPONSE_INVALID",
|
|
2476
|
+
"Studio answered the proposal without an ETag, and the approval ask must be pinned "
|
|
2477
|
+
"to the exact proposal version it was made against",
|
|
2478
|
+
)
|
|
2479
|
+
journal.state.setdefault("etags", {})["proposal"] = etag
|
|
2480
|
+
journal.save()
|
|
2481
|
+
held = record.get("approval_request_id")
|
|
2482
|
+
if isinstance(held, str) and _UUID_TEXT.fullmatch(held) is not None:
|
|
2483
|
+
return etag, client.read(GET_APPROVAL_PATH.format(approval_request_id=held))
|
|
2484
|
+
return etag, None
|
|
2485
|
+
|
|
2486
|
+
|
|
2487
|
+
def _adopt_approval(
|
|
2488
|
+
journal: Journal, client: StudioProposeClient, record: Mapping[str, Any]
|
|
2489
|
+
) -> None:
|
|
2490
|
+
"""Hold the approval Studio already opened for this proposal as this bundle's own."""
|
|
2491
|
+
|
|
2492
|
+
approval_request_id = _uuid_response(record, "approval_request_id", "the approval request")
|
|
2493
|
+
journal.state["resources"]["approval"] = dict(record)
|
|
2494
|
+
journal.state["operations"].pop("approval", None)
|
|
2495
|
+
journal.state["operations"]["approval"] = {
|
|
2496
|
+
"operation": "approval",
|
|
2497
|
+
"state": "adopted",
|
|
2498
|
+
"path": GET_APPROVAL_PATH.format(approval_request_id=approval_request_id),
|
|
2499
|
+
}
|
|
2500
|
+
journal.save()
|
|
2501
|
+
|
|
2502
|
+
|
|
2503
|
+
# --------------------------------------------------------------------------------------------
|
|
2504
|
+
# The receipts
|
|
2505
|
+
# --------------------------------------------------------------------------------------------
|
|
2506
|
+
|
|
2507
|
+
|
|
2508
|
+
def _common(client: StudioProposeClient, journal: Journal, *, stage: str) -> dict[str, Any]:
|
|
2509
|
+
ids = dict(journal.state.get("ids", {}))
|
|
2510
|
+
payload = {
|
|
2511
|
+
"schema_version": PROPOSE_SCHEMA,
|
|
2512
|
+
"lane": "hosted",
|
|
2513
|
+
"bundle": str(journal.path.parent.parent),
|
|
2514
|
+
"journal": str(journal.path),
|
|
2515
|
+
"through": stage,
|
|
2516
|
+
"workspace_id": journal.state["workspace_id"],
|
|
2517
|
+
"ids": ids,
|
|
2518
|
+
}
|
|
2519
|
+
dataset_id = ids.get("dataset_id")
|
|
2520
|
+
if isinstance(dataset_id, str):
|
|
2521
|
+
payload["dataset_dashboard_url"] = dataset_dashboard_url(
|
|
2522
|
+
client.session.cloud_url, dataset_id
|
|
2523
|
+
)
|
|
2524
|
+
return payload
|
|
2525
|
+
|
|
2526
|
+
|
|
2527
|
+
def _stage_receipt(client: StudioProposeClient, journal: Journal, *, stage: str) -> dict[str, Any]:
|
|
2528
|
+
"""The receipt for a ``--through`` stop before the proposal exists.
|
|
2529
|
+
|
|
2530
|
+
Its status is ``authoring_stage_complete`` for every one of the four early stages, and the
|
|
2531
|
+
payload names which. After ``--through sources`` it carries the registered source ids under
|
|
2532
|
+
``ids.sources`` so the agent can write ``recipe.json`` against them; that is the whole reason
|
|
2533
|
+
the stage exists.
|
|
2534
|
+
"""
|
|
2535
|
+
|
|
2536
|
+
payload = _common(client, journal, stage=stage)
|
|
2537
|
+
payload["status"] = "authoring_stage_complete"
|
|
2538
|
+
next_stage = STAGES[STAGES.index(stage) + 1]
|
|
2539
|
+
payload["note"] = (
|
|
2540
|
+
f"Stopped after the {stage} stage. Re-run mr-data propose to continue through the "
|
|
2541
|
+
f"{next_stage} stage"
|
|
2542
|
+
+ (
|
|
2543
|
+
"; the registered source ids are under ids.sources for recipe.json to reference"
|
|
2544
|
+
if stage == "sources"
|
|
2545
|
+
else ""
|
|
2546
|
+
)
|
|
2547
|
+
+ "."
|
|
2548
|
+
)
|
|
2549
|
+
return payload
|
|
2550
|
+
|
|
2551
|
+
|
|
2552
|
+
def _proposal_receipt(
|
|
2553
|
+
client: StudioProposeClient, journal: Journal, *, stage: str
|
|
2554
|
+
) -> dict[str, Any]:
|
|
2555
|
+
"""The receipt once the proposal exists: the proposal, its digest, and what settles it.
|
|
2556
|
+
|
|
2557
|
+
When the approval was opened too, the receipt reads the approval back and says which of four
|
|
2558
|
+
things is true, because each is a different next step: the approval is still PENDING, and a
|
|
2559
|
+
person settles it at the dashboard address (exit 2); it is APPROVED and no Build has started,
|
|
2560
|
+
and ``mr-data build --hosted`` queues one; it is approved and the confirmation STARTED the
|
|
2561
|
+
Build in the same act -- Studio #124 -- and the run is named here because ``mr-data build``
|
|
2562
|
+
would refuse it rather than replay it; or it was declined (exit 2). Both commands carry
|
|
2563
|
+
``--hosted``, because on a full install the bare names are the local lane's. Nothing here
|
|
2564
|
+
decides anything.
|
|
2565
|
+
"""
|
|
2566
|
+
|
|
2567
|
+
payload = _common(client, journal, stage=stage)
|
|
2568
|
+
ids = journal.state.get("ids", {})
|
|
2569
|
+
recipe_proposal_id = ids.get("recipe_proposal_id")
|
|
2570
|
+
recipe_digest = ids.get("recipe_digest")
|
|
2571
|
+
payload["recipe_proposal_id"] = recipe_proposal_id
|
|
2572
|
+
payload["recipe_digest"] = recipe_digest
|
|
2573
|
+
build_command = (
|
|
2574
|
+
f"mr-data build --recipe-proposal {recipe_proposal_id} --recipe-digest {recipe_digest} "
|
|
2575
|
+
"--hosted"
|
|
2576
|
+
)
|
|
2577
|
+
if stage == "proposal":
|
|
2578
|
+
payload["status"] = "recipe_proposed"
|
|
2579
|
+
payload["build_command"] = build_command
|
|
2580
|
+
payload["note"] = (
|
|
2581
|
+
"The recipe proposal exists and nothing has approved it. Re-run mr-data propose to "
|
|
2582
|
+
"open its approval; then a person settles it and you build."
|
|
2583
|
+
)
|
|
2584
|
+
return payload
|
|
2585
|
+
approval_request_id = str(ids.get("approval_request_id"))
|
|
2586
|
+
payload["approval_request_id"] = approval_request_id
|
|
2587
|
+
payload["dashboard_url"] = approval_dashboard_url(client.session.cloud_url, approval_request_id)
|
|
2588
|
+
payload["never_decides"] = (
|
|
2589
|
+
"this command opened an approval; it did not grant one. A Table recipe is confirmed by a "
|
|
2590
|
+
"signed-in editor in the dashboard, which the command line cannot be."
|
|
2591
|
+
)
|
|
2592
|
+
approval = client.read(GET_APPROVAL_PATH.format(approval_request_id=approval_request_id))
|
|
2593
|
+
approval_status = approval.get("status")
|
|
2594
|
+
payload["approval_status"] = approval_status
|
|
2595
|
+
if approval_status == "pending":
|
|
2596
|
+
payload["status"] = "recipe_proposal_approval_requested"
|
|
2597
|
+
payload["approve_command"] = (
|
|
2598
|
+
f"mr-data recipe-approve --recipe {approval_request_id} "
|
|
2599
|
+
'--reason "why you are asking" --hosted'
|
|
2600
|
+
)
|
|
2601
|
+
payload["build_command"] = build_command
|
|
2602
|
+
payload["note"] = (
|
|
2603
|
+
"The proposal is open and waiting on a person. Open the dashboard address above, or "
|
|
2604
|
+
"run the recipe-approve command, so a human settles it; then build with the build "
|
|
2605
|
+
"command -- unless their confirmation started the Build, which re-running this "
|
|
2606
|
+
"command reports."
|
|
2607
|
+
)
|
|
2608
|
+
return payload
|
|
2609
|
+
if approval_status != "approved":
|
|
2610
|
+
payload["status"] = "recipe_proposal_not_approved"
|
|
2611
|
+
payload["note"] = (
|
|
2612
|
+
f"The approval settled as {approval_status}; nothing will build from this proposal. "
|
|
2613
|
+
"Author the recipe again in a new bundle that adopts the Dataset, question and "
|
|
2614
|
+
"sources by id."
|
|
2615
|
+
)
|
|
2616
|
+
return payload
|
|
2617
|
+
started = client.approval_start(approval_request_id)
|
|
2618
|
+
if started is None:
|
|
2619
|
+
payload["status"] = "recipe_proposal_approved"
|
|
2620
|
+
payload["build_command"] = build_command
|
|
2621
|
+
payload["note"] = (
|
|
2622
|
+
"A person approved the recipe and no Build has been started for it; queue the first "
|
|
2623
|
+
"hosted Build with the build command. It replays to the same run if one is started "
|
|
2624
|
+
"meanwhile."
|
|
2625
|
+
)
|
|
2626
|
+
return payload
|
|
2627
|
+
run = started.get("run") if isinstance(started.get("run"), Mapping) else {}
|
|
2628
|
+
run_id = run.get("run_id")
|
|
2629
|
+
if not isinstance(run_id, str):
|
|
2630
|
+
raise ProposeError(
|
|
2631
|
+
"THIN_PROPOSE_RESPONSE_INVALID", "Studio named a started Build without a run"
|
|
2632
|
+
)
|
|
2633
|
+
payload["status"] = "recipe_build_queued"
|
|
2634
|
+
payload["run_id"] = run_id
|
|
2635
|
+
payload["run_dashboard_url"] = run_dashboard_url(client.session.cloud_url, run_id)
|
|
2636
|
+
payload["watch_command"] = f"mr-data watch {run_id} --hosted"
|
|
2637
|
+
payload["note"] = (
|
|
2638
|
+
"The confirmation started the first hosted Build in the same act; there is nothing left "
|
|
2639
|
+
"to queue, and mr-data build would refuse rather than repeat it. Follow this run."
|
|
2640
|
+
)
|
|
2641
|
+
return payload
|
|
2642
|
+
|
|
2643
|
+
|
|
2644
|
+
def propose_exit_code(payload: Mapping[str, Any]) -> int:
|
|
2645
|
+
"""``0`` when the requested work is done, ``2`` when a person still has to decide, or did not.
|
|
2646
|
+
|
|
2647
|
+
A ``--through`` stop before the approval, a proposal created without opening its approval, an
|
|
2648
|
+
approved proposal, and a Build the approval already started all did exactly what was asked
|
|
2649
|
+
and exit 0. An approval a human has not settled exits 2 and names the dashboard, the way
|
|
2650
|
+
``mr-data approve`` exits 2 while an ask is pending -- because a script gating on this must
|
|
2651
|
+
not read "asked" as "approved" -- and so does one a human declined.
|
|
2652
|
+
"""
|
|
2653
|
+
|
|
2654
|
+
if payload.get("status") in {
|
|
2655
|
+
"recipe_proposal_approval_requested",
|
|
2656
|
+
"recipe_proposal_not_approved",
|
|
2657
|
+
}:
|
|
2658
|
+
return 2
|
|
2659
|
+
return 0
|
|
2660
|
+
|
|
2661
|
+
|
|
2662
|
+
# --------------------------------------------------------------------------------------------
|
|
2663
|
+
# The command line, declared once for both parsers
|
|
2664
|
+
# --------------------------------------------------------------------------------------------
|
|
2665
|
+
|
|
2666
|
+
COMMAND = "propose"
|
|
2667
|
+
COMMAND_HELP = "author one hosted recipe proposal from a bundle and open its approval"
|
|
2668
|
+
|
|
2669
|
+
|
|
2670
|
+
def declare_arguments(parser: argparse.ArgumentParser) -> None:
|
|
2671
|
+
"""Add ``propose``'s arguments to ``parser``.
|
|
2672
|
+
|
|
2673
|
+
Called by both front doors -- the thin router's parser and ``cli``'s -- so the argument surface
|
|
2674
|
+
is declared once and the two cannot disagree, the property ``note`` and the research commands
|
|
2675
|
+
have by the same construction.
|
|
2676
|
+
"""
|
|
2677
|
+
|
|
2678
|
+
parser.add_argument(
|
|
2679
|
+
"bundle_dir",
|
|
2680
|
+
help="the directory holding the documents to author from; its .mr-data journal is written "
|
|
2681
|
+
"here",
|
|
2682
|
+
)
|
|
2683
|
+
parser.add_argument(
|
|
2684
|
+
"--through",
|
|
2685
|
+
choices=STAGES,
|
|
2686
|
+
help="stop after this stage instead of opening the approval; "
|
|
2687
|
+
"--through sources registers the sources first so recipe.json can reference their ids",
|
|
2688
|
+
)
|
|
2689
|
+
parser.add_argument(
|
|
2690
|
+
"--research-session",
|
|
2691
|
+
dest="research_session",
|
|
2692
|
+
metavar="UUID",
|
|
2693
|
+
help="the research session this recipe was explored in; its id and run are sealed into "
|
|
2694
|
+
"the hosted bootstrap evidence",
|
|
2695
|
+
)
|
|
2696
|
+
|
|
2697
|
+
|
|
2698
|
+
__all__ = [
|
|
2699
|
+
"COMMAND",
|
|
2700
|
+
"COMMAND_HELP",
|
|
2701
|
+
"CREATE_DATASET_PATH",
|
|
2702
|
+
"CREATE_QUESTION_PATH",
|
|
2703
|
+
"CREATE_RECIPE_PROPOSAL_PATH",
|
|
2704
|
+
"CREATE_TABLE_PLAN_PATH",
|
|
2705
|
+
"DEFAULT_SCHEDULE",
|
|
2706
|
+
"GET_CONNECTOR_PATH",
|
|
2707
|
+
"GET_DATASET_PATH",
|
|
2708
|
+
"GET_QUESTION_PATH",
|
|
2709
|
+
"GET_RECIPE_PROPOSAL_PATH",
|
|
2710
|
+
"GET_REQUIREMENTS_PATH",
|
|
2711
|
+
"GET_SOURCE_PATH",
|
|
2712
|
+
"GET_TABLE_PLAN_PATH",
|
|
2713
|
+
"HOSTED_BOOTSTRAP_ORIGIN",
|
|
2714
|
+
"JOURNAL_DIR",
|
|
2715
|
+
"JOURNAL_FILE",
|
|
2716
|
+
"JOURNAL_SCHEMA",
|
|
2717
|
+
"MAX_DOCUMENT_BYTES",
|
|
2718
|
+
"MAX_JOURNAL_BYTES",
|
|
2719
|
+
"PINS",
|
|
2720
|
+
"PROPOSE_SCHEMA",
|
|
2721
|
+
"PUT_REQUIREMENTS_PATH",
|
|
2722
|
+
"RECIPE_MEDIA_TYPE",
|
|
2723
|
+
"RECIPE_SCHEMA_VERSIONS",
|
|
2724
|
+
"REGISTER_CONNECTOR_PATH",
|
|
2725
|
+
"REGISTER_SOURCE_PATH",
|
|
2726
|
+
"REQUEST_RECIPE_APPROVAL_PATH",
|
|
2727
|
+
"STAGES",
|
|
2728
|
+
"WORKER_POLICY_PREFLIGHT_PATH",
|
|
2729
|
+
"Journal",
|
|
2730
|
+
"ProposeError",
|
|
2731
|
+
"StudioProposeClient",
|
|
2732
|
+
"approval_body",
|
|
2733
|
+
"backfill_policy_digest",
|
|
2734
|
+
"bootstrap_evidence",
|
|
2735
|
+
"check_activation",
|
|
2736
|
+
"check_dataset",
|
|
2737
|
+
"check_question",
|
|
2738
|
+
"check_recipe",
|
|
2739
|
+
"check_sources",
|
|
2740
|
+
"check_table_plan",
|
|
2741
|
+
"connector_authority",
|
|
2742
|
+
"connector_body",
|
|
2743
|
+
"dataset_body",
|
|
2744
|
+
"declare_arguments",
|
|
2745
|
+
"execution_plan_digest",
|
|
2746
|
+
"harness_version",
|
|
2747
|
+
"journal_ids",
|
|
2748
|
+
"plan_content",
|
|
2749
|
+
"preflight_body",
|
|
2750
|
+
"proposal_body",
|
|
2751
|
+
"propose",
|
|
2752
|
+
"propose_exit_code",
|
|
2753
|
+
"question_body",
|
|
2754
|
+
"read_journal",
|
|
2755
|
+
"requirements_body",
|
|
2756
|
+
"source_body",
|
|
2757
|
+
"table_plan_body",
|
|
2758
|
+
"validation_policy_digest",
|
|
2759
|
+
]
|