mostlyright-data 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mostlyright/data_harness/__init__.py +158 -0
- mostlyright/data_harness/acquisition/__init__.py +55 -0
- mostlyright/data_harness/acquisition/http.py +2773 -0
- mostlyright/data_harness/acquisition/parsing.py +809 -0
- mostlyright/data_harness/acquisition/ranges.py +495 -0
- mostlyright/data_harness/acquisition/result_download.py +360 -0
- mostlyright/data_harness/acquisition/retention_admission.py +248 -0
- mostlyright/data_harness/acquisition/sandbox.py +4888 -0
- mostlyright/data_harness/acquisition/url_policy.py +530 -0
- mostlyright/data_harness/agent_runtime.py +2743 -0
- mostlyright/data_harness/assets/logo-ink.svg +31 -0
- mostlyright/data_harness/backends/__init__.py +28 -0
- mostlyright/data_harness/backends/pandas_backend.py +350 -0
- mostlyright/data_harness/backends/polars_backend.py +366 -0
- mostlyright/data_harness/backends/protocol.py +124 -0
- mostlyright/data_harness/backends/reference.py +83 -0
- mostlyright/data_harness/backends/registry.py +55 -0
- mostlyright/data_harness/backends/restrictions.py +126 -0
- mostlyright/data_harness/canonical.py +333 -0
- mostlyright/data_harness/catalog_job.py +625 -0
- mostlyright/data_harness/cli.py +5398 -0
- mostlyright/data_harness/contracts.py +53 -0
- mostlyright/data_harness/coordinator.py +1307 -0
- mostlyright/data_harness/deploy.py +924 -0
- mostlyright/data_harness/deploy_target.py +312 -0
- mostlyright/data_harness/deployment_evidence.py +1067 -0
- mostlyright/data_harness/event_presentation.py +576 -0
- mostlyright/data_harness/events.py +2152 -0
- mostlyright/data_harness/fast_delimited.py +239 -0
- mostlyright/data_harness/fleet.py +237 -0
- mostlyright/data_harness/formats.py +236 -0
- mostlyright/data_harness/governors.py +1163 -0
- mostlyright/data_harness/hosted_bootstrap.py +972 -0
- mostlyright/data_harness/hosted_crawler.py +1115 -0
- mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
- mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
- mostlyright/data_harness/hosted_crawler_job.py +1277 -0
- mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
- mostlyright/data_harness/hosted_dataset.py +1500 -0
- mostlyright/data_harness/hosted_deploy.py +3037 -0
- mostlyright/data_harness/hosted_handoff.py +62 -0
- mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
- mostlyright/data_harness/hosted_ingestion_job.py +356 -0
- mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
- mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
- mostlyright/data_harness/hosted_session_worker.py +3554 -0
- mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
- mostlyright/data_harness/hosted_worker.py +6784 -0
- mostlyright/data_harness/ingestion/__init__.py +56 -0
- mostlyright/data_harness/ingestion/contracts.py +461 -0
- mostlyright/data_harness/ingestion/faults.py +42 -0
- mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
- mostlyright/data_harness/ingestion/spool.py +130 -0
- mostlyright/data_harness/ingestion/store.py +885 -0
- mostlyright/data_harness/key_seam.py +434 -0
- mostlyright/data_harness/linux_process_boundary.py +262 -0
- mostlyright/data_harness/local_contracts.py +2880 -0
- mostlyright/data_harness/local_search/__init__.py +5 -0
- mostlyright/data_harness/local_search/build_index.py +1087 -0
- mostlyright/data_harness/local_search/contracts.py +920 -0
- mostlyright/data_harness/local_search/query_trace.py +266 -0
- mostlyright/data_harness/local_search/retrieval.py +700 -0
- mostlyright/data_harness/local_search/sealed.py +474 -0
- mostlyright/data_harness/local_search/service.py +784 -0
- mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
- mostlyright/data_harness/nbrender/__init__.py +12 -0
- mostlyright/data_harness/nbrender/chrome.py +359 -0
- mostlyright/data_harness/nbrender/code_body.py +266 -0
- mostlyright/data_harness/nbrender/document.py +407 -0
- mostlyright/data_harness/nbrender/frame.py +275 -0
- mostlyright/data_harness/nbrender/interactive.py +337 -0
- mostlyright/data_harness/nbrender/markdown_body.py +477 -0
- mostlyright/data_harness/nbrender/mr_components.py +134 -0
- mostlyright/data_harness/nbrender/outputs_data.py +595 -0
- mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
- mostlyright/data_harness/nbrender/outputs_source.py +260 -0
- mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
- mostlyright/data_harness/nbrender/outputs_text.py +400 -0
- mostlyright/data_harness/nbrender/parse.py +394 -0
- mostlyright/data_harness/nbrender/status.py +40 -0
- mostlyright/data_harness/nbrender/tokens.py +1295 -0
- mostlyright/data_harness/notebook.py +1710 -0
- mostlyright/data_harness/offline.py +2049 -0
- mostlyright/data_harness/operation_registry.py +1007 -0
- mostlyright/data_harness/operator_setup.py +239 -0
- mostlyright/data_harness/pipeline.py +6428 -0
- mostlyright/data_harness/plan_graph.py +2026 -0
- mostlyright/data_harness/preparation/__init__.py +104 -0
- mostlyright/data_harness/preparation/contracts.py +1017 -0
- mostlyright/data_harness/preparation/engine.py +221 -0
- mostlyright/data_harness/preparation/errors.py +14 -0
- mostlyright/data_harness/preparation/gates.py +751 -0
- mostlyright/data_harness/preparation/joins.py +574 -0
- mostlyright/data_harness/preparation/profile.py +384 -0
- mostlyright/data_harness/preparation/table.py +217 -0
- mostlyright/data_harness/preparation/transforms.py +568 -0
- mostlyright/data_harness/progress_events.py +534 -0
- mostlyright/data_harness/readers/__init__.py +46 -0
- mostlyright/data_harness/readers/containers.py +963 -0
- mostlyright/data_harness/readers/contracts.py +542 -0
- mostlyright/data_harness/readers/delimited.py +257 -0
- mostlyright/data_harness/readers/grib2/__init__.py +33 -0
- mostlyright/data_harness/readers/grib2/admission.py +722 -0
- mostlyright/data_harness/readers/grib2/decode.py +1009 -0
- mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
- mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
- mostlyright/data_harness/readers/json_tabular.py +485 -0
- mostlyright/data_harness/readers/registry.py +514 -0
- mostlyright/data_harness/readers/samples/README.md +110 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples.py +582 -0
- mostlyright/data_harness/readers/spreadsheet.py +803 -0
- mostlyright/data_harness/readers/tabular.py +510 -0
- mostlyright/data_harness/recipe.py +5321 -0
- mostlyright/data_harness/repair/__init__.py +78 -0
- mostlyright/data_harness/repair/adapters.py +274 -0
- mostlyright/data_harness/repair/contracts.py +872 -0
- mostlyright/data_harness/repair/coordinator.py +1099 -0
- mostlyright/data_harness/repair/errors.py +16 -0
- mostlyright/data_harness/review.py +2533 -0
- mostlyright/data_harness/rowset.py +283 -0
- mostlyright/data_harness/serving.py +1975 -0
- mostlyright/data_harness/serving_edge.py +590 -0
- mostlyright/data_harness/serving_http.py +1031 -0
- mostlyright/data_harness/session_probes.py +759 -0
- mostlyright/data_harness/signing.py +101 -0
- mostlyright/data_harness/source_discovery.py +898 -0
- mostlyright/data_harness/sources/__init__.py +209 -0
- mostlyright/data_harness/sources/_adapter_steps.py +213 -0
- mostlyright/data_harness/sources/adapters.py +1214 -0
- mostlyright/data_harness/sources/cadence.py +1428 -0
- mostlyright/data_harness/sources/cadence_emission.py +453 -0
- mostlyright/data_harness/sources/cadence_history.py +546 -0
- mostlyright/data_harness/sources/catalog/__init__.py +17 -0
- mostlyright/data_harness/sources/catalog/admission.py +477 -0
- mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
- mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
- mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
- mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
- mostlyright/data_harness/sources/catalog/channel.py +523 -0
- mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
- mostlyright/data_harness/sources/catalog/contracts.py +825 -0
- mostlyright/data_harness/sources/catalog/coverage.py +137 -0
- mostlyright/data_harness/sources/catalog/delta.py +1340 -0
- mostlyright/data_harness/sources/catalog/embedding.py +532 -0
- mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
- mostlyright/data_harness/sources/catalog/fill.py +3889 -0
- mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
- mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
- mostlyright/data_harness/sources/catalog/gating.py +374 -0
- mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
- mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
- mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
- mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
- mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
- mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
- mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
- mostlyright/data_harness/sources/catalog/health.py +447 -0
- mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
- mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
- mostlyright/data_harness/sources/catalog/neural.py +1618 -0
- mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
- mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
- mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
- mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
- mostlyright/data_harness/sources/catalog/recommend.py +171 -0
- mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
- mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
- mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
- mostlyright/data_harness/sources/catalog/sealed.py +560 -0
- mostlyright/data_harness/sources/catalog/search.py +230 -0
- mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
- mostlyright/data_harness/sources/catalog/update.py +891 -0
- mostlyright/data_harness/sources/collections.py +815 -0
- mostlyright/data_harness/sources/contracts.py +2223 -0
- mostlyright/data_harness/sources/deletion.py +761 -0
- mostlyright/data_harness/sources/fitness.py +162 -0
- mostlyright/data_harness/sources/governance.py +163 -0
- mostlyright/data_harness/sources/hosted.py +173 -0
- mostlyright/data_harness/sources/integration.py +218 -0
- mostlyright/data_harness/sources/range_reader.py +418 -0
- mostlyright/data_harness/sources/registry.py +514 -0
- mostlyright/data_harness/sources/rights_rule.py +59 -0
- mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
- mostlyright/data_harness/sources/sports.py +521 -0
- mostlyright/data_harness/sources/stream.py +524 -0
- mostlyright/data_harness/sources/stream_connector.py +418 -0
- mostlyright/data_harness/sources/stream_recorder.py +1404 -0
- mostlyright/data_harness/studio_boundary.py +2019 -0
- mostlyright/data_harness/thin/__init__.py +37 -0
- mostlyright/data_harness/thin/acquire.py +1137 -0
- mostlyright/data_harness/thin/acquire_cancel.py +579 -0
- mostlyright/data_harness/thin/approvals.py +617 -0
- mostlyright/data_harness/thin/commands.py +406 -0
- mostlyright/data_harness/thin/download.py +194 -0
- mostlyright/data_harness/thin/narrative.py +589 -0
- mostlyright/data_harness/thin/parity.py +1070 -0
- mostlyright/data_harness/thin/propose.py +2759 -0
- mostlyright/data_harness/thin/research.py +1663 -0
- mostlyright/data_harness/thin/router.py +924 -0
- mostlyright/data_harness/thin/runs.py +519 -0
- mostlyright/data_harness/thin/session.py +281 -0
- mostlyright/data_harness/thin/stream.py +501 -0
- mostlyright/data_harness/thin/transport.py +187 -0
- mostlyright/data_harness/thin/vocabulary.py +368 -0
- mostlyright/data_harness/thin/workers.py +164 -0
- mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
- mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
- mostlyright/data_harness/unit_flow.py +927 -0
- mostlyright/data_harness/units.py +572 -0
- mostlyright/data_harness/ux/__init__.py +9 -0
- mostlyright/data_harness/ux/approve.py +485 -0
- mostlyright/data_harness/ux/author_yaml.py +597 -0
- mostlyright/data_harness/ux/cloud_auth.py +447 -0
- mostlyright/data_harness/ux/commands/__init__.py +260 -0
- mostlyright/data_harness/ux/commands/approve.py +136 -0
- mostlyright/data_harness/ux/commands/auth.py +744 -0
- mostlyright/data_harness/ux/commands/author.py +79 -0
- mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
- mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
- mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
- mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
- mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
- mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
- mostlyright/data_harness/ux/commands/deploy.py +134 -0
- mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
- mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
- mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
- mostlyright/data_harness/ux/commands/diff.py +74 -0
- mostlyright/data_harness/ux/commands/index.py +84 -0
- mostlyright/data_harness/ux/commands/inventory.py +47 -0
- mostlyright/data_harness/ux/commands/list_builds.py +143 -0
- mostlyright/data_harness/ux/commands/login.py +63 -0
- mostlyright/data_harness/ux/commands/peek.py +236 -0
- mostlyright/data_harness/ux/commands/plan_check.py +90 -0
- mostlyright/data_harness/ux/commands/preflight.py +97 -0
- mostlyright/data_harness/ux/commands/record.py +107 -0
- mostlyright/data_harness/ux/commands/review_setup.py +47 -0
- mostlyright/data_harness/ux/commands/search.py +440 -0
- mostlyright/data_harness/ux/commands/show.py +61 -0
- mostlyright/data_harness/ux/commands/whoami.py +37 -0
- mostlyright/data_harness/ux/credential_native.py +551 -0
- mostlyright/data_harness/ux/credential_store.py +1055 -0
- mostlyright/data_harness/ux/credentials.py +631 -0
- mostlyright/data_harness/ux/diffing.py +444 -0
- mostlyright/data_harness/ux/headline.py +671 -0
- mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
- mostlyright/data_harness/ux/hosted_run_status.py +619 -0
- mostlyright/data_harness/ux/inventory.py +427 -0
- mostlyright/data_harness/ux/local_review.py +375 -0
- mostlyright/data_harness/ux/login.py +691 -0
- mostlyright/data_harness/ux/path_kind.py +147 -0
- mostlyright/data_harness/ux/peek.py +1000 -0
- mostlyright/data_harness/ux/plain_file.py +178 -0
- mostlyright/data_harness/ux/plan_check.py +311 -0
- mostlyright/data_harness/ux/preflight.py +918 -0
- mostlyright/data_harness/ux/readers.py +1124 -0
- mostlyright/data_harness/ux/remediation.py +2195 -0
- mostlyright/data_harness/ux/render.py +657 -0
- mostlyright/data_harness/ux/workload.py +1077 -0
- mostlyright/data_harness/viewer.py +3713 -0
- mostlyright/data_harness/visual_run/__init__.py +83 -0
- mostlyright/data_harness/visual_run/authoring.py +235 -0
- mostlyright/data_harness/visual_run/contracts.py +673 -0
- mostlyright/data_harness/visual_run/materialize.py +486 -0
- mostlyright/data_harness/visual_run/observations.py +874 -0
- mostlyright/data_harness/visual_run/query.py +259 -0
- mostlyright/data_harness/visual_run/reducer.py +280 -0
- mostlyright/data_harness/visual_run/sdk.py +892 -0
- mostlyright/data_harness/visual_run/store.py +584 -0
- mostlyright/data_harness/visual_run/transport.py +239 -0
- mostlyright/data_harness/watch.py +2999 -0
- mostlyright_data-0.9.0.dist-info/METADATA +607 -0
- mostlyright_data-0.9.0.dist-info/RECORD +314 -0
- mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
- mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
|
@@ -0,0 +1,2880 @@
|
|
|
1
|
+
"""Versioned, immutable Harness-private execution contracts.
|
|
2
|
+
|
|
3
|
+
These types describe local agent proposals and deterministic execution inputs. They deliberately
|
|
4
|
+
exclude Studio tenancy, authorization, persistence, and release fields; cross-repository values
|
|
5
|
+
are mapped to Studio's generated client only at the hosted boundary.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
from collections.abc import Iterable, Mapping, Sequence
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
from datetime import datetime
|
|
14
|
+
from pathlib import PurePosixPath
|
|
15
|
+
from typing import Any, ClassVar
|
|
16
|
+
from urllib.parse import parse_qsl, unquote_plus, urlsplit, urlunsplit
|
|
17
|
+
|
|
18
|
+
from mostlyright.data_harness.canonical import canonical_json_bytes, canonical_sha256
|
|
19
|
+
from mostlyright.data_harness.formats import PLAN_DATA_FORMATS
|
|
20
|
+
from mostlyright.data_harness.operation_registry import (
|
|
21
|
+
GRAPH_OPERATION_CONTRACT_VERSION,
|
|
22
|
+
RETIRED_OPERATIONS,
|
|
23
|
+
OperationRegistryError,
|
|
24
|
+
resolve_operation,
|
|
25
|
+
thaw_parameter,
|
|
26
|
+
)
|
|
27
|
+
from mostlyright.data_harness.units import (
|
|
28
|
+
NO_UNIT,
|
|
29
|
+
Unit,
|
|
30
|
+
UnitError,
|
|
31
|
+
is_count_unit,
|
|
32
|
+
legacy_aliases,
|
|
33
|
+
resolve_declared_unit,
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
QUESTION_SCHEMA_VERSION = "local-question.v1"
|
|
37
|
+
OUTPUT_GRAIN_SCHEMA_VERSION = "local-output-grain.v1"
|
|
38
|
+
REQUIREMENTS_SCHEMA_VERSION = "local-requirements.v1"
|
|
39
|
+
SOURCE_PROPOSAL_SCHEMA_VERSION = "local-source-proposal.v2"
|
|
40
|
+
PLAN_SCHEMA_VERSION = "local-table-plan.v1"
|
|
41
|
+
_LEGACY_PLAN_SCHEMA_VERSION = "local-plan.v1"
|
|
42
|
+
OPERATION_CONTRACT_VERSION = "local-operations.v1"
|
|
43
|
+
GRAPH_PLAN_SCHEMA_VERSION = "local-graph-table-plan.v1"
|
|
44
|
+
VALIDATION_POLICY_VERSION = "local-validation.v1"
|
|
45
|
+
# The first semantics generation carries the fifteen hand-grown tokens and nothing else; the
|
|
46
|
+
# second carries a unit code, which is any expression the grammar in :mod:`units` resolves against
|
|
47
|
+
# its pinned table. The two are read side by side forever: a document sealed under the first
|
|
48
|
+
# generation says what vocabulary its author was writing against, and re-reading it must not widen
|
|
49
|
+
# that. New documents are authored at the second.
|
|
50
|
+
TABLE_SEMANTICS_V1 = "table-semantics.v1"
|
|
51
|
+
TABLE_SEMANTICS_V2 = "table-semantics.v2"
|
|
52
|
+
TABLE_SEMANTICS_VERSIONS = (TABLE_SEMANTICS_V1, TABLE_SEMANTICS_V2)
|
|
53
|
+
TABLE_SEMANTICS_VERSION = TABLE_SEMANTICS_V2
|
|
54
|
+
|
|
55
|
+
MAX_IDENTIFIER_LENGTH = 64
|
|
56
|
+
MAX_COLUMN_COUNT = 256
|
|
57
|
+
MAX_SOURCE_COUNT = 64
|
|
58
|
+
MAX_OPERATION_COUNT = 256
|
|
59
|
+
MAX_GRAPH_NODE_COUNT = 512
|
|
60
|
+
MAX_TEXT_LENGTH = 16_384
|
|
61
|
+
MAX_SHORT_TEXT_LENGTH = 2_000
|
|
62
|
+
MAX_PATH_LENGTH = 1_024
|
|
63
|
+
MAX_EVIDENCE_COUNT = 64
|
|
64
|
+
MAX_OUTPUT_ROWS = 100_000
|
|
65
|
+
GRAPH_MAX_OUTPUT_ROWS = 1_000_000
|
|
66
|
+
|
|
67
|
+
_IDENTIFIER = re.compile(r"^[a-z][a-z0-9]*(?:[-_][a-z0-9]+)*$")
|
|
68
|
+
_COLUMN = re.compile(r"^[a-z][a-z0-9_]{0,62}$")
|
|
69
|
+
_SHA256 = re.compile(r"^[0-9a-f]{64}$")
|
|
70
|
+
_UTC_TIMESTAMP = re.compile(
|
|
71
|
+
r"^(?P<date>[0-9]{4}-[0-9]{2}-[0-9]{2})"
|
|
72
|
+
r"T(?P<time>[0-9]{2}:[0-9]{2}:[0-9]{2})"
|
|
73
|
+
r"(?P<fraction>\.[0-9]{1,6})?Z$"
|
|
74
|
+
)
|
|
75
|
+
_SENSITIVE_LOCATOR_QUERY_KEY = re.compile(
|
|
76
|
+
r"(?:^|[_-])(?:api[_-]?key|authorization|bearer|credential|password|private[_-]?key|"
|
|
77
|
+
r"secret|sig|signature|signed|token)(?:$|[_-])",
|
|
78
|
+
re.IGNORECASE,
|
|
79
|
+
)
|
|
80
|
+
_SECRET_LOCATOR_VALUE = re.compile(
|
|
81
|
+
r"^(?:Bearer\s+|Basic\s+|sk[-_]|gh[opusr]_|AIza|AKIA|ASIA|eyJ[A-Za-z0-9_-]*\.)",
|
|
82
|
+
re.IGNORECASE,
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
# The complete set of boundaries Python's text renderer treats as a new line. Tabs and ordinary
|
|
86
|
+
# Unicode remain valid within a line; U+009B and bidi controls are deliberately not decided here.
|
|
87
|
+
# This is structural line safety, not a terminal-control policy.
|
|
88
|
+
_DISPLAY_LINE_BOUNDARIES = frozenset(
|
|
89
|
+
{"\n", "\r", "\v", "\f", "\x1c", "\x1d", "\x1e", "\x85", "\u2028", "\u2029"}
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
_QUESTION_FIELDS = frozenset({"schema_version", "question_id", "text", "created_at"})
|
|
93
|
+
_TARGET_POLICIES = frozenset({"none", "required_if_supportable"})
|
|
94
|
+
_FEASIBILITY_DECISIONS = frozenset({"supportable", "supportable_with_limits", "unsupported"})
|
|
95
|
+
_FEASIBILITY_REASONS = frozenset(
|
|
96
|
+
{
|
|
97
|
+
"sufficient_evidence",
|
|
98
|
+
"limited_coverage",
|
|
99
|
+
"unacceptable_delay",
|
|
100
|
+
"rights_unclear",
|
|
101
|
+
"no_reliable_source",
|
|
102
|
+
"target_not_observable",
|
|
103
|
+
}
|
|
104
|
+
)
|
|
105
|
+
# Narrower than the model's coarse family; see the vocabulary map in sources/contracts.py.
|
|
106
|
+
_SOURCE_CLASSES = frozenset(
|
|
107
|
+
{
|
|
108
|
+
"external_adapter",
|
|
109
|
+
"user_file",
|
|
110
|
+
"user_url",
|
|
111
|
+
"user_api",
|
|
112
|
+
"database_extract",
|
|
113
|
+
"webhook",
|
|
114
|
+
"stream",
|
|
115
|
+
}
|
|
116
|
+
)
|
|
117
|
+
_LOCATOR_KINDS = frozenset({"relative_path", "https_url", "artifact_reference"})
|
|
118
|
+
# Mixes encodings with access shapes; see the vocabulary map in sources/contracts.py.
|
|
119
|
+
# This is the plan-contract set, not the wire-format set: dropping to DATA_FORMATS here
|
|
120
|
+
# would silently remove api and stream from plan validation.
|
|
121
|
+
_DATA_FORMATS = PLAN_DATA_FORMATS
|
|
122
|
+
# No unknown state here, unlike model liveness; see the vocabulary map in sources/contracts.py.
|
|
123
|
+
_LIVE_ENDPOINTS = frozenset({"live", "degraded", "delayed", "dead", "not_applicable"})
|
|
124
|
+
# Same spelling in all three layers; see the vocabulary map in sources/contracts.py.
|
|
125
|
+
_PROPOSAL_RIGHTS = frozenset({"approved", "conditional", "unclear", "prohibited"})
|
|
126
|
+
# Stage-specific triplet; see the vocabulary map in sources/contracts.py.
|
|
127
|
+
_FITNESS_DECISIONS = frozenset({"selected", "eligible", "rejected"})
|
|
128
|
+
|
|
129
|
+
# Rights basis a build reads a file under, not a rights adjudication: this unknown is not
|
|
130
|
+
# _PROPOSAL_RIGHTS.unclear. See the vocabulary map in sources/contracts.py.
|
|
131
|
+
_RIGHTS_STATUSES = frozenset(
|
|
132
|
+
{
|
|
133
|
+
"project_owned",
|
|
134
|
+
"public_domain",
|
|
135
|
+
"permissive_license",
|
|
136
|
+
"authorized_internal",
|
|
137
|
+
"user_authorized",
|
|
138
|
+
"unknown",
|
|
139
|
+
"prohibited",
|
|
140
|
+
}
|
|
141
|
+
)
|
|
142
|
+
_PERMISSIONS = frozenset({"local_use", "redistribute", "host"})
|
|
143
|
+
_ACQUISITION_METHODS = frozenset({"checked_in_fixture", "user_upload", "authorized_export"})
|
|
144
|
+
_OUTPUT_INTENTS = frozenset({"local_use", "redistribute", "host"})
|
|
145
|
+
# The one authoritative logical-type vocabulary shared by Recipe declarations and the
|
|
146
|
+
# deterministic local transform path. ``_CAST_TYPES`` remains as a private compatibility alias
|
|
147
|
+
# for callers that inspect the plan contract, but every accepted logical type is executable.
|
|
148
|
+
LOGICAL_TYPES = frozenset({"string", "int64", "float64", "boolean", "date", "timestamp_utc"})
|
|
149
|
+
_CAST_TYPES = LOGICAL_TYPES
|
|
150
|
+
_CLEANING_OPERATIONS = frozenset({"trim", "empty_to_null", "rename", "cast"})
|
|
151
|
+
# The whole of what a ``local-plan.v1`` join may declare, frozen at one kind by the ruling in
|
|
152
|
+
# ``docs/TRANSFORMS.md``. It is written out as a name rather than left inline because the set is a
|
|
153
|
+
# decision about sealed meaning and not a list that grows: the v1 join rule is named in the
|
|
154
|
+
# ``mandatory_gates`` list ``validation_policy_digest`` hashes, and every approved v1 Recipe is
|
|
155
|
+
# bound to that digest. ``JoinSpec`` below says why each of the graph's other two kinds is refused
|
|
156
|
+
# here; join-kind growth happens in ``operation_registry.GRAPH_JOIN_KINDS`` instead.
|
|
157
|
+
JOIN_KINDS = frozenset({"left"})
|
|
158
|
+
_JOIN_CARDINALITIES = frozenset({"one_to_one", "many_to_one"})
|
|
159
|
+
SEMANTIC_TYPES = frozenset(
|
|
160
|
+
{"identifier", "entity", "category", "measure", "dimension", "target", "other"}
|
|
161
|
+
)
|
|
162
|
+
# The fifteen tokens the hand-grown list held. They are no longer the vocabulary: a declared unit
|
|
163
|
+
# is a code the grammar resolves, and these fourteen names plus ``none`` are read as the codes they
|
|
164
|
+
# always meant. What they still are is the whole of what ``table-semantics.v1`` may declare, and
|
|
165
|
+
# a frozen generation cannot be a function of a data file that a later change might extend -- so
|
|
166
|
+
# the fifteen are written out here, and the equality below binds them to the pinned alias table.
|
|
167
|
+
# Adding an alias to that table without deciding what it means for the frozen generation is an
|
|
168
|
+
# explicit import-time failure rather than a silent widening of sealed meaning.
|
|
169
|
+
UNITS = frozenset(
|
|
170
|
+
{
|
|
171
|
+
"none",
|
|
172
|
+
"count",
|
|
173
|
+
"ratio",
|
|
174
|
+
"percent",
|
|
175
|
+
"celsius",
|
|
176
|
+
"fahrenheit",
|
|
177
|
+
"kelvin",
|
|
178
|
+
"meter",
|
|
179
|
+
"kilometer",
|
|
180
|
+
"mile",
|
|
181
|
+
"microgram_per_cubic_meter",
|
|
182
|
+
"second",
|
|
183
|
+
"minute",
|
|
184
|
+
"hour",
|
|
185
|
+
"hectopascal",
|
|
186
|
+
}
|
|
187
|
+
)
|
|
188
|
+
assert UNITS == frozenset({NO_UNIT}) | frozenset(legacy_aliases())
|
|
189
|
+
UNIT_STATES = frozenset({"declared", "normalized", "not_applicable", "unknown"})
|
|
190
|
+
NORMALIZATIONS = frozenset(
|
|
191
|
+
{"z_score", "min_max", "unit_interval", "percent_of_total", "index_100", "log", "other"}
|
|
192
|
+
)
|
|
193
|
+
CATEGORY_STATUSES = frozenset({"complete", "partial", "not_applicable", "unknown"})
|
|
194
|
+
SEMANTIC_EVIDENCE_KINDS = frozenset({"source_id"})
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
class ContractError(ValueError):
|
|
198
|
+
"""A typed private-contract validation failure with an exact field path."""
|
|
199
|
+
|
|
200
|
+
def __init__(self, path: str, code: str, detail: str) -> None:
|
|
201
|
+
self.path = path
|
|
202
|
+
self.code = code
|
|
203
|
+
self.detail = detail
|
|
204
|
+
super().__init__(f"{path}: {detail} [{code}]")
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
class _CanonicalContract:
|
|
208
|
+
"""Common deterministic serialization behavior for private contracts."""
|
|
209
|
+
|
|
210
|
+
schema_version: str
|
|
211
|
+
EXPECTED_SCHEMA_VERSION: ClassVar[str]
|
|
212
|
+
|
|
213
|
+
def to_dict(self) -> dict[str, Any]: # pragma: no cover - abstract-by-convention
|
|
214
|
+
raise NotImplementedError
|
|
215
|
+
|
|
216
|
+
@property
|
|
217
|
+
def canonical_bytes(self) -> bytes:
|
|
218
|
+
return canonical_json_bytes(self.to_dict())
|
|
219
|
+
|
|
220
|
+
@property
|
|
221
|
+
def digest(self) -> str:
|
|
222
|
+
return canonical_sha256(self.to_dict())
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
@dataclass(frozen=True)
|
|
226
|
+
class LocalQuestion(_CanonicalContract):
|
|
227
|
+
"""A local question proposal without hosted identity or authorization fields."""
|
|
228
|
+
|
|
229
|
+
question_id: str
|
|
230
|
+
text: str
|
|
231
|
+
created_at: str
|
|
232
|
+
schema_version: str = QUESTION_SCHEMA_VERSION
|
|
233
|
+
|
|
234
|
+
EXPECTED_SCHEMA_VERSION: ClassVar[str] = QUESTION_SCHEMA_VERSION
|
|
235
|
+
|
|
236
|
+
def __post_init__(self) -> None:
|
|
237
|
+
_version(self.schema_version, QUESTION_SCHEMA_VERSION, "question.schema_version")
|
|
238
|
+
_identifier(self.question_id, "question.question_id")
|
|
239
|
+
_text(self.text, "question.text", minimum=1, maximum=12_000)
|
|
240
|
+
_utc_timestamp(self.created_at, "question.created_at")
|
|
241
|
+
|
|
242
|
+
def to_dict(self) -> dict[str, Any]:
|
|
243
|
+
return {
|
|
244
|
+
"schema_version": self.schema_version,
|
|
245
|
+
"question_id": self.question_id,
|
|
246
|
+
"text": self.text,
|
|
247
|
+
"created_at": self.created_at,
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
@classmethod
|
|
251
|
+
def from_value(cls, value: Any) -> LocalQuestion:
|
|
252
|
+
data = _object(value, "question")
|
|
253
|
+
_exact_fields(data, _QUESTION_FIELDS, "question")
|
|
254
|
+
return cls(
|
|
255
|
+
schema_version=_required_text(
|
|
256
|
+
data["schema_version"],
|
|
257
|
+
"question.schema_version",
|
|
258
|
+
maximum=64,
|
|
259
|
+
),
|
|
260
|
+
question_id=_required_text(
|
|
261
|
+
data["question_id"],
|
|
262
|
+
"question.question_id",
|
|
263
|
+
maximum=MAX_IDENTIFIER_LENGTH,
|
|
264
|
+
),
|
|
265
|
+
text=_required_text(data["text"], "question.text", maximum=12_000),
|
|
266
|
+
created_at=_required_text(
|
|
267
|
+
data["created_at"],
|
|
268
|
+
"question.created_at",
|
|
269
|
+
maximum=32,
|
|
270
|
+
),
|
|
271
|
+
)
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
@dataclass(frozen=True)
|
|
275
|
+
class OutputGrain(_CanonicalContract):
|
|
276
|
+
"""Ordered local output key definition."""
|
|
277
|
+
|
|
278
|
+
columns: tuple[str, ...]
|
|
279
|
+
description: str
|
|
280
|
+
schema_version: str = OUTPUT_GRAIN_SCHEMA_VERSION
|
|
281
|
+
|
|
282
|
+
EXPECTED_SCHEMA_VERSION: ClassVar[str] = OUTPUT_GRAIN_SCHEMA_VERSION
|
|
283
|
+
|
|
284
|
+
def __post_init__(self) -> None:
|
|
285
|
+
_version(
|
|
286
|
+
self.schema_version,
|
|
287
|
+
OUTPUT_GRAIN_SCHEMA_VERSION,
|
|
288
|
+
"output_grain.schema_version",
|
|
289
|
+
)
|
|
290
|
+
_column_tuple(
|
|
291
|
+
self.columns,
|
|
292
|
+
"output_grain.columns",
|
|
293
|
+
nonempty=True,
|
|
294
|
+
maximum=16,
|
|
295
|
+
)
|
|
296
|
+
_text(
|
|
297
|
+
self.description,
|
|
298
|
+
"output_grain.description",
|
|
299
|
+
minimum=1,
|
|
300
|
+
maximum=1_000,
|
|
301
|
+
)
|
|
302
|
+
|
|
303
|
+
def to_dict(self) -> dict[str, Any]:
|
|
304
|
+
return {
|
|
305
|
+
"schema_version": self.schema_version,
|
|
306
|
+
"columns": list(self.columns),
|
|
307
|
+
"description": self.description,
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
@classmethod
|
|
311
|
+
def from_value(cls, value: Any, path: str = "output_grain") -> OutputGrain:
|
|
312
|
+
data = _object(value, path)
|
|
313
|
+
_exact_fields(data, {"schema_version", "columns", "description"}, path)
|
|
314
|
+
return cls(
|
|
315
|
+
schema_version=_required_text(
|
|
316
|
+
data["schema_version"],
|
|
317
|
+
f"{path}.schema_version",
|
|
318
|
+
maximum=64,
|
|
319
|
+
),
|
|
320
|
+
columns=_parse_column_array(
|
|
321
|
+
data["columns"],
|
|
322
|
+
f"{path}.columns",
|
|
323
|
+
nonempty=True,
|
|
324
|
+
maximum=16,
|
|
325
|
+
),
|
|
326
|
+
description=_required_text(
|
|
327
|
+
data["description"],
|
|
328
|
+
f"{path}.description",
|
|
329
|
+
maximum=1_000,
|
|
330
|
+
),
|
|
331
|
+
)
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
@dataclass(frozen=True)
|
|
335
|
+
class TimeRange:
|
|
336
|
+
"""A half-open UTC time range used by private requirements and proposals."""
|
|
337
|
+
|
|
338
|
+
start_inclusive: str
|
|
339
|
+
end_exclusive: str
|
|
340
|
+
|
|
341
|
+
def __post_init__(self) -> None:
|
|
342
|
+
start = _utc_timestamp(self.start_inclusive, "time_range.start_inclusive")
|
|
343
|
+
end = _utc_timestamp(self.end_exclusive, "time_range.end_exclusive")
|
|
344
|
+
if start >= end:
|
|
345
|
+
raise ContractError(
|
|
346
|
+
"time_range.end_exclusive",
|
|
347
|
+
"RANGE_ORDER",
|
|
348
|
+
"must be later than time_range.start_inclusive",
|
|
349
|
+
)
|
|
350
|
+
|
|
351
|
+
def to_dict(self) -> dict[str, str]:
|
|
352
|
+
return {
|
|
353
|
+
"start_inclusive": self.start_inclusive,
|
|
354
|
+
"end_exclusive": self.end_exclusive,
|
|
355
|
+
}
|
|
356
|
+
|
|
357
|
+
@classmethod
|
|
358
|
+
def from_value(cls, value: Any, path: str) -> TimeRange:
|
|
359
|
+
data = _object(value, path)
|
|
360
|
+
_exact_fields(data, {"start_inclusive", "end_exclusive"}, path)
|
|
361
|
+
start = _required_text(
|
|
362
|
+
data["start_inclusive"],
|
|
363
|
+
f"{path}.start_inclusive",
|
|
364
|
+
maximum=32,
|
|
365
|
+
)
|
|
366
|
+
end = _required_text(
|
|
367
|
+
data["end_exclusive"],
|
|
368
|
+
f"{path}.end_exclusive",
|
|
369
|
+
maximum=32,
|
|
370
|
+
)
|
|
371
|
+
try:
|
|
372
|
+
return cls(start_inclusive=start, end_exclusive=end)
|
|
373
|
+
except ContractError as exc:
|
|
374
|
+
if exc.path.startswith("time_range."):
|
|
375
|
+
suffix = exc.path.removeprefix("time_range.")
|
|
376
|
+
raise ContractError(f"{path}.{suffix}", exc.code, exc.detail) from None
|
|
377
|
+
raise
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
@dataclass(frozen=True)
|
|
381
|
+
class Feasibility:
|
|
382
|
+
"""Deterministically bounded feasibility proposal."""
|
|
383
|
+
|
|
384
|
+
decision: str
|
|
385
|
+
reason_codes: tuple[str, ...]
|
|
386
|
+
narrative: str
|
|
387
|
+
|
|
388
|
+
def __post_init__(self) -> None:
|
|
389
|
+
_choice(self.decision, _FEASIBILITY_DECISIONS, "feasibility.decision")
|
|
390
|
+
_choice_tuple(
|
|
391
|
+
self.reason_codes,
|
|
392
|
+
_FEASIBILITY_REASONS,
|
|
393
|
+
"feasibility.reason_codes",
|
|
394
|
+
maximum=16,
|
|
395
|
+
)
|
|
396
|
+
_text(
|
|
397
|
+
self.narrative,
|
|
398
|
+
"feasibility.narrative",
|
|
399
|
+
minimum=1,
|
|
400
|
+
maximum=4_000,
|
|
401
|
+
)
|
|
402
|
+
if self.decision == "supportable" and "sufficient_evidence" not in self.reason_codes:
|
|
403
|
+
raise ContractError(
|
|
404
|
+
"feasibility.reason_codes",
|
|
405
|
+
"MISSING_REASON",
|
|
406
|
+
"supportable feasibility requires sufficient_evidence",
|
|
407
|
+
)
|
|
408
|
+
if self.decision == "unsupported" and not self.reason_codes:
|
|
409
|
+
raise ContractError(
|
|
410
|
+
"feasibility.reason_codes",
|
|
411
|
+
"EMPTY_COLLECTION",
|
|
412
|
+
"unsupported feasibility requires at least one reason code",
|
|
413
|
+
)
|
|
414
|
+
|
|
415
|
+
def to_dict(self) -> dict[str, Any]:
|
|
416
|
+
return {
|
|
417
|
+
"decision": self.decision,
|
|
418
|
+
"reason_codes": list(self.reason_codes),
|
|
419
|
+
"narrative": self.narrative,
|
|
420
|
+
}
|
|
421
|
+
|
|
422
|
+
@classmethod
|
|
423
|
+
def from_value(cls, value: Any, path: str) -> Feasibility:
|
|
424
|
+
data = _object(value, path)
|
|
425
|
+
_exact_fields(data, {"decision", "reason_codes", "narrative"}, path)
|
|
426
|
+
try:
|
|
427
|
+
return cls(
|
|
428
|
+
decision=_required_text(
|
|
429
|
+
data["decision"],
|
|
430
|
+
f"{path}.decision",
|
|
431
|
+
maximum=64,
|
|
432
|
+
),
|
|
433
|
+
reason_codes=_parse_choice_array(
|
|
434
|
+
data["reason_codes"],
|
|
435
|
+
f"{path}.reason_codes",
|
|
436
|
+
_FEASIBILITY_REASONS,
|
|
437
|
+
nonempty=False,
|
|
438
|
+
maximum=16,
|
|
439
|
+
),
|
|
440
|
+
narrative=_required_text(
|
|
441
|
+
data["narrative"],
|
|
442
|
+
f"{path}.narrative",
|
|
443
|
+
maximum=4_000,
|
|
444
|
+
),
|
|
445
|
+
)
|
|
446
|
+
except ContractError as exc:
|
|
447
|
+
if exc.path.startswith("feasibility."):
|
|
448
|
+
suffix = exc.path.removeprefix("feasibility.")
|
|
449
|
+
raise ContractError(f"{path}.{suffix}", exc.code, exc.detail) from None
|
|
450
|
+
raise
|
|
451
|
+
|
|
452
|
+
|
|
453
|
+
@dataclass(frozen=True)
|
|
454
|
+
class LocalRequirements(_CanonicalContract):
|
|
455
|
+
"""Question-derived local dataset requirements."""
|
|
456
|
+
|
|
457
|
+
requirements_id: str
|
|
458
|
+
question_id: str
|
|
459
|
+
population: str
|
|
460
|
+
time_range: TimeRange
|
|
461
|
+
output_grain: OutputGrain
|
|
462
|
+
required_fields: tuple[str, ...]
|
|
463
|
+
target_policy: str
|
|
464
|
+
success_criteria: tuple[str, ...]
|
|
465
|
+
feasibility: Feasibility
|
|
466
|
+
created_at: str
|
|
467
|
+
schema_version: str = REQUIREMENTS_SCHEMA_VERSION
|
|
468
|
+
|
|
469
|
+
EXPECTED_SCHEMA_VERSION: ClassVar[str] = REQUIREMENTS_SCHEMA_VERSION
|
|
470
|
+
|
|
471
|
+
def __post_init__(self) -> None:
|
|
472
|
+
_version(
|
|
473
|
+
self.schema_version,
|
|
474
|
+
REQUIREMENTS_SCHEMA_VERSION,
|
|
475
|
+
"requirements.schema_version",
|
|
476
|
+
)
|
|
477
|
+
_identifier(self.requirements_id, "requirements.requirements_id")
|
|
478
|
+
_identifier(self.question_id, "requirements.question_id")
|
|
479
|
+
_text(
|
|
480
|
+
self.population,
|
|
481
|
+
"requirements.population",
|
|
482
|
+
minimum=1,
|
|
483
|
+
maximum=MAX_SHORT_TEXT_LENGTH,
|
|
484
|
+
)
|
|
485
|
+
if not isinstance(self.time_range, TimeRange):
|
|
486
|
+
raise ContractError(
|
|
487
|
+
"requirements.time_range",
|
|
488
|
+
"TYPE",
|
|
489
|
+
"must be a TimeRange",
|
|
490
|
+
)
|
|
491
|
+
if not isinstance(self.output_grain, OutputGrain):
|
|
492
|
+
raise ContractError(
|
|
493
|
+
"requirements.output_grain",
|
|
494
|
+
"TYPE",
|
|
495
|
+
"must be an OutputGrain",
|
|
496
|
+
)
|
|
497
|
+
_column_tuple(
|
|
498
|
+
self.required_fields,
|
|
499
|
+
"requirements.required_fields",
|
|
500
|
+
nonempty=True,
|
|
501
|
+
maximum=MAX_COLUMN_COUNT,
|
|
502
|
+
)
|
|
503
|
+
missing_grain = set(self.output_grain.columns) - set(self.required_fields)
|
|
504
|
+
if missing_grain:
|
|
505
|
+
raise ContractError(
|
|
506
|
+
"requirements.required_fields",
|
|
507
|
+
"GRAIN_FIELD_MISSING",
|
|
508
|
+
f"must contain output-grain columns {sorted(missing_grain)}",
|
|
509
|
+
)
|
|
510
|
+
_choice(self.target_policy, _TARGET_POLICIES, "requirements.target_policy")
|
|
511
|
+
_text_tuple(
|
|
512
|
+
self.success_criteria,
|
|
513
|
+
"requirements.success_criteria",
|
|
514
|
+
nonempty=True,
|
|
515
|
+
maximum=64,
|
|
516
|
+
item_maximum=1_000,
|
|
517
|
+
)
|
|
518
|
+
if not isinstance(self.feasibility, Feasibility):
|
|
519
|
+
raise ContractError(
|
|
520
|
+
"requirements.feasibility",
|
|
521
|
+
"TYPE",
|
|
522
|
+
"must be Feasibility",
|
|
523
|
+
)
|
|
524
|
+
_utc_timestamp(self.created_at, "requirements.created_at")
|
|
525
|
+
|
|
526
|
+
def to_dict(self) -> dict[str, Any]:
|
|
527
|
+
return {
|
|
528
|
+
"schema_version": self.schema_version,
|
|
529
|
+
"requirements_id": self.requirements_id,
|
|
530
|
+
"question_id": self.question_id,
|
|
531
|
+
"population": self.population,
|
|
532
|
+
"time_range": self.time_range.to_dict(),
|
|
533
|
+
"output_grain": self.output_grain.to_dict(),
|
|
534
|
+
"required_fields": list(self.required_fields),
|
|
535
|
+
"target_policy": self.target_policy,
|
|
536
|
+
"success_criteria": list(self.success_criteria),
|
|
537
|
+
"feasibility": self.feasibility.to_dict(),
|
|
538
|
+
"created_at": self.created_at,
|
|
539
|
+
}
|
|
540
|
+
|
|
541
|
+
@classmethod
|
|
542
|
+
def from_value(cls, value: Any) -> LocalRequirements:
|
|
543
|
+
path = "requirements"
|
|
544
|
+
data = _object(value, path)
|
|
545
|
+
_exact_fields(
|
|
546
|
+
data,
|
|
547
|
+
{
|
|
548
|
+
"schema_version",
|
|
549
|
+
"requirements_id",
|
|
550
|
+
"question_id",
|
|
551
|
+
"population",
|
|
552
|
+
"time_range",
|
|
553
|
+
"output_grain",
|
|
554
|
+
"required_fields",
|
|
555
|
+
"target_policy",
|
|
556
|
+
"success_criteria",
|
|
557
|
+
"feasibility",
|
|
558
|
+
"created_at",
|
|
559
|
+
},
|
|
560
|
+
path,
|
|
561
|
+
)
|
|
562
|
+
return cls(
|
|
563
|
+
schema_version=_required_text(
|
|
564
|
+
data["schema_version"],
|
|
565
|
+
f"{path}.schema_version",
|
|
566
|
+
maximum=64,
|
|
567
|
+
),
|
|
568
|
+
requirements_id=_required_text(
|
|
569
|
+
data["requirements_id"],
|
|
570
|
+
f"{path}.requirements_id",
|
|
571
|
+
maximum=MAX_IDENTIFIER_LENGTH,
|
|
572
|
+
),
|
|
573
|
+
question_id=_required_text(
|
|
574
|
+
data["question_id"],
|
|
575
|
+
f"{path}.question_id",
|
|
576
|
+
maximum=MAX_IDENTIFIER_LENGTH,
|
|
577
|
+
),
|
|
578
|
+
population=_required_text(
|
|
579
|
+
data["population"],
|
|
580
|
+
f"{path}.population",
|
|
581
|
+
maximum=MAX_SHORT_TEXT_LENGTH,
|
|
582
|
+
),
|
|
583
|
+
time_range=TimeRange.from_value(data["time_range"], f"{path}.time_range"),
|
|
584
|
+
output_grain=OutputGrain.from_value(
|
|
585
|
+
data["output_grain"],
|
|
586
|
+
f"{path}.output_grain",
|
|
587
|
+
),
|
|
588
|
+
required_fields=_parse_column_array(
|
|
589
|
+
data["required_fields"],
|
|
590
|
+
f"{path}.required_fields",
|
|
591
|
+
nonempty=True,
|
|
592
|
+
maximum=MAX_COLUMN_COUNT,
|
|
593
|
+
),
|
|
594
|
+
target_policy=_required_text(
|
|
595
|
+
data["target_policy"],
|
|
596
|
+
f"{path}.target_policy",
|
|
597
|
+
maximum=64,
|
|
598
|
+
),
|
|
599
|
+
success_criteria=_parse_text_array(
|
|
600
|
+
data["success_criteria"],
|
|
601
|
+
f"{path}.success_criteria",
|
|
602
|
+
nonempty=True,
|
|
603
|
+
maximum=64,
|
|
604
|
+
item_maximum=1_000,
|
|
605
|
+
),
|
|
606
|
+
feasibility=Feasibility.from_value(data["feasibility"], f"{path}.feasibility"),
|
|
607
|
+
created_at=_required_text(
|
|
608
|
+
data["created_at"],
|
|
609
|
+
f"{path}.created_at",
|
|
610
|
+
maximum=32,
|
|
611
|
+
),
|
|
612
|
+
)
|
|
613
|
+
|
|
614
|
+
|
|
615
|
+
@dataclass(frozen=True)
|
|
616
|
+
class EvidenceReference:
|
|
617
|
+
"""One immutable, non-secret source-proposal evidence reference."""
|
|
618
|
+
|
|
619
|
+
uri: str
|
|
620
|
+
observed_at: str
|
|
621
|
+
content_sha256: str
|
|
622
|
+
|
|
623
|
+
def __post_init__(self) -> None:
|
|
624
|
+
_evidence_uri(self.uri, "evidence.uri")
|
|
625
|
+
_utc_timestamp(self.observed_at, "evidence.observed_at")
|
|
626
|
+
_sha256(self.content_sha256, "evidence.content_sha256")
|
|
627
|
+
|
|
628
|
+
def to_dict(self) -> dict[str, str]:
|
|
629
|
+
return {
|
|
630
|
+
"uri": self.uri,
|
|
631
|
+
"observed_at": self.observed_at,
|
|
632
|
+
"content_sha256": self.content_sha256,
|
|
633
|
+
}
|
|
634
|
+
|
|
635
|
+
@classmethod
|
|
636
|
+
def from_value(cls, value: Any, path: str) -> EvidenceReference:
|
|
637
|
+
data = _object(value, path)
|
|
638
|
+
_exact_fields(data, {"uri", "observed_at", "content_sha256"}, path)
|
|
639
|
+
try:
|
|
640
|
+
return cls(
|
|
641
|
+
uri=_required_text(data["uri"], f"{path}.uri", maximum=2_048),
|
|
642
|
+
observed_at=_required_text(
|
|
643
|
+
data["observed_at"],
|
|
644
|
+
f"{path}.observed_at",
|
|
645
|
+
maximum=32,
|
|
646
|
+
),
|
|
647
|
+
content_sha256=_required_text(
|
|
648
|
+
data["content_sha256"],
|
|
649
|
+
f"{path}.content_sha256",
|
|
650
|
+
maximum=64,
|
|
651
|
+
),
|
|
652
|
+
)
|
|
653
|
+
except ContractError as exc:
|
|
654
|
+
if exc.path.startswith("evidence."):
|
|
655
|
+
suffix = exc.path.removeprefix("evidence.")
|
|
656
|
+
raise ContractError(f"{path}.{suffix}", exc.code, exc.detail) from None
|
|
657
|
+
raise
|
|
658
|
+
|
|
659
|
+
|
|
660
|
+
@dataclass(frozen=True)
|
|
661
|
+
class SourceProposal(_CanonicalContract):
|
|
662
|
+
"""A bounded agent proposal; it is never an acquisition authorization."""
|
|
663
|
+
|
|
664
|
+
proposal_id: str
|
|
665
|
+
question_id: str
|
|
666
|
+
source_id: str
|
|
667
|
+
source_class: str
|
|
668
|
+
display_name: str
|
|
669
|
+
locator_kind: str
|
|
670
|
+
locator: str
|
|
671
|
+
data_format: str
|
|
672
|
+
observed_at: str
|
|
673
|
+
historical_coverage: TimeRange
|
|
674
|
+
live_endpoint: str
|
|
675
|
+
publication_delay_seconds: int
|
|
676
|
+
rights_status: str
|
|
677
|
+
fitness_decision: str
|
|
678
|
+
evidence: tuple[EvidenceReference, ...]
|
|
679
|
+
schema_version: str = SOURCE_PROPOSAL_SCHEMA_VERSION
|
|
680
|
+
|
|
681
|
+
EXPECTED_SCHEMA_VERSION: ClassVar[str] = SOURCE_PROPOSAL_SCHEMA_VERSION
|
|
682
|
+
|
|
683
|
+
def __post_init__(self) -> None:
|
|
684
|
+
_version(
|
|
685
|
+
self.schema_version,
|
|
686
|
+
SOURCE_PROPOSAL_SCHEMA_VERSION,
|
|
687
|
+
"source_proposal.schema_version",
|
|
688
|
+
)
|
|
689
|
+
_identifier(self.proposal_id, "source_proposal.proposal_id")
|
|
690
|
+
_identifier(self.question_id, "source_proposal.question_id")
|
|
691
|
+
_identifier(self.source_id, "source_proposal.source_id")
|
|
692
|
+
_choice(self.source_class, _SOURCE_CLASSES, "source_proposal.source_class")
|
|
693
|
+
_text(
|
|
694
|
+
self.display_name,
|
|
695
|
+
"source_proposal.display_name",
|
|
696
|
+
minimum=1,
|
|
697
|
+
maximum=200,
|
|
698
|
+
)
|
|
699
|
+
_choice(self.locator_kind, _LOCATOR_KINDS, "source_proposal.locator_kind")
|
|
700
|
+
validate_source_locator(self.locator, self.locator_kind, "source_proposal.locator")
|
|
701
|
+
_choice(self.data_format, _DATA_FORMATS, "source_proposal.data_format")
|
|
702
|
+
_utc_timestamp(self.observed_at, "source_proposal.observed_at")
|
|
703
|
+
if not isinstance(self.historical_coverage, TimeRange):
|
|
704
|
+
raise ContractError(
|
|
705
|
+
"source_proposal.historical_coverage",
|
|
706
|
+
"TYPE",
|
|
707
|
+
"must be a TimeRange",
|
|
708
|
+
)
|
|
709
|
+
_choice(self.live_endpoint, _LIVE_ENDPOINTS, "source_proposal.live_endpoint")
|
|
710
|
+
_bounded_int(
|
|
711
|
+
self.publication_delay_seconds,
|
|
712
|
+
"source_proposal.publication_delay_seconds",
|
|
713
|
+
minimum=0,
|
|
714
|
+
maximum=10 * 365 * 24 * 60 * 60,
|
|
715
|
+
)
|
|
716
|
+
_choice(self.rights_status, _PROPOSAL_RIGHTS, "source_proposal.rights_status")
|
|
717
|
+
_choice(
|
|
718
|
+
self.fitness_decision,
|
|
719
|
+
_FITNESS_DECISIONS,
|
|
720
|
+
"source_proposal.fitness_decision",
|
|
721
|
+
)
|
|
722
|
+
_typed_tuple(
|
|
723
|
+
self.evidence,
|
|
724
|
+
EvidenceReference,
|
|
725
|
+
"source_proposal.evidence",
|
|
726
|
+
nonempty=True,
|
|
727
|
+
maximum=MAX_EVIDENCE_COUNT,
|
|
728
|
+
)
|
|
729
|
+
evidence_keys = tuple(
|
|
730
|
+
(item.uri, item.observed_at, item.content_sha256) for item in self.evidence
|
|
731
|
+
)
|
|
732
|
+
_require_unique(evidence_keys, "source_proposal.evidence")
|
|
733
|
+
if self.fitness_decision == "selected" and self.rights_status not in {
|
|
734
|
+
"approved",
|
|
735
|
+
"conditional",
|
|
736
|
+
}:
|
|
737
|
+
raise ContractError(
|
|
738
|
+
"source_proposal.rights_status",
|
|
739
|
+
"RIGHTS_NOT_SELECTABLE",
|
|
740
|
+
"selected proposals require approved or conditional rights",
|
|
741
|
+
)
|
|
742
|
+
if self.fitness_decision == "selected" and self.live_endpoint in {"dead", "delayed"}:
|
|
743
|
+
raise ContractError(
|
|
744
|
+
"source_proposal.live_endpoint",
|
|
745
|
+
"ENDPOINT_NOT_SELECTABLE",
|
|
746
|
+
"selected proposals cannot have a dead or delayed endpoint",
|
|
747
|
+
)
|
|
748
|
+
|
|
749
|
+
def to_dict(self) -> dict[str, Any]:
|
|
750
|
+
return {
|
|
751
|
+
"schema_version": self.schema_version,
|
|
752
|
+
"proposal_id": self.proposal_id,
|
|
753
|
+
"question_id": self.question_id,
|
|
754
|
+
"source_id": self.source_id,
|
|
755
|
+
"source_class": self.source_class,
|
|
756
|
+
"display_name": self.display_name,
|
|
757
|
+
"locator_kind": self.locator_kind,
|
|
758
|
+
"locator": self.locator,
|
|
759
|
+
"data_format": self.data_format,
|
|
760
|
+
"observed_at": self.observed_at,
|
|
761
|
+
"historical_coverage": self.historical_coverage.to_dict(),
|
|
762
|
+
"live_endpoint": self.live_endpoint,
|
|
763
|
+
"publication_delay_seconds": self.publication_delay_seconds,
|
|
764
|
+
"rights_status": self.rights_status,
|
|
765
|
+
"fitness_decision": self.fitness_decision,
|
|
766
|
+
"evidence": [item.to_dict() for item in self.evidence],
|
|
767
|
+
}
|
|
768
|
+
|
|
769
|
+
@classmethod
|
|
770
|
+
def from_value(cls, value: Any) -> SourceProposal:
|
|
771
|
+
path = "source_proposal"
|
|
772
|
+
data = _object(value, path)
|
|
773
|
+
_exact_fields(
|
|
774
|
+
data,
|
|
775
|
+
{
|
|
776
|
+
"schema_version",
|
|
777
|
+
"proposal_id",
|
|
778
|
+
"question_id",
|
|
779
|
+
"source_id",
|
|
780
|
+
"source_class",
|
|
781
|
+
"display_name",
|
|
782
|
+
"locator_kind",
|
|
783
|
+
"locator",
|
|
784
|
+
"data_format",
|
|
785
|
+
"observed_at",
|
|
786
|
+
"historical_coverage",
|
|
787
|
+
"live_endpoint",
|
|
788
|
+
"publication_delay_seconds",
|
|
789
|
+
"rights_status",
|
|
790
|
+
"fitness_decision",
|
|
791
|
+
"evidence",
|
|
792
|
+
},
|
|
793
|
+
path,
|
|
794
|
+
)
|
|
795
|
+
evidence = tuple(
|
|
796
|
+
EvidenceReference.from_value(item, f"{path}.evidence[{index}]")
|
|
797
|
+
for index, item in enumerate(
|
|
798
|
+
_array(
|
|
799
|
+
data["evidence"],
|
|
800
|
+
f"{path}.evidence",
|
|
801
|
+
nonempty=True,
|
|
802
|
+
maximum=MAX_EVIDENCE_COUNT,
|
|
803
|
+
)
|
|
804
|
+
)
|
|
805
|
+
)
|
|
806
|
+
return cls(
|
|
807
|
+
schema_version=_required_text(
|
|
808
|
+
data["schema_version"],
|
|
809
|
+
f"{path}.schema_version",
|
|
810
|
+
maximum=64,
|
|
811
|
+
),
|
|
812
|
+
proposal_id=_required_text(
|
|
813
|
+
data["proposal_id"],
|
|
814
|
+
f"{path}.proposal_id",
|
|
815
|
+
maximum=MAX_IDENTIFIER_LENGTH,
|
|
816
|
+
),
|
|
817
|
+
question_id=_required_text(
|
|
818
|
+
data["question_id"],
|
|
819
|
+
f"{path}.question_id",
|
|
820
|
+
maximum=MAX_IDENTIFIER_LENGTH,
|
|
821
|
+
),
|
|
822
|
+
source_id=_required_text(
|
|
823
|
+
data["source_id"],
|
|
824
|
+
f"{path}.source_id",
|
|
825
|
+
maximum=MAX_IDENTIFIER_LENGTH,
|
|
826
|
+
),
|
|
827
|
+
source_class=_required_text(
|
|
828
|
+
data["source_class"],
|
|
829
|
+
f"{path}.source_class",
|
|
830
|
+
maximum=64,
|
|
831
|
+
),
|
|
832
|
+
display_name=_required_text(
|
|
833
|
+
data["display_name"],
|
|
834
|
+
f"{path}.display_name",
|
|
835
|
+
maximum=200,
|
|
836
|
+
),
|
|
837
|
+
locator_kind=_required_text(
|
|
838
|
+
data["locator_kind"],
|
|
839
|
+
f"{path}.locator_kind",
|
|
840
|
+
maximum=64,
|
|
841
|
+
),
|
|
842
|
+
locator=_required_text(
|
|
843
|
+
data["locator"],
|
|
844
|
+
f"{path}.locator",
|
|
845
|
+
maximum=MAX_PATH_LENGTH,
|
|
846
|
+
),
|
|
847
|
+
data_format=_required_text(
|
|
848
|
+
data["data_format"],
|
|
849
|
+
f"{path}.data_format",
|
|
850
|
+
maximum=64,
|
|
851
|
+
),
|
|
852
|
+
observed_at=_required_text(
|
|
853
|
+
data["observed_at"],
|
|
854
|
+
f"{path}.observed_at",
|
|
855
|
+
maximum=32,
|
|
856
|
+
),
|
|
857
|
+
historical_coverage=TimeRange.from_value(
|
|
858
|
+
data["historical_coverage"],
|
|
859
|
+
f"{path}.historical_coverage",
|
|
860
|
+
),
|
|
861
|
+
live_endpoint=_required_text(
|
|
862
|
+
data["live_endpoint"],
|
|
863
|
+
f"{path}.live_endpoint",
|
|
864
|
+
maximum=64,
|
|
865
|
+
),
|
|
866
|
+
publication_delay_seconds=_integer(
|
|
867
|
+
data["publication_delay_seconds"],
|
|
868
|
+
f"{path}.publication_delay_seconds",
|
|
869
|
+
),
|
|
870
|
+
rights_status=_required_text(
|
|
871
|
+
data["rights_status"],
|
|
872
|
+
f"{path}.rights_status",
|
|
873
|
+
maximum=64,
|
|
874
|
+
),
|
|
875
|
+
fitness_decision=_required_text(
|
|
876
|
+
data["fitness_decision"],
|
|
877
|
+
f"{path}.fitness_decision",
|
|
878
|
+
maximum=64,
|
|
879
|
+
),
|
|
880
|
+
evidence=evidence,
|
|
881
|
+
)
|
|
882
|
+
|
|
883
|
+
|
|
884
|
+
@dataclass(frozen=True)
|
|
885
|
+
class RightsSpec:
|
|
886
|
+
"""The rights basis claimed for one source and the uses it permits."""
|
|
887
|
+
|
|
888
|
+
status: str
|
|
889
|
+
evidence: str
|
|
890
|
+
permissions: tuple[str, ...]
|
|
891
|
+
|
|
892
|
+
def __post_init__(self) -> None:
|
|
893
|
+
_choice(self.status, _RIGHTS_STATUSES, "rights.status")
|
|
894
|
+
_text(self.evidence, "rights.evidence", minimum=1, maximum=MAX_TEXT_LENGTH)
|
|
895
|
+
_choice_tuple(
|
|
896
|
+
self.permissions,
|
|
897
|
+
_PERMISSIONS,
|
|
898
|
+
"rights.permissions",
|
|
899
|
+
maximum=len(_PERMISSIONS),
|
|
900
|
+
nonempty=True,
|
|
901
|
+
)
|
|
902
|
+
|
|
903
|
+
def to_dict(self) -> dict[str, Any]:
|
|
904
|
+
return {
|
|
905
|
+
"status": self.status,
|
|
906
|
+
"evidence": self.evidence,
|
|
907
|
+
"permissions": list(self.permissions),
|
|
908
|
+
}
|
|
909
|
+
|
|
910
|
+
|
|
911
|
+
@dataclass(frozen=True)
|
|
912
|
+
class SourceSpec:
|
|
913
|
+
"""One input file the build reads.
|
|
914
|
+
|
|
915
|
+
Carries the source identifier, the path relative to the build input root, the format,
|
|
916
|
+
the stated origin, how the file was acquired, and the rights it is read under.
|
|
917
|
+
"""
|
|
918
|
+
|
|
919
|
+
source_id: str
|
|
920
|
+
path: str
|
|
921
|
+
format: str
|
|
922
|
+
origin: str
|
|
923
|
+
acquisition_method: str
|
|
924
|
+
rights: RightsSpec
|
|
925
|
+
|
|
926
|
+
def __post_init__(self) -> None:
|
|
927
|
+
_identifier(self.source_id, "source.id")
|
|
928
|
+
_relative_posix_path(self.path, "source.path")
|
|
929
|
+
# Build input encoding, narrower than _DATA_FORMATS; see sources/contracts.py.
|
|
930
|
+
_choice(self.format, frozenset({"csv"}), "source.format")
|
|
931
|
+
_text(self.origin, "source.origin", minimum=1, maximum=4_000)
|
|
932
|
+
_choice(
|
|
933
|
+
self.acquisition_method,
|
|
934
|
+
_ACQUISITION_METHODS,
|
|
935
|
+
"source.acquisition_method",
|
|
936
|
+
)
|
|
937
|
+
if not isinstance(self.rights, RightsSpec):
|
|
938
|
+
raise ContractError("source.rights", "TYPE", "must be a RightsSpec")
|
|
939
|
+
|
|
940
|
+
def to_dict(self) -> dict[str, Any]:
|
|
941
|
+
return {
|
|
942
|
+
"id": self.source_id,
|
|
943
|
+
"path": self.path,
|
|
944
|
+
"format": self.format,
|
|
945
|
+
"origin": self.origin,
|
|
946
|
+
"acquisition_method": self.acquisition_method,
|
|
947
|
+
"rights": self.rights.to_dict(),
|
|
948
|
+
}
|
|
949
|
+
|
|
950
|
+
|
|
951
|
+
@dataclass(frozen=True)
|
|
952
|
+
class CleaningStep:
|
|
953
|
+
"""One cleaning operation applied to named columns of one source.
|
|
954
|
+
|
|
955
|
+
``trim`` and ``empty_to_null`` take plain column names. ``rename`` and ``cast`` take
|
|
956
|
+
``(column, target)`` pairs, where the target is the new column name or the cast type.
|
|
957
|
+
"""
|
|
958
|
+
|
|
959
|
+
source_id: str
|
|
960
|
+
operation: str
|
|
961
|
+
columns: tuple[str, ...] | tuple[tuple[str, str], ...]
|
|
962
|
+
|
|
963
|
+
def __post_init__(self) -> None:
|
|
964
|
+
_identifier(self.source_id, "cleaning.source")
|
|
965
|
+
_choice(self.operation, _CLEANING_OPERATIONS, "cleaning.operation")
|
|
966
|
+
if not isinstance(self.columns, tuple) or not self.columns:
|
|
967
|
+
raise ContractError(
|
|
968
|
+
"cleaning.columns",
|
|
969
|
+
"TYPE",
|
|
970
|
+
"must be a non-empty immutable tuple",
|
|
971
|
+
)
|
|
972
|
+
if self.operation in {"trim", "empty_to_null"}:
|
|
973
|
+
_column_tuple(
|
|
974
|
+
self.columns,
|
|
975
|
+
"cleaning.columns",
|
|
976
|
+
nonempty=True,
|
|
977
|
+
maximum=MAX_COLUMN_COUNT,
|
|
978
|
+
)
|
|
979
|
+
return
|
|
980
|
+
pairs = self.columns
|
|
981
|
+
if len(pairs) > MAX_COLUMN_COUNT:
|
|
982
|
+
raise ContractError(
|
|
983
|
+
"cleaning.columns",
|
|
984
|
+
"COLLECTION_LIMIT",
|
|
985
|
+
f"exceeds {MAX_COLUMN_COUNT} entries",
|
|
986
|
+
)
|
|
987
|
+
sources: list[str] = []
|
|
988
|
+
targets: list[str] = []
|
|
989
|
+
for index, pair in enumerate(pairs):
|
|
990
|
+
if (
|
|
991
|
+
not isinstance(pair, tuple)
|
|
992
|
+
or len(pair) != 2
|
|
993
|
+
or not all(isinstance(item, str) for item in pair)
|
|
994
|
+
):
|
|
995
|
+
raise ContractError(
|
|
996
|
+
f"cleaning.columns[{index}]",
|
|
997
|
+
"TYPE",
|
|
998
|
+
"must be a (column, value) string pair",
|
|
999
|
+
)
|
|
1000
|
+
source, target = pair
|
|
1001
|
+
_column_name(source, f"cleaning.columns[{index}].source")
|
|
1002
|
+
if self.operation == "rename":
|
|
1003
|
+
_column_name(target, f"cleaning.columns[{index}].target")
|
|
1004
|
+
else:
|
|
1005
|
+
_choice(target, _CAST_TYPES, f"cleaning.columns[{index}].type")
|
|
1006
|
+
sources.append(source)
|
|
1007
|
+
targets.append(target)
|
|
1008
|
+
_require_unique(sources, "cleaning.columns.sources")
|
|
1009
|
+
if self.operation == "rename":
|
|
1010
|
+
_require_unique(targets, "cleaning.columns.targets")
|
|
1011
|
+
|
|
1012
|
+
def to_dict(self) -> dict[str, Any]:
|
|
1013
|
+
columns: Any
|
|
1014
|
+
if self.operation in {"rename", "cast"}:
|
|
1015
|
+
columns = dict(self.columns)
|
|
1016
|
+
else:
|
|
1017
|
+
columns = list(self.columns)
|
|
1018
|
+
return {
|
|
1019
|
+
"source": self.source_id,
|
|
1020
|
+
"operation": self.operation,
|
|
1021
|
+
"columns": columns,
|
|
1022
|
+
}
|
|
1023
|
+
|
|
1024
|
+
|
|
1025
|
+
@dataclass(frozen=True)
|
|
1026
|
+
class JoinSpec:
|
|
1027
|
+
"""The single join a v1 plan performs.
|
|
1028
|
+
|
|
1029
|
+
Names two distinct sources, the key columns, the join kind, and the cardinality. The
|
|
1030
|
+
build enforces the declared cardinality rather than inferring it.
|
|
1031
|
+
|
|
1032
|
+
``kind`` is ``left`` and stays ``left``, while the graph vocabulary carries ``left``, ``inner``
|
|
1033
|
+
and ``anti``. That is a settled ruling rather than an omission or a pending edit, and
|
|
1034
|
+
``docs/TRANSFORMS.md`` records it. ``pipeline._join`` refuses the first unmatched left row and
|
|
1035
|
+
any row multiplier other than one, unconditionally, because a v1 plan has no way to declare
|
|
1036
|
+
anything else.
|
|
1037
|
+
|
|
1038
|
+
Under that rule ``inner`` cannot differ from ``left`` by a single row, so it would be a second
|
|
1039
|
+
spelling of one behaviour: two names an author would have to choose between with nothing to
|
|
1040
|
+
choose on, and no capability behind either. Admitting it would touch no digest, which makes it
|
|
1041
|
+
cheap rather than worth doing.
|
|
1042
|
+
|
|
1043
|
+
``anti`` is the stronger case, and it starts from the fact that nothing here dispatches on
|
|
1044
|
+
``kind`` at all. ``pipeline._join`` and both backends ignore it; its only reader in the whole
|
|
1045
|
+
software is ``ux.approve._join_sentence``, which renders it into the sentence a person is shown
|
|
1046
|
+
before approving. So widening this set alone would neither produce an anti join nor refuse one:
|
|
1047
|
+
the run would be the left join it always was, the word would be sealed into the plan bytes, and
|
|
1048
|
+
the approval sentence would say ``(anti join)`` about a result that is nothing of the kind.
|
|
1049
|
+
That is fail-open, and it lands on the one surface a person is asked to read.
|
|
1050
|
+
|
|
1051
|
+
Making the kind mean what it says requires withdrawing the v1 rule, because an anti join's
|
|
1052
|
+
output is exactly the rows that rule refuses. That rule is named in the ``mandatory_gates``
|
|
1053
|
+
list ``validation_policy_digest`` hashes, which every approved v1 Recipe is bound to.
|
|
1054
|
+
Withdrawing a name retires all of them; leaving the name in place while changing what it
|
|
1055
|
+
governs would silently change what they meant, which is worse.
|
|
1056
|
+
|
|
1057
|
+
Plans that need those kinds use the ``local-graph-table-plan.v1`` graph, where the join is a
|
|
1058
|
+
node carrying its own declared ``unmatched_policy`` and the same question is answered per node.
|
|
1059
|
+
"""
|
|
1060
|
+
|
|
1061
|
+
left: str
|
|
1062
|
+
right: str
|
|
1063
|
+
on: tuple[str, ...]
|
|
1064
|
+
kind: str
|
|
1065
|
+
cardinality: str
|
|
1066
|
+
|
|
1067
|
+
def __post_init__(self) -> None:
|
|
1068
|
+
_identifier(self.left, "join.left")
|
|
1069
|
+
_identifier(self.right, "join.right")
|
|
1070
|
+
if self.left == self.right:
|
|
1071
|
+
raise ContractError("join.right", "SAME_SOURCE", "must differ from join.left")
|
|
1072
|
+
_column_tuple(self.on, "join.on", nonempty=True, maximum=16)
|
|
1073
|
+
_choice(self.kind, JOIN_KINDS, "join.kind")
|
|
1074
|
+
_choice(self.cardinality, _JOIN_CARDINALITIES, "join.cardinality")
|
|
1075
|
+
|
|
1076
|
+
def to_dict(self) -> dict[str, Any]:
|
|
1077
|
+
return {
|
|
1078
|
+
"left": self.left,
|
|
1079
|
+
"right": self.right,
|
|
1080
|
+
"on": list(self.on),
|
|
1081
|
+
"kind": self.kind,
|
|
1082
|
+
"cardinality": self.cardinality,
|
|
1083
|
+
}
|
|
1084
|
+
|
|
1085
|
+
|
|
1086
|
+
@dataclass(frozen=True)
|
|
1087
|
+
class QualitySpec:
|
|
1088
|
+
"""The gates the joined output must pass.
|
|
1089
|
+
|
|
1090
|
+
A minimum row count, plus the columns that must contain no nulls.
|
|
1091
|
+
"""
|
|
1092
|
+
|
|
1093
|
+
min_rows: int
|
|
1094
|
+
not_null: tuple[str, ...]
|
|
1095
|
+
|
|
1096
|
+
def __post_init__(self) -> None:
|
|
1097
|
+
_bounded_int(
|
|
1098
|
+
self.min_rows,
|
|
1099
|
+
"quality.min_rows",
|
|
1100
|
+
minimum=1,
|
|
1101
|
+
maximum=GRAPH_MAX_OUTPUT_ROWS,
|
|
1102
|
+
)
|
|
1103
|
+
_column_tuple(
|
|
1104
|
+
self.not_null,
|
|
1105
|
+
"quality.not_null",
|
|
1106
|
+
nonempty=False,
|
|
1107
|
+
maximum=MAX_COLUMN_COUNT,
|
|
1108
|
+
)
|
|
1109
|
+
|
|
1110
|
+
def to_dict(self) -> dict[str, Any]:
|
|
1111
|
+
return {"min_rows": self.min_rows, "not_null": list(self.not_null)}
|
|
1112
|
+
|
|
1113
|
+
|
|
1114
|
+
@dataclass(frozen=True)
|
|
1115
|
+
class SemanticEvidenceRef:
|
|
1116
|
+
"""A bounded pointer to one source sealed alongside a semantic claim."""
|
|
1117
|
+
|
|
1118
|
+
kind: str
|
|
1119
|
+
ref: str
|
|
1120
|
+
|
|
1121
|
+
def __post_init__(self) -> None:
|
|
1122
|
+
_choice(self.kind, SEMANTIC_EVIDENCE_KINDS, "semantics.columns.evidence_refs.kind")
|
|
1123
|
+
_text(
|
|
1124
|
+
self.ref,
|
|
1125
|
+
"semantics.columns.evidence_refs.ref",
|
|
1126
|
+
minimum=1,
|
|
1127
|
+
maximum=128,
|
|
1128
|
+
)
|
|
1129
|
+
_identifier(self.ref, "semantics.columns.evidence_refs.ref")
|
|
1130
|
+
|
|
1131
|
+
def to_dict(self) -> dict[str, str]:
|
|
1132
|
+
return {"kind": self.kind, "ref": self.ref}
|
|
1133
|
+
|
|
1134
|
+
|
|
1135
|
+
@dataclass(frozen=True)
|
|
1136
|
+
class CategoryDefinition:
|
|
1137
|
+
"""One bounded code-to-label definition from declared source semantics."""
|
|
1138
|
+
|
|
1139
|
+
code: str
|
|
1140
|
+
label: str
|
|
1141
|
+
|
|
1142
|
+
def __post_init__(self) -> None:
|
|
1143
|
+
_text(self.code, "semantics.columns.categories.code", minimum=1, maximum=64)
|
|
1144
|
+
_text(self.label, "semantics.columns.categories.label", minimum=1, maximum=120)
|
|
1145
|
+
if not is_single_plain_line(self.code) or not is_single_plain_line(self.label):
|
|
1146
|
+
raise ContractError(
|
|
1147
|
+
"semantics.columns.categories",
|
|
1148
|
+
"NONCANONICAL_TEXT",
|
|
1149
|
+
"codes and labels must each be one plain line",
|
|
1150
|
+
)
|
|
1151
|
+
|
|
1152
|
+
def to_dict(self) -> dict[str, str]:
|
|
1153
|
+
return {"code": self.code, "label": self.label}
|
|
1154
|
+
|
|
1155
|
+
|
|
1156
|
+
@dataclass(frozen=True)
|
|
1157
|
+
class SemanticCoverage:
|
|
1158
|
+
"""Declared dataset coverage without inferred temporal or population facts."""
|
|
1159
|
+
|
|
1160
|
+
start_inclusive: str | None
|
|
1161
|
+
end_exclusive: str | None
|
|
1162
|
+
population: str
|
|
1163
|
+
completeness_note: str
|
|
1164
|
+
|
|
1165
|
+
def __post_init__(self) -> None:
|
|
1166
|
+
if (self.start_inclusive is None) != (self.end_exclusive is None):
|
|
1167
|
+
raise ContractError(
|
|
1168
|
+
"semantics.coverage.time_range",
|
|
1169
|
+
"FIELDS",
|
|
1170
|
+
"start_inclusive and end_exclusive must both be set or both be null",
|
|
1171
|
+
)
|
|
1172
|
+
for name, value in (
|
|
1173
|
+
("start_inclusive", self.start_inclusive),
|
|
1174
|
+
("end_exclusive", self.end_exclusive),
|
|
1175
|
+
):
|
|
1176
|
+
if value is not None:
|
|
1177
|
+
_text(value, f"semantics.coverage.time_range.{name}", minimum=1, maximum=128)
|
|
1178
|
+
if not is_single_plain_line(value):
|
|
1179
|
+
raise ContractError(
|
|
1180
|
+
f"semantics.coverage.time_range.{name}",
|
|
1181
|
+
"NONCANONICAL_TEXT",
|
|
1182
|
+
"must be one plain line",
|
|
1183
|
+
)
|
|
1184
|
+
_text(self.population, "semantics.coverage.population", minimum=1, maximum=2_000)
|
|
1185
|
+
_text(
|
|
1186
|
+
self.completeness_note,
|
|
1187
|
+
"semantics.coverage.completeness_note",
|
|
1188
|
+
minimum=1,
|
|
1189
|
+
maximum=1_000,
|
|
1190
|
+
)
|
|
1191
|
+
|
|
1192
|
+
def to_dict(self) -> dict[str, Any]:
|
|
1193
|
+
return {
|
|
1194
|
+
"time_range": None
|
|
1195
|
+
if self.start_inclusive is None
|
|
1196
|
+
else {
|
|
1197
|
+
"start_inclusive": self.start_inclusive,
|
|
1198
|
+
"end_exclusive": self.end_exclusive,
|
|
1199
|
+
},
|
|
1200
|
+
"population": self.population,
|
|
1201
|
+
"completeness_note": self.completeness_note,
|
|
1202
|
+
}
|
|
1203
|
+
|
|
1204
|
+
|
|
1205
|
+
@dataclass(frozen=True)
|
|
1206
|
+
class ColumnSemantics:
|
|
1207
|
+
"""Sealed human-readable meaning of exactly one selected physical column."""
|
|
1208
|
+
|
|
1209
|
+
name: str
|
|
1210
|
+
display_label: str
|
|
1211
|
+
description: str
|
|
1212
|
+
semantic_type: str
|
|
1213
|
+
unit: str
|
|
1214
|
+
unit_state: str
|
|
1215
|
+
normalization: str | None
|
|
1216
|
+
reversible: bool | None
|
|
1217
|
+
categories: tuple[CategoryDefinition, ...] | None
|
|
1218
|
+
categories_status: str
|
|
1219
|
+
evidence_refs: tuple[SemanticEvidenceRef, ...]
|
|
1220
|
+
|
|
1221
|
+
def __post_init__(self) -> None:
|
|
1222
|
+
_column_name(self.name, "semantics.columns.name")
|
|
1223
|
+
_text(self.display_label, "semantics.columns.display_label", minimum=1, maximum=120)
|
|
1224
|
+
if not is_single_plain_line(self.display_label):
|
|
1225
|
+
raise ContractError(
|
|
1226
|
+
"semantics.columns.display_label",
|
|
1227
|
+
"NONCANONICAL_TEXT",
|
|
1228
|
+
"must be one plain line",
|
|
1229
|
+
)
|
|
1230
|
+
_text(self.description, "semantics.columns.description", minimum=1, maximum=1_000)
|
|
1231
|
+
_choice(self.semantic_type, SEMANTIC_TYPES, "semantics.columns.semantic_type")
|
|
1232
|
+
resolved = _declared_unit(self.unit, "semantics.columns.unit")
|
|
1233
|
+
_choice(self.unit_state, UNIT_STATES, "semantics.columns.unit_state")
|
|
1234
|
+
if self.normalization is not None:
|
|
1235
|
+
_choice(
|
|
1236
|
+
self.normalization,
|
|
1237
|
+
NORMALIZATIONS,
|
|
1238
|
+
"semantics.columns.normalization",
|
|
1239
|
+
)
|
|
1240
|
+
if self.reversible is not None and type(self.reversible) is not bool:
|
|
1241
|
+
raise ContractError("semantics.columns.reversible", "TYPE", "must be a boolean or null")
|
|
1242
|
+
|
|
1243
|
+
if self.unit_state == "declared":
|
|
1244
|
+
if (
|
|
1245
|
+
self.unit == NO_UNIT
|
|
1246
|
+
or self.normalization is not None
|
|
1247
|
+
or self.reversible is not None
|
|
1248
|
+
):
|
|
1249
|
+
raise ContractError(
|
|
1250
|
+
"semantics.columns.unit_state",
|
|
1251
|
+
"ENUM",
|
|
1252
|
+
"declared units require a physical unit and no normalization fields",
|
|
1253
|
+
)
|
|
1254
|
+
elif self.unit_state == "normalized":
|
|
1255
|
+
if (
|
|
1256
|
+
self.unit != NO_UNIT
|
|
1257
|
+
or self.normalization is None
|
|
1258
|
+
or type(self.reversible) is not bool
|
|
1259
|
+
):
|
|
1260
|
+
raise ContractError(
|
|
1261
|
+
"semantics.columns.unit_state",
|
|
1262
|
+
"ENUM",
|
|
1263
|
+
"normalized values require unit none, a normalization, and reversible boolean",
|
|
1264
|
+
)
|
|
1265
|
+
elif self.unit != NO_UNIT or self.normalization is not None or self.reversible is not None:
|
|
1266
|
+
raise ContractError(
|
|
1267
|
+
"semantics.columns.unit_state",
|
|
1268
|
+
"ENUM",
|
|
1269
|
+
"not_applicable and unknown require unit none and no normalization fields",
|
|
1270
|
+
)
|
|
1271
|
+
if self.semantic_type == "measure" and self.unit_state == "not_applicable":
|
|
1272
|
+
raise ContractError(
|
|
1273
|
+
"semantics.columns.unit_state",
|
|
1274
|
+
"ENUM",
|
|
1275
|
+
"measure fields require a declared, normalized, or unknown unit state",
|
|
1276
|
+
)
|
|
1277
|
+
# The rule is about the unit rather than about the spelling of it. ``count`` and
|
|
1278
|
+
# ``{count}`` are one unit under two names, and a dimensionless ratio is a different one,
|
|
1279
|
+
# so the question asked here is which unit the code resolved to.
|
|
1280
|
+
if (
|
|
1281
|
+
self.semantic_type not in {"measure", "target"}
|
|
1282
|
+
and self.unit_state == "declared"
|
|
1283
|
+
and resolved is not None
|
|
1284
|
+
and not is_count_unit(resolved)
|
|
1285
|
+
):
|
|
1286
|
+
raise ContractError(
|
|
1287
|
+
"semantics.columns.unit",
|
|
1288
|
+
"ENUM",
|
|
1289
|
+
"non-measure and non-target fields may declare only count units",
|
|
1290
|
+
)
|
|
1291
|
+
|
|
1292
|
+
_choice(
|
|
1293
|
+
self.categories_status,
|
|
1294
|
+
CATEGORY_STATUSES,
|
|
1295
|
+
"semantics.columns.categories_status",
|
|
1296
|
+
)
|
|
1297
|
+
if self.categories is not None:
|
|
1298
|
+
_typed_tuple(
|
|
1299
|
+
self.categories,
|
|
1300
|
+
CategoryDefinition,
|
|
1301
|
+
"semantics.columns.categories",
|
|
1302
|
+
nonempty=True,
|
|
1303
|
+
maximum=64,
|
|
1304
|
+
)
|
|
1305
|
+
_require_unique(
|
|
1306
|
+
(item.code for item in self.categories),
|
|
1307
|
+
"semantics.columns.categories.code",
|
|
1308
|
+
)
|
|
1309
|
+
if self.categories_status in {"complete", "partial"} and not self.categories:
|
|
1310
|
+
raise ContractError(
|
|
1311
|
+
"semantics.columns.categories",
|
|
1312
|
+
"ENUM",
|
|
1313
|
+
"complete or partial category status requires definitions",
|
|
1314
|
+
)
|
|
1315
|
+
if self.categories_status in {"not_applicable", "unknown"} and self.categories is not None:
|
|
1316
|
+
raise ContractError(
|
|
1317
|
+
"semantics.columns.categories",
|
|
1318
|
+
"ENUM",
|
|
1319
|
+
"not_applicable or unknown category status cannot carry definitions",
|
|
1320
|
+
)
|
|
1321
|
+
_typed_tuple(
|
|
1322
|
+
self.evidence_refs,
|
|
1323
|
+
SemanticEvidenceRef,
|
|
1324
|
+
"semantics.columns.evidence_refs",
|
|
1325
|
+
nonempty=True,
|
|
1326
|
+
maximum=8,
|
|
1327
|
+
)
|
|
1328
|
+
_require_unique(
|
|
1329
|
+
((item.kind, item.ref) for item in self.evidence_refs),
|
|
1330
|
+
"semantics.columns.evidence_refs",
|
|
1331
|
+
)
|
|
1332
|
+
|
|
1333
|
+
def to_dict(self) -> dict[str, Any]:
|
|
1334
|
+
return {
|
|
1335
|
+
"name": self.name,
|
|
1336
|
+
"display_label": self.display_label,
|
|
1337
|
+
"description": self.description,
|
|
1338
|
+
"semantic_type": self.semantic_type,
|
|
1339
|
+
"unit": self.unit,
|
|
1340
|
+
"unit_state": self.unit_state,
|
|
1341
|
+
"normalization": self.normalization,
|
|
1342
|
+
"reversible": self.reversible,
|
|
1343
|
+
"categories": None
|
|
1344
|
+
if self.categories is None
|
|
1345
|
+
else [item.to_dict() for item in sorted(self.categories, key=lambda item: item.code)],
|
|
1346
|
+
"categories_status": self.categories_status,
|
|
1347
|
+
"evidence_refs": [
|
|
1348
|
+
item.to_dict()
|
|
1349
|
+
for item in sorted(self.evidence_refs, key=lambda item: (item.kind, item.ref))
|
|
1350
|
+
],
|
|
1351
|
+
}
|
|
1352
|
+
|
|
1353
|
+
|
|
1354
|
+
@dataclass(frozen=True)
|
|
1355
|
+
class TableSemantics(_CanonicalContract):
|
|
1356
|
+
"""One complete, digestable semantic document for a selected Table shape."""
|
|
1357
|
+
|
|
1358
|
+
summary: str
|
|
1359
|
+
grain_statement: str
|
|
1360
|
+
coverage: SemanticCoverage
|
|
1361
|
+
limitations: tuple[str, ...]
|
|
1362
|
+
columns: tuple[ColumnSemantics, ...]
|
|
1363
|
+
schema_version: str = TABLE_SEMANTICS_VERSION
|
|
1364
|
+
|
|
1365
|
+
EXPECTED_SCHEMA_VERSION: ClassVar[str] = TABLE_SEMANTICS_VERSION
|
|
1366
|
+
|
|
1367
|
+
def __post_init__(self) -> None:
|
|
1368
|
+
_choice_version(self.schema_version, TABLE_SEMANTICS_VERSIONS, "semantics.schema_version")
|
|
1369
|
+
_text(self.summary, "semantics.summary", minimum=1, maximum=1_000)
|
|
1370
|
+
_text(self.grain_statement, "semantics.grain_statement", minimum=1, maximum=400)
|
|
1371
|
+
if not isinstance(self.coverage, SemanticCoverage):
|
|
1372
|
+
raise ContractError("semantics.coverage", "TYPE", "must be SemanticCoverage")
|
|
1373
|
+
_text_tuple(
|
|
1374
|
+
self.limitations,
|
|
1375
|
+
"semantics.limitations",
|
|
1376
|
+
nonempty=False,
|
|
1377
|
+
maximum=32,
|
|
1378
|
+
item_maximum=1_000,
|
|
1379
|
+
)
|
|
1380
|
+
_typed_tuple(
|
|
1381
|
+
self.columns,
|
|
1382
|
+
ColumnSemantics,
|
|
1383
|
+
"semantics.columns",
|
|
1384
|
+
nonempty=True,
|
|
1385
|
+
maximum=MAX_COLUMN_COUNT,
|
|
1386
|
+
)
|
|
1387
|
+
_require_unique((item.name for item in self.columns), "semantics.columns.name")
|
|
1388
|
+
if self.schema_version == TABLE_SEMANTICS_V1:
|
|
1389
|
+
# The first generation's vocabulary is closed and stays closed. Widening it in place
|
|
1390
|
+
# would rewrite what a sealed document meant, which is the one thing a version is for.
|
|
1391
|
+
for index, column in enumerate(self.columns):
|
|
1392
|
+
if column.unit not in UNITS:
|
|
1393
|
+
raise ContractError(
|
|
1394
|
+
f"semantics.columns[{index}].unit",
|
|
1395
|
+
"UNIT_UNSUPPORTED_IN_VERSION",
|
|
1396
|
+
f"{TABLE_SEMANTICS_V1} carries only {sorted(UNITS)}; a unit code "
|
|
1397
|
+
f"requires {TABLE_SEMANTICS_V2}",
|
|
1398
|
+
)
|
|
1399
|
+
|
|
1400
|
+
def validate_for_select(self, select: tuple[str, ...]) -> None:
|
|
1401
|
+
if tuple(item.name for item in self.columns) != select:
|
|
1402
|
+
raise ContractError(
|
|
1403
|
+
"semantics.columns",
|
|
1404
|
+
"REFERENCE_MISMATCH",
|
|
1405
|
+
"must follow the exact selected-column order",
|
|
1406
|
+
)
|
|
1407
|
+
|
|
1408
|
+
def validate_for_sources(self, source_ids: tuple[str, ...]) -> None:
|
|
1409
|
+
"""Bind every semantic citation to a source in the sealed plan/Recipe."""
|
|
1410
|
+
|
|
1411
|
+
allowed = frozenset(source_ids)
|
|
1412
|
+
for column in self.columns:
|
|
1413
|
+
for evidence in column.evidence_refs:
|
|
1414
|
+
if evidence.ref not in allowed:
|
|
1415
|
+
raise ContractError(
|
|
1416
|
+
"semantics.columns.evidence_refs.ref",
|
|
1417
|
+
"REFERENCE_MISMATCH",
|
|
1418
|
+
"must reference a source sealed in the same plan or Recipe",
|
|
1419
|
+
)
|
|
1420
|
+
|
|
1421
|
+
def to_dict(self) -> dict[str, Any]:
|
|
1422
|
+
return {
|
|
1423
|
+
"schema_version": self.schema_version,
|
|
1424
|
+
"summary": self.summary,
|
|
1425
|
+
"grain_statement": self.grain_statement,
|
|
1426
|
+
"coverage": self.coverage.to_dict(),
|
|
1427
|
+
"limitations": list(self.limitations),
|
|
1428
|
+
"columns": [item.to_dict() for item in self.columns],
|
|
1429
|
+
}
|
|
1430
|
+
|
|
1431
|
+
|
|
1432
|
+
@dataclass(frozen=True)
|
|
1433
|
+
class TablePlan(_CanonicalContract):
|
|
1434
|
+
"""The complete plan a build executes.
|
|
1435
|
+
|
|
1436
|
+
Names the question, the output intent, the output grain, at least two sources, the
|
|
1437
|
+
cleaning steps, exactly one join, the selected output columns, and the quality gates.
|
|
1438
|
+
The output intent is one of ``local_use``, ``redistribute``, or ``host``, and it gates
|
|
1439
|
+
the rights check: a build refuses unless every source permits that intent. The plan is
|
|
1440
|
+
closed: a build derives every derived member from this value and the raw source bytes
|
|
1441
|
+
alone, which is what lets verification replay the derivation and compare bytes.
|
|
1442
|
+
"""
|
|
1443
|
+
|
|
1444
|
+
question: str
|
|
1445
|
+
output_intent: str
|
|
1446
|
+
grain: tuple[str, ...]
|
|
1447
|
+
sources: tuple[SourceSpec, ...]
|
|
1448
|
+
cleaning: tuple[CleaningStep, ...]
|
|
1449
|
+
join: JoinSpec
|
|
1450
|
+
select: tuple[str, ...]
|
|
1451
|
+
quality: QualitySpec
|
|
1452
|
+
schema_version: str = PLAN_SCHEMA_VERSION
|
|
1453
|
+
|
|
1454
|
+
EXPECTED_SCHEMA_VERSION: ClassVar[str] = PLAN_SCHEMA_VERSION
|
|
1455
|
+
|
|
1456
|
+
def __post_init__(self) -> None:
|
|
1457
|
+
_version(self.schema_version, PLAN_SCHEMA_VERSION, "plan.schema_version")
|
|
1458
|
+
_text(self.question, "plan.question", minimum=1, maximum=MAX_TEXT_LENGTH)
|
|
1459
|
+
_choice(self.output_intent, _OUTPUT_INTENTS, "plan.output_intent")
|
|
1460
|
+
_column_tuple(self.grain, "plan.grain", nonempty=True, maximum=16)
|
|
1461
|
+
_typed_tuple(
|
|
1462
|
+
self.sources,
|
|
1463
|
+
SourceSpec,
|
|
1464
|
+
"plan.sources",
|
|
1465
|
+
nonempty=True,
|
|
1466
|
+
minimum=2,
|
|
1467
|
+
maximum=MAX_SOURCE_COUNT,
|
|
1468
|
+
)
|
|
1469
|
+
source_ids = tuple(source.source_id for source in self.sources)
|
|
1470
|
+
_require_unique(source_ids, "plan.sources.id")
|
|
1471
|
+
_typed_tuple(
|
|
1472
|
+
self.cleaning,
|
|
1473
|
+
CleaningStep,
|
|
1474
|
+
"plan.cleaning",
|
|
1475
|
+
nonempty=False,
|
|
1476
|
+
maximum=MAX_OPERATION_COUNT,
|
|
1477
|
+
)
|
|
1478
|
+
unknown_cleaning_sources = sorted(
|
|
1479
|
+
{step.source_id for step in self.cleaning} - set(source_ids)
|
|
1480
|
+
)
|
|
1481
|
+
if unknown_cleaning_sources:
|
|
1482
|
+
raise ContractError(
|
|
1483
|
+
"plan.cleaning",
|
|
1484
|
+
"UNKNOWN_SOURCE",
|
|
1485
|
+
f"references unknown sources {unknown_cleaning_sources}",
|
|
1486
|
+
)
|
|
1487
|
+
if not isinstance(self.join, JoinSpec):
|
|
1488
|
+
raise ContractError("plan.join", "TYPE", "must be a JoinSpec")
|
|
1489
|
+
if self.join.left not in source_ids:
|
|
1490
|
+
raise ContractError("plan.join.left", "UNKNOWN_SOURCE", "is not in plan.sources")
|
|
1491
|
+
if self.join.right not in source_ids:
|
|
1492
|
+
raise ContractError("plan.join.right", "UNKNOWN_SOURCE", "is not in plan.sources")
|
|
1493
|
+
_column_tuple(
|
|
1494
|
+
self.select,
|
|
1495
|
+
"plan.select",
|
|
1496
|
+
nonempty=True,
|
|
1497
|
+
maximum=MAX_COLUMN_COUNT,
|
|
1498
|
+
)
|
|
1499
|
+
missing_grain = sorted(set(self.grain) - set(self.select))
|
|
1500
|
+
if missing_grain:
|
|
1501
|
+
raise ContractError(
|
|
1502
|
+
"plan.grain",
|
|
1503
|
+
"GRAIN_NOT_SELECTED",
|
|
1504
|
+
f"contains columns absent from plan.select: {missing_grain}",
|
|
1505
|
+
)
|
|
1506
|
+
if not isinstance(self.quality, QualitySpec):
|
|
1507
|
+
raise ContractError("plan.quality", "TYPE", "must be a QualitySpec")
|
|
1508
|
+
_bounded_int(
|
|
1509
|
+
self.quality.min_rows,
|
|
1510
|
+
"plan.quality.min_rows",
|
|
1511
|
+
minimum=1,
|
|
1512
|
+
maximum=MAX_OUTPUT_ROWS,
|
|
1513
|
+
)
|
|
1514
|
+
if not isinstance(self.quality, QualitySpec):
|
|
1515
|
+
raise ContractError("plan.quality", "TYPE", "must be a QualitySpec")
|
|
1516
|
+
missing_quality = sorted(set(self.quality.not_null) - set(self.select))
|
|
1517
|
+
if missing_quality:
|
|
1518
|
+
raise ContractError(
|
|
1519
|
+
"plan.quality.not_null",
|
|
1520
|
+
"QUALITY_COLUMN_NOT_SELECTED",
|
|
1521
|
+
f"contains columns absent from plan.select: {missing_quality}",
|
|
1522
|
+
)
|
|
1523
|
+
|
|
1524
|
+
def to_dict(self) -> dict[str, Any]:
|
|
1525
|
+
return {
|
|
1526
|
+
"schema_version": self.schema_version,
|
|
1527
|
+
"question": self.question,
|
|
1528
|
+
"output_intent": self.output_intent,
|
|
1529
|
+
"grain": list(self.grain),
|
|
1530
|
+
"sources": [source.to_dict() for source in self.sources],
|
|
1531
|
+
"cleaning": [step.to_dict() for step in self.cleaning],
|
|
1532
|
+
"join": self.join.to_dict(),
|
|
1533
|
+
"select": list(self.select),
|
|
1534
|
+
"quality": self.quality.to_dict(),
|
|
1535
|
+
}
|
|
1536
|
+
|
|
1537
|
+
@classmethod
|
|
1538
|
+
def from_value(cls, value: Any) -> TablePlan:
|
|
1539
|
+
return parse_table_plan(value)
|
|
1540
|
+
|
|
1541
|
+
|
|
1542
|
+
@dataclass(frozen=True)
|
|
1543
|
+
class GraphNode:
|
|
1544
|
+
"""One ordered node in a closed table-plan graph."""
|
|
1545
|
+
|
|
1546
|
+
node_id: str
|
|
1547
|
+
operation: str
|
|
1548
|
+
operation_version: str
|
|
1549
|
+
inputs: tuple[str, ...]
|
|
1550
|
+
parameters: tuple[tuple[str, Any], ...]
|
|
1551
|
+
|
|
1552
|
+
def __post_init__(self) -> None:
|
|
1553
|
+
_identifier(self.node_id, "plan.nodes.id")
|
|
1554
|
+
_text(self.operation, "plan.nodes.operation", minimum=1, maximum=64)
|
|
1555
|
+
_text(self.operation_version, "plan.nodes.operation_version", minimum=1, maximum=32)
|
|
1556
|
+
_text_tuple(
|
|
1557
|
+
self.inputs,
|
|
1558
|
+
"plan.nodes.inputs",
|
|
1559
|
+
nonempty=False,
|
|
1560
|
+
maximum=MAX_GRAPH_NODE_COUNT,
|
|
1561
|
+
item_maximum=MAX_IDENTIFIER_LENGTH,
|
|
1562
|
+
)
|
|
1563
|
+
for index, value in enumerate(self.inputs):
|
|
1564
|
+
_identifier(value, f"plan.nodes.inputs[{index}]")
|
|
1565
|
+
if not isinstance(self.parameters, tuple) or any(
|
|
1566
|
+
not isinstance(item, tuple) or len(item) != 2 or not isinstance(item[0], str)
|
|
1567
|
+
for item in self.parameters
|
|
1568
|
+
):
|
|
1569
|
+
raise ContractError("plan.nodes.parameters", "TYPE", "must be frozen parameter pairs")
|
|
1570
|
+
|
|
1571
|
+
def parameter(self, name: str) -> Any:
|
|
1572
|
+
return dict(self.parameters)[name]
|
|
1573
|
+
|
|
1574
|
+
def to_dict(self) -> dict[str, Any]:
|
|
1575
|
+
return {
|
|
1576
|
+
"id": self.node_id,
|
|
1577
|
+
"operation": self.operation,
|
|
1578
|
+
"operation_version": self.operation_version,
|
|
1579
|
+
"inputs": list(self.inputs),
|
|
1580
|
+
"parameters": thaw_parameter(self.parameters),
|
|
1581
|
+
}
|
|
1582
|
+
|
|
1583
|
+
|
|
1584
|
+
@dataclass(frozen=True)
|
|
1585
|
+
class GraphTablePlan(_CanonicalContract):
|
|
1586
|
+
"""A versioned ordered DAG with one declared terminal Table node."""
|
|
1587
|
+
|
|
1588
|
+
operation_contract_version: str
|
|
1589
|
+
question: str
|
|
1590
|
+
output_intent: str
|
|
1591
|
+
grain: tuple[str, ...]
|
|
1592
|
+
sources: tuple[SourceSpec, ...]
|
|
1593
|
+
nodes: tuple[GraphNode, ...]
|
|
1594
|
+
terminal: str
|
|
1595
|
+
output_columns: tuple[str, ...]
|
|
1596
|
+
quality: QualitySpec
|
|
1597
|
+
schema_version: str = GRAPH_PLAN_SCHEMA_VERSION
|
|
1598
|
+
|
|
1599
|
+
EXPECTED_SCHEMA_VERSION: ClassVar[str] = GRAPH_PLAN_SCHEMA_VERSION
|
|
1600
|
+
|
|
1601
|
+
def __post_init__(self) -> None:
|
|
1602
|
+
_version(self.schema_version, GRAPH_PLAN_SCHEMA_VERSION, "plan.schema_version")
|
|
1603
|
+
_version(
|
|
1604
|
+
self.operation_contract_version,
|
|
1605
|
+
GRAPH_OPERATION_CONTRACT_VERSION,
|
|
1606
|
+
"plan.operation_contract_version",
|
|
1607
|
+
)
|
|
1608
|
+
_text(self.question, "plan.question", minimum=1, maximum=MAX_TEXT_LENGTH)
|
|
1609
|
+
_choice(self.output_intent, _OUTPUT_INTENTS, "plan.output_intent")
|
|
1610
|
+
_column_tuple(self.grain, "plan.grain", nonempty=True, maximum=16)
|
|
1611
|
+
_typed_tuple(
|
|
1612
|
+
self.sources,
|
|
1613
|
+
SourceSpec,
|
|
1614
|
+
"plan.sources",
|
|
1615
|
+
nonempty=True,
|
|
1616
|
+
maximum=MAX_SOURCE_COUNT,
|
|
1617
|
+
)
|
|
1618
|
+
source_ids = tuple(source.source_id for source in self.sources)
|
|
1619
|
+
_require_unique(source_ids, "plan.sources.id")
|
|
1620
|
+
_typed_tuple(
|
|
1621
|
+
self.nodes,
|
|
1622
|
+
GraphNode,
|
|
1623
|
+
"plan.nodes",
|
|
1624
|
+
nonempty=True,
|
|
1625
|
+
maximum=MAX_GRAPH_NODE_COUNT,
|
|
1626
|
+
)
|
|
1627
|
+
node_ids = tuple(node.node_id for node in self.nodes)
|
|
1628
|
+
if len(set(node_ids)) != len(node_ids):
|
|
1629
|
+
raise ContractError("plan.nodes.id", "GRAPH_NODE_DUPLICATE", "must be unique")
|
|
1630
|
+
_identifier(self.terminal, "plan.terminal")
|
|
1631
|
+
if self.terminal not in set(node_ids):
|
|
1632
|
+
raise ContractError("plan.terminal", "GRAPH_TERMINAL_UNKNOWN", "is not a graph node")
|
|
1633
|
+
_column_tuple(
|
|
1634
|
+
self.output_columns,
|
|
1635
|
+
"plan.output_columns",
|
|
1636
|
+
nonempty=True,
|
|
1637
|
+
maximum=MAX_COLUMN_COUNT,
|
|
1638
|
+
)
|
|
1639
|
+
missing_grain = sorted(set(self.grain) - set(self.output_columns))
|
|
1640
|
+
if missing_grain:
|
|
1641
|
+
raise ContractError(
|
|
1642
|
+
"plan.grain",
|
|
1643
|
+
"GRAIN_NOT_SELECTED",
|
|
1644
|
+
f"contains columns absent from plan.output_columns: {missing_grain}",
|
|
1645
|
+
)
|
|
1646
|
+
if not isinstance(self.quality, QualitySpec):
|
|
1647
|
+
raise ContractError("plan.quality", "TYPE", "must be a QualitySpec")
|
|
1648
|
+
missing_quality = sorted(set(self.quality.not_null) - set(self.output_columns))
|
|
1649
|
+
if missing_quality:
|
|
1650
|
+
raise ContractError(
|
|
1651
|
+
"plan.quality.not_null",
|
|
1652
|
+
"QUALITY_COLUMN_NOT_SELECTED",
|
|
1653
|
+
f"contains columns absent from plan.output_columns: {missing_quality}",
|
|
1654
|
+
)
|
|
1655
|
+
self._validate_graph(source_ids)
|
|
1656
|
+
|
|
1657
|
+
@property
|
|
1658
|
+
def select(self) -> tuple[str, ...]:
|
|
1659
|
+
"""Compatibility name used by Recipe output-semantic validation."""
|
|
1660
|
+
|
|
1661
|
+
return self.output_columns
|
|
1662
|
+
|
|
1663
|
+
def _validate_graph(self, source_ids: tuple[str, ...]) -> None:
|
|
1664
|
+
seen: set[str] = set()
|
|
1665
|
+
source_bindings: list[str] = []
|
|
1666
|
+
node_by_id = {node.node_id: node for node in self.nodes}
|
|
1667
|
+
for index, node in enumerate(self.nodes):
|
|
1668
|
+
try:
|
|
1669
|
+
operation = resolve_operation(node.operation, node.operation_version)
|
|
1670
|
+
except OperationRegistryError as exc:
|
|
1671
|
+
raise ContractError(
|
|
1672
|
+
f"plan.nodes[{index}].operation", exc.code, exc.detail
|
|
1673
|
+
) from None
|
|
1674
|
+
minimum, maximum = operation.input_arity
|
|
1675
|
+
if not minimum <= len(node.inputs) <= maximum:
|
|
1676
|
+
raise ContractError(
|
|
1677
|
+
f"plan.nodes[{index}].inputs",
|
|
1678
|
+
"OPERATION_ARITY",
|
|
1679
|
+
f"requires {minimum}..{maximum} inputs",
|
|
1680
|
+
)
|
|
1681
|
+
unknown = [input_id for input_id in node.inputs if input_id not in seen]
|
|
1682
|
+
if unknown:
|
|
1683
|
+
raise ContractError(
|
|
1684
|
+
f"plan.nodes[{index}].inputs",
|
|
1685
|
+
"GRAPH_INPUT_ORDER",
|
|
1686
|
+
f"must reference earlier nodes; unavailable={unknown}",
|
|
1687
|
+
)
|
|
1688
|
+
if node.operation == "source":
|
|
1689
|
+
source_id = node.parameter("source")
|
|
1690
|
+
if not isinstance(source_id, str) or source_id not in set(source_ids):
|
|
1691
|
+
raise ContractError(
|
|
1692
|
+
f"plan.nodes[{index}].parameters.source",
|
|
1693
|
+
"GRAPH_SOURCE_UNKNOWN",
|
|
1694
|
+
"must bind one declared source",
|
|
1695
|
+
)
|
|
1696
|
+
source_bindings.append(source_id)
|
|
1697
|
+
if node.operation == "prediction_label":
|
|
1698
|
+
if node.node_id != self.terminal:
|
|
1699
|
+
raise ContractError(
|
|
1700
|
+
f"plan.nodes[{index}]",
|
|
1701
|
+
"PREDICTION_LABEL_POSITION",
|
|
1702
|
+
"prediction_label must be the terminal node so its future value cannot "
|
|
1703
|
+
"feed another transform",
|
|
1704
|
+
)
|
|
1705
|
+
output_column = node.parameter("output_column")
|
|
1706
|
+
if output_column not in self.output_columns:
|
|
1707
|
+
raise ContractError(
|
|
1708
|
+
f"plan.nodes[{index}].parameters.output_column",
|
|
1709
|
+
"PREDICTION_LABEL_OUTPUT",
|
|
1710
|
+
"must be selected as an output column",
|
|
1711
|
+
)
|
|
1712
|
+
seen.add(node.node_id)
|
|
1713
|
+
if len(set(source_bindings)) != len(source_bindings):
|
|
1714
|
+
raise ContractError(
|
|
1715
|
+
"plan.nodes.parameters.source",
|
|
1716
|
+
"GRAPH_SOURCE_DUPLICATE",
|
|
1717
|
+
"a declared source may have only one source node",
|
|
1718
|
+
)
|
|
1719
|
+
if set(source_bindings) != set(source_ids):
|
|
1720
|
+
raise ContractError(
|
|
1721
|
+
"plan.nodes.parameters.source",
|
|
1722
|
+
"GRAPH_SOURCE_SET",
|
|
1723
|
+
"source nodes must bind every declared source exactly once",
|
|
1724
|
+
)
|
|
1725
|
+
reachable: set[str] = set()
|
|
1726
|
+
|
|
1727
|
+
def visit(node_id: str) -> None:
|
|
1728
|
+
if node_id in reachable:
|
|
1729
|
+
return
|
|
1730
|
+
reachable.add(node_id)
|
|
1731
|
+
for input_id in node_by_id[node_id].inputs:
|
|
1732
|
+
visit(input_id)
|
|
1733
|
+
|
|
1734
|
+
visit(self.terminal)
|
|
1735
|
+
unreachable = sorted(set(node_by_id) - reachable)
|
|
1736
|
+
if unreachable:
|
|
1737
|
+
raise ContractError(
|
|
1738
|
+
"plan.nodes",
|
|
1739
|
+
"GRAPH_NODE_UNREACHABLE",
|
|
1740
|
+
f"nodes are outside terminal ancestry: {unreachable}",
|
|
1741
|
+
)
|
|
1742
|
+
|
|
1743
|
+
def to_dict(self) -> dict[str, Any]:
|
|
1744
|
+
return {
|
|
1745
|
+
"schema_version": self.schema_version,
|
|
1746
|
+
"operation_contract_version": self.operation_contract_version,
|
|
1747
|
+
"question": self.question,
|
|
1748
|
+
"output_intent": self.output_intent,
|
|
1749
|
+
"grain": list(self.grain),
|
|
1750
|
+
"sources": [source.to_dict() for source in self.sources],
|
|
1751
|
+
"nodes": [node.to_dict() for node in self.nodes],
|
|
1752
|
+
"terminal": self.terminal,
|
|
1753
|
+
"output_columns": list(self.output_columns),
|
|
1754
|
+
"quality": self.quality.to_dict(),
|
|
1755
|
+
}
|
|
1756
|
+
|
|
1757
|
+
@classmethod
|
|
1758
|
+
def from_value(cls, value: Any) -> GraphTablePlan:
|
|
1759
|
+
return parse_graph_table_plan(value)
|
|
1760
|
+
|
|
1761
|
+
|
|
1762
|
+
def is_single_plain_line(value: object) -> bool:
|
|
1763
|
+
"""Return whether text contains no boundary that renders as another display line.
|
|
1764
|
+
|
|
1765
|
+
This deliberately answers only the structural question. A tab remains part of one line, and
|
|
1766
|
+
U+009B and bidi controls remain outside this rule pending an explicit terminal-safety policy.
|
|
1767
|
+
"""
|
|
1768
|
+
|
|
1769
|
+
return isinstance(value, str) and not any(
|
|
1770
|
+
boundary in value for boundary in _DISPLAY_LINE_BOUNDARIES
|
|
1771
|
+
)
|
|
1772
|
+
|
|
1773
|
+
|
|
1774
|
+
def parse_question(value: Any) -> LocalQuestion:
|
|
1775
|
+
return LocalQuestion.from_value(value)
|
|
1776
|
+
|
|
1777
|
+
|
|
1778
|
+
def parse_requirements(value: Any) -> LocalRequirements:
|
|
1779
|
+
return LocalRequirements.from_value(value)
|
|
1780
|
+
|
|
1781
|
+
|
|
1782
|
+
def parse_source_proposal(value: Any) -> SourceProposal:
|
|
1783
|
+
return SourceProposal.from_value(value)
|
|
1784
|
+
|
|
1785
|
+
|
|
1786
|
+
def parse_table_semantics(value: Any) -> TableSemantics:
|
|
1787
|
+
"""Parse the closed human-readable semantics document without inferring any claim."""
|
|
1788
|
+
|
|
1789
|
+
path = "semantics"
|
|
1790
|
+
data = _object(value, path)
|
|
1791
|
+
_exact_fields(
|
|
1792
|
+
data,
|
|
1793
|
+
{"schema_version", "summary", "grain_statement", "coverage", "limitations", "columns"},
|
|
1794
|
+
path,
|
|
1795
|
+
)
|
|
1796
|
+
coverage_data = _object(data["coverage"], f"{path}.coverage")
|
|
1797
|
+
_exact_fields(
|
|
1798
|
+
coverage_data,
|
|
1799
|
+
{"time_range", "population", "completeness_note"},
|
|
1800
|
+
f"{path}.coverage",
|
|
1801
|
+
)
|
|
1802
|
+
time_range = coverage_data["time_range"]
|
|
1803
|
+
start_inclusive: str | None = None
|
|
1804
|
+
end_exclusive: str | None = None
|
|
1805
|
+
if time_range is not None:
|
|
1806
|
+
time_data = _object(time_range, f"{path}.coverage.time_range")
|
|
1807
|
+
_exact_fields(
|
|
1808
|
+
time_data,
|
|
1809
|
+
{"start_inclusive", "end_exclusive"},
|
|
1810
|
+
f"{path}.coverage.time_range",
|
|
1811
|
+
)
|
|
1812
|
+
start_inclusive = _required_text(
|
|
1813
|
+
time_data["start_inclusive"],
|
|
1814
|
+
f"{path}.coverage.time_range.start_inclusive",
|
|
1815
|
+
maximum=128,
|
|
1816
|
+
)
|
|
1817
|
+
end_exclusive = _required_text(
|
|
1818
|
+
time_data["end_exclusive"],
|
|
1819
|
+
f"{path}.coverage.time_range.end_exclusive",
|
|
1820
|
+
maximum=128,
|
|
1821
|
+
)
|
|
1822
|
+
|
|
1823
|
+
columns: list[ColumnSemantics] = []
|
|
1824
|
+
for index, raw_column in enumerate(
|
|
1825
|
+
_array(data["columns"], f"{path}.columns", nonempty=True, maximum=MAX_COLUMN_COUNT)
|
|
1826
|
+
):
|
|
1827
|
+
column_path = f"{path}.columns[{index}]"
|
|
1828
|
+
column = _object(raw_column, column_path)
|
|
1829
|
+
_exact_fields(
|
|
1830
|
+
column,
|
|
1831
|
+
{
|
|
1832
|
+
"name",
|
|
1833
|
+
"display_label",
|
|
1834
|
+
"description",
|
|
1835
|
+
"semantic_type",
|
|
1836
|
+
"unit",
|
|
1837
|
+
"unit_state",
|
|
1838
|
+
"normalization",
|
|
1839
|
+
"reversible",
|
|
1840
|
+
"categories",
|
|
1841
|
+
"categories_status",
|
|
1842
|
+
"evidence_refs",
|
|
1843
|
+
},
|
|
1844
|
+
column_path,
|
|
1845
|
+
)
|
|
1846
|
+
categories: tuple[CategoryDefinition, ...] | None = None
|
|
1847
|
+
if column["categories"] is not None:
|
|
1848
|
+
parsed_categories: list[CategoryDefinition] = []
|
|
1849
|
+
for category_index, raw_category in enumerate(
|
|
1850
|
+
_array(
|
|
1851
|
+
column["categories"],
|
|
1852
|
+
f"{column_path}.categories",
|
|
1853
|
+
nonempty=True,
|
|
1854
|
+
maximum=64,
|
|
1855
|
+
)
|
|
1856
|
+
):
|
|
1857
|
+
category_path = f"{column_path}.categories[{category_index}]"
|
|
1858
|
+
category = _object(raw_category, category_path)
|
|
1859
|
+
_exact_fields(category, {"code", "label"}, category_path)
|
|
1860
|
+
parsed_categories.append(
|
|
1861
|
+
CategoryDefinition(
|
|
1862
|
+
code=_required_text(category["code"], f"{category_path}.code", maximum=64),
|
|
1863
|
+
label=_required_text(
|
|
1864
|
+
category["label"], f"{category_path}.label", maximum=120
|
|
1865
|
+
),
|
|
1866
|
+
)
|
|
1867
|
+
)
|
|
1868
|
+
categories = tuple(parsed_categories)
|
|
1869
|
+
|
|
1870
|
+
evidence_refs: list[SemanticEvidenceRef] = []
|
|
1871
|
+
for evidence_index, raw_evidence in enumerate(
|
|
1872
|
+
_array(
|
|
1873
|
+
column["evidence_refs"],
|
|
1874
|
+
f"{column_path}.evidence_refs",
|
|
1875
|
+
nonempty=False,
|
|
1876
|
+
maximum=8,
|
|
1877
|
+
)
|
|
1878
|
+
):
|
|
1879
|
+
evidence_path = f"{column_path}.evidence_refs[{evidence_index}]"
|
|
1880
|
+
evidence = _object(raw_evidence, evidence_path)
|
|
1881
|
+
_exact_fields(evidence, {"kind", "ref"}, evidence_path)
|
|
1882
|
+
evidence_refs.append(
|
|
1883
|
+
SemanticEvidenceRef(
|
|
1884
|
+
kind=_required_text(evidence["kind"], f"{evidence_path}.kind", maximum=64),
|
|
1885
|
+
ref=_required_text(evidence["ref"], f"{evidence_path}.ref", maximum=128),
|
|
1886
|
+
)
|
|
1887
|
+
)
|
|
1888
|
+
|
|
1889
|
+
reversible = column["reversible"]
|
|
1890
|
+
if reversible is not None and type(reversible) is not bool:
|
|
1891
|
+
raise ContractError(f"{column_path}.reversible", "TYPE", "must be a boolean or null")
|
|
1892
|
+
normalization = column["normalization"]
|
|
1893
|
+
if normalization is not None and not isinstance(normalization, str):
|
|
1894
|
+
raise ContractError(f"{column_path}.normalization", "TYPE", "must be a string or null")
|
|
1895
|
+
columns.append(
|
|
1896
|
+
ColumnSemantics(
|
|
1897
|
+
name=_required_text(column["name"], f"{column_path}.name", maximum=64),
|
|
1898
|
+
display_label=_required_text(
|
|
1899
|
+
column["display_label"], f"{column_path}.display_label", maximum=120
|
|
1900
|
+
),
|
|
1901
|
+
description=_required_text(
|
|
1902
|
+
column["description"], f"{column_path}.description", maximum=1_000
|
|
1903
|
+
),
|
|
1904
|
+
semantic_type=_required_text(
|
|
1905
|
+
column["semantic_type"], f"{column_path}.semantic_type", maximum=64
|
|
1906
|
+
),
|
|
1907
|
+
unit=_required_text(column["unit"], f"{column_path}.unit", maximum=64),
|
|
1908
|
+
unit_state=_required_text(
|
|
1909
|
+
column["unit_state"], f"{column_path}.unit_state", maximum=64
|
|
1910
|
+
),
|
|
1911
|
+
normalization=normalization,
|
|
1912
|
+
reversible=reversible,
|
|
1913
|
+
categories=categories,
|
|
1914
|
+
categories_status=_required_text(
|
|
1915
|
+
column["categories_status"],
|
|
1916
|
+
f"{column_path}.categories_status",
|
|
1917
|
+
maximum=64,
|
|
1918
|
+
),
|
|
1919
|
+
evidence_refs=tuple(evidence_refs),
|
|
1920
|
+
)
|
|
1921
|
+
)
|
|
1922
|
+
|
|
1923
|
+
return TableSemantics(
|
|
1924
|
+
schema_version=_required_text(data["schema_version"], f"{path}.schema_version", maximum=64),
|
|
1925
|
+
summary=_required_text(data["summary"], f"{path}.summary", maximum=1_000),
|
|
1926
|
+
grain_statement=_required_text(
|
|
1927
|
+
data["grain_statement"], f"{path}.grain_statement", maximum=400
|
|
1928
|
+
),
|
|
1929
|
+
coverage=SemanticCoverage(
|
|
1930
|
+
start_inclusive=start_inclusive,
|
|
1931
|
+
end_exclusive=end_exclusive,
|
|
1932
|
+
population=_required_text(
|
|
1933
|
+
coverage_data["population"], f"{path}.coverage.population", maximum=2_000
|
|
1934
|
+
),
|
|
1935
|
+
completeness_note=_required_text(
|
|
1936
|
+
coverage_data["completeness_note"],
|
|
1937
|
+
f"{path}.coverage.completeness_note",
|
|
1938
|
+
maximum=1_000,
|
|
1939
|
+
),
|
|
1940
|
+
),
|
|
1941
|
+
limitations=_parse_text_array(
|
|
1942
|
+
data["limitations"],
|
|
1943
|
+
f"{path}.limitations",
|
|
1944
|
+
nonempty=False,
|
|
1945
|
+
maximum=32,
|
|
1946
|
+
item_maximum=1_000,
|
|
1947
|
+
),
|
|
1948
|
+
columns=tuple(columns),
|
|
1949
|
+
)
|
|
1950
|
+
|
|
1951
|
+
|
|
1952
|
+
def parse_table_plan(value: Any) -> TablePlan:
|
|
1953
|
+
"""Parse an untrusted JSON-compatible value into the closed execution plan."""
|
|
1954
|
+
|
|
1955
|
+
path = "plan"
|
|
1956
|
+
data = _object(value, path)
|
|
1957
|
+
_exact_fields(
|
|
1958
|
+
data,
|
|
1959
|
+
{
|
|
1960
|
+
"schema_version",
|
|
1961
|
+
"question",
|
|
1962
|
+
"output_intent",
|
|
1963
|
+
"grain",
|
|
1964
|
+
"sources",
|
|
1965
|
+
"cleaning",
|
|
1966
|
+
"join",
|
|
1967
|
+
"select",
|
|
1968
|
+
"quality",
|
|
1969
|
+
},
|
|
1970
|
+
path,
|
|
1971
|
+
)
|
|
1972
|
+
sources = tuple(
|
|
1973
|
+
_parse_source(item, f"{path}.sources[{index}]")
|
|
1974
|
+
for index, item in enumerate(
|
|
1975
|
+
_array(
|
|
1976
|
+
data["sources"],
|
|
1977
|
+
f"{path}.sources",
|
|
1978
|
+
nonempty=True,
|
|
1979
|
+
minimum=2,
|
|
1980
|
+
maximum=MAX_SOURCE_COUNT,
|
|
1981
|
+
)
|
|
1982
|
+
)
|
|
1983
|
+
)
|
|
1984
|
+
_require_unique((source.source_id for source in sources), f"{path}.sources.id")
|
|
1985
|
+
source_ids = frozenset(source.source_id for source in sources)
|
|
1986
|
+
cleaning = tuple(
|
|
1987
|
+
_parse_cleaning(item, f"{path}.cleaning[{index}]", source_ids)
|
|
1988
|
+
for index, item in enumerate(
|
|
1989
|
+
_array(
|
|
1990
|
+
data["cleaning"],
|
|
1991
|
+
f"{path}.cleaning",
|
|
1992
|
+
nonempty=False,
|
|
1993
|
+
maximum=MAX_OPERATION_COUNT,
|
|
1994
|
+
)
|
|
1995
|
+
)
|
|
1996
|
+
)
|
|
1997
|
+
return TablePlan(
|
|
1998
|
+
schema_version=_required_text(
|
|
1999
|
+
data["schema_version"],
|
|
2000
|
+
f"{path}.schema_version",
|
|
2001
|
+
maximum=64,
|
|
2002
|
+
),
|
|
2003
|
+
question=_required_text(
|
|
2004
|
+
data["question"],
|
|
2005
|
+
f"{path}.question",
|
|
2006
|
+
maximum=MAX_TEXT_LENGTH,
|
|
2007
|
+
),
|
|
2008
|
+
output_intent=_required_text(
|
|
2009
|
+
data["output_intent"],
|
|
2010
|
+
f"{path}.output_intent",
|
|
2011
|
+
maximum=64,
|
|
2012
|
+
),
|
|
2013
|
+
grain=_parse_column_array(
|
|
2014
|
+
data["grain"],
|
|
2015
|
+
f"{path}.grain",
|
|
2016
|
+
nonempty=True,
|
|
2017
|
+
maximum=16,
|
|
2018
|
+
),
|
|
2019
|
+
sources=sources,
|
|
2020
|
+
cleaning=cleaning,
|
|
2021
|
+
join=_parse_join(data["join"], f"{path}.join", source_ids),
|
|
2022
|
+
select=_parse_column_array(
|
|
2023
|
+
data["select"],
|
|
2024
|
+
f"{path}.select",
|
|
2025
|
+
nonempty=True,
|
|
2026
|
+
maximum=MAX_COLUMN_COUNT,
|
|
2027
|
+
),
|
|
2028
|
+
quality=_parse_quality(data["quality"], f"{path}.quality"),
|
|
2029
|
+
)
|
|
2030
|
+
|
|
2031
|
+
|
|
2032
|
+
def parse_graph_table_plan(value: Any) -> GraphTablePlan:
|
|
2033
|
+
"""Parse an untrusted v2 graph without executing any node."""
|
|
2034
|
+
|
|
2035
|
+
path = "plan"
|
|
2036
|
+
data = _object(value, path)
|
|
2037
|
+
_exact_fields(
|
|
2038
|
+
data,
|
|
2039
|
+
{
|
|
2040
|
+
"schema_version",
|
|
2041
|
+
"operation_contract_version",
|
|
2042
|
+
"question",
|
|
2043
|
+
"output_intent",
|
|
2044
|
+
"grain",
|
|
2045
|
+
"sources",
|
|
2046
|
+
"nodes",
|
|
2047
|
+
"terminal",
|
|
2048
|
+
"output_columns",
|
|
2049
|
+
"quality",
|
|
2050
|
+
},
|
|
2051
|
+
path,
|
|
2052
|
+
)
|
|
2053
|
+
operation_contract = _required_text(
|
|
2054
|
+
data["operation_contract_version"],
|
|
2055
|
+
f"{path}.operation_contract_version",
|
|
2056
|
+
maximum=64,
|
|
2057
|
+
)
|
|
2058
|
+
if operation_contract != GRAPH_OPERATION_CONTRACT_VERSION:
|
|
2059
|
+
raise ContractError(
|
|
2060
|
+
f"{path}.operation_contract_version",
|
|
2061
|
+
"PLAN_OPERATION_PAIR",
|
|
2062
|
+
f"{GRAPH_PLAN_SCHEMA_VERSION} requires {GRAPH_OPERATION_CONTRACT_VERSION}",
|
|
2063
|
+
)
|
|
2064
|
+
sources = tuple(
|
|
2065
|
+
_parse_source(item, f"{path}.sources[{index}]")
|
|
2066
|
+
for index, item in enumerate(
|
|
2067
|
+
_array(data["sources"], f"{path}.sources", nonempty=True, maximum=MAX_SOURCE_COUNT)
|
|
2068
|
+
)
|
|
2069
|
+
)
|
|
2070
|
+
nodes: list[GraphNode] = []
|
|
2071
|
+
for index, item in enumerate(
|
|
2072
|
+
_array(data["nodes"], f"{path}.nodes", nonempty=True, maximum=MAX_GRAPH_NODE_COUNT)
|
|
2073
|
+
):
|
|
2074
|
+
node_path = f"{path}.nodes[{index}]"
|
|
2075
|
+
node_data = _object(item, node_path)
|
|
2076
|
+
_exact_fields(
|
|
2077
|
+
node_data,
|
|
2078
|
+
{"id", "operation", "operation_version", "inputs", "parameters"},
|
|
2079
|
+
node_path,
|
|
2080
|
+
)
|
|
2081
|
+
operation_name = _required_text(
|
|
2082
|
+
node_data["operation"], f"{node_path}.operation", maximum=64
|
|
2083
|
+
)
|
|
2084
|
+
operation_version = _required_text(
|
|
2085
|
+
node_data["operation_version"], f"{node_path}.operation_version", maximum=32
|
|
2086
|
+
)
|
|
2087
|
+
try:
|
|
2088
|
+
operation = resolve_operation(operation_name, operation_version)
|
|
2089
|
+
parameters = operation.validate_parameters(
|
|
2090
|
+
node_data["parameters"], path=f"{node_path}.parameters"
|
|
2091
|
+
)
|
|
2092
|
+
except OperationRegistryError as exc:
|
|
2093
|
+
raise ContractError(exc.path, exc.code, exc.detail) from None
|
|
2094
|
+
nodes.append(
|
|
2095
|
+
GraphNode(
|
|
2096
|
+
node_id=_required_text(node_data["id"], f"{node_path}.id", maximum=64),
|
|
2097
|
+
operation=operation_name,
|
|
2098
|
+
operation_version=operation_version,
|
|
2099
|
+
inputs=tuple(
|
|
2100
|
+
_required_text(value, f"{node_path}.inputs[{input_index}]", maximum=64)
|
|
2101
|
+
for input_index, value in enumerate(
|
|
2102
|
+
_array(
|
|
2103
|
+
node_data["inputs"],
|
|
2104
|
+
f"{node_path}.inputs",
|
|
2105
|
+
nonempty=False,
|
|
2106
|
+
maximum=MAX_GRAPH_NODE_COUNT,
|
|
2107
|
+
)
|
|
2108
|
+
)
|
|
2109
|
+
),
|
|
2110
|
+
parameters=parameters,
|
|
2111
|
+
)
|
|
2112
|
+
)
|
|
2113
|
+
return GraphTablePlan(
|
|
2114
|
+
schema_version=_required_text(data["schema_version"], f"{path}.schema_version", maximum=64),
|
|
2115
|
+
operation_contract_version=operation_contract,
|
|
2116
|
+
question=_required_text(data["question"], f"{path}.question", maximum=MAX_TEXT_LENGTH),
|
|
2117
|
+
output_intent=_required_text(data["output_intent"], f"{path}.output_intent", maximum=64),
|
|
2118
|
+
grain=_parse_column_array(data["grain"], f"{path}.grain", nonempty=True, maximum=16),
|
|
2119
|
+
sources=sources,
|
|
2120
|
+
nodes=tuple(nodes),
|
|
2121
|
+
terminal=_required_text(data["terminal"], f"{path}.terminal", maximum=64),
|
|
2122
|
+
output_columns=_parse_column_array(
|
|
2123
|
+
data["output_columns"],
|
|
2124
|
+
f"{path}.output_columns",
|
|
2125
|
+
nonempty=True,
|
|
2126
|
+
maximum=MAX_COLUMN_COUNT,
|
|
2127
|
+
),
|
|
2128
|
+
quality=_parse_quality(data["quality"], f"{path}.quality"),
|
|
2129
|
+
)
|
|
2130
|
+
|
|
2131
|
+
|
|
2132
|
+
def parse_plan_document(value: Any) -> TablePlan | GraphTablePlan:
|
|
2133
|
+
"""Dispatch exactly one accepted plan generation by its schema version.
|
|
2134
|
+
|
|
2135
|
+
This is the reader for a plan of any age, including one read back out of a sealed Build, so it
|
|
2136
|
+
understands every operation the registry still holds. :func:`parse_fresh_plan_document` is the
|
|
2137
|
+
entry point for newly authored bytes.
|
|
2138
|
+
"""
|
|
2139
|
+
|
|
2140
|
+
data = _object(value, "plan")
|
|
2141
|
+
schema_version = data.get("schema_version")
|
|
2142
|
+
if schema_version == PLAN_SCHEMA_VERSION:
|
|
2143
|
+
return parse_table_plan(value)
|
|
2144
|
+
if schema_version == GRAPH_PLAN_SCHEMA_VERSION:
|
|
2145
|
+
return parse_graph_table_plan(value)
|
|
2146
|
+
if schema_version == _LEGACY_PLAN_SCHEMA_VERSION:
|
|
2147
|
+
detail = (
|
|
2148
|
+
f"found {schema_version!r}, supported {PLAN_SCHEMA_VERSION!r} (renamed in V3); "
|
|
2149
|
+
f"set plan.schema_version to {PLAN_SCHEMA_VERSION!r} and retry"
|
|
2150
|
+
)
|
|
2151
|
+
else:
|
|
2152
|
+
detail = (
|
|
2153
|
+
f"found {schema_version!r}, supported one of "
|
|
2154
|
+
f"{[PLAN_SCHEMA_VERSION, GRAPH_PLAN_SCHEMA_VERSION]}"
|
|
2155
|
+
)
|
|
2156
|
+
raise ContractError("plan.schema_version", "VERSION", detail)
|
|
2157
|
+
|
|
2158
|
+
|
|
2159
|
+
def refuse_retired_operations(plan: TablePlan | GraphTablePlan) -> None:
|
|
2160
|
+
"""Refuse a plan that names an operation the graph vocabulary has retired.
|
|
2161
|
+
|
|
2162
|
+
A retired operation is one the admission contract in ``docs/TRANSFORMS.md`` no longer admits.
|
|
2163
|
+
It stays in the registry because a Build sealed with it, and a Recipe approved with it, both
|
|
2164
|
+
re-parse their stored plan through that table to verify -- dropping the entry would make
|
|
2165
|
+
already-sealed bytes unreadable, which is a worse thing than the operation continuing to
|
|
2166
|
+
exist. So the line is drawn by age of the bytes rather than by the table: this is called on
|
|
2167
|
+
newly authored bytes and on nothing else, exactly as
|
|
2168
|
+
:func:`mostlyright.data_harness.recipe.parse_fresh_recipe` calls
|
|
2169
|
+
``require_resolvable_reader`` for a Reader pin whose implementation has since been retired.
|
|
2170
|
+
"""
|
|
2171
|
+
|
|
2172
|
+
if not isinstance(plan, GraphTablePlan):
|
|
2173
|
+
return
|
|
2174
|
+
for index, node in enumerate(plan.nodes):
|
|
2175
|
+
if node.operation in RETIRED_OPERATIONS:
|
|
2176
|
+
raise ContractError(
|
|
2177
|
+
f"plan.nodes[{index}].operation",
|
|
2178
|
+
"OPERATION_RETIRED",
|
|
2179
|
+
f"operation {node.operation!r} is retired and cannot be named by a new plan",
|
|
2180
|
+
)
|
|
2181
|
+
|
|
2182
|
+
|
|
2183
|
+
def refuse_retired_operation_names(value: Any) -> None:
|
|
2184
|
+
"""Say "retired" about a node before saying anything else about it.
|
|
2185
|
+
|
|
2186
|
+
The structural rules a retired operation still carries -- where it may sit, what it must
|
|
2187
|
+
select -- fire during the parse, ahead of the retirement check that runs after it. Their advice
|
|
2188
|
+
is to correct the offending field and run the command again, which for a node that cannot be in
|
|
2189
|
+
a new plan at all sends the author round a loop that ends here anyway, and the author is
|
|
2190
|
+
frequently an agent that will act on it. So the retirement is read off the raw document first.
|
|
2191
|
+
|
|
2192
|
+
Anything that is not a well-formed node list is left alone, so a malformed document still gets
|
|
2193
|
+
the ordinary parse error that describes its actual shape rather than this one.
|
|
2194
|
+
"""
|
|
2195
|
+
|
|
2196
|
+
if not isinstance(value, Mapping):
|
|
2197
|
+
return
|
|
2198
|
+
nodes = value.get("nodes")
|
|
2199
|
+
if not isinstance(nodes, list):
|
|
2200
|
+
return
|
|
2201
|
+
for index, node in enumerate(nodes):
|
|
2202
|
+
if not isinstance(node, Mapping):
|
|
2203
|
+
continue
|
|
2204
|
+
operation = node.get("operation")
|
|
2205
|
+
if isinstance(operation, str) and operation in RETIRED_OPERATIONS:
|
|
2206
|
+
raise ContractError(
|
|
2207
|
+
f"plan.nodes[{index}].operation",
|
|
2208
|
+
"OPERATION_RETIRED",
|
|
2209
|
+
f"operation {operation!r} is retired and cannot be named by a new plan",
|
|
2210
|
+
)
|
|
2211
|
+
|
|
2212
|
+
|
|
2213
|
+
def refuse_inconsistent_units(plan: TablePlan | GraphTablePlan) -> None:
|
|
2214
|
+
"""Refuse a plan whose declared units contradict each other or the arithmetic on them.
|
|
2215
|
+
|
|
2216
|
+
This is on the authoring path and on nothing else, for the same reason the retirement check is.
|
|
2217
|
+
The analysis reads a declaration that did not exist before it, so nothing already sealed can
|
|
2218
|
+
trip it today -- but it is an analysis rather than a table, and a later pass that sees more will
|
|
2219
|
+
refuse more. Running it where sealed bytes are read would make that later pass retroactive:
|
|
2220
|
+
a plan sealed under this version would stop verifying under the next one, which is the outcome
|
|
2221
|
+
the age-of-the-bytes line exists to prevent.
|
|
2222
|
+
"""
|
|
2223
|
+
|
|
2224
|
+
if not isinstance(plan, GraphTablePlan):
|
|
2225
|
+
return
|
|
2226
|
+
from mostlyright.data_harness.unit_flow import UnitFlowError
|
|
2227
|
+
from mostlyright.data_harness.unit_flow import refuse_inconsistent_units as _refuse
|
|
2228
|
+
|
|
2229
|
+
try:
|
|
2230
|
+
_refuse(plan)
|
|
2231
|
+
except UnitFlowError as exc:
|
|
2232
|
+
raise ContractError(exc.path, exc.code, exc.detail) from None
|
|
2233
|
+
|
|
2234
|
+
|
|
2235
|
+
def parse_fresh_plan_document(value: Any) -> TablePlan | GraphTablePlan:
|
|
2236
|
+
"""Parse a plan entering the authoring lifecycle, and hold it to the checks age decides.
|
|
2237
|
+
|
|
2238
|
+
Three checks, in this order. The first reads retired names off the raw document so retirement
|
|
2239
|
+
is what the author is told; the second holds for a caller that passes an already-parsed plan
|
|
2240
|
+
through :func:`refuse_retired_operations` instead; the third is the two unit gates, which read
|
|
2241
|
+
a declaration and the arithmetic on it. All three are refusals for newly authored bytes and for
|
|
2242
|
+
nothing else.
|
|
2243
|
+
"""
|
|
2244
|
+
|
|
2245
|
+
refuse_retired_operation_names(value)
|
|
2246
|
+
plan = parse_plan_document(value)
|
|
2247
|
+
refuse_retired_operations(plan)
|
|
2248
|
+
refuse_inconsistent_units(plan)
|
|
2249
|
+
return plan
|
|
2250
|
+
|
|
2251
|
+
|
|
2252
|
+
def _parse_source(value: Any, path: str) -> SourceSpec:
|
|
2253
|
+
data = _object(value, path)
|
|
2254
|
+
_exact_fields(
|
|
2255
|
+
data,
|
|
2256
|
+
{"id", "path", "format", "origin", "acquisition_method", "rights"},
|
|
2257
|
+
path,
|
|
2258
|
+
)
|
|
2259
|
+
rights_path = f"{path}.rights"
|
|
2260
|
+
rights_data = _object(data["rights"], rights_path)
|
|
2261
|
+
_exact_fields(rights_data, {"status", "evidence", "permissions"}, rights_path)
|
|
2262
|
+
try:
|
|
2263
|
+
rights = RightsSpec(
|
|
2264
|
+
status=_required_text(
|
|
2265
|
+
rights_data["status"],
|
|
2266
|
+
f"{rights_path}.status",
|
|
2267
|
+
maximum=64,
|
|
2268
|
+
),
|
|
2269
|
+
evidence=_required_text(
|
|
2270
|
+
rights_data["evidence"],
|
|
2271
|
+
f"{rights_path}.evidence",
|
|
2272
|
+
maximum=MAX_TEXT_LENGTH,
|
|
2273
|
+
),
|
|
2274
|
+
permissions=_parse_choice_array(
|
|
2275
|
+
rights_data["permissions"],
|
|
2276
|
+
f"{rights_path}.permissions",
|
|
2277
|
+
_PERMISSIONS,
|
|
2278
|
+
nonempty=True,
|
|
2279
|
+
maximum=len(_PERMISSIONS),
|
|
2280
|
+
),
|
|
2281
|
+
)
|
|
2282
|
+
except ContractError as exc:
|
|
2283
|
+
if exc.path.startswith("rights."):
|
|
2284
|
+
suffix = exc.path.removeprefix("rights.")
|
|
2285
|
+
raise ContractError(f"{rights_path}.{suffix}", exc.code, exc.detail) from None
|
|
2286
|
+
raise
|
|
2287
|
+
try:
|
|
2288
|
+
return SourceSpec(
|
|
2289
|
+
source_id=_required_text(
|
|
2290
|
+
data["id"],
|
|
2291
|
+
f"{path}.id",
|
|
2292
|
+
maximum=MAX_IDENTIFIER_LENGTH,
|
|
2293
|
+
),
|
|
2294
|
+
path=_required_text(data["path"], f"{path}.path", maximum=MAX_PATH_LENGTH),
|
|
2295
|
+
format=_required_text(data["format"], f"{path}.format", maximum=16),
|
|
2296
|
+
origin=_required_text(data["origin"], f"{path}.origin", maximum=4_000),
|
|
2297
|
+
acquisition_method=_required_text(
|
|
2298
|
+
data["acquisition_method"],
|
|
2299
|
+
f"{path}.acquisition_method",
|
|
2300
|
+
maximum=64,
|
|
2301
|
+
),
|
|
2302
|
+
rights=rights,
|
|
2303
|
+
)
|
|
2304
|
+
except ContractError as exc:
|
|
2305
|
+
if exc.path.startswith("source."):
|
|
2306
|
+
suffix = exc.path.removeprefix("source.")
|
|
2307
|
+
raise ContractError(f"{path}.{suffix}", exc.code, exc.detail) from None
|
|
2308
|
+
raise
|
|
2309
|
+
|
|
2310
|
+
|
|
2311
|
+
def _parse_cleaning(
|
|
2312
|
+
value: Any,
|
|
2313
|
+
path: str,
|
|
2314
|
+
source_ids: frozenset[str],
|
|
2315
|
+
) -> CleaningStep:
|
|
2316
|
+
data = _object(value, path)
|
|
2317
|
+
_exact_fields(data, {"source", "operation", "columns"}, path)
|
|
2318
|
+
source_id = _required_text(
|
|
2319
|
+
data["source"],
|
|
2320
|
+
f"{path}.source",
|
|
2321
|
+
maximum=MAX_IDENTIFIER_LENGTH,
|
|
2322
|
+
)
|
|
2323
|
+
if source_id not in source_ids:
|
|
2324
|
+
raise ContractError(
|
|
2325
|
+
f"{path}.source",
|
|
2326
|
+
"UNKNOWN_SOURCE",
|
|
2327
|
+
f"must be one of {sorted(source_ids)}",
|
|
2328
|
+
)
|
|
2329
|
+
operation = _required_text(
|
|
2330
|
+
data["operation"],
|
|
2331
|
+
f"{path}.operation",
|
|
2332
|
+
maximum=64,
|
|
2333
|
+
)
|
|
2334
|
+
_choice(operation, _CLEANING_OPERATIONS, f"{path}.operation")
|
|
2335
|
+
if operation in {"trim", "empty_to_null"}:
|
|
2336
|
+
columns: tuple[str, ...] | tuple[tuple[str, str], ...] = _parse_column_array(
|
|
2337
|
+
data["columns"],
|
|
2338
|
+
f"{path}.columns",
|
|
2339
|
+
nonempty=True,
|
|
2340
|
+
maximum=MAX_COLUMN_COUNT,
|
|
2341
|
+
)
|
|
2342
|
+
else:
|
|
2343
|
+
mapping = _object(data["columns"], f"{path}.columns")
|
|
2344
|
+
if not mapping:
|
|
2345
|
+
raise ContractError(
|
|
2346
|
+
f"{path}.columns",
|
|
2347
|
+
"EMPTY_COLLECTION",
|
|
2348
|
+
"cannot be empty",
|
|
2349
|
+
)
|
|
2350
|
+
if len(mapping) > MAX_COLUMN_COUNT:
|
|
2351
|
+
raise ContractError(
|
|
2352
|
+
f"{path}.columns",
|
|
2353
|
+
"COLLECTION_LIMIT",
|
|
2354
|
+
f"exceeds {MAX_COLUMN_COUNT} entries",
|
|
2355
|
+
)
|
|
2356
|
+
pairs: list[tuple[str, str]] = []
|
|
2357
|
+
for key, item in mapping.items():
|
|
2358
|
+
source_column = _column_name(key, f"{path}.columns.<key>")
|
|
2359
|
+
target = _required_text(
|
|
2360
|
+
item,
|
|
2361
|
+
f"{path}.columns[{key!r}]",
|
|
2362
|
+
maximum=MAX_IDENTIFIER_LENGTH,
|
|
2363
|
+
)
|
|
2364
|
+
if operation == "rename":
|
|
2365
|
+
_column_name(target, f"{path}.columns[{key!r}]")
|
|
2366
|
+
else:
|
|
2367
|
+
_choice(target, _CAST_TYPES, f"{path}.columns[{key!r}]")
|
|
2368
|
+
pairs.append((source_column, target))
|
|
2369
|
+
columns = tuple(pairs)
|
|
2370
|
+
try:
|
|
2371
|
+
return CleaningStep(source_id=source_id, operation=operation, columns=columns)
|
|
2372
|
+
except ContractError as exc:
|
|
2373
|
+
if exc.path.startswith("cleaning."):
|
|
2374
|
+
suffix = exc.path.removeprefix("cleaning.")
|
|
2375
|
+
raise ContractError(f"{path}.{suffix}", exc.code, exc.detail) from None
|
|
2376
|
+
raise
|
|
2377
|
+
|
|
2378
|
+
|
|
2379
|
+
def _parse_join(value: Any, path: str, source_ids: frozenset[str]) -> JoinSpec:
|
|
2380
|
+
data = _object(value, path)
|
|
2381
|
+
_exact_fields(data, {"left", "right", "on", "kind", "cardinality"}, path)
|
|
2382
|
+
left = _required_text(data["left"], f"{path}.left", maximum=MAX_IDENTIFIER_LENGTH)
|
|
2383
|
+
right = _required_text(data["right"], f"{path}.right", maximum=MAX_IDENTIFIER_LENGTH)
|
|
2384
|
+
if left not in source_ids:
|
|
2385
|
+
raise ContractError(
|
|
2386
|
+
f"{path}.left",
|
|
2387
|
+
"UNKNOWN_SOURCE",
|
|
2388
|
+
f"must be one of {sorted(source_ids)}",
|
|
2389
|
+
)
|
|
2390
|
+
if right not in source_ids:
|
|
2391
|
+
raise ContractError(
|
|
2392
|
+
f"{path}.right",
|
|
2393
|
+
"UNKNOWN_SOURCE",
|
|
2394
|
+
f"must be one of {sorted(source_ids)}",
|
|
2395
|
+
)
|
|
2396
|
+
try:
|
|
2397
|
+
return JoinSpec(
|
|
2398
|
+
left=left,
|
|
2399
|
+
right=right,
|
|
2400
|
+
on=_parse_column_array(
|
|
2401
|
+
data["on"],
|
|
2402
|
+
f"{path}.on",
|
|
2403
|
+
nonempty=True,
|
|
2404
|
+
maximum=16,
|
|
2405
|
+
),
|
|
2406
|
+
kind=_required_text(data["kind"], f"{path}.kind", maximum=32),
|
|
2407
|
+
cardinality=_required_text(
|
|
2408
|
+
data["cardinality"],
|
|
2409
|
+
f"{path}.cardinality",
|
|
2410
|
+
maximum=32,
|
|
2411
|
+
),
|
|
2412
|
+
)
|
|
2413
|
+
except ContractError as exc:
|
|
2414
|
+
if exc.path.startswith("join."):
|
|
2415
|
+
suffix = exc.path.removeprefix("join.")
|
|
2416
|
+
raise ContractError(f"{path}.{suffix}", exc.code, exc.detail) from None
|
|
2417
|
+
raise
|
|
2418
|
+
|
|
2419
|
+
|
|
2420
|
+
def _parse_quality(value: Any, path: str) -> QualitySpec:
|
|
2421
|
+
data = _object(value, path)
|
|
2422
|
+
_exact_fields(data, {"min_rows", "not_null"}, path)
|
|
2423
|
+
try:
|
|
2424
|
+
return QualitySpec(
|
|
2425
|
+
min_rows=_integer(data["min_rows"], f"{path}.min_rows"),
|
|
2426
|
+
not_null=_parse_column_array(
|
|
2427
|
+
data["not_null"],
|
|
2428
|
+
f"{path}.not_null",
|
|
2429
|
+
nonempty=False,
|
|
2430
|
+
maximum=MAX_COLUMN_COUNT,
|
|
2431
|
+
),
|
|
2432
|
+
)
|
|
2433
|
+
except ContractError as exc:
|
|
2434
|
+
if exc.path.startswith("quality."):
|
|
2435
|
+
suffix = exc.path.removeprefix("quality.")
|
|
2436
|
+
raise ContractError(f"{path}.{suffix}", exc.code, exc.detail) from None
|
|
2437
|
+
raise
|
|
2438
|
+
|
|
2439
|
+
|
|
2440
|
+
def _object(value: Any, path: str) -> Mapping[str, Any]:
|
|
2441
|
+
if not isinstance(value, Mapping) or any(not isinstance(key, str) for key in value):
|
|
2442
|
+
raise ContractError(path, "TYPE", "must be an object with string keys")
|
|
2443
|
+
return value
|
|
2444
|
+
|
|
2445
|
+
|
|
2446
|
+
def _array(
|
|
2447
|
+
value: Any,
|
|
2448
|
+
path: str,
|
|
2449
|
+
*,
|
|
2450
|
+
nonempty: bool,
|
|
2451
|
+
maximum: int,
|
|
2452
|
+
minimum: int = 0,
|
|
2453
|
+
) -> Sequence[Any]:
|
|
2454
|
+
if not isinstance(value, list):
|
|
2455
|
+
raise ContractError(path, "TYPE", "must be an array")
|
|
2456
|
+
if nonempty and not value:
|
|
2457
|
+
raise ContractError(path, "EMPTY_COLLECTION", "cannot be empty")
|
|
2458
|
+
if len(value) < minimum:
|
|
2459
|
+
raise ContractError(path, "COLLECTION_MINIMUM", f"must contain at least {minimum} entries")
|
|
2460
|
+
if len(value) > maximum:
|
|
2461
|
+
raise ContractError(path, "COLLECTION_LIMIT", f"exceeds {maximum} entries")
|
|
2462
|
+
return value
|
|
2463
|
+
|
|
2464
|
+
|
|
2465
|
+
def _exact_fields(data: Mapping[str, Any], expected: Iterable[str], path: str) -> None:
|
|
2466
|
+
expected_set = frozenset(expected)
|
|
2467
|
+
missing = sorted(expected_set - data.keys())
|
|
2468
|
+
extra = sorted(data.keys() - expected_set)
|
|
2469
|
+
if missing or extra:
|
|
2470
|
+
raise ContractError(
|
|
2471
|
+
path,
|
|
2472
|
+
"FIELDS",
|
|
2473
|
+
f"fields invalid; missing={missing}, unknown={extra}",
|
|
2474
|
+
)
|
|
2475
|
+
|
|
2476
|
+
|
|
2477
|
+
def _required_text(value: Any, path: str, *, maximum: int) -> str:
|
|
2478
|
+
if not isinstance(value, str):
|
|
2479
|
+
raise ContractError(path, "TYPE", "must be a string")
|
|
2480
|
+
_text(value, path, minimum=1, maximum=maximum)
|
|
2481
|
+
return value
|
|
2482
|
+
|
|
2483
|
+
|
|
2484
|
+
def _text(value: Any, path: str, *, minimum: int, maximum: int) -> str:
|
|
2485
|
+
if not isinstance(value, str):
|
|
2486
|
+
raise ContractError(path, "TYPE", "must be a string")
|
|
2487
|
+
if value != value.strip():
|
|
2488
|
+
raise ContractError(path, "NONCANONICAL_TEXT", "must not have surrounding whitespace")
|
|
2489
|
+
if len(value) < minimum:
|
|
2490
|
+
raise ContractError(path, "STRING_MINIMUM", f"must contain at least {minimum} characters")
|
|
2491
|
+
if len(value) > maximum:
|
|
2492
|
+
raise ContractError(path, "STRING_LIMIT", f"exceeds {maximum} characters")
|
|
2493
|
+
for char in value:
|
|
2494
|
+
if ord(char) < 0x20 and char not in {"\t", "\n", "\r"}:
|
|
2495
|
+
raise ContractError(path, "CONTROL_CHARACTER", "contains a control character")
|
|
2496
|
+
return value
|
|
2497
|
+
|
|
2498
|
+
|
|
2499
|
+
def _identifier(value: Any, path: str) -> str:
|
|
2500
|
+
text = _text(
|
|
2501
|
+
value,
|
|
2502
|
+
path,
|
|
2503
|
+
minimum=1,
|
|
2504
|
+
maximum=MAX_IDENTIFIER_LENGTH,
|
|
2505
|
+
)
|
|
2506
|
+
if not _IDENTIFIER.fullmatch(text):
|
|
2507
|
+
raise ContractError(
|
|
2508
|
+
path,
|
|
2509
|
+
"IDENTIFIER",
|
|
2510
|
+
"must be a canonical lowercase identifier",
|
|
2511
|
+
)
|
|
2512
|
+
return text
|
|
2513
|
+
|
|
2514
|
+
|
|
2515
|
+
def _column_name(value: Any, path: str) -> str:
|
|
2516
|
+
text = _text(
|
|
2517
|
+
value,
|
|
2518
|
+
path,
|
|
2519
|
+
minimum=1,
|
|
2520
|
+
maximum=MAX_IDENTIFIER_LENGTH,
|
|
2521
|
+
)
|
|
2522
|
+
if not _COLUMN.fullmatch(text):
|
|
2523
|
+
raise ContractError(
|
|
2524
|
+
path,
|
|
2525
|
+
"COLUMN_IDENTIFIER",
|
|
2526
|
+
"must be a canonical lowercase snake-case column name",
|
|
2527
|
+
)
|
|
2528
|
+
return text
|
|
2529
|
+
|
|
2530
|
+
|
|
2531
|
+
def _version(value: Any, expected: str, path: str) -> str:
|
|
2532
|
+
text = _text(value, path, minimum=1, maximum=64)
|
|
2533
|
+
if text != expected:
|
|
2534
|
+
raise ContractError(path, "VERSION", f"must be {expected!r}")
|
|
2535
|
+
return text
|
|
2536
|
+
|
|
2537
|
+
|
|
2538
|
+
def _choice_version(value: Any, accepted: tuple[str, ...], path: str) -> str:
|
|
2539
|
+
"""Pin a schema version to one of the generations still read."""
|
|
2540
|
+
|
|
2541
|
+
text = _text(value, path, minimum=1, maximum=64)
|
|
2542
|
+
if text not in accepted:
|
|
2543
|
+
raise ContractError(path, "VERSION", f"must be one of {list(accepted)}")
|
|
2544
|
+
return text
|
|
2545
|
+
|
|
2546
|
+
|
|
2547
|
+
def _choice(value: Any, choices: frozenset[str], path: str) -> str:
|
|
2548
|
+
if not isinstance(value, str) or value not in choices:
|
|
2549
|
+
raise ContractError(path, "ENUM", f"must be one of {sorted(choices)}")
|
|
2550
|
+
return value
|
|
2551
|
+
|
|
2552
|
+
|
|
2553
|
+
def _declared_unit(value: Any, path: str) -> Unit | None:
|
|
2554
|
+
"""Accept ``none`` or one unit code the grammar resolves, and refuse anything else by name.
|
|
2555
|
+
|
|
2556
|
+
Returning the resolved unit rather than the string is what lets the rules above ask what the
|
|
2557
|
+
code means instead of how it was spelled. ``none`` resolves to nothing on purpose: it is the
|
|
2558
|
+
absence of a unit, not a dimensionless one.
|
|
2559
|
+
"""
|
|
2560
|
+
|
|
2561
|
+
text = _text(value, path, minimum=1, maximum=64)
|
|
2562
|
+
if text == NO_UNIT:
|
|
2563
|
+
return None
|
|
2564
|
+
try:
|
|
2565
|
+
return resolve_declared_unit(text)
|
|
2566
|
+
except UnitError as exc:
|
|
2567
|
+
raise ContractError(
|
|
2568
|
+
path, "UNIT_CODE", f"{text!r} is not a unit code: {exc.detail}"
|
|
2569
|
+
) from None
|
|
2570
|
+
|
|
2571
|
+
|
|
2572
|
+
def _integer(value: Any, path: str) -> int:
|
|
2573
|
+
if type(value) is not int:
|
|
2574
|
+
raise ContractError(path, "INTEGER", "must be an integer (booleans are not integers)")
|
|
2575
|
+
return value
|
|
2576
|
+
|
|
2577
|
+
|
|
2578
|
+
def _bounded_int(value: Any, path: str, *, minimum: int, maximum: int) -> int:
|
|
2579
|
+
integer = _integer(value, path)
|
|
2580
|
+
if integer < minimum or integer > maximum:
|
|
2581
|
+
raise ContractError(
|
|
2582
|
+
path,
|
|
2583
|
+
"INTEGER_RANGE",
|
|
2584
|
+
f"must be between {minimum} and {maximum}",
|
|
2585
|
+
)
|
|
2586
|
+
return integer
|
|
2587
|
+
|
|
2588
|
+
|
|
2589
|
+
def _utc_timestamp(value: Any, path: str) -> datetime:
|
|
2590
|
+
text = _text(value, path, minimum=1, maximum=32)
|
|
2591
|
+
if not _UTC_TIMESTAMP.fullmatch(text):
|
|
2592
|
+
raise ContractError(
|
|
2593
|
+
path,
|
|
2594
|
+
"UTC_TIMESTAMP",
|
|
2595
|
+
"must be a zero-padded RFC 3339 UTC timestamp ending in Z",
|
|
2596
|
+
)
|
|
2597
|
+
try:
|
|
2598
|
+
parsed = datetime.fromisoformat(text.removesuffix("Z") + "+00:00")
|
|
2599
|
+
except ValueError:
|
|
2600
|
+
raise ContractError(path, "UTC_TIMESTAMP", "must be a valid UTC timestamp") from None
|
|
2601
|
+
if parsed.utcoffset() is None or parsed.utcoffset().total_seconds() != 0:
|
|
2602
|
+
raise ContractError(path, "UTC_TIMESTAMP", "must use UTC Z")
|
|
2603
|
+
return parsed
|
|
2604
|
+
|
|
2605
|
+
|
|
2606
|
+
def _sha256(value: Any, path: str) -> str:
|
|
2607
|
+
if not isinstance(value, str) or not _SHA256.fullmatch(value):
|
|
2608
|
+
raise ContractError(path, "SHA256", "must be a lowercase 64-character SHA-256 digest")
|
|
2609
|
+
return value
|
|
2610
|
+
|
|
2611
|
+
|
|
2612
|
+
def _relative_posix_path(value: Any, path: str) -> str:
|
|
2613
|
+
text = _text(value, path, minimum=1, maximum=MAX_PATH_LENGTH)
|
|
2614
|
+
pure = PurePosixPath(text)
|
|
2615
|
+
if (
|
|
2616
|
+
"\\" in text
|
|
2617
|
+
or "\x00" in text
|
|
2618
|
+
or pure.is_absolute()
|
|
2619
|
+
or not pure.parts
|
|
2620
|
+
or any(part in {"", ".", ".."} for part in pure.parts)
|
|
2621
|
+
or pure.as_posix() != text
|
|
2622
|
+
or text.endswith("/")
|
|
2623
|
+
or (pure.parts and ":" in pure.parts[0])
|
|
2624
|
+
):
|
|
2625
|
+
raise ContractError(
|
|
2626
|
+
path,
|
|
2627
|
+
"POSIX_PATH",
|
|
2628
|
+
"must be canonical normalized relative POSIX text",
|
|
2629
|
+
)
|
|
2630
|
+
return text
|
|
2631
|
+
|
|
2632
|
+
|
|
2633
|
+
def _decoded_query_component(value: str) -> str:
|
|
2634
|
+
"""Bound repeated decoding so concealed credential keys cannot enter durable evidence."""
|
|
2635
|
+
|
|
2636
|
+
decoded = value
|
|
2637
|
+
for _ in range(3):
|
|
2638
|
+
replacement = unquote_plus(decoded)
|
|
2639
|
+
if replacement == decoded:
|
|
2640
|
+
break
|
|
2641
|
+
decoded = replacement
|
|
2642
|
+
return decoded
|
|
2643
|
+
|
|
2644
|
+
|
|
2645
|
+
def validate_source_locator(value: Any, kind: str, path: str) -> str:
|
|
2646
|
+
"""Validate a durable source locator without admitting credentials or signed URLs."""
|
|
2647
|
+
|
|
2648
|
+
text = _text(value, path, minimum=1, maximum=MAX_PATH_LENGTH)
|
|
2649
|
+
if kind == "relative_path":
|
|
2650
|
+
return _relative_posix_path(text, path)
|
|
2651
|
+
if kind == "artifact_reference":
|
|
2652
|
+
if not text.startswith("artifact://") or len(text) <= len("artifact://"):
|
|
2653
|
+
raise ContractError(path, "ARTIFACT_REFERENCE", "must be an artifact:// reference")
|
|
2654
|
+
if any(char.isspace() for char in text):
|
|
2655
|
+
raise ContractError(path, "ARTIFACT_REFERENCE", "must not contain whitespace")
|
|
2656
|
+
try:
|
|
2657
|
+
parsed_artifact = urlsplit(text)
|
|
2658
|
+
except ValueError as error:
|
|
2659
|
+
raise ContractError(
|
|
2660
|
+
path, "ARTIFACT_REFERENCE", "must be a valid artifact:// reference"
|
|
2661
|
+
) from error
|
|
2662
|
+
if (
|
|
2663
|
+
parsed_artifact.username is not None
|
|
2664
|
+
or parsed_artifact.password is not None
|
|
2665
|
+
or parsed_artifact.fragment
|
|
2666
|
+
):
|
|
2667
|
+
raise ContractError(
|
|
2668
|
+
path,
|
|
2669
|
+
"ARTIFACT_REFERENCE",
|
|
2670
|
+
"must not contain embedded credentials or a fragment",
|
|
2671
|
+
)
|
|
2672
|
+
for query_key, query_value in parse_qsl(parsed_artifact.query, keep_blank_values=True):
|
|
2673
|
+
if _SENSITIVE_LOCATOR_QUERY_KEY.search(
|
|
2674
|
+
_decoded_query_component(query_key)
|
|
2675
|
+
) or _SECRET_LOCATOR_VALUE.search(_decoded_query_component(query_value)):
|
|
2676
|
+
raise ContractError(
|
|
2677
|
+
path,
|
|
2678
|
+
"ARTIFACT_REFERENCE",
|
|
2679
|
+
"must not contain credentials or signed/secret query material",
|
|
2680
|
+
)
|
|
2681
|
+
return text
|
|
2682
|
+
try:
|
|
2683
|
+
parsed = urlsplit(text)
|
|
2684
|
+
except ValueError as error:
|
|
2685
|
+
raise ContractError(path, "HTTPS_URL", "must be a valid canonical HTTPS URL") from error
|
|
2686
|
+
try:
|
|
2687
|
+
port = parsed.port
|
|
2688
|
+
except ValueError:
|
|
2689
|
+
port = None
|
|
2690
|
+
invalid_port = True
|
|
2691
|
+
else:
|
|
2692
|
+
invalid_port = False
|
|
2693
|
+
if (
|
|
2694
|
+
parsed.scheme != "https"
|
|
2695
|
+
or not parsed.hostname
|
|
2696
|
+
or parsed.username is not None
|
|
2697
|
+
or parsed.password is not None
|
|
2698
|
+
or parsed.fragment
|
|
2699
|
+
or parsed.hostname != parsed.hostname.lower()
|
|
2700
|
+
or invalid_port
|
|
2701
|
+
or port == 443
|
|
2702
|
+
):
|
|
2703
|
+
raise ContractError(
|
|
2704
|
+
path,
|
|
2705
|
+
"HTTPS_URL",
|
|
2706
|
+
"must be a credential-free canonical HTTPS URL without a fragment",
|
|
2707
|
+
)
|
|
2708
|
+
for query_key, query_value in parse_qsl(parsed.query, keep_blank_values=True):
|
|
2709
|
+
decoded_key = _decoded_query_component(query_key)
|
|
2710
|
+
decoded_value = _decoded_query_component(query_value)
|
|
2711
|
+
if _SENSITIVE_LOCATOR_QUERY_KEY.search(decoded_key) or _SECRET_LOCATOR_VALUE.search(
|
|
2712
|
+
decoded_value
|
|
2713
|
+
):
|
|
2714
|
+
raise ContractError(
|
|
2715
|
+
path,
|
|
2716
|
+
"HTTPS_URL",
|
|
2717
|
+
"must not contain credentials or signed/secret query material",
|
|
2718
|
+
)
|
|
2719
|
+
return text
|
|
2720
|
+
|
|
2721
|
+
|
|
2722
|
+
def source_locator_identity(kind: str, locator: str) -> tuple[str, str]:
|
|
2723
|
+
"""Return a stable identity for already-validated proposal locators."""
|
|
2724
|
+
|
|
2725
|
+
validate_source_locator(locator, kind, "source_proposal.locator")
|
|
2726
|
+
if kind != "https_url":
|
|
2727
|
+
return kind, locator
|
|
2728
|
+
parsed = urlsplit(locator)
|
|
2729
|
+
normalized = urlunsplit((parsed.scheme, parsed.netloc, parsed.path or "/", parsed.query, ""))
|
|
2730
|
+
return kind, normalized
|
|
2731
|
+
|
|
2732
|
+
|
|
2733
|
+
def _evidence_uri(value: Any, path: str) -> str:
|
|
2734
|
+
text = _text(value, path, minimum=1, maximum=2_048)
|
|
2735
|
+
parsed = urlsplit(text)
|
|
2736
|
+
if parsed.scheme not in {"https", "fixture"} or not parsed.netloc:
|
|
2737
|
+
raise ContractError(path, "EVIDENCE_URI", "must be an https:// or fixture:// URI")
|
|
2738
|
+
if parsed.username is not None or parsed.password is not None:
|
|
2739
|
+
raise ContractError(path, "EVIDENCE_URI", "must not embed credentials")
|
|
2740
|
+
return text
|
|
2741
|
+
|
|
2742
|
+
|
|
2743
|
+
def _require_unique(values: Iterable[Any], path: str) -> None:
|
|
2744
|
+
items = tuple(values)
|
|
2745
|
+
if len(items) != len(set(items)):
|
|
2746
|
+
raise ContractError(path, "DUPLICATE", "must contain unique entries")
|
|
2747
|
+
|
|
2748
|
+
|
|
2749
|
+
def _column_tuple(
|
|
2750
|
+
values: Any,
|
|
2751
|
+
path: str,
|
|
2752
|
+
*,
|
|
2753
|
+
nonempty: bool,
|
|
2754
|
+
maximum: int,
|
|
2755
|
+
) -> tuple[str, ...]:
|
|
2756
|
+
if not isinstance(values, tuple):
|
|
2757
|
+
raise ContractError(path, "TYPE", "must be an immutable tuple")
|
|
2758
|
+
if nonempty and not values:
|
|
2759
|
+
raise ContractError(path, "EMPTY_COLLECTION", "cannot be empty")
|
|
2760
|
+
if len(values) > maximum:
|
|
2761
|
+
raise ContractError(path, "COLLECTION_LIMIT", f"exceeds {maximum} entries")
|
|
2762
|
+
for index, item in enumerate(values):
|
|
2763
|
+
_column_name(item, f"{path}[{index}]")
|
|
2764
|
+
_require_unique(values, path)
|
|
2765
|
+
return values
|
|
2766
|
+
|
|
2767
|
+
|
|
2768
|
+
def _text_tuple(
|
|
2769
|
+
values: Any,
|
|
2770
|
+
path: str,
|
|
2771
|
+
*,
|
|
2772
|
+
nonempty: bool,
|
|
2773
|
+
maximum: int,
|
|
2774
|
+
item_maximum: int,
|
|
2775
|
+
) -> tuple[str, ...]:
|
|
2776
|
+
if not isinstance(values, tuple):
|
|
2777
|
+
raise ContractError(path, "TYPE", "must be an immutable tuple")
|
|
2778
|
+
if nonempty and not values:
|
|
2779
|
+
raise ContractError(path, "EMPTY_COLLECTION", "cannot be empty")
|
|
2780
|
+
if len(values) > maximum:
|
|
2781
|
+
raise ContractError(path, "COLLECTION_LIMIT", f"exceeds {maximum} entries")
|
|
2782
|
+
for index, item in enumerate(values):
|
|
2783
|
+
_text(item, f"{path}[{index}]", minimum=1, maximum=item_maximum)
|
|
2784
|
+
_require_unique(values, path)
|
|
2785
|
+
return values
|
|
2786
|
+
|
|
2787
|
+
|
|
2788
|
+
def _choice_tuple(
|
|
2789
|
+
values: Any,
|
|
2790
|
+
choices: frozenset[str],
|
|
2791
|
+
path: str,
|
|
2792
|
+
*,
|
|
2793
|
+
maximum: int,
|
|
2794
|
+
nonempty: bool = False,
|
|
2795
|
+
) -> tuple[str, ...]:
|
|
2796
|
+
if not isinstance(values, tuple):
|
|
2797
|
+
raise ContractError(path, "TYPE", "must be an immutable tuple")
|
|
2798
|
+
if nonempty and not values:
|
|
2799
|
+
raise ContractError(path, "EMPTY_COLLECTION", "cannot be empty")
|
|
2800
|
+
if len(values) > maximum:
|
|
2801
|
+
raise ContractError(path, "COLLECTION_LIMIT", f"exceeds {maximum} entries")
|
|
2802
|
+
for index, item in enumerate(values):
|
|
2803
|
+
_choice(item, choices, f"{path}[{index}]")
|
|
2804
|
+
_require_unique(values, path)
|
|
2805
|
+
return values
|
|
2806
|
+
|
|
2807
|
+
|
|
2808
|
+
def _typed_tuple(
|
|
2809
|
+
values: Any,
|
|
2810
|
+
expected_type: type[Any],
|
|
2811
|
+
path: str,
|
|
2812
|
+
*,
|
|
2813
|
+
nonempty: bool,
|
|
2814
|
+
maximum: int,
|
|
2815
|
+
minimum: int = 0,
|
|
2816
|
+
) -> tuple[Any, ...]:
|
|
2817
|
+
if not isinstance(values, tuple):
|
|
2818
|
+
raise ContractError(path, "TYPE", "must be an immutable tuple")
|
|
2819
|
+
if nonempty and not values:
|
|
2820
|
+
raise ContractError(path, "EMPTY_COLLECTION", "cannot be empty")
|
|
2821
|
+
if len(values) < minimum:
|
|
2822
|
+
raise ContractError(path, "COLLECTION_MINIMUM", f"must contain at least {minimum} entries")
|
|
2823
|
+
if len(values) > maximum:
|
|
2824
|
+
raise ContractError(path, "COLLECTION_LIMIT", f"exceeds {maximum} entries")
|
|
2825
|
+
for index, item in enumerate(values):
|
|
2826
|
+
if not isinstance(item, expected_type):
|
|
2827
|
+
raise ContractError(
|
|
2828
|
+
f"{path}[{index}]",
|
|
2829
|
+
"TYPE",
|
|
2830
|
+
f"must be {expected_type.__name__}",
|
|
2831
|
+
)
|
|
2832
|
+
return values
|
|
2833
|
+
|
|
2834
|
+
|
|
2835
|
+
def _parse_column_array(
|
|
2836
|
+
value: Any,
|
|
2837
|
+
path: str,
|
|
2838
|
+
*,
|
|
2839
|
+
nonempty: bool,
|
|
2840
|
+
maximum: int,
|
|
2841
|
+
) -> tuple[str, ...]:
|
|
2842
|
+
items = _array(value, path, nonempty=nonempty, maximum=maximum)
|
|
2843
|
+
result = tuple(_column_name(item, f"{path}[{index}]") for index, item in enumerate(items))
|
|
2844
|
+
_require_unique(result, path)
|
|
2845
|
+
return result
|
|
2846
|
+
|
|
2847
|
+
|
|
2848
|
+
def _parse_text_array(
|
|
2849
|
+
value: Any,
|
|
2850
|
+
path: str,
|
|
2851
|
+
*,
|
|
2852
|
+
nonempty: bool,
|
|
2853
|
+
maximum: int,
|
|
2854
|
+
item_maximum: int,
|
|
2855
|
+
) -> tuple[str, ...]:
|
|
2856
|
+
items = _array(value, path, nonempty=nonempty, maximum=maximum)
|
|
2857
|
+
result = tuple(
|
|
2858
|
+
_required_text(item, f"{path}[{index}]", maximum=item_maximum)
|
|
2859
|
+
for index, item in enumerate(items)
|
|
2860
|
+
)
|
|
2861
|
+
_require_unique(result, path)
|
|
2862
|
+
return result
|
|
2863
|
+
|
|
2864
|
+
|
|
2865
|
+
def _parse_choice_array(
|
|
2866
|
+
value: Any,
|
|
2867
|
+
path: str,
|
|
2868
|
+
choices: frozenset[str],
|
|
2869
|
+
*,
|
|
2870
|
+
nonempty: bool,
|
|
2871
|
+
maximum: int,
|
|
2872
|
+
) -> tuple[str, ...]:
|
|
2873
|
+
items = _array(value, path, nonempty=nonempty, maximum=maximum)
|
|
2874
|
+
result: list[str] = []
|
|
2875
|
+
for index, item in enumerate(items):
|
|
2876
|
+
text = _required_text(item, f"{path}[{index}]", maximum=64)
|
|
2877
|
+
_choice(text, choices, f"{path}[{index}]")
|
|
2878
|
+
result.append(text)
|
|
2879
|
+
_require_unique(result, path)
|
|
2880
|
+
return tuple(result)
|