mostlyright-data 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mostlyright/data_harness/__init__.py +158 -0
- mostlyright/data_harness/acquisition/__init__.py +55 -0
- mostlyright/data_harness/acquisition/http.py +2773 -0
- mostlyright/data_harness/acquisition/parsing.py +809 -0
- mostlyright/data_harness/acquisition/ranges.py +495 -0
- mostlyright/data_harness/acquisition/result_download.py +360 -0
- mostlyright/data_harness/acquisition/retention_admission.py +248 -0
- mostlyright/data_harness/acquisition/sandbox.py +4888 -0
- mostlyright/data_harness/acquisition/url_policy.py +530 -0
- mostlyright/data_harness/agent_runtime.py +2743 -0
- mostlyright/data_harness/assets/logo-ink.svg +31 -0
- mostlyright/data_harness/backends/__init__.py +28 -0
- mostlyright/data_harness/backends/pandas_backend.py +350 -0
- mostlyright/data_harness/backends/polars_backend.py +366 -0
- mostlyright/data_harness/backends/protocol.py +124 -0
- mostlyright/data_harness/backends/reference.py +83 -0
- mostlyright/data_harness/backends/registry.py +55 -0
- mostlyright/data_harness/backends/restrictions.py +126 -0
- mostlyright/data_harness/canonical.py +333 -0
- mostlyright/data_harness/catalog_job.py +625 -0
- mostlyright/data_harness/cli.py +5398 -0
- mostlyright/data_harness/contracts.py +53 -0
- mostlyright/data_harness/coordinator.py +1307 -0
- mostlyright/data_harness/deploy.py +924 -0
- mostlyright/data_harness/deploy_target.py +312 -0
- mostlyright/data_harness/deployment_evidence.py +1067 -0
- mostlyright/data_harness/event_presentation.py +576 -0
- mostlyright/data_harness/events.py +2152 -0
- mostlyright/data_harness/fast_delimited.py +239 -0
- mostlyright/data_harness/fleet.py +237 -0
- mostlyright/data_harness/formats.py +236 -0
- mostlyright/data_harness/governors.py +1163 -0
- mostlyright/data_harness/hosted_bootstrap.py +972 -0
- mostlyright/data_harness/hosted_crawler.py +1115 -0
- mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
- mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
- mostlyright/data_harness/hosted_crawler_job.py +1277 -0
- mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
- mostlyright/data_harness/hosted_dataset.py +1500 -0
- mostlyright/data_harness/hosted_deploy.py +3037 -0
- mostlyright/data_harness/hosted_handoff.py +62 -0
- mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
- mostlyright/data_harness/hosted_ingestion_job.py +356 -0
- mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
- mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
- mostlyright/data_harness/hosted_session_worker.py +3554 -0
- mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
- mostlyright/data_harness/hosted_worker.py +6784 -0
- mostlyright/data_harness/ingestion/__init__.py +56 -0
- mostlyright/data_harness/ingestion/contracts.py +461 -0
- mostlyright/data_harness/ingestion/faults.py +42 -0
- mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
- mostlyright/data_harness/ingestion/spool.py +130 -0
- mostlyright/data_harness/ingestion/store.py +885 -0
- mostlyright/data_harness/key_seam.py +434 -0
- mostlyright/data_harness/linux_process_boundary.py +262 -0
- mostlyright/data_harness/local_contracts.py +2880 -0
- mostlyright/data_harness/local_search/__init__.py +5 -0
- mostlyright/data_harness/local_search/build_index.py +1087 -0
- mostlyright/data_harness/local_search/contracts.py +920 -0
- mostlyright/data_harness/local_search/query_trace.py +266 -0
- mostlyright/data_harness/local_search/retrieval.py +700 -0
- mostlyright/data_harness/local_search/sealed.py +474 -0
- mostlyright/data_harness/local_search/service.py +784 -0
- mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
- mostlyright/data_harness/nbrender/__init__.py +12 -0
- mostlyright/data_harness/nbrender/chrome.py +359 -0
- mostlyright/data_harness/nbrender/code_body.py +266 -0
- mostlyright/data_harness/nbrender/document.py +407 -0
- mostlyright/data_harness/nbrender/frame.py +275 -0
- mostlyright/data_harness/nbrender/interactive.py +337 -0
- mostlyright/data_harness/nbrender/markdown_body.py +477 -0
- mostlyright/data_harness/nbrender/mr_components.py +134 -0
- mostlyright/data_harness/nbrender/outputs_data.py +595 -0
- mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
- mostlyright/data_harness/nbrender/outputs_source.py +260 -0
- mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
- mostlyright/data_harness/nbrender/outputs_text.py +400 -0
- mostlyright/data_harness/nbrender/parse.py +394 -0
- mostlyright/data_harness/nbrender/status.py +40 -0
- mostlyright/data_harness/nbrender/tokens.py +1295 -0
- mostlyright/data_harness/notebook.py +1710 -0
- mostlyright/data_harness/offline.py +2049 -0
- mostlyright/data_harness/operation_registry.py +1007 -0
- mostlyright/data_harness/operator_setup.py +239 -0
- mostlyright/data_harness/pipeline.py +6428 -0
- mostlyright/data_harness/plan_graph.py +2026 -0
- mostlyright/data_harness/preparation/__init__.py +104 -0
- mostlyright/data_harness/preparation/contracts.py +1017 -0
- mostlyright/data_harness/preparation/engine.py +221 -0
- mostlyright/data_harness/preparation/errors.py +14 -0
- mostlyright/data_harness/preparation/gates.py +751 -0
- mostlyright/data_harness/preparation/joins.py +574 -0
- mostlyright/data_harness/preparation/profile.py +384 -0
- mostlyright/data_harness/preparation/table.py +217 -0
- mostlyright/data_harness/preparation/transforms.py +568 -0
- mostlyright/data_harness/progress_events.py +534 -0
- mostlyright/data_harness/readers/__init__.py +46 -0
- mostlyright/data_harness/readers/containers.py +963 -0
- mostlyright/data_harness/readers/contracts.py +542 -0
- mostlyright/data_harness/readers/delimited.py +257 -0
- mostlyright/data_harness/readers/grib2/__init__.py +33 -0
- mostlyright/data_harness/readers/grib2/admission.py +722 -0
- mostlyright/data_harness/readers/grib2/decode.py +1009 -0
- mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
- mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
- mostlyright/data_harness/readers/json_tabular.py +485 -0
- mostlyright/data_harness/readers/registry.py +514 -0
- mostlyright/data_harness/readers/samples/README.md +110 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples.py +582 -0
- mostlyright/data_harness/readers/spreadsheet.py +803 -0
- mostlyright/data_harness/readers/tabular.py +510 -0
- mostlyright/data_harness/recipe.py +5321 -0
- mostlyright/data_harness/repair/__init__.py +78 -0
- mostlyright/data_harness/repair/adapters.py +274 -0
- mostlyright/data_harness/repair/contracts.py +872 -0
- mostlyright/data_harness/repair/coordinator.py +1099 -0
- mostlyright/data_harness/repair/errors.py +16 -0
- mostlyright/data_harness/review.py +2533 -0
- mostlyright/data_harness/rowset.py +283 -0
- mostlyright/data_harness/serving.py +1975 -0
- mostlyright/data_harness/serving_edge.py +590 -0
- mostlyright/data_harness/serving_http.py +1031 -0
- mostlyright/data_harness/session_probes.py +759 -0
- mostlyright/data_harness/signing.py +101 -0
- mostlyright/data_harness/source_discovery.py +898 -0
- mostlyright/data_harness/sources/__init__.py +209 -0
- mostlyright/data_harness/sources/_adapter_steps.py +213 -0
- mostlyright/data_harness/sources/adapters.py +1214 -0
- mostlyright/data_harness/sources/cadence.py +1428 -0
- mostlyright/data_harness/sources/cadence_emission.py +453 -0
- mostlyright/data_harness/sources/cadence_history.py +546 -0
- mostlyright/data_harness/sources/catalog/__init__.py +17 -0
- mostlyright/data_harness/sources/catalog/admission.py +477 -0
- mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
- mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
- mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
- mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
- mostlyright/data_harness/sources/catalog/channel.py +523 -0
- mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
- mostlyright/data_harness/sources/catalog/contracts.py +825 -0
- mostlyright/data_harness/sources/catalog/coverage.py +137 -0
- mostlyright/data_harness/sources/catalog/delta.py +1340 -0
- mostlyright/data_harness/sources/catalog/embedding.py +532 -0
- mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
- mostlyright/data_harness/sources/catalog/fill.py +3889 -0
- mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
- mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
- mostlyright/data_harness/sources/catalog/gating.py +374 -0
- mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
- mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
- mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
- mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
- mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
- mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
- mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
- mostlyright/data_harness/sources/catalog/health.py +447 -0
- mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
- mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
- mostlyright/data_harness/sources/catalog/neural.py +1618 -0
- mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
- mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
- mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
- mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
- mostlyright/data_harness/sources/catalog/recommend.py +171 -0
- mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
- mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
- mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
- mostlyright/data_harness/sources/catalog/sealed.py +560 -0
- mostlyright/data_harness/sources/catalog/search.py +230 -0
- mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
- mostlyright/data_harness/sources/catalog/update.py +891 -0
- mostlyright/data_harness/sources/collections.py +815 -0
- mostlyright/data_harness/sources/contracts.py +2223 -0
- mostlyright/data_harness/sources/deletion.py +761 -0
- mostlyright/data_harness/sources/fitness.py +162 -0
- mostlyright/data_harness/sources/governance.py +163 -0
- mostlyright/data_harness/sources/hosted.py +173 -0
- mostlyright/data_harness/sources/integration.py +218 -0
- mostlyright/data_harness/sources/range_reader.py +418 -0
- mostlyright/data_harness/sources/registry.py +514 -0
- mostlyright/data_harness/sources/rights_rule.py +59 -0
- mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
- mostlyright/data_harness/sources/sports.py +521 -0
- mostlyright/data_harness/sources/stream.py +524 -0
- mostlyright/data_harness/sources/stream_connector.py +418 -0
- mostlyright/data_harness/sources/stream_recorder.py +1404 -0
- mostlyright/data_harness/studio_boundary.py +2019 -0
- mostlyright/data_harness/thin/__init__.py +37 -0
- mostlyright/data_harness/thin/acquire.py +1137 -0
- mostlyright/data_harness/thin/acquire_cancel.py +579 -0
- mostlyright/data_harness/thin/approvals.py +617 -0
- mostlyright/data_harness/thin/commands.py +406 -0
- mostlyright/data_harness/thin/download.py +194 -0
- mostlyright/data_harness/thin/narrative.py +589 -0
- mostlyright/data_harness/thin/parity.py +1070 -0
- mostlyright/data_harness/thin/propose.py +2759 -0
- mostlyright/data_harness/thin/research.py +1663 -0
- mostlyright/data_harness/thin/router.py +924 -0
- mostlyright/data_harness/thin/runs.py +519 -0
- mostlyright/data_harness/thin/session.py +281 -0
- mostlyright/data_harness/thin/stream.py +501 -0
- mostlyright/data_harness/thin/transport.py +187 -0
- mostlyright/data_harness/thin/vocabulary.py +368 -0
- mostlyright/data_harness/thin/workers.py +164 -0
- mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
- mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
- mostlyright/data_harness/unit_flow.py +927 -0
- mostlyright/data_harness/units.py +572 -0
- mostlyright/data_harness/ux/__init__.py +9 -0
- mostlyright/data_harness/ux/approve.py +485 -0
- mostlyright/data_harness/ux/author_yaml.py +597 -0
- mostlyright/data_harness/ux/cloud_auth.py +447 -0
- mostlyright/data_harness/ux/commands/__init__.py +260 -0
- mostlyright/data_harness/ux/commands/approve.py +136 -0
- mostlyright/data_harness/ux/commands/auth.py +744 -0
- mostlyright/data_harness/ux/commands/author.py +79 -0
- mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
- mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
- mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
- mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
- mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
- mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
- mostlyright/data_harness/ux/commands/deploy.py +134 -0
- mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
- mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
- mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
- mostlyright/data_harness/ux/commands/diff.py +74 -0
- mostlyright/data_harness/ux/commands/index.py +84 -0
- mostlyright/data_harness/ux/commands/inventory.py +47 -0
- mostlyright/data_harness/ux/commands/list_builds.py +143 -0
- mostlyright/data_harness/ux/commands/login.py +63 -0
- mostlyright/data_harness/ux/commands/peek.py +236 -0
- mostlyright/data_harness/ux/commands/plan_check.py +90 -0
- mostlyright/data_harness/ux/commands/preflight.py +97 -0
- mostlyright/data_harness/ux/commands/record.py +107 -0
- mostlyright/data_harness/ux/commands/review_setup.py +47 -0
- mostlyright/data_harness/ux/commands/search.py +440 -0
- mostlyright/data_harness/ux/commands/show.py +61 -0
- mostlyright/data_harness/ux/commands/whoami.py +37 -0
- mostlyright/data_harness/ux/credential_native.py +551 -0
- mostlyright/data_harness/ux/credential_store.py +1055 -0
- mostlyright/data_harness/ux/credentials.py +631 -0
- mostlyright/data_harness/ux/diffing.py +444 -0
- mostlyright/data_harness/ux/headline.py +671 -0
- mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
- mostlyright/data_harness/ux/hosted_run_status.py +619 -0
- mostlyright/data_harness/ux/inventory.py +427 -0
- mostlyright/data_harness/ux/local_review.py +375 -0
- mostlyright/data_harness/ux/login.py +691 -0
- mostlyright/data_harness/ux/path_kind.py +147 -0
- mostlyright/data_harness/ux/peek.py +1000 -0
- mostlyright/data_harness/ux/plain_file.py +178 -0
- mostlyright/data_harness/ux/plan_check.py +311 -0
- mostlyright/data_harness/ux/preflight.py +918 -0
- mostlyright/data_harness/ux/readers.py +1124 -0
- mostlyright/data_harness/ux/remediation.py +2195 -0
- mostlyright/data_harness/ux/render.py +657 -0
- mostlyright/data_harness/ux/workload.py +1077 -0
- mostlyright/data_harness/viewer.py +3713 -0
- mostlyright/data_harness/visual_run/__init__.py +83 -0
- mostlyright/data_harness/visual_run/authoring.py +235 -0
- mostlyright/data_harness/visual_run/contracts.py +673 -0
- mostlyright/data_harness/visual_run/materialize.py +486 -0
- mostlyright/data_harness/visual_run/observations.py +874 -0
- mostlyright/data_harness/visual_run/query.py +259 -0
- mostlyright/data_harness/visual_run/reducer.py +280 -0
- mostlyright/data_harness/visual_run/sdk.py +892 -0
- mostlyright/data_harness/visual_run/store.py +584 -0
- mostlyright/data_harness/visual_run/transport.py +239 -0
- mostlyright/data_harness/watch.py +2999 -0
- mostlyright_data-0.9.0.dist-info/METADATA +607 -0
- mostlyright_data-0.9.0.dist-info/RECORD +314 -0
- mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
- mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
|
@@ -0,0 +1,1077 @@
|
|
|
1
|
+
"""The workload a build would run, judged against the ceilings that will judge it.
|
|
2
|
+
|
|
3
|
+
``mr-data preflight`` used to ask only what this *machine* can do. That is a real question and it
|
|
4
|
+
is still asked, but it is not the question a person who is about to deploy has: theirs is whether
|
|
5
|
+
*this job* can run at all. Today the first honest answer to that arrives as an opaque failure
|
|
6
|
+
inside a container, after the sources have been uploaded and the deployment approved, which is
|
|
7
|
+
the shape of both #234 and #235.
|
|
8
|
+
|
|
9
|
+
This module answers it before any of that, from two things and nothing else: the plan, and one
|
|
10
|
+
confined ``stat`` per source. No source is opened or parsed, nothing is fetched, and no graph is
|
|
11
|
+
executed. That bound is what makes the answer cheap enough to be the first thing a person runs.
|
|
12
|
+
|
|
13
|
+
The stat is confined the way the build's own read is confined -- component by component, with
|
|
14
|
+
``O_NOFOLLOW``, through the pipeline's own ``_require_source_directory`` and
|
|
15
|
+
``_require_source_file``. A plain path join would let the kernel follow a directory link, which
|
|
16
|
+
sizes a file the build refuses and stats something outside the folder this command was pointed at.
|
|
17
|
+
|
|
18
|
+
**Every ceiling here is imported, never restated.** ``MAX_SOURCE_BYTES`` is the pipeline's own
|
|
19
|
+
constant, ``HARD_MAX_RETAINED_BYTES`` is the graph kernel's own, ``ReaderBudgets`` is the Reader
|
|
20
|
+
contract's own. A number written down twice is a number that drifts, and a preflight that judged
|
|
21
|
+
a workload against a stale copy of a ceiling would be worse than no preflight: it would say a job
|
|
22
|
+
fits and then the engine would refuse it. :class:`Ceiling` therefore carries the constant's own
|
|
23
|
+
spelling beside the value, so the rendering names what a reader can go and look at.
|
|
24
|
+
|
|
25
|
+
**A claim is only ever as strong as the observation under it.** Three kinds of claim are
|
|
26
|
+
distinguished, because they license different conclusions:
|
|
27
|
+
|
|
28
|
+
``MEASURED``
|
|
29
|
+
The exact number the engine will see -- a file's size on disk is the size the build reads.
|
|
30
|
+
Over the ceiling means refused, and under it means admitted.
|
|
31
|
+
|
|
32
|
+
``AT_LEAST``
|
|
33
|
+
A floor. The sealed Build holds the raw sources *and* the derived members, so the raw total is
|
|
34
|
+
a floor under its size. A floor over the ceiling refuses, because the real value is at least
|
|
35
|
+
the floor. A floor *under* it settles nothing and is reported undecided -- the symmetric
|
|
36
|
+
mistake to the one below, and the one that let "at least 199 bytes, which is within
|
|
37
|
+
MAX_CANDIDATE_BYTES" be printed about a Build whose size nothing here had established.
|
|
38
|
+
|
|
39
|
+
``AT_MOST``
|
|
40
|
+
A bound in the other direction. A CSV row costs at least one byte, so a source cannot hold
|
|
41
|
+
more rows than it holds bytes. A bound *under* the ceiling proves the ceiling cannot be
|
|
42
|
+
reached; a bound over it proves nothing at all, and is reported as undecided rather than as a
|
|
43
|
+
refusal. This is the direction that keeps the guarantee in
|
|
44
|
+
``tests/test_ux_workload.py`` true: no verdict here says a job fits that the engine
|
|
45
|
+
then refuses.
|
|
46
|
+
|
|
47
|
+
**What cannot be known here is said, not guessed.** How many rows a join emits, how much the
|
|
48
|
+
kernel retains while it runs, and -- the one #234 was read as until its own follow-up found the
|
|
49
|
+
harness's eager allocation instead -- how much memory the machine that will execute this job
|
|
50
|
+
actually has, are all outside what a plan and a stat can answer. :func:`unknowns_for` names each
|
|
51
|
+
of them with the ceiling that will be applied and where it is applied, so the report is honest
|
|
52
|
+
about its own edges instead of quietly implying it checked. It
|
|
53
|
+
is derived from the workload rather than fixed, because a sentence about the graph kernel printed
|
|
54
|
+
over a tabular plan describes a check that did not happen.
|
|
55
|
+
|
|
56
|
+
Nothing here prints, writes, or reaches a network. The result is a value
|
|
57
|
+
:mod:`mostlyright.data_harness.ux.preflight` renders alongside its own checks.
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
from __future__ import annotations
|
|
61
|
+
|
|
62
|
+
import os
|
|
63
|
+
from collections.abc import Mapping, Sequence
|
|
64
|
+
from dataclasses import dataclass
|
|
65
|
+
from pathlib import Path, PurePosixPath
|
|
66
|
+
from typing import Any
|
|
67
|
+
|
|
68
|
+
from mostlyright.data_harness.local_contracts import (
|
|
69
|
+
GRAPH_PLAN_SCHEMA_VERSION,
|
|
70
|
+
MAX_COLUMN_COUNT,
|
|
71
|
+
MAX_GRAPH_NODE_COUNT,
|
|
72
|
+
MAX_OPERATION_COUNT,
|
|
73
|
+
MAX_OUTPUT_ROWS,
|
|
74
|
+
MAX_SOURCE_COUNT,
|
|
75
|
+
PLAN_SCHEMA_VERSION,
|
|
76
|
+
)
|
|
77
|
+
from mostlyright.data_harness.local_contracts import (
|
|
78
|
+
GraphTablePlan as GraphDatasetPlan,
|
|
79
|
+
)
|
|
80
|
+
from mostlyright.data_harness.local_contracts import (
|
|
81
|
+
TablePlan as DatasetPlan,
|
|
82
|
+
)
|
|
83
|
+
from mostlyright.data_harness.offline import (
|
|
84
|
+
_DEFINITION_FILES,
|
|
85
|
+
CONTROL_PATH,
|
|
86
|
+
DEFINITION_PATH,
|
|
87
|
+
INPUT_ROOT_PATH,
|
|
88
|
+
)
|
|
89
|
+
from mostlyright.data_harness.pipeline import (
|
|
90
|
+
MAX_CANDIDATE_BYTES,
|
|
91
|
+
MAX_CANDIDATE_MEMBER_BYTES,
|
|
92
|
+
MAX_COLUMNS,
|
|
93
|
+
MAX_FIELD_BYTES,
|
|
94
|
+
MAX_ROWS,
|
|
95
|
+
MAX_SOURCE_BYTES,
|
|
96
|
+
)
|
|
97
|
+
from mostlyright.data_harness.plan_graph import (
|
|
98
|
+
HARD_MAX_INTERMEDIATE_ROWS,
|
|
99
|
+
HARD_MAX_SOURCE_ROWS,
|
|
100
|
+
GraphResourcePolicy,
|
|
101
|
+
)
|
|
102
|
+
from mostlyright.data_harness.ux.path_kind import (
|
|
103
|
+
A_FOLDER,
|
|
104
|
+
NOTHING_THERE,
|
|
105
|
+
kind_at,
|
|
106
|
+
presence_at,
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
# The file a workbench folder keeps its plan in, taken from the coordinator's own table rather than
|
|
110
|
+
# spelled again here: a rename there would otherwise leave this reporting that the workload could
|
|
111
|
+
# not be read.
|
|
112
|
+
_PLAN_DEFINITION = _DEFINITION_FILES["plan"]
|
|
113
|
+
|
|
114
|
+
# What kind of workload was named, in the words the report uses. These are values a person reads.
|
|
115
|
+
A_WORKBENCH = "a workbench folder"
|
|
116
|
+
A_PLAN = "a plan"
|
|
117
|
+
A_RECIPE = "a recipe"
|
|
118
|
+
|
|
119
|
+
# Which engine a plan runs on. Both spellings are display words derived from the plan's own schema
|
|
120
|
+
# version, which is the contract value and is never renamed here.
|
|
121
|
+
GRAPH_EXECUTION = "the graph engine"
|
|
122
|
+
TABULAR_EXECUTION = "the tabular engine"
|
|
123
|
+
|
|
124
|
+
# How strong one observation is. See the module docstring: only the first two license a refusal.
|
|
125
|
+
MEASURED = "measured"
|
|
126
|
+
AT_LEAST = "at least"
|
|
127
|
+
AT_MOST = "at most"
|
|
128
|
+
|
|
129
|
+
# What one judged fact came out as.
|
|
130
|
+
WITHIN = "within"
|
|
131
|
+
OVER = "over"
|
|
132
|
+
UNDECIDED = "undecided"
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
# The two ways a named workload fails to become facts. Both are literals at their raise sites and
|
|
136
|
+
# both carry their own remediation, because a person who typed a path that is not a plan is not
|
|
137
|
+
# helped by a paragraph about packaging a Build for Studio, which is what the ``WORKLOAD`` prefix
|
|
138
|
+
# family answers.
|
|
139
|
+
WORKLOAD_UNREADABLE = "WORKLOAD_UNREADABLE"
|
|
140
|
+
WORKLOAD_INVALID = "WORKLOAD_INVALID"
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
class WorkloadUnreadable(Exception):
|
|
144
|
+
"""The named workload could not be read, with the typed code and plain reason why.
|
|
145
|
+
|
|
146
|
+
Raised rather than reported, because preflight's caller turns it back into a *check* -- a
|
|
147
|
+
workload that cannot be read is an answer to the question preflight asks, and it belongs on
|
|
148
|
+
the same line as every other answer rather than as a traceback.
|
|
149
|
+
|
|
150
|
+
``refused_as`` is the code of the refusal underneath, when there was one: the contract reader
|
|
151
|
+
and the recipe reader raise their own typed codes, and those are more specific than either of
|
|
152
|
+
the two above. It is carried beside the code rather than *as* it so that every raise site here
|
|
153
|
+
writes a literal, which is what keeps these two codes visible to the gate that proves every
|
|
154
|
+
typed code names a fix.
|
|
155
|
+
"""
|
|
156
|
+
|
|
157
|
+
def __init__(self, code: str, detail: str, *, refused_as: str | None = None) -> None:
|
|
158
|
+
self.code = code
|
|
159
|
+
self.detail = detail
|
|
160
|
+
self.refused_as = refused_as
|
|
161
|
+
super().__init__(detail)
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
@dataclass(frozen=True)
|
|
165
|
+
class Ceiling:
|
|
166
|
+
"""One fact about this workload, and the exact bound the engine will judge it against.
|
|
167
|
+
|
|
168
|
+
``limit_name`` is the constant's own spelling -- ``plan_graph.HARD_MAX_RETAINED_BYTES``, not a
|
|
169
|
+
prose paraphrase -- so a reader who doubts the number can open the file it names. ``code`` is
|
|
170
|
+
the typed code the engine raises when the bound is crossed, so this report and the refusal it
|
|
171
|
+
predicts say one word rather than two.
|
|
172
|
+
"""
|
|
173
|
+
|
|
174
|
+
fact: str
|
|
175
|
+
strength: str
|
|
176
|
+
observed: int | None
|
|
177
|
+
limit: int
|
|
178
|
+
limit_name: str
|
|
179
|
+
code: str
|
|
180
|
+
subject: str | None = None
|
|
181
|
+
|
|
182
|
+
@property
|
|
183
|
+
def verdict(self) -> str:
|
|
184
|
+
"""``over`` only where the observation licenses it; see the module docstring."""
|
|
185
|
+
|
|
186
|
+
if self.observed is None:
|
|
187
|
+
return UNDECIDED
|
|
188
|
+
if self.observed <= self.limit:
|
|
189
|
+
# The mirror image of the rule below, and the one that was missing. A *floor* under the
|
|
190
|
+
# ceiling licenses nothing: the real value is somewhere above the floor and may be
|
|
191
|
+
# above the ceiling too. Reporting that as "within" would be this report saying a job
|
|
192
|
+
# fits that the engine then refuses, which is the one thing it may never do.
|
|
193
|
+
return UNDECIDED if self.strength == AT_LEAST else WITHIN
|
|
194
|
+
# A bound in the "at most" direction says nothing when it sits above the ceiling: the real
|
|
195
|
+
# number is somewhere under the bound and may be under the ceiling too. Calling that a
|
|
196
|
+
# refusal would stop work that would have finished, which is the expensive direction to be
|
|
197
|
+
# wrong in.
|
|
198
|
+
return UNDECIDED if self.strength == AT_MOST else OVER
|
|
199
|
+
|
|
200
|
+
@property
|
|
201
|
+
def refuses(self) -> bool:
|
|
202
|
+
return self.verdict == OVER
|
|
203
|
+
|
|
204
|
+
@property
|
|
205
|
+
def sentence(self) -> str:
|
|
206
|
+
"""This fact, its bound, and what it came to, as one plain line."""
|
|
207
|
+
|
|
208
|
+
named = f"{self.subject}: " if self.subject else ""
|
|
209
|
+
if self.observed is None:
|
|
210
|
+
return (
|
|
211
|
+
f"{named}{self.fact} is not knowable here; the bound is {self.limit} "
|
|
212
|
+
f"({self.limit_name}), refused as {self.code}"
|
|
213
|
+
)
|
|
214
|
+
strength = "" if self.strength == MEASURED else f"{self.strength} "
|
|
215
|
+
judged = {
|
|
216
|
+
WITHIN: "which is within",
|
|
217
|
+
OVER: "which is over",
|
|
218
|
+
UNDECIDED: "which decides nothing about",
|
|
219
|
+
}[self.verdict]
|
|
220
|
+
return (
|
|
221
|
+
f"{named}{self.fact} is {strength}{self.observed}, {judged} the "
|
|
222
|
+
f"{self.limit} of {self.limit_name} ({self.code})"
|
|
223
|
+
)
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
@dataclass(frozen=True)
|
|
227
|
+
class Refused:
|
|
228
|
+
"""A refusal this preflight established that is not a number against a ceiling.
|
|
229
|
+
|
|
230
|
+
Some of what stops a build is not a quantity. A source reached through a directory symlink, a
|
|
231
|
+
source with two hard links, and a recipe pinning a Reader coordinate this build does not ship
|
|
232
|
+
are each refused outright, with no number to compare -- and each is refused by the same typed
|
|
233
|
+
code the engine raises, because the whole point is that one answer is given twice rather than
|
|
234
|
+
two answers given once.
|
|
235
|
+
"""
|
|
236
|
+
|
|
237
|
+
subject: str
|
|
238
|
+
code: str
|
|
239
|
+
sentence: str
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
@dataclass(frozen=True)
|
|
243
|
+
class SourceFacts:
|
|
244
|
+
"""One source of this workload: where it is, how big it is, and what bounds it."""
|
|
245
|
+
|
|
246
|
+
source_id: str
|
|
247
|
+
path: str
|
|
248
|
+
data_format: str
|
|
249
|
+
looked_at: str
|
|
250
|
+
bytes_on_disk: int | None
|
|
251
|
+
unreadable: str | None
|
|
252
|
+
reader: str | None
|
|
253
|
+
reader_budgets: Mapping[str, int] | None
|
|
254
|
+
ceilings: tuple[Ceiling, ...]
|
|
255
|
+
refused: tuple[Refused, ...] = ()
|
|
256
|
+
|
|
257
|
+
@property
|
|
258
|
+
def sentence(self) -> str:
|
|
259
|
+
if self.unreadable is not None:
|
|
260
|
+
return f"{self.path} ({self.data_format}): {self.unreadable}"
|
|
261
|
+
opened_by = ""
|
|
262
|
+
if self.reader is not None:
|
|
263
|
+
# The budgets are part of the answer, not a detail behind it: "which Reader, under
|
|
264
|
+
# which budgets" is one of the facts this command was asked for, and a family name on
|
|
265
|
+
# its own does not say what that family will admit.
|
|
266
|
+
under = (
|
|
267
|
+
", ".join(f"{name} {value}" for name, value in sorted(self.reader_budgets.items()))
|
|
268
|
+
if self.reader_budgets
|
|
269
|
+
else "budgets this build cannot resolve"
|
|
270
|
+
)
|
|
271
|
+
opened_by = f", read by {self.reader} under {under}"
|
|
272
|
+
return f"{self.path} ({self.data_format}): {self.bytes_on_disk} bytes{opened_by}"
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
@dataclass(frozen=True)
|
|
276
|
+
class Workload:
|
|
277
|
+
"""Everything a plan and one stat per source can say about whether this job can run."""
|
|
278
|
+
|
|
279
|
+
kind: str
|
|
280
|
+
named: str
|
|
281
|
+
plan_schema_version: str
|
|
282
|
+
execution: str
|
|
283
|
+
sources: tuple[SourceFacts, ...]
|
|
284
|
+
ceilings: tuple[Ceiling, ...]
|
|
285
|
+
unknowns: tuple[tuple[str, str], ...]
|
|
286
|
+
|
|
287
|
+
@property
|
|
288
|
+
def every_ceiling(self) -> tuple[Ceiling, ...]:
|
|
289
|
+
"""Every judged fact, per source first and then aggregate, in the order they are met."""
|
|
290
|
+
|
|
291
|
+
per_source = [ceiling for source in self.sources for ceiling in source.ceilings]
|
|
292
|
+
return (*per_source, *self.ceilings)
|
|
293
|
+
|
|
294
|
+
@property
|
|
295
|
+
def refusals(self) -> tuple[Ceiling, ...]:
|
|
296
|
+
return tuple(ceiling for ceiling in self.every_ceiling if ceiling.refuses)
|
|
297
|
+
|
|
298
|
+
@property
|
|
299
|
+
def refused(self) -> tuple[Refused, ...]:
|
|
300
|
+
return tuple(item for source in self.sources for item in source.refused)
|
|
301
|
+
|
|
302
|
+
@property
|
|
303
|
+
def blocked(self) -> bool:
|
|
304
|
+
return bool(self.refusals) or bool(self.refused)
|
|
305
|
+
|
|
306
|
+
@property
|
|
307
|
+
def sentence(self) -> str:
|
|
308
|
+
"""The one line the preflight check carries: what was read, and what it came to."""
|
|
309
|
+
|
|
310
|
+
counted = len(self.sources)
|
|
311
|
+
noun = "source" if counted == 1 else "sources"
|
|
312
|
+
read = f"{self.kind} at {self.named}, {self.plan_schema_version} on {self.execution}, "
|
|
313
|
+
stopped: list[str] = [item.sentence for item in self.refused]
|
|
314
|
+
stopped.extend(ceiling.sentence for ceiling in self.refusals)
|
|
315
|
+
if not stopped:
|
|
316
|
+
return (
|
|
317
|
+
f"{read}{counted} {noun}; nothing this preflight can measure is over an "
|
|
318
|
+
"engine ceiling"
|
|
319
|
+
)
|
|
320
|
+
rest = len(stopped) - 1
|
|
321
|
+
more = f", and {rest} more" if rest else ""
|
|
322
|
+
return f"{read}{counted} {noun}; {stopped[0]}{more}"
|
|
323
|
+
|
|
324
|
+
@property
|
|
325
|
+
def remediation(self) -> tuple[str, ...]:
|
|
326
|
+
"""What to do about every refusal, each line said once, in the order they are met."""
|
|
327
|
+
|
|
328
|
+
lines: list[str] = []
|
|
329
|
+
for item in self.refused:
|
|
330
|
+
for line in _remediation_for_refusal(item.code, item.sentence, item.subject):
|
|
331
|
+
if line not in lines:
|
|
332
|
+
lines.append(line)
|
|
333
|
+
for ceiling in self.refusals:
|
|
334
|
+
for line in _remediation_for(ceiling):
|
|
335
|
+
if line not in lines:
|
|
336
|
+
lines.append(line)
|
|
337
|
+
return tuple(lines)
|
|
338
|
+
|
|
339
|
+
def to_block(self) -> dict[str, Any]:
|
|
340
|
+
"""The structured block preflight carries alongside its checks."""
|
|
341
|
+
|
|
342
|
+
width = len(str(max(len(self.sources), 1)))
|
|
343
|
+
bound_width = len(str(max(len(self.every_ceiling), 1)))
|
|
344
|
+
block: dict[str, Any] = {
|
|
345
|
+
"what it is": (
|
|
346
|
+
f"{self.kind} at {self.named}, written as {self.plan_schema_version}, "
|
|
347
|
+
f"which runs on {self.execution}"
|
|
348
|
+
),
|
|
349
|
+
"its sources": {
|
|
350
|
+
f"source {index:0{width}d}": source.sentence
|
|
351
|
+
for index, source in enumerate(self.sources, start=1)
|
|
352
|
+
},
|
|
353
|
+
"what it needs, against what is allowed": {
|
|
354
|
+
f"bound {index:0{bound_width}d}": ceiling.sentence
|
|
355
|
+
for index, ceiling in enumerate(self.every_ceiling, start=1)
|
|
356
|
+
},
|
|
357
|
+
"what this cannot know": dict(self.unknowns),
|
|
358
|
+
}
|
|
359
|
+
if self.refused:
|
|
360
|
+
block["what stops it outright"] = {
|
|
361
|
+
f"{item.subject} ({item.code})": item.sentence for item in self.refused
|
|
362
|
+
}
|
|
363
|
+
return block
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
# ------------------------------------------------------------------------------------------------
|
|
367
|
+
# What cannot be answered from a plan and a stat
|
|
368
|
+
# ------------------------------------------------------------------------------------------------
|
|
369
|
+
#
|
|
370
|
+
# Written down rather than left out. An omitted fact reads as a checked one, and the whole value of
|
|
371
|
+
# this command is that a person can tell the difference between "this was judged and it is fine"
|
|
372
|
+
# and "nothing here judged this".
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
def unknowns_for(*, graph: bool, recipe: bool) -> tuple[tuple[str, str], ...]:
|
|
376
|
+
"""What this reading of *this* workload could not answer, and what will answer it.
|
|
377
|
+
|
|
378
|
+
Derived rather than fixed. A static list said "accounted per node by the graph kernel" about a
|
|
379
|
+
tabular plan that never reaches the kernel, and pointed at "the bound above" for a bound only
|
|
380
|
+
the graph branch produces -- two sentences that described a check that had not happened. Each
|
|
381
|
+
entry below names the bound it is about by its own words rather than by its position, because
|
|
382
|
+
the rendering sorts these keys and "above" is not a place.
|
|
383
|
+
"""
|
|
384
|
+
|
|
385
|
+
# Imported here rather than at the top: `hosted_worker` is the hosted execution surface and
|
|
386
|
+
# pulls the Studio boundary with it, which a local preflight has no business loading to print
|
|
387
|
+
# one number.
|
|
388
|
+
from mostlyright.data_harness.hosted_worker import MAX_HOSTED_CANDIDATE_BYTES
|
|
389
|
+
|
|
390
|
+
rows = HARD_MAX_SOURCE_ROWS if graph else MAX_ROWS
|
|
391
|
+
unknowns: list[tuple[str, str]] = [
|
|
392
|
+
(
|
|
393
|
+
"how many rows a source really holds",
|
|
394
|
+
"counted while the source is streamed, not before. What is knowable here is the bound "
|
|
395
|
+
f"its bytes allow, and the ceiling this plan runs under is {rows}",
|
|
396
|
+
),
|
|
397
|
+
(
|
|
398
|
+
"how wide a source is, and how long its longest field is",
|
|
399
|
+
"both are read out of the source's own header and rows, which this command does not "
|
|
400
|
+
f"open. The build refuses a source over {MAX_COLUMNS} columns as CSV_COLUMNS and a "
|
|
401
|
+
f"field over {MAX_FIELD_BYTES} bytes as CSV_FIELD. The column bound reported above is "
|
|
402
|
+
"the plan's own declared output width, which is a different number",
|
|
403
|
+
),
|
|
404
|
+
(
|
|
405
|
+
"how many rows this job finally produces",
|
|
406
|
+
"the result of running it. What is reported above is the row count the plan's own "
|
|
407
|
+
"quality gate demands, which is a floor the plan states rather than an estimate",
|
|
408
|
+
),
|
|
409
|
+
(
|
|
410
|
+
"how much memory and processing the machine that runs this job has",
|
|
411
|
+
"no contract this command can read states it. Every bound above is the engine's own "
|
|
412
|
+
"ceiling, held whatever the container is given, so a workload that passes here has "
|
|
413
|
+
"not been shown to fit any particular hosted container -- see issue #234, where "
|
|
414
|
+
"hosted runs died with MemoryError while the worker's measured footprint sat far "
|
|
415
|
+
"below its container's limit",
|
|
416
|
+
),
|
|
417
|
+
(
|
|
418
|
+
"whether the sealed Build fits the hosted handoff envelope",
|
|
419
|
+
f"a hosted Build member is capped at {MAX_HOSTED_CANDIDATE_BYTES} bytes "
|
|
420
|
+
"(hosted_worker.MAX_HOSTED_CANDIDATE_BYTES, refused as CANDIDATE_ENVELOPE_INVALID) "
|
|
421
|
+
"and the Courier contract caps are checked by the acquire path. Neither is judged "
|
|
422
|
+
"here: both are about the packaged Build, which does not exist until this job has run",
|
|
423
|
+
),
|
|
424
|
+
]
|
|
425
|
+
if graph:
|
|
426
|
+
unknowns.append(
|
|
427
|
+
(
|
|
428
|
+
"how many rows the steps in between hold",
|
|
429
|
+
"decided per node while the graph runs, because a join's output size is a "
|
|
430
|
+
"property of the data rather than of the plan; the ceiling is "
|
|
431
|
+
f"{HARD_MAX_INTERMEDIATE_ROWS} (plan_graph.HARD_MAX_INTERMEDIATE_ROWS), refused "
|
|
432
|
+
"as RESOURCE_POLICY_INVALID",
|
|
433
|
+
)
|
|
434
|
+
)
|
|
435
|
+
unknowns.append(
|
|
436
|
+
(
|
|
437
|
+
"how many bytes the run itself holds while it runs",
|
|
438
|
+
"accounted per node by the graph kernel. What is knowable here is the part "
|
|
439
|
+
"already spoken for, which is the bound named raw source bytes plus one decode "
|
|
440
|
+
"transient",
|
|
441
|
+
)
|
|
442
|
+
)
|
|
443
|
+
unknowns.append(
|
|
444
|
+
(
|
|
445
|
+
"how many bytes a source arrived as before it was normalized",
|
|
446
|
+
"a Reader's input budget is applied to the bytes fetched from a publisher, and what "
|
|
447
|
+
"is on disk here is the normalized CSV that decode produced. The fetched size is not "
|
|
448
|
+
"observable from a plan and a stat; it is refused as READER_BUDGET at acquisition"
|
|
449
|
+
+ (
|
|
450
|
+
", against the pinned family's max_input_bytes"
|
|
451
|
+
if recipe
|
|
452
|
+
else ", before a source becomes a plan source"
|
|
453
|
+
),
|
|
454
|
+
)
|
|
455
|
+
)
|
|
456
|
+
if not recipe:
|
|
457
|
+
unknowns.append(
|
|
458
|
+
(
|
|
459
|
+
"which reader opens a source, and under which budgets",
|
|
460
|
+
"a plan source is already-normalized CSV and pins no Reader family. A family is "
|
|
461
|
+
"pinned on a recipe source; point --workload at the recipe to have its "
|
|
462
|
+
"coordinate and its narrowed budgets reported",
|
|
463
|
+
)
|
|
464
|
+
)
|
|
465
|
+
return tuple(unknowns)
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
# ------------------------------------------------------------------------------------------------
|
|
469
|
+
# Reading the workload
|
|
470
|
+
# ------------------------------------------------------------------------------------------------
|
|
471
|
+
|
|
472
|
+
|
|
473
|
+
def read_workload(workload: Path | str, *, input_root: Path | str | None = None) -> Workload:
|
|
474
|
+
"""Derive the bounded facts of one workload, without opening a source or running anything.
|
|
475
|
+
|
|
476
|
+
``workload`` is a workbench folder, a plan file, or a frozen recipe file. A workbench folder
|
|
477
|
+
carries its own sources under ``inputs/`` and its own plan under ``definitions/``; the other
|
|
478
|
+
two name their sources relative to ``input_root``, and report the sizes as unknowable when no
|
|
479
|
+
input folder was given, because a relative path with nothing to resolve it against is not a
|
|
480
|
+
file this command may guess at.
|
|
481
|
+
"""
|
|
482
|
+
|
|
483
|
+
named = Path(workload)
|
|
484
|
+
if kind_at(named / CONTROL_PATH) == A_FOLDER:
|
|
485
|
+
return _from_workbench(named)
|
|
486
|
+
return _from_document(named, input_root=input_root)
|
|
487
|
+
|
|
488
|
+
|
|
489
|
+
def _from_workbench(workspace: Path) -> Workload:
|
|
490
|
+
"""A workbench folder: the plan it recorded, and the sources it already holds."""
|
|
491
|
+
|
|
492
|
+
plan_path = workspace / DEFINITION_PATH / _PLAN_DEFINITION
|
|
493
|
+
plan = _plan_from(_read_document(plan_path))
|
|
494
|
+
return _facts(
|
|
495
|
+
plan,
|
|
496
|
+
kind=A_WORKBENCH,
|
|
497
|
+
named=str(workspace),
|
|
498
|
+
sources_under=workspace / INPUT_ROOT_PATH,
|
|
499
|
+
readers={},
|
|
500
|
+
)
|
|
501
|
+
|
|
502
|
+
|
|
503
|
+
def _from_document(document: Path, *, input_root: Path | str | None) -> Workload:
|
|
504
|
+
"""A plan file or a recipe file, whose sources are read from the named input folder."""
|
|
505
|
+
|
|
506
|
+
value = _read_document(document)
|
|
507
|
+
schema_version = value.get("schema_version") if isinstance(value, Mapping) else None
|
|
508
|
+
if isinstance(schema_version, str) and schema_version.startswith("frozen-recipe."):
|
|
509
|
+
plan, readers, kind = _recipe(value)
|
|
510
|
+
else:
|
|
511
|
+
plan, readers, kind = _plan_from(value), {}, A_PLAN
|
|
512
|
+
return _facts(
|
|
513
|
+
plan,
|
|
514
|
+
kind=kind,
|
|
515
|
+
named=str(document),
|
|
516
|
+
sources_under=None if input_root is None else Path(input_root),
|
|
517
|
+
readers=readers,
|
|
518
|
+
)
|
|
519
|
+
|
|
520
|
+
|
|
521
|
+
def _read_document(path: Path) -> Any:
|
|
522
|
+
"""One JSON document, read through the command line's own bounded reader.
|
|
523
|
+
|
|
524
|
+
Not ``path.read_bytes()``. Every document reader in this package goes through
|
|
525
|
+
``cli._read_json`` -- open once, ask the descriptor what it is, read it under a bound -- and a
|
|
526
|
+
preflight pointed at a FIFO would otherwise wait for a writer that never comes, which is the
|
|
527
|
+
recorded failure this reader exists to prevent.
|
|
528
|
+
"""
|
|
529
|
+
|
|
530
|
+
# Imported here, not at the top: this package is loaded while the command line is still being
|
|
531
|
+
# set up.
|
|
532
|
+
from mostlyright.data_harness import cli
|
|
533
|
+
|
|
534
|
+
try:
|
|
535
|
+
return cli._read_json(path)
|
|
536
|
+
except Exception as error:
|
|
537
|
+
raise WorkloadUnreadable(
|
|
538
|
+
"WORKLOAD_UNREADABLE",
|
|
539
|
+
f"{path} could not be read as a plan or a recipe: {error}",
|
|
540
|
+
refused_as=_inner_code(error),
|
|
541
|
+
) from error
|
|
542
|
+
|
|
543
|
+
|
|
544
|
+
def _inner_code(error: BaseException) -> str | None:
|
|
545
|
+
"""The typed code of the refusal underneath, when the reader that raised it carried one."""
|
|
546
|
+
|
|
547
|
+
code = getattr(error, "code", None)
|
|
548
|
+
return code if isinstance(code, str) else None
|
|
549
|
+
|
|
550
|
+
|
|
551
|
+
def _plan_from(value: Any) -> DatasetPlan | GraphDatasetPlan:
|
|
552
|
+
"""The plan this document holds, parsed by the contract's own reader."""
|
|
553
|
+
|
|
554
|
+
from mostlyright.data_harness.local_contracts import parse_plan_document
|
|
555
|
+
|
|
556
|
+
try:
|
|
557
|
+
return parse_plan_document(value)
|
|
558
|
+
except Exception as error:
|
|
559
|
+
raise WorkloadUnreadable(
|
|
560
|
+
"WORKLOAD_INVALID",
|
|
561
|
+
f"that is not a plan this build can read: {error}",
|
|
562
|
+
refused_as=_inner_code(error),
|
|
563
|
+
) from error
|
|
564
|
+
|
|
565
|
+
|
|
566
|
+
def _recipe(value: Any) -> tuple[DatasetPlan | GraphDatasetPlan, dict[str, Any], str]:
|
|
567
|
+
"""A frozen recipe: the plan it froze, and the Reader each of its sources is pinned to."""
|
|
568
|
+
|
|
569
|
+
from mostlyright.data_harness.recipe import parse_frozen_recipe
|
|
570
|
+
|
|
571
|
+
try:
|
|
572
|
+
recipe = parse_frozen_recipe(value)
|
|
573
|
+
except Exception as error:
|
|
574
|
+
raise WorkloadUnreadable(
|
|
575
|
+
"WORKLOAD_INVALID",
|
|
576
|
+
f"that is not a recipe this build can read: {error}",
|
|
577
|
+
refused_as=_inner_code(error),
|
|
578
|
+
) from error
|
|
579
|
+
readers = {source.plan_source_path: source for source in recipe.sources}
|
|
580
|
+
return recipe.transform_plan, readers, A_RECIPE
|
|
581
|
+
|
|
582
|
+
|
|
583
|
+
# ------------------------------------------------------------------------------------------------
|
|
584
|
+
# Judging it
|
|
585
|
+
# ------------------------------------------------------------------------------------------------
|
|
586
|
+
|
|
587
|
+
|
|
588
|
+
def _facts(
|
|
589
|
+
plan: DatasetPlan | GraphDatasetPlan,
|
|
590
|
+
*,
|
|
591
|
+
kind: str,
|
|
592
|
+
named: str,
|
|
593
|
+
sources_under: Path | None,
|
|
594
|
+
readers: Mapping[str, Any],
|
|
595
|
+
) -> Workload:
|
|
596
|
+
graph = isinstance(plan, GraphDatasetPlan)
|
|
597
|
+
sources = tuple(
|
|
598
|
+
_source_facts(
|
|
599
|
+
source,
|
|
600
|
+
sources_under=sources_under,
|
|
601
|
+
reader=readers.get(source.path),
|
|
602
|
+
graph=graph,
|
|
603
|
+
)
|
|
604
|
+
for source in plan.sources
|
|
605
|
+
)
|
|
606
|
+
return Workload(
|
|
607
|
+
kind=kind,
|
|
608
|
+
named=named,
|
|
609
|
+
plan_schema_version=GRAPH_PLAN_SCHEMA_VERSION if graph else PLAN_SCHEMA_VERSION,
|
|
610
|
+
execution=GRAPH_EXECUTION if graph else TABULAR_EXECUTION,
|
|
611
|
+
sources=sources,
|
|
612
|
+
ceilings=_aggregate_ceilings(plan, sources, graph=graph),
|
|
613
|
+
unknowns=unknowns_for(graph=graph, recipe=kind == A_RECIPE),
|
|
614
|
+
)
|
|
615
|
+
|
|
616
|
+
|
|
617
|
+
def _source_facts(
|
|
618
|
+
source: Any,
|
|
619
|
+
*,
|
|
620
|
+
sources_under: Path | None,
|
|
621
|
+
reader: Any,
|
|
622
|
+
graph: bool,
|
|
623
|
+
) -> SourceFacts:
|
|
624
|
+
"""One source, stated once by the build's own observation and judged against its ceilings.
|
|
625
|
+
|
|
626
|
+
The observation is the writer's, deliberately. ``pipeline._read_confined_source_from_handle``
|
|
627
|
+
walks every component of the relative path with ``O_NOFOLLOW`` and ``dir_fd``, applies
|
|
628
|
+
``_require_source_directory`` to each folder and ``_require_source_file`` to the leaf, and only
|
|
629
|
+
then refuses ``SOURCE_TOO_LARGE`` from ``st_size`` -- before it reads a byte. A preflight that
|
|
630
|
+
resolved the path with an ordinary join instead would let the kernel follow a directory link,
|
|
631
|
+
which reports the size of a file the build refuses outright and also stats something outside
|
|
632
|
+
the folder this command was pointed at. So the walk is done here the same way, and the two
|
|
633
|
+
refusals it can raise are reported under the codes the build raises them under.
|
|
634
|
+
"""
|
|
635
|
+
|
|
636
|
+
budgets, unresolved = _reader_facts(reader)
|
|
637
|
+
coordinate = _reader_coordinate(reader)
|
|
638
|
+
refused = () if unresolved is None else (Refused(source.source_id, *unresolved),)
|
|
639
|
+
|
|
640
|
+
def unmeasured(reason: str) -> SourceFacts:
|
|
641
|
+
return SourceFacts(
|
|
642
|
+
source_id=source.source_id,
|
|
643
|
+
path=source.path,
|
|
644
|
+
data_format=source.format,
|
|
645
|
+
looked_at="" if sources_under is None else str(sources_under / source.path),
|
|
646
|
+
bytes_on_disk=None,
|
|
647
|
+
unreadable=reason,
|
|
648
|
+
reader=coordinate,
|
|
649
|
+
reader_budgets=budgets,
|
|
650
|
+
ceilings=(),
|
|
651
|
+
refused=refused,
|
|
652
|
+
)
|
|
653
|
+
|
|
654
|
+
if sources_under is None:
|
|
655
|
+
return unmeasured(
|
|
656
|
+
"no input folder was named, so there is nothing to resolve this relative path "
|
|
657
|
+
"against and its size was not guessed at"
|
|
658
|
+
)
|
|
659
|
+
from mostlyright.data_harness.pipeline import BuildError
|
|
660
|
+
|
|
661
|
+
try:
|
|
662
|
+
info = _confined_stat(sources_under, source.path)
|
|
663
|
+
except BuildError as stopped:
|
|
664
|
+
# The build's own refusal, carried through as it was raised. No second class and no second
|
|
665
|
+
# vocabulary: the code a person quotes here is the code the build would have printed, and
|
|
666
|
+
# `finding_id` is where `pipeline.BuildError` keeps it.
|
|
667
|
+
return SourceFacts(
|
|
668
|
+
source_id=source.source_id,
|
|
669
|
+
path=source.path,
|
|
670
|
+
data_format=source.format,
|
|
671
|
+
looked_at=str(sources_under / source.path),
|
|
672
|
+
bytes_on_disk=None,
|
|
673
|
+
unreadable=str(stopped),
|
|
674
|
+
reader=coordinate,
|
|
675
|
+
reader_budgets=budgets,
|
|
676
|
+
ceilings=(),
|
|
677
|
+
refused=(
|
|
678
|
+
*refused,
|
|
679
|
+
Refused(source.source_id, stopped.finding_id, str(stopped)),
|
|
680
|
+
),
|
|
681
|
+
)
|
|
682
|
+
except OSError as error:
|
|
683
|
+
where = sources_under / source.path
|
|
684
|
+
found = presence_at(where)
|
|
685
|
+
if found == NOTHING_THERE:
|
|
686
|
+
return unmeasured(f"there is {found} at {where} to read")
|
|
687
|
+
# A path that is not there is the commonest case by far, and it is worth its own sentence.
|
|
688
|
+
# Everything else is reported by the errno the walk actually met, and by nothing else: a
|
|
689
|
+
# second look at a path whose stat has just failed would put a noun in front of a person
|
|
690
|
+
# that this command did not establish, which `scripts/path_kind_gate.py` is right to refuse.
|
|
691
|
+
return unmeasured(f"{where} could not be measured: {error.strerror or error}")
|
|
692
|
+
size = info.st_size
|
|
693
|
+
return SourceFacts(
|
|
694
|
+
source_id=source.source_id,
|
|
695
|
+
path=source.path,
|
|
696
|
+
data_format=source.format,
|
|
697
|
+
looked_at=str(sources_under / source.path),
|
|
698
|
+
bytes_on_disk=size,
|
|
699
|
+
unreadable=None,
|
|
700
|
+
reader=coordinate,
|
|
701
|
+
reader_budgets=budgets,
|
|
702
|
+
ceilings=_source_ceilings(source.source_id, size, budgets=budgets, graph=graph),
|
|
703
|
+
refused=refused,
|
|
704
|
+
)
|
|
705
|
+
|
|
706
|
+
|
|
707
|
+
def _confined_stat(root: Path, relative: str) -> os.stat_result:
|
|
708
|
+
"""Stat one source the way the build reaches it: component by component, following nothing.
|
|
709
|
+
|
|
710
|
+
``pipeline._require_source_directory`` and ``pipeline._require_source_file`` are the build's
|
|
711
|
+
own tests, called here rather than re-implemented, so the shapes this reports and the shapes
|
|
712
|
+
the build refuses cannot become two lists. Their ``BuildError`` is allowed to propagate exactly
|
|
713
|
+
as raised -- ``SOURCE_SYMLINK``, ``SOURCE_NOT_DIRECTORY``, ``SOURCE_NOT_REGULAR``,
|
|
714
|
+
``SOURCE_HARDLINK`` -- because this command exists to say the build's refusal early rather than
|
|
715
|
+
to invent a second class and a second vocabulary for it.
|
|
716
|
+
|
|
717
|
+
Nothing is read and nothing is kept: the descriptors opened are folder descriptors on the way
|
|
718
|
+
down, and every one of them is closed before this returns.
|
|
719
|
+
"""
|
|
720
|
+
|
|
721
|
+
from mostlyright.data_harness.pipeline import (
|
|
722
|
+
_require_source_directory,
|
|
723
|
+
_require_source_file,
|
|
724
|
+
)
|
|
725
|
+
|
|
726
|
+
flags = os.O_RDONLY | getattr(os, "O_DIRECTORY", 0) | getattr(os, "O_NOFOLLOW", 0)
|
|
727
|
+
parts = PurePosixPath(relative).parts
|
|
728
|
+
opened: list[int] = []
|
|
729
|
+
try:
|
|
730
|
+
parent_fd = os.open(root, flags)
|
|
731
|
+
opened.append(parent_fd)
|
|
732
|
+
walked = ""
|
|
733
|
+
for part in parts[:-1]:
|
|
734
|
+
walked = f"{walked}/{part}" if walked else part
|
|
735
|
+
named = os.stat(part, dir_fd=parent_fd, follow_symlinks=False)
|
|
736
|
+
_require_source_directory(named, walked)
|
|
737
|
+
parent_fd = os.open(part, flags, dir_fd=parent_fd)
|
|
738
|
+
opened.append(parent_fd)
|
|
739
|
+
leaf = os.stat(parts[-1], dir_fd=parent_fd, follow_symlinks=False)
|
|
740
|
+
_require_source_file(leaf, relative)
|
|
741
|
+
return leaf
|
|
742
|
+
finally:
|
|
743
|
+
for descriptor in opened:
|
|
744
|
+
os.close(descriptor)
|
|
745
|
+
|
|
746
|
+
|
|
747
|
+
def _source_ceilings(
|
|
748
|
+
source_id: str,
|
|
749
|
+
size: int,
|
|
750
|
+
*,
|
|
751
|
+
budgets: Mapping[str, int] | None,
|
|
752
|
+
graph: bool,
|
|
753
|
+
) -> tuple[Ceiling, ...]:
|
|
754
|
+
ceilings = [
|
|
755
|
+
Ceiling(
|
|
756
|
+
fact="bytes on disk",
|
|
757
|
+
strength=MEASURED,
|
|
758
|
+
observed=size,
|
|
759
|
+
limit=MAX_SOURCE_BYTES,
|
|
760
|
+
limit_name="pipeline.MAX_SOURCE_BYTES",
|
|
761
|
+
code="SOURCE_TOO_LARGE",
|
|
762
|
+
subject=source_id,
|
|
763
|
+
),
|
|
764
|
+
Ceiling(
|
|
765
|
+
fact="bytes it adds to the sealed Build as one member",
|
|
766
|
+
strength=MEASURED,
|
|
767
|
+
observed=size,
|
|
768
|
+
limit=MAX_CANDIDATE_MEMBER_BYTES,
|
|
769
|
+
limit_name="pipeline.MAX_CANDIDATE_MEMBER_BYTES",
|
|
770
|
+
code="CANDIDATE_TOO_LARGE",
|
|
771
|
+
subject=source_id,
|
|
772
|
+
),
|
|
773
|
+
# One CSV row costs at least the byte that ends it, so a source cannot hold more rows than
|
|
774
|
+
# it holds bytes. Under the ceiling that settles the question; over it, nothing is decided
|
|
775
|
+
# -- see the module docstring on AT_MOST.
|
|
776
|
+
Ceiling(
|
|
777
|
+
fact="rows its bytes allow",
|
|
778
|
+
strength=AT_MOST,
|
|
779
|
+
observed=size,
|
|
780
|
+
limit=HARD_MAX_SOURCE_ROWS if graph else MAX_ROWS,
|
|
781
|
+
limit_name=("plan_graph.HARD_MAX_SOURCE_ROWS" if graph else "pipeline.MAX_ROWS"),
|
|
782
|
+
code="CSV_ROWS",
|
|
783
|
+
subject=source_id,
|
|
784
|
+
),
|
|
785
|
+
]
|
|
786
|
+
if budgets is not None:
|
|
787
|
+
# The *output* budget, not the input one. What is on disk at a recipe source's
|
|
788
|
+
# `plan_source_path` is the normalized CSV a decode produced, and
|
|
789
|
+
# `ReaderResult.validate_for` holds exactly that against `max_output_bytes`. The input
|
|
790
|
+
# budget applies to the bytes
|
|
791
|
+
# fetched from the publisher, which are gone by the time this file exists -- judging this
|
|
792
|
+
# file by that ceiling refused normalized sources between the two, and the bulk families
|
|
793
|
+
# put 16x between them. The fetched size is named in the unknowns instead.
|
|
794
|
+
#
|
|
795
|
+
# This is where a raster refuses: the weather.grib2 families take the plain ReaderBudgets
|
|
796
|
+
# defaults rather than the bulk ones, so both of their ceilings are 16 MiB, which no
|
|
797
|
+
# published model file fits under.
|
|
798
|
+
ceilings.append(
|
|
799
|
+
Ceiling(
|
|
800
|
+
fact="bytes its pinned Reader may emit",
|
|
801
|
+
strength=MEASURED,
|
|
802
|
+
observed=size,
|
|
803
|
+
limit=budgets["max_output_bytes"],
|
|
804
|
+
limit_name="the pinned Reader family's max_output_bytes",
|
|
805
|
+
code="READER_BUDGET",
|
|
806
|
+
subject=source_id,
|
|
807
|
+
)
|
|
808
|
+
)
|
|
809
|
+
return tuple(ceilings)
|
|
810
|
+
|
|
811
|
+
|
|
812
|
+
def _aggregate_ceilings(
|
|
813
|
+
plan: DatasetPlan | GraphDatasetPlan,
|
|
814
|
+
sources: Sequence[SourceFacts],
|
|
815
|
+
*,
|
|
816
|
+
graph: bool,
|
|
817
|
+
) -> tuple[Ceiling, ...]:
|
|
818
|
+
"""The bounds that are about the whole job rather than about one source of it."""
|
|
819
|
+
|
|
820
|
+
sized = [source.bytes_on_disk for source in sources if source.bytes_on_disk is not None]
|
|
821
|
+
complete = len(sized) == len(sources) and bool(sources)
|
|
822
|
+
raw_total = sum(sized) if complete else None
|
|
823
|
+
decode_peak = max(sized, default=0) if complete else None
|
|
824
|
+
ceilings = [
|
|
825
|
+
Ceiling(
|
|
826
|
+
fact="sources this plan declares",
|
|
827
|
+
strength=MEASURED,
|
|
828
|
+
observed=len(plan.sources),
|
|
829
|
+
limit=MAX_SOURCE_COUNT,
|
|
830
|
+
limit_name="local_contracts.MAX_SOURCE_COUNT",
|
|
831
|
+
code="COLLECTION_LIMIT",
|
|
832
|
+
),
|
|
833
|
+
# The plan's own declared output width against the contract's ceiling for it. It is
|
|
834
|
+
# deliberately *not* judged against `CSV_COLUMNS`, which the build raises about a source's
|
|
835
|
+
# header width -- a different number, unknowable from a stat, and named in the unknowns.
|
|
836
|
+
Ceiling(
|
|
837
|
+
fact="output columns this plan declares",
|
|
838
|
+
strength=MEASURED,
|
|
839
|
+
observed=len(plan.select),
|
|
840
|
+
limit=MAX_COLUMN_COUNT,
|
|
841
|
+
limit_name="local_contracts.MAX_COLUMN_COUNT",
|
|
842
|
+
code="COLLECTION_LIMIT",
|
|
843
|
+
),
|
|
844
|
+
]
|
|
845
|
+
if graph:
|
|
846
|
+
assert isinstance(plan, GraphDatasetPlan)
|
|
847
|
+
policy = GraphResourcePolicy()
|
|
848
|
+
ceilings.extend(
|
|
849
|
+
(
|
|
850
|
+
Ceiling(
|
|
851
|
+
fact="graph nodes this plan declares",
|
|
852
|
+
strength=MEASURED,
|
|
853
|
+
observed=len(plan.nodes),
|
|
854
|
+
limit=MAX_GRAPH_NODE_COUNT,
|
|
855
|
+
limit_name="local_contracts.MAX_GRAPH_NODE_COUNT",
|
|
856
|
+
code="COLLECTION_LIMIT",
|
|
857
|
+
),
|
|
858
|
+
Ceiling(
|
|
859
|
+
fact="graph operations this plan declares",
|
|
860
|
+
strength=MEASURED,
|
|
861
|
+
observed=sum(1 for node in plan.nodes if node.operation != "source"),
|
|
862
|
+
limit=MAX_GRAPH_NODE_COUNT,
|
|
863
|
+
limit_name="local_contracts.MAX_GRAPH_NODE_COUNT",
|
|
864
|
+
code="COLLECTION_LIMIT",
|
|
865
|
+
),
|
|
866
|
+
# `pipeline._replay_graph_derivations` refuses this before it executes a node.
|
|
867
|
+
Ceiling(
|
|
868
|
+
fact="rows the plan's own quality gate demands",
|
|
869
|
+
strength=MEASURED,
|
|
870
|
+
observed=plan.quality.min_rows,
|
|
871
|
+
limit=policy.max_output_rows,
|
|
872
|
+
limit_name="plan_graph.HARD_MAX_OUTPUT_ROWS",
|
|
873
|
+
code="ROW_BUDGET_EXCEEDED",
|
|
874
|
+
),
|
|
875
|
+
Ceiling(
|
|
876
|
+
fact="raw source bytes the run must hold at once",
|
|
877
|
+
strength=MEASURED,
|
|
878
|
+
observed=raw_total,
|
|
879
|
+
limit=policy.max_retained_bytes,
|
|
880
|
+
limit_name="plan_graph.HARD_MAX_RETAINED_BYTES",
|
|
881
|
+
code="BYTE_BUDGET_EXCEEDED",
|
|
882
|
+
),
|
|
883
|
+
# The kernel keeps the raw bytes and one decoded-text transient beside its own
|
|
884
|
+
# tables, and refuses when what is left of the envelope falls below one byte. Said
|
|
885
|
+
# here as the sum against one byte less than the envelope, which is the same test.
|
|
886
|
+
Ceiling(
|
|
887
|
+
fact="raw source bytes plus one decode transient",
|
|
888
|
+
strength=MEASURED,
|
|
889
|
+
observed=None if raw_total is None else raw_total + (decode_peak or 0),
|
|
890
|
+
limit=policy.max_retained_bytes - 1,
|
|
891
|
+
limit_name="plan_graph.HARD_MAX_RETAINED_BYTES, less the one byte the "
|
|
892
|
+
"execution itself must be left",
|
|
893
|
+
code="BYTE_BUDGET_EXCEEDED",
|
|
894
|
+
),
|
|
895
|
+
)
|
|
896
|
+
)
|
|
897
|
+
else:
|
|
898
|
+
ceilings.extend(
|
|
899
|
+
(
|
|
900
|
+
Ceiling(
|
|
901
|
+
fact="cleaning steps this plan declares",
|
|
902
|
+
strength=MEASURED,
|
|
903
|
+
observed=len(plan.cleaning),
|
|
904
|
+
limit=MAX_OPERATION_COUNT,
|
|
905
|
+
limit_name="local_contracts.MAX_OPERATION_COUNT",
|
|
906
|
+
code="COLLECTION_LIMIT",
|
|
907
|
+
),
|
|
908
|
+
Ceiling(
|
|
909
|
+
fact="rows the plan's own quality gate demands",
|
|
910
|
+
strength=MEASURED,
|
|
911
|
+
observed=plan.quality.min_rows,
|
|
912
|
+
limit=MAX_OUTPUT_ROWS,
|
|
913
|
+
limit_name="local_contracts.MAX_OUTPUT_ROWS",
|
|
914
|
+
code="ROW_BUDGET_EXCEEDED",
|
|
915
|
+
),
|
|
916
|
+
)
|
|
917
|
+
)
|
|
918
|
+
ceilings.append(
|
|
919
|
+
# A floor, not a measurement: the sealed Build holds every raw source as a member and the
|
|
920
|
+
# derived members besides. Over the ceiling it still refuses, because the real size is at
|
|
921
|
+
# least this.
|
|
922
|
+
Ceiling(
|
|
923
|
+
fact="bytes the sealed Build holds",
|
|
924
|
+
strength=AT_LEAST,
|
|
925
|
+
observed=raw_total,
|
|
926
|
+
limit=MAX_CANDIDATE_BYTES,
|
|
927
|
+
limit_name="pipeline.MAX_CANDIDATE_BYTES",
|
|
928
|
+
code="CANDIDATE_TOO_LARGE",
|
|
929
|
+
)
|
|
930
|
+
)
|
|
931
|
+
return tuple(ceilings)
|
|
932
|
+
|
|
933
|
+
|
|
934
|
+
def _reader_coordinate(reader: Any) -> str | None:
|
|
935
|
+
"""``family@version`` for a recipe source that pins one, and ``None`` for anything else."""
|
|
936
|
+
|
|
937
|
+
if reader is None or getattr(reader, "family_id", None) is None:
|
|
938
|
+
return None
|
|
939
|
+
return f"{reader.family_id}@{reader.family_version}"
|
|
940
|
+
|
|
941
|
+
|
|
942
|
+
def _reader_facts(reader: Any) -> tuple[dict[str, int] | None, tuple[str, str] | None]:
|
|
943
|
+
"""The budgets one decode will really run under, or the refusal that says there will be none.
|
|
944
|
+
|
|
945
|
+
This is ``acquisition.sandbox``'s own call -- ``family.default_budgets.narrowed_by(caps)`` --
|
|
946
|
+
made against the same registry, so the numbers reported here are the numbers the clean room
|
|
947
|
+
will apply rather than a second opinion about them.
|
|
948
|
+
|
|
949
|
+
The two ways it can fail are answered differently, and conflating them was a false pass. A
|
|
950
|
+
coordinate the Toolbox does not hold is a *refusal*: ``recipe.py`` resolves the same pin on the
|
|
951
|
+
execution path and ``READER_FAMILY_UNKNOWN`` is what it gets, so a preflight that dropped the
|
|
952
|
+
budgets and said nothing reported "read by weather.grib2@99.0.0" about a Reader that does not
|
|
953
|
+
exist, and passed. A family whose optional dependency is simply not installed here is not that:
|
|
954
|
+
the pin is real and this machine cannot answer for its budgets, which is said in the source's
|
|
955
|
+
own sentence rather than left blank.
|
|
956
|
+
"""
|
|
957
|
+
|
|
958
|
+
if reader is None or getattr(reader, "family_id", None) is None:
|
|
959
|
+
return None, None
|
|
960
|
+
from mostlyright.data_harness.readers.contracts import ReaderError
|
|
961
|
+
from mostlyright.data_harness.readers.registry import TOOLBOX
|
|
962
|
+
|
|
963
|
+
coordinate = f"{reader.family_id}@{reader.family_version}"
|
|
964
|
+
try:
|
|
965
|
+
family = TOOLBOX.resolve(reader.family_id, reader.family_version)
|
|
966
|
+
except ReaderError as error:
|
|
967
|
+
if error.code == "READER_DEPENDENCY_UNAVAILABLE":
|
|
968
|
+
return None, None
|
|
969
|
+
offered = ", ".join(f"{name}@{version}" for name, version in TOOLBOX.families())
|
|
970
|
+
return None, (
|
|
971
|
+
error.code,
|
|
972
|
+
f"this recipe pins {coordinate}, which this build does not have. It has: {offered}",
|
|
973
|
+
)
|
|
974
|
+
caps = None if reader.resource_caps is None else dict(reader.resource_caps.to_dict())
|
|
975
|
+
try:
|
|
976
|
+
budgets = family.default_budgets.narrowed_by(caps)
|
|
977
|
+
except ReaderError as error:
|
|
978
|
+
return None, (error.code, f"the caps {coordinate} is narrowed by are refused: {error}")
|
|
979
|
+
return {
|
|
980
|
+
"max_input_bytes": budgets.max_input_bytes,
|
|
981
|
+
"max_output_bytes": budgets.max_output_bytes,
|
|
982
|
+
"max_declared_cells": budgets.max_declared_cells,
|
|
983
|
+
"max_rows": budgets.max_rows,
|
|
984
|
+
"max_columns": budgets.max_columns,
|
|
985
|
+
}, None
|
|
986
|
+
|
|
987
|
+
|
|
988
|
+
# ------------------------------------------------------------------------------------------------
|
|
989
|
+
# What to do about a refusal
|
|
990
|
+
# ------------------------------------------------------------------------------------------------
|
|
991
|
+
|
|
992
|
+
|
|
993
|
+
def _remediation_for_refusal(code: str, sentence: str, subject: str) -> tuple[str, ...]:
|
|
994
|
+
"""What to do about a refusal that is not a number, in the words of the refusal it predicts."""
|
|
995
|
+
|
|
996
|
+
from mostlyright.data_harness.ux.remediation import remediation_for
|
|
997
|
+
|
|
998
|
+
return (
|
|
999
|
+
f"{subject}: {sentence}. A build refuses this as {code}. Nothing was uploaded and "
|
|
1000
|
+
"nothing was run.",
|
|
1001
|
+
*remediation_for(code),
|
|
1002
|
+
)
|
|
1003
|
+
|
|
1004
|
+
|
|
1005
|
+
def _remediation_for(ceiling: Ceiling) -> tuple[str, ...]:
|
|
1006
|
+
"""What to do about one crossed bound, in the words of the refusal it predicts.
|
|
1007
|
+
|
|
1008
|
+
The error-remediation map answers by code, and these codes are shared with the mid-run
|
|
1009
|
+
refusals, so the map's paragraph is the right one and is used. What it cannot say is the thing
|
|
1010
|
+
a person most needs to hear here, which is that a bigger machine is not the answer: every
|
|
1011
|
+
bound above is a semantic ceiling of the engine, and CPU and RAM do not move any of them. That
|
|
1012
|
+
sentence is added for the byte and row bounds, where the instinct to ask for more memory is
|
|
1013
|
+
strongest.
|
|
1014
|
+
"""
|
|
1015
|
+
|
|
1016
|
+
from mostlyright.data_harness.ux.remediation import remediation_for
|
|
1017
|
+
|
|
1018
|
+
lines = list(remediation_for(ceiling.code))
|
|
1019
|
+
named = f" ({ceiling.subject})" if ceiling.subject else ""
|
|
1020
|
+
# The typed code goes in the sentence, not only in the record: this payload never passes
|
|
1021
|
+
# through the command line's error boundary, which is where a refusal's `Code:` line is
|
|
1022
|
+
# printed, so a person reading the plain rendering would otherwise have nothing to quote.
|
|
1023
|
+
lines.insert(
|
|
1024
|
+
0,
|
|
1025
|
+
f"This workload's {ceiling.fact}{named} is {ceiling.observed}, over the "
|
|
1026
|
+
f"{ceiling.limit} that {ceiling.limit_name} allows, which a build refuses as "
|
|
1027
|
+
f"{ceiling.code}. Nothing was uploaded and nothing was run.",
|
|
1028
|
+
)
|
|
1029
|
+
if ceiling.code in _NOT_A_MACHINE_SIZE:
|
|
1030
|
+
lines.append(_PARTITIONED)
|
|
1031
|
+
return tuple(lines)
|
|
1032
|
+
|
|
1033
|
+
|
|
1034
|
+
# The bounds a bigger machine does not move. Every one of them is a ceiling of the deterministic
|
|
1035
|
+
# engine, held whatever the container is given, so answering one with more RAM is answering the
|
|
1036
|
+
# wrong question -- which is the mistake #234 records being made in the other direction.
|
|
1037
|
+
_NOT_A_MACHINE_SIZE = frozenset(
|
|
1038
|
+
{
|
|
1039
|
+
"BYTE_BUDGET_EXCEEDED",
|
|
1040
|
+
"CANDIDATE_TOO_LARGE",
|
|
1041
|
+
"CSV_ROWS",
|
|
1042
|
+
"READER_BUDGET",
|
|
1043
|
+
"ROW_BUDGET_EXCEEDED",
|
|
1044
|
+
"SOURCE_TOO_LARGE",
|
|
1045
|
+
}
|
|
1046
|
+
)
|
|
1047
|
+
|
|
1048
|
+
_PARTITIONED = (
|
|
1049
|
+
"A larger machine does not raise this bound: it is a ceiling of the deterministic engine, "
|
|
1050
|
+
"which runs in memory and holds it whatever CPU and RAM it is given. A source of this size "
|
|
1051
|
+
"needs partitioned execution -- bounded shards manifested, executed and merged -- which is "
|
|
1052
|
+
"separate, unbuilt work, and is not more memory."
|
|
1053
|
+
)
|
|
1054
|
+
|
|
1055
|
+
|
|
1056
|
+
__all__ = [
|
|
1057
|
+
"AT_LEAST",
|
|
1058
|
+
"AT_MOST",
|
|
1059
|
+
"A_PLAN",
|
|
1060
|
+
"A_RECIPE",
|
|
1061
|
+
"A_WORKBENCH",
|
|
1062
|
+
"GRAPH_EXECUTION",
|
|
1063
|
+
"MEASURED",
|
|
1064
|
+
"OVER",
|
|
1065
|
+
"TABULAR_EXECUTION",
|
|
1066
|
+
"UNDECIDED",
|
|
1067
|
+
"WITHIN",
|
|
1068
|
+
"WORKLOAD_INVALID",
|
|
1069
|
+
"WORKLOAD_UNREADABLE",
|
|
1070
|
+
"Ceiling",
|
|
1071
|
+
"Refused",
|
|
1072
|
+
"SourceFacts",
|
|
1073
|
+
"Workload",
|
|
1074
|
+
"WorkloadUnreadable",
|
|
1075
|
+
"read_workload",
|
|
1076
|
+
"unknowns_for",
|
|
1077
|
+
]
|